diff --git a/Makefile b/Makefile index bab8885..a296113 100644 --- a/Makefile +++ b/Makefile @@ -9,16 +9,18 @@ COMPOSE := docker compose -f deploy/docker-compose.yaml COMPOSE_MIXED := docker compose -f deploy/docker-compose.mixed.yaml # Which engine backends the mixed stack runs (comma-separated subset of -# vllm,sglang,llamacpp). Override on the CLI: `make up-mixed ENGINES=vllm,sglang`. +# vllm,sglang,llamacpp,trtllm). Override on the CLI: `make up-mixed ENGINES=vllm,sglang`. # The FIRST engine listed also serves the dashboard /api (all backends share Postgres, # so any can). Compose profiles gate the rest, so unlisted backends never start. -ENGINES ?= vllm,sglang,llamacpp +# NOTE: trtllm pulls a large (~60GB) TensorRT-LLM image; drop it from ENGINES if you +# don't need it (e.g. `make up-mixed ENGINES=vllm,sglang,llamacpp`). +ENGINES ?= vllm,sglang,llamacpp,trtllm DASHBOARD_BACKEND := $(shell echo "$(ENGINES)" | cut -d, -f1)-backend MIXED_ENV := COMPOSE_PROFILES=$(ENGINES) DASHBOARD_BACKEND=$(DASHBOARD_BACKEND) # Engines NOT selected this run. `docker compose up` (even with --remove-orphans) # leaves a profiled-but-deselected service running, so we stop+remove them explicitly # to make switching engine sets clean. Computed as ALL − ENGINES. -_ALL_ENGINES := vllm sglang llamacpp +_ALL_ENGINES := vllm sglang llamacpp trtllm _comma := , _space := $(empty) $(empty) _SELECTED := $(subst $(_comma),$(_space),$(ENGINES)) @@ -28,7 +30,7 @@ _DESELECTED_SVCS := $(addsuffix -backend,$(_DESELECTED)) .PHONY: help test test-backend test-router test-schema \ dev-backend dev-frontend build-frontend install-frontend \ up down logs ps build up-mixed down-mixed logs-mixed \ - up-vllm up-sglang up-llamacpp + up-vllm up-sglang up-llamacpp up-trtllm help: @echo "Targets:" @@ -89,6 +91,9 @@ up-sglang: up-llamacpp: $(MAKE) up-mixed ENGINES=llamacpp +up-trtllm: + $(MAKE) up-mixed ENGINES=trtllm + # down removes the whole project (all profiles) regardless of the current selection. down-mixed: COMPOSE_PROFILES=vllm,sglang,llamacpp $(COMPOSE_MIXED) down diff --git a/README.md b/README.md index a79f849..793bda9 100644 --- a/README.md +++ b/README.md @@ -29,7 +29,8 @@ becomes a routable model; the router load-balances across instances; and a bundl ## Highlights - **One router controls the whole fleet** — a single OpenAI- & Anthropic-compatible origin fronts every model. Route by the `model` field across `/v1/chat/completions`, `/v1/messages`, `/v1/embeddings`, `/v1/rerank`, `/v1/score`, `/tokenize` and more; the router resolves the group and load-balances its instances, so clients never address an instance directly. -- **Three inference engines, one control plane — vLLM, SGLang & llama.cpp** — choose the engine per model (the *Add Model* dialog has an engine selector); an engine-aware scheduler places each model on a backend that can run it, and the same router / dashboard / monitoring front them all. Run a **vLLM-only** stack (`make up`) or a **mixed** fleet (`make up-mixed`) that adds SGLang and a llama.cpp (GGUF / CPU-offload) backend. See [docs/mixed-engine-deployment.md](docs/mixed-engine-deployment.md). +- **Four inference engines, one control plane — vLLM, SGLang, llama.cpp & TensorRT-LLM** — choose the engine per model (the *Add Model* dialog has an engine selector); an engine-aware scheduler places each model on a backend that can run it, and the same router / dashboard / monitoring front them all. Run a **vLLM-only** stack (`make up`) or a **mixed** fleet (`make up-mixed`) that adds SGLang, llama.cpp (GGUF / CPU-offload) and TensorRT-LLM backends. See [docs/mixed-engine-deployment.md](docs/mixed-engine-deployment.md). +- **TensorRT-LLM, two ways** — a `trtllm` group runs a **pre-built TRT engine** (`engine_dir` → `--backend tensorrt`, lowest TTFT) or **HF weights directly** (no `engine_dir` → `--backend pytorch`, no build step, any TRT-LLM-supported arch); monitoring is identical either way. Build engines from the UI: the **Model Library**'s *Convert to TRT* action runs `trtllm-bench` (one architecture-agnostic path, not per-model scripts) to compile a cached HF model into an engine, then pick it in *Add Model*. See [docs/trtllm-launcher-impl-design_zh-TW.md](docs/trtllm-launcher-impl-design_zh-TW.md). - **Add a model by pasting `vllm serve …`** — parsed into a form and layered on as a dynamic overlay; the router hot-reloads, no `config.yaml` edits. - **Lifecycle + self-healing** — per-instance state machine (`stopped → starting → ready → sleeping → failed`), VRAM pre-flight guard, GPU auto-placement, crash auto-restart with backoff. - **Autoscaling with a warm-standby tier** — per group, keep `min_ready` replicas warm and scale up on queue depth (wake first, else cold-start) to `max_ready`; fold idle replicas back down `ready → sleep → stop`. vLLM **sleep mode** (level-1) frees a replica's VRAM but wakes in seconds, so scaling down needn't mean a minute-long cold start. Set it from config.yaml or the dashboard; a live Grafana dashboard + alerts are bundled. @@ -41,7 +42,7 @@ becomes a routable model; the router load-balances across instances; and a bundl - **Lifecycle alerting** — discrete model events (crash, restart-budget exhausted, recovered) pushed to Slack / Discord / a generic webhook, with per-sink severity floors and per-model cooldown; configured via env or the admin **Notifications** page (one-click test). Complements Grafana's metric alerts. - **Playground** — OpenAI-compatible chat (streaming) / completions / embeddings / reranking, with reasoning display. - **Benchmark & evaluate** — evalscope load tests (concurrency, arrival-rate, SLA auto-tune) plus 30+ accuracy datasets with LLM-as-judge. -- **Libraries** — browse / pre-download HF model weights & datasets from the UI; tool-calling parser helper; LoRA support. +- **Libraries** — browse / pre-download HF model weights & datasets from the UI; the Model Library groups cached weights into **Models / GGUF / TensorRT-LLM engines**, builds TRT engines from a cached model, and *Add Model* lets you pick a downloaded model (and a built TRT engine) instead of typing paths; tool-calling parser helper; LoRA support (incl. PEFT→GGUF conversion). - **Multi-user & audit** — role-based control (`viewer`/`operator`/`admin`) via named operator credentials, with a redacted **audit log** of every change; plus mint/revoke API keys with per-key usage attribution, rate limits and **token quotas** (total / daily / monthly). The env admin token and open local-dev still work unchanged. - **Config versioning & backup** — the dynamic-model overlay (where every runtime change lives) is snapshotted on each mutation; export it as a portable file, import to restore, and roll back to any past version with a side-by-side diff — all from the admin **Config Versions** page. `config.yaml` is never rewritten. @@ -63,8 +64,8 @@ make up # build + start the whole stack **Two deployment modes:** - **`make up`** — the default **vLLM-only** stack. -- **`make up-mixed`** — a **multi-engine** fleet: vLLM, SGLang and llama.cpp backends sharing one Postgres, router, dashboard and Grafana. Add a model from *Add Model → engine: `sglang` / `llamacpp`* and it is auto-placed on a backend that can run it. `make down-mixed` / `make logs-mixed` manage it. See [docs/mixed-engine-deployment.md](docs/mixed-engine-deployment.md). - - **Pick which engines run**: `make up-mixed ENGINES=vllm,sglang` (any subset), or the shortcuts `make up-vllm` / `up-sglang` / `up-llamacpp`. Unlisted engines never start (their VRAM stays free); the first engine listed serves the dashboard API, and switching sets stops the de-selected backends automatically. +- **`make up-mixed`** — a **multi-engine** fleet: vLLM, SGLang, llama.cpp and TensorRT-LLM backends sharing one Postgres, router, dashboard and Grafana (all four by default; drop any via `ENGINES=…`). Add a model from *Add Model → engine: `sglang` / `llamacpp` / `trtllm`* and it is auto-placed on a backend that can run it. `make down-mixed` / `make logs-mixed` manage it. See [docs/mixed-engine-deployment.md](docs/mixed-engine-deployment.md). + - **Pick which engines run**: `make up-mixed ENGINES=vllm,sglang,llamacpp,trtllm` (any subset), or the shortcuts `make up-vllm` / `up-sglang` / `up-llamacpp` / `up-trtllm`. Unlisted engines never start (their VRAM stays free); the first engine listed serves the dashboard API, and switching sets stops the de-selected backends automatically. The dashboard shows the **whole fleet regardless of which backend serves the API** (nginx additionally pins the TRT engine-build endpoints to the TensorRT-LLM backend, so *Convert to TRT* works even when another engine serves the dashboard). ```bash curl http://localhost:8887/v1/models # router: configured model groups @@ -140,12 +141,13 @@ The **router only routes** — the **backend owns model lifecycle**. The fronten backend, and Grafana sit behind nginx on a single origin; backend, router, and Prometheus share one network namespace so the spawned vLLM instances are reachable on `localhost`. -### Mixed vLLM + SGLang + llama.cpp (`make up-mixed`) +### Mixed vLLM + SGLang + llama.cpp + TensorRT-LLM (`make up-mixed`) Each engine runs as its own backend container (they can't share a netns), sharing one Postgres (scheduling / desired intent), one router, one dashboard and one monitoring stack. Each backend publishes its ready instances as **routable addresses** to a shared file_sd that -Prometheus scrapes. See [docs/mixed-engine-deployment.md](docs/mixed-engine-deployment.md). +Prometheus scrapes (TensorRT-LLM is scraped at `/prometheus/metrics` and exposes `trtllm_*`). +See [docs/mixed-engine-deployment.md](docs/mixed-engine-deployment.md). ```mermaid flowchart LR @@ -168,37 +170,49 @@ flowchart LR BL["backend · :5073
NODE_ENGINES=llamacpp"] LINS["llama-server instances"] end + subgraph tbe["TensorRT-LLM backend (engine-trtllm.Dockerfile)"] + BT["backend · :5074
NODE_ENGINES=trtllm"] + TINS["trtllm-serve instances"] + end Client --> FE FE -->|/api| BV + FE -->|/api/trt| BT FE -->|/v1| RT FE -->|/grafana| GF BV -->|launch| VINS BS -->|launch| SINS BL -->|launch| LINS + BT -->|launch / build engine| TINS RT -->|route| VINS RT -->|route| SINS RT -->|route| LINS + RT -->|route| TINS BV <-->|leader/schedule| PG BS <-->|converge desired| PG BL <-->|converge desired| PG + BT <-->|converge desired| PG PR -->|scrape| VINS PR -->|scrape| SINS PR -->|scrape| LINS + PR -->|scrape /prometheus/metrics| TINS GF -->|query| PR ``` The leader's **engine-aware scheduler** places each model on a backend that can run its engine; a control action landing on the wrong node is deferred to the owning one. SGLang serves OpenMetrics, so Prometheus stores its metrics as `sglang_*` (underscore), while vLLM and -llama.cpp keep colons (`vllm:*`, `llamacpp:*`). +llama.cpp keep colons (`vllm:*`, `llamacpp:*`); TensorRT-LLM exposes `trtllm_*` at +`/prometheus/metrics`. Each engine has its own bundled Grafana dashboard and alert rules. ## Documentation | Topic | | |---|---| | Deployment & topology | [docs/deployment.md](docs/deployment.md) | -| Mixed-engine (vLLM + SGLang) | [docs/mixed-engine-deployment.md](docs/mixed-engine-deployment.md) | +| Mixed-engine (vLLM + SGLang + llama.cpp) | [docs/mixed-engine-deployment.md](docs/mixed-engine-deployment.md) | +| TensorRT-LLM engine (dual backend + Convert to TRT) | [docs/trtllm-launcher-impl-design_zh-TW.md](docs/trtllm-launcher-impl-design_zh-TW.md) | +| Adding a new engine | [docs/adding-a-new-engine_zh-TW.md](docs/adding-a-new-engine_zh-TW.md) | | Configuration (`config.yaml`) | [docs/configuration.md](docs/configuration.md) | | Features in depth | [docs/features.md](docs/features.md) | | Monitoring (Prometheus + Grafana) | [docs/monitoring.md](docs/monitoring.md) | diff --git a/README_zh-CN.md b/README_zh-CN.md index 8b8ddf3..3af70fd 100644 --- a/README_zh-CN.md +++ b/README_zh-CN.md @@ -30,7 +30,8 @@ ## 功能亮點 - **一個 router 掌控整個集群** — 單一 OpenAI 與 Anthropic 相容入口統管所有模型。以 `model` 欄位路由 `/v1/chat/completions`、`/v1/messages`、`/v1/embeddings`、`/v1/rerank`、`/v1/score`、`/tokenize` 等端點;router 自動解析群組並在實例間負載平衡,客戶端永遠不直接連到單一實例。 -- **三種推理引擎、同一個控制平面 — vLLM、SGLang 與 llama.cpp** — 每顆模型可各自選引擎(*新增模型*對話框有引擎選擇器);engine-aware 排程器把每顆模型擺到「跑得動它」的 backend 上,並由同一個 router/控制台/監控統一前置。可只跑 **vLLM**(`make up`),或跑**混合**集群(`make up-mixed`)加上 SGLang 與 llama.cpp(GGUF/CPU offload)backend。見 [docs/mixed-engine-deployment_zh-CN.md](docs/mixed-engine-deployment_zh-CN.md)。 +- **四種推理引擎、同一個控制平面 — vLLM、SGLang、llama.cpp 與 TensorRT-LLM** — 每顆模型可各自選引擎(*新增模型*對話框有引擎選擇器);engine-aware 排程器把每顆模型擺到「跑得動它」的 backend 上,並由同一個 router/控制台/監控統一前置。可只跑 **vLLM**(`make up`),或跑**混合**集群(`make up-mixed`)加上 SGLang、llama.cpp(GGUF/CPU offload)與 TensorRT-LLM backend。見 [docs/mixed-engine-deployment_zh-CN.md](docs/mixed-engine-deployment_zh-CN.md)。 +- **TensorRT-LLM,兩種跑法** — `trtllm` group 可跑**預先編譯的 TRT engine**(`engine_dir` → `--backend tensorrt`,最低 TTFT),或**直接吃 HF 權重**(不給 `engine_dir` → `--backend pytorch`,免編譯、支援 TRT-LLM 所有架構);兩者監控完全一致。可從 UI 建 engine:**模型庫**的 *轉成 TRT* 用 `trtllm-bench`(一條 arch-agnostic 路徑,非逐模型腳本)把已快取的 HF 模型編成 engine,再到 *新增模型* 挑它。見 [docs/trtllm-launcher-impl-design_zh-TW.md](docs/trtllm-launcher-impl-design_zh-TW.md)。 - **貼上 `vllm serve …` 即可新增模型** — 解析成表單、以動態 overlay 疊加;router 熱重載。 - **生命週期** — 每實例狀態機(`stopped → starting → ready → sleeping → failed`)、VRAM 預檢防呆、GPU 自動擺放、崩潰指數退避自動重啟。 - **自動擴縮(含暖待命層)** — 每群組保留 `min_ready` 暖機副本,依佇列深度擴容(優先喚醒、其次冷啟)到 `max_ready`;閒置時逐階縮回 `ready → sleep → stop`。vLLM **sleep mode**(level-1)釋放副本 VRAM 但秒級喚醒,所以縮容不必付出數分鐘冷啟代價。config.yaml 或控制台皆可設定,內建即時 Grafana 面板與告警。 @@ -42,7 +43,7 @@ - **生命週期告警** — 離散的模型事件(崩潰、退避用盡、復原)推到 Slack/Discord/通用 webhook,含每個 sink 自訂嚴重度門檻與 per-model 去重;用環境變數或 admin「通知」頁(含一鍵測試)設定。與 Grafana 指標告警互補。 - **Playground** — OpenAI 相容的 chat(串流)/completions/embeddings/reranking。 - **壓測與評測** — LLM 壓測(並發、到達率、SLA 自動調優)+ 30+ 個準確度資料集與 LLM-as-judge。 -- **資料庫** — 在 UI 瀏覽/預下載 HF 權重與資料集;工具調用 parser 助手;LoRA 支援。 +- **資料庫** — 在 UI 瀏覽/預下載 HF 權重與資料集;模型庫把已快取權重分成 **一般模型/GGUF/TensorRT-LLM 引擎** 三區,可從已快取模型建 TRT engine,*新增模型* 也能直接挑已下載的模型(與已建的 TRT engine)而非手打路徑;工具調用 parser 助手;LoRA 支援(含 PEFT→GGUF 轉換)。 - **多使用者與稽核** — 以具名 operator 憑證做角色控管(`viewer`/`operator`/`admin`),並有脫敏的**稽核日誌**記錄每次變更;另可發行/撤銷 API 金鑰,帶 per-key 用量歸屬、速率上限與 **token 額度**(總量/每日/每月)。env 管理員權杖與本機 dev 開放模式維持不變。 - **設定版本化與備份** — 動態模型 overlay(所有 runtime 改動所在)每次變更都會自動快照;可一鍵匯出成可攜檔備份、匯入還原,也能在 admin「設定版本」頁看歷史、並排 diff 與一鍵回滾到任一版。`config.yaml` 永遠不會被改寫。 @@ -63,8 +64,8 @@ make up # 建置並啟動整套服務 **兩種啟動方式:** - **`make up`** — 預設的**純 vLLM** 集群。 -- **`make up-mixed`** — **多引擎**混合集群:vLLM、SGLang 與 llama.cpp backend 共用同一顆 Postgres、router、控制台與 Grafana。從 *新增模型 → 引擎:`sglang` / `llamacpp`* 新增的模型會自動擺到跑得動它的 backend。對應 `make down-mixed`/`make logs-mixed`。見 [docs/mixed-engine-deployment_zh-CN.md](docs/mixed-engine-deployment_zh-CN.md)。 - - **可選要跑哪些引擎**:`make up-mixed ENGINES=vllm,sglang`(任意子集),或捷徑 `make up-vllm` / `up-sglang` / `up-llamacpp`。沒列到的引擎不會起(VRAM 留著);第一個引擎負責控制台 API,切換引擎組合時會自動停掉沒選到的 backend。 +- **`make up-mixed`** — **多引擎**混合集群:vLLM、SGLang、llama.cpp 與 TensorRT-LLM backend 共用同一顆 Postgres、router、控制台與 Grafana(預設四個全開;用 `ENGINES=…` 可拿掉不要的)。從 *新增模型 → 引擎:`sglang` / `llamacpp` / `trtllm`* 新增的模型會自動擺到跑得動它的 backend。對應 `make down-mixed`/`make logs-mixed`。見 [docs/mixed-engine-deployment_zh-CN.md](docs/mixed-engine-deployment_zh-CN.md)。 + - **可選要跑哪些引擎**:`make up-mixed ENGINES=vllm,sglang,llamacpp,trtllm`(任意子集),或捷徑 `make up-vllm` / `up-sglang` / `up-llamacpp` / `up-trtllm`。沒列到的引擎不會起(VRAM 留著);第一個引擎負責控制台 API,切換引擎組合時會自動停掉沒選到的 backend。控制台**不論由哪個 backend 服務都會顯示整個 fleet**(nginx 另外把 TRT engine 建置端點固定路由到 TensorRT-LLM backend,所以即使別的引擎在服務控制台,*轉成 TRT* 仍可用)。 ```bash curl http://localhost:8887/v1/models # router:列出設定的模型群組 @@ -194,7 +195,9 @@ leader 的 **engine-aware 排程器**把每顆模型擺到「跑得動它引擎 | 主題 | | |---|---| | 部署與架構 | [docs/deployment_zh-CN.md](docs/deployment_zh-CN.md) | -| 混合引擎(vLLM + SGLang) | [docs/mixed-engine-deployment_zh-CN.md](docs/mixed-engine-deployment_zh-CN.md) | +| 混合引擎(vLLM + SGLang + llama.cpp) | [docs/mixed-engine-deployment_zh-CN.md](docs/mixed-engine-deployment_zh-CN.md) | +| TensorRT-LLM 引擎(雙 backend + 轉成 TRT) | [docs/trtllm-launcher-impl-design_zh-TW.md](docs/trtllm-launcher-impl-design_zh-TW.md) | +| 新增一個引擎 | [docs/adding-a-new-engine_zh-TW.md](docs/adding-a-new-engine_zh-TW.md) | | 配置(`config.yaml`) | [docs/configuration_zh-CN.md](docs/configuration_zh-CN.md) | | 功能特色(詳細) | [docs/features_zh-CN.md](docs/features_zh-CN.md) | | 監控(Prometheus + Grafana) | [docs/monitoring_zh-CN.md](docs/monitoring_zh-CN.md) | diff --git a/apps/backend/app/api/engines.py b/apps/backend/app/api/engines.py new file mode 100644 index 0000000..c364fd5 --- /dev/null +++ b/apps/backend/app/api/engines.py @@ -0,0 +1,22 @@ +"""The registered inference engines and their capabilities. + +Single source of engine knowledge for the dashboard: the frontend reads this to +build the engine picker and to gate form fields on `capabilities` / +`inapplicable_keys`, instead of hardcoding engine names and re-deriving what each +engine can do (which drifts from the backend's launcher CAP_* flags). See +docs/multi-engine-extensibility-review_zh-TW.md §5 (P1). +""" +from fastapi import APIRouter, Depends + +from app.api.deps import get_manager +from app.llmops.manager import ModelManager + +router = APIRouter(prefix="/engines", tags=["engines"]) + + +@router.get("") +async def list_engines(manager: ModelManager = Depends(get_manager)): + """The registered LLM engines, each with: + name, capabilities (CAP_* strings), lora_endpoint_prefix, metric_prefix, + inapplicable_keys (greyed-out model_config keys), paste_example.""" + return {"engines": manager.engine_catalogue()} diff --git a/apps/backend/app/api/models.py b/apps/backend/app/api/models.py index 6424ee7..913e546 100644 --- a/apps/backend/app/api/models.py +++ b/apps/backend/app/api/models.py @@ -14,6 +14,7 @@ from app.api.schemas import ModelView from app.core.auth import require_operator from app.llmops.manager import ( + EngineUnavailable, GpuUnavailable, LoraRuntimeError, ModelAlreadyRunning, @@ -213,7 +214,7 @@ async def start_model(key: str, force: bool = False, manager: ModelManager = Dep raise HTTPException(status.HTTP_404_NOT_FOUND, f"unknown model: {key}") except ModelAlreadyRunning: raise HTTPException(status.HTTP_409_CONFLICT, f"model already running: {key}") - except (VRAMInsufficient, GpuUnavailable) as e: + except (VRAMInsufficient, GpuUnavailable, EngineUnavailable) as e: raise HTTPException(status.HTTP_409_CONFLICT, str(e)) diff --git a/apps/backend/app/api/trt.py b/apps/backend/app/api/trt.py new file mode 100644 index 0000000..972bef0 --- /dev/null +++ b/apps/backend/app/api/trt.py @@ -0,0 +1,90 @@ +"""TensorRT-LLM engine build ("convert to TRT") endpoints. Mirrors api/lora.py. + +Builds a pre-built TRT engine from an HF model via trtllm-bench (unified, +architecture-agnostic). A completed build lands under the engines root and is usable +by a trtllm group with engine_dir pointing at it (--backend tensorrt, lowest TTFT). +Only available on the TensorRT-LLM backend image (trtllm-bench present). +""" +from __future__ import annotations + +import asyncio +from typing import Optional + +from fastapi import APIRouter, Depends, HTTPException, Request, status +from pydantic import BaseModel, Field + +from app.core.auth import require_operator +from app.services import trt_convert +from app.services.trt_convert import BuildParams + +router = APIRouter(tags=["trt"]) + + +def _mgr(request: Request): + return request.app.state.trt_convert_manager + + +class TrtBuildRequest(BaseModel): + model_tag: str = Field(min_length=1) + tp_size: int = Field(default=1, ge=1) + pp_size: int = Field(default=1, ge=1) + max_seq_len: int = Field(default=2048, ge=1) + max_batch_size: int = Field(default=4, ge=1) + max_num_tokens: int = Field(default=8192, ge=1) + quantization: Optional[str] = None # one of trt_convert.QUANTIZATIONS, or null + + +@router.get("/trt/engines") +async def list_trt_engines(request: Request): + """Built engines on disk + whether the build tooling is available here + the + quantization options the UI can offer.""" + mgr = _mgr(request) + loop = asyncio.get_event_loop() + engines = await loop.run_in_executor(None, mgr.list_engines) + return { + "available": mgr.available(), + "root": trt_convert.engines_root(), + "quantizations": list(trt_convert.QUANTIZATIONS), + "engines": engines, + } + + +@router.get("/trt/conversions") +async def list_trt_conversions(request: Request): + mgr = _mgr(request) + return {"available": mgr.available(), "jobs": mgr.list()} + + +@router.post("/trt/convert", status_code=status.HTTP_202_ACCEPTED, + dependencies=[Depends(require_operator)]) +async def start_trt_convert(body: TrtBuildRequest, request: Request): + """Kick off an async HF->TRT-engine build. Returns the job (poll /trt/conversions).""" + mgr = _mgr(request) + if not mgr.available(): + raise HTTPException( + status.HTTP_503_SERVICE_UNAVAILABLE, + "TensorRT-LLM engine build tooling (trtllm-bench) isn't available on this backend", + ) + try: + params = BuildParams( + model_tag=body.model_tag, tp_size=body.tp_size, pp_size=body.pp_size, + max_seq_len=body.max_seq_len, max_batch_size=body.max_batch_size, + max_num_tokens=body.max_num_tokens, quantization=body.quantization, + ) + return mgr.start(params) + except ValueError as e: + raise HTTPException(status.HTTP_400_BAD_REQUEST, str(e)) + + +@router.delete("/trt/engines/{cache_key}", dependencies=[Depends(require_operator)]) +async def delete_trt_engine(cache_key: str, request: Request): + """Delete a built engine directory by its cache key (the engine dir name).""" + mgr = _mgr(request) + loop = asyncio.get_event_loop() + try: + existed = await loop.run_in_executor(None, mgr.delete_engine, cache_key) + except ValueError as e: + raise HTTPException(status.HTTP_400_BAD_REQUEST, str(e)) + if not existed: + raise HTTPException(status.HTTP_404_NOT_FOUND, f"no engine '{cache_key}'") + return {"deleted": cache_key} diff --git a/apps/backend/app/llmops/instance.py b/apps/backend/app/llmops/instance.py index d20086b..61b851b 100644 --- a/apps/backend/app/llmops/instance.py +++ b/apps/backend/app/llmops/instance.py @@ -35,6 +35,11 @@ class LaunchSpec: engine: str = "vllm" capabilities: frozenset = field(default_factory=frozenset) model_tag: Optional[str] = None + # The name the engine advertises at /v1/models (--served-model-name / --alias): + # `served_model_name` if set, else the model_tag. Used to verify a process's + # identity on boot adoption — when served_model_name is set, /v1/models shows ONLY + # this, not the model_tag. Defaults to model_tag when unset. + served_name: Optional[str] = None # True when launched with --enable-sleep-mode + VLLM_SERVER_DEV_MODE=1, so the # /sleep, /wake_up and /is_sleeping dev endpoints are available. sleep_enabled: bool = False diff --git a/apps/backend/app/llmops/launchers.py b/apps/backend/app/llmops/launchers.py index 522ceac..7a97b78 100644 --- a/apps/backend/app/llmops/launchers.py +++ b/apps/backend/app/llmops/launchers.py @@ -14,6 +14,7 @@ import json import logging import os +import shutil import sys import tempfile from typing import Protocol @@ -137,6 +138,7 @@ def build_vllm_cli_args(model_cfg: dict) -> list[str]: CAP_METRICS_VLLM = "metrics_vllm" # exposes vLLM-format Prometheus metrics (waiting queue, …) CAP_METRICS_SGLANG = "metrics_sglang" # exposes sglang:* Prometheus metrics (the router parses these) CAP_METRICS_LLAMACPP = "metrics_llamacpp" # exposes llamacpp:* Prometheus metrics (no kv-usage dim) +CAP_METRICS_TRTLLM = "metrics_trtllm" # exposes trtllm_* Prometheus metrics at /prometheus/metrics # Sentinel engine name for non-LLM launchers (embedding server): they aren't # selected by an engine choice, so they register under one fixed value. @@ -156,6 +158,16 @@ class Launcher(Protocol): # here so the LoRA caller reads it from the launcher instead of branching on the # engine name (Low#3). "" default. lora_endpoint_prefix: str = "" + # --- catalogue metadata (GET /api/engines) — the launcher is the single source of + # engine knowledge, so the frontend gates on these instead of hardcoding engine + # names/capabilities (P1). --- + # Prometheus metric-name prefix this engine exposes (e.g. "vllm" -> vllm:*). "" = none. + metric_prefix: str = "" + # model_config keys that don't apply to this engine (its CLI builder drops them); + # the dashboard greys them out. Empty = every key applies. + inapplicable_keys: frozenset[str] = frozenset() + # An example launch command for the Add-Model paste box. "" = no example. + paste_example: str = "" def keys(self, config) -> list[str]: """All instance keys this launcher defines in the config (its engine only).""" @@ -165,6 +177,14 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: """Resolve the LaunchSpec for one key.""" ... + def available(self) -> bool: + """Whether this engine's runtime is present in the running image. Optional + hook — callers treat a missing method as always-available, so only engines + that can be absent (TensorRT-LLM, whose trtllm-serve ships only in its own + image) need implement it. Lets the collapsed / non-trtllm backend reject a + trtllm model up-front with a clear error instead of crashing at spawn (§8.4).""" + ... + class VllmLauncher: kind = ModelKind.LLM @@ -173,6 +193,10 @@ class VllmLauncher: CAP_SLEEP, CAP_RUNTIME_LORA, CAP_LORA_MODULES, CAP_KV_TRANSFER, CAP_METRICS_VLLM, }) lora_endpoint_prefix = "/v1" # vLLM serves /v1/{load,unload}_lora_adapter + metric_prefix = "vllm" + paste_example = ("CUDA_VISIBLE_DEVICES=0 vllm serve Qwen/Qwen2.5-3B-Instruct " + "--port 8020 --dtype float16 --max-model-len 4096 " + "--gpu-memory-utilization 0.85") def keys(self, config) -> list[str]: out: list[str] = [] @@ -251,6 +275,7 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: port=inst.port, probe_url=f"http://{inst.host}:{inst.port}/health", model_tag=engine.settings.model_tag, + served_name=merged.get("served_model_name") or engine.settings.model_tag, sleep_enabled=sleep_enabled, ) @@ -389,6 +414,10 @@ class SglangLauncher: # See docs/multi-backend-engine-design_zh-CN.md §5.2. capabilities = frozenset({CAP_RUNTIME_LORA, CAP_LORA_MODULES, CAP_METRICS_SGLANG}) lora_endpoint_prefix = "" # SGLang serves /{load,unload}_lora_adapter (no /v1) + metric_prefix = "sglang" + paste_example = ("CUDA_VISIBLE_DEVICES=0 python -m sglang.launch_server " + "--model-path Qwen/Qwen3-0.6B --port 8030 --context-length 4096 " + "--mem-fraction-static 0.85") def keys(self, config) -> list[str]: out: list[str] = [] @@ -442,6 +471,7 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: port=inst.port, probe_url=f"http://{inst.host}:{inst.port}/health", model_tag=engine.settings.model_tag, + served_name=merged.get("served_model_name") or engine.settings.model_tag, ) @@ -551,6 +581,13 @@ class LlamacppLauncher: # metrics_llamacpp: launches with --metrics; the router normalizes llamacpp:* into # the same {waiting,running} load shape (no kv-usage dim). See §3/§10 of the design doc. capabilities = frozenset({CAP_LORA_MODULES, CAP_METRICS_LLAMACPP}) + metric_prefix = "llamacpp" + # These vLLM/SGLang knobs have no llama.cpp equivalent (the CLI builder drops + # them) — the dashboard greys them out. Sourced from _LLAMACPP_DROP_KEYS. + inapplicable_keys = _LLAMACPP_DROP_KEYS + paste_example = ("CUDA_VISIBLE_DEVICES=0 llama-server -hf " + "Qwen/Qwen2.5-0.5B-Instruct-GGUF:Q4_K_M -a qwen25-05b-gguf " + "-c 4096 -ngl 99 --port 8091") def keys(self, config) -> list[str]: out: list[str] = [] @@ -606,4 +643,203 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: port=inst.port, probe_url=f"http://{inst.host}:{inst.port}/health", model_tag=engine.settings.model_tag, + served_name=merged.get("served_model_name") or engine.settings.model_tag, + ) + + +# --- TensorRT-LLM (engine: trtllm) ------------------------------------------- +# Phase 1 "bring-your-own-engine": serve a PRE-BUILT TRT engine directory with +# `trtllm-serve serve --backend tensorrt`. The Phase 2 auto-build +# (convert_checkpoint + trtllm-build via a prepare() hook) is a separate step. +# See docs/trtllm-launcher-impl-design_zh-TW.md. + +# engine-neutral key -> trtllm-serve flag (single underscore, NOT kebab-case). +_TRTLLM_PARAM_MAP = { + "max_model_len": "max_seq_len", + "gpu_memory_utilization": "kv_cache_free_gpu_memory_fraction", + "tensor_parallel_size": "tp_size", + "pipeline_parallel_size": "pp_size", +} +# trtllm-native serve flags allowed to pass through as --. trtllm-serve is a +# strict click CLI (it errors on unknown args — e.g. enable_iter_perf_stats), so +# UNLIKE llama.cpp we do NOT pass arbitrary extra keys through; only this whitelist. +_TRTLLM_PASSTHROUGH = frozenset({ + "max_batch_size", "max_num_tokens", "kv_cache_dtype", + "trust_remote_code", "hf_revision", "chat_template", +}) +# vLLM/SGLang/llama.cpp knobs with no trtllm-serve equivalent (build-time or other +# engine) — greyed out in the dashboard (inapplicable_keys) and never emitted. +_TRTLLM_DROP_KEYS = frozenset({ + "n_gpu_layers", "gguf_quant", "model_file", "hf_file", + "enforce_eager", "dtype", "quantization", +}) +# Consumed specially by build_spec / build_trtllm_cli_args (addressing / served name / +# host / port / meta) — never emitted by the generic loop. +_TRTLLM_SKIP_CLI_KEYS = frozenset({ + "engine_dir", "tokenizer", "model_tag", "served_model_name", + "id", "cuda_device", "host", "port", + "enable_lora", "lora_modules", _LORA_RUNTIME_KEY, + "max_lora_rank", "max_loras", "lora_target_modules", "fully_sharded_loras", +}) | _ROUTER_ONLY_KEYS | _TRTLLM_DROP_KEYS + + +def build_trtllm_cli_args(model_cfg: dict, *, perf_yaml_path: str) -> list[str]: + """dict -> ``trtllm-serve`` CLI args (the part after ``trtllm-serve``). + + Two launch modes, chosen by whether ``engine_dir`` is set: + - ``engine_dir`` present -> ``serve --backend tensorrt`` (BYO a + pre-built TensorRT engine; lowest TTFT, but engine is GPU-arch/version-bound). + - ``engine_dir`` absent -> ``serve --backend pytorch`` (load HF + weights directly, no build step). ``model_tag`` is the positional. + The tokenizer isn't inside a tensorrt engine dir, so ``--tokenizer`` is always + emitted (``tokenizer`` override, else ``model_tag``); for pytorch it's a harmless + restatement of the model's own tokenizer. ``--served_model_name`` makes /v1/models + advertise a stable name (router forward_name + boot adopt identity). + + Engine-neutral keys map via ``_TRTLLM_PARAM_MAP`` (``max_model_len`` -> + ``--max_seq_len`` etc.); a small whitelist of trtllm-native flags passes through; + everything else is dropped (trtllm-serve rejects unknown args). ``--extra_llm_api_options`` + points at a per-instance YAML carrying ``return_perf_metrics: true`` — WITHOUT it + the server exposes no ``/prometheus/metrics`` (verified true for BOTH backends). + """ + model_tag = model_cfg.get("model_tag") + engine_dir = model_cfg.get("engine_dir") + if engine_dir: + positional, backend = str(engine_dir), "tensorrt" + elif model_tag: + positional, backend = str(model_tag), "pytorch" + else: + raise ValueError( + "trtllm model_config needs either 'engine_dir' (a pre-built TRT engine " + "directory -> --backend tensorrt) or 'model_tag' (HF weights -> " + "--backend pytorch)") + served = model_cfg.get("served_model_name") or model_tag + tokenizer = model_cfg.get("tokenizer") or model_tag + host = model_cfg.get("host") or "localhost" + port = model_cfg.get("port") + + args: list[str] = ["serve", positional, "--backend", backend, + "--host", str(host)] + if port is not None: + args += ["--port", str(port)] + if tokenizer: + args += ["--tokenizer", str(tokenizer)] + if served: + args += ["--served_model_name", str(served)] + + for key, value in model_cfg.items(): + if value is None or key in _TRTLLM_SKIP_CLI_KEYS: + continue + if key in _TRTLLM_PARAM_MAP: + flag = "--" + _TRTLLM_PARAM_MAP[key] + elif key in _TRTLLM_PASSTHROUGH: + flag = "--" + key + else: + continue # unknown -> drop (trtllm-serve is strict) + if isinstance(value, bool): + if value: # store_true: True -> present; False -> omit + args.append(flag) + else: + args += [flag, str(value)] + + args += ["--extra_llm_api_options", str(perf_yaml_path)] + return args + + +class TrtllmLauncher: + kind = ModelKind.LLM + engine = "trtllm" + # Serves either a pre-built TRT engine (engine_dir -> --backend tensorrt) or HF + # weights directly (model_tag only -> --backend pytorch) over OpenAI /v1. Both + # expose the same Prometheus trtllm_* metrics at /prometheus/metrics, so monitoring + # is identical. Capabilities are advertised conservatively for BOTH modes: metrics + # only — NO runtime LoRA (no /v1/load_lora_adapter) and NO kv_transfer. (Sleep is + # pytorch-backend-only and left as a follow-up so the dashboard doesn't offer a + # button that 500s on tensorrt-mode instances.) Static LoRA needs a YAML + # lora_config. See docs/trtllm-launcher-impl-design_zh-TW.md §7. + capabilities = frozenset({CAP_METRICS_TRTLLM}) + lora_endpoint_prefix = "" + metric_prefix = "trtllm" + inapplicable_keys = _TRTLLM_DROP_KEYS + paste_example = ("trtllm-serve serve /engines/qwen3-06b_tp1_fp16 --backend tensorrt " + "--tokenizer Qwen/Qwen3-0.6B --served_model_name qwen3-06b-trt " + "--host 0.0.0.0 --port 8050") + + def available(self) -> bool: + # trtllm-serve ships only in the dedicated TensorRT-LLM image; on the + # collapsed vLLM image which(...) is None, so start() rejects trtllm models + # up-front instead of crash-looping at spawn. See §8.4. + return shutil.which("trtllm-serve") is not None + + def keys(self, config) -> list[str]: + out: list[str] = [] + for model_tag, engine in config.LLM_engines.items(): + if getattr(engine.settings, "engine", "vllm") != self.engine: + continue + for inst in engine.instances: + out.append(f"{model_tag}::{inst.id}") + return out + + def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: + model_tag, _, instance_id = key.partition("::") + engine = config.LLM_engines.get(model_tag) + if engine is None: + raise KeyError(f"model group '{model_tag}' not in config") + inst = next((i for i in engine.instances if i.id == instance_id), None) + if inst is None: + raise KeyError(f"instance '{instance_id}' not in group '{model_tag}'") + + merged: dict = engine.settings.model_dump(by_alias=False) + merged.update(inst.model_dump()) + + env: dict[str, str] = {} + if merged.get("tensor_parallel_size", 1) == 1: + cuda_device = merged.pop("cuda_device", None) + if cuda_device is not None: + env["CUDA_VISIBLE_DEVICES"] = str(cuda_device) + merged.pop("id", None) + + # HA split deploys: bind to a routable interface (shares LLMOPS_VLLM_BIND_HOST + # with the other engines); local probe/record stay localhost. + bind_host = os.environ.get("LLMOPS_VLLM_BIND_HOST", "").strip() + cli_cfg = {**merged, "host": bind_host} if bind_host else merged + + # Per-instance YAML enabling Prometheus metrics. WITHOUT this the server has no + # /prometheus/metrics. The required keys differ by backend (both verified live): + # - tensorrt: `return_perf_metrics` ALONE gives the full trtllm_* set. Adding + # enable_iter_perf_stats makes the tensorrt backend abort at load + # ("_TrtLLM got invalid argument: enable_iter_perf_stats"). + # - pytorch: `return_perf_metrics` alone only yields request-level metrics + # (prompt/generation tokens, latency histograms); the iteration-level GAUGES + # the router + dashboard rely on (num_requests_running/waiting, + # kv_cache_utilization) need enable_iter_perf_stats too. + # engine_dir present => tensorrt mode; absent => pytorch mode (see cli builder). + is_pytorch = not merged.get("engine_dir") + perf_yaml_path = os.path.join(LOG_DIR, f"{model_tag}__{instance_id}.trtllm-perf.yaml") + try: + os.makedirs(LOG_DIR, exist_ok=True) + with open(perf_yaml_path, "w", encoding="utf-8") as f: + f.write("return_perf_metrics: true\n") + if is_pytorch: + f.write("enable_iter_perf_stats: true\n") + except OSError: + logger.warning("trtllm: could not write perf YAML at %s", perf_yaml_path) + + command = ["trtllm-serve"] + build_trtllm_cli_args(cli_cfg, perf_yaml_path=perf_yaml_path) + log_path = os.path.join(LOG_DIR, f"{model_tag}__{instance_id}.log") + # readiness: /health binds only after the engine loads (~35s for a small model, + # much longer for big engines) — the reconciler's not-yet-200 handling covers it. + return LaunchSpec( + key=key, + kind=self.kind, + engine=self.engine, + capabilities=self.capabilities, + command=command, + env=env, + log_path=log_path, + host=inst.host, + port=inst.port, + probe_url=f"http://{inst.host}:{inst.port}/health", + model_tag=engine.settings.model_tag, + served_name=merged.get("served_model_name") or engine.settings.model_tag, ) diff --git a/apps/backend/app/llmops/manager.py b/apps/backend/app/llmops/manager.py index 364e64a..1735f65 100644 --- a/apps/backend/app/llmops/manager.py +++ b/apps/backend/app/llmops/manager.py @@ -17,7 +17,7 @@ from app.llmops.events import emit_transition from app.llmops.instance import ModelInstance from app.llmops.launchers import CAP_RUNTIME_LORA, CAP_SLEEP, Launcher -from app.llmops.process import spawn_process, terminate_process_group +from app.llmops.process import free_port, spawn_process, terminate_process_group from app.llmops.registry import ModelRegistry from app.llmops.state import Desired, ModelKind, ModelState @@ -65,6 +65,12 @@ class GpuUnavailable(RuntimeError): """The pinned cuda_device doesn't exist on this host (pre-flight guard).""" +class EngineUnavailable(RuntimeError): + """This image lacks the engine's runtime binary (e.g. a trtllm model on the + collapsed vLLM image, which has no trtllm-serve). Rejected up-front instead of + crash-looping at spawn.""" + + class LoraRuntimeError(RuntimeError): """A runtime LoRA load/unload against a vLLM instance failed.""" @@ -75,8 +81,36 @@ class SleepError(RuntimeError): precondition failure (wrong state / not sleep-capable).""" +def unregistered_engine_groups(config, launchers: list[Launcher]) -> dict[str, str]: + """LLM groups whose `engine` no registered launcher claims — {group: engine}. + + The registered launchers are the single authority for which engines exist + (P1). A group with an engine no launcher handles is a *phantom*: schema-valid but + silently unrunnable — build_registry would create no instance for it, so it + vanishes from the backend while the router still lists it and black-holes its + traffic. Callers fail loud on a non-empty result instead of dropping it silently.""" + known = {l.engine for l in launchers} + out: dict[str, str] = {} + for group, engine in config.LLM_engines.items(): + name = getattr(engine.settings, "engine", "vllm") + if name not in known: + out[group] = name + return out + + def build_registry(config, config_path: str, launchers: list[Launcher]) -> ModelRegistry: - """Enumerate every instance every launcher defines, all STOPPED initially.""" + """Enumerate every instance every launcher defines, all STOPPED initially. + + Logs a loud error for any LLM group whose engine no launcher claims (a phantom + engine — see unregistered_engine_groups); it produces no instance, so surfacing it + here keeps it from disappearing silently.""" + phantom = unregistered_engine_groups(config, launchers) + if phantom: + logger.error( + "config has group(s) with an engine no launcher handles (they will NOT " + "run — add a launcher or fix the engine): %s", + ", ".join(f"{g} (engine={e})" for g, e in sorted(phantom.items())), + ) registry = ModelRegistry() for launcher in launchers: for key in launcher.keys(config): @@ -481,14 +515,42 @@ async def unschedulable_reasons(self) -> dict[str, str]: for key, want in desired.items(): if want != Desired.RUNNING.value: continue - group = key.split("::")[0] - engine = getattr( - getattr(self.config.LLM_engines.get(group), "settings", None), - "engine", "vllm") + # Resolve the engine from the registry (same source as the scheduler's + # _track_unschedulable), NOT config.LLM_engines — a non-LLM group like + # `embedding` isn't in LLM_engines and would fall back to "vllm", hiding the + # very case that most needs the flag (bespoke embedding can't be placed in a + # mixed fleet). A key absent from the registry (overlay not synced) is + # skipped rather than guessed. (verification gap #1) + inst = self.registry.get(key) + engine = getattr(inst, "engine", None) + if engine is None: + continue if not any(node_supports(n, engine) for n in nodes): out[key] = f"no live node runs engine '{engine}'" return out + def engine_catalogue(self) -> list[dict]: + """The registered LLM engines + their capability metadata — the single source + the dashboard (and any external consumer) reads instead of re-hardcoding which + engines exist and what each can do (P1). Deduped by engine name, sorted. Serves + GET /api/engines; the frontend gates its form on `capabilities` / + `inapplicable_keys` rather than engine-name conditionals.""" + out: list[dict] = [] + seen: set[str] = set() + for (kind, engine), launcher in self._launchers.items(): + if kind != ModelKind.LLM or engine in seen: + continue + seen.add(engine) + out.append({ + "name": engine, + "capabilities": sorted(launcher.capabilities), + "lora_endpoint_prefix": getattr(launcher, "lora_endpoint_prefix", ""), + "metric_prefix": getattr(launcher, "metric_prefix", ""), + "inapplicable_keys": sorted(getattr(launcher, "inapplicable_keys", frozenset())), + "paste_example": getattr(launcher, "paste_example", ""), + }) + return sorted(out, key=lambda e: e["name"]) + async def get(self, key: str) -> ModelInstance: return self._require(key) @@ -572,6 +634,14 @@ async def start(self, key: str, force: bool = False, reset_restart: bool = True) if key in await self.foreign_assignments(): return await self._defer_to_owner(inst, Desired.RUNNING) launcher = self._launcher_for(inst) + # Fail loud if this image lacks the engine's runtime (a trtllm model on the + # collapsed vLLM image has no trtllm-serve) — reject up-front instead of + # crash-looping at spawn. Optional launcher hook; absent = always available. + if not getattr(launcher, "available", lambda: True)(): + raise EngineUnavailable( + f"engine '{inst.engine}' runtime is not installed in this image; " + f"cannot start {key} — deploy it on the {inst.engine} backend image/profile." + ) # Re-resolve the spec (config may have changed) outside the lock so the # GPU pre-flight's nvidia-smi call never extends the critical section. @@ -608,6 +678,13 @@ async def start(self, key: str, force: bool = False, reset_restart: bool = True) # Spawn outside the lock — Popen returns immediately but still does IO. loop = asyncio.get_event_loop() + # Clear a teardown race on the port: a just-killed engine's detached worker + # processes (e.g. TensorRT-LLM's executor) may still hold the socket, which + # would make this spawn fail with EADDRINUSE and burn a restart. free_port + # kills any lingering holder. No-op when the port is already free (common case). + if not await loop.run_in_executor(None, free_port, spec.port): + logger.warning("port %s still in use after free_port; starting %s anyway", + spec.port, key) try: proc = await loop.run_in_executor(None, spawn_process, spec) except Exception as e: @@ -1229,6 +1306,14 @@ async def delete_overlay_model(self, key: str) -> None: await self.store.delete_assignment(key) except Exception: logger.warning("Failed to clear assignment for deleted %s", key, exc_info=True) + # Clear the shared observed row too, else fleet_views (HA/store-preferred + # mode) re-adds the just-deleted key from the store until the row's TTL + # expires (~30s) — the model lingers on the dashboard after delete. + if hasattr(self.store, "remove_instance_observed"): + try: + await self.store.remove_instance_observed(key) + except Exception: + logger.warning("Failed to clear observed for deleted %s", key, exc_info=True) logger.info("Deleted dynamic model %s", key) # -- Config export / import (versioning) ---------------------------------- diff --git a/apps/backend/app/llmops/process.py b/apps/backend/app/llmops/process.py index 9650517..dddf496 100644 --- a/apps/backend/app/llmops/process.py +++ b/apps/backend/app/llmops/process.py @@ -109,6 +109,97 @@ def terminate_process_group(proc: subprocess.Popen, timeout: float = 10.0) -> No pass +def port_is_free(port: int) -> bool: + """Whether 0.0.0.0:`port` can be bound right now (with SO_REUSEADDR, matching how + the engine servers bind). False if a live process still holds a listening socket.""" + import socket + + s = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + try: + s.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + s.bind(("0.0.0.0", int(port))) + return True + except OSError: + return False + finally: + s.close() + + +def _pids_holding_port(port: int) -> set[int]: + """PIDs with an open socket bound to `port` (any TCP state), via /proc. Same PID + namespace as the caller (the backend + the engine subprocesses it spawns).""" + import glob + + inodes: set[str] = set() + hexport = f"{int(port):04X}" + for f in ("/proc/net/tcp", "/proc/net/tcp6"): + try: + lines = open(f).read().splitlines()[1:] + except OSError: + continue + for ln in lines: + p = ln.split() + if len(p) < 10: + continue + # p[1] = local_address "HEXIP:HEXPORT"; p[9] = inode + if p[1].rsplit(":", 1)[-1].upper() == hexport: + inodes.add(p[9]) + if not inodes: + return set() + pids: set[int] = set() + for fd in glob.glob("/proc/[0-9]*/fd/*"): + try: + tgt = os.readlink(fd) + except OSError: + continue + if tgt.startswith("socket:[") and tgt[8:-1] in inodes: + try: + pids.add(int(fd.split("/", 3)[2])) + except (ValueError, IndexError): + pass + return pids + + +def free_port(port: int, timeout: float = 150.0) -> bool: + """Make `port` bindable, then return whether it is. Blocking; run in an executor. + + Two things can hold a just-killed engine's port: + 1. a *detached* worker process still holding the listening socket via an inherited + fd — notably TensorRT-LLM, whose executor workers reparent to init in their own + process group/session and so escape the process-group kill. We find these via + /proc and SIGKILL them. + 2. kernel connection-teardown states (FIN_WAIT2/TIME_WAIT) left by in-flight + connections (health probes, metric scrapes) when the server was SIGKILLed + rather than closed gracefully. These have no owning process and SO_REUSEADDR + does not override them, so the only remedy is to wait out the kernel timeout + (~60s). We poll until bindable. + Without this a crash/kill-triggered restart fails with EADDRINUSE and burns the + restart budget in a loop. This converts it into a single patient wait (safe: + start_timeout is far larger). No-op — returns True immediately — when the port is + already free, which is the graceful-stop common case.""" + import time + + if port_is_free(port): + return True + deadline = time.monotonic() + max(0.0, timeout) + logged = False + while True: + for pid in _pids_holding_port(port): + try: + os.kill(pid, signal.SIGKILL) + logger.warning("free_port %s: killed lingering holder pid %s", port, pid) + except (ProcessLookupError, PermissionError): + pass + if port_is_free(port): + return True + if time.monotonic() >= deadline: + return port_is_free(port) + if not logged: + logger.info("free_port %s: held (kernel socket teardown); waiting…", port) + logged = True + time.sleep(1.0) + + def kill_process_group(proc: subprocess.Popen) -> None: """SIGKILL the whole process group immediately, no graceful grace period. diff --git a/apps/backend/app/llmops/reconciler.py b/apps/backend/app/llmops/reconciler.py index 119204f..204a4a4 100644 --- a/apps/backend/app/llmops/reconciler.py +++ b/apps/backend/app/llmops/reconciler.py @@ -391,7 +391,12 @@ async def _serves_model(http_client, inst: ModelInstance) -> bool: except Exception: return False group = inst.key.split("::")[0] - wanted = {c for c in (inst.model_tag, group) if c} + # When served_model_name is set, /v1/models advertises ONLY it (not the model_tag), + # and all three launchers always pass --served-model-name/--alias. So match on the + # served name too, else a model with a custom served name is wrongly rejected on + # boot adoption → a second process spawns on its port → crash loop. (verification P0) + served = getattr(getattr(inst, "spec", None), "served_name", None) + wanted = {c for c in (inst.model_tag, group, served) if c} return bool(ids & wanted) diff --git a/apps/backend/app/main.py b/apps/backend/app/main.py index 1cb3cfc..8468f59 100644 --- a/apps/backend/app/main.py +++ b/apps/backend/app/main.py @@ -23,9 +23,11 @@ from app.api import datasets as dataset_routes from app.api import downloads as download_routes from app.api import embedding as embedding_routes +from app.api import engines as engine_routes from app.api import eval as eval_routes from app.api import lora as lora_routes from app.api import metrics as metrics_routes +from app.api import trt as trt_routes from app.api import models as model_routes from app.api import sso as sso_routes from app.api import perf as perf_routes @@ -36,7 +38,8 @@ from app.core.logging import setup_logging from app.core.settings import BackendSettings from app.core.store import LLMOpsStore -from app.llmops.launchers import EmbeddingLauncher, LlamacppLauncher, SglangLauncher, VllmLauncher +from app.llmops.launchers import (EmbeddingLauncher, LlamacppLauncher, SglangLauncher, + TrtllmLauncher, VllmLauncher) from app.llmops.manager import ModelManager, build_registry from app.llmops.autoscaler import autoscaler_loop from app.llmops.scheduler import Scheduler @@ -49,6 +52,7 @@ from app.services.dataset_downloads import DatasetDownloadManager from app.services.downloads import DownloadManager from app.services.lora_convert import LoraConvertManager +from app.services.trt_convert import TrtConvertManager from app.services.lora_downloads import LoraDownloadManager from app.services.gpu_service import get_gpu_processes_with_info from app.services.overlay import build_merged_config, hydrate_overlay_from_store, overlay_path @@ -126,7 +130,8 @@ async def lifespan(app: FastAPI): # Base config.yaml + dynamically-added models (overlay), merged into one view. config = build_merged_config(config_path) - launchers = [VllmLauncher(), SglangLauncher(), LlamacppLauncher(), EmbeddingLauncher()] + launchers = [VllmLauncher(), SglangLauncher(), LlamacppLauncher(), TrtllmLauncher(), + EmbeddingLauncher()] registry = build_registry(config, config_path, launchers) # `or` (not get's default) so an env var set-but-empty (as the compose env # passes it) still falls back instead of yielding "" — an empty router_url @@ -151,6 +156,7 @@ async def lifespan(app: FastAPI): app.state.dataset_download_manager = DatasetDownloadManager() app.state.lora_download_manager = LoraDownloadManager() app.state.lora_convert_manager = LoraConvertManager() + app.state.trt_convert_manager = TrtConvertManager() perf_root = os.path.join(os.path.dirname(store.db_path), "perf") app.state.perf_manager = PerfManager(store, manager, settings, perf_root, router_url) eval_root = os.path.join(os.path.dirname(store.db_path), "eval") @@ -304,6 +310,8 @@ def create_app() -> FastAPI: # Snapshots the overlay whenever a request changes it (for history/rollback). install_config_version_middleware(app) app.include_router(model_routes.router, prefix="/api") + app.include_router(engine_routes.router, prefix="/api") + app.include_router(trt_routes.router, prefix="/api") app.include_router(system_routes.router, prefix="/api") app.include_router(config_routes.router, prefix="/api") app.include_router(observability_routes.router, prefix="/api") diff --git a/apps/backend/app/services/trt_convert.py b/apps/backend/app/services/trt_convert.py new file mode 100644 index 0000000..81e84a3 --- /dev/null +++ b/apps/backend/app/services/trt_convert.py @@ -0,0 +1,297 @@ +"""Build a TensorRT-LLM engine from an HF model, for the trtllm engine's tensorrt +backend (engine_dir mode). The model library's "convert to TRT" action. + +Mirrors lora_convert: an async job manager wrapping a blocking subprocess. Here the +subprocess is ``trtllm-bench -m -w build ...`` — the UNIFIED, +architecture-agnostic engine builder. It auto-detects the model architecture (unlike +the legacy per-arch ``convert_checkpoint.py`` scripts), so ONE code path covers every +architecture TensorRT-LLM supports. The pre-built engine gives the lowest TTFT, but is +bound to the GPU architecture + TRT-LLM version, so the cache key + manifest record +both and a mismatch forces a rebuild. + +Only the TensorRT-LLM image ships ``trtllm-bench``; ``bench_available()`` gates on it, +so on other backends the feature is simply unavailable (like GGUF conversion is). + +Engines land under the shared engines root (see ``engines_root()`` — by default a +``trt-engines/`` dir in the same HF cache the Model Library uses, overridable with +``LLMOPS_TRT_ENGINES_DIR``), so a completed build is cached alongside the downloaded +models and immediately usable by a trtllm group whose ``engine_dir`` points at it. +""" +from __future__ import annotations + +import asyncio +import glob +import json +import logging +import os +import re +import shutil +import subprocess +import time +from dataclasses import asdict, dataclass, field +from typing import Optional + +logger = logging.getLogger(__name__) + +# Quantization algorithms `trtllm-bench build -q` accepts. None = no quantization. +QUANTIZATIONS = ( + "W8A16", "W4A16", "W4A16_AWQ", "W4A8_AWQ", "W4A16_GPTQ", "FP8", "INT8", "NVFP4", +) +# Cap the build subprocess; large models build for minutes, but never hang forever. +BUILD_TIMEOUT_S = 3600.0 + + +def engines_root() -> str: + """Where built engines live. Defaults to a ``trt-engines/`` sibling of the HF hub + cache — i.e. the SAME host mount as the Model Library's downloaded weights + (HF_HOME=/hf -> ~/.cache/huggingface), so engines are cached alongside models and + visible to every backend. It sits beside ``hub/`` (not inside it) so + huggingface_hub's scan_cache_dir ignores it. Override with LLMOPS_TRT_ENGINES_DIR.""" + override = os.environ.get("LLMOPS_TRT_ENGINES_DIR") + if override: + return override + try: + from huggingface_hub.constants import HF_HUB_CACHE + return os.path.join(os.path.dirname(HF_HUB_CACHE), "trt-engines") + except Exception: + return "/engines" + + +def bench_available() -> bool: + """Whether trtllm-bench is present (only the TensorRT-LLM image ships it).""" + return shutil.which("trtllm-bench") is not None + + +def _trtllm_version() -> str: + try: + from importlib.metadata import version + return version("tensorrt_llm") + except Exception: + return "unknown" + + +def _gpu_compute_cap() -> str: + """SM compute capability of GPU 0 as digits (e.g. '86'). Empty if undeterminable. + Part of the cache key because a serialized engine can't deserialize on a different + GPU architecture.""" + try: + out = subprocess.run( + ["nvidia-smi", "--query-gpu=compute_cap", "--format=csv,noheader"], + capture_output=True, text=True, timeout=10, + ) + first = (out.stdout or "").strip().splitlines()[0].strip() + return first.replace(".", "") # "8.6" -> "86" + except Exception: + return "" + + +def _sanitize(s: str) -> str: + return re.sub(r"[^A-Za-z0-9._-]", "_", s.strip()) + + +def _dir_size(path: str) -> int: + """Total bytes of files directly under `path` (engine dirs are flat).""" + total = 0 + try: + with os.scandir(path) as it: + for e in it: + if e.is_file(follow_symlinks=False): + total += e.stat().st_size + except OSError: + pass + return total + + +@dataclass(frozen=True) +class BuildParams: + """The knobs that determine (and identify) a built engine. Anything that changes + the engine bytes belongs in the cache key.""" + model_tag: str + tp_size: int = 1 + pp_size: int = 1 + max_seq_len: int = 2048 + max_batch_size: int = 4 + max_num_tokens: int = 8192 + quantization: Optional[str] = None # one of QUANTIZATIONS, or None + + def cache_key(self) -> str: + q = (self.quantization or "none").lower() + return "_".join([ + _sanitize(self.model_tag), + f"tp{self.tp_size}pp{self.pp_size}", + f"seq{self.max_seq_len}", f"bs{self.max_batch_size}", f"nt{self.max_num_tokens}", + q, + f"sm{_gpu_compute_cap() or 'x'}", + f"trt{_sanitize(_trtllm_version())}", + ]) + + +def engine_dir_for(params: BuildParams) -> str: + return os.path.join(engines_root(), params.cache_key()) + + +def is_built(params: BuildParams) -> bool: + d = engine_dir_for(params) + return os.path.isfile(os.path.join(d, "manifest.json")) and bool( + glob.glob(os.path.join(d, "rank*.engine"))) + + +def _validate(params: BuildParams) -> None: + if not (params.model_tag or "").strip(): + raise ValueError("model_tag is required") + if params.quantization is not None and params.quantization not in QUANTIZATIONS: + raise ValueError( + f"unknown quantization '{params.quantization}'; expected one of {list(QUANTIZATIONS)}") + for name in ("tp_size", "pp_size", "max_seq_len", "max_batch_size", "max_num_tokens"): + if int(getattr(params, name)) < 1: + raise ValueError(f"{name} must be >= 1") + + +def build(params: BuildParams) -> str: + """Build (or reuse) the engine for ``params``; return its engine_dir. Blocking — + call via run_in_executor. + + Runs ``trtllm-bench build`` into a temp workspace, moves the produced engine dir + (the one holding ``rank*.engine``) to a flat ``//`` and + writes ``manifest.json``. Idempotent: a valid existing build is returned as-is. + + trtllm-bench inherits the process env; the backend is launched via a login shell + (compose ``command: bash -lc``), so LD_LIBRARY_PATH for the TRT libs is already set + — same reason the launcher can spawn trtllm-serve directly. + """ + if not bench_available(): + raise RuntimeError( + "trtllm-bench is not available on this backend (only the TensorRT-LLM image ships it)") + _validate(params) + final_dir = engine_dir_for(params) + if is_built(params): + logger.info("trt engine already built, reusing: %s", final_dir) + return final_dir + + ws = os.path.join(engines_root(), ".trt-build", params.cache_key()) + shutil.rmtree(ws, ignore_errors=True) + os.makedirs(ws, exist_ok=True) + cmd = [ + "trtllm-bench", "-m", params.model_tag, "-w", ws, "build", + "--tp_size", str(params.tp_size), "--pp_size", str(params.pp_size), + "--max_seq_len", str(params.max_seq_len), + "--max_batch_size", str(params.max_batch_size), + "--max_num_tokens", str(params.max_num_tokens), + ] + if params.quantization: + cmd += ["--quantization", params.quantization] + + logger.info("Building TRT engine [%s]: %s", params.cache_key(), " ".join(cmd)) + proc = subprocess.run(cmd, capture_output=True, text=True, timeout=BUILD_TIMEOUT_S) + if proc.returncode != 0: + shutil.rmtree(ws, ignore_errors=True) + tail = (proc.stderr or proc.stdout or "").strip()[-800:] + raise RuntimeError(f"trtllm-bench build failed (rc={proc.returncode}): {tail}") + + produced = glob.glob(os.path.join(ws, "**", "rank0.engine"), recursive=True) + if not produced: + shutil.rmtree(ws, ignore_errors=True) + raise RuntimeError("trtllm-bench build reported success but produced no engine") + engine_src = os.path.dirname(produced[0]) + + shutil.rmtree(final_dir, ignore_errors=True) + os.makedirs(os.path.dirname(final_dir), exist_ok=True) + shutil.move(engine_src, final_dir) + shutil.rmtree(ws, ignore_errors=True) + try: # drop the now-empty .trt-build/ parent so it doesn't clutter the cache + os.rmdir(os.path.dirname(ws)) + except OSError: + pass + + manifest = { + "model_tag": params.model_tag, + "trtllm_version": _trtllm_version(), + "compute_capability": _gpu_compute_cap(), + "tp_size": params.tp_size, "pp_size": params.pp_size, + "max_seq_len": params.max_seq_len, "max_batch_size": params.max_batch_size, + "max_num_tokens": params.max_num_tokens, "quantization": params.quantization, + "cache_key": params.cache_key(), "built_at": time.time(), + } + with open(os.path.join(final_dir, "manifest.json"), "w", encoding="utf-8") as f: + json.dump(manifest, f, indent=2) + logger.info("Built TRT engine -> %s", final_dir) + return final_dir + + +@dataclass +class TrtBuildJob: + key: str # cache key (also the engine dir name) + params: dict # BuildParams as a dict, for the UI + state: str = "pending" # pending | building | completed | failed + engine_dir: Optional[str] = None + error: Optional[str] = None + started_at: float = field(default_factory=time.time) + updated_at: float = field(default_factory=time.time) + + +class TrtConvertManager: + """In-memory HF->TRT-engine build jobs, keyed by cache key. Mirrors LoraConvertManager.""" + + def __init__(self) -> None: + self._jobs: dict[str, TrtBuildJob] = {} + self._tasks: dict[str, asyncio.Task] = {} + + def available(self) -> bool: + return bench_available() + + def list(self) -> list[dict]: + return [asdict(j) for j in sorted(self._jobs.values(), key=lambda j: j.started_at, reverse=True)] + + def list_engines(self) -> list[dict]: + """Built engines on disk (a manifest.json under the engines root). Blocking.""" + out: list[dict] = [] + for mf in glob.glob(os.path.join(engines_root(), "*", "manifest.json")): + try: + with open(mf, encoding="utf-8") as f: + m = json.load(f) + except (OSError, json.JSONDecodeError): + continue + d = os.path.dirname(mf) + m["engine_dir"] = d + m["size_on_disk"] = _dir_size(d) + out.append(m) + return sorted(out, key=lambda m: m.get("built_at", 0), reverse=True) + + def delete_engine(self, cache_key: str) -> bool: + """Delete a built engine directory by its cache key. Returns whether it existed. + The key is a single path segment (no traversal) resolved under the engines root.""" + if not cache_key or "/" in cache_key or "\\" in cache_key or cache_key in (".", ".."): + raise ValueError("invalid engine key") + d = os.path.join(engines_root(), cache_key) + if not os.path.isfile(os.path.join(d, "manifest.json")): + return False + shutil.rmtree(d, ignore_errors=True) + logger.info("Deleted TRT engine %s", cache_key) + return True + + def start(self, params: BuildParams) -> dict: + _validate(params) + key = params.cache_key() + existing = self._jobs.get(key) + if existing and existing.state in ("pending", "building"): + return asdict(existing) # already in flight — idempotent + job = TrtBuildJob(key=key, params=asdict(params)) + self._jobs[key] = job + self._tasks[key] = asyncio.create_task(self._run(job, params)) + return asdict(job) + + async def _run(self, job: TrtBuildJob, params: BuildParams) -> None: + loop = asyncio.get_event_loop() + job.state = "building" + job.updated_at = time.time() + try: + job.engine_dir = await loop.run_in_executor(None, build, params) + job.state = "completed" + logger.info("Built TRT engine %s -> %s", job.key, job.engine_dir) + except Exception as e: + job.state = "failed" + job.error = str(e) + logger.warning("TRT engine build failed for %s: %s", job.key, e) + finally: + job.updated_at = time.time() + self._tasks.pop(job.key, None) diff --git a/apps/backend/tests/api/test_engines_routes.py b/apps/backend/tests/api/test_engines_routes.py new file mode 100644 index 0000000..4454302 --- /dev/null +++ b/apps/backend/tests/api/test_engines_routes.py @@ -0,0 +1,19 @@ +import pytest + +pytestmark = pytest.mark.api + + +def test_list_engines_returns_registered_launchers(client): + # The catalogue reflects the launchers the manager actually registered (P1) — the + # test app registers vLLM. The phantom trtllm (no launcher) never appears. + r = client.get("/api/engines") + assert r.status_code == 200 + engines = {e["name"]: e for e in r.json()["engines"]} + assert "vllm" in engines + assert "trtllm" not in engines + vllm = engines["vllm"] + # Capability metadata + example command drive the frontend's gating (no hardcoding). + assert "sleep" in vllm["capabilities"] + assert vllm["metric_prefix"] == "vllm" + assert vllm["lora_endpoint_prefix"] == "/v1" + assert vllm["paste_example"] diff --git a/apps/backend/tests/api/test_trt_routes.py b/apps/backend/tests/api/test_trt_routes.py new file mode 100644 index 0000000..5c90aaf --- /dev/null +++ b/apps/backend/tests/api/test_trt_routes.py @@ -0,0 +1,76 @@ +"""TRT engine build API (app/api/trt.py).""" +import pytest + +from app.services import trt_convert as T + +pytestmark = pytest.mark.api + + +def test_list_engines_reports_availability_and_quantizations(client, monkeypatch): + monkeypatch.setattr(T, "bench_available", lambda: True) + r = client.get("/api/trt/engines") + assert r.status_code == 200 + body = r.json() + assert body["available"] is True + assert "FP8" in body["quantizations"] + assert isinstance(body["engines"], list) + + +def test_convert_unavailable_returns_503(auth_client, monkeypatch): + monkeypatch.setattr(T, "bench_available", lambda: False) + r = auth_client.post("/api/trt/convert", json={"model_tag": "Qwen/Qwen3-0.6B"}, + headers={"Authorization": "Bearer secret-admin"}) + assert r.status_code == 503 + + +def test_convert_requires_operator(auth_client): + # Auth is on (auth_client sets admin_token) and no token is sent -> the + # require_operator dependency rejects before reaching the manager. + r = auth_client.post("/api/trt/convert", json={"model_tag": "Qwen/Qwen3-0.6B"}) + assert r.status_code in (401, 403) + + +def test_convert_starts_job_and_passes_params(auth_client, monkeypatch): + monkeypatch.setattr(T, "bench_available", lambda: True) + monkeypatch.setattr(T, "_gpu_compute_cap", lambda: "86") + monkeypatch.setattr(T, "_trtllm_version", lambda: "1.3.0rc20") + captured = {} + + def fake_start(self, params): + captured["params"] = params + return {"key": params.cache_key(), "params": {}, "state": "pending"} + + monkeypatch.setattr(T.TrtConvertManager, "start", fake_start) + r = auth_client.post( + "/api/trt/convert", + json={"model_tag": "Qwen/Qwen3-0.6B", "max_batch_size": 8, + "max_num_tokens": 16384, "quantization": "FP8"}, + headers={"Authorization": "Bearer secret-admin"}, + ) + assert r.status_code == 202 + p = captured["params"] + assert p.model_tag == "Qwen/Qwen3-0.6B" and p.max_batch_size == 8 + assert p.max_num_tokens == 16384 and p.quantization == "FP8" + + +def test_delete_engine_requires_operator(auth_client): + r = auth_client.delete("/api/trt/engines/somekey") + assert r.status_code in (401, 403) + + +def test_delete_missing_engine_returns_404(auth_client): + r = auth_client.delete("/api/trt/engines/does-not-exist", + headers={"Authorization": "Bearer secret-admin"}) + assert r.status_code == 404 + + +def test_convert_rejects_bad_quantization(auth_client, monkeypatch): + monkeypatch.setattr(T, "bench_available", lambda: True) + monkeypatch.setattr(T, "_gpu_compute_cap", lambda: "86") + monkeypatch.setattr(T, "_trtllm_version", lambda: "1.3.0rc20") + r = auth_client.post( + "/api/trt/convert", + json={"model_tag": "Qwen/Qwen3-0.6B", "quantization": "BOGUS"}, + headers={"Authorization": "Bearer secret-admin"}, + ) + assert r.status_code == 400 diff --git a/apps/backend/tests/conftest.py b/apps/backend/tests/conftest.py index c4e115f..4739f1e 100644 --- a/apps/backend/tests/conftest.py +++ b/apps/backend/tests/conftest.py @@ -572,6 +572,8 @@ def app(monkeypatch): application.state.lora_download_manager = LoraDownloadManager() from app.services.lora_convert import LoraConvertManager application.state.lora_convert_manager = LoraConvertManager() + from app.services.trt_convert import TrtConvertManager + application.state.trt_convert_manager = TrtConvertManager() application.state.perf_manager = PerfManager( store, application.state.manager, settings, str(BACKEND_ROOT), "http://127.0.0.1:8887" ) diff --git a/apps/backend/tests/unit/test_ha_safety.py b/apps/backend/tests/unit/test_ha_safety.py index 82ee95d..d5be6c3 100644 --- a/apps/backend/tests/unit/test_ha_safety.py +++ b/apps/backend/tests/unit/test_ha_safety.py @@ -5,8 +5,8 @@ from app.core.settings import BackendSettings from app.llmops.launchers import EmbeddingLauncher, VllmLauncher -from app.llmops.manager import (ModelAlreadyRunning, ModelConflict, ModelManager, - build_registry) +from app.llmops.manager import (EngineUnavailable, ModelAlreadyRunning, + ModelConflict, ModelManager, build_registry) from app.llmops.state import Desired, ModelState from schema import load_config @@ -42,6 +42,9 @@ def __init__(self): async def list_instance_observed(self, ts=None): return list(self.observed) + async def remove_instance_observed(self, key): + self.observed = [o for o in self.observed if o.get("key") != key] + async def list_instance_desired(self): return dict(self.desired) @@ -71,8 +74,11 @@ def _manager(tmp_path, store=None, node_engines=None, instance_id="node-A"): config = load_config(str(cfg_path)) launchers = [VllmLauncher(), EmbeddingLauncher()] registry = build_registry(config, str(cfg_path), launchers) + # vram_guard=False so start()'s _vram_preflight doesn't hit the real GPU and raise + # VRAMInsufficient when the host GPU is busy — these tests assert control-flow, not + # placement. settings = BackendSettings(instance_id=instance_id, - node_engines=node_engines or []) + node_engines=node_engines or [], vram_guard=False) mgr = ModelManager( registry, launchers, None, config, str(cfg_path), settings, store=store, overlay_path=str(overlay_path), @@ -145,6 +151,20 @@ async def test_edit_allowed_when_stopped_everywhere(tmp_path): # ---- High#1: stop/start race -------------------------------------------- +async def test_delete_overlay_model_clears_observed(tmp_path): + """Deleting an overlay model must clear its shared observed row — else fleet_views + (store-preferred mode) re-adds the deleted key from the store until the row's TTL + expires, so it lingers on the dashboard ~30s after delete.""" + store = FakeStore() + mgr, _ = _manager(tmp_path, store=store) + await mgr.create_overlay_model( + "NewGroup", {"id": "x", "host": "localhost", "port": 9099}, {"model_tag": "org/new"}) + # The owning node had backfilled an observed row for it. + store.observed = [{"key": "NewGroup::x", "node_id": "node-A", "state": "stopped"}] + await mgr.delete_overlay_model("NewGroup::x") + assert all(o["key"] != "NewGroup::x" for o in store.observed) + + async def test_start_rejected_while_stopping(tmp_path): mgr, _ = _manager(tmp_path) inst = mgr.registry.get("Qwen3-0.6B::a") @@ -153,6 +173,17 @@ async def test_start_rejected_while_stopping(tmp_path): await mgr.start("Qwen3-0.6B::a") +async def test_start_rejected_when_engine_runtime_missing(tmp_path): + """start() must reject up-front (not crash at spawn) when the launcher reports + its engine runtime is absent from this image (e.g. a trtllm model on the + collapsed vLLM image whose trtllm-serve is missing).""" + mgr, _ = _manager(tmp_path) + launcher = mgr._launcher_for(mgr.registry.get("Qwen3-0.6B::a")) + launcher.available = lambda: False # simulate the runtime not being installed + with pytest.raises(EngineUnavailable): + await mgr.start("Qwen3-0.6B::a") + + async def test_stop_cas_finalize_does_not_clobber_new_process(tmp_path): """stop()'s finalizer must not reset state to STOPPED when a newer process (different proc handle) has taken the slot during the drain window.""" @@ -346,3 +377,40 @@ async def test_unschedulable_empty_in_collapsed_mode(tmp_path): # No store / SQLite (db_url None) -> nothing computed (single host runs everything). mgr, _ = _manager(tmp_path, store=None) assert await mgr.unschedulable_reasons() == {} + + +EMBED_CONFIG_YAML = CONFIG_YAML + """ +embedding_server: + host: localhost + port: 8005 + cuda_device: 0 + embedding_models: + m3e: + model_name: moka-ai/m3e-base + max_length: 512 + use_gpu: true + reranking_models: + bge: + model_name: BAAI/bge-reranker-large + max_length: 512 +""" + + +async def test_unschedulable_uses_registry_engine_for_embedding(tmp_path): + # embedding's engine is the sentinel 'default'; no node advertises it, so it must + # be flagged even when a vLLM node exists. The old code resolved engine from + # config.LLM_engines (no 'embedding' group) and fell back to vllm, hiding it. (#1) + cfg = tmp_path / "config.yaml" + cfg.write_text(EMBED_CONFIG_YAML, encoding="utf-8") + config = load_config(str(cfg)) + launchers = [VllmLauncher(), EmbeddingLauncher()] + registry = build_registry(config, str(cfg), launchers) + store = FakeStore() + store.desired = {"embedding::default": Desired.RUNNING.value} + store.nodes = [{"node_id": "n1", "engines": '["vllm"]'}] + mgr = ModelManager(registry, launchers, None, config, str(cfg), + BackendSettings(instance_id="node-A"), store=store, + overlay_path=str(tmp_path / "o.json")) + reasons = await mgr.unschedulable_reasons() + assert "embedding::default" in reasons + assert "default" in reasons["embedding::default"] diff --git a/apps/backend/tests/unit/test_launchers.py b/apps/backend/tests/unit/test_launchers.py index c9a41a5..ed822f4 100644 --- a/apps/backend/tests/unit/test_launchers.py +++ b/apps/backend/tests/unit/test_launchers.py @@ -4,11 +4,12 @@ from app.llmops.launchers import (CAP_KV_TRANSFER, CAP_LORA_MODULES, CAP_METRICS_LLAMACPP, CAP_METRICS_SGLANG, - CAP_RUNTIME_LORA, CAP_SLEEP, EMBEDDING_KEY, - ENGINE_DEFAULT, EmbeddingLauncher, LlamacppLauncher, - SglangLauncher, VllmLauncher, _write_effective_config, + CAP_METRICS_TRTLLM, CAP_RUNTIME_LORA, CAP_SLEEP, + EMBEDDING_KEY, ENGINE_DEFAULT, EmbeddingLauncher, + LlamacppLauncher, SglangLauncher, TrtllmLauncher, + VllmLauncher, _write_effective_config, build_llamacpp_cli_args, build_sglang_cli_args, - build_vllm_cli_args) + build_trtllm_cli_args, build_vllm_cli_args) from app.llmops.state import ModelKind from schema import RootConfig from tests.conftest import FAKE_CONFIG @@ -620,3 +621,132 @@ def test_llamacpp_bind_host_env_overrides_only_the_bind_address(monkeypatch): assert spec.command[spec.command.index("--host") + 1] == "0.0.0.0" # binds all assert spec.host == "localhost" # record unchanged assert spec.probe_url == "http://localhost:8100/health" # local probe unchanged + + +# ---- TensorRT-LLM launcher (docs/trtllm-launcher-impl-design_zh-TW.md) -------- + +def _trtllm_config(extra: dict | None = None) -> RootConfig: + mc = {"model_tag": "Qwen/Qwen3-0.6B", "engine": "trtllm", + "engine_dir": "/engines/qwen3-06b_tp1_fp16"} + if extra: + mc.update(extra) + return RootConfig.model_validate({ + "server": {"host": "0.0.0.0", "port": 8887}, + "LLM_engines": {"T": { + "instances": [{"id": "a", "host": "localhost", "port": 8050, "cuda_device": 0}], + "model_config": mc, + }}, + }) + + +def test_trtllm_args_engine_dir_positional_and_tensorrt_backend(): + args = build_trtllm_cli_args( + {"model_tag": "Qwen/Qwen3-0.6B", "engine_dir": "/engines/q3", + "port": 8050, "host": "0.0.0.0"}, + perf_yaml_path="/logs/T__a.trtllm-perf.yaml") + assert args[:4] == ["serve", "/engines/q3", "--backend", "tensorrt"] + # tokenizer defaults to model_tag; served name defaults to model_tag + assert args[args.index("--tokenizer") + 1] == "Qwen/Qwen3-0.6B" + assert args[args.index("--served_model_name") + 1] == "Qwen/Qwen3-0.6B" + assert args[args.index("--host") + 1] == "0.0.0.0" + assert args[args.index("--port") + 1] == "8050" + # perf YAML always passed (else no /prometheus/metrics) + assert args[args.index("--extra_llm_api_options") + 1] == "/logs/T__a.trtllm-perf.yaml" + + +def test_trtllm_args_param_map_and_passthrough_and_drops(): + args = build_trtllm_cli_args( + {"engine_dir": "/e", "model_tag": "m", + "max_model_len": 2048, "gpu_memory_utilization": 0.4, + "tensor_parallel_size": 2, "max_batch_size": 4, + "n_gpu_layers": 99, "enforce_eager": True, "dtype": "float16", # all dropped + "some_unknown_flag": "x"}, # dropped (strict CLI) + perf_yaml_path="/p.yaml") + assert args[args.index("--max_seq_len") + 1] == "2048" + assert args[args.index("--kv_cache_free_gpu_memory_fraction") + 1] == "0.4" + assert args[args.index("--tp_size") + 1] == "2" + assert args[args.index("--max_batch_size") + 1] == "4" + for dropped in ("--n_gpu_layers", "--enforce_eager", "--dtype", "--some_unknown_flag", + "--n-gpu-layers"): + assert dropped not in args + + +def test_trtllm_args_pytorch_backend_from_model_tag(): + # No engine_dir -> serve the HF model_tag with --backend pytorch. + args = build_trtllm_cli_args( + {"model_tag": "Qwen/Qwen3-0.6B", "port": 8061, "host": "0.0.0.0", + "max_model_len": 2048}, + perf_yaml_path="/logs/T__a.trtllm-perf.yaml") + assert args[:4] == ["serve", "Qwen/Qwen3-0.6B", "--backend", "pytorch"] + assert args[args.index("--tokenizer") + 1] == "Qwen/Qwen3-0.6B" + assert args[args.index("--served_model_name") + 1] == "Qwen/Qwen3-0.6B" + assert args[args.index("--max_seq_len") + 1] == "2048" + # metrics YAML still passed -> monitoring identical to tensorrt mode + assert args[args.index("--extra_llm_api_options") + 1] == "/logs/T__a.trtllm-perf.yaml" + + +def test_trtllm_args_requires_engine_dir_or_model_tag(): + with pytest.raises(ValueError, match="engine_dir.*model_tag|model_tag"): + build_trtllm_cli_args({}, perf_yaml_path="/p.yaml") + + +def test_trtllm_build_spec_command_and_capabilities(tmp_path, monkeypatch): + import app.llmops.launchers as L + monkeypatch.setattr(L, "LOG_DIR", str(tmp_path)) + spec = TrtllmLauncher().build_spec( + _trtllm_config({"served_model_name": "qwen3-06b-trt", "max_model_len": 2048}), + "config.yaml", "T::a") + assert spec.command[0] == "trtllm-serve" + assert spec.command[1:3] == ["serve", "/engines/qwen3-06b_tp1_fp16"] + assert "--backend" in spec.command and "tensorrt" in spec.command + assert spec.engine == "trtllm" + assert spec.served_name == "qwen3-06b-trt" + assert CAP_METRICS_TRTLLM in spec.capabilities + assert CAP_SLEEP not in spec.capabilities and CAP_RUNTIME_LORA not in spec.capabilities + assert spec.env.get("CUDA_VISIBLE_DEVICES") == "0" + # perf YAML was written and referenced; tensorrt mode must NOT set + # enable_iter_perf_stats (the tensorrt backend aborts on it). + perf = spec.command[spec.command.index("--extra_llm_api_options") + 1] + with open(perf) as f: + content = f.read() + assert "return_perf_metrics: true" in content + assert "enable_iter_perf_stats" not in content + + +def test_trtllm_build_spec_pytorch_when_no_engine_dir(tmp_path, monkeypatch): + import app.llmops.launchers as L + monkeypatch.setattr(L, "LOG_DIR", str(tmp_path)) + # A trtllm group with NO engine_dir -> pytorch backend serving the HF model_tag. + cfg = RootConfig.model_validate({ + "server": {"host": "0.0.0.0", "port": 8887}, + "LLM_engines": {"T": { + "instances": [{"id": "a", "host": "localhost", "port": 8061, "cuda_device": 0}], + "model_config": {"model_tag": "Qwen/Qwen3-0.6B", "engine": "trtllm", + "served_model_name": "qwen3-pt", "max_model_len": 2048}, + }}, + }) + spec = TrtllmLauncher().build_spec(cfg, "config.yaml", "T::a") + assert spec.command[1:5] == ["serve", "Qwen/Qwen3-0.6B", "--backend", "pytorch"] + assert spec.served_name == "qwen3-pt" + assert CAP_METRICS_TRTLLM in spec.capabilities + # pytorch mode needs BOTH keys so the iteration-level gauges (running/waiting/ + # kv_cache_utilization) the router + dashboard use are exposed. + perf = spec.command[spec.command.index("--extra_llm_api_options") + 1] + with open(perf) as f: + content = f.read() + assert "return_perf_metrics: true" in content + assert "enable_iter_perf_stats: true" in content + + +def test_trtllm_launcher_keys_filters_by_engine(): + keys = TrtllmLauncher().keys(_trtllm_config()) + assert keys == ["T::a"] + + +def test_trtllm_available_reflects_trtllm_serve_presence(monkeypatch): + import app.llmops.launchers as L + monkeypatch.setattr(L.shutil, "which", lambda name: "/usr/bin/trtllm-serve" + if name == "trtllm-serve" else None) + assert TrtllmLauncher().available() is True + monkeypatch.setattr(L.shutil, "which", lambda name: None) + assert TrtllmLauncher().available() is False diff --git a/apps/backend/tests/unit/test_manager_engine.py b/apps/backend/tests/unit/test_manager_engine.py index 09d786a..bfebd61 100644 --- a/apps/backend/tests/unit/test_manager_engine.py +++ b/apps/backend/tests/unit/test_manager_engine.py @@ -168,13 +168,14 @@ async def test_load_lora_rejected_when_engine_lacks_capability(tmp_path): async def test_create_overlay_model_rejects_unregistered_engine(tmp_path): - # trtllm is a valid engine name in the schema, but no launcher is registered - # for it here — the manager must refuse cleanly, not KeyError into a 500. + # llamacpp is a valid schema engine, but no launcher is registered for it in this + # fake manager (only vllm + the sglang fake) — the manager must refuse cleanly, + # not KeyError into a 500. mgr = _manager_with_fake(tmp_path) with pytest.raises(ModelConflict, match="unsupported engine"): await mgr.create_overlay_model( "Brand", {"id": "z", "host": "localhost", "port": 8040}, - {"model_tag": "org/brand", "engine": "trtllm"}, + {"model_tag": "org/brand", "engine": "llamacpp"}, ) @@ -309,3 +310,24 @@ async def test_owning_node_api_url_none_when_local(tmp_path): BackendSettings(instance_id="vllm-node"), store=store, overlay_path=str(tmp_path/"o.json")) assert await mgr.owning_node_api_url("S::a") is None # local -> read locally + + +def test_unregistered_engine_groups_detects_phantom(tmp_path): + # A group whose engine no registered launcher claims must be reported (fail loud), + # not silently dropped from the registry. (P1) + from app.llmops.manager import unregistered_engine_groups + from app.llmops.launchers import VllmLauncher + from schema import load_config + cfg = tmp_path / "c.yaml" + cfg.write_text( + "server:\n port: 8887\n" + "LLM_engines:\n" + " A:\n instances:\n - {id: a, host: localhost, port: 8001}\n" + " model_config: {model_tag: org/a}\n" # vllm (default) + " B:\n instances:\n - {id: b, host: localhost, port: 8002}\n" + " model_config: {model_tag: org/b, engine: sglang}\n", # no sglang launcher + encoding="utf-8", + ) + config = load_config(str(cfg)) + phantom = unregistered_engine_groups(config, [VllmLauncher()]) + assert phantom == {"B": "sglang"} diff --git a/apps/backend/tests/unit/test_process.py b/apps/backend/tests/unit/test_process.py index ea5fb45..5b4f028 100644 --- a/apps/backend/tests/unit/test_process.py +++ b/apps/backend/tests/unit/test_process.py @@ -36,3 +36,33 @@ def test_terminate_process_group_graceful(): # SIGTERM ends `sleep` well under the grace period. assert proc.poll() is not None assert time.perf_counter() - start < 5.0 + + +def test_port_is_free_detects_live_listener(): + import socket + from app.llmops.process import port_is_free + s = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + s.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + s.bind(("0.0.0.0", 0)); s.listen(1) + port = s.getsockname()[1] + assert port_is_free(port) is False + s.close() + assert port_is_free(port) is True + + +def test_free_port_kills_the_holder(): + import socket + from app.llmops.process import free_port, port_is_free + probe = socket.socket(); probe.bind(("0.0.0.0", 0)); port = probe.getsockname()[1]; probe.close() + child = subprocess.Popen([sys.executable, "-c", + "import socket,time;s=socket.socket();s.setsockopt(socket.SOL_SOCKET,socket.SO_REUSEADDR,1);" + f"s.bind(('0.0.0.0',{port}));s.listen(1);time.sleep(60)"]) + try: + for _ in range(40): + if not port_is_free(port): break + time.sleep(0.05) + assert port_is_free(port) is False + assert free_port(port, timeout=10.0) is True + assert child.poll() is not None + finally: + if child.poll() is None: child.kill() diff --git a/apps/backend/tests/unit/test_reconciler.py b/apps/backend/tests/unit/test_reconciler.py index 9e16ee0..e235aa9 100644 --- a/apps/backend/tests/unit/test_reconciler.py +++ b/apps/backend/tests/unit/test_reconciler.py @@ -324,6 +324,17 @@ async def test_adopt_skips_port_serving_wrong_model(): assert reg.get(HEALTHY).state == ModelState.STOPPED +async def test_adopt_matches_custom_served_name(): + # A model with served_model_name set advertises ONLY that name at /v1/models (not + # its model_tag or group). Adoption must still succeed via the spec's served_name, + # else a second process spawns on its port -> crash loop. (verification P0) + reg = _registry() + reg.get(HEALTHY).spec.served_name = "my-served-alias" + client = FakeHTTPClient(healthy_ports={8002}, served_ids=["my-served-alias"]) + await adopt_running(reg, client, _settings()) + assert reg.get(HEALTHY).state == ModelState.READY + + async def test_adopt_respects_persisted_stopped_intent(): # A model the user had stopped (persisted desired=stopped) whose process survived # a backend restart is adopted READY but NOT resurrected to desired=running. diff --git a/apps/backend/tests/unit/test_trt_convert.py b/apps/backend/tests/unit/test_trt_convert.py new file mode 100644 index 0000000..6c10680 --- /dev/null +++ b/apps/backend/tests/unit/test_trt_convert.py @@ -0,0 +1,173 @@ +"""HF -> TensorRT-LLM engine build service (app/services/trt_convert.py). + +The build subprocess (trtllm-bench) is mocked; these assert the wiring: cache key, +validation, the build's move-and-manifest, idempotent reuse, and engine listing. +""" +import json +import os + +import pytest + +from app.services import trt_convert as T +from app.services.trt_convert import BuildParams, TrtConvertManager + +pytestmark = pytest.mark.unit + + +@pytest.fixture +def env(tmp_path, monkeypatch): + monkeypatch.setenv("LLMOPS_TRT_ENGINES_DIR", str(tmp_path)) + # Deterministic cache key (no real GPU / package lookup). + monkeypatch.setattr(T, "_gpu_compute_cap", lambda: "86") + monkeypatch.setattr(T, "_trtllm_version", lambda: "1.3.0rc20") + monkeypatch.setattr(T, "bench_available", lambda: True) + return tmp_path + + +def test_engines_root_defaults_beside_hf_cache(monkeypatch, tmp_path): + # No override -> co-locate with the HF model cache under a trt-engines/ sibling + # of hub/ (same mount the Model Library uses). + monkeypatch.delenv("LLMOPS_TRT_ENGINES_DIR", raising=False) + import huggingface_hub.constants as C + monkeypatch.setattr(C, "HF_HUB_CACHE", str(tmp_path / "hf" / "hub")) + assert T.engines_root() == str(tmp_path / "hf" / "trt-engines") + # explicit override still wins + monkeypatch.setenv("LLMOPS_TRT_ENGINES_DIR", "/custom/engines") + assert T.engines_root() == "/custom/engines" + + +def test_cache_key_encodes_every_build_param(env): + k = BuildParams("Qwen/Qwen3-0.6B", tp_size=2, max_seq_len=4096, max_batch_size=8, + max_num_tokens=16384, quantization="FP8").cache_key() + # model sanitized (no slash), and each shape/quant/arch/version present + assert "Qwen_Qwen3-0.6B" in k and "tp2pp1" in k and "seq4096" in k + assert "bs8" in k and "nt16384" in k and "fp8" in k and "sm86" in k and "trt1.3.0rc20" in k + # a different knob -> a different key (no cache collision) + assert BuildParams("Qwen/Qwen3-0.6B", max_seq_len=2048).cache_key() != \ + BuildParams("Qwen/Qwen3-0.6B", max_seq_len=4096).cache_key() + + +def test_validate_rejects_bad_quant_and_dims(env): + with pytest.raises(ValueError, match="quantization"): + T._validate(BuildParams("m", quantization="BOGUS")) + with pytest.raises(ValueError, match="max_seq_len"): + T._validate(BuildParams("m", max_seq_len=0)) + with pytest.raises(ValueError, match="model_tag"): + T._validate(BuildParams(" ")) + + +def test_build_unavailable_raises(env, monkeypatch): + monkeypatch.setattr(T, "bench_available", lambda: False) + with pytest.raises(RuntimeError, match="trtllm-bench is not available"): + T.build(BuildParams("Qwen/Qwen3-0.6B")) + + +def _fake_bench(engine_leaf="rank0.engine"): + """Return a subprocess.run stand-in that plants a produced engine under the -w + workspace (mimicking trtllm-bench's //tp_1_pp_1/rank0.engine).""" + def run(cmd, **kwargs): + ws = cmd[cmd.index("-w") + 1] + out = os.path.join(ws, "Model", "tp_1_pp_1") + os.makedirs(out, exist_ok=True) + open(os.path.join(out, engine_leaf), "w").close() + open(os.path.join(out, "config.json"), "w").close() + + class R: + returncode = 0 + stdout = "ENGINE SAVED" + stderr = "" + return R() + return run + + +def test_build_moves_engine_and_writes_manifest(env, monkeypatch): + monkeypatch.setattr(T.subprocess, "run", _fake_bench()) + params = BuildParams("Qwen/Qwen3-0.6B", max_seq_len=2048) + engine_dir = T.build(params) + + assert engine_dir == T.engine_dir_for(params) + assert os.path.isfile(os.path.join(engine_dir, "rank0.engine")) + with open(os.path.join(engine_dir, "manifest.json")) as f: + m = json.load(f) + assert m["model_tag"] == "Qwen/Qwen3-0.6B" + assert m["compute_capability"] == "86" and m["trtllm_version"] == "1.3.0rc20" + assert m["max_seq_len"] == 2048 and m["cache_key"] == params.cache_key() + # the temp build workspace was cleaned up + assert not os.path.isdir(os.path.join(str(env), ".trt-build", params.cache_key())) + + +def test_build_is_idempotent_reuses_existing(env, monkeypatch): + calls = {"n": 0} + real = _fake_bench() + + def counting(cmd, **kwargs): + calls["n"] += 1 + return real(cmd, **kwargs) + + monkeypatch.setattr(T.subprocess, "run", counting) + params = BuildParams("Qwen/Qwen3-0.6B") + T.build(params) + T.build(params) # second call must hit the cache, not rebuild + assert calls["n"] == 1 + + +def test_build_failure_surfaces_stderr(env, monkeypatch): + def failing(cmd, **kwargs): + class R: + returncode = 1 + stdout = "" + stderr = "boom: unsupported architecture" + return R() + monkeypatch.setattr(T.subprocess, "run", failing) + with pytest.raises(RuntimeError, match="unsupported architecture"): + T.build(BuildParams("Weird/Model")) + + +def test_list_engines_reads_manifests(env, monkeypatch): + monkeypatch.setattr(T.subprocess, "run", _fake_bench()) + T.build(BuildParams("Qwen/Qwen3-0.6B")) + engines = TrtConvertManager().list_engines() + assert len(engines) == 1 + assert engines[0]["model_tag"] == "Qwen/Qwen3-0.6B" + assert engines[0]["engine_dir"] == T.engine_dir_for(BuildParams("Qwen/Qwen3-0.6B")) + assert engines[0]["size_on_disk"] >= 0 # summed from the engine dir + + +def test_delete_engine_removes_dir(env, monkeypatch): + monkeypatch.setattr(T.subprocess, "run", _fake_bench()) + params = BuildParams("Qwen/Qwen3-0.6B") + T.build(params) + mgr = TrtConvertManager() + assert os.path.isdir(T.engine_dir_for(params)) + assert mgr.delete_engine(params.cache_key()) is True + assert not os.path.isdir(T.engine_dir_for(params)) + assert mgr.delete_engine(params.cache_key()) is False # already gone -> False + + +def test_delete_engine_rejects_path_traversal(env): + mgr = TrtConvertManager() + for bad in ("../etc", "a/b", "..", "."): + with pytest.raises(ValueError): + mgr.delete_engine(bad) + + +async def test_manager_start_runs_build_and_completes(env, monkeypatch): + monkeypatch.setattr(T.subprocess, "run", _fake_bench()) + mgr = TrtConvertManager() + job = mgr.start(BuildParams("Qwen/Qwen3-0.6B")) + assert job["state"] in ("pending", "building") + await mgr._tasks[job["key"]] # await the build task + done = mgr.list()[0] + assert done["state"] == "completed" + assert done["engine_dir"] == T.engine_dir_for(BuildParams("Qwen/Qwen3-0.6B")) + + +async def test_manager_start_idempotent_while_in_flight(env, monkeypatch): + monkeypatch.setattr(T.subprocess, "run", _fake_bench()) + mgr = TrtConvertManager() + p = BuildParams("Qwen/Qwen3-0.6B") + j1 = mgr.start(p) + j2 = mgr.start(p) # same key, still in flight -> same job, no duplicate task + assert j1["key"] == j2["key"] + assert len(mgr.list()) == 1 + await mgr._tasks[j1["key"]] diff --git a/apps/frontend_llmops/eslint.config.ts b/apps/frontend_llmops/eslint.config.ts index 89fdc89..1dd4a54 100644 --- a/apps/frontend_llmops/eslint.config.ts +++ b/apps/frontend_llmops/eslint.config.ts @@ -22,5 +22,13 @@ export default defineConfigWithVueTs( ...pluginOxlint.buildFromOxlintConfigFile('.oxlintrc.json'), + // The local UI kit mirrors shadcn/reka naming (Button, Card, Dialog…) where + // single-word primitives are the convention, not a collision risk. + { + name: 'app/ui-kit-single-word-names', + files: ['src/components/ui/**/*.vue'], + rules: { 'vue/multi-word-component-names': 'off' }, + }, + skipFormatting, ) diff --git a/apps/frontend_llmops/src/components/AddModelDialog.vue b/apps/frontend_llmops/src/components/AddModelDialog.vue index cfbd817..15c52f9 100644 --- a/apps/frontend_llmops/src/components/AddModelDialog.vue +++ b/apps/frontend_llmops/src/components/AddModelDialog.vue @@ -6,6 +6,7 @@ import Dialog from '@/components/ui/Dialog.vue' import Input from '@/components/ui/Input.vue' import Textarea from '@/components/ui/Textarea.vue' import Button from '@/components/ui/Button.vue' +import Select from '@/components/ui/Select.vue' import Badge from '@/components/ui/Badge.vue' import { api } from '@/lib/api' import { toast } from '@/lib/toast' @@ -15,7 +16,7 @@ import { useResourcesStore } from '@/stores/resources' import { formatBytes } from '@/lib/utils' import { ROUTING_STRATEGIES, routingStrategyLabel } from '@/lib/routingStrategies' import { KV_SHARE_PRESET, isKvShared } from '@/lib/kvSharing' -import type { CachedModel, DownloadJob, KvTransferConfig, LoraAdapter, LoraModule, SettingValue } from '@/types/api' +import type { CachedModel, DownloadJob, EngineInfo, KvTransferConfig, LoraAdapter, LoraModule, SettingValue, TrtEngine } from '@/types/api' const open = defineModel('open', { default: false }) const props = defineProps<{ mode?: 'create' | 'edit'; editKey?: string | null }>() @@ -43,23 +44,57 @@ const modelTag = ref('') // Inference engine for the whole group. Different engines map the same concepts to // different launch flags; the launcher translates. vLLM is the default. const engine = ref('vllm') -const ENGINE_OPTIONS = ['vllm', 'sglang', 'llamacpp', 'trtllm'] +// Registered engines + capabilities from the backend (single source of truth — the +// form gates on these instead of hardcoding engine names/features). Loaded on mount; +// falls back to a vLLM-only stub so the dialog works before the fetch resolves. (P1) +const engineCatalogue = ref([ + { name: 'vllm', capabilities: ['sleep', 'kv_transfer'], lora_endpoint_prefix: '/v1', metric_prefix: 'vllm', inapplicable_keys: [], paste_example: '' }, +]) +async function loadEngines() { + try { + const { engines } = await api.listEngines() + if (engines.length) engineCatalogue.value = engines + } catch { + /* keep the fallback stub */ + } +} +loadEngines() +const ENGINE_OPTIONS = computed(() => engineCatalogue.value.map((e) => e.name)) +const engineMeta = (name: string) => engineCatalogue.value.find((e) => e.name === name) +const selectedEngine = computed(() => engineMeta(engine.value)) // Which engine's command the paste box expects — drives the example placeholder and // is sent to the parser so an ambiguous command still parses for the right engine. const pasteEngine = ref('vllm') -const PASTE_PLACEHOLDER: Record = { - vllm: 'CUDA_VISIBLE_DEVICES=0 vllm serve Qwen/Qwen2.5-3B-Instruct --port 8020 --dtype float16 --max-model-len 4096 --gpu-memory-utilization 0.85', - sglang: 'CUDA_VISIBLE_DEVICES=0 python -m sglang.launch_server --model-path Qwen/Qwen3-0.6B --port 8030 --context-length 4096 --mem-fraction-static 0.85', - llamacpp: 'CUDA_VISIBLE_DEVICES=0 llama-server -hf Qwen/Qwen2.5-0.5B-Instruct-GGUF:Q4_K_M -a qwen25-05b-gguf -c 4096 -ngl 99 --port 8091', -} -const pastePlaceholder = computed(() => PASTE_PLACEHOLDER[pasteEngine.value] ?? PASTE_PLACEHOLDER.vllm) -// Capability gating — the two engines support different things, so the form only -// shows what the selected engine actually has (see launchers.py CAP_* flags). +const pastePlaceholder = computed( + () => engineMeta(pasteEngine.value)?.paste_example || engineMeta('vllm')?.paste_example || '', +) +// Engine-specific UI panels still key on the name (presentation, not capability); +// capability gating derives from the API's capability list (drift-free vs CAP_*). const engineIsVllm = computed(() => engine.value === 'vllm') const engineIsSglang = computed(() => engine.value === 'sglang') const engineIsLlamacpp = computed(() => engine.value === 'llamacpp') -const engineHasSleep = computed(() => engine.value === 'vllm') // /sleep + /wake_up (vLLM only) -const engineHasKvShare = computed(() => engine.value === 'vllm') // OffloadingConnector (vLLM only) +const engineIsTrtllm = computed(() => engine.value === 'trtllm') +// TensorRT-LLM: an empty engine_dir runs HF weights (pytorch backend); a pre-built +// engine dir runs --backend tensorrt. Offer the engines already built in the Library. +const trtEngineDir = ref('') +const trtEngines = ref([]) +async function loadTrtEngines() { + try { + trtEngines.value = (await api.listTrtEngines()).engines + } catch { + /* feature unavailable here — leave empty */ + } +} +// Engines built for the entered model_tag first; fall back to all so a match is always +// pickable even if the tag isn't typed yet. +const trtEngineOptions = computed(() => { + const forTag = trtEngines.value.filter((e) => e.model_tag === modelTag.value.trim()) + return forTag.length ? forTag : trtEngines.value +}) +const engineHasSleep = computed(() => selectedEngine.value?.capabilities.includes('sleep') ?? false) +const engineHasKvShare = computed(() => selectedEngine.value?.capabilities.includes('kv_transfer') ?? false) +// model_config keys the selected engine ignores — greyed out / dropped on submit. +const inapplicableKeys = computed(() => new Set(selectedEngine.value?.inapplicable_keys ?? [])) const params = ref<{ key: string; value: string }[]>([]) // Router-only load-balancing policy for the group. Lives in model_config but is // NOT a vLLM flag, so it's edited as its own field and kept out of the raw param @@ -195,6 +230,7 @@ function reset() { routingStrategy.value = '' kvShared.value = false sleepMode.value = false + trtEngineDir.value = '' loras.value = [] } @@ -240,10 +276,9 @@ function prefillForEdit() { routingStrategy.value = String(cfg.settings.routing_strategy ?? '') kvShared.value = isKvShared(cfg.settings) sleepMode.value = !!cfg.settings.enable_sleep_mode - // vLLM/SGLang compute knobs that the schema serialises for every group but that - // don't apply to llama.cpp (the launcher drops them). Filtered out when editing a - // llamacpp model so they don't clutter the raw param editor or round-trip on save. - const llamacppInapplicable = new Set(['gpu_memory_utilization', 'tensor_parallel_size', 'dtype']) + // Knobs the selected engine ignores (e.g. gpu_memory_utilization/dtype for + // llama.cpp) come from the engine catalogue, so they don't clutter the raw param + // editor or round-trip on save. Drift-free vs the backend's drop lists. params.value = extractLoras( Object.entries(cfg.settings).filter( ([k2, v]) => @@ -253,12 +288,15 @@ function prefillForEdit() { v !== null && k2 !== 'model_tag' && k2 !== 'engine' && + k2 !== 'engine_dir' && k2 !== 'routing_strategy' && k2 !== 'kv_transfer_config' && k2 !== 'enable_sleep_mode' && - !(engine.value === 'llamacpp' && llamacppInapplicable.has(k2)), + !inapplicableKeys.value.has(k2), ), ).map(([k2, v]) => ({ key: k2, value: String(v) })) + // TensorRT-LLM engine_dir is managed by its own picker, not the raw param list. + trtEngineDir.value = String(cfg.settings.engine_dir ?? '') warnings.value = [] parsed.value = true // skip the paste/parse step } @@ -275,6 +313,7 @@ watch(open, (v) => { if (isEdit.value) prefillForEdit() void loadCache() void loadDownloads() + void loadTrtEngines() void loadLoraLibrary() downloadPoll = setInterval(loadDownloads, 1500) // reflect background progress }) @@ -572,6 +611,7 @@ async function submit() { kv_transfer_config?: KvTransferConfig enable_sleep_mode?: boolean engine?: string + engine_dir?: string } = { model_tag: modelTag.value, // Engine for the whole group; default vllm keeps existing configs unchanged. @@ -589,10 +629,13 @@ async function submit() { kk !== 'routing_strategy' && kk !== 'kv_transfer_config' && kk !== 'enable_sleep_mode' && - kk !== 'engine' + kk !== 'engine' && + kk !== 'engine_dir' ) settings[kk] = coerce(value) } + // TensorRT-LLM: a picked pre-built engine -> --backend tensorrt; empty -> pytorch. + if (engineIsTrtllm.value && trtEngineDir.value) settings.engine_dir = trtEngineDir.value // Router-only load-balancing policy; '' inherits the global default, so only // send it when explicitly chosen. if (routingStrategy.value) settings.routing_strategy = routingStrategy.value @@ -745,39 +788,60 @@ async function submit() { + +