From b4de0a0cfa87585cffd2ed5dc9e43e737f708a22 Mon Sep 17 00:00:00 2001 From: max Date: Sat, 4 Jul 2026 11:29:35 +0800 Subject: [PATCH 01/20] fix(adopt): match served_model_name on adoption; unschedulable uses registry engine MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Address the review-fixes verification (docs/architecture-review-fixes-verification_zh-TW.md). P0 — adopt regression (M#6): _serves_model only matched {model_tag, group}, but when served_model_name is set /v1/models advertises ONLY that name, and all three launchers always pass --served-model-name/--alias. So a model with a custom served name was rejected on boot adoption → a second process spawned on its port → crash loop. Add a served_name field to LaunchSpec (each launcher fills served_model_name or model_tag) and match on it in _serves_model. Gap #1 — unschedulable_reasons (M#5 UI): resolved engine from config.LLM_engines, so the embedding group (engine "default", absent from LLM_engines) fell back to "vllm" and was never flagged — the case the flag most needs in a mixed fleet. Resolve engine from the registry instead (same source as the scheduler), so "default" is correctly detected. Recorded as roadmap/known-bounded in the fixes doc: #2 fencing-token monotonicity after release, #3 reclaim_local availability cost with same-engine nodes, #4 inf into the autoscaler, #5 router live-set failover tail. Tests: backend 487, router 127, store 41 (+ test_adopt_matches_custom_served_name, test_unschedulable_uses_registry_engine_for_embedding). Co-Authored-By: Claude Opus 4.8 --- apps/backend/app/llmops/instance.py | 5 + apps/backend/app/llmops/launchers.py | 3 + apps/backend/app/llmops/manager.py | 14 ++- apps/backend/app/llmops/reconciler.py | 7 +- apps/backend/tests/unit/test_ha_safety.py | 37 ++++++ apps/backend/tests/unit/test_reconciler.py | 11 ++ ...tecture-review-fixes-verification_zh-TW.md | 112 ++++++++++++++++++ docs/architecture-review-fixes_zh-TW.md | 33 +++++- 8 files changed, 214 insertions(+), 8 deletions(-) create mode 100644 docs/architecture-review-fixes-verification_zh-TW.md diff --git a/apps/backend/app/llmops/instance.py b/apps/backend/app/llmops/instance.py index d20086b..61b851b 100644 --- a/apps/backend/app/llmops/instance.py +++ b/apps/backend/app/llmops/instance.py @@ -35,6 +35,11 @@ class LaunchSpec: engine: str = "vllm" capabilities: frozenset = field(default_factory=frozenset) model_tag: Optional[str] = None + # The name the engine advertises at /v1/models (--served-model-name / --alias): + # `served_model_name` if set, else the model_tag. Used to verify a process's + # identity on boot adoption — when served_model_name is set, /v1/models shows ONLY + # this, not the model_tag. Defaults to model_tag when unset. + served_name: Optional[str] = None # True when launched with --enable-sleep-mode + VLLM_SERVER_DEV_MODE=1, so the # /sleep, /wake_up and /is_sleeping dev endpoints are available. sleep_enabled: bool = False diff --git a/apps/backend/app/llmops/launchers.py b/apps/backend/app/llmops/launchers.py index 522ceac..e7aaf5d 100644 --- a/apps/backend/app/llmops/launchers.py +++ b/apps/backend/app/llmops/launchers.py @@ -251,6 +251,7 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: port=inst.port, probe_url=f"http://{inst.host}:{inst.port}/health", model_tag=engine.settings.model_tag, + served_name=merged.get("served_model_name") or engine.settings.model_tag, sleep_enabled=sleep_enabled, ) @@ -442,6 +443,7 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: port=inst.port, probe_url=f"http://{inst.host}:{inst.port}/health", model_tag=engine.settings.model_tag, + served_name=merged.get("served_model_name") or engine.settings.model_tag, ) @@ -606,4 +608,5 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: port=inst.port, probe_url=f"http://{inst.host}:{inst.port}/health", model_tag=engine.settings.model_tag, + served_name=merged.get("served_model_name") or engine.settings.model_tag, ) diff --git a/apps/backend/app/llmops/manager.py b/apps/backend/app/llmops/manager.py index 364e64a..77dc887 100644 --- a/apps/backend/app/llmops/manager.py +++ b/apps/backend/app/llmops/manager.py @@ -481,10 +481,16 @@ async def unschedulable_reasons(self) -> dict[str, str]: for key, want in desired.items(): if want != Desired.RUNNING.value: continue - group = key.split("::")[0] - engine = getattr( - getattr(self.config.LLM_engines.get(group), "settings", None), - "engine", "vllm") + # Resolve the engine from the registry (same source as the scheduler's + # _track_unschedulable), NOT config.LLM_engines — a non-LLM group like + # `embedding` isn't in LLM_engines and would fall back to "vllm", hiding the + # very case that most needs the flag (bespoke embedding can't be placed in a + # mixed fleet). A key absent from the registry (overlay not synced) is + # skipped rather than guessed. (verification gap #1) + inst = self.registry.get(key) + engine = getattr(inst, "engine", None) + if engine is None: + continue if not any(node_supports(n, engine) for n in nodes): out[key] = f"no live node runs engine '{engine}'" return out diff --git a/apps/backend/app/llmops/reconciler.py b/apps/backend/app/llmops/reconciler.py index 119204f..204a4a4 100644 --- a/apps/backend/app/llmops/reconciler.py +++ b/apps/backend/app/llmops/reconciler.py @@ -391,7 +391,12 @@ async def _serves_model(http_client, inst: ModelInstance) -> bool: except Exception: return False group = inst.key.split("::")[0] - wanted = {c for c in (inst.model_tag, group) if c} + # When served_model_name is set, /v1/models advertises ONLY it (not the model_tag), + # and all three launchers always pass --served-model-name/--alias. So match on the + # served name too, else a model with a custom served name is wrongly rejected on + # boot adoption → a second process spawns on its port → crash loop. (verification P0) + served = getattr(getattr(inst, "spec", None), "served_name", None) + wanted = {c for c in (inst.model_tag, group, served) if c} return bool(ids & wanted) diff --git a/apps/backend/tests/unit/test_ha_safety.py b/apps/backend/tests/unit/test_ha_safety.py index 82ee95d..c5c19ef 100644 --- a/apps/backend/tests/unit/test_ha_safety.py +++ b/apps/backend/tests/unit/test_ha_safety.py @@ -346,3 +346,40 @@ async def test_unschedulable_empty_in_collapsed_mode(tmp_path): # No store / SQLite (db_url None) -> nothing computed (single host runs everything). mgr, _ = _manager(tmp_path, store=None) assert await mgr.unschedulable_reasons() == {} + + +EMBED_CONFIG_YAML = CONFIG_YAML + """ +embedding_server: + host: localhost + port: 8005 + cuda_device: 0 + embedding_models: + m3e: + model_name: moka-ai/m3e-base + max_length: 512 + use_gpu: true + reranking_models: + bge: + model_name: BAAI/bge-reranker-large + max_length: 512 +""" + + +async def test_unschedulable_uses_registry_engine_for_embedding(tmp_path): + # embedding's engine is the sentinel 'default'; no node advertises it, so it must + # be flagged even when a vLLM node exists. The old code resolved engine from + # config.LLM_engines (no 'embedding' group) and fell back to vllm, hiding it. (#1) + cfg = tmp_path / "config.yaml" + cfg.write_text(EMBED_CONFIG_YAML, encoding="utf-8") + config = load_config(str(cfg)) + launchers = [VllmLauncher(), EmbeddingLauncher()] + registry = build_registry(config, str(cfg), launchers) + store = FakeStore() + store.desired = {"embedding::default": Desired.RUNNING.value} + store.nodes = [{"node_id": "n1", "engines": '["vllm"]'}] + mgr = ModelManager(registry, launchers, None, config, str(cfg), + BackendSettings(instance_id="node-A"), store=store, + overlay_path=str(tmp_path / "o.json")) + reasons = await mgr.unschedulable_reasons() + assert "embedding::default" in reasons + assert "default" in reasons["embedding::default"] diff --git a/apps/backend/tests/unit/test_reconciler.py b/apps/backend/tests/unit/test_reconciler.py index 9e16ee0..e235aa9 100644 --- a/apps/backend/tests/unit/test_reconciler.py +++ b/apps/backend/tests/unit/test_reconciler.py @@ -324,6 +324,17 @@ async def test_adopt_skips_port_serving_wrong_model(): assert reg.get(HEALTHY).state == ModelState.STOPPED +async def test_adopt_matches_custom_served_name(): + # A model with served_model_name set advertises ONLY that name at /v1/models (not + # its model_tag or group). Adoption must still succeed via the spec's served_name, + # else a second process spawns on its port -> crash loop. (verification P0) + reg = _registry() + reg.get(HEALTHY).spec.served_name = "my-served-alias" + client = FakeHTTPClient(healthy_ports={8002}, served_ids=["my-served-alias"]) + await adopt_running(reg, client, _settings()) + assert reg.get(HEALTHY).state == ModelState.READY + + async def test_adopt_respects_persisted_stopped_intent(): # A model the user had stopped (persisted desired=stopped) whose process survived # a backend restart is adopted READY but NOT resurrected to desired=running. diff --git a/docs/architecture-review-fixes-verification_zh-TW.md b/docs/architecture-review-fixes-verification_zh-TW.md new file mode 100644 index 0000000..62b32b0 --- /dev/null +++ b/docs/architecture-review-fixes-verification_zh-TW.md @@ -0,0 +1,112 @@ +# 架構審查修復 — 複審報告 + +> 複審日期:2026-07-04。對象:[architecture-review-fixes_zh-TW.md](architecture-review-fixes_zh-TW.md) +> 所列修復(PR #12 `cb79ec3`、PR #13 `6a91a42`)。方法:逐條對照實際 diff(非僅文檔宣稱), +> 並重跑相關測試套件。 +> +> **結論:修復品質很好,C#1~C#3、H#1 等關鍵項目都正確落地,測試屬實(HA 相關 69 passed、 +> router 32 passed、store 41 passed)。發現一個新引入的真 bug(M#6 的 adopt 驗證)需要修, +> 另有五個小縫建議記錄。** + +--- + +## 🔴 P0 — 新引入的迴歸:M#6 的 adopt 身分驗證漏了 `served_model_name` + +**位置**:`apps/backend/app/llmops/reconciler.py` `_serves_model` + +```python +wanted = {c for c in (inst.model_tag, group) if c} +return bool(ids & wanted) +``` + +**問題**:比對集合只有 `{model_tag, group 名}`,但三個引擎都支援 `served_model_name`, +而且設定之後 `/v1/models` 廣告的**只有 served name,不含 model_tag**: + +- vLLM:`--served-model-name` 經通用 CLI 路徑透傳;且 **paste-command 匯入會自動寫入 + `served_model_name`**(`apps/backend/app/services/vllm_command.py:378`)——不是罕見配置。 +- SGLang / llama.cpp:launcher 恆定發出 `--served-model-name` / `--alias` + = `served_model_name or model_tag`。 + +**觸發序列**: +1. 模型設了 `served_model_name`(≠ model_tag 且 ≠ group 名),正常運行中; +2. backend 容器重啟(推理行程存活)→ `adopt_running` 呼叫 `_serves_model` → 比對不到 → + **拒絕領養**,本地 state 停在 STOPPED; +3. store 內 desired=running → `replay_desired` / `converge_desired` 在**同一個 port** 上 + spawn 第二份行程 → bind 失敗 → FAILED → auto-restart 迴圈。 + +**影響**:backend 重啟(常規運維動作)後,設了 served name 的模型進入 port 衝突 crash loop。 +這正是原審查 M#6 想避免的「錯誤領養」的反面——修復把合法的自家行程當外人,後果比原問題更吵。 + +**建議修法**: +- 最乾淨:三個 launcher 的 `build_spec` 都已算出 `served` 變數,把它存進 `LaunchSpec` + (如 `served_name` 欄位),`_serves_model` 把 `inst.spec.served_name` 加進 `wanted`; +- 快修:`_serves_model` 從 `manager.config` 的 group settings 讀 `served_model_name` 加入比對。 +- **補回歸測試**:「`served_model_name` 有設定時 adopt 成功」——現有三個 adopt 測試 + (`test_reconciler.py`)都未涵蓋此 case。 + +--- + +## 🟡 小縫(不擋合併,建議記錄/擇機處理) + +### 1. `unschedulable_reasons` 對 embedding 判斷錯誤(M#5 UI 在最需要的 case 上失效) +- 位置:`apps/backend/app/llmops/manager.py` `unschedulable_reasons` +- 問題:engine 從 `config.LLM_engines.get(group)` 解析;`embedding::default` 的 group + (`embedding`)不在 `LLM_engines` → fallback 成 `"vllm"`。mixed 模式下 embedding 明明永遠 + 排不上(engine=`default`、無 node 宣告),只要有 vllm node 存在,UI 就**不會**顯示 + unschedulable——與 scheduler 端 `_track_unschedulable`(用 registry、正確解析 `default`) + 不一致。M#2 文檔雖已標注 mixed 不支援 bespoke embedding,但 M#5 的 UI 恰好在這個最需要 + 提示的場景失效。 +- 建議:`unschedulable_reasons` 改從 registry 取 engine(與 scheduler 同源)。 + +### 2. Fencing token 在 graceful release 後不單調 +- 位置:`packages/llmops-store/llmops_store.py` `release_leader`(DELETE row) +- 問題:release 刪除 lease row,下一任 acquire 重新 INSERT `token=1`。一個更早、token 較大的 + 殭屍 leader 的遲到寫入會通過 `fence < current` 檢查。需多重故障疊加才觸發,嚴重度低。 +- 建議:release 改為「把 `expires_at` 設為過期」而保留 row(token 即真正單調)。 + +### 3. `reclaim_local` 的可用性代價(多節點同引擎時) +- 位置:`apps/backend/app/llmops/reconciler.py` converge 的 reclaim 分支 +- 問題:reclaim 觸發條件只看「assignment 指向別的存活 node」。未來兩個同引擎 node 時, + 一次 >30s(`node_ttl`)的 DB / heartbeat 抖動 → scheduler 搬走 assignment → 原 node 恢復後 + **殺掉自己健康服務中的行程**,而新 node 還在冷啟。目前一引擎一 node 拓撲下 reclaim 是 + 休眠的(無可搬移的候選),現階段無害。 +- 建議:reclaim 前多查一個條件——owner node 的 `instances_live` / observed 已有該 key 的 + READY row 才殺本地行程,否則等下一輪。 + +### 4. H#2 的 inf 會流進 autoscaler,變成持續擴容壓力 +- 位置:`vllm_metrics_client.py` fetch 的名稱錯配 → inf;`load_monitor.py` `aggregate_load` + 不濾 inf +- 問題:名稱錯配時 `waiting_per_replica = inf` → 每個 cooldown 週期都想 scale-up 到 + max_ready。比原本的「靜默縮容」好(吵 = 可發現),且 scrape 失敗本來就會產生 inf, + 不算迴歸——但值得知道這個行為。 +- 建議:可考慮 `aggregate_load` 把 inf 樣本當 missing 處理,讓 warning log 作為唯一訊號。 + +### 5. Router live-set 的 failover 尾巴 +- 位置:`routing_strategies.py` `select_instance` 的 live 過濾 +- 問題:live 候選全部試過失敗後,fallback 回非 live 的 config 實例(mixed 下是必死的 + router-container localhost)。被 `max_attempts ≤ 3` 限制住,影響小。 +- 建議:可在 fallback 分支略過已知非 live 的實例,或維持現狀(有界)。 + +--- + +## ✅ 特別驗證過、確認沒問題的點 + +| 疑點 | 驗證結果 | +|---|---| +| `resync_registry` 的 skipped 會不會永久卡住(變更不再重試)? | 不會。backend 的 `hydrate_overlay_from_store` 每 tick 都回 True(無 short-circuit),skipped 變更每 5s 重試,instance 停止後即套用。 | +| `_defer_to_owner` 清 assignment 的條件 | 正確——只在「persisted desired 真的改變」時清,封掉了 C#3 的 min_ready→defer→delete_assignment churn。 | +| `stop()` CAS 與 `start()` 拒絕 STOPPING | 組合正確;stale 分支提前 return,不會誤記 STOPPED 事件或清掉新行程。 | +| `fleet_state_snapshot` / `_live_elsewhere` 的資料形狀 | 吻合——`list_instance_observed` 回傳 parsed view JSON(含 `key`/`state`)外加 `node_id`,兩個消費者的欄位存取一致。 | +| `import_overlay` 與跨 node sync 的 guard 劃分 | 清楚——admin import 走 `protect_live=False`(自己已 force-stop),週期 sync 走預設 `protect_live=True`。 | +| `_FleetInstanceView` 屬性委派 | autoscaler / load_monitor 用到的 `key`、`kind`、`spec.sleep_enabled` 都經 `__getattr__` 正確委派;`state` 為覆寫值。 | +| Scheduler `_resolve_engine` 對已刪除模型的殘留 desired row | 回 None → 跳過,不會以 "vllm" 錯置;`_track_unschedulable` 也排除 None(等 overlay 同步,不誤報)。 | +| 測試宣稱 | 屬實:`test_ha_safety.py` + `test_converge_desired.py` + `test_scheduler.py` + `test_reconciler.py` 69 passed;router 兩個套件 32 passed;store 41 passed(35 skipped 為 PG-only)。 | + +--- + +## 處理順序建議 + +1. **修 P0(`_serves_model` + served_model_name)**——下次重啟生產環境前完成,附回歸測試。 +2. 小縫 #1(unschedulable UI 對 embedding)順手可修(改用 registry 解析 engine)。 +3. 小縫 #2、#3 掛進多機 roadmap(與既有的 DB-clock、in-flight 聚合並列)。 +4. 小縫 #4、#5 記錄即可,行為有界且可發現。 diff --git a/docs/architecture-review-fixes_zh-TW.md b/docs/architecture-review-fixes_zh-TW.md index 09f07a1..ea0f559 100644 --- a/docs/architecture-review-fixes_zh-TW.md +++ b/docs/architecture-review-fixes_zh-TW.md @@ -103,15 +103,27 @@ replica 都算得出,不依賴誰是 leader);前端 ModelGroupCard 對這種實例顯示「無法排程」警示 badge。 [manager.py](../apps/backend/app/llmops/manager.py) `unschedulable_reasons` / [models.py](../apps/backend/app/api/models.py) / `ModelGroupCard.vue` +- **複審 #1 修正**:engine 原本從 `config.LLM_engines.get(group)` 解析,`embedding::default` 的 group + 不在 LLM_engines → fallback 成 vllm,導致 mixed 下最該提示的「bespoke embedding 排不上」反而不顯示。 + 改成從 registry 取 engine(與 scheduler `_track_unschedulable` 同源),`default` 引擎正確判為 unschedulable。 - **測試**:`test_ha_safety.py::{test_unschedulable_reason_when_no_node_runs_engine, - test_not_unschedulable_when_a_node_runs_engine, test_unschedulable_empty_in_collapsed_mode}` + test_not_unschedulable_when_a_node_runs_engine, test_unschedulable_empty_in_collapsed_mode, + test_unschedulable_uses_registry_engine_for_embedding}` ### M#6 — adopt_running 只憑 /health 200 無條件領養 ✅ -- adopt 前對 LLM 實例查 `/v1/models` 比對 model_tag / group served name,對不上就不 adopt;且不再把 +- adopt 前對 LLM 實例查 `/v1/models` 比對 model_tag / group / **served_name**,對不上就不 adopt;且不再把 desired 從 stopped 強制改成 running(尊重 store 已持久化 intent)。 [reconciler.py](../apps/backend/app/llmops/reconciler.py) `adopt_running` / `_serves_model` +- **複審 P0 修正**:初版 `_serves_model` 只比對 `{model_tag, group}`,漏了 `served_model_name`——而設了 + served name 時 `/v1/models` 只廣告 served name(不含 model_tag),三個 launcher 又恆定發 + `--served-model-name`/`--alias`。導致設了 served name 的模型在 backend 重啟後被拒絕領養 → 同 port 起 + 第二份行程 → crash loop。修法:`LaunchSpec` 新增 `served_name`(三個 launcher 都填 + `served_model_name or model_tag`),`_serves_model` 把它加進比對集。 + [instance.py](../apps/backend/app/llmops/instance.py) `LaunchSpec.served_name` / + [launchers.py](../apps/backend/app/llmops/launchers.py) - **測試**:`test_reconciler.py::{test_adopt_running_marks_healthy_unmanaged, - test_adopt_skips_port_serving_wrong_model, test_adopt_respects_persisted_stopped_intent}` + test_adopt_skips_port_serving_wrong_model, test_adopt_respects_persisted_stopped_intent, + test_adopt_matches_custom_served_name}` ## Low @@ -150,3 +162,18 @@ - **M#3 跨 worker in-flight 聚合**:把 in-flight 落到 store,讓 `ROUTER_WORKERS>1` 的 drain 正確。 - **H#3 VRAM admission/reservation**:多模型併發排程的資源預留協議(審查 §4.4)。 - **L#2 控制端點認證**:協同 frontend/backend/router/nginx 的 admin-token 化。 +- **複審 #2 fencing token 單調性**:`release_leader` 目前 DELETE lease row,下一任 acquire 重置 `token=1`, + 極端多重故障下更早的殭屍 leader 遲到寫入可能通過 `fence < current`。修法:release 改為把 `expires_at` + 設過期而保留 row(token 真正單調)。嚴重度低,掛多機 roadmap。 +- **複審 #3 reclaim_local 可用性代價**:未來兩個同引擎 node 時,一次 >`node_ttl` 的 heartbeat 抖動可能讓 + scheduler 搬走 assignment,原 node 恢復後 reclaim 殺掉自己健康行程而新 node 還在冷啟。修法:reclaim 前 + 多查「owner 的 instances_live/observed 已有該 key 的 READY row」才殺。一引擎一 node 拓撲下 reclaim 休眠、 + 現階段無害,掛多機 roadmap。 + +## 已知、有界、可發現(記錄即可) + +- **複審 #4 H#2 的 inf 流進 autoscaler**:名稱錯配 → `waiting_per_replica=inf` → 每個 cooldown 週期都想 + scale-up 到 max_ready。比原本「靜默縮容」好(吵=可發現),且 scrape 失敗本來就產生 inf,非迴歸。可考慮 + 讓 `aggregate_load` 把 inf 樣本當 missing、只留 warning log 作為訊號。 +- **複審 #5 Router live-set failover 尾巴**:live 候選全試過失敗後 fallback 回非 live 的 config 實例(mixed + 下是必死的 router-container localhost),受 `max_attempts ≤ 3` 限制,影響有界。 From cf609c87ec6223bac716d70950c1580a3719ba7a Mon Sep 17 00:00:00 2001 From: max Date: Sat, 4 Jul 2026 16:13:56 +0800 Subject: [PATCH 02/20] =?UTF-8?q?feat(engines):=20converge=20engine=20know?= =?UTF-8?q?ledge=20=E2=80=94=20fail-loud=20phantom=20+=20/api/engines=20(P?= =?UTF-8?q?1)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Implement the extensibility review's P1 (docs/multi-engine-extensibility-review_zh-TW.md §5) so a 4th engine costs "one launcher + one table + one image", and fix the current `trtllm` phantom-engine trap. P1.1 — launcher is the single authority; fail loud on a phantom engine: - Drop `trtllm` from the schema engine Literal (no launcher backs it) — a config using it now fails to load instead of silently vanishing from the backend while the router black-holes its traffic. - unregistered_engine_groups(config, launchers): build_registry logs a loud error for any LLM group whose engine no registered launcher claims. - Remove `trtllm` from the frontend engine list/order/colour maps. P1.2 — GET /api/engines capabilities API (kills the frontend capability drift): - Launchers gain catalogue metadata (metric_prefix, inapplicable_keys, paste_example); manager.engine_catalogue() serves it at /api/engines. - AddModelDialog consumes it: ENGINE_OPTIONS, engineHasSleep (caps.includes('sleep')), engineHasKvShare, paste placeholder and inapplicable-key masking are all API-driven instead of hardcoding CAP_* knowledge. P1.3 — router startup self-check: - unknown_metric_engines(config) warns at load/reload for any engine with no METRIC_NAMES_BY_ENGINE entry, making backend↔router table drift visible. P2/P3 (shared engine-catalog package, Launcher.available()/prepare()/readiness_probe hooks, engine_args namespace, unified Grafana) remain roadmap for when a 4th engine lands. Tests: backend 489, router 129, schema 5, frontend type-check + build green. Co-Authored-By: Claude Opus 4.8 --- apps/backend/app/api/engines.py | 22 ++ apps/backend/app/llmops/launchers.py | 25 ++ apps/backend/app/llmops/manager.py | 52 +++- apps/backend/app/main.py | 2 + apps/backend/tests/api/test_engines_routes.py | 19 ++ .../backend/tests/unit/test_manager_engine.py | 28 +- .../src/components/AddModelDialog.vue | 49 +-- .../src/components/ModelGroupCard.vue | 1 - apps/frontend_llmops/src/lib/api.ts | 4 + apps/frontend_llmops/src/types/api.ts | 11 + apps/frontend_llmops/src/views/ModelsView.vue | 3 +- apps/router-server/src/llm_router/main.py | 11 + .../src/llm_router/vllm_metrics_client.py | 13 + .../tests/unit/test_vllm_metrics_client.py | 17 ++ ...multi-engine-extensibility-review_zh-TW.md | 279 ++++++++++++++++++ packages/config-schema/schema.py | 8 +- 16 files changed, 518 insertions(+), 26 deletions(-) create mode 100644 apps/backend/app/api/engines.py create mode 100644 apps/backend/tests/api/test_engines_routes.py create mode 100644 docs/multi-engine-extensibility-review_zh-TW.md diff --git a/apps/backend/app/api/engines.py b/apps/backend/app/api/engines.py new file mode 100644 index 0000000..c364fd5 --- /dev/null +++ b/apps/backend/app/api/engines.py @@ -0,0 +1,22 @@ +"""The registered inference engines and their capabilities. + +Single source of engine knowledge for the dashboard: the frontend reads this to +build the engine picker and to gate form fields on `capabilities` / +`inapplicable_keys`, instead of hardcoding engine names and re-deriving what each +engine can do (which drifts from the backend's launcher CAP_* flags). See +docs/multi-engine-extensibility-review_zh-TW.md §5 (P1). +""" +from fastapi import APIRouter, Depends + +from app.api.deps import get_manager +from app.llmops.manager import ModelManager + +router = APIRouter(prefix="/engines", tags=["engines"]) + + +@router.get("") +async def list_engines(manager: ModelManager = Depends(get_manager)): + """The registered LLM engines, each with: + name, capabilities (CAP_* strings), lora_endpoint_prefix, metric_prefix, + inapplicable_keys (greyed-out model_config keys), paste_example.""" + return {"engines": manager.engine_catalogue()} diff --git a/apps/backend/app/llmops/launchers.py b/apps/backend/app/llmops/launchers.py index e7aaf5d..dd5c1ed 100644 --- a/apps/backend/app/llmops/launchers.py +++ b/apps/backend/app/llmops/launchers.py @@ -156,6 +156,16 @@ class Launcher(Protocol): # here so the LoRA caller reads it from the launcher instead of branching on the # engine name (Low#3). "" default. lora_endpoint_prefix: str = "" + # --- catalogue metadata (GET /api/engines) — the launcher is the single source of + # engine knowledge, so the frontend gates on these instead of hardcoding engine + # names/capabilities (P1). --- + # Prometheus metric-name prefix this engine exposes (e.g. "vllm" -> vllm:*). "" = none. + metric_prefix: str = "" + # model_config keys that don't apply to this engine (its CLI builder drops them); + # the dashboard greys them out. Empty = every key applies. + inapplicable_keys: frozenset[str] = frozenset() + # An example launch command for the Add-Model paste box. "" = no example. + paste_example: str = "" def keys(self, config) -> list[str]: """All instance keys this launcher defines in the config (its engine only).""" @@ -173,6 +183,10 @@ class VllmLauncher: CAP_SLEEP, CAP_RUNTIME_LORA, CAP_LORA_MODULES, CAP_KV_TRANSFER, CAP_METRICS_VLLM, }) lora_endpoint_prefix = "/v1" # vLLM serves /v1/{load,unload}_lora_adapter + metric_prefix = "vllm" + paste_example = ("CUDA_VISIBLE_DEVICES=0 vllm serve Qwen/Qwen2.5-3B-Instruct " + "--port 8020 --dtype float16 --max-model-len 4096 " + "--gpu-memory-utilization 0.85") def keys(self, config) -> list[str]: out: list[str] = [] @@ -390,6 +404,10 @@ class SglangLauncher: # See docs/multi-backend-engine-design_zh-CN.md §5.2. capabilities = frozenset({CAP_RUNTIME_LORA, CAP_LORA_MODULES, CAP_METRICS_SGLANG}) lora_endpoint_prefix = "" # SGLang serves /{load,unload}_lora_adapter (no /v1) + metric_prefix = "sglang" + paste_example = ("CUDA_VISIBLE_DEVICES=0 python -m sglang.launch_server " + "--model-path Qwen/Qwen3-0.6B --port 8030 --context-length 4096 " + "--mem-fraction-static 0.85") def keys(self, config) -> list[str]: out: list[str] = [] @@ -553,6 +571,13 @@ class LlamacppLauncher: # metrics_llamacpp: launches with --metrics; the router normalizes llamacpp:* into # the same {waiting,running} load shape (no kv-usage dim). See §3/§10 of the design doc. capabilities = frozenset({CAP_LORA_MODULES, CAP_METRICS_LLAMACPP}) + metric_prefix = "llamacpp" + # These vLLM/SGLang knobs have no llama.cpp equivalent (the CLI builder drops + # them) — the dashboard greys them out. Sourced from _LLAMACPP_DROP_KEYS. + inapplicable_keys = _LLAMACPP_DROP_KEYS + paste_example = ("CUDA_VISIBLE_DEVICES=0 llama-server -hf " + "Qwen/Qwen2.5-0.5B-Instruct-GGUF:Q4_K_M -a qwen25-05b-gguf " + "-c 4096 -ngl 99 --port 8091") def keys(self, config) -> list[str]: out: list[str] = [] diff --git a/apps/backend/app/llmops/manager.py b/apps/backend/app/llmops/manager.py index 77dc887..d494bff 100644 --- a/apps/backend/app/llmops/manager.py +++ b/apps/backend/app/llmops/manager.py @@ -75,8 +75,36 @@ class SleepError(RuntimeError): precondition failure (wrong state / not sleep-capable).""" +def unregistered_engine_groups(config, launchers: list[Launcher]) -> dict[str, str]: + """LLM groups whose `engine` no registered launcher claims — {group: engine}. + + The registered launchers are the single authority for which engines exist + (P1). A group with an engine no launcher handles is a *phantom*: schema-valid but + silently unrunnable — build_registry would create no instance for it, so it + vanishes from the backend while the router still lists it and black-holes its + traffic. Callers fail loud on a non-empty result instead of dropping it silently.""" + known = {l.engine for l in launchers} + out: dict[str, str] = {} + for group, engine in config.LLM_engines.items(): + name = getattr(engine.settings, "engine", "vllm") + if name not in known: + out[group] = name + return out + + def build_registry(config, config_path: str, launchers: list[Launcher]) -> ModelRegistry: - """Enumerate every instance every launcher defines, all STOPPED initially.""" + """Enumerate every instance every launcher defines, all STOPPED initially. + + Logs a loud error for any LLM group whose engine no launcher claims (a phantom + engine — see unregistered_engine_groups); it produces no instance, so surfacing it + here keeps it from disappearing silently.""" + phantom = unregistered_engine_groups(config, launchers) + if phantom: + logger.error( + "config has group(s) with an engine no launcher handles (they will NOT " + "run — add a launcher or fix the engine): %s", + ", ".join(f"{g} (engine={e})" for g, e in sorted(phantom.items())), + ) registry = ModelRegistry() for launcher in launchers: for key in launcher.keys(config): @@ -495,6 +523,28 @@ async def unschedulable_reasons(self) -> dict[str, str]: out[key] = f"no live node runs engine '{engine}'" return out + def engine_catalogue(self) -> list[dict]: + """The registered LLM engines + their capability metadata — the single source + the dashboard (and any external consumer) reads instead of re-hardcoding which + engines exist and what each can do (P1). Deduped by engine name, sorted. Serves + GET /api/engines; the frontend gates its form on `capabilities` / + `inapplicable_keys` rather than engine-name conditionals.""" + out: list[dict] = [] + seen: set[str] = set() + for (kind, engine), launcher in self._launchers.items(): + if kind != ModelKind.LLM or engine in seen: + continue + seen.add(engine) + out.append({ + "name": engine, + "capabilities": sorted(launcher.capabilities), + "lora_endpoint_prefix": getattr(launcher, "lora_endpoint_prefix", ""), + "metric_prefix": getattr(launcher, "metric_prefix", ""), + "inapplicable_keys": sorted(getattr(launcher, "inapplicable_keys", frozenset())), + "paste_example": getattr(launcher, "paste_example", ""), + }) + return sorted(out, key=lambda e: e["name"]) + async def get(self, key: str) -> ModelInstance: return self._require(key) diff --git a/apps/backend/app/main.py b/apps/backend/app/main.py index 1cb3cfc..670c133 100644 --- a/apps/backend/app/main.py +++ b/apps/backend/app/main.py @@ -23,6 +23,7 @@ from app.api import datasets as dataset_routes from app.api import downloads as download_routes from app.api import embedding as embedding_routes +from app.api import engines as engine_routes from app.api import eval as eval_routes from app.api import lora as lora_routes from app.api import metrics as metrics_routes @@ -304,6 +305,7 @@ def create_app() -> FastAPI: # Snapshots the overlay whenever a request changes it (for history/rollback). install_config_version_middleware(app) app.include_router(model_routes.router, prefix="/api") + app.include_router(engine_routes.router, prefix="/api") app.include_router(system_routes.router, prefix="/api") app.include_router(config_routes.router, prefix="/api") app.include_router(observability_routes.router, prefix="/api") diff --git a/apps/backend/tests/api/test_engines_routes.py b/apps/backend/tests/api/test_engines_routes.py new file mode 100644 index 0000000..4454302 --- /dev/null +++ b/apps/backend/tests/api/test_engines_routes.py @@ -0,0 +1,19 @@ +import pytest + +pytestmark = pytest.mark.api + + +def test_list_engines_returns_registered_launchers(client): + # The catalogue reflects the launchers the manager actually registered (P1) — the + # test app registers vLLM. The phantom trtllm (no launcher) never appears. + r = client.get("/api/engines") + assert r.status_code == 200 + engines = {e["name"]: e for e in r.json()["engines"]} + assert "vllm" in engines + assert "trtllm" not in engines + vllm = engines["vllm"] + # Capability metadata + example command drive the frontend's gating (no hardcoding). + assert "sleep" in vllm["capabilities"] + assert vllm["metric_prefix"] == "vllm" + assert vllm["lora_endpoint_prefix"] == "/v1" + assert vllm["paste_example"] diff --git a/apps/backend/tests/unit/test_manager_engine.py b/apps/backend/tests/unit/test_manager_engine.py index 09d786a..bfebd61 100644 --- a/apps/backend/tests/unit/test_manager_engine.py +++ b/apps/backend/tests/unit/test_manager_engine.py @@ -168,13 +168,14 @@ async def test_load_lora_rejected_when_engine_lacks_capability(tmp_path): async def test_create_overlay_model_rejects_unregistered_engine(tmp_path): - # trtllm is a valid engine name in the schema, but no launcher is registered - # for it here — the manager must refuse cleanly, not KeyError into a 500. + # llamacpp is a valid schema engine, but no launcher is registered for it in this + # fake manager (only vllm + the sglang fake) — the manager must refuse cleanly, + # not KeyError into a 500. mgr = _manager_with_fake(tmp_path) with pytest.raises(ModelConflict, match="unsupported engine"): await mgr.create_overlay_model( "Brand", {"id": "z", "host": "localhost", "port": 8040}, - {"model_tag": "org/brand", "engine": "trtllm"}, + {"model_tag": "org/brand", "engine": "llamacpp"}, ) @@ -309,3 +310,24 @@ async def test_owning_node_api_url_none_when_local(tmp_path): BackendSettings(instance_id="vllm-node"), store=store, overlay_path=str(tmp_path/"o.json")) assert await mgr.owning_node_api_url("S::a") is None # local -> read locally + + +def test_unregistered_engine_groups_detects_phantom(tmp_path): + # A group whose engine no registered launcher claims must be reported (fail loud), + # not silently dropped from the registry. (P1) + from app.llmops.manager import unregistered_engine_groups + from app.llmops.launchers import VllmLauncher + from schema import load_config + cfg = tmp_path / "c.yaml" + cfg.write_text( + "server:\n port: 8887\n" + "LLM_engines:\n" + " A:\n instances:\n - {id: a, host: localhost, port: 8001}\n" + " model_config: {model_tag: org/a}\n" # vllm (default) + " B:\n instances:\n - {id: b, host: localhost, port: 8002}\n" + " model_config: {model_tag: org/b, engine: sglang}\n", # no sglang launcher + encoding="utf-8", + ) + config = load_config(str(cfg)) + phantom = unregistered_engine_groups(config, [VllmLauncher()]) + assert phantom == {"B": "sglang"} diff --git a/apps/frontend_llmops/src/components/AddModelDialog.vue b/apps/frontend_llmops/src/components/AddModelDialog.vue index cfbd817..bf51f84 100644 --- a/apps/frontend_llmops/src/components/AddModelDialog.vue +++ b/apps/frontend_llmops/src/components/AddModelDialog.vue @@ -15,7 +15,7 @@ import { useResourcesStore } from '@/stores/resources' import { formatBytes } from '@/lib/utils' import { ROUTING_STRATEGIES, routingStrategyLabel } from '@/lib/routingStrategies' import { KV_SHARE_PRESET, isKvShared } from '@/lib/kvSharing' -import type { CachedModel, DownloadJob, KvTransferConfig, LoraAdapter, LoraModule, SettingValue } from '@/types/api' +import type { CachedModel, DownloadJob, EngineInfo, KvTransferConfig, LoraAdapter, LoraModule, SettingValue } from '@/types/api' const open = defineModel('open', { default: false }) const props = defineProps<{ mode?: 'create' | 'edit'; editKey?: string | null }>() @@ -43,23 +43,39 @@ const modelTag = ref('') // Inference engine for the whole group. Different engines map the same concepts to // different launch flags; the launcher translates. vLLM is the default. const engine = ref('vllm') -const ENGINE_OPTIONS = ['vllm', 'sglang', 'llamacpp', 'trtllm'] +// Registered engines + capabilities from the backend (single source of truth — the +// form gates on these instead of hardcoding engine names/features). Loaded on mount; +// falls back to a vLLM-only stub so the dialog works before the fetch resolves. (P1) +const engineCatalogue = ref([ + { name: 'vllm', capabilities: ['sleep', 'kv_transfer'], lora_endpoint_prefix: '/v1', metric_prefix: 'vllm', inapplicable_keys: [], paste_example: '' }, +]) +async function loadEngines() { + try { + const { engines } = await api.listEngines() + if (engines.length) engineCatalogue.value = engines + } catch { + /* keep the fallback stub */ + } +} +loadEngines() +const ENGINE_OPTIONS = computed(() => engineCatalogue.value.map((e) => e.name)) +const engineMeta = (name: string) => engineCatalogue.value.find((e) => e.name === name) +const selectedEngine = computed(() => engineMeta(engine.value)) // Which engine's command the paste box expects — drives the example placeholder and // is sent to the parser so an ambiguous command still parses for the right engine. const pasteEngine = ref('vllm') -const PASTE_PLACEHOLDER: Record = { - vllm: 'CUDA_VISIBLE_DEVICES=0 vllm serve Qwen/Qwen2.5-3B-Instruct --port 8020 --dtype float16 --max-model-len 4096 --gpu-memory-utilization 0.85', - sglang: 'CUDA_VISIBLE_DEVICES=0 python -m sglang.launch_server --model-path Qwen/Qwen3-0.6B --port 8030 --context-length 4096 --mem-fraction-static 0.85', - llamacpp: 'CUDA_VISIBLE_DEVICES=0 llama-server -hf Qwen/Qwen2.5-0.5B-Instruct-GGUF:Q4_K_M -a qwen25-05b-gguf -c 4096 -ngl 99 --port 8091', -} -const pastePlaceholder = computed(() => PASTE_PLACEHOLDER[pasteEngine.value] ?? PASTE_PLACEHOLDER.vllm) -// Capability gating — the two engines support different things, so the form only -// shows what the selected engine actually has (see launchers.py CAP_* flags). +const pastePlaceholder = computed( + () => engineMeta(pasteEngine.value)?.paste_example || engineMeta('vllm')?.paste_example || '', +) +// Engine-specific UI panels still key on the name (presentation, not capability); +// capability gating derives from the API's capability list (drift-free vs CAP_*). const engineIsVllm = computed(() => engine.value === 'vllm') const engineIsSglang = computed(() => engine.value === 'sglang') const engineIsLlamacpp = computed(() => engine.value === 'llamacpp') -const engineHasSleep = computed(() => engine.value === 'vllm') // /sleep + /wake_up (vLLM only) -const engineHasKvShare = computed(() => engine.value === 'vllm') // OffloadingConnector (vLLM only) +const engineHasSleep = computed(() => selectedEngine.value?.capabilities.includes('sleep') ?? false) +const engineHasKvShare = computed(() => selectedEngine.value?.capabilities.includes('kv_transfer') ?? false) +// model_config keys the selected engine ignores — greyed out / dropped on submit. +const inapplicableKeys = computed(() => new Set(selectedEngine.value?.inapplicable_keys ?? [])) const params = ref<{ key: string; value: string }[]>([]) // Router-only load-balancing policy for the group. Lives in model_config but is // NOT a vLLM flag, so it's edited as its own field and kept out of the raw param @@ -240,10 +256,9 @@ function prefillForEdit() { routingStrategy.value = String(cfg.settings.routing_strategy ?? '') kvShared.value = isKvShared(cfg.settings) sleepMode.value = !!cfg.settings.enable_sleep_mode - // vLLM/SGLang compute knobs that the schema serialises for every group but that - // don't apply to llama.cpp (the launcher drops them). Filtered out when editing a - // llamacpp model so they don't clutter the raw param editor or round-trip on save. - const llamacppInapplicable = new Set(['gpu_memory_utilization', 'tensor_parallel_size', 'dtype']) + // Knobs the selected engine ignores (e.g. gpu_memory_utilization/dtype for + // llama.cpp) come from the engine catalogue, so they don't clutter the raw param + // editor or round-trip on save. Drift-free vs the backend's drop lists. params.value = extractLoras( Object.entries(cfg.settings).filter( ([k2, v]) => @@ -256,7 +271,7 @@ function prefillForEdit() { k2 !== 'routing_strategy' && k2 !== 'kv_transfer_config' && k2 !== 'enable_sleep_mode' && - !(engine.value === 'llamacpp' && llamacppInapplicable.has(k2)), + !inapplicableKeys.value.has(k2), ), ).map(([k2, v]) => ({ key: k2, value: String(v) })) warnings.value = [] diff --git a/apps/frontend_llmops/src/components/ModelGroupCard.vue b/apps/frontend_llmops/src/components/ModelGroupCard.vue index 234d7e5..e23fcb5 100644 --- a/apps/frontend_llmops/src/components/ModelGroupCard.vue +++ b/apps/frontend_llmops/src/components/ModelGroupCard.vue @@ -143,7 +143,6 @@ const ENGINE_COLORS: Record = { vllm: 'var(--chart-1)', sglang: 'var(--chart-2)', llamacpp: 'var(--chart-3)', - trtllm: 'var(--chart-4)', } const engineColor = computed(() => ENGINE_COLORS[engine.value] ?? 'var(--chart-1)') const modelTag = computed(() => { diff --git a/apps/frontend_llmops/src/lib/api.ts b/apps/frontend_llmops/src/lib/api.ts index ccd3436..bcd77d5 100644 --- a/apps/frontend_llmops/src/lib/api.ts +++ b/apps/frontend_llmops/src/lib/api.ts @@ -4,6 +4,7 @@ import type { AuditEntry, CacheInfo, ConfigDiff, + EngineInfo, ConfigSummary, ConfigVersion, CostSummary, @@ -138,6 +139,9 @@ export const api = { // ---- Dashboard Backend ---------------------------------------------------- listModels: () => request(API_BASE, '/api/models'), getModel: (key: string) => request(API_BASE, `/api/models/${enc(key)}`), + // Registered engines + capabilities — the dashboard gates its form on this instead + // of hardcoding engine names/features (P1). + listEngines: () => request<{ engines: EngineInfo[] }>(API_BASE, '/api/engines'), startModel: (key: string, force = false) => request(API_BASE, `/api/models/${enc(key)}/start${force ? '?force=true' : ''}`, { method: 'POST', diff --git a/apps/frontend_llmops/src/types/api.ts b/apps/frontend_llmops/src/types/api.ts index bdde82b..ba5dc52 100644 --- a/apps/frontend_llmops/src/types/api.ts +++ b/apps/frontend_llmops/src/types/api.ts @@ -25,6 +25,17 @@ export interface ModelStartupMetrics { gpu_mem_util?: { current: number | null; effective: number | null; suggested: number | null } } +// One registered inference engine (GET /api/engines) — the single source the form +// gates on instead of hardcoding engine names/capabilities. +export interface EngineInfo { + name: string + capabilities: string[] + lora_endpoint_prefix: string + metric_prefix: string + inapplicable_keys: string[] + paste_example: string +} + export interface ModelView { key: string kind: ModelKind diff --git a/apps/frontend_llmops/src/views/ModelsView.vue b/apps/frontend_llmops/src/views/ModelsView.vue index 4cdfb06..4259ef0 100644 --- a/apps/frontend_llmops/src/views/ModelsView.vue +++ b/apps/frontend_llmops/src/views/ModelsView.vue @@ -31,9 +31,8 @@ const ENGINE_COLORS: Record = { vllm: 'var(--chart-1)', sglang: 'var(--chart-2)', llamacpp: 'var(--chart-3)', - trtllm: 'var(--chart-4)', } -const ENGINE_ORDER = ['vllm', 'sglang', 'llamacpp', 'trtllm'] +const ENGINE_ORDER = ['vllm', 'sglang', 'llamacpp'] const engineOf = (m: ModelView) => m.engine ?? 'vllm' const drawerOpen = ref(false) const selectedKey = ref(null) diff --git a/apps/router-server/src/llm_router/main.py b/apps/router-server/src/llm_router/main.py index 7f99bf8..413cba7 100644 --- a/apps/router-server/src/llm_router/main.py +++ b/apps/router-server/src/llm_router/main.py @@ -87,6 +87,17 @@ async def lifespan(app: FastAPI): if await hydrate_overlay_from_store(app.state.store): app.state.config = load_config_with_overlay(app.state.config_path) + # P1 self-check: warn if any config group uses an engine the router has no metric + # table for (backend↔router engine drift), turning a silent metrics gap into a log. + from src.llm_router.vllm_metrics_client import unknown_metric_engines + unknown = unknown_metric_engines(app.state.config) + if unknown: + logger.warning( + "config uses engine(s) with no METRIC_NAMES_BY_ENGINE entry: %s — their " + "/metrics won't parse (treated as unreachable). Add a metric table.", + ", ".join(sorted(unknown)), + ) + metrics_task = asyncio.create_task(poll_metrics_forever(app, interval=1.0)) app.state.metrics_task = metrics_task try: diff --git a/apps/router-server/src/llm_router/vllm_metrics_client.py b/apps/router-server/src/llm_router/vllm_metrics_client.py index ff2f9d5..3d982fa 100644 --- a/apps/router-server/src/llm_router/vllm_metrics_client.py +++ b/apps/router-server/src/llm_router/vllm_metrics_client.py @@ -103,6 +103,19 @@ def engine_sleep_capable(engine: str) -> bool: return engine in ENGINE_SLEEP_CAPABLE +def unknown_metric_engines(config: dict) -> set[str]: + """Engine names used by config groups that have no METRIC_NAMES_BY_ENGINE entry. + The router would parse their /metrics with the vLLM name table, match nothing, and + (post-H#2) treat them as unreachable. Surfacing them at load/reload turns a silent + metrics gap into a visible warning, catching backend↔router engine-table drift (P1 + self-check). See docs/multi-engine-extensibility-review_zh-TW.md §5.""" + used = { + (mc.get("model_config") or {}).get("engine", "vllm") + for mc in (config.get("LLM_engines") or {}).values() + } + return {e for e in used if e not in METRIC_NAMES_BY_ENGINE} + + class VLLMMetricsClient: # Default (vLLM) names; engine-specific lookups use METRIC_NAMES_BY_ENGINE. METRIC_NAMES = METRIC_NAMES_BY_ENGINE["vllm"] diff --git a/apps/router-server/tests/unit/test_vllm_metrics_client.py b/apps/router-server/tests/unit/test_vllm_metrics_client.py index 67dc539..681fade 100644 --- a/apps/router-server/tests/unit/test_vllm_metrics_client.py +++ b/apps/router-server/tests/unit/test_vllm_metrics_client.py @@ -136,3 +136,20 @@ def test_engine_sleep_capable_only_vllm(): assert engine_sleep_capable("sglang") is False assert engine_sleep_capable("llamacpp") is False assert engine_sleep_capable("unknown") is False + + +def test_unknown_metric_engines_flags_missing_table(): + from src.llm_router.vllm_metrics_client import unknown_metric_engines + config = {"LLM_engines": { + "a": {"model_config": {"engine": "vllm"}}, + "b": {"model_config": {"engine": "sglang"}}, + "c": {"model_config": {"engine": "trtllm"}}, # no metric table + "d": {"model_config": {}}, # defaults vllm + }} + assert unknown_metric_engines(config) == {"trtllm"} + + +def test_unknown_metric_engines_empty_when_all_known(): + from src.llm_router.vllm_metrics_client import unknown_metric_engines + config = {"LLM_engines": {"a": {"model_config": {"engine": "llamacpp"}}}} + assert unknown_metric_engines(config) == set() diff --git a/docs/multi-engine-extensibility-review_zh-TW.md b/docs/multi-engine-extensibility-review_zh-TW.md new file mode 100644 index 0000000..8e5c05c --- /dev/null +++ b/docs/multi-engine-extensibility-review_zh-TW.md @@ -0,0 +1,279 @@ +# 多引擎架構可擴展性評估 — 新增第 N 個引擎要付多少代價? + +> 評估日期:2026-07-04。問題:目前三引擎(vLLM / SGLang / llama.cpp)的架構,未來要加第四個 +> 引擎(如 TensorRT-LLM、MLC、Ollama…)時,**需要重構嗎?還是只要修改?** 方法:實際盤點 +> 「新增一個引擎」會觸碰的每一個檔案與每一處 engine 分支(backend / router / frontend / +> schema / deploy),並以 SGLang、llama.cpp 兩次實際加入的歷史為經驗證據。 + +--- + +## 1. 結論(TL;DR) + +**不需要重構。核心接縫選對了,而且已經被驗證過兩次。** Launcher protocol +(`(kind, engine)` dispatch)+ capability frozenset + engine-neutral 參數 + per-engine 指標名 +對照表,這套設計讓 SGLang 和 llama.cpp 的加入都是**純增量**(新增類別 + 表格條目),沒有動過 +核心狀態機、reconciler、scheduler 或 router 的主流程——這是「架構對了」最有力的證據。 + +但「加一個引擎的代價」不在核心,而在**外圍**:引擎清單與引擎知識散落在 ≥ 5 個彼此不互查的 +地方(schema Literal、launcher 註冊、前端硬編碼、Makefile、router 對照表),每加一個引擎要 +人肉同步一輪,漏一處就是靜默 drift。**現在的程式碼裡就躺著一個現成的 drift 證據:`trtllm`** +(詳見 §3.1)。 + +建議:加第四個引擎之前,先做 §5 的 P1 收斂(約 1–2 天工作量,收斂引擎目錄 + capabilities +API),之後每個新引擎的成本就從「改 5 個 app、~14 個觸點」降到「一個 launcher 類別 + 一條 +指標表 + 一個 Dockerfile」。 + +| 問題 | 答案 | +|---|---| +| 需要重購(rebuild)嗎? | **不需要**。核心抽象(Launcher / capability / 指標正規化)方向正確且已兩次實證。 | +| 需要修改嗎? | **需要,但都是收斂性修改**:把散落的引擎知識集中成單一目錄,並補上跨層一致性檢查。 | +| 今天直接加第四個引擎可行嗎? | 可行,約 1–2 天(launcher + 測試)+ 前端/部署零散修改;但會再複製一輪硬編碼,drift 面積繼續變大。 | + +--- + +## 2. 現況:新增第 4 個引擎的完整觸點清單 + +以下是照抄 SGLang / llama.cpp 加入路徑、逐檔盤點出的**實際**觸點。分三級: +✅ = 設計良好,增量修改;⚠️ = 能用但是硬編碼複製;❌ = 缺失/陷阱。 + +### 2.1 Backend(核心——設計良好) + +| # | 檔案 | 要做什麼 | 評級 | +|---|---|---|---| +| 1 | `apps/backend/app/llmops/launchers.py` | 新 Launcher 類別:`engine` 名、`capabilities`、`lora_endpoint_prefix`、`keys()`、`build_spec()` + 一個純函式 CLI builder(參數翻譯/skip 清單) | ✅ 這是主要工作量,且完全隔離、可單元測試 | +| 2 | `apps/backend/app/main.py:129` | launchers 清單加一項(1 行) | ✅ | +| 3 | `packages/config-schema/schema.py:62` | `engine: Literal[...]` 加一個值 | ⚠️ 與 #2 無交叉檢查(見 §3.1) | +| 4 | `apps/backend/app/services/vllm_command.py` | (選配)paste-command 解析器 + `parse_command` 的 sniff 規則 | ⚠️ 每引擎一個 if 分支 | +| 5 | `apps/backend/app/perf/manager.py:74`、`lora_convert.py` | 只有引擎有特殊格式時才需要(llama.cpp 的 GGUF tokenizer / 轉檔) | ✅ 屬引擎本質差異,難免 | + +**核心流程不用動**:manager 的 sleep/LoRA gating 走 `_llm_engine_capabilities()`(launcher 查表)、 +`_post_lora` 走 `lora_endpoint_prefix`、reconciler/probe 是通用 `/health`、scheduler 的 +`node_supports()` 對引擎名是不透明字串(node 以 JSON 宣告)、`_defer_to_owner` 引擎無關、 +prometheus_targets 的 `engine` label 通用。**這些是架構的「對」的部分。** + +### 2.2 Router(單檔集中——設計良好) + +| # | 檔案 | 要做什麼 | 評級 | +|---|---|---|---| +| 6 | `vllm_metrics_client.py:59-85` | `METRIC_NAMES_BY_ENGINE` 加一條(5 個指標名) | ✅ 單一宣告點 | +| 7 | `vllm_metrics_client.py` `ENGINE_SLEEP_CAPABLE` | 引擎若支援 sleep 才加 | ⚠️ 與 backend 的 `CAP_SLEEP` 是**兩份平行真相**(見 §3.3) | + +metrics_poller 讀 config 的 `model_config.engine` 自動選 parser,無需修改;未知引擎名 fallback +到 vLLM 名稱表 → 解析全 miss → 經 H#2 修復後回 unreachable + warning(fail-loud,可接受)。 + +### 2.3 Frontend(最差的一層——全部硬編碼) + +| # | 檔案 | 要做什麼 | 評級 | +|---|---|---|---| +| 8 | `AddModelDialog.vue:46` | `ENGINE_OPTIONS = ['vllm','sglang','llamacpp','trtllm']` 加值 | ❌ 硬編碼,且**已經**含有一個不存在的引擎 | +| 9 | `AddModelDialog.vue:58-62` | `engineIsVllm / engineIsSglang / engineIsLlamacpp / engineHasSleep / engineHasKvShare` — 每引擎一組 computed | ❌ 能力判斷用引擎名硬編,與 backend 的 CAP_* 完全脫鉤 | +| 10 | `AddModelDialog.vue:50, 246, 362, 432` | paste placeholder、`llamacppInapplicable` 參數遮蔽集、LoRA 格式分流、tool-calling 文案 — 各一份 per-engine 表 | ⚠️ | +| 11 | `i18n/locales/*.ts` | 引擎相關文案 | ⚠️ | + +**根因:backend 沒有任何 API 暴露 launcher 的 capabilities**(`grep capabilities app/api/` 為空)。 +前端被迫把「哪些引擎存在、各自會什麼」重新發明一遍,這是 drift 的最大來源。 + +### 2.4 Deploy / 監控 + +| # | 檔案 | 要做什麼 | 評級 | +|---|---|---|---| +| 12 | `deploy/engine-.Dockerfile` | 新引擎 base image + backend 程式碼 | ✅ 模式清楚(已有三個範本) | +| 13 | `deploy/docker-compose.mixed.yaml` | 新 service(profile、`LLMOPS_NODE_ENGINES=`、`LLMOPS_PROMETHEUS_SD_PATH=/sd/targets-.json`) | ✅ compose profile + Makefile `ENGINES=` 已參數化 | +| 14 | `Makefile:12` | ENGINES 預設清單加值 | ⚠️ 又一份引擎清單 | +| 15 | Grafana dashboards | 新引擎指標前綴的 panel(sglang 已有獨立 dashboard 目錄) | ⚠️ 每引擎一套,無法避免但需記入成本 | + +Prometheus 的 file_sd 用 glob 自動發現 `targets-*.json`,**無需改 scrape config**——這是設計好的部分。 + +--- + +## 3. 結構性風險(加引擎會放大的問題) + +### 3.1 引擎清單有 ≥5 份,彼此不互查——`trtllm` 是現成的 drift 實證 + +引擎清單目前同時存在於:schema `Literal`、`main.py` launcher 註冊、前端 `ENGINE_OPTIONS`、 +Makefile `ENGINES`、router 指標表。其中 **schema 和前端已經含有 `trtllm`,但沒有任何 +launcher 實作它**。後果(今天就能觸發): + +- `config.yaml` 寫一個 `engine: trtllm` 的群組 → **schema 驗證通過** → `build_registry` 時每個 + launcher 都不認領 → 該群組在 backend **靜默消失**(registry 沒有 instance、dashboard 看不到、 + 沒有任何錯誤); +- 但 **router 照樣從 config 讀到這個群組** → `/v1/models` 列出它、請求被路由到 config 位址 → + 黑洞 503; +- dashboard 的 Add Model 下拉可以選 trtllm → 送出 → `create_overlay_model` 500 + (`unsupported engine — no launcher registered`)。 + +同一個引擎名,三層三種行為(靜默消失 / 黑洞路由 / 500)。這不是假設性風險,是**現在**的行為。 + +### 3.2 前端零 capability 來源 + +`engineHasSleep = engine === 'vllm'` 這類判斷,等於把 backend `CAP_*` 的知識用另一種語言再寫 +一遍。backend 加了能力(例如未來 SGLang 支援 sleep)或加了引擎,前端不會跟著動——UI gating +與實際能力脫鉤只是時間問題。原多引擎設計文件說「UI 一律以 capability gate,而非引擎名」, +**前端這層目前完全沒有做到**(它拿不到 capability,想做也做不到)。 + +### 3.3 backend 與 router 各持一份「引擎→行為」表 + +backend:launcher 的 `capabilities` / `lora_endpoint_prefix`。router:`METRIC_NAMES_BY_ENGINE` / +`ENGINE_SLEEP_CAPABLE`。兩邊靠人工同步(例:sleep 能力在 backend 是 `CAP_SLEEP`、在 router 是 +另一個 frozenset)。router 只消費 config dict、拿不到 launcher 物件,是這個分裂的結構原因。 +三個引擎時還管得住;五、六個引擎時,「backend 說會 sleep、router 不去 probe /is_sleeping」這 +種半套 drift 幾乎必然發生。 + +### 3.4 collapsed(單機單 image)模式沒有「這台跑得動嗎」的檢查 + +mixed 模式有 `LLMOPS_NODE_ENGINES` gate;collapsed 模式 `node_engines` 為空 → `_node_can_run` +恆 True → 在 vLLM-only image 裡新增一個 sglang 群組並 start → spawn +`python -m sglang.launch_server` → module 不存在 → FAILED + auto-restart 消耗重啟預算。 +失敗有出口(log tail 可見),但體驗是「crash 三次才知道 image 沒裝」,而不是 create/start +當下的明確拒絕。引擎越多、image 排列組合越多,這個坑越常踩。 + +### 3.5 新引擎必須滿足的**隱含契約**(沒有寫在任何 interface 上) + +目前的 Launcher/LaunchSpec 模型隱含假設每個引擎都是: + +1. **單行程、CLI 旗標啟動**(`command: list[str]` + `spawn_process` + process-group kill); +2. **HTTP `/health` 200 = ready**(probe 通用); +3. **OpenAI 相容 `/v1` 端點**,且 `/v1/models` 廣告 served name(router forward、adopt 身分驗證都靠它); +4. **Prometheus text `/metrics`**,指標名有獨特前綴; +5. **啟動進度反映在 log 檔成長**(progress-aware timeout 靠 `os.path.getsize`); +6. **綁 port 即可服務**(無需預熱/編譯階段)。 + +vLLM / SGLang / llama.cpp 恰好都符合,所以這些假設從未被挑戰。但候選的第四引擎很可能踩線: +**TensorRT-LLM**(需要離線 engine build,「啟動」包含編譯/載入 artifact,不是 spawn-and-probe; +若走 Triton 是多行程)、**disaggregated prefill/decode 架構**(一個「instance」= 多個行程)、 +**Ollama**(一個 daemon 管多模型,不是一行程一模型)。這些不需要重構才能支援,但需要在 +Launcher protocol 上**預留擴充點**(見 §5 P2),否則屆時會被迫在 manager/reconciler 裡塞 +engine 特判——那才是架構開始腐爛的起點。 + +### 3.6 `model_config` 的 extra="allow" 共用命名空間 + +engine-neutral 鍵 + 各引擎原生鍵全部混在同一層 dict,靠每個 launcher 的 skip/drop 清單自保 +(`_SKIP_CLI_KEYS`、`_SGLANG_SKIP_CLI_KEYS`、`_LLAMACPP_SKIP_CLI_KEYS` + `_ROUTER_ONLY_KEYS`)。 +三個引擎已出現「vLLM 的 LoRA 旗標會讓 llama-server 直接 exit」這種跨引擎污染,靠 drop 清單 +擋住。引擎每多一個,清單維護是 O(引擎數 × 雜項鍵數)。尚可管理,但方向上應該收斂 +(見 §5 P2 的 `engine_args` 建議)。 + +--- + +## 4. 為什麼**不**建議重構(現在的「對」) + +1. **Dispatch 接縫正確**:`(kind, engine)` → Launcher,加引擎不碰既有 launcher; +2. **capability gate 在 backend 核心是真的**:sleep / runtime-LoRA / autoscaler 降級全部查 + capability,不查引擎名(唯一殘留的 engine-name 分支已在 L#3 修復中改掉); +3. **指標正規化單表**:五個概念指標名 → 同一個 `VLLMInstanceMetrics` 形狀,downstream + (routing score、autoscaler、dashboard)完全引擎無關; +4. **排程/HA 對引擎名不透明**:node 宣告字串集合、scheduler 集合比對,加引擎零修改; +5. **部署已參數化**:compose profiles + `ENGINES=`、file_sd glob、per-node SD 檔; +6. **兩次實證**:SGLang 與 llama.cpp 的 git 歷史顯示核心(manager/reconciler/scheduler/router + 主流程)在兩次引擎加入中幾乎零改動。 + +重構(例如抽成 plugin 系統、動態載入 entry-point)在目前規模是過度設計:引擎數 ≤ 6、 +全部 in-tree、發佈同步,Launcher 類別 + 表格條目的成本已經夠低。**問題不是抽象不夠, +是同一份知識抄了太多份。** + +--- + +## 5. 具體建議(分階段) + +### P1 — 加第四個引擎**之前**做(收斂知識,約 1–2 天) + +1. **單一引擎目錄(single source of truth)** + - 以「已註冊 launcher」為唯一權威:`build_merged_config` / overlay 驗證時,對每個群組的 + `engine` 檢查是否有 launcher 認領,**沒有就 fail loud**(config 拒載或明確 error, + 不再靜默消失); + - schema 的 `Literal` 移除 `trtllm`(或改為執行期驗證,Literal 只留註解用途); + - 前端 `ENGINE_OPTIONS` 移除 `trtllm`,改吃 #2 的 API; + - router `/reload` 時對未知 engine 名 log warning(現在 H#2 的 unreachable 已兜底, + 補一句啟動期警告即可)。 + +2. **`GET /api/engines` capabilities API** + - 從註冊的 launchers 生成:`[{ name, capabilities: [...], lora_endpoint_prefix, + metric_prefix, paste_example, inapplicable_keys }]`; + - 前端刪掉 `ENGINE_OPTIONS`、`engineIsVllm/engineHasSleep/...`、`llamacppInapplicable`、 + `PASTE_PLACEHOLDER`,全部改由 API 驅動(`engineHasSleep` → + `caps.includes('sleep')`); + - 這一步之後,**加引擎的前端成本歸零**(除了真正引擎特有的文案)。 + +3. **router 的引擎表加啟動自檢** + - router 起動時比對 config 中出現的 engine 名 vs `METRIC_NAMES_BY_ENGINE` 鍵集,缺了就 + warning——把 §3.3 的 drift 從「靜默」變「可見」。(真正合併兩份表需要共享套件, + 放 P2。) + +### P2 — 加第四個引擎**時**順手做 + +4. **共享 engine-catalog 套件**(比照 `packages/config-schema` / `packages/llmops-store` 的 + 既有模式):把 `METRIC_NAMES_BY_ENGINE`、sleep 能力、`lora_endpoint_prefix`、engine 名清單 + 放進 `packages/engine-catalog`,backend launcher 與 router 都 import 它——§3.3 的兩份真相 + 合併成一份。 + +5. **Launcher 加 `available()` 自檢**:檢查引擎 binary/module 是否存在於本 image + (`shutil.which("llama-server")` / import probe),collapsed 模式下 create/start 時直接拒絕 + 並給明確訊息,取代「crash 三次才發現」(§3.4)。 + +6. **Launcher protocol 預留兩個可選 hook**(為 §3.5 的非典型引擎鋪路,現有引擎不實作、 + 行為零變): + - `async prepare(spec) -> None`:啟動前置(TRT engine build / artifact 下載),manager 在 + spawn 前呼叫,期間 state 可停留 STARTING; + - `readiness_probe(spec) -> ProbeSpec`:允許覆寫預設 `/health`(自訂路徑/判準)。 + 有這兩個縫,TRT-LLM 類引擎就能 in-protocol 支援,不必在 manager 裡塞特判。 + +7. **`engine_args` 巢狀命名空間(漸進)**:schema 允許 + `model_config.engine_args: dict`(原生旗標放這裡,engine-neutral 鍵留頂層),launcher 的 + CLI builder 優先消費它;頂層 extra keys 保留向後相容。skip/drop 清單從此不再隨引擎數 + 線性成長。 + +### P3 — 記錄即可(與多機 roadmap 並列) + +8. 每引擎一套 Grafana dashboard 是固有成本;可考慮用 label(`engine`)+ 指標 alias recording + rules 做一套統一 dashboard,但投資報酬率視引擎數而定。 +9. registry「目錄 vs 本地狀態」的語意拆分(前次審查 §1 弱點 1)——引擎數增加不會惡化它, + 多機化才會,維持在多機 roadmap。 + +--- + +## 6. 成本估算 + +| 情境 | 工作量(粗估) | +|---|---| +| **今天**直接加第四個典型引擎(單行程、OpenAI 相容、有 /metrics) | launcher + CLI builder + 測試 ~1 天;router 表 10 分鐘;前端硬編碼補一輪 ~0.5 天;Dockerfile + compose + Makefile ~0.5 天;Grafana 視需求。**共 ~2–3 天**,且 drift 面積 +1 | +| 先做 P1(~1–2 天)再加第四個引擎 | launcher + 測試 ~1 天;router 表 10 分鐘;前端 **0**;deploy 同上。**共 ~1.5–2 天**,drift 面積不變 | +| 加一個**非典型**引擎(TRT-LLM 類) | 上述 + P2-6 的兩個 hook(~1 天,一次性)+ 該引擎的 prepare 實作 | + +**最終建議**:不重構;按 P1 → 加引擎 → P2 的順序走。P1 的兩項(引擎目錄一致性檢查 + +`/api/engines`)是報酬率最高的投資——它們同時修掉現存的 `trtllm` 陷阱(§3.1)和前端 +capability 脫鉤(§3.2),並讓之後每一個引擎的邊際成本降到「一個類別 + 一條表 + 一個 image」。 + +--- + +## 附錄:P1 已實作(2026-07-04) + +§5 的 P1 三項全部落地(P2/P3 維持 roadmap): + +- **P1.1 引擎目錄一致性 / fail-loud**: + - schema `Literal` 移除 phantom `trtllm`,只留 `vllm/sglang/llamacpp`(有 launcher 的三個); + engine:trtllm 的 config 現在**載入即報錯**,不再靜默消失 / 黑洞路由。 + - `manager.unregistered_engine_groups(config, launchers)`:以「已註冊 launcher」為權威, + `build_registry` 對無 launcher 認領的群組**記 loud error**(不再無聲丟棄)。 + - 前端 `ENGINE_OPTIONS` / `ENGINE_ORDER` / 顏色表的 `trtllm` 全部移除。 + - [schema.py](../packages/config-schema/schema.py) / [manager.py](../apps/backend/app/llmops/manager.py) + `unregistered_engine_groups` / `build_registry` +- **P1.2 `GET /api/engines` capabilities API**: + - launcher 新增目錄 metadata(`metric_prefix`、`inapplicable_keys`、`paste_example`), + `manager.engine_catalogue()` 產生 `[{name, capabilities, lora_endpoint_prefix, metric_prefix, + inapplicable_keys, paste_example}]`,由 `GET /api/engines` 提供。 + - 前端 `AddModelDialog` 改吃 API:`ENGINE_OPTIONS`、`engineHasSleep`(→`caps.includes('sleep')`)、 + `engineHasKvShare`(→`caps.includes('kv_transfer')`)、paste placeholder、inapplicable 遮蔽鍵 + 全部 API 驅動,不再硬編 CAP_* 知識。之後加引擎的前端成本≈0(除引擎特有文案)。 + - [engines.py](../apps/backend/app/api/engines.py) / [launchers.py](../apps/backend/app/llmops/launchers.py) / + `AddModelDialog.vue` +- **P1.3 router 啟動自檢**:`unknown_metric_engines(config)` 在 lifespan 比對 config 引擎名 vs + `METRIC_NAMES_BY_ENGINE` 鍵集,缺表就 warning——把 §3.3 的 backend↔router drift 從靜默變可見。 + [vllm_metrics_client.py](../apps/router-server/src/llm_router/vllm_metrics_client.py) / `main.py` + +測試:backend 489、router 129、frontend type-check + build 綠;新增 +`test_engines_routes.py`、`test_unregistered_engine_groups_detects_phantom`、 +`test_unknown_metric_engines_*`。 + +**仍為 roadmap**(§5 P2/P3,加第四個引擎時再做):shared `packages/engine-catalog`(合併 backend/router +兩份表)、`Launcher.available()` self-check(§3.4 collapsed image 明確拒絕)、Launcher `prepare()`/ +`readiness_probe()` hook(§3.5 非典型引擎)、`engine_args` 巢狀命名空間(§3.6)、統一 Grafana dashboard。 diff --git a/packages/config-schema/schema.py b/packages/config-schema/schema.py index da6774b..81eba0a 100644 --- a/packages/config-schema/schema.py +++ b/packages/config-schema/schema.py @@ -58,8 +58,12 @@ class EngineModelConfig(BaseModel): # behaviour, byte-for-byte unchanged. Engine-specific flags ride extra="allow" # and are interpreted by that engine's arg builder. A group is single-engine # (all its instances are replicas of one model); mix engines across groups. - # See docs/multi-backend-engine-design_zh-CN.md. - engine: Literal["vllm", "sglang", "llamacpp", "trtllm"] = "vllm" + # See docs/multi-backend-engine-design_zh-CN.md. Every value here MUST have a + # registered Launcher (apps/backend/app/main.py) — a value with no launcher is a + # phantom engine: schema-valid but silently unrunnable. The backend cross-checks + # this on load (validate_registered_engines) and fails loud rather than dropping + # the group silently. Keep this list == the registered engine names. + engine: Literal["vllm", "sglang", "llamacpp"] = "vllm" # Router-facing endpoint kind (NOT an engine flag — the launcher skips it): # chat -> /v1/chat/completions + /v1/completions (a generate model) # embed -> /v1/embeddings (a vLLM pooling embedding model) From 6d129395b053c96cdce96e27e0f650af7ebcc8ac Mon Sep 17 00:00:00 2001 From: max Date: Sat, 4 Jul 2026 19:26:39 +0800 Subject: [PATCH 03/20] docs(engines): add-a-new-engine guide + TensorRT-LLM research/validation/impl-design MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Doc suite for adding a 4th inference engine, and the TensorRT-LLM (trtllm) case: - adding-a-new-engine_zh-TW.md: generic per-engine touchpoint checklist + the implicit Launcher/LaunchSpec contract a new engine must satisfy. - trtllm_research_checklist_zh-TW.md: what to research, benchmarked vs the 3 engines. - trtllm-engine-conversion-guide.md / trtllm-launcher-research.md: spec + conversion workflow (research), updated with verified findings. - trtllm-conversion-validation_zh-TW.md: hands-on validation on sm_86 8GB — HF -> checkpoint -> TRT engine build works, trtllm-serve --backend tensorrt serves and /v1/* infer; corrected metrics (needs return_perf_metrics:true, scrape /prometheus/metrics; the 5 trtllm_* names verified) and sleep (release_memory is AsyncLLM/pytorch-only, 500 on engine backend) with real wall-clock timings. - trtllm-launcher-impl-design_zh-TW.md: engine-first implementation design, phased (Phase 1 BYO pre-built engine; Phase 2 prepare()/auto-build + artifact cache), with per-engine metrics path, LD_LIBRARY_PATH handling, capability table, risks. Co-Authored-By: Claude Opus 4.8 --- docs/adding-a-new-engine_zh-TW.md | 257 ++++++++++ docs/trtllm-conversion-validation_zh-TW.md | 199 ++++++++ docs/trtllm-engine-conversion-guide.md | 526 +++++++++++++++++++++ docs/trtllm-launcher-impl-design_zh-TW.md | 386 +++++++++++++++ docs/trtllm-launcher-research.md | 417 ++++++++++++++++ docs/trtllm_research_checklist_zh-TW.md | 148 ++++++ 6 files changed, 1933 insertions(+) create mode 100644 docs/adding-a-new-engine_zh-TW.md create mode 100644 docs/trtllm-conversion-validation_zh-TW.md create mode 100644 docs/trtllm-engine-conversion-guide.md create mode 100644 docs/trtllm-launcher-impl-design_zh-TW.md create mode 100644 docs/trtllm-launcher-research.md create mode 100644 docs/trtllm_research_checklist_zh-TW.md diff --git a/docs/adding-a-new-engine_zh-TW.md b/docs/adding-a-new-engine_zh-TW.md new file mode 100644 index 0000000..cc59664 --- /dev/null +++ b/docs/adding-a-new-engine_zh-TW.md @@ -0,0 +1,257 @@ +# 新增一個推理引擎要注意什麼(實作指南) + +> 對象:想在既有的 vLLM / SGLang / llama.cpp 之上,加入第 N 個推理引擎(例:TensorRT-LLM、 +> MLC、Ollama…)的人。本文照抄 SGLang、llama.cpp 兩次實際加入的路徑,列出**每一個要動的檔案**、 +> 新引擎必須滿足的**隱含契約**,以及最容易踩的坑。 +> +> 背景設計見 [multi-backend-engine-design_zh-CN.md](multi-backend-engine-design_zh-CN.md); +> 可擴展性評估與 P1 收斂見 [multi-engine-extensibility-review_zh-TW.md](multi-engine-extensibility-review_zh-TW.md)。 + +--- + +## 0. 先確認:你的引擎符合「隱含契約」嗎? + +現有的 `Launcher` / `LaunchSpec` 模型**隱含假設**每個引擎是這樣的(vLLM/SGLang/llama.cpp 剛好都符合): + +1. **單行程、用 CLI 旗標啟動** — `command: list[str]` 由 `spawn_process` 拉起,停止用 process-group kill。 +2. **HTTP `/health` 200 = ready** — reconciler 用通用 probe(llama.cpp 是「載入後才綁 port」→ 啟動期 + connection-refused,也被當 not-ready,OK)。 +3. **OpenAI 相容 `/v1`**,且 `/v1/models` 廣告 served name(router 轉發、boot adopt 身分驗證都靠它)。 +4. **Prometheus text `/metrics`**,指標名有**獨特前綴**(`vllm:` / `sglang:` / `llamacpp:`)。 +5. **啟動進度反映在 log 檔成長** — progress-aware timeout 靠 `os.path.getsize(log)`。 +6. **綁 port 即可服務**(不需離線編譯 / 預熱階段)。 + +**若你的引擎踩線**(例:TensorRT-LLM 需離線 build engine、Triton 是多行程;Ollama 是一個 daemon 管多模型; +disaggregated prefill/decode 一個「instance」= 多行程),先別硬塞——那需要在 Launcher protocol 上補 +`prepare()` / `readiness_probe()` hook(見 §7 roadmap),否則會被迫在 manager/reconciler 裡塞 engine 特判, +那是架構開始腐爛的起點。**典型引擎(單行程、OpenAI 相容、有 /metrics)才是本文的 happy path。** + +--- + +## 1. 觸點總表(照這張清單逐項做) + +✅ = 設計良好、純增量;⚠️ = 需複製一份表;🔴 = 做錯會靜默出事。 + +| # | 檔案 | 要做什麼 | 級別 | +|---|---|---|---| +| **Backend(主要工作量)** |||| +| 1 | `apps/backend/app/llmops/launchers.py` | 新 `Launcher` 類別 + 純函式 CLI builder | ✅ | +| 2 | `apps/backend/app/main.py`(launchers 清單,約 :129) | 註冊新 launcher(1 行) | ✅ 🔴 漏了 = 引擎不存在 | +| 3 | `packages/config-schema/schema.py`(`engine: Literal[...]`,約 :62) | 加上新引擎名 | ⚠️ 🔴 見 §4 | +| 4 | `apps/backend/app/services/vllm_command.py` | (選配)paste-command 解析器 + sniff 規則 | ⚠️ | +| 5 | `apps/backend/app/perf/manager.py`、`app/services/lora_convert.py` | 只有引擎有特殊 tokenizer / 轉檔格式時才需要 | ✅ | +| **Router(單檔集中)** |||| +| 6 | `apps/router-server/src/llm_router/vllm_metrics_client.py` `METRIC_NAMES_BY_ENGINE` | 加一條(5 個指標名) | ✅ 單一宣告點 | +| 7 | 同檔 `ENGINE_SLEEP_CAPABLE` | 引擎若支援 sleep 才加 | ⚠️ | +| **Frontend(P1 之後基本免動)** |||| +| 8 | — | `AddModelDialog` 已改吃 `GET /api/engines`,capability/選單/遮蔽鍵自動帶出 | ✅ 引擎特有 UI 文案才需補 | +| 9 | `apps/frontend_llmops/src/views/ModelsView.vue`、`ModelGroupCard.vue` 顏色表 | (選配)給引擎一個 badge 顏色 | ⚠️ 純顯示 | +| 10 | `i18n/locales/*.ts` | (選配)引擎特有文案 | ⚠️ | +| **Deploy / 監控** |||| +| 11 | `deploy/engine-.Dockerfile` | 新 base image + backend 程式碼(照三個範本抄) | ✅ | +| 12 | `deploy/docker-compose.mixed.yaml` | 新 backend service(profile、`LLMOPS_NODE_ENGINES=`、SD path) | ✅ | +| 13 | `Makefile`(`_ALL_ENGINES`) | 引擎清單加值 | ⚠️ | +| 14 | `deploy/grafana/dashboards//` | (選配)該引擎指標前綴的 dashboard | ⚠️ | + +> **核心流程完全不用動**:manager 的 sleep/LoRA gating 走 capability、`_post_lora` 走 +> `lora_endpoint_prefix`、reconciler/probe 是通用 `/health`、scheduler `node_supports()` 對引擎名是 +> 不透明字串、`_defer_to_owner` 引擎無關、Prometheus file_sd 用 glob 自動發現 `targets-*.json`。**這些是 +> 「架構對了」的部分,不要去碰。** + +--- + +## 2. 寫 Launcher 類別(#1,主要工作) + +在 `launchers.py` 加一個類別。以 `LlamacppLauncher` 為最貼近的範本。必須提供: + +```python +class MyEngineLauncher: + kind = ModelKind.LLM + engine = "myengine" # (kind, engine) 是 dispatch key,必須全域唯一 + + # --- 能力宣告:callers 一律 gate 在這些 capability 上,不看引擎名 --- + capabilities = frozenset({ + CAP_LORA_MODULES, # 啟動時 --lora-modules 靜態掛載 + CAP_METRICS_MYENGINE, # 有 Prometheus /metrics(需在 §6 定義新 CAP 常數) + # CAP_SLEEP # 只有真的有 /sleep+/wake_up 才加(見 §5 坑) + # CAP_RUNTIME_LORA # 只有真的能 runtime 熱掛 LoRA 才加 + # CAP_KV_TRANSFER # 只有真的能跨實例共享 KV 才加 + }) + + # --- LoRA runtime 端點前綴(vLLM 是 "/v1",SGLang 是 "") --- + lora_endpoint_prefix = "" + + # --- 目錄 metadata(GET /api/engines;前端靠它自動長出 UI,不要漏)--- + metric_prefix = "myengine" # Prometheus 指標名前綴(myengine:*) + inapplicable_keys = frozenset({ # 這個引擎沒有、CLI builder 會 drop 的 model_config 鍵 + "gpu_memory_utilization", ... # 前端會把這些欄位 grey out + }) + paste_example = "CUDA_VISIBLE_DEVICES=0 my-engine-server --model ... --port 8099" + + def keys(self, config) -> list[str]: + # 回傳「engine 屬於自己」的所有 :: + out = [] + for tag, eng in config.LLM_engines.items(): + if getattr(eng.settings, "engine", "vllm") != self.engine: + continue + out.extend(f"{tag}::{i.id}" for i in eng.instances) + return out + + def build_spec(self, config, config_path, key) -> LaunchSpec: + # 解析 merged model_config -> LaunchSpec(見下) + ... +``` + +### `build_spec` 必須填對的 `LaunchSpec` 欄位 + +照抄現有 launcher 的 merge 樣板(`merged = engine.settings.model_dump(); merged.update(inst.model_dump())`, +單卡 `cuda_device`→`CUDA_VISIBLE_DEVICES`,`merged.pop("id")`),然後: + +| 欄位 | 說明 / 坑 | +|---|---| +| `command` | `["my-engine-server", *build_myengine_cli_args(cli_cfg)]` | +| `host` / `port` | 用 `inst.host` / `inst.port`;bind host 走 `LLMOPS_VLLM_BIND_HOST`(共用,0.0.0.0 供跨容器路由) | +| `probe_url` | `http://{host}:{port}/health` | +| `engine` / `capabilities` | `self.engine` / `self.capabilities` | +| `model_tag` | `engine.settings.model_tag` | +| **`served_name`** | **`merged.get("served_model_name") or engine.settings.model_tag`** — 🔴 漏了會害 boot adopt 認不得自己、在同 port 起第二份行程 → crash loop(見 §5) | +| `sleep_enabled` | 只有支援 sleep 的引擎才 `True`,且要在 env 開對應 dev flag | + +### CLI builder(純函式,好單測) + +比照 `build_llamacpp_cli_args`:把 engine-neutral 的 `model_config` 鍵翻譯成該引擎的旗標。三個要點: + +- **skip/drop 清單**:`_ROUTER_ONLY_KEYS`(`routing_strategy`/`kind`/`engine`——永遠不進 CLI)必扣掉; + 加上你引擎的 `_MYENGINE_SKIP_CLI_KEYS`。🔴 **跨引擎污染是真的**:vLLM 的 LoRA 旗標會讓 llama-server + 直接 exit——你的引擎也要把不認得的鍵 drop 掉,別原封不動透傳。 +- **參數改名**:像 llama.cpp 的 `_LLAMACPP_PARAM_MAP = {"max_model_len": "ctx-size"}`。 +- **bool 語意**:注意 store_true 旗標有沒有 `--no-xxx` 反向式(llama.cpp 沒有,別亂合成)。 + +`inapplicable_keys`(#1 的 metadata)通常 = 你的 `_MYENGINE_DROP_KEYS`,兩者對齊,前端才會把該遮的欄位遮掉。 + +--- + +## 3. 註冊(#2)—— 漏了就「引擎不存在」 + +```python +# apps/backend/app/main.py (約 :129) +launchers = [VllmLauncher(), SglangLauncher(), LlamacppLauncher(), MyEngineLauncher(), EmbeddingLauncher()] +``` + +**所有 engine image 都註冊全部 launcher**(build_spec 只是組指令、不需要該引擎裝在本機);哪個 node 真的跑得動, +由 `LLMOPS_NODE_ENGINES` gate。所以這行漏了 = 該引擎在**每個** backend 都不存在。 + +--- + +## 4. schema Literal(#3)—— 別造出 phantom engine + +```python +# packages/config-schema/schema.py (約 :62) +engine: Literal["vllm", "sglang", "llamacpp", "myengine"] = "vllm" +``` + +🔴 **這個 Literal 必須 == 已註冊的 launcher 引擎名集合。** 只加 Literal、忘了 launcher(或反過來)= +**phantom engine**:schema 驗過、但沒人認領 → 群組在 backend 靜默消失、router 照樣列在 `/v1/models` 並黑洞路由。 +(`trtllm` 曾經就是這種鬼——P1 已清掉。) + +保險:backend 啟動時 `unregistered_engine_groups()` 會對「有 engine、無 launcher」的群組記 loud error; +router 啟動時 `unknown_metric_engines()` 會對「有 engine、無指標表」warning。加引擎時**看一下這兩條 log**確認沒漏。 + +--- + +## 5. 三個最容易踩的坑 + +1. **`served_name` 沒填** → boot 重啟後,設了 `served_model_name` 的模型 adopt 失敗、同 port 起第二份 → + bind 失敗 → FAILED → auto-restart 迴圈。`build_spec` 一定要填 `served_name`。 +2. **capability 亂宣告**:`CAP_SLEEP` / `CAP_RUNTIME_LORA` / `CAP_KV_TRANSFER` 只有引擎**真的**支援才加。 + router 的 `ENGINE_SLEEP_CAPABLE`(#7)也要對齊——backend 說會 sleep、router 不去 probe `/is_sleeping` + 是典型的半套 drift。 +3. **指標名沒有獨特前綴 / router 表沒加**:`METRIC_NAMES_BY_ENGINE` 缺你的引擎 → router 用 vLLM 名表解析 → + 全 miss → (經 H#2)回 unreachable + warning。routing 會避開它、autoscaler 拿不到訊號。#6 一定要補。 + +--- + +## 6. Router(#6、#7) + +```python +# apps/router-server/src/llm_router/vllm_metrics_client.py +METRIC_NAMES_BY_ENGINE = { + ..., + "myengine": { + "running": "myengine:requests_running", + "waiting": "myengine:requests_waiting", + "kv_cache_usage_perc": "myengine:kv_usage", # 沒有就指一個不存在的名(→ 0.0),別亂填 + "prompt_tokens": "myengine:prompt_tokens_total", + "generation_tokens": "myengine:generation_tokens_total", + }, +} +ENGINE_SLEEP_CAPABLE = frozenset({"vllm", ...}) # 只有真的有 sleep/wake 才加 myengine +``` + +`metrics_poller` 會自動用 `model_config.engine` 選對的指標表,無需再改。backend 若有定義新 `CAP_METRICS_MYENGINE` +常數,記得在 `launchers.py` 頂部宣告。 + +--- + +## 7. Deploy(#11、#12、#13) + +- **Dockerfile**:`deploy/engine-myengine.Dockerfile`,`FROM <該引擎官方 base image>`,把 backend/router + 程式碼 COPY 進去(照 `engine-llamacpp.Dockerfile` 抄——注意 base 可能自帶錯的 HEALTHCHECK/ENTRYPOINT 要清、 + `LD_LIBRARY_PATH` 之類的坑)。 +- **compose**(`docker-compose.mixed.yaml`):加一個 `myengine-backend` service: + ```yaml + myengine-backend: + profiles: ["myengine"] + build: { context: .., dockerfile: deploy/engine-myengine.Dockerfile } + environment: + - LLMOPS_NODE_ENGINES=myengine + - LLMOPS_PROMETHEUS_SD_PATH=/sd/targets-myengine.json # Prometheus glob 自動抓,不用改 scrape config + - LLMOPS_NODE_HOST=mixed-myengine-backend + - ...(照其他 backend 抄) + ``` +- **Makefile**:`_ALL_ENGINES := vllm sglang llamacpp myengine`(讓 `make up-mixed ENGINES=...` 認得它、 + 切換時能正確停掉)。 + +Prometheus / file_sd / compose profile / `ENGINES=` 都已參數化,以上就是全部。 + +--- + +## 8. 測試(照現有樣板) + +- `tests/unit/test_launchers.py`:新 launcher 的 `keys()` / `build_spec()`(含 `served_name`)、CLI builder + 的參數翻譯 + skip 清單。 +- `tests/unit/test_vllm_command.py`:若有寫 paste 解析器。 +- `tests/api/test_engines_routes.py`:`GET /api/engines` 會自動帶出新引擎(capability/metadata 對不對)。 +- router `tests/unit/test_vllm_metrics_client.py`:新指標表能正規化成 `VLLMInstanceMetrics` 形狀。 +- 加完跑 `make test`(backend + router + schema)。 + +--- + +## 9. 驗收清單(TL;DR) + +- [ ] `MyEngineLauncher`:`engine`、`capabilities`、`lora_endpoint_prefix`、`metric_prefix`、 + `inapplicable_keys`、`paste_example`、`keys()`、`build_spec()`(含 **`served_name`**)。 +- [ ] CLI builder 純函式 + skip/drop 清單(扣掉 `_ROUTER_ONLY_KEYS` + 自己不認得的鍵)。 +- [ ] `main.py` 註冊 launcher。 +- [ ] schema `Literal` 加引擎名(== launcher 集合)。 +- [ ] router `METRIC_NAMES_BY_ENGINE` +(如支援)`ENGINE_SLEEP_CAPABLE`。 +- [ ] `engine-myengine.Dockerfile` + compose service(profile / NODE_ENGINES / SD path)+ Makefile 清單。 +- [ ] 測試齊全,`make test` 綠。 +- [ ] 啟動看 backend `unregistered_engine_groups` / router `unknown_metric_engines` 兩條 log **沒有**警告。 +- [ ] 前端 Add Model 選單自動出現新引擎、capability 欄位正確(不用改前端;若沒出現→ `/api/engines` 沒帶到,回頭查 metadata)。 + +--- + +## 附:非典型引擎的 roadmap(§0 踩線時) + +若第 N 個引擎不符合 §0 契約,先做這些 protocol 擴充(現有引擎不實作、行為零變): + +- `Launcher.available()`:`shutil.which(...)` / import probe 檢查 binary 在不在本 image,collapsed 模式 + create/start 時直接明確拒絕(取代「crash 三次才知道 image 沒裝」)。 +- `async Launcher.prepare(spec)`:啟動前置(TRT engine build / artifact 下載),manager 在 spawn 前呼叫, + 期間 state 停 STARTING。 +- `Launcher.readiness_probe(spec)`:覆寫預設 `/health`(自訂路徑/判準)。 +- `model_config.engine_args` 巢狀命名空間:原生旗標放這裡,engine-neutral 鍵留頂層,skip/drop 清單不再隨 + 引擎數線性成長。 + +詳見 [multi-engine-extensibility-review_zh-TW.md](multi-engine-extensibility-review_zh-TW.md) §5 P2。 diff --git a/docs/trtllm-conversion-validation_zh-TW.md b/docs/trtllm-conversion-validation_zh-TW.md new file mode 100644 index 0000000..b652d80 --- /dev/null +++ b/docs/trtllm-conversion-validation_zh-TW.md @@ -0,0 +1,199 @@ +# TensorRT-LLM 轉換 + 端點驗證報告 + +> 驗證日期:2026-07-04。目的:在寫 `TrtllmLauncher` **之前**,先實測(a)能不能把本地 LLM 轉成 TRT engine、 +> (b)轉出來能不能 serve 且端點可用。**結論:兩者都成功。** 對標 +> [trtllm-engine-conversion-guide.md](trtllm-engine-conversion-guide.md) 與 +> [trtllm-launcher-research.md](trtllm-launcher-research.md),並修正其中幾個與實測不符的點(尤其 metrics 端點)。 + +## TL;DR + +| 項目 | 結果 | +|---|---| +| 拉 image `nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20`(63.4GB) | ✅ 免 NGC 登入即可 pull | +| HF → TRT-LLM checkpoint(`convert_checkpoint.py`) | ✅ **wall-clock ~15s**(工具印的「Total time…2s」只是內層轉換迴圈,不含 import/load),產出 `config.json` + `rank0.safetensors`(~1.3–1.5GB 真權重) | +| checkpoint → TRT engine(`trtllm-build`) | ✅ **wall-clock ~24s**(內層 build ~10s),產出 `rank0.engine`(~1.2–1.5GB),build 期 peak GPU ~1.2GB | +| serve engine(`--backend tensorrt`) | ✅ `/health` 200(~35s),推理正常,VRAM ~5.3GB | +| serve HF 目錄(`--backend pytorch`,免 build) | ✅ `/health` 200(~40s),推理正常 | +| `/v1/models`、`/v1/chat/completions`、`/v1/completions` | ✅ 全部正常生成 | +| Prometheus 指標(`/prometheus/metrics`) | ✅ 需 `return_perf_metrics: true`;五個 `trtllm_*` 指標實測全有(見「深入調查」§A) | +| Sleep(`/release_memory`) | ❌ engine backend 回 500(僅 pytorch backend 支援;不宣告 CAP_SLEEP) | + +**在「官方不支援」的 RTX 3060 Ti(sm_86 consumer Ampere, 8GB)上,0.5B 模型的轉換 + serve 全部跑通。** + +--- + +## 環境 + +- GPU:**RTX 3060 Ti,compute cap 8.6(sm_86 consumer Ampere),8GB**,driver 591.86(WSL2)。 + - 不在 TRT-LLM 官方支援清單(A100/H100/L40…),但 sm_86 kernel 照樣 build + run。 +- 模型:`Qwen/Qwen2.5-0.5B-Instruct`(本地 HF safetensors,完整)。 +- image:`nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20`,pull 下來 **63.4GB(未壓縮)**,`tensorrt_llm 1.3.0rc20`。 + +### ⚠️ 坑 1:container 內要用 login shell + +image 的 `LD_LIBRARY_PATH`(含 `/usr/local/tensorrt/lib`)是靠**登入 shell 的 profile** 設的,不在 Docker +`Config.Env`。所以 `docker exec ... python3 -c "import tensorrt_llm"` 會炸 +`ImportError: libnvonnxparser.so.10`。**一律用 `docker exec bash -lc '...'`**(login shell)才會對。 +→ launcher spawn 進程時,command 或 entrypoint 要確保這條 LD_LIBRARY_PATH 生效。 + +--- + +## 步驟與實測輸出 + +啟動 container(照 research doc): +```bash +docker run -d --name trtllm-test --gpus all --ipc=host \ + --ulimit memlock=-1 --ulimit stack=67108864 -p 8000:8000 \ + -v /home/max/trtllm-test:/work \ + -v /home/max/.cache/huggingface:/root/.cache/huggingface \ + nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20 sleep infinity +``` + +### 1. 轉換(Route B:真的產 engine artifact)—— 成功 + +Qwen 的 `convert_checkpoint.py` **image 內已內建**(`/app/tensorrt_llm/examples/models/core/qwen/`,與 wheel +同版,不用 clone)。 + +```bash +# HF -> TRT-LLM checkpoint (wall-clock ~15s) +python3 convert_checkpoint.py --model_dir \ + --output_dir /work/models/qwen25-05b_ckpt_tp1_fp16 --dtype float16 --tp_size 1 --pp_size 1 +# -> config.json + rank0.safetensors (real fp16 weights, ~1.3-1.5GB) + +# checkpoint -> TRT engine (wall-clock ~24s, peak GPU ~1.2GB) +trtllm-build --checkpoint_dir /work/models/qwen25-05b_ckpt_tp1_fp16 \ + --output_dir /work/engines/qwen25-05b_tp1_fp16_isl1024_osl1024_bs4 \ + --max_batch_size 4 --max_input_len 1024 --max_seq_len 2048 --max_num_tokens 2048 \ + --gemm_plugin auto --gpt_attention_plugin auto --kv_cache_type paged +# -> config.json + rank0.engine (serialized TRT engine, ~1.2-1.5GB) +``` + +> **時間釐清(你問「2 秒會不會太快」):** 那個「2s」是工具印的**內層轉換迴圈**時間,不是 wall-clock。 +> 用 `time` 實測 **Qwen3-0.6B**:`convert_checkpoint.py` **real 15.1s**、`trtllm-build` **real 23.6s** +> (內層各印 2s / 10s,差額是 python/torch/tensorrt import + 載入 + 序列化)。**端到端約 40s**。 +> 而且是真的:產出 `rank0.engine` 1.53GB,serve 起來 `/v1/chat/completions` 回「red, blue, green」(正確)。 +> ⚠️ 這是 0.5–0.6B 的數字;**大模型(8B+)的 build 會到分鐘級**,所以 `prepare()` 的 build 絕不能算進普通 +> startup timeout。 + +> ⚠️ **坑 2:`convert_checkpoint.py` 印 LEGACY WARNING** —「This is part of the legacy TensorRT engine-build +> workflow. New projects should use the PyTorch backend.」而且 +> [research doc](trtllm-launcher-research.md) 已知 **`v1.3.0rc20` 是最後一個支援 TensorRT backend 的版本,下一版移除**。 +> → engine 轉換路線能用,但屬 legacy;若要做 engine artifact 快取,要接受它會被淘汰。 + +### 2. serve engine(`--backend tensorrt`)—— 成功 + +```bash +trtllm-serve serve /work/engines/qwen25-05b_tp1_fp16_isl1024_osl1024_bs4 \ + --backend tensorrt --host 0.0.0.0 --port 8000 \ + --tokenizer --served_model_name qwen25-05b-trt \ + --max_batch_size 4 --kv_cache_free_gpu_memory_fraction 0.4 +``` +`/health` 200(~35s);engine 目錄不含 tokenizer,**必須 `--tokenizer` 指回 HF 目錄**(坑 3)。 + +### 3. serve HF 目錄(`--backend pytorch`)—— 成功(免 build,launcher 首選路線) + +```bash +trtllm-serve serve --backend pytorch --host 0.0.0.0 --port 8000 \ + --served_model_name qwen25-05b-pt --max_seq_len 2048 --max_batch_size 4 \ + --kv_cache_free_gpu_memory_fraction 0.4 +``` +`/health` 200(~40s),直接吃 HF 目錄、**不需要 convert/build**。這條就是 launcher 第一版該走的(spawn-and-probe, +無 artifact 相容性問題)。 + +### 4. 端點驗證(兩種 backend 都測) + +| 端點 | 結果 | +|---|---| +| `GET /health` | 200 | +| `GET /v1/models` | `{"id":"qwen25-05b-trt","owned_by":"tensorrt_llm"}` —**廣告 served_name**(adopt 身分驗證可行 ✅) | +| `POST /v1/chat/completions` | ✅ 實際生成:「Hello! How can I assist you today?」 | +| `POST /v1/completions` | ✅ 生成 | + +--- + +## 深入調查後的定論(metrics / sleep) + +> 前一版本報告在這兩點有誤(沒開對旗標)。以下為讀 `openai_server.py` 原始碼 + 在 **TensorRT engine +> backend** 上實測的**定論**。 + +### A. Prometheus 指標:**可用**,但要開 `return_perf_metrics`(research doc 的名稱是對的) + +`/prometheus/metrics` 是**條件式註冊**的——`openai_server.py` 只在 `generator.args.return_perf_metrics == true` +時才 `mount_metrics()`(建 `MetricsCollector` + `make_asgi_app`)。沒開就 404(我第一版就是沒開才誤判)。 + +**正確做法(engine backend 實測通過):** +- 啟動時經 `--extra_llm_api_options ` 傳 `return_perf_metrics: true`。 +- scrape **`/prometheus/metrics`**(不是 `/metrics`)。`/metrics` 是另一個 **JSON** 端點(request 級 perf, + 非 Prometheus)。 +- 🔴 **`enable_iter_perf_stats` 不能用在 TensorRT backend**——實測 `ValueError: _TrtLLM got invalid argument: + enable_iter_perf_stats`(那是 pytorch/autodeploy backend 專屬)。TRT engine 只要 `return_perf_metrics: true` + 就夠,kv/running/waiting 都有。 + +**實測 `/prometheus/metrics` 上 research doc 的五個名稱全部存在(engine backend, prefix `trtllm_`):** + +| 概念 | 指標名 | 實測值 | +|---|---|---| +| running | `trtllm_num_requests_running` | ✅ present | +| waiting | `trtllm_num_requests_waiting` | ✅ present | +| kv usage | `trtllm_kv_cache_utilization` | ✅ present | +| prompt tokens | `trtllm_prompt_tokens_total` | ✅ 8.0 | +| generation tokens | `trtllm_generation_tokens_total` | ✅ 64.0 | + +還附帶大量其他(`trtllm_kv_cache_{free,used,max}_blocks`、`trtllm_gpu_memory_usage_bytes`、 +`trtllm_e2e_request_latency_seconds`、TTFT、`trtllm_num_{context,generation,paused}_requests`…)。 +label 有 `model_name` 與 `engine_type`(實測顯示 `engine_type="unknown"`,小瑕疵)。 + +→ **router 的 `METRIC_NAMES_BY_ENGINE` 機制可以套**:`METRIC_NAMES_BY_ENGINE["trtllm"]` 用上表五個名(text 格式、 +底線非冒號),scrape path 用 `/prometheus/metrics`。**launcher 的 build_spec 必須自動寫一個含 `return_perf_metrics: +true` 的暫存 YAML 並帶 `--extra_llm_api_options`,否則沒有指標。** + +### B. 實際端點清單(openapi.json,authoritative) + +``` +GET /health, /health_generate, /version, /server_info, /energy_metrics +GET /metrics, /perf_metrics # JSON(request 級),非 Prometheus +GET /prometheus/metrics # Prometheus text,僅 return_perf_metrics=true 時掛載 +POST /v1/chat/completions, /v1/completions GET /v1/models +POST /v1/responses GET/DELETE /v1/responses/{id} +POST /kv_cache_events, /release_memory, /resume_memory, /update_weights, /steady_clock_offset +``` + +### C. Sleep:`/release_memory`+`/resume_memory` **存在但對 engine backend 不可用** + +原始碼:`release_memory` = `collective_rpc('sleep')`、`resume_memory` = `collective_rpc('wakeup')`——確實是 sleep/wake +機制。**但三個端點(含 `/update_weights`)都 `assert isinstance(self.generator, AsyncLLM)`**,而 AsyncLLM 是 +**PyTorch backend** 的 generator。 + +**實測(TensorRT engine backend):`POST /release_memory` → 500、`POST /resume_memory` → 500。** + +→ **engine 路線不宣告 `CAP_SLEEP`**(與 research doc 結論一致,但原因是「端點存在、僅 pytorch backend 支援」)。 +若哪天改走 pytorch backend 才有可能宣告 sleep-like。 + +--- + +## 對 Launcher 實作的結論(engine-first) + +> 本 launcher 目標是 **TRT engine 模型**,所以走 `--backend tensorrt`(不走 pytorch)。 + +1. **`prepare()` 是必要的,不是選配**。engine 路線需要:先 `convert_checkpoint.py` + `trtllm-build`(或使用者提供 + pre-built engine 目錄)→ 才 spawn `trtllm-serve --backend tensorrt`。build 期不可算進普通 startup timeout。 + engine artifact 要做相容性快取(見 [conversion guide](trtllm-engine-conversion-guide.md) §「輸出目錄建議」 + 的 cache key:GPU compute cap / TRT-LLM ver / TRT ver / container tag / TP/PP / max seq·batch·token / quant)。 + ⚠️ **注意**:`v1.3.0rc20` 是最後一個支援 TensorRT backend 的版本,下一版移除——升級策略要先想。 +2. **啟動契約符合**:單行程、CLI 旗標、OpenAI `/v1`、`/health` 200 ready、`/v1/models` 廣告 served_name + → 可照 vLLM/SGLang 樣板做 `build_spec`,但 command 是 `trtllm-serve serve --backend tensorrt + --tokenizer --served_model_name <> --host 0.0.0.0 --port <> --extra_llm_api_options `。 +3. **metrics 可用**(§A):`METRIC_NAMES_BY_ENGINE["trtllm"]` = 上表五名、scrape `/prometheus/metrics`;launcher 要 + 自動產含 `return_perf_metrics: true` 的 YAML。**router 端要能對 trtllm 用 `/prometheus/metrics` 這個非標準路徑 + scrape**(現有 metrics_poller 是打 `/metrics`,需支援 per-engine 的 metrics path)。 +4. **capability**:`{CAP_LORA_MODULES?(需 YAML), CAP_METRICS_TRTLLM}`。**不宣告** `CAP_SLEEP`(§C)、 + `CAP_RUNTIME_LORA`、`CAP_KV_TRANSFER`。 +5. **坑**:(1) LD_LIBRARY_PATH 要 login-shell 生效(Dockerfile ENTRYPOINT / command 要處理); + (2) engine 模式要帶 `--tokenizer`;(3) `--host 0.0.0.0`;(4) 沒帶 perf YAML 就完全沒有指標。 +6. **硬體**:sm_86 8GB 能跑 0.5B 的轉換+serve;真要跑大模型仍需官方支援的卡(A100/H100/L40)。 + +## 產物與清理 + +- engine/checkpoint 產物保留在 host:`/home/max/trtllm-test/{models,engines}/`。 +- 測試 container `trtllm-test` 仍在(idle,serve 已停,VRAM 已釋放回 baseline ~1.2GB)。 +- 清理:`docker rm -f trtllm-test`;image 很大(63GB),要回收再 `docker rmi nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20`。 diff --git a/docs/trtllm-engine-conversion-guide.md b/docs/trtllm-engine-conversion-guide.md new file mode 100644 index 0000000..072f094 --- /dev/null +++ b/docs/trtllm-engine-conversion-guide.md @@ -0,0 +1,526 @@ +# TensorRT-LLM Engine 轉換指南 + +查詢日期: 2026-07-04 +基準版本: TensorRT-LLM `1.3.0rc20` / NGC `nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20` + +## 先講結論 + +如果只是要用 `trtllm-serve` 跑模型, 最新 TensorRT-LLM 預設可以直接吃 Hugging Face model name 或本地 HF checkpoint, 不一定要先轉 TensorRT engine。 + +如果你要的是「產出可快取、可交給 TensorRT backend/Triton/legacy runtime 使用的 TensorRT-LLM engine 目錄」, 才需要轉換。流程通常是: + +```text +HF checkpoint + -> TensorRT-LLM checkpoint + -> TensorRT engine directory + -> trtllm-serve --backend tensorrt / run.py / Triton backend +``` + +官方最新文件也提醒: `convert_checkpoint.py` / `trtllm-build` / `run.py` 這套是 legacy workflow, 新專案優先用 `trtllm-serve` 或 LLM Python API。不過若本專案要做 engine artifact 快取, 仍可參考這套流程。 + +## 來源 + +- `trtllm-serve` CLI: https://nvidia.github.io/TensorRT-LLM/commands/trtllm-serve/trtllm-serve.html +- `trtllm-build` CLI: https://nvidia.github.io/TensorRT-LLM/latest/commands/trtllm-build.html +- TensorRT-LLM checkpoint workflow: https://nvidia.github.io/TensorRT-LLM/latest/legacy/architecture/checkpoint.html +- TensorRT-LLM build workflow: https://nvidia.github.io/TensorRT-LLM/architecture/workflow.html +- LLaMA example workflow: https://github.com/NVIDIA/TensorRT-LLM/blob/main/examples/models/core/llama/README.md +- Container images: https://nvidia.github.io/TensorRT-LLM/installation/containers.html +- Supported hardware: https://nvidia.github.io/TensorRT-LLM/supported-hardware.html +- TensorRT engine compatibility: https://docs.nvidia.com/deeplearning/tensorrt/latest/inference-library/engine-compatibility.html + +## 重要限制 + +### 1. Engine 不是通用格式 + +TensorRT engine 會和 build 環境強相關。cache key 至少要包含: + +- GPU compute capability / GPU 型號族群 +- TensorRT-LLM 版本 +- TensorRT 版本 +- CUDA/container tag +- `tp_size`, `pp_size`, `cp_size` +- `max_batch_size`, `max_input_len`, `max_seq_len`, `max_num_tokens` +- dtype / quantization / KV cache dtype + +TensorRT 文件說明, engine 預設會檢查 TensorRT runtime version 與 GPU compute capability。不匹配時可能無法 deserialize。 + +### 2. Build 要在 NVIDIA GPU 環境做 + +`trtllm-build` 通常需要 NVIDIA GPU。官方支援硬體包含: + +- Blackwell: B200, GB200, B300, GB300, DGX Spark +- Hopper: H100, H200, GH200 +- Ada: L20, L40/L40S +- Ampere: A100 + +### 3. 版本要對齊 + +不要拿不同版本 repo 的 `examples/.../convert_checkpoint.py` 搭配另一個版本的 `tensorrt_llm` wheel。官方 build workflow 特別提醒 examples scripts 可能使用 internal/unstable API, 版本不一致容易壞。 + +## 推薦環境: NGC release container + +官方 `release` image 已安裝 pre-built wheel, 適合直接跑 `trtllm-serve`, `trtllm-build`, examples。 + +```bash +docker pull nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20 +``` + +啟動 container: + +```bash +docker run --rm -it \ + --gpus all \ + --ipc=host \ + --ulimit memlock=-1 \ + --ulimit stack=67108864 \ + -p 8000:8000 \ + -v $PWD/models:/models \ + -v $HOME/.cache/huggingface:/root/.cache/huggingface \ + nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20 \ + /bin/bash +``` + +進 container 後檢查: + +```bash +nvidia-smi +python3 - <<'PY' +import tensorrt_llm +print("tensorrt_llm", tensorrt_llm.__version__) +PY +trtllm-build --help | head +trtllm-serve --help | head +``` + +若 container 內沒有你要的 example source, 用同版本 tag clone: + +```bash +cd /workspace +git clone --branch v1.3.0rc20 --depth 1 https://github.com/NVIDIA/TensorRT-LLM.git +cd TensorRT-LLM +``` + +## Hugging Face 權重準備 + +登入 Hugging Face: + +```bash +huggingface-cli login +``` + +下載模型到本地。範例用 LLaMA 類模型, 但實際 repo 請換成你有權限的模型: + +```bash +export HF_MODEL_ID="meta-llama/Llama-3.1-8B-Instruct" +export MODEL_DIR="/models/hf/llama-3.1-8b-instruct" + +huggingface-cli download "$HF_MODEL_ID" \ + --local-dir "$MODEL_DIR" \ + --local-dir-use-symlinks False +``` + +確認至少有: + +```text +config.json +tokenizer.json / tokenizer.model / tokenizer_config.json +model-*.safetensors 或 pytorch_model*.bin +``` + +## 路線 A: 不轉 engine, 直接用 trtllm-serve + +這是新專案優先路線: + +```bash +trtllm-serve serve "$MODEL_DIR" \ + --host 0.0.0.0 \ + --port 8000 \ + --backend pytorch \ + --max_seq_len 4096 \ + --max_batch_size 8 \ + --max_num_tokens 8192 \ + --tp_size 1 \ + --trust_remote_code +``` + +測試: + +```bash +curl http://localhost:8000/health + +curl http://localhost:8000/v1/models + +curl http://localhost:8000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -d '{ + "model": "'$MODEL_DIR'", + "messages": [{"role": "user", "content": "Say hello in one sentence."}], + "max_tokens": 32 + }' +``` + +## 路線 B: 轉 TensorRT-LLM checkpoint + build TensorRT engine + +這是「真的轉 engine artifact」的流程。下面以 LLaMA/Mistral compatible 模型為例。 + +### 0. 設定變數 + +```bash +export TRTLLM_REPO="/workspace/TensorRT-LLM" +export MODEL_DIR="/models/hf/llama-3.1-8b-instruct" +export WORK_DIR="/models/trtllm/llama-3.1-8b-instruct" + +export DTYPE="float16" # A100/H100 常可用 float16 或 bfloat16 +export TP_SIZE=1 +export PP_SIZE=1 +export MAX_INPUT_LEN=2048 +export MAX_OUTPUT_LEN=512 +export MAX_SEQ_LEN=$((MAX_INPUT_LEN + MAX_OUTPUT_LEN)) +export MAX_BATCH_SIZE=8 +export MAX_NUM_TOKENS=8192 +export WORKERS=1 + +export CKPT_DIR="$WORK_DIR/tllm_checkpoint_tp${TP_SIZE}_pp${PP_SIZE}_${DTYPE}" +export ENGINE_DIR="$WORK_DIR/engine_tp${TP_SIZE}_pp${PP_SIZE}_${DTYPE}_isl${MAX_INPUT_LEN}_osl${MAX_OUTPUT_LEN}_bs${MAX_BATCH_SIZE}" +``` + +### 1. 安裝 example requirements + +```bash +cd "$TRTLLM_REPO/examples/models/core/llama" +pip install --upgrade -r requirements.txt +``` + +### 2. HF checkpoint -> TensorRT-LLM checkpoint + +```bash +cd "$TRTLLM_REPO/examples/models/core/llama" + +python3 convert_checkpoint.py \ + --model_dir "$MODEL_DIR" \ + --output_dir "$CKPT_DIR" \ + --dtype "$DTYPE" \ + --tp_size "$TP_SIZE" \ + --pp_size "$PP_SIZE" +``` + +轉完後檢查: + +```bash +find "$CKPT_DIR" -maxdepth 1 -type f | sort +``` + +你應該會看到 config 與 rank weight 檔。官方 checkpoint convention 是一個 config JSON 加上一個或多個 rank weights 檔。 + +### 3. TensorRT-LLM checkpoint -> TensorRT engine + +```bash +trtllm-build \ + --checkpoint_dir "$CKPT_DIR" \ + --output_dir "$ENGINE_DIR" \ + --max_batch_size "$MAX_BATCH_SIZE" \ + --max_input_len "$MAX_INPUT_LEN" \ + --max_seq_len "$MAX_SEQ_LEN" \ + --max_num_tokens "$MAX_NUM_TOKENS" \ + --gemm_plugin auto \ + --gpt_attention_plugin auto \ + --kv_cache_type paged \ + --workers "$WORKERS" +``` + +常用 build 參數: + +| 參數 | 用途 | +|---|---| +| `--checkpoint_dir` | TensorRT-LLM checkpoint 目錄 | +| `--output_dir` | engine 輸出目錄 | +| `--max_batch_size` | engine 可 schedule 的最大 request 數 | +| `--max_input_len` | 單一 request 最大 prompt/input 長度 | +| `--max_seq_len` | 單一 request 最大總長度, prompt + output | +| `--max_num_tokens` | 單 batch 移除 padding 後的最大 token 數 | +| `--gemm_plugin auto` | 自動依 dtype 啟用 GEMM plugin | +| `--gpt_attention_plugin auto` | 自動依 dtype 啟用 GPT attention plugin | +| `--kv_cache_type paged` | 使用 paged KV cache | +| `--workers` | 平行 build workers, 單節點有效 | +| `--fast_build` | 加快 build, 但可能影響效能/相容性 | +| `--dry_run` | 跑 build 流程但不實際產 engine, 可 debug | +| `--monitor_memory` | build 期間監控記憶體 | + +轉完後檢查: + +```bash +find "$ENGINE_DIR" -maxdepth 2 -type f | sort +du -sh "$ENGINE_DIR" +``` + +### 4. 用 examples/run.py 驗證 engine + +單 GPU: + +```bash +cd "$TRTLLM_REPO/examples/models/core/llama" + +python3 ../../../run.py \ + --engine_dir "$ENGINE_DIR" \ + --tokenizer_dir "$MODEL_DIR" \ + --max_output_len "$MAX_OUTPUT_LEN" \ + --input_text "Hello, my name is" +``` + +多 GPU, 例如 `TP_SIZE=2`: + +```bash +mpirun -n "$TP_SIZE" --allow-run-as-root \ + python3 ../../../run.py \ + --engine_dir "$ENGINE_DIR" \ + --tokenizer_dir "$MODEL_DIR" \ + --max_output_len "$MAX_OUTPUT_LEN" \ + --input_text "Hello, my name is" +``` + +## 路線 C: 用已 build engine 啟動 trtllm-serve + +`trtllm-serve serve` 的 `MODEL` 可是 TensorRT engine path。因為 `--backend` 預設是 `pytorch`, 用 engine 時建議明確指定 `tensorrt`: + +```bash +trtllm-serve serve "$ENGINE_DIR" \ + --backend tensorrt \ + --host 0.0.0.0 \ + --port 8000 \ + --tokenizer "$MODEL_DIR" \ + --served_model_name "llama-3.1-8b-trt" \ + --tp_size "$TP_SIZE" \ + --pp_size "$PP_SIZE" +``` + +測試: + +```bash +curl http://localhost:8000/health +curl http://localhost:8000/v1/models + +curl http://localhost:8000/v1/completions \ + -H "Content-Type: application/json" \ + -d '{ + "model": "llama-3.1-8b-trt", + "prompt": "The capital of France is", + "max_tokens": 16 + }' +``` + +## 多 GPU build 範例 + +### TP=2 + +```bash +export TP_SIZE=2 +export PP_SIZE=1 +export WORKERS=2 +export CKPT_DIR="$WORK_DIR/tllm_checkpoint_tp2_fp16" +export ENGINE_DIR="$WORK_DIR/engine_tp2_fp16" + +python3 convert_checkpoint.py \ + --model_dir "$MODEL_DIR" \ + --output_dir "$CKPT_DIR" \ + --dtype float16 \ + --tp_size "$TP_SIZE" + +trtllm-build \ + --checkpoint_dir "$CKPT_DIR" \ + --output_dir "$ENGINE_DIR" \ + --gemm_plugin auto \ + --gpt_attention_plugin auto \ + --max_input_len 2048 \ + --max_seq_len 2560 \ + --max_batch_size 8 \ + --max_num_tokens 8192 \ + --workers "$WORKERS" +``` + +### TP=2, PP=2 + +```bash +export TP_SIZE=2 +export PP_SIZE=2 +export WORKERS=4 +export CKPT_DIR="$WORK_DIR/tllm_checkpoint_tp2_pp2_fp16" +export ENGINE_DIR="$WORK_DIR/engine_tp2_pp2_fp16" + +python3 convert_checkpoint.py \ + --model_dir "$MODEL_DIR" \ + --output_dir "$CKPT_DIR" \ + --dtype float16 \ + --tp_size "$TP_SIZE" \ + --pp_size "$PP_SIZE" + +trtllm-build \ + --checkpoint_dir "$CKPT_DIR" \ + --output_dir "$ENGINE_DIR" \ + --gemm_plugin auto \ + --gpt_attention_plugin auto \ + --max_input_len 2048 \ + --max_seq_len 2560 \ + --max_batch_size 8 \ + --max_num_tokens 8192 \ + --workers "$WORKERS" +``` + +注意: `--workers` 目前只支援 single node parallel build。多節點 inference 可以 build 在單節點完成, 但 run/serve 時要用對應 MPI/Slurm/launcher。 + +## 量化範例 + +### INT8 weight-only + +```bash +export CKPT_DIR="$WORK_DIR/tllm_checkpoint_int8wo" +export ENGINE_DIR="$WORK_DIR/engine_int8wo" + +python3 convert_checkpoint.py \ + --model_dir "$MODEL_DIR" \ + --output_dir "$CKPT_DIR" \ + --dtype float16 \ + --use_weight_only \ + --weight_only_precision int8 \ + --tp_size "$TP_SIZE" + +trtllm-build \ + --checkpoint_dir "$CKPT_DIR" \ + --output_dir "$ENGINE_DIR" \ + --gemm_plugin auto \ + --gpt_attention_plugin auto +``` + +### NVFP4 / FP8 KV cache + +官方 LLaMA example 對 NVFP4 建議先用 `examples/quantization/quantize.py` 產 TensorRT-LLM checkpoint, 再 build engine。 + +```bash +export CKPT_DIR="$WORK_DIR/tllm_checkpoint_nvfp4" +export ENGINE_DIR="$WORK_DIR/engine_nvfp4" + +cd "$TRTLLM_REPO" + +python3 examples/quantization/quantize.py \ + --model_dir "$MODEL_DIR" \ + --output_dir "$CKPT_DIR" \ + --dtype float16 \ + --qformat nvfp4 \ + --kv_cache_dtype fp8 \ + --tp_size "$TP_SIZE" + +trtllm-build \ + --checkpoint_dir "$CKPT_DIR" \ + --output_dir "$ENGINE_DIR" \ + --use_paged_context_fmha enable \ + --use_fp8_context_fmha enable +``` + +硬體注意: + +- FP8 主要看 Ada/Hopper/Blackwell 能力。 +- NVFP4/FP4 主要看 Blackwell 世代能力。 +- 不支援的 GPU 上 build 或 runtime 可能失敗。 + +## 輸出目錄建議 + +建議不要只用 model name 當目錄, 要把 build 條件寫進路徑: + +```text +/models/trtllm/ + llama-3.1-8b-instruct/ + hf/ + checkpoints/ + tp1_pp1_fp16/ + engines/ + trtllm-1.3.0rc20_cuda13_sm90_tp1_pp1_fp16_isl2048_osl512_bs8_nt8192/ +``` + +建議另外存一份 manifest: + +```json +{ + "model_id": "meta-llama/Llama-3.1-8B-Instruct", + "trtllm_version": "1.3.0rc20", + "container": "nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20", + "dtype": "float16", + "tp_size": 1, + "pp_size": 1, + "max_input_len": 2048, + "max_output_len": 512, + "max_seq_len": 2560, + "max_batch_size": 8, + "max_num_tokens": 8192, + "gpu_name": "NVIDIA H100", + "compute_capability": "9.0" +} +``` + +## Launcher 實作建議 + +如果本專案要自動化轉換, 建議設計成 `prepare()`: + +1. 檢查 engine manifest 是否存在且相容。 +2. 若不存在, 執行 `convert_checkpoint.py` 或 `quantize.py`。 +3. 執行 `trtllm-build`。 +4. 寫入 manifest。 +5. `build_spec.command` 才 spawn `trtllm-serve serve "$ENGINE_DIR" --backend tensorrt ...`。 + +不要把 build 時間算進普通 startup timeout。中大型模型 build 可能非常久, 而且可能吃滿 CPU/GPU/磁碟。 + +## 常見錯誤 + +| 問題 | 可能原因 | 解法 | +|---|---|---| +| `ModuleNotFoundError` 或 convert script import error | example 版本和 installed wheel 不一致 | 使用同一個 NGC release image, 或 clone 同一版 tag | +| engine deserialize 失敗 | GPU compute capability 或 TensorRT version 不匹配 | 在目標 GPU/相同 container tag 重新 build | +| CUDA OOM during build | `max_batch_size` / `max_num_tokens` / seq len 太大 | 降低 build shape, 或增加 GPU/TP | +| run 時 TP rank 數不對 | engine 是 TP=2, 但只用 1 GPU 啟動 | 用 `mpirun -n 2` 或 `trtllm-serve --tp_size 2` | +| tokenizer 找不到 | engine 目錄不一定含 tokenizer | serve/run 時指定 `--tokenizer` / `--tokenizer_dir` 指向 HF checkpoint | +| FP8/NVFP4 build 失敗 | GPU 不支援或 checkpoint/量化流程不符 | 確認硬體與 ModelOpt 量化輸出 | + +## 最短可執行版 + +```bash +docker run --rm -it \ + --gpus all --ipc=host \ + --ulimit memlock=-1 --ulimit stack=67108864 \ + -p 8000:8000 \ + -v $PWD/models:/models \ + -v $HOME/.cache/huggingface:/root/.cache/huggingface \ + nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20 /bin/bash + +cd /workspace +git clone --branch v1.3.0rc20 --depth 1 https://github.com/NVIDIA/TensorRT-LLM.git +cd /workspace/TensorRT-LLM/examples/models/core/llama +pip install --upgrade -r requirements.txt + +export MODEL_DIR=/models/hf/llama-3.1-8b-instruct +export CKPT_DIR=/models/trtllm/llama-3.1-8b-instruct/checkpoints/tp1_fp16 +export ENGINE_DIR=/models/trtllm/llama-3.1-8b-instruct/engines/tp1_fp16_isl2048_osl512_bs8 + +python3 convert_checkpoint.py \ + --model_dir "$MODEL_DIR" \ + --output_dir "$CKPT_DIR" \ + --dtype float16 \ + --tp_size 1 + +trtllm-build \ + --checkpoint_dir "$CKPT_DIR" \ + --output_dir "$ENGINE_DIR" \ + --max_batch_size 8 \ + --max_input_len 2048 \ + --max_seq_len 2560 \ + --max_num_tokens 8192 \ + --gemm_plugin auto \ + --gpt_attention_plugin auto \ + --kv_cache_type paged + +trtllm-serve serve "$ENGINE_DIR" \ + --backend tensorrt \ + --host 0.0.0.0 \ + --port 8000 \ + --tokenizer "$MODEL_DIR" \ + --served_model_name llama-trt +``` + diff --git a/docs/trtllm-launcher-impl-design_zh-TW.md b/docs/trtllm-launcher-impl-design_zh-TW.md new file mode 100644 index 0000000..bb824db --- /dev/null +++ b/docs/trtllm-launcher-impl-design_zh-TW.md @@ -0,0 +1,386 @@ +# TrtllmLauncher 實作設計(第四個引擎) + +> 前置閱讀:[adding-a-new-engine_zh-TW.md](adding-a-new-engine_zh-TW.md)(觸點總表 + 隱含契約)、 +> [trtllm-launcher-research.md](trtllm-launcher-research.md)(規格盤點)、 +> [trtllm-conversion-validation_zh-TW.md](trtllm-conversion-validation_zh-TW.md)(**已實測**轉換 + 端點 + metrics/sleep)。 +> 基準:TensorRT-LLM `v1.3.0rc20`,`nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20`。 +> **目標:serve TRT engine 模型(`--backend tensorrt`),不是 pytorch backend。** + +--- + +## 0. trtllm 與前三個引擎最大的三個差異(實測得出,必須在實作處理) + +vLLM / SGLang / llama.cpp 都符合「spawn-and-probe 單行程」契約,加它們是純增量。**trtllm 踩了三條線:** + +1. **🔴 啟動前要有 engine 目錄(build artifact)。** `trtllm-serve --backend tensorrt` 吃的是 `trtllm-build` + 產出的 engine 目錄,不是 HF repo。所以「啟動」多了一個離線 build 階段(0.6B 實測 ~40s,8B+ 到分鐘級)。 + → 需要 **`prepare()` hook**,或第一版走「使用者自備 engine 目錄」(BYO,見 §1)。 +2. **🟡 metrics 端點不同。** 指標在 **`/prometheus/metrics`**(不是 `/metrics`),且**只有啟動時傳 + `return_perf_metrics: true` 才掛載**。router 現有 `metrics_poller` 把路徑寫死成 `/metrics` + ([vllm_metrics_client.py:128](../apps/router-server/src/llm_router/vllm_metrics_client.py))→ 要加 **per-engine + metrics path**。 +3. **🟡 `LD_LIBRARY_PATH` 要 login-shell 才對。** image 靠登入 profile 設 `/usr/local/tensorrt/lib` 等;而 + `spawn_process` 是**直接跑 command、非 login shell**([process.py:73](../apps/backend/app/llmops/process.py)) + → 直接 spawn `trtllm-serve` 會 `ImportError: libnvonnxparser.so.10`。要在 image/command 層處理(見 §9)。 + +其餘契約 trtllm 都符合:單行程、CLI 旗標、OpenAI `/v1`、`/health` 200 ready、`/v1/models` 廣告 served_name。 + +--- + +## 1. 分兩期落地(關鍵決策) + +不要一開始就做 `prepare()` 那套 build/cache 基建。分兩期: + +### Phase 1 —「BYO engine」(先做,可照樣板,零 protocol 擴充) +- 使用者**自己在對的 GPU 上先 build 好 engine 目錄**(照 conversion guide),config 指過去。 +- launcher 只做:`trtllm-serve serve --backend tensorrt --tokenizer ...`。 +- **不需要動 Launcher protocol / manager**——build_spec 產 command 即可,和 vLLM/SGLang 一樣是純增量。 +- 已實測這條可跑(validation doc:serve pre-built engine → 端點正常)。 + +### Phase 2 —「auto-build」(之後做,才需要擴 protocol) +- launcher `prepare()`:engine 不存在/不相容 → 自動 `convert_checkpoint.py` + `trtllm-build` + 寫 manifest。 +- 需要:Launcher protocol 加 `prepare()`、manager 在 spawn 前呼叫、build 期不觸發 startup timeout(見 §8)。 + +> 本文件 §2–§7、§9–§11 是 **Phase 1** 的完整實作;§8 是 **Phase 2** 的設計。 + +--- + +## 2. 檔案改動清單(對照 [adding-a-new-engine §1](adding-a-new-engine_zh-TW.md)) + +| # | 檔案 | Phase 1 要做 | +|---|---|---| +| 1 | `apps/backend/app/llmops/launchers.py` | `TrtllmLauncher` + `build_trtllm_cli_args` + 產 perf YAML | +| 2 | `apps/backend/app/main.py:129` | 註冊 `TrtllmLauncher()` | +| 3 | `packages/config-schema/schema.py:62` | `engine: Literal[..., "trtllm"]`;新增選配 key `engine_dir`、`tokenizer` | +| 6 | `apps/router-server/src/llm_router/vllm_metrics_client.py` | `METRIC_NAMES_BY_ENGINE["trtllm"]` + **`METRIC_PATH_BY_ENGINE`**(per-engine path) | +| 11 | `deploy/engine-trtllm.Dockerfile` | 新 image(FROM TRT-LLM release)+ 處理 LD_LIBRARY_PATH | +| 12 | `deploy/docker-compose.mixed.yaml` | `trtllm-backend` service(profile / ipc:host / ulimits / engine volume / SD path) | +| 13 | `Makefile` `_ALL_ENGINES` | 加 `trtllm` | +| 8/9/10 | 前端 | **P1 之後免動**(`/api/engines` 自動帶出);顏色 badge 選配 | + +--- + +## 3. config 資料模型 + +### Phase 1(BYO engine)—— config.yaml 範例 +```yaml +LLM_engines: + Qwen3-0.6B-trt: + instances: + - id: trt-qwen3 + host: localhost + port: 8050 + cuda_device: 0 + model_config: + engine: trtllm + model_tag: "Qwen/Qwen3-0.6B" # HF repo:給 served_name 預設 + tokenizer 預設來源 + engine_dir: "/engines/qwen3-06b_tp1_fp16_isl1024_osl1024_bs4" # 容器內 pre-built engine 目錄(必填) + tokenizer: "Qwen/Qwen3-0.6B" # 選配;預設 = model_tag + max_model_len: 2048 # -> --max_seq_len(需 ≤ engine build 的 max_seq_len) + gpu_memory_utilization: 0.4 # -> --kv_cache_free_gpu_memory_fraction + tensor_parallel_size: 1 # -> --tp_size(需 == engine build 的 tp) +``` +> ⚠️ serve 期的 `max_model_len` / `tp_size` **不能超過 engine build 時的形狀**(engine 綁死)。launcher 可在 +> build_spec 讀 `engine_dir/config.json` 做 sanity check(選配)。 + +### Phase 2(auto-build)—— 只給 model_tag +省略 `engine_dir`;`prepare()` 依 model_tag + 形狀參數自動 build,寫進 artifact cache(見 §8)。 + +--- + +## 4. `TrtllmLauncher` class + +```python +class TrtllmLauncher: + kind = ModelKind.LLM + engine = "trtllm" + # 不宣告 sleep / runtime_lora / kv_transfer(§7) + capabilities = frozenset({CAP_METRICS_TRTLLM}) # LoRA 若做 YAML 靜態再加 CAP_LORA_MODULES + lora_endpoint_prefix = "" # 無 runtime LoRA 端點 + + # 目錄 metadata(GET /api/engines,前端自動帶) + metric_prefix = "trtllm" + inapplicable_keys = _TRTLLM_DROP_KEYS # 見 §5 + paste_example = ("trtllm-serve serve /engines/qwen3-06b_... --backend tensorrt " + "--tokenizer Qwen/Qwen3-0.6B --served_model_name qwen3-06b-trt " + "--host 0.0.0.0 --port 8050") + + def keys(self, config) -> list[str]: + # 同其他 launcher:engine == "trtllm" 的群組展開 :: + ... + + def build_spec(self, config, config_path, key) -> LaunchSpec: + # merged model_config;engine_dir 定址;產 perf YAML;組 command + ... + return LaunchSpec( + key=key, kind=self.kind, engine=self.engine, capabilities=self.capabilities, + command=command, env=env, log_path=log_path, + host=inst.host, port=inst.port, + probe_url=f"http://{inst.host}:{inst.port}/health", + model_tag=engine.settings.model_tag, + served_name=merged.get("served_model_name") or engine.settings.model_tag, # 🔴 別漏 + ) +``` + +新增 CAP 常數:`CAP_METRICS_TRTLLM = "metrics_trtllm"`(launchers.py 頂部,對照 `CAP_METRICS_LLAMACPP`)。 + +--- + +## 5. arg builder:`build_trtllm_cli_args` + 參數表 + perf YAML + +### command 樣板(build_spec 組出) +``` +trtllm-serve serve \ + --backend tensorrt \ + --tokenizer \ + --served_model_name \ + --host --port \ + --max_seq_len \ + --kv_cache_free_gpu_memory_fraction \ + --tp_size --pp_size \ + --max_batch_size <..> --max_num_tokens <..> \ + --extra_llm_api_options +``` + +### `_TRTLLM_PARAM_MAP`(engine-neutral 鍵 → trtllm 旗標) +```python +_TRTLLM_PARAM_MAP = { + "max_model_len": "max_seq_len", + "gpu_memory_utilization": "kv_cache_free_gpu_memory_fraction", + "tensor_parallel_size": "tp_size", + "pipeline_parallel_size": "pp_size", + "max_batch_size": "max_batch_size", + "max_num_tokens": "max_num_tokens", +} +``` +CLI flag 用**單底線**(trtllm 是 `--max_seq_len`,不是 `--max-seq-len`)——注意別套用 llama.cpp 那種 `_`→`-`。 + +### `_TRTLLM_DROP_KEYS` / `inapplicable_keys`(前端 grey out + CLI 不透傳) +```python +_TRTLLM_DROP_KEYS = frozenset({ + "n_gpu_layers", # llama.cpp offload,不適用 + "gguf_quant", "model_file", # GGUF 專屬 + "enforce_eager", # vLLM eager/CUDA-graph 語意,不適用 + "dtype", # engine build 期決定,serve 期無 top-level flag + "quantization", # FP8/NVFP4 是 build/checkpoint recipe,不等於 vLLM --quantization +}) +``` +外加 `_ROUTER_ONLY_KEYS`(`routing_strategy`/`kind`/`engine`)與 `engine_dir`/`tokenizer`/`served_model_name` +(build_spec 已消化,不進 generic CLI)一律 skip。 + +### perf YAML(**metrics 的關鍵**) +build_spec 要**為每個 instance 產一個暫存 YAML**並帶 `--extra_llm_api_options`: +```yaml +return_perf_metrics: true +``` +建議寫在 `LOG_DIR` 或 overlay 旁(如 `/trtllm/.perf.yaml`),路徑放進 command。 +🔴 **不要放 `enable_iter_perf_stats`**(TensorRT backend 會 `ValueError: _TrtLLM got invalid argument`)。 +沒帶這個 YAML → `/prometheus/metrics` 404、完全沒指標。 + +--- + +## 6. 監控指標(router 改動) + +### 6.1 `METRIC_NAMES_BY_ENGINE["trtllm"]`(五個名,已實測存在) +```python +"trtllm": { + "running": "trtllm_num_requests_running", + "waiting": "trtllm_num_requests_waiting", + "kv_cache_usage_perc": "trtllm_kv_cache_utilization", + "prompt_tokens": "trtllm_prompt_tokens_total", + "generation_tokens": "trtllm_generation_tokens_total", +}, +``` +名稱是**底線**(非 vLLM 的冒號),但 `parse_metrics` 的 regex 兩者都吃,無需改 parser。 + +### 6.2 per-engine metrics path(**新的 router 改動**) +`fetch()` 現在寫死 `base_url + "/metrics"`。改成 per-engine: +```python +METRIC_PATH_BY_ENGINE = {"trtllm": "/prometheus/metrics"} # 其餘預設 "/metrics" +# fetch(): +metrics_url = base_url.rstrip("/") + METRIC_PATH_BY_ENGINE.get(engine, "/metrics") +``` +`metrics_poller` 已經把 `engine` 傳進 `fetch_many`([metrics_poller.py](../apps/router-server/src/llm_router/metrics_poller.py)), +所以只改 `fetch()` 一處。`unknown_metric_engines()` 啟動自檢已涵蓋 trtllm(有表就不 warn)。 + +### 6.3 sleep gating +`ENGINE_SLEEP_CAPABLE` **不加 trtllm**(§7)。 + +--- + +## 7. capabilities 總表(對照) + +| capability | vllm | sglang | llamacpp | **trtllm** | 依據 | +|---|---|---|---|---|---| +| `CAP_SLEEP` | ✅ | ✗ | ✗ | **✗** | `/release_memory` 僅 pytorch backend,engine backend 回 500(實測) | +| `CAP_RUNTIME_LORA` | ✅ | ✅ | ✗ | **✗** | 無 vLLM 式 `/v1/load_lora_adapter`,只有 request `extra_body.lora_request` | +| `CAP_LORA_MODULES` | ✅ | ✅ | ✅ | **⚠️ 延後** | 要產 YAML `lora_config`,非 `--lora-modules`;Phase 1 先不做 | +| `CAP_KV_TRANSFER` | ✅ | ✗ | ✗ | **✗** | 是 disaggregated/connector,踩 instance=多行程假設 | +| `CAP_METRICS_*` | ✅ | ✅ | ✅ | **✅** | `/prometheus/metrics`(需 return_perf_metrics) | + +--- + +## 8. Phase 2:`prepare()` hook + engine artifact cache(設計) + +### 8.1 Launcher protocol 擴充(選配 hook,現有引擎不實作、行為零變) +```python +class Launcher(Protocol): + async def prepare(self, config, config_path, key) -> None: # 預設 no-op + ... +``` +`TrtllmLauncher.prepare()`: +1. 依 (model_tag, dtype, tp, pp, max_input/seq_len, max_batch, max_num_tokens, quant) 算 **cache key**。 +2. 查 `//manifest.json` 是否存在且相容(GPU compute cap / TRT-LLM ver / TRT ver / + container tag 全部要 match——engine 綁環境,實測不匹配無法 deserialize)。 +3. 不存在/不相容 → `convert_checkpoint.py`(依 arch 選 example 目錄)+ `trtllm-build` → 寫 manifest。 +4. 把算出的 `engine_dir` 記在 spec,build_spec 用它組 command。 + +### 8.2 manager 呼叫時機 + 不觸發 startup timeout(關鍵) +`manager.start()` 在 `spawn_process` 前 `await launcher.prepare(...)`。但 build 可能到分鐘級,不能被 +reconciler 的 progress-aware startup timeout 誤殺。兩個選項: +- **(建議)新增 `ModelState.PREPARING`**:start 先進 PREPARING(不進 STARTING),build 在 executor 跑、 + 輸出寫進 log_path(progress-aware timeout 靠 log 成長,天然不誤殺);build 完才 STARTING + spawn。 + 需在 state enum、reconciler、fleet view、前端 badge 各補一個狀態(對照 SLEEPING 的加法)。 +- (簡)build 前不進任何「受 timeout 監管」狀態,build 期間週期更新 `last_progress_at`;較 hack。 + +### 8.3 artifact 目錄與 cache key(對照 conversion guide §「輸出目錄建議」) +``` +/engines/trtllm//__sm_tp_pp__isl<>_osl<>_bs<>_nt<>/ + config.json rank*.engine manifest.json +``` +manifest 至少含:model_id、trtllm_version、container、gpu_name、compute_capability、tp/pp、shapes、dtype/quant。 +放獨立 volume(見 §10),**不可跨 GPU 架構共用**。 + +### 8.4 `available()` 自檢(collapsed 模式明確拒絕) +`TrtllmLauncher.available()` → `shutil.which("trtllm-serve") is not None`;collapsed 模式(非 trtllm image) +create/start 時直接明確拒絕,取代「crash 才發現」。 + +--- + +## 9. `deploy/engine-trtllm.Dockerfile` + +```dockerfile +FROM nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20 + +WORKDIR /app +# 🔴 LD_LIBRARY_PATH:base 靠 login profile 設 /usr/local/tensorrt/lib 等,而 launcher 是直接 spawn +# (非 login shell)。兩個處理法,擇一: +# (A) 把 login-shell 的 LD_LIBRARY_PATH 烤進 ENV(部署確定、無 per-spawn 成本,但換 image 版要更新): +# ENV LD_LIBRARY_PATH=/usr/local/tensorrt/lib:/usr/local/cuda/lib64:/usr/local/ucx/lib:\ +# /usr/local/lib/python3.12/dist-packages/torch/lib:...(以實機 `bash -lc 'echo $LD_LIBRARY_PATH'` 為準) +# (B) launcher 的 command 包一層 `bash -lc "exec trtllm-serve ..."`(版本 robust;exec 保留 PID/process-group, +# process.py 的 group kill 仍有效)。← 建議 +COPY apps/backend/requirements.txt /tmp/backend-req.txt +COPY apps/router-server/requirements.txt /tmp/router-req.txt +# 只啟動 trtllm:移除 vllm/sglang/bitsandbytes/pytest(base 沒有這些且不需要);tensorrt_llm 已在 base。 +RUN sed -i -E '/^(vllm|sglang|bitsandbytes.*|pytest.*)$/d' /tmp/router-req.txt /tmp/backend-req.txt \ + && pip install --no-cache-dir -r /tmp/backend-req.txt -r /tmp/router-req.txt +COPY apps/backend ./apps/backend +COPY apps/router-server ./apps/router-server +COPY packages ./packages +ENTRYPOINT [] +CMD ["bash"] +HEALTHCHECK NONE +``` +> ⚠️ base image 已 63GB;疊上 backend deps 後更大。build/pull 時間要有心理準備。 + +--- + +## 10. compose + Makefile + schema + +### compose service(對照 sglang-backend,加 engine artifact volume + ulimits) +```yaml +trtllm-backend: + profiles: ["trtllm"] + build: { context: .., dockerfile: deploy/engine-trtllm.Dockerfile } + image: llmops-engine-trtllm:latest + container_name: mixed-trtllm-backend + working_dir: /app/apps/backend + command: bash -lc "uvicorn main:app --host 0.0.0.0 --port 5000" # login shell:確保 LD_LIBRARY_PATH + env_file: .env + depends_on: { postgres: { condition: service_healthy } } + environment: + - LLMOPS_NODE_ENGINES=trtllm + - LLMOPS_NODE_HOST=mixed-trtllm-backend + - LLMOPS_PROMETHEUS_SD_PATH=/sd/targets-trtllm.json + - LLMOPS_VLLM_BIND_HOST=0.0.0.0 + - LLMOPS_ROUTER_URL=http://router:8887 + - LLMOPS_DB_URL=postgresql://llmops:llmops@postgres:5432/llmops + - ...(照其他 backend) + volumes: + - ../packages/config-schema/config.yaml:/app/packages/config-schema/config.yaml + - mixed-trtllm-data:/app/data + - mixed-sd:/sd + - mixed-trtllm-engines:/engines # Phase 2 的 engine artifact cache + - ${HF_CACHE_DIR:-${HOME}/.cache/huggingface}:/root/.cache/huggingface + ipc: host + shm_size: "16gb" + ulimits: { memlock: -1, stack: 67108864 } + deploy: { resources: { reservations: { devices: [{ driver: nvidia, capabilities: [gpu] }] } } } + restart: unless-stopped +``` +> `command` 用 `bash -lc "uvicorn ..."` 讓 backend 進程本身就在 login 環境 → 它 spawn 的 trtllm-serve 繼承對的 +> `LD_LIBRARY_PATH`(這樣連 §9 的 (B) 包 bash 都可省)。**這是處理 LD_LIBRARY_PATH 最省事的一招。** + +### schema +`engine: Literal["vllm", "sglang", "llamacpp", "trtllm"]`;`LLMEngine.settings` 加選配 `engine_dir: Optional[str]`、 +`tokenizer: Optional[str]`(extra="allow" 其實也能塞,但明列較清楚)。 + +### Makefile +`_ALL_ENGINES := vllm sglang llamacpp trtllm`;`make up-trtllm` 捷徑(對照 up-sglang)。 + +--- + +## 11. 前端 + +P1 之後**免動**:`/api/engines` 會自動帶出 trtllm(name/capabilities/inapplicable_keys/paste_example), +Add Model 選單、capability 欄位、遮蔽鍵全自動。選配:`ModelsView`/`ModelGroupCard` 給 trtllm 一個 badge 顏色。 + +--- + +## 12. 測試計畫 + +- `test_launchers.py`:`TrtllmLauncher.build_spec`(command 含 `--backend tensorrt` / `--extra_llm_api_options` / + `served_name`)、`build_trtllm_cli_args` 的 param map + drop keys、perf YAML 內容(`return_perf_metrics: true`、 + **不含** `enable_iter_perf_stats`)。 +- `test_vllm_metrics_client.py`:`METRIC_PATH_BY_ENGINE["trtllm"] == "/prometheus/metrics"`;`fetch` 對 trtllm 打對路徑; + 五個名解析成 `VLLMInstanceMetrics`。 +- `test_engines_routes.py`:`/api/engines` 帶出 trtllm(caps 不含 sleep)。 +- `test_manager_engine.py`:engine=trtllm 有 launcher → 不再是 phantom。 +- Phase 2:`prepare()` 的 cache 命中/未命中、manifest 相容性、PREPARING 狀態轉移。 + +--- + +## 13. 落地步驟(每步可獨立 commit,跑全測) + +**Phase 1:** +1. schema Literal + `engine_dir`/`tokenizer` key。 +2. router:`METRIC_NAMES_BY_ENGINE["trtllm"]` + `METRIC_PATH_BY_ENGINE` + `fetch` 用 per-engine path + 測試。 +3. `TrtllmLauncher` + `build_trtllm_cli_args` + perf YAML + main.py 註冊 + 測試。 +4. `engine-trtllm.Dockerfile` + compose service(`command: bash -lc`)+ Makefile。 +5. live 驗證:用已 build 好的 engine(`/home/max/trtllm-test/engines/qwen3-06b_...`),從 dashboard 起、經 router 推理、看 `/prometheus/metrics` 有進 Prometheus。 + +**Phase 2:** +6. Launcher protocol 加 `prepare()`/`available()` + `ModelState.PREPARING` + manager 呼叫 + reconciler/前端狀態。 +7. `TrtllmLauncher.prepare()`:convert + build + manifest + cache key + 相容性檢查。 + +--- + +## 14. 契約缺口 / 風險 / open questions + +- **🔴 TensorRT backend 淘汰**:`v1.3.0rc20` 是**最後一個**支援 `--backend tensorrt` 的版本,下一版移除。→ 要嘛 + pin 在此版,要嘛接受未來得改走 pytorch backend(那時 engine build 不再需要,但 sleep 才可用)。**這是決定 + engine-first 值不值得的最大變數,先跟團隊確認。** +- **engine 綁 GPU 架構/版本**:mixed 多機或換卡要重 build;artifact cache key 必含 compute cap + 版本。 +- **build timeout / 資源**:大模型 build 到分鐘級且吃 CPU/GPU/磁碟,`prepare()` 不能算進 startup timeout(§8.2)。 +- **硬體支援**:官方支援 A100/H100/L40 等;consumer 卡(如驗證用的 sm_86 8GB)能跑小模型但非官方支援。 +- **LoRA**:要做只能走 YAML `lora_config` 靜態掛載 + request `extra_body`,非 vLLM 式端點——Phase 1 先不做。 + +--- + +## 15. 不做什麼(第一版明確排除) + +- `--backend pytorch`(本 launcher 目標是 engine)、TensorRT engine 以外的 backend。 +- disaggregated / KV connector(踩 instance=多行程假設,要 protocol 擴充)。 +- runtime LoRA 熱掛、sleep/wake(engine backend 不支援)。 +- `/metrics`、`/perf_metrics` 的 JSON(用 `/prometheus/metrics` 就好)。 diff --git a/docs/trtllm-launcher-research.md b/docs/trtllm-launcher-research.md new file mode 100644 index 0000000..4f3a1c5 --- /dev/null +++ b/docs/trtllm-launcher-research.md @@ -0,0 +1,417 @@ +# TensorRT-LLM Launcher 實作前規格盤點 + +查詢日期: 2026-07-04 +主要版本基準: TensorRT-LLM `v1.3.0rc20` / NGC `nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20` + +> **✅ 已實測驗證(2026-07-04):** 本 launcher 目標是 **TRT engine 模型**,走 `--backend tensorrt`。已在 sm_86 +> 8GB 實測轉換 + serve + 端點,並讀原始碼確認 metrics/sleep 行為。詳見 +> [trtllm-conversion-validation_zh-TW.md](trtllm-conversion-validation_zh-TW.md)。以下 §4、§5、§9 已依實測更新 +> (標「實測」處)。**engine-first 前提下,`prepare()`(engine build/artifact)是必要的,非選配。** + +重點警訊: GitHub release 顯示 `v1.3.0rc20` 是最後一個支援 TensorRT backend 的 RC, 下一版會移除 TensorRT backend。因此若本專案要新增 `trtllm`, 建議第一版以 `trtllm-serve` 的預設 PyTorch backend 為主, TensorRT engine 路線視為進階/legacy/build-artifact 模式。 + +## 來源 + +- TensorRT-LLM `trtllm-serve` 官方文件: https://nvidia.github.io/TensorRT-LLM/commands/trtllm-serve/trtllm-serve.html +- TensorRT-LLM Quick Start: https://nvidia.github.io/TensorRT-LLM/quick-start-guide.html +- TensorRT-LLM Prometheus Metrics example: https://nvidia.github.io/TensorRT-LLM/examples/prometheus_metrics.html +- TensorRT-LLM metrics collector source: https://github.com/NVIDIA/TensorRT-LLM/blob/main/tensorrt_llm/metrics/collector.py +- TensorRT-LLM LoRA feature doc: https://nvidia.github.io/TensorRT-LLM/features/lora.html +- TensorRT-LLM KV Cache Connector doc: https://nvidia.github.io/TensorRT-LLM/features/kv-cache-connector.html +- TensorRT-LLM container images doc: https://nvidia.github.io/TensorRT-LLM/installation/containers.html +- TensorRT-LLM supported hardware: https://nvidia.github.io/TensorRT-LLM/supported-hardware.html +- TensorRT-LLM GitHub releases: https://github.com/NVIDIA/TensorRT-LLM/releases +- TensorRT engine compatibility: https://docs.nvidia.com/deeplearning/tensorrt/latest/inference-library/engine-compatibility.html + +## 總結結論 + +建議新增 launcher 時把 `trtllm` 分成兩條路: + +1. `trtllm-serve` / PyTorch backend: 最貼近本專案現有契約。可用 HF repo 或本地 checkpoint 直接啟動, 單一 HTTP server, 有 OpenAI `/v1`, 有 `/health`, 有 `--host`, `--port`, `--served_model_name`, 也有 Prometheus 指標端點。這條可以先照 vLLM/SGLang 樣板做。 +2. TensorRT engine backend: 有 artifact/cache/build 相容性問題, 且 `v1.3.0rc20` release 已宣告下一版會移除 TensorRT backend。若仍要支援, 應視為 legacy/advanced mode, 需要 `prepare()` 或 engine artifact 管理, 不建議第一版混進普通 `spawn-and-probe`。 + +## 第 0 步: 六個生死題 + +| # | 結論 | TRT-LLM 最新答案 | Launcher 影響 | +|---|---|---|---| +| 0.1 啟動前需要離線 build engine 嗎? | ⚠️ 分 backend | `trtllm-serve` 的 `MODEL` 可為 model name、HF checkpoint path、TensorRT engine path。Quick Start 直接示範 `trtllm-serve "TinyLlama/TinyLlama-1.1B-Chat-v1.0"`。但 TensorRT engine 路線仍涉及 build artifact, TensorRT plan 相容性也需管理。 | 第一版用預設 PyTorch backend 可不做 `prepare()`。若要 `--backend tensorrt`/engine path, 建議加 `prepare()` 或使用者提供 engine path。 | +| 0.2 單行程還是多行程? | ⚠️ 基本 serve 單 command, 分散式/TP 可能多 rank | `trtllm-serve serve` 是單一 CLI 入口。多節點範例用 `trtllm-llmapi-launch trtllm-serve ...`、Slurm `srun`,disaggregated 有獨立 subcommand/worker。TP/PP 在單機多 GPU 下可能由 launcher/內部 worker 管理, 需以 process group kill。 | aggregated `trtllm-serve serve` 可先支援。disaggregated/Slurm/多節點不要先納入 instance=single-process 假設。 | +| 0.3 有 OpenAI 相容 `/v1` 嗎? | ✅ 符合契約 | 官方列出 `/v1/models`, `/v1/completions`, `/v1/chat/completions`; 文件也提到 Responses API examples。 | router 可直接轉發 OpenAI `/v1`。 | +| 0.4 CLI 旗標還是 config 檔? | ✅/⚠️ 兩者都有 | 有 CLI flags;也有 `--config`/`--extra_llm_api_options` YAML。官方說 CLI 顯式值優先於 YAML。巢狀性能項常走 YAML。 | 常用啟動參數用 CLI。LoRA、CUDA graph、KV cache dtype 等巢狀設定可由 launcher 產生暫存 YAML, 或第一版不支援。 | +| 0.5 engine artifact 綁 GPU/版本嗎? | 🔴 TensorRT engine 路線踩線 | TensorRT plan 預設記錄 TensorRT runtime version 與 compute capability;不匹配可能無法 deserialize。硬體相容模式可放寬, 但可能損失效能。 | engine cache key 至少要含 model、TRT-LLM/TensorRT version、CUDA/container tag、GPU compute capability、TP/PP、max seq/batch/token、quantization。 | +| 0.6 ready 訊號是什麼? | ✅ 符合契約 | `trtllm-serve` 支援 `/health`。也有 `/metrics`, `/version`。 | readiness probe 可先用 `GET /health == 200`。若 build/load 長, 要拉長 startup timeout 或看 log growth。 | + +## 1. 啟動 / serving + +推薦第一版 template: + +```bash +trtllm-serve serve \ + --host 0.0.0.0 \ + --port \ + --served_model_name \ + --max_seq_len \ + --max_batch_size \ + --max_num_tokens \ + --kv_cache_free_gpu_memory_fraction \ + --tp_size \ + --pp_size \ + --trust_remote_code +``` + +備註: + +- `trtllm-serve serve [OPTIONS] MODEL` 是文件列出的 OpenAI-compatible server command。 +- `MODEL` 可為 model name、HF checkpoint path、TensorRT engine path。 +- `--backend` 可選 `pytorch | tensorrt | _autodeploy`, 預設是 `pytorch`。 +- `--served_model_name` 存在;若未指定, API model name 使用 model path。可滿足 adopt 身分驗證。 +- `--host` 預設 `localhost`, 本專案應覆寫成 `0.0.0.0`。 +- `--port` 預設 `8000`。 +- `--grpc` 會改跑 gRPC server, 不應用於 OpenAI HTTP launcher。 + +## 2. 參數盤點 + +### `_TRTLLM_PARAM_MAP` 建議 + +| engine-neutral key | TRT-LLM flag / config | serve 期或 build 期 | 建議處理 | +|---|---|---|---| +| `max_model_len` | `--max_seq_len` | serve 期;TensorRT engine 模式可能也是 build 形狀的一部分 | 直接 map。說明為 prompt+output 總長。 | +| `gpu_memory_utilization` | `--kv_cache_free_gpu_memory_fraction` / alias `--free_gpu_memory_fraction` | serve 期 | 直接 map, 但語意不同: 是模型/緩衝配置後保留給 KV cache 的 free GPU memory fraction。 | +| `tensor_parallel_size` | `--tp_size` / `--tensor_parallel_size` | serve 期;engine artifact 可能需一致 | 直接 map。 | +| `pipeline_parallel_size` | `--pp_size` / `--pipeline_parallel_size` | serve 期;engine artifact 可能需一致 | 直接 map。 | +| `max_batch_size` | `--max_batch_size` | serve 期;TensorRT engine build config 也常需要 | 若前端已有 neutral key, map。 | +| `max_num_tokens` | `--max_num_tokens` | serve 期;TensorRT engine build config 也常需要 | 建議新增 TRT-LLM-specific option。 | +| `dtype` | 無明確 top-level `trtllm-serve --dtype`; PyTorch backend 多由 checkpoint/model metadata 決定 | 多數情況不適合 generic serve flag | 第一版 drop 或放 `--config` 進階欄。 | +| `quantization` | 預量化模型如 `nvidia/Qwen3-8B-FP8`; `kv_cache_dtype` 可設 KV cache | 多為模型/checkpoint/build 期 | 不要把 generic quantization 直接轉 flag。 | +| `kv_cache_dtype` | `--kv_cache_dtype auto|fp8|nvfp4` | serve 期, prototype | 可做 TRT-LLM-specific option。 | +| `chunked prefill` | `--enable_chunked_prefill` | serve 期, prototype | 可支援, 但標 beta/prototype。 | +| `speculative decoding` | 文件有 feature/examples, 但 `trtllm-serve` top-level flag 不像 vLLM 那樣簡單 | 多走 config/模型 recipe | 第一版不做 generic map。 | + +### `_TRTLLM_DROP_KEYS` / `inapplicable_keys` 建議 + +| key | 原因 | +|---|---| +| `n_gpu_layers` | TensorRT-LLM 是 GPU serving, 不走 llama.cpp 逐層 CPU/GPU offload 語意。 | +| `enforce_eager` | TRT-LLM 不是 vLLM eager/CUDA graph 切換語意。 | +| `gguf_quant` | TRT-LLM 不吃 GGUF 量化檔語意。 | +| `model_file` | 若現有 key 指 llama.cpp `.gguf`, 對 TRT-LLM 不適用。 | +| generic `quantization` | TRT-LLM 的 FP8/NVFP4/INT4 等多半是 model/checkpoint/build/config recipe, 不宜直接等同 vLLM `--quantization`。 | + +### build 期 vs serve 期 + +| 類別 | 參數/資料 | 建議 | +|---|---|---| +| 直接 serve 期 | `--host`, `--port`, `--served_model_name`, `--max_seq_len`, `--max_batch_size`, `--max_num_tokens`, `--kv_cache_free_gpu_memory_fraction`, `--tp_size`, `--pp_size`, `--trust_remote_code`, `--hf_revision`, `--chat_template` | 可由 `build_spec.command` 直接生成。 | +| YAML/config 期 | `lora_config`, `peft_cache_config`, `cuda_graph_config`, `kv_cache_config`, `moe_config`, `enable_iter_perf_stats` | 需 launcher 產生暫存 YAML 或暫不支援。 | +| build/artifact 期 | TensorRT engine path、TensorRT backend、engine hardware/version compatibility、某些 quantization/build shape/parallelism | 建議等 `prepare()` hook 或手動提供 engine path。 | + +## 3. LoRA + +| 面向 | TRT-LLM 最新答案 | 判斷 | +|---|---|---| +| 啟動時靜態掛載 | 透過 YAML `lora_config` 設 `lora_dir`, `max_lora_rank`, `max_loras`, `max_cpu_loras`, `lora_target_modules` 等。 | ⚠️ 能用但要 launcher 產 config 檔。 | +| runtime 熱掛 endpoint | 沒查到 vLLM 風格 `/v1/load_lora_adapter` / unload endpoint。官方 `trtllm-serve with LoRA` 是 request 的 `extra_body.lora_request` 指定 `lora_name`, `lora_int_id`, `lora_path`。 | 不宣告 `CAP_RUNTIME_LORA`。 | +| adapter 格式 | PyTorch backend 支援 HuggingFace 與 NeMo LoRA 格式。MoE LoRA 也支援 HF PEFT per-expert layout, 但限制較多。 | 可宣告靜態/請求式 LoRA 能力, 但需要 protocol/UI 配合。 | +| max_lora_rank / max_loras | `lora_config.max_lora_rank`, `max_loras`, `max_cpu_loras`。 | YAML config。 | + +建議 capability: + +- `CAP_LORA_MODULES`: 可加, 但實作不是 vLLM `--lora-modules NAME=PATH` 格式;要轉成 YAML 或 request `extra_body`。 +- `CAP_RUNTIME_LORA`: 不加。 + +## 4. 監控指標 + +TRT-LLM 現在有兩種 metrics: + +1. `/metrics`: 官方 `trtllm-serve` 文件描述為 runtime iteration statistics, 回傳 JSON-like runtime stats, 且部分 stats 可能被 poll 後移除。 +2. `/prometheus/metrics`: Prometheus exposition text, 範例明確使用這個 URL, prefix 是 `trtllm_`。 + +router 若需要 Prometheus 正規化, 建議 scrape `/prometheus/metrics`, 不是 `/metrics`。 + +> **✅ 實測修正(engine backend, v1.3.0rc20):** +> - `/prometheus/metrics` 是**條件式掛載**——`openai_server.py` 只在 `return_perf_metrics == true` 時才 +> `mount_metrics()`。**沒開就 404。** launcher 必須經 `--extra_llm_api_options ` 傳 +> `return_perf_metrics: true`(build_spec 要自動產這個暫存 YAML),否則完全沒有 Prometheus 指標。 +> - 🔴 **`enable_iter_perf_stats` 不能用在 TensorRT backend**(實測 `ValueError: _TrtLLM got invalid argument`, +> 那是 pytorch/autodeploy 專屬)。TRT engine 只要 `return_perf_metrics: true` 就夠。 +> - 下表五個名稱**實測全部存在**於 `/prometheus/metrics`(prefix `trtllm_`,text 底線格式,label 含 +> `model_name`/`engine_type`):`trtllm_num_requests_running`、`trtllm_num_requests_waiting`、 +> `trtllm_kv_cache_utilization`、`trtllm_prompt_tokens_total`、`trtllm_generation_tokens_total`。 +> → §9.4 的 `METRIC_NAMES_BY_ENGINE["trtllm"]` 正確,scrape path 用 `/prometheus/metrics`。 +> - ⚠️ router 現有 `metrics_poller` 是打 `/metrics`,trtllm 需要 `/prometheus/metrics`——router 要 +> 支援 **per-engine 的 metrics path**(不是所有引擎都 `/metrics`)。 + +建議 `METRIC_NAMES_BY_ENGINE["trtllm"]`: + +| 概念 | TRT-LLM Prometheus metric | 備註 | +|---|---|---| +| running | `trtllm_num_requests_running` | Gauge, active requests。 | +| waiting | `trtllm_num_requests_waiting` | Gauge, queued requests。 | +| KV cache 使用率 | `trtllm_kv_cache_utilization` | Gauge。也有 `trtllm_kv_cache_hit_rate`, `*_used_blocks`, `*_free_blocks`。 | +| prompt tokens | `trtllm_prompt_tokens_total` | Counter。 | +| generation tokens | `trtllm_generation_tokens_total` | Counter。 | + +其他可用: + +- `trtllm_request_success_total` +- `trtllm_request_error_total` +- `trtllm_e2e_request_latency_seconds` +- `trtllm_time_to_first_token_seconds` +- `trtllm_time_per_output_token_seconds` +- `trtllm_request_queue_time_seconds` +- `trtllm_gpu_memory_usage_bytes` +- `trtllm_num_context_requests` +- `trtllm_num_generation_requests` +- `trtllm_spec_decode_*` + +開啟方式: + +- Prometheus example 沒顯示需要額外 flag。 +- iteration-level stats 對 PyTorch backend 可能需要在 YAML 設 `enable_iter_perf_stats: true`, 且官方提醒可能有些效能影響。 + +## 5. Sleep / 暖待命 + +未找到 `trtllm-serve` 提供 `/sleep`, `/wake_up`, `/is_sleeping` 或類似 vLLM sleep mode 的官方 endpoint/flag。 + +> **✅ 實測修正:** 其實**有** `POST /release_memory` + `POST /resume_memory`——原始碼裡分別是 +> `collective_rpc('sleep')` / `collective_rpc('wakeup')`,語意就是 sleep/wake。**但**三個端點(含 +> `/update_weights`)都 `assert isinstance(self.generator, AsyncLLM)`,AsyncLLM 是 **PyTorch backend** 的 +> generator。**實測在 TensorRT engine backend 上 `POST /release_memory` / `/resume_memory` 都回 500。** +> → engine-first 路線**確定不宣告 `CAP_SLEEP`**(結論不變,但原因是「端點存在、僅 pytorch backend 支援」)。 + +建議: + +- 不宣告 `CAP_SLEEP`。 +- router `ENGINE_SLEEP_CAPABLE` 不加 `trtllm`。 + +## 6. KV transfer / disaggregated + +TRT-LLM 確實有: + +- `trtllm-serve disaggregated` subcommand。 +- `--server_role` 可指定 `CONTEXT`/prefill、`GENERATION`/decode 等角色。 +- KV Cache Connector, 支援外部 KV cache、offloading、prefill/decode 分離、P2P/cache sharing 等進階場景。 + +但這會踩本專案目前 instance 假設: + +- disaggregated mode 不是單一 OpenAI server process 即可完整代表一個 instance。 +- 需要 metadata server / worker / role / cluster URI 等設定。 +- KV connector 是開發者 API/config, 不是 vLLM 那種簡單 `kv_transfer_config` 直通。 + +建議: + +- 第一版只支援 aggregated `trtllm-serve serve`。 +- 不宣告 `CAP_KV_TRANSFER`。 +- 後續若支援 disagg, 應新增 protocol: 一個 logical instance = 多 role/process/service。 + +## 7. 部署 / Docker + +推薦 base image: + +```Dockerfile +FROM nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20 +``` + +依官方 container doc: + +- `release` image 是 runtime image, 安裝好 pre-built wheel, ready to use。 +- `devel` image 是 build/dev 環境。 +- NGC 最新 tag 查到 `1.3.0rc20`, updated 2026-06-29/30 UTC, compressed size 約 20.69 GB。 +- 官方 run 範例使用 `--ipc host`, `--gpus all`, `--ulimit memlock=-1`, `--ulimit stack=67108864`。 +- 文件提醒 local devel run 要設 `--ipc=host`, 否則可能 bus error。 + +compose 建議: + +```yaml +services: + engine-trtllm: + image: nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20 + ipc: host + shm_size: "16gb" + ulimits: + memlock: -1 + stack: 67108864 + deploy: + resources: + reservations: + devices: + - capabilities: [gpu] + volumes: + - hf-cache:/root/.cache/huggingface + - trtllm-cache:/models/trtllm +``` + +GPU / CUDA / 硬體: + +- TensorRT-LLM supported hardware: Blackwell B200/GB200/B300/GB300/DGX Spark, Hopper H100/H200/GH200, Ada L20/L40/L40S, Ampere A100。 +- pip doc 查到當前 Linux wheel 前置需求含 CUDA Toolkit 13.1、PyTorch 2.10.0 CUDA 13.0 wheel、Python 3.12 sanity check。用 NGC release image 可避免自行配版本。 +- FP8/FP4/NVFP4 等能力依硬體而定;文件範例提醒跑 `nvidia/Qwen3-8B-FP8` 前要確認 GPU 支援 FP8。 + +artifact 策略: + +- PyTorch backend: 主要 cache HF model/checkpoint。 +- TensorRT engine backend: engine 目錄應放獨立 volume, cache key 要含 GPU compute capability / TensorRT-LLM version / TensorRT version / CUDA/container tag / TP/PP / max seq/batch/token / quantization。不要跨不同 GPU 架構盲用。 + +## 8. paste-command 解析 + +建議 sniff: + +- `trtllm-serve` +- `python -m tensorrt_llm.commands.serve` +- `trtllm-llmapi-launch trtllm-serve` + +第一版 parser 支援: + +- positional `MODEL` +- `--host` +- `--port` +- `--served_model_name` +- `--backend` +- `--max_seq_len` +- `--max_batch_size` +- `--max_num_tokens` +- `--kv_cache_free_gpu_memory_fraction` / `--free_gpu_memory_fraction` +- `--tp_size` / `--tensor_parallel_size` +- `--pp_size` / `--pipeline_parallel_size` +- `--trust_remote_code` +- `--config` / `--extra_llm_api_options` + +遇到以下指令先標為 unsupported/advanced: + +- `trtllm-serve disaggregated ...` +- `trtllm-serve disaggregated_mpi_worker ...` +- `trtllm-llmapi-launch ...` 多節點/Slurm +- `--grpc` + +## 9. 交付項對應 + +### 9.1 六題答案 + +> **本 launcher 目標是 TRT engine 模型 → 走 `--backend tensorrt`(不是 pytorch)。** 因此: + +- 使用 `trtllm-serve serve --backend tensorrt --tokenizer `。 +- **`prepare()` 是必要的**:先 `convert_checkpoint.py` + `trtllm-build`(或使用者提供 pre-built engine 目錄), + 才 spawn server。build 期不可算進普通 startup timeout。 +- 需要 engine artifact **相容性快取 + 檢查**(cache key:GPU compute cap / TRT-LLM ver / TRT ver / container + tag / TP·PP / max seq·batch·token / quant)。實測 build 出的 engine 綁環境。 +- ⚠️ **`v1.3.0rc20` 是最後一個支援 TensorRT backend 的版本,下一版移除**——升級策略要先想(屆時要嘛 pin 版本、 + 要嘛改走 pytorch backend)。 +- 不支援 disaggregated。 + +> (若哪天要退而求其次:`--backend pytorch` 直接吃 HF 目錄可免 build/artifact,但那不是本 launcher 的目標, +> 且 sleep 端點也只有 pytorch backend 能用。) + +### 9.2 啟動指令模板 + +```bash +trtllm-serve serve "${MODEL}" \ + --host 0.0.0.0 \ + --port "${PORT}" \ + --served_model_name "${SERVED_MODEL_NAME}" \ + --max_seq_len "${MAX_MODEL_LEN}" \ + --kv_cache_free_gpu_memory_fraction "${GPU_MEMORY_UTILIZATION}" \ + --tp_size "${TENSOR_PARALLEL_SIZE}" \ + --pp_size "${PIPELINE_PARALLEL_SIZE}" +``` + +### 9.3 三張參數表 + +最小 `_TRTLLM_PARAM_MAP`: + +```python +_TRTLLM_PARAM_MAP = { + "max_model_len": "--max_seq_len", + "gpu_memory_utilization": "--kv_cache_free_gpu_memory_fraction", + "tensor_parallel_size": "--tp_size", + "pipeline_parallel_size": "--pp_size", + "max_batch_size": "--max_batch_size", + "max_num_tokens": "--max_num_tokens", +} +``` + +最小 `_TRTLLM_DROP_KEYS`: + +```python +_TRTLLM_DROP_KEYS = { + "n_gpu_layers", + "gguf_quant", + "model_file", + "enforce_eager", + "quantization", +} +``` + +TRT-LLM-specific 可選: + +```python +_TRTLLM_EXTRA_KEYS = { + "kv_cache_dtype": "--kv_cache_dtype", + "enable_chunked_prefill": "--enable_chunked_prefill", + "trust_remote_code": "--trust_remote_code", + "hf_revision": "--hf_revision", + "chat_template": "--chat_template", +} +``` + +### 9.4 五個指標名 + +```python +METRIC_NAMES_BY_ENGINE["trtllm"] = { + "running": "trtllm_num_requests_running", + "waiting": "trtllm_num_requests_waiting", + "kv_cache_usage": "trtllm_kv_cache_utilization", + "prompt_tokens": "trtllm_prompt_tokens_total", + "generation_tokens": "trtllm_generation_tokens_total", +} +``` + +scrape path: + +```text +/prometheus/metrics +``` + +### 9.5 capability 集 + +建議第一版: + +```python +TRTLLM_CAPABILITIES = { + CAP_OPENAI_V1, + CAP_METRICS, + CAP_LORA_MODULES, # only if launcher can emit YAML / request extra_body policy +} +``` + +不建議第一版宣告: + +```python +CAP_SLEEP +CAP_RUNTIME_LORA +CAP_KV_TRANSFER +``` + +### 9.6 base image + artifact 策略 + +- base image: `nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20` +- compose: `ipc: host`, GPU access, memlock/stack ulimits, HF cache volume。 +- engine artifact volume: only for TensorRT engine/legacy path, e.g. `/models/trtllm/engines`. +- cache key: include GPU compute capability and TRT/container versions。 + +## 最後建議 + +短期實作順序(engine-first): + +1. 新增 `trtllm` engine,`trtllm-serve serve --backend tensorrt` aggregated OpenAI HTTP。 +2. **`prepare()` hook**:engine build(`convert_checkpoint.py` + `trtllm-build`)或吃使用者給的 engine 目錄 + + artifact 相容性快取。**這是 engine-first 的核心工作,不是後續。** +3. readiness 使用 `/health`(實測 engine 載入 ~35s)。 +4. metrics:launcher 產含 `return_perf_metrics: true` 的 YAML 傳 `--extra_llm_api_options`;router scrape + **`/prometheus/metrics`**(非 `/metrics`),`METRIC_NAMES_BY_ENGINE["trtllm"]` 用 §9.4 五名(已實測)。 + → router 的 metrics path 要能 per-engine(不是全部 `/metrics`)。 +5. 參數:drop generic quantization/dtype,保留 `max_seq_len`, `tp_size`, `pp_size`, KV cache fraction、 + `max_batch_size`, `max_num_tokens`。注意有些是 **build 期**形狀(engine 綁),不能 serve 期動態改。 +6. capability:`CAP_METRICS_TRTLLM`(+ LoRA 若做 YAML 靜態)。**不宣告 CAP_SLEEP / CAP_RUNTIME_LORA / + CAP_KV_TRANSFER**(sleep 端點僅 pytorch backend、實測 engine backend 回 500)。 +7. disaggregated / KV transfer / pytorch-backend sleep 放到後續 protocol 擴充。 + diff --git a/docs/trtllm_research_checklist_zh-TW.md b/docs/trtllm_research_checklist_zh-TW.md new file mode 100644 index 0000000..66f0851 --- /dev/null +++ b/docs/trtllm_research_checklist_zh-TW.md @@ -0,0 +1,148 @@ +# 加入 TensorRT-LLM(trtllm)前要查的資料清單 + +> 目的:在動手寫 `TrtllmLauncher` 之前,先把「這個引擎怎麼啟動 / 加速 / 部署」問清楚。做法:對標 +> 現有三個引擎(vLLM / SGLang / llama.cpp)**已知**的做法,逐項列出 TRT-LLM 要確認的問題。你把最右欄填完, +> 就等於完成了實作前的規格盤點。 +> +> 搭配閱讀:[adding-a-new-engine_zh-TW.md](adding-a-new-engine_zh-TW.md)(觸點總表 + 隱含契約)。 +> TRT-LLM 文件的兩條主線:**`trtllm-serve`**(較新的 OpenAI 相容 server)與 **Triton + +> tensorrtllm_backend**(較舊、多行程)。**優先確認 `trtllm-serve` 路線**,因為它最貼近本專案的 +> 「單行程 + CLI + OpenAI /v1」模型。 +> +> 圖例:填「要查」欄時,標 **✅ 符合契約** / **⚠️ 能用但要處理** / **🔴 契約踩線(可能要 protocol hook)**。 + +--- + +## 🔴 第 0 步:先回答這 6 個「生死題」(決定成本是 2 天還是 2 週) + +這 6 題決定 TRT-LLM 是「照樣板加」還是「要先擴 Launcher protocol」。**先查這個,別先寫 code。** + +| # | 生死題 | 為什麼致命 | vLLM/SGLang/llama.cpp 的答案 | TRT-LLM 要查 | +|---|---|---|---|---| +| 0.1 | **啟動前需要離線 build engine 嗎?**(`trtllm-build` / `trtllm-serve` 能不能吃 HF repo 直接跑?) | 若「啟動」= 編譯 + 載入 artifact,就不是 spawn-and-probe → 需要 `prepare()` hook,否則 build 期會被當 startup timeout 殺掉 | 都不用:給 HF repo / GGUF,spawn 即載入 | `trtllm-serve ` 能否自動 build?還是必須先 `trtllm-build` 產出 engine 目錄再指給 server? build 要多久? | +| 0.2 | **是單行程還是多行程?**(trtllm-serve 單行程?Triton 是不是 server+backend 多進程?) | 本專案 `spawn_process` + process-group kill 假設一個 instance = 一個 process | 都是單行程 | trtllm-serve 是單一 process 嗎?TP>1 時是 `mpirun`/多 rank 嗎(還算同一 process group 可一起 kill 嗎)? | +| 0.3 | **有 OpenAI 相容 `/v1`(chat/completions + models)嗎?** | router 轉發、身分驗證全靠 `/v1`;沒有就要寫 adapter | 都有 | trtllm-serve 提供 `/v1/chat/completions`、`/v1/completions`、`/v1/models` 嗎? | +| 0.4 | **用 CLI 旗標啟動嗎?還是只吃 config 檔?** | Launcher 產 `command: list[str]`;若只吃 YAML/json config,要改成寫檔 | 都是 CLI 旗標 | trtllm-serve 的參數是 CLI flags 還是 `--extra_llm_api_options a.yaml`?哪些只能走檔? | +| 0.5 | **engine artifact 是否綁 GPU 架構 / 版本?** | TRT engine 通常**綁 compute capability + TRT 版本**,換卡要重 build → 影響部署與快取策略 | 無(權重可攜) | 同一份 build 能跨 GPU 型號嗎?混合部署(和別的引擎同機)時 engine 目錄放哪、多大、怎麼快取? | +| 0.6 | **ready 訊號是什麼?** | reconciler 用 `/health` 200 + log 成長判 ready | `/health` 200(llama.cpp 載入後才綁 port) | trtllm-serve 有 `/health`?載入/build 期間 `/health` 回什麼?log 會不會持續成長(progress timeout 靠它)? | + +> **若 0.1 = 要離線 build,或 0.2 = 多行程** → 先看 [adding-a-new-engine_zh-TW.md](adding-a-new-engine_zh-TW.md) +> §7 的 `prepare()` / `available()` / `readiness_probe()` roadmap,這是實作前就要決定的架構縫。 + +--- + +## 1. 啟動 / serving(build_spec 的 `command`) + +| 面向 | vLLM | SGLang | llama.cpp | TRT-LLM 要查 | +|---|---|---|---|---| +| 啟動指令 | `vllm serve ` | `python -m sglang.launch_server --model-path ` | `llama-server -hf [:quant]` / `-m ` | 二進位/模組名?`trtllm-serve `? | +| 指定 port | `--port` | `--port` | `--port` | ? | +| served name(/v1/models 廣告用,router 轉發 + adopt 驗身分) | `--served-model-name` | `--served-model-name` | `-a` / `--alias` | 有 `--served-model-name` 對應嗎?**沒有的話 adopt 身分驗證要另想** | +| 綁 host(跨容器路由,本專案用 0.0.0.0) | `--host` | `--host` | `--host` | ? | +| 模型來源 | HF repo / 本地路徑 | HF repo | HF GGUF repo / 本地 .gguf | HF repo?還是必須先 build 出的 engine 目錄? | + +--- + +## 2. 加速 / 效能參數(engine-neutral 鍵 → TRT-LLM 的對應) + +這張表決定 `build__cli_args` 的**參數翻譯**與 **`inapplicable_keys`**(前端會 grey out 的鍵)。 +逐一確認每個 engine-neutral 鍵在 TRT-LLM 是「改名對應」還是「不適用」。 + +| engine-neutral 鍵 | vLLM | SGLang | llama.cpp | TRT-LLM 要查(對應旗標 / 不適用?) | +|---|---|---|---|---| +| `max_model_len` | `--max-model-len` | `--context-length` | `-c` / `--ctx-size` | `--max_seq_len`?`--max_input_len` + `--max_num_tokens`?是 build 期還是 serve 期參數? | +| `gpu_memory_utilization` | `--gpu-memory-utilization` | `--mem-fraction-static` | 不適用(用 n_gpu_layers) | KV cache free fraction?`--kv_cache_free_gpu_memory_fraction`? | +| `tensor_parallel_size` | `--tensor-parallel-size` | `--tp-size` | 不適用 | `--tp_size`?是 build 期決定(engine 綁 TP)還是 serve 期? | +| `dtype` | `--dtype` | `--dtype` | 不適用(GGUF 內建) | build 期指定?`--dtype float16/bfloat16`? | +| `quantization` | `--quantization`(bitsandbytes…) | quant 旗標 | GGUF 量化檔(`gguf_quant` 選 Q4_K_M…) | FP8 / INT4-AWQ / INT8-SmoothQuant?**是 build 期產生量化 engine 嗎?** | +| `pipeline_parallel_size` | `--pipeline-parallel-size` | pp | 不適用 | `--pp_size`? | +| `enforce_eager` | `--enforce-eager` | — | 不適用 | 無對應?(TRT 本來就是 AOT 編譯) | +| max batch / in-flight batching | 自動 | 自動 | `--parallel` | `--max_batch_size`?in-flight batching 預設開嗎? | +| chunked prefill / prefix cache | `--enable-chunked-prefill` 等 | radix cache | `--cache-reuse` | `--enable_chunked_context`?`--use_paged_context_fmha`? | +| speculative decoding | `--speculative-...` | — | `--draft` | TRT-LLM 的 spec-dec / Medusa / Eagle 怎麼開? | +| n_gpu_layers(CPU offload) | 不適用 | 不適用 | `-ngl` | 不適用(TRT 全 GPU)→ 進 `inapplicable_keys` | + +> 產出:一張「TRT-LLM 專屬旗標」清單 + 一張「不適用鍵」清單(= `_TRTLLM_DROP_KEYS` / launcher 的 +> `inapplicable_keys`)+ 一張「參數改名表」(= `_TRTLLM_PARAM_MAP`)。**特別標出哪些是 build 期 vs serve 期** +> ——build 期參數不能在 `build_spec` 動態改,得走 §0.1 的 build 流程。 + +--- + +## 3. LoRA(capability:`CAP_LORA_MODULES` / `CAP_RUNTIME_LORA`) + +| 面向 | vLLM | SGLang | llama.cpp | TRT-LLM 要查 | +|---|---|---|---|---| +| 啟動時靜態掛載 | `--lora-modules NAME=PATH(JSON)` | `--lora-paths NAME=PATH` | `--lora ` | 有嗎?格式?`--lora_dir`?需要 build 期 `--use_lora_plugin` 嗎? | +| runtime 熱掛(load/unload 端點) | `POST /v1/load_lora_adapter` | `POST /load_lora_adapter`(無 /v1) | 不支援(僅啟動時 + 可調 scale) | 有 runtime load/unload 端點嗎?→ 決定要不要宣告 `CAP_RUNTIME_LORA` + `lora_endpoint_prefix` | +| adapter 格式 | HF PEFT | HF PEFT | GGUF(需轉檔) | HF PEFT?還是要 TRT 專屬轉換? | +| max_lora_rank / max_loras | `--max-lora-rank` / `--max-loras` | 同 | 不適用 | 對應旗標?build 期還是 serve 期? | + +--- + +## 4. 監控指標(router `METRIC_NAMES_BY_ENGINE` 要加一條) + +需要**五個概念指標**的實際名稱,router 才能正規化成統一形狀給 routing/autoscaler 用。 + +| 概念 | vLLM | SGLang(OpenMetrics,`:`→`_`) | llama.cpp | TRT-LLM 要查 | +|---|---|---|---|---| +| 開啟指標的旗標 | 預設有 | `--enable-metrics` | `--metrics` | 有 Prometheus `/metrics` 嗎?要不要旗標開? | +| 指標名前綴 | `vllm:` | `sglang:`(入庫變 `sglang_`) | `llamacpp:` | 前綴是什麼?(要獨特,別和別人撞) | +| running(處理中) | `vllm:num_requests_running` | `sglang:num_running_reqs` | `llamacpp:requests_processing` | ? | +| waiting(排隊) | `vllm:num_requests_waiting` | `sglang:num_queue_reqs` | `llamacpp:requests_deferred` | ? | +| KV cache 使用率 | `vllm:kv_cache_usage_perc` | `sglang:token_usage` | 無(指不存在名→0) | 有 KV 使用率指標嗎?沒有就比照 llama.cpp | +| prompt / generation tokens | `vllm:prompt_tokens` / `generation_tokens` | `sglang:*_total` | `llamacpp:prompt_tokens_total` / `tokens_predicted_total` | ? | +| 格式 | Prometheus text | OpenMetrics | Prometheus text | text 還是 OpenMetrics?(影響 `:` 會不會被正規化) | + +--- + +## 5. Sleep / 暖待命(capability:`CAP_SLEEP`) + +| 面向 | vLLM | SGLang | llama.cpp | TRT-LLM 要查 | +|---|---|---|---|---| +| 有 /sleep + /wake_up + /is_sleeping? | 有(`--enable-sleep-mode` + `VLLM_SERVER_DEV_MODE=1`) | 無 | 無 | 有把權重 page 到 CPU、釋放 VRAM、行程續活的機制嗎?**多半沒有 → 不宣告 CAP_SLEEP**,router `ENGINE_SLEEP_CAPABLE` 也別加 | + +--- + +## 6. KV transfer / disaggregated(capability:`CAP_KV_TRANSFER`) + +| 面向 | vLLM | SGLang | llama.cpp | TRT-LLM 要查 | +|---|---|---|---|---| +| 跨實例共享 KV | 有(OffloadingConnector,`kv_transfer_config`) | — | 無 | 有跨實例 KV 池嗎? | +| disaggregated prefill/decode | 有 | 有 | 無 | **TRT-LLM 有 disagg**——若要用,一個「instance」是不是 = prefill + decode 多行程?那會踩 §0.2 契約,先確認要不要支援,還是先只做 aggregated 單行程 | + +--- + +## 7. 部署 / Docker(`engine-trtllm.Dockerfile` + compose) + +| 面向 | vLLM | SGLang | llama.cpp | TRT-LLM 要查 | +|---|---|---|---|---| +| 官方 base image | `vllm/vllm-openai:latest` | SGLang 官方 image | `ghcr.io/ggml-org/llama.cpp:server-cuda` | `nvcr.io/nvidia/tritonserver:*-trtllm-python-py3`?還是有 `tensorrt_llm` release image?哪個含 `trtllm-serve`? | +| GPU / CUDA / 驅動需求 | 一般 | 一般 | 一般 | TRT-LLM 綁特定 CUDA / TRT / 驅動版本?compute capability 下限?(A100/H100 才有 FP8) | +| base image 的雷 | ENTRYPOINT 要清 | — | HEALTHCHECK/ENTRYPOINT 要清、`LD_LIBRARY_PATH` | base 有沒有自帶 ENTRYPOINT/HEALTHCHECK 要清?python 環境齊全嗎(要裝 backend deps)? | +| engine artifact 存放 | 無 | 無 | 無(HF/GGUF 快取) | build 出的 engine 目錄放哪?要不要獨立 volume?多大?能否預先 build 進 image? | +| compose 需要的 env | 標準 | `ipc: host` + `shm_size` | 無特殊 | 需要 `ipc: host` / `shm_size` / `--gpus` 特殊設定嗎?TP>1 要 `mpirun` 嗎? | + +> compose 其餘照樣板:`profiles: ["trtllm"]`、`LLMOPS_NODE_ENGINES=trtllm`、 +> `LLMOPS_PROMETHEUS_SD_PATH=/sd/targets-trtllm.json`、Makefile `_ALL_ENGINES` 加 `trtllm`。 + +--- + +## 8. paste-command 解析(選配,`vllm_command.py`) + +若想支援在 Add Model 貼上 `trtllm-serve ...` 自動解析:查 TRT-LLM 啟動指令的**完整旗標集**與**簡寫** +(像 llama.cpp 的 `-c`/`-ngl`),寫一個 `parse_trtllm_command` + 在 dispatcher 加 sniff 規則 +(認得指令特徵字,如 `trtllm-serve` / `trtllm-build`)。**非必要,可最後做。** + +--- + +## 9. 交付:查完這張清單你應該產出 + +1. **§0 六題的明確答案**(尤其 0.1 build / 0.2 多行程)→ 決定走「樣板加」還是「先擴 protocol」。 +2. **啟動指令模板**:`trtllm-serve <...> --port <> --served-model-name <>`(→ `paste_example` + `build_spec`)。 +3. **三張參數表**:改名(`_TRTLLM_PARAM_MAP`)、drop/不適用(`inapplicable_keys`)、build 期 vs serve 期。 +4. **五個指標名**(→ `METRIC_NAMES_BY_ENGINE["trtllm"]`)+ 是否有 KV 指標。 +5. **capability 集**:大概率是 `{CAP_LORA_MODULES?, CAP_METRICS_TRTLLM}`——sleep/kv_transfer/runtime_lora + 有沒有,逐一以「真的支援才加」為準。 +6. **base image + GPU 需求 + engine artifact 策略**(→ Dockerfile + compose volume)。 + +拿到這些,就能照 [adding-a-new-engine_zh-TW.md](adding-a-new-engine_zh-TW.md) 的 §9 驗收清單開工。 From 68da159f0bb22c4e65ea77fb6d278b8481785414 Mon Sep 17 00:00:00 2001 From: max Date: Sat, 4 Jul 2026 19:42:32 +0800 Subject: [PATCH 04/20] feat(trtllm): add TensorRT-LLM as the 4th engine (Phase 1: bring-your-own-engine) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Serve a pre-built TensorRT engine via `trtllm-serve serve --backend tensorrt`, following docs/trtllm-launcher-impl-design_zh-TW.md. - schema: `engine: trtllm` + engine_dir/tokenizer fields. - TrtllmLauncher + build_trtllm_cli_args: engine_dir positional, --backend tensorrt, --tokenizer/--served_model_name, param map (max_model_len->--max_seq_len, gpu_memory_utilization->--kv_cache_free_gpu_memory_fraction, tp/pp), a strict whitelist (trtllm-serve rejects unknown args), and a per-instance perf YAML (return_perf_metrics: true) passed via --extra_llm_api_options — without it there is no /prometheus/metrics. Capabilities: CAP_METRICS_TRTLLM only (no sleep/runtime-lora/ kv-transfer). Registered in main.py. - router: METRIC_NAMES_BY_ENGINE[trtllm] (5 verified trtllm_* names) + per-engine METRIC_PATH_BY_ENGINE so trtllm is scraped at /prometheus/metrics, not /metrics. - deploy: engine-trtllm.Dockerfile (FROM the TRT-LLM release image); mixed compose trtllm-backend service (profile, ipc:host + memlock/stack ulimits, /engines mount, `command: bash -lc "uvicorn …"` so spawned trtllm-serve inherits the login-shell LD_LIBRARY_PATH); Makefile ENGINES/up-trtllm. Live-verified end-to-end (RTX 3060 Ti, pre-built Qwen3-0.6B engine): backend loads the trtllm group, the launcher spawns trtllm-serve, model reaches READY (~45s), direct and router inference work, and the rebuilt router scrapes /prometheus/metrics into real per-instance load numbers. Tests: backend 494, router 131, schema 5. Co-Authored-By: Claude Opus 4.8 --- Makefile | 7 +- apps/backend/app/llmops/launchers.py | 171 ++++++++++++++++++ apps/backend/app/main.py | 6 +- apps/backend/tests/unit/test_launchers.py | 87 ++++++++- .../src/llm_router/vllm_metrics_client.py | 22 ++- .../tests/unit/test_vllm_metrics_client.py | 34 +++- deploy/docker-compose.mixed.yaml | 53 ++++++ deploy/engine-trtllm.Dockerfile | 48 +++++ packages/config-schema/schema.py | 9 +- 9 files changed, 425 insertions(+), 12 deletions(-) create mode 100644 deploy/engine-trtllm.Dockerfile diff --git a/Makefile b/Makefile index bab8885..b81a6bc 100644 --- a/Makefile +++ b/Makefile @@ -18,7 +18,7 @@ MIXED_ENV := COMPOSE_PROFILES=$(ENGINES) DASHBOARD_BACKEND=$(DASHBOARD_BACKEND) # Engines NOT selected this run. `docker compose up` (even with --remove-orphans) # leaves a profiled-but-deselected service running, so we stop+remove them explicitly # to make switching engine sets clean. Computed as ALL − ENGINES. -_ALL_ENGINES := vllm sglang llamacpp +_ALL_ENGINES := vllm sglang llamacpp trtllm _comma := , _space := $(empty) $(empty) _SELECTED := $(subst $(_comma),$(_space),$(ENGINES)) @@ -28,7 +28,7 @@ _DESELECTED_SVCS := $(addsuffix -backend,$(_DESELECTED)) .PHONY: help test test-backend test-router test-schema \ dev-backend dev-frontend build-frontend install-frontend \ up down logs ps build up-mixed down-mixed logs-mixed \ - up-vllm up-sglang up-llamacpp + up-vllm up-sglang up-llamacpp up-trtllm help: @echo "Targets:" @@ -89,6 +89,9 @@ up-sglang: up-llamacpp: $(MAKE) up-mixed ENGINES=llamacpp +up-trtllm: + $(MAKE) up-mixed ENGINES=trtllm + # down removes the whole project (all profiles) regardless of the current selection. down-mixed: COMPOSE_PROFILES=vllm,sglang,llamacpp $(COMPOSE_MIXED) down diff --git a/apps/backend/app/llmops/launchers.py b/apps/backend/app/llmops/launchers.py index dd5c1ed..eb91a6c 100644 --- a/apps/backend/app/llmops/launchers.py +++ b/apps/backend/app/llmops/launchers.py @@ -137,6 +137,7 @@ def build_vllm_cli_args(model_cfg: dict) -> list[str]: CAP_METRICS_VLLM = "metrics_vllm" # exposes vLLM-format Prometheus metrics (waiting queue, …) CAP_METRICS_SGLANG = "metrics_sglang" # exposes sglang:* Prometheus metrics (the router parses these) CAP_METRICS_LLAMACPP = "metrics_llamacpp" # exposes llamacpp:* Prometheus metrics (no kv-usage dim) +CAP_METRICS_TRTLLM = "metrics_trtllm" # exposes trtllm_* Prometheus metrics at /prometheus/metrics # Sentinel engine name for non-LLM launchers (embedding server): they aren't # selected by an engine choice, so they register under one fixed value. @@ -635,3 +636,173 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: model_tag=engine.settings.model_tag, served_name=merged.get("served_model_name") or engine.settings.model_tag, ) + + +# --- TensorRT-LLM (engine: trtllm) ------------------------------------------- +# Phase 1 "bring-your-own-engine": serve a PRE-BUILT TRT engine directory with +# `trtllm-serve serve --backend tensorrt`. The Phase 2 auto-build +# (convert_checkpoint + trtllm-build via a prepare() hook) is a separate step. +# See docs/trtllm-launcher-impl-design_zh-TW.md. + +# engine-neutral key -> trtllm-serve flag (single underscore, NOT kebab-case). +_TRTLLM_PARAM_MAP = { + "max_model_len": "max_seq_len", + "gpu_memory_utilization": "kv_cache_free_gpu_memory_fraction", + "tensor_parallel_size": "tp_size", + "pipeline_parallel_size": "pp_size", +} +# trtllm-native serve flags allowed to pass through as --. trtllm-serve is a +# strict click CLI (it errors on unknown args — e.g. enable_iter_perf_stats), so +# UNLIKE llama.cpp we do NOT pass arbitrary extra keys through; only this whitelist. +_TRTLLM_PASSTHROUGH = frozenset({ + "max_batch_size", "max_num_tokens", "kv_cache_dtype", + "trust_remote_code", "hf_revision", "chat_template", +}) +# vLLM/SGLang/llama.cpp knobs with no trtllm-serve equivalent (build-time or other +# engine) — greyed out in the dashboard (inapplicable_keys) and never emitted. +_TRTLLM_DROP_KEYS = frozenset({ + "n_gpu_layers", "gguf_quant", "model_file", "hf_file", + "enforce_eager", "dtype", "quantization", +}) +# Consumed specially by build_spec / build_trtllm_cli_args (addressing / served name / +# host / port / meta) — never emitted by the generic loop. +_TRTLLM_SKIP_CLI_KEYS = frozenset({ + "engine_dir", "tokenizer", "model_tag", "served_model_name", + "id", "cuda_device", "host", "port", + "enable_lora", "lora_modules", _LORA_RUNTIME_KEY, + "max_lora_rank", "max_loras", "lora_target_modules", "fully_sharded_loras", +}) | _ROUTER_ONLY_KEYS | _TRTLLM_DROP_KEYS + + +def build_trtllm_cli_args(model_cfg: dict, *, perf_yaml_path: str) -> list[str]: + """dict -> ``trtllm-serve`` CLI args (the part after ``trtllm-serve``). + + Phase 1 serves a pre-built TensorRT engine directory (``engine_dir``, required) + with ``--backend tensorrt``. ``MODEL`` is the positional engine dir. The tokenizer + is not inside the engine dir, so ``--tokenizer`` is always emitted (``tokenizer`` + override, else ``model_tag``). ``--served_model_name`` makes /v1/models advertise a + stable name (router forward_name + boot adopt identity). + + Engine-neutral keys map via ``_TRTLLM_PARAM_MAP`` (``max_model_len`` -> + ``--max_seq_len`` etc.); a small whitelist of trtllm-native flags passes through; + everything else is dropped (trtllm-serve rejects unknown args). ``--extra_llm_api_options`` + points at a per-instance YAML carrying ``return_perf_metrics: true`` — WITHOUT it + the server exposes no ``/prometheus/metrics``. + """ + engine_dir = model_cfg.get("engine_dir") + if not engine_dir: + raise ValueError( + "trtllm model_config must provide 'engine_dir' (a pre-built TRT engine " + "directory inside the container); Phase 1 is bring-your-own-engine") + model_tag = model_cfg.get("model_tag") + served = model_cfg.get("served_model_name") or model_tag + tokenizer = model_cfg.get("tokenizer") or model_tag + host = model_cfg.get("host") or "localhost" + port = model_cfg.get("port") + + args: list[str] = ["serve", str(engine_dir), "--backend", "tensorrt", + "--host", str(host)] + if port is not None: + args += ["--port", str(port)] + if tokenizer: + args += ["--tokenizer", str(tokenizer)] + if served: + args += ["--served_model_name", str(served)] + + for key, value in model_cfg.items(): + if value is None or key in _TRTLLM_SKIP_CLI_KEYS: + continue + if key in _TRTLLM_PARAM_MAP: + flag = "--" + _TRTLLM_PARAM_MAP[key] + elif key in _TRTLLM_PASSTHROUGH: + flag = "--" + key + else: + continue # unknown -> drop (trtllm-serve is strict) + if isinstance(value, bool): + if value: # store_true: True -> present; False -> omit + args.append(flag) + else: + args += [flag, str(value)] + + args += ["--extra_llm_api_options", str(perf_yaml_path)] + return args + + +class TrtllmLauncher: + kind = ModelKind.LLM + engine = "trtllm" + # Serves a pre-built TRT engine over OpenAI /v1. Metrics only (Prometheus trtllm_* + # at /prometheus/metrics). NO sleep (/release_memory is AsyncLLM/pytorch-only — + # 500 on the engine backend), NO runtime LoRA (no /v1/load_lora_adapter), NO + # kv_transfer (that's disaggregated). Static LoRA needs a YAML lora_config — not + # done in Phase 1. See docs/trtllm-launcher-impl-design_zh-TW.md §7. + capabilities = frozenset({CAP_METRICS_TRTLLM}) + lora_endpoint_prefix = "" + metric_prefix = "trtllm" + inapplicable_keys = _TRTLLM_DROP_KEYS + paste_example = ("trtllm-serve serve /engines/qwen3-06b_tp1_fp16 --backend tensorrt " + "--tokenizer Qwen/Qwen3-0.6B --served_model_name qwen3-06b-trt " + "--host 0.0.0.0 --port 8050") + + def keys(self, config) -> list[str]: + out: list[str] = [] + for model_tag, engine in config.LLM_engines.items(): + if getattr(engine.settings, "engine", "vllm") != self.engine: + continue + for inst in engine.instances: + out.append(f"{model_tag}::{inst.id}") + return out + + def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: + model_tag, _, instance_id = key.partition("::") + engine = config.LLM_engines.get(model_tag) + if engine is None: + raise KeyError(f"model group '{model_tag}' not in config") + inst = next((i for i in engine.instances if i.id == instance_id), None) + if inst is None: + raise KeyError(f"instance '{instance_id}' not in group '{model_tag}'") + + merged: dict = engine.settings.model_dump(by_alias=False) + merged.update(inst.model_dump()) + + env: dict[str, str] = {} + if merged.get("tensor_parallel_size", 1) == 1: + cuda_device = merged.pop("cuda_device", None) + if cuda_device is not None: + env["CUDA_VISIBLE_DEVICES"] = str(cuda_device) + merged.pop("id", None) + + # HA split deploys: bind to a routable interface (shares LLMOPS_VLLM_BIND_HOST + # with the other engines); local probe/record stay localhost. + bind_host = os.environ.get("LLMOPS_VLLM_BIND_HOST", "").strip() + cli_cfg = {**merged, "host": bind_host} if bind_host else merged + + # Per-instance YAML enabling Prometheus metrics. WITHOUT this the server has no + # /prometheus/metrics. (enable_iter_perf_stats is deliberately NOT set — the + # TensorRT backend rejects it.) + perf_yaml_path = os.path.join(LOG_DIR, f"{model_tag}__{instance_id}.trtllm-perf.yaml") + try: + os.makedirs(LOG_DIR, exist_ok=True) + with open(perf_yaml_path, "w", encoding="utf-8") as f: + f.write("return_perf_metrics: true\n") + except OSError: + logger.warning("trtllm: could not write perf YAML at %s", perf_yaml_path) + + command = ["trtllm-serve"] + build_trtllm_cli_args(cli_cfg, perf_yaml_path=perf_yaml_path) + log_path = os.path.join(LOG_DIR, f"{model_tag}__{instance_id}.log") + # readiness: /health binds only after the engine loads (~35s for a small model, + # much longer for big engines) — the reconciler's not-yet-200 handling covers it. + return LaunchSpec( + key=key, + kind=self.kind, + engine=self.engine, + capabilities=self.capabilities, + command=command, + env=env, + log_path=log_path, + host=inst.host, + port=inst.port, + probe_url=f"http://{inst.host}:{inst.port}/health", + model_tag=engine.settings.model_tag, + served_name=merged.get("served_model_name") or engine.settings.model_tag, + ) diff --git a/apps/backend/app/main.py b/apps/backend/app/main.py index 670c133..c82a7b3 100644 --- a/apps/backend/app/main.py +++ b/apps/backend/app/main.py @@ -37,7 +37,8 @@ from app.core.logging import setup_logging from app.core.settings import BackendSettings from app.core.store import LLMOpsStore -from app.llmops.launchers import EmbeddingLauncher, LlamacppLauncher, SglangLauncher, VllmLauncher +from app.llmops.launchers import (EmbeddingLauncher, LlamacppLauncher, SglangLauncher, + TrtllmLauncher, VllmLauncher) from app.llmops.manager import ModelManager, build_registry from app.llmops.autoscaler import autoscaler_loop from app.llmops.scheduler import Scheduler @@ -127,7 +128,8 @@ async def lifespan(app: FastAPI): # Base config.yaml + dynamically-added models (overlay), merged into one view. config = build_merged_config(config_path) - launchers = [VllmLauncher(), SglangLauncher(), LlamacppLauncher(), EmbeddingLauncher()] + launchers = [VllmLauncher(), SglangLauncher(), LlamacppLauncher(), TrtllmLauncher(), + EmbeddingLauncher()] registry = build_registry(config, config_path, launchers) # `or` (not get's default) so an env var set-but-empty (as the compose env # passes it) still falls back instead of yielding "" — an empty router_url diff --git a/apps/backend/tests/unit/test_launchers.py b/apps/backend/tests/unit/test_launchers.py index c9a41a5..e51ae4b 100644 --- a/apps/backend/tests/unit/test_launchers.py +++ b/apps/backend/tests/unit/test_launchers.py @@ -4,11 +4,12 @@ from app.llmops.launchers import (CAP_KV_TRANSFER, CAP_LORA_MODULES, CAP_METRICS_LLAMACPP, CAP_METRICS_SGLANG, - CAP_RUNTIME_LORA, CAP_SLEEP, EMBEDDING_KEY, - ENGINE_DEFAULT, EmbeddingLauncher, LlamacppLauncher, - SglangLauncher, VllmLauncher, _write_effective_config, + CAP_METRICS_TRTLLM, CAP_RUNTIME_LORA, CAP_SLEEP, + EMBEDDING_KEY, ENGINE_DEFAULT, EmbeddingLauncher, + LlamacppLauncher, SglangLauncher, TrtllmLauncher, + VllmLauncher, _write_effective_config, build_llamacpp_cli_args, build_sglang_cli_args, - build_vllm_cli_args) + build_trtllm_cli_args, build_vllm_cli_args) from app.llmops.state import ModelKind from schema import RootConfig from tests.conftest import FAKE_CONFIG @@ -620,3 +621,81 @@ def test_llamacpp_bind_host_env_overrides_only_the_bind_address(monkeypatch): assert spec.command[spec.command.index("--host") + 1] == "0.0.0.0" # binds all assert spec.host == "localhost" # record unchanged assert spec.probe_url == "http://localhost:8100/health" # local probe unchanged + + +# ---- TensorRT-LLM launcher (docs/trtllm-launcher-impl-design_zh-TW.md) -------- + +def _trtllm_config(extra: dict | None = None) -> RootConfig: + mc = {"model_tag": "Qwen/Qwen3-0.6B", "engine": "trtllm", + "engine_dir": "/engines/qwen3-06b_tp1_fp16"} + if extra: + mc.update(extra) + return RootConfig.model_validate({ + "server": {"host": "0.0.0.0", "port": 8887}, + "LLM_engines": {"T": { + "instances": [{"id": "a", "host": "localhost", "port": 8050, "cuda_device": 0}], + "model_config": mc, + }}, + }) + + +def test_trtllm_args_engine_dir_positional_and_tensorrt_backend(): + args = build_trtllm_cli_args( + {"model_tag": "Qwen/Qwen3-0.6B", "engine_dir": "/engines/q3", + "port": 8050, "host": "0.0.0.0"}, + perf_yaml_path="/logs/T__a.trtllm-perf.yaml") + assert args[:4] == ["serve", "/engines/q3", "--backend", "tensorrt"] + # tokenizer defaults to model_tag; served name defaults to model_tag + assert args[args.index("--tokenizer") + 1] == "Qwen/Qwen3-0.6B" + assert args[args.index("--served_model_name") + 1] == "Qwen/Qwen3-0.6B" + assert args[args.index("--host") + 1] == "0.0.0.0" + assert args[args.index("--port") + 1] == "8050" + # perf YAML always passed (else no /prometheus/metrics) + assert args[args.index("--extra_llm_api_options") + 1] == "/logs/T__a.trtllm-perf.yaml" + + +def test_trtllm_args_param_map_and_passthrough_and_drops(): + args = build_trtllm_cli_args( + {"engine_dir": "/e", "model_tag": "m", + "max_model_len": 2048, "gpu_memory_utilization": 0.4, + "tensor_parallel_size": 2, "max_batch_size": 4, + "n_gpu_layers": 99, "enforce_eager": True, "dtype": "float16", # all dropped + "some_unknown_flag": "x"}, # dropped (strict CLI) + perf_yaml_path="/p.yaml") + assert args[args.index("--max_seq_len") + 1] == "2048" + assert args[args.index("--kv_cache_free_gpu_memory_fraction") + 1] == "0.4" + assert args[args.index("--tp_size") + 1] == "2" + assert args[args.index("--max_batch_size") + 1] == "4" + for dropped in ("--n_gpu_layers", "--enforce_eager", "--dtype", "--some_unknown_flag", + "--n-gpu-layers"): + assert dropped not in args + + +def test_trtllm_args_requires_engine_dir(): + with pytest.raises(ValueError, match="engine_dir"): + build_trtllm_cli_args({"model_tag": "m"}, perf_yaml_path="/p.yaml") + + +def test_trtllm_build_spec_command_and_capabilities(tmp_path, monkeypatch): + import app.llmops.launchers as L + monkeypatch.setattr(L, "LOG_DIR", str(tmp_path)) + spec = TrtllmLauncher().build_spec( + _trtllm_config({"served_model_name": "qwen3-06b-trt", "max_model_len": 2048}), + "config.yaml", "T::a") + assert spec.command[0] == "trtllm-serve" + assert spec.command[1:3] == ["serve", "/engines/qwen3-06b_tp1_fp16"] + assert "--backend" in spec.command and "tensorrt" in spec.command + assert spec.engine == "trtllm" + assert spec.served_name == "qwen3-06b-trt" + assert CAP_METRICS_TRTLLM in spec.capabilities + assert CAP_SLEEP not in spec.capabilities and CAP_RUNTIME_LORA not in spec.capabilities + assert spec.env.get("CUDA_VISIBLE_DEVICES") == "0" + # perf YAML was written and referenced + perf = spec.command[spec.command.index("--extra_llm_api_options") + 1] + with open(perf) as f: + assert "return_perf_metrics: true" in f.read() + + +def test_trtllm_launcher_keys_filters_by_engine(): + keys = TrtllmLauncher().keys(_trtllm_config()) + assert keys == ["T::a"] diff --git a/apps/router-server/src/llm_router/vllm_metrics_client.py b/apps/router-server/src/llm_router/vllm_metrics_client.py index 3d982fa..d231513 100644 --- a/apps/router-server/src/llm_router/vllm_metrics_client.py +++ b/apps/router-server/src/llm_router/vllm_metrics_client.py @@ -85,7 +85,26 @@ def to_dict(self): "prompt_tokens": "llamacpp:prompt_tokens_total", "generation_tokens": "llamacpp:tokens_predicted_total", }, + "trtllm": { + # TensorRT-LLM exposes trtllm_* Prometheus text at /prometheus/metrics (NOT + # /metrics — see METRIC_PATH_BY_ENGINE), and only when the server was launched + # with return_perf_metrics: true. Names verified live on the tensorrt engine + # backend (v1.3.0rc20). See docs/trtllm-conversion-validation_zh-TW.md §A. + "running": "trtllm_num_requests_running", + "waiting": "trtllm_num_requests_waiting", + "kv_cache_usage_perc": "trtllm_kv_cache_utilization", + "prompt_tokens": "trtllm_prompt_tokens_total", + "generation_tokens": "trtllm_generation_tokens_total", + }, +} + +# Most engines serve their metrics at /metrics. TensorRT-LLM serves Prometheus +# text at /prometheus/metrics instead (its /metrics is a different JSON endpoint). +# fetch() looks the path up here per engine. See docs/trtllm-launcher-impl-design §6.2. +METRIC_PATH_BY_ENGINE = { + "trtllm": "/prometheus/metrics", } +DEFAULT_METRIC_PATH = "/metrics" # Engines that expose vLLM's /sleep + /wake_up (level-1 warm standby that frees VRAM). @@ -125,7 +144,8 @@ def __init__(self, http_client: httpx.AsyncClient, timeout: float = 2.0) -> None self.timeout = timeout async def fetch(self, base_url: str, engine: str = "vllm") -> Optional[VLLMInstanceMetrics]: - metrics_url = base_url.rstrip("/") + "/metrics" + path = METRIC_PATH_BY_ENGINE.get(engine, DEFAULT_METRIC_PATH) + metrics_url = base_url.rstrip("/") + path resp = await self.http_client.get(metrics_url, timeout=self.timeout) resp.raise_for_status() diff --git a/apps/router-server/tests/unit/test_vllm_metrics_client.py b/apps/router-server/tests/unit/test_vllm_metrics_client.py index 681fade..e22ff3f 100644 --- a/apps/router-server/tests/unit/test_vllm_metrics_client.py +++ b/apps/router-server/tests/unit/test_vllm_metrics_client.py @@ -143,13 +143,43 @@ def test_unknown_metric_engines_flags_missing_table(): config = {"LLM_engines": { "a": {"model_config": {"engine": "vllm"}}, "b": {"model_config": {"engine": "sglang"}}, - "c": {"model_config": {"engine": "trtllm"}}, # no metric table + "c": {"model_config": {"engine": "mystery"}}, # no metric table "d": {"model_config": {}}, # defaults vllm }} - assert unknown_metric_engines(config) == {"trtllm"} + assert unknown_metric_engines(config) == {"mystery"} def test_unknown_metric_engines_empty_when_all_known(): from src.llm_router.vllm_metrics_client import unknown_metric_engines config = {"LLM_engines": {"a": {"model_config": {"engine": "llamacpp"}}}} assert unknown_metric_engines(config) == set() + + +class _PathCapturingClient: + def __init__(self, text): + self._text = text + self.url = None + async def get(self, url, timeout=None): + self.url = url + return _FakeResp(self._text) + + +async def test_trtllm_scrapes_prometheus_metrics_path(): + # trtllm serves Prometheus text at /prometheus/metrics, not /metrics. + text = ("trtllm_num_requests_running{model_name=\"m\"} 2\n" + "trtllm_num_requests_waiting{model_name=\"m\"} 1\n" + "trtllm_kv_cache_utilization{model_name=\"m\"} 0.3\n" + "trtllm_prompt_tokens_total{model_name=\"m\"} 8\n" + "trtllm_generation_tokens_total{model_name=\"m\"} 64\n") + client = VLLMMetricsClient(http_client=_PathCapturingClient(text)) + m = await client.fetch("http://localhost:8050", engine="trtllm") + assert client.http_client.url == "http://localhost:8050/prometheus/metrics" + assert m.running == 2.0 and m.waiting == 1.0 + assert m.kv_cache_usage_perc == 0.3 + assert m.prompt_tokens == 8.0 and m.generation_tokens == 64.0 + + +async def test_default_engine_still_uses_metrics_path(): + client = VLLMMetricsClient(http_client=_PathCapturingClient("vllm:num_requests_running 0\n")) + await client.fetch("http://localhost:8002", engine="vllm") + assert client.http_client.url == "http://localhost:8002/metrics" diff --git a/deploy/docker-compose.mixed.yaml b/deploy/docker-compose.mixed.yaml index 503747d..236a3df 100644 --- a/deploy/docker-compose.mixed.yaml +++ b/deploy/docker-compose.mixed.yaml @@ -151,6 +151,58 @@ services: deploy: { resources: { reservations: { devices: [{ driver: nvidia, capabilities: [gpu] }] } } } restart: unless-stopped + # Fourth engine backend: TensorRT-LLM. FROM the TRT-LLM release image so it can spawn + # `trtllm-serve --backend tensorrt` over a PRE-BUILT engine directory (Phase 1 + # bring-your-own-engine — mount the engine dir at /engines). Declares + # LLMOPS_NODE_ENGINES=trtllm. `command: bash -lc "uvicorn …"` is REQUIRED: the base + # sets TensorRT's LD_LIBRARY_PATH only via a login-shell profile, so the backend must + # run through a login shell for the trtllm-serve subprocesses it spawns to import + # tensorrt_llm. Needs ipc:host + memlock/stack ulimits (TRT-LLM guidance). + # See docs/trtllm-launcher-impl-design_zh-TW.md. + trtllm-backend: + profiles: ["trtllm"] + build: { context: .., dockerfile: deploy/engine-trtllm.Dockerfile } + image: llmops-engine-trtllm:latest + container_name: mixed-trtllm-backend + working_dir: /app/apps/backend + command: bash -lc "uvicorn main:app --host 0.0.0.0 --port 5000" + env_file: .env + depends_on: + postgres: { condition: service_healthy } + environment: + - LLM_ROUTER_SERVER_CONFIG_PATH=/app/packages/config-schema/config.yaml + - LLMOPS_DB_URL=postgresql://llmops:llmops@postgres:5432/llmops + - LLMOPS_DB_PATH=/app/data/llmops.db + - LLMOPS_OVERLAY_PATH=/app/data/dynamic_models.json + - LLMOPS_INSTANCE_ID=trtllm-node + - LLMOPS_NODE_HOST=mixed-trtllm-backend + - LLMOPS_NODE_ENGINES=trtllm + - LLMOPS_NODE_API_URL=http://mixed-trtllm-backend:5000 + - LLMOPS_VLLM_BIND_HOST=0.0.0.0 + - LLMOPS_ROUTER_URL=http://router:8887 + - LLMOPS_PROMETHEUS_SD_PATH=/sd/targets-trtllm.json + - HF_HOME=/hf + - MODELSCOPE_CACHE=/modelscope/hub + - LLMOPS_LORA_DIR=/lora + - NVIDIA_VISIBLE_DEVICES=${NVIDIA_VISIBLE_DEVICES:-all} + volumes: + - ../packages/config-schema/config.yaml:/app/packages/config-schema/config.yaml + - mixed-trtllm-data:/app/data + - mixed-sd:/sd # file_sd scrape targets (shared with Prometheus) + # Pre-built TRT engine dirs mounted at /engines (Phase 1 bring-your-own; Phase 2 + # can also write build artifacts here). Point TRTLLM_ENGINES_DIR at your engines. + - ${TRTLLM_ENGINES_DIR:-${HOME}/trtllm-test/engines}:/engines + - ${HF_CACHE_DIR:-${HOME}/.cache/huggingface}:/hf + - ${MODELSCOPE_CACHE_DIR:-${HOME}/.cache/modelscope}:/modelscope + - ${LORA_ADAPTERS_DIR:-${HOME}/.cache/lora_adapters}:/lora + - /etc/localtime:/etc/localtime:ro + ports: ["${MIXED_TRTLLM_PORT:-5074}:5000"] + ipc: host + shm_size: "16gb" + ulimits: { memlock: -1, stack: 67108864 } + deploy: { resources: { reservations: { devices: [{ driver: nvidia, capabilities: [gpu] }] } } } + restart: unless-stopped + router: image: llmops-engine:latest container_name: mixed-router @@ -264,6 +316,7 @@ volumes: mixed-vllm-data: mixed-sglang-data: mixed-llamacpp-data: + mixed-trtllm-data: mixed-grafana-data: mixed-sd: # shared file_sd scrape-target dir (backends write, Prometheus reads) mixed-kv-cache: # cross-instance KV-cache root diff --git a/deploy/engine-trtllm.Dockerfile b/deploy/engine-trtllm.Dockerfile new file mode 100644 index 0000000..6d7eb81 --- /dev/null +++ b/deploy/engine-trtllm.Dockerfile @@ -0,0 +1,48 @@ +# syntax=docker/dockerfile:1 +# +# TensorRT-LLM variant of the engine image (see engine.Dockerfile for vLLM, +# engine-sglang.Dockerfile for SGLang, engine-llamacpp.Dockerfile for llama.cpp). +# Multi-backend design: each inference engine gets its own backend image, built +# FROM that engine's official base, because the launcher spawns the engine as a +# subprocess *inside this container* (apps/backend/app/llmops/process.py). So +# "which engines a backend can launch" == "what's installed in its image". Here the +# base provides `trtllm-serve`. See docs/trtllm-launcher-impl-design_zh-TW.md. +# +# Phase 1 "bring-your-own-engine": serves a PRE-BUILT TRT engine directory with +# `trtllm-serve serve --backend tensorrt`. (Phase 2 auto-build via a +# prepare() hook is a later step.) +# +# NOTE (LD_LIBRARY_PATH): the base sets TensorRT's lib path (/usr/local/tensorrt/lib +# …) only via a *login-shell* profile, not Docker ENV. The backend spawns trtllm-serve +# with the inherited env (apps/backend/app/llmops/process.py runs the command directly, +# not a login shell), so the compose service must launch the backend through a login +# shell — `command: bash -lc "uvicorn …"` — so trtllm-serve inherits the right +# LD_LIBRARY_PATH. Otherwise `import tensorrt_llm` fails on libnvonnxparser.so.10. +FROM nvcr.io/nvidia/tensorrt-llm/release:1.3.0rc20 + +WORKDIR /app + +COPY apps/backend/requirements.txt /tmp/backend-req.txt +COPY apps/router-server/requirements.txt /tmp/router-req.txt +# This image only launches TensorRT-LLM: +# - vllm / sglang: other engines, not installed here (vllm is multi-GB). +# - bitsandbytes: vLLM on-the-fly quant; TRT-LLM quantizes at engine-build time. +# - pytest*: dev-only. +# tensorrt_llm itself is already in the base image. +RUN sed -i -E '/^(vllm|sglang|bitsandbytes.*|pytest.*)$/d' /tmp/router-req.txt /tmp/backend-req.txt \ + && pip install --no-cache-dir -r /tmp/backend-req.txt -r /tmp/router-req.txt + +# App code + shared packages — same layout as the other engine images so the in-code +# sys.path bootstrap and default config/overlay/db paths resolve to /app. +COPY apps/backend ./apps/backend +COPY apps/router-server ./apps/router-server +COPY packages ./packages + +# The TRT-LLM base sets its own ENTRYPOINT; clear it so the compose `command:` +# (bash -lc "uvicorn …") runs verbatim. +ENTRYPOINT [] +CMD ["bash"] + +# The base ships no healthcheck relevant to the control-plane backend (uvicorn :5000); +# match the other engine images (no healthcheck). +HEALTHCHECK NONE diff --git a/packages/config-schema/schema.py b/packages/config-schema/schema.py index 81eba0a..65d6065 100644 --- a/packages/config-schema/schema.py +++ b/packages/config-schema/schema.py @@ -63,7 +63,7 @@ class EngineModelConfig(BaseModel): # phantom engine: schema-valid but silently unrunnable. The backend cross-checks # this on load (validate_registered_engines) and fails loud rather than dropping # the group silently. Keep this list == the registered engine names. - engine: Literal["vllm", "sglang", "llamacpp"] = "vllm" + engine: Literal["vllm", "sglang", "llamacpp", "trtllm"] = "vllm" # Router-facing endpoint kind (NOT an engine flag — the launcher skips it): # chat -> /v1/chat/completions + /v1/completions (a generate model) # embed -> /v1/embeddings (a vLLM pooling embedding model) @@ -76,6 +76,13 @@ class EngineModelConfig(BaseModel): max_model_len: Optional[int] = None gpu_memory_utilization: Optional[float] = None tensor_parallel_size: int = 1 + # TensorRT-LLM (engine: trtllm) only, Phase 1 "bring-your-own-engine": the path to + # a pre-built TRT engine directory (inside the container) that trtllm-serve serves + # with --backend tensorrt. `tokenizer` overrides the tokenizer source (a HF repo or + # dir); defaults to model_tag. Both ride the trtllm launcher; other engines ignore + # them. See docs/trtllm-launcher-impl-design_zh-TW.md. + engine_dir: Optional[str] = None + tokenizer: Optional[str] = None # LoRA: `enable_lora`/`max_lora_rank`/… flow through extra="allow" as plain # vLLM flags; `lora_modules` is typed so the dashboard can render + manage the # adapters and the launcher can emit the multi-arg `--lora-modules` form. From 04f43ffa6586f0d7cfeccd5ec3742122c4954fa1 Mon Sep 17 00:00:00 2001 From: max Date: Sat, 4 Jul 2026 20:29:05 +0800 Subject: [PATCH 05/20] fix(trtllm): free the port before (re)start so a crashed engine recovers cleanly MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit TensorRT-LLM's executor workers reparent to init in their own process group/session, so they escape terminate_process_group; and a SIGKILL/crash of trtllm-serve leaves in-flight connections (health probes, metric scrapes) in kernel FIN_WAIT2/TIME_WAIT holding the port — which SO_REUSEADDR does not override. The auto-restart then failed with EADDRINUSE and burned the whole restart budget (3/3) in a crash-loop. Add free_port(): before spawning, SIGKILL any process still holding the port (via /proc socket-inode lookup) and then wait for the port to become bindable (kernel teardown ~60s). Called in manager.start() before spawn_process — a no-op that returns immediately when the port is free (the graceful-stop common case), so only a crash/kill restart pays the wait. This turns the crash-loop into a single patient restart (verified: recovers in ~90s, one "attempt 1/3", zero EADDRINUSE). Also make test_ha_safety's manager use vram_guard=False so start()'s _vram_preflight doesn't hit the real GPU and flake when the host GPU is busy. Co-Authored-By: Claude Opus 4.8 --- apps/backend/app/llmops/manager.py | 9 ++- apps/backend/app/llmops/process.py | 91 +++++++++++++++++++++++ apps/backend/tests/unit/test_ha_safety.py | 5 +- apps/backend/tests/unit/test_process.py | 30 ++++++++ 4 files changed, 133 insertions(+), 2 deletions(-) diff --git a/apps/backend/app/llmops/manager.py b/apps/backend/app/llmops/manager.py index d494bff..e5f2303 100644 --- a/apps/backend/app/llmops/manager.py +++ b/apps/backend/app/llmops/manager.py @@ -17,7 +17,7 @@ from app.llmops.events import emit_transition from app.llmops.instance import ModelInstance from app.llmops.launchers import CAP_RUNTIME_LORA, CAP_SLEEP, Launcher -from app.llmops.process import spawn_process, terminate_process_group +from app.llmops.process import free_port, spawn_process, terminate_process_group from app.llmops.registry import ModelRegistry from app.llmops.state import Desired, ModelKind, ModelState @@ -664,6 +664,13 @@ async def start(self, key: str, force: bool = False, reset_restart: bool = True) # Spawn outside the lock — Popen returns immediately but still does IO. loop = asyncio.get_event_loop() + # Clear a teardown race on the port: a just-killed engine's detached worker + # processes (e.g. TensorRT-LLM's executor) may still hold the socket, which + # would make this spawn fail with EADDRINUSE and burn a restart. free_port + # kills any lingering holder. No-op when the port is already free (common case). + if not await loop.run_in_executor(None, free_port, spec.port): + logger.warning("port %s still in use after free_port; starting %s anyway", + spec.port, key) try: proc = await loop.run_in_executor(None, spawn_process, spec) except Exception as e: diff --git a/apps/backend/app/llmops/process.py b/apps/backend/app/llmops/process.py index 9650517..dddf496 100644 --- a/apps/backend/app/llmops/process.py +++ b/apps/backend/app/llmops/process.py @@ -109,6 +109,97 @@ def terminate_process_group(proc: subprocess.Popen, timeout: float = 10.0) -> No pass +def port_is_free(port: int) -> bool: + """Whether 0.0.0.0:`port` can be bound right now (with SO_REUSEADDR, matching how + the engine servers bind). False if a live process still holds a listening socket.""" + import socket + + s = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + try: + s.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + s.bind(("0.0.0.0", int(port))) + return True + except OSError: + return False + finally: + s.close() + + +def _pids_holding_port(port: int) -> set[int]: + """PIDs with an open socket bound to `port` (any TCP state), via /proc. Same PID + namespace as the caller (the backend + the engine subprocesses it spawns).""" + import glob + + inodes: set[str] = set() + hexport = f"{int(port):04X}" + for f in ("/proc/net/tcp", "/proc/net/tcp6"): + try: + lines = open(f).read().splitlines()[1:] + except OSError: + continue + for ln in lines: + p = ln.split() + if len(p) < 10: + continue + # p[1] = local_address "HEXIP:HEXPORT"; p[9] = inode + if p[1].rsplit(":", 1)[-1].upper() == hexport: + inodes.add(p[9]) + if not inodes: + return set() + pids: set[int] = set() + for fd in glob.glob("/proc/[0-9]*/fd/*"): + try: + tgt = os.readlink(fd) + except OSError: + continue + if tgt.startswith("socket:[") and tgt[8:-1] in inodes: + try: + pids.add(int(fd.split("/", 3)[2])) + except (ValueError, IndexError): + pass + return pids + + +def free_port(port: int, timeout: float = 150.0) -> bool: + """Make `port` bindable, then return whether it is. Blocking; run in an executor. + + Two things can hold a just-killed engine's port: + 1. a *detached* worker process still holding the listening socket via an inherited + fd — notably TensorRT-LLM, whose executor workers reparent to init in their own + process group/session and so escape the process-group kill. We find these via + /proc and SIGKILL them. + 2. kernel connection-teardown states (FIN_WAIT2/TIME_WAIT) left by in-flight + connections (health probes, metric scrapes) when the server was SIGKILLed + rather than closed gracefully. These have no owning process and SO_REUSEADDR + does not override them, so the only remedy is to wait out the kernel timeout + (~60s). We poll until bindable. + Without this a crash/kill-triggered restart fails with EADDRINUSE and burns the + restart budget in a loop. This converts it into a single patient wait (safe: + start_timeout is far larger). No-op — returns True immediately — when the port is + already free, which is the graceful-stop common case.""" + import time + + if port_is_free(port): + return True + deadline = time.monotonic() + max(0.0, timeout) + logged = False + while True: + for pid in _pids_holding_port(port): + try: + os.kill(pid, signal.SIGKILL) + logger.warning("free_port %s: killed lingering holder pid %s", port, pid) + except (ProcessLookupError, PermissionError): + pass + if port_is_free(port): + return True + if time.monotonic() >= deadline: + return port_is_free(port) + if not logged: + logger.info("free_port %s: held (kernel socket teardown); waiting…", port) + logged = True + time.sleep(1.0) + + def kill_process_group(proc: subprocess.Popen) -> None: """SIGKILL the whole process group immediately, no graceful grace period. diff --git a/apps/backend/tests/unit/test_ha_safety.py b/apps/backend/tests/unit/test_ha_safety.py index c5c19ef..919a5d0 100644 --- a/apps/backend/tests/unit/test_ha_safety.py +++ b/apps/backend/tests/unit/test_ha_safety.py @@ -71,8 +71,11 @@ def _manager(tmp_path, store=None, node_engines=None, instance_id="node-A"): config = load_config(str(cfg_path)) launchers = [VllmLauncher(), EmbeddingLauncher()] registry = build_registry(config, str(cfg_path), launchers) + # vram_guard=False so start()'s _vram_preflight doesn't hit the real GPU and raise + # VRAMInsufficient when the host GPU is busy — these tests assert control-flow, not + # placement. settings = BackendSettings(instance_id=instance_id, - node_engines=node_engines or []) + node_engines=node_engines or [], vram_guard=False) mgr = ModelManager( registry, launchers, None, config, str(cfg_path), settings, store=store, overlay_path=str(overlay_path), diff --git a/apps/backend/tests/unit/test_process.py b/apps/backend/tests/unit/test_process.py index ea5fb45..5b4f028 100644 --- a/apps/backend/tests/unit/test_process.py +++ b/apps/backend/tests/unit/test_process.py @@ -36,3 +36,33 @@ def test_terminate_process_group_graceful(): # SIGTERM ends `sleep` well under the grace period. assert proc.poll() is not None assert time.perf_counter() - start < 5.0 + + +def test_port_is_free_detects_live_listener(): + import socket + from app.llmops.process import port_is_free + s = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + s.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + s.bind(("0.0.0.0", 0)); s.listen(1) + port = s.getsockname()[1] + assert port_is_free(port) is False + s.close() + assert port_is_free(port) is True + + +def test_free_port_kills_the_holder(): + import socket + from app.llmops.process import free_port, port_is_free + probe = socket.socket(); probe.bind(("0.0.0.0", 0)); port = probe.getsockname()[1]; probe.close() + child = subprocess.Popen([sys.executable, "-c", + "import socket,time;s=socket.socket();s.setsockopt(socket.SOL_SOCKET,socket.SO_REUSEADDR,1);" + f"s.bind(('0.0.0.0',{port}));s.listen(1);time.sleep(60)"]) + try: + for _ in range(40): + if not port_is_free(port): break + time.sleep(0.05) + assert port_is_free(port) is False + assert free_port(port, timeout=10.0) is True + assert child.poll() is not None + finally: + if child.poll() is None: child.kill() From b4a4e749db173e794d19520051cc090972d10fd3 Mon Sep 17 00:00:00 2001 From: max Date: Sat, 4 Jul 2026 20:29:44 +0800 Subject: [PATCH 06/20] =?UTF-8?q?docs(trtllm):=20document=20the=20crash/ki?= =?UTF-8?q?ll=20port-teardown=20restart=20fix=20(=C2=A713b)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Opus 4.8 --- docs/trtllm-launcher-impl-design_zh-TW.md | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/docs/trtllm-launcher-impl-design_zh-TW.md b/docs/trtllm-launcher-impl-design_zh-TW.md index bb824db..2084d4e 100644 --- a/docs/trtllm-launcher-impl-design_zh-TW.md +++ b/docs/trtllm-launcher-impl-design_zh-TW.md @@ -366,6 +366,22 @@ Add Model 選單、capability 欄位、遮蔽鍵全自動。選配:`ModelsView`/ --- +## 13b. 重啟/崩潰的 port 釋放(實作時踩到、已修) + +trtllm 有一個 vLLM/SGLang/llama.cpp 沒有的行為:**`trtllm-serve` 的 executor worker 會 reparent 到 init、 +自成 process group/session**,所以 `terminate_process_group`(killpg)殺不到它們;而且**被 SIGKILL/崩潰**時, +in-flight 連線(health 探測、metric scrape)會在 kernel 留下 `FIN_WAIT2`/`TIME_WAIT`,**連 SO_REUSEADDR 都無法 +覆蓋**地佔住 port ~60s。結果:auto-restart 一直 `EADDRINUSE`,把 restart budget(3 次)在幾秒內燒光 → 模型卡死。 + +**修法**([process.py](../apps/backend/app/llmops/process.py) `free_port`,[manager.py](../apps/backend/app/llmops/manager.py) `start`): +spawn 前先 `free_port(port)`——(1) 用 /proc socket-inode 找出仍持有 port 的 process 並 SIGKILL;(2) 再輪詢 +等到 port 可 bind(kernel teardown ~60s)。**port 本來就空時立即回傳(graceful stop 的常態,零成本)**,只有 +崩潰/被殺的重啟才付這個等待。效果:crash-loop 變成**單一次耐心重啟**(實測 ~90s 回到 READY、只用一次 +`attempt 1/3`、零 EADDRINUSE)。graceful stop/start 仍是 ~2s,不受影響。 + +> 注:這是 trtllm 特有(其他引擎 worker 在同 process group,killpg 直接收乾淨、port 立即釋放)。`free_port` +> 對其他引擎是 no-op。 + ## 14. 契約缺口 / 風險 / open questions - **🔴 TensorRT backend 淘汰**:`v1.3.0rc20` 是**最後一個**支援 `--backend tensorrt` 的版本,下一版移除。→ 要嘛 From 7543d3accf9e451e08ed546be3fa6ef74bfbba5e Mon Sep 17 00:00:00 2001 From: max Date: Sat, 4 Jul 2026 20:50:59 +0800 Subject: [PATCH 07/20] feat(trtllm): scrape /prometheus/metrics + Grafana dashboard embedded in Monitoring TensorRT-LLM metrics were never reaching Prometheus: the `engines` scrape job hit /metrics uniformly, but trtllm-serve exposes its Prometheus text at /prometheus/metrics (/metrics returns the router SD JSON, which fails to parse), so the target stayed down and no trtllm_* series existed. - prometheus.mixed.yml: relabel __metrics_path__ -> /prometheus/metrics for engine=trtllm only (others keep /metrics). Mirrors the router's METRIC_PATH_BY_ENGINE. Verified: target up, 58 trtllm_* series with correct engine/instance labels. - Add deploy/grafana/dashboards/trtllm/trtllm-dashboard.json (13 panels: running/waiting, KV utilization/hit-rate, prompt/gen throughput, TTFT/TPOT/E2E p95, completed-rate, context/generation batch, GPU memory). All PromQL validated against live traffic. - Embed it in the Monitoring view as a new "TensorRT-LLM" tab (MonitoringView.vue + en/zh-TW i18n), matching the other per-engine dashboards. Co-Authored-By: Claude Opus 4.8 --- apps/frontend_llmops/src/i18n/locales/en.ts | 1 + .../frontend_llmops/src/i18n/locales/zh-TW.ts | 1 + .../src/views/MonitoringView.vue | 5 +- .../dashboards/trtllm/trtllm-dashboard.json | 729 ++++++++++++++++++ deploy/prometheus.mixed.yml | 9 + 5 files changed, 744 insertions(+), 1 deletion(-) create mode 100644 deploy/grafana/dashboards/trtllm/trtllm-dashboard.json diff --git a/apps/frontend_llmops/src/i18n/locales/en.ts b/apps/frontend_llmops/src/i18n/locales/en.ts index b62e4d6..d4a60f7 100644 --- a/apps/frontend_llmops/src/i18n/locales/en.ts +++ b/apps/frontend_llmops/src/i18n/locales/en.ts @@ -214,6 +214,7 @@ export default { sglang: 'SGLang', sglangDashboard: 'Dashboard', llamacppDashboard: 'Dashboard', + trtllmDashboard: 'Dashboard', shared: 'Shared', gpu: 'GPU', host: 'Host', diff --git a/apps/frontend_llmops/src/i18n/locales/zh-TW.ts b/apps/frontend_llmops/src/i18n/locales/zh-TW.ts index 7f0e0ad..6b855db 100644 --- a/apps/frontend_llmops/src/i18n/locales/zh-TW.ts +++ b/apps/frontend_llmops/src/i18n/locales/zh-TW.ts @@ -206,6 +206,7 @@ export default { sglang: 'SGLang', sglangDashboard: '儀表板', llamacppDashboard: '儀表板', + trtllmDashboard: '儀表板', shared: '共用', gpu: 'GPU', host: '主機', diff --git a/apps/frontend_llmops/src/views/MonitoringView.vue b/apps/frontend_llmops/src/views/MonitoringView.vue index a106df0..8010617 100644 --- a/apps/frontend_llmops/src/views/MonitoringView.vue +++ b/apps/frontend_llmops/src/views/MonitoringView.vue @@ -10,7 +10,7 @@ import { useModelsStore } from '@/stores/models' // from the provisioned dashboards (deploy/grafana/dashboards). const BASE = '/grafana/d' -type DashboardId = 'overview' | 'autoscaling' | 'capacity' | 'perf' | 'query' | 'sglang' | 'llamacpp' | 'gpu' | 'host' +type DashboardId = 'overview' | 'autoscaling' | 'capacity' | 'perf' | 'query' | 'sglang' | 'llamacpp' | 'trtllm' | 'gpu' | 'host' const ranges = [ { label: '15m', from: 'now-15m' }, @@ -34,6 +34,8 @@ const dashboards = computed( { id: 'sglang', group: 'sglang', label: t('monitoring.sglangDashboard'), icon: Boxes, path: `${BASE}/sglang-dashboard/sglang-dashboard` }, // llama.cpp (llamacpp:* metrics — no KV-usage panel, see the dashboard note) { id: 'llamacpp', group: 'llamacpp', label: t('monitoring.llamacppDashboard'), icon: Boxes, path: `${BASE}/llamacpp-dashboard/llama-cpp-dashboard` }, + // TensorRT-LLM (trtllm_* metrics scraped from /prometheus/metrics) + { id: 'trtllm', group: 'trtllm', label: t('monitoring.trtllmDashboard'), icon: Boxes, path: `${BASE}/trtllm-dashboard/tensorrt-llm-dashboard` }, // Shared / cross-engine + infra { id: 'autoscaling', group: 'shared', label: t('monitoring.autoscaling'), icon: Layers, path: `${BASE}/llmops-autoscaling/autoscaling` }, { id: 'gpu', group: 'shared', label: t('monitoring.gpu'), icon: Server, path: `${BASE}/Oxed_c6Wz/nvidia-dcgm-exporter-dashboard` }, @@ -47,6 +49,7 @@ const groupedTabs = computed(() => { key: 'vllm', label: 'vLLM' }, { key: 'sglang', label: 'SGLang' }, { key: 'llamacpp', label: 'llama.cpp' }, + { key: 'trtllm', label: 'TensorRT-LLM' }, { key: 'shared', label: t('monitoring.shared') }, ] as const ) diff --git a/deploy/grafana/dashboards/trtllm/trtllm-dashboard.json b/deploy/grafana/dashboards/trtllm/trtllm-dashboard.json new file mode 100644 index 0000000..c308818 --- /dev/null +++ b/deploy/grafana/dashboards/trtllm/trtllm-dashboard.json @@ -0,0 +1,729 @@ +{ + "annotations": { + "list": [ + { + "builtIn": 1, + "datasource": { + "type": "grafana", + "uid": "-- Grafana --" + }, + "enable": true, + "hide": true, + "type": "dashboard" + } + ] + }, + "editable": true, + "fiscalYearStartMonth": 0, + "graphTooltip": 1, + "id": null, + "links": [], + "panels": [ + { + "id": 1, + "type": "text", + "title": "", + "datasource": null, + "gridPos": { + "h": 3, + "w": 24, + "x": 0, + "y": 0 + }, + "options": { + "mode": "markdown", + "content": "### TensorRT-LLM fleet\nMetrics scraped from `trtllm-serve` at `/prometheus/metrics` (`trtllm_*`), exposed only when the server is launched with `return_perf_metrics: true` (perf YAML) on the **tensorrt** backend (v1.3.0rc20). **Note:** the tensorrt backend has no sleep/wake; sleep panels intentionally absent. See `docs/trtllm-launcher-impl-design_zh-TW.md`." + } + }, + { + "id": 2, + "type": "timeseries", + "title": "Requests running", + "description": "trtllm_num_requests_running \u2014 requests currently decoding.", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 3 + }, + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "lineInterpolation": "smooth", + "fillOpacity": 12, + "showPoints": "never", + "lineWidth": 2 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "editorMode": "code", + "expr": "trtllm_num_requests_running{engine=\"trtllm\", instance=~\"$instance\"}", + "legendFormat": "{{instance}}", + "range": true, + "refId": "A" + } + ] + }, + { + "id": 3, + "type": "timeseries", + "title": "Requests waiting (queued)", + "description": "trtllm_num_requests_waiting \u2014 requests queued for a slot (the autoscaler's queue signal).", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 3 + }, + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "lineInterpolation": "smooth", + "fillOpacity": 12, + "showPoints": "never", + "lineWidth": 2 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "editorMode": "code", + "expr": "trtllm_num_requests_waiting{engine=\"trtllm\", instance=~\"$instance\"}", + "legendFormat": "{{instance}}", + "range": true, + "refId": "A" + } + ] + }, + { + "id": 4, + "type": "timeseries", + "title": "KV cache utilization", + "description": "trtllm_kv_cache_utilization \u2014 fraction of the KV cache in use.", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 11 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit", + "custom": { + "drawStyle": "line", + "lineInterpolation": "smooth", + "fillOpacity": 12, + "showPoints": "never", + "lineWidth": 2 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "editorMode": "code", + "expr": "trtllm_kv_cache_utilization{engine=\"trtllm\", instance=~\"$instance\"}", + "legendFormat": "{{instance}}", + "range": true, + "refId": "A" + } + ] + }, + { + "id": 5, + "type": "timeseries", + "title": "KV cache hit rate", + "description": "trtllm_kv_cache_hit_rate \u2014 block-reuse hit rate.", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 11 + }, + "fieldConfig": { + "defaults": { + "unit": "percentunit", + "custom": { + "drawStyle": "line", + "lineInterpolation": "smooth", + "fillOpacity": 12, + "showPoints": "never", + "lineWidth": 2 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "editorMode": "code", + "expr": "trtllm_kv_cache_hit_rate{engine=\"trtllm\", instance=~\"$instance\"}", + "legendFormat": "{{instance}}", + "range": true, + "refId": "A" + } + ] + }, + { + "id": 6, + "type": "timeseries", + "title": "Prompt tokens rate", + "description": "rate(trtllm_prompt_tokens_total) \u2014 prompt (prefill) tokens processed per second.", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 19 + }, + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "lineInterpolation": "smooth", + "fillOpacity": 12, + "showPoints": "never", + "lineWidth": 2 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "editorMode": "code", + "expr": "rate(trtllm_prompt_tokens_total{engine=\"trtllm\", instance=~\"$instance\"}[$__rate_interval])", + "legendFormat": "{{instance}}", + "range": true, + "refId": "A" + } + ] + }, + { + "id": 7, + "type": "timeseries", + "title": "Generation tokens rate", + "description": "rate(trtllm_generation_tokens_total) \u2014 generated (decode) tokens per second.", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 19 + }, + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "lineInterpolation": "smooth", + "fillOpacity": 12, + "showPoints": "never", + "lineWidth": 2 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "editorMode": "code", + "expr": "rate(trtllm_generation_tokens_total{engine=\"trtllm\", instance=~\"$instance\"}[$__rate_interval])", + "legendFormat": "{{instance}}", + "range": true, + "refId": "A" + } + ] + }, + { + "id": 8, + "type": "timeseries", + "title": "Time to first token (p95)", + "description": "p95 of trtllm_time_to_first_token_seconds \u2014 prefill latency.", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 27 + }, + "fieldConfig": { + "defaults": { + "unit": "s", + "custom": { + "drawStyle": "line", + "lineInterpolation": "smooth", + "fillOpacity": 12, + "showPoints": "never", + "lineWidth": 2 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "editorMode": "code", + "expr": "histogram_quantile(0.95, sum by (le, instance) (rate(trtllm_time_to_first_token_seconds_bucket{engine=\"trtllm\", instance=~\"$instance\"}[$__rate_interval])))", + "legendFormat": "{{instance}}", + "range": true, + "refId": "A" + } + ] + }, + { + "id": 9, + "type": "timeseries", + "title": "Time per output token (p95)", + "description": "p95 of trtllm_time_per_output_token_seconds \u2014 inter-token decode latency.", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 27 + }, + "fieldConfig": { + "defaults": { + "unit": "s", + "custom": { + "drawStyle": "line", + "lineInterpolation": "smooth", + "fillOpacity": 12, + "showPoints": "never", + "lineWidth": 2 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "editorMode": "code", + "expr": "histogram_quantile(0.95, sum by (le, instance) (rate(trtllm_time_per_output_token_seconds_bucket{engine=\"trtllm\", instance=~\"$instance\"}[$__rate_interval])))", + "legendFormat": "{{instance}}", + "range": true, + "refId": "A" + } + ] + }, + { + "id": 10, + "type": "timeseries", + "title": "End-to-end request latency (p95)", + "description": "p95 of trtllm_e2e_request_latency_seconds \u2014 full request latency.", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 35 + }, + "fieldConfig": { + "defaults": { + "unit": "s", + "custom": { + "drawStyle": "line", + "lineInterpolation": "smooth", + "fillOpacity": 12, + "showPoints": "never", + "lineWidth": 2 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "editorMode": "code", + "expr": "histogram_quantile(0.95, sum by (le, instance) (rate(trtllm_e2e_request_latency_seconds_bucket{engine=\"trtllm\", instance=~\"$instance\"}[$__rate_interval])))", + "legendFormat": "{{instance}}", + "range": true, + "refId": "A" + } + ] + }, + { + "id": 11, + "type": "timeseries", + "title": "Completed requests rate", + "description": "rate(trtllm_num_requests_completed_total) \u2014 requests finished per second.", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 35 + }, + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "lineInterpolation": "smooth", + "fillOpacity": 12, + "showPoints": "never", + "lineWidth": 2 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "editorMode": "code", + "expr": "rate(trtllm_num_requests_completed_total{engine=\"trtllm\", instance=~\"$instance\"}[$__rate_interval])", + "legendFormat": "{{instance}}", + "range": true, + "refId": "A" + } + ] + }, + { + "id": 12, + "type": "timeseries", + "title": "Scheduled requests (context / generation)", + "description": "trtllm_num_context_requests (prefill) and trtllm_num_generation_requests (decode) in the active batch.", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 0, + "y": 43 + }, + "fieldConfig": { + "defaults": { + "unit": "short", + "custom": { + "drawStyle": "line", + "lineInterpolation": "smooth", + "fillOpacity": 12, + "showPoints": "never", + "lineWidth": 2 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "editorMode": "code", + "expr": "trtllm_num_context_requests{engine=\"trtllm\", instance=~\"$instance\"}", + "legendFormat": "{{instance}} context", + "range": true, + "refId": "A" + }, + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "editorMode": "code", + "expr": "trtllm_num_generation_requests{engine=\"trtllm\", instance=~\"$instance\"}", + "legendFormat": "{{instance}} generation", + "range": true, + "refId": "B" + } + ] + }, + { + "id": 13, + "type": "timeseries", + "title": "GPU memory usage", + "description": "trtllm_gpu_memory_usage_bytes \u2014 device memory held by the engine.", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "gridPos": { + "h": 8, + "w": 12, + "x": 12, + "y": 43 + }, + "fieldConfig": { + "defaults": { + "unit": "bytes", + "custom": { + "drawStyle": "line", + "lineInterpolation": "smooth", + "fillOpacity": 12, + "showPoints": "never", + "lineWidth": 2 + } + }, + "overrides": [] + }, + "options": { + "legend": { + "displayMode": "list", + "placement": "bottom", + "showLegend": true + }, + "tooltip": { + "mode": "multi", + "sort": "desc" + } + }, + "targets": [ + { + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "editorMode": "code", + "expr": "trtllm_gpu_memory_usage_bytes{engine=\"trtllm\", instance=~\"$instance\"}", + "legendFormat": "{{instance}}", + "range": true, + "refId": "A" + } + ] + } + ], + "preload": false, + "refresh": "5s", + "schemaVersion": 41, + "tags": [ + "trtllm", + "tensorrt-llm", + "llm" + ], + "templating": { + "list": [ + { + "name": "instance", + "type": "query", + "datasource": { + "type": "prometheus", + "uid": "prometheus" + }, + "query": { + "qryType": 1, + "query": "label_values(trtllm_num_requests_running, instance)", + "refId": "PrometheusVariableQueryEditor-VariableQuery" + }, + "refresh": 2, + "includeAll": true, + "multi": true, + "allValue": ".*", + "current": { + "text": "All", + "value": "$__all" + }, + "sort": 1 + } + ] + }, + "time": { + "from": "now-30m", + "to": "now" + }, + "timepicker": {}, + "timezone": "", + "title": "TensorRT-LLM Dashboard", + "uid": "trtllm-dashboard", + "version": 1 +} diff --git a/deploy/prometheus.mixed.yml b/deploy/prometheus.mixed.yml index 77ad4a5..af3d0bb 100644 --- a/deploy/prometheus.mixed.yml +++ b/deploy/prometheus.mixed.yml @@ -25,6 +25,15 @@ scrape_configs: - source_labels: [group, instance_id] separator: "::" target_label: instance + # Per-engine metrics path: vLLM/SGLang/llama.cpp expose Prometheus text at + # /metrics (the job default), but TensorRT-LLM's trtllm-serve exposes it at + # /prometheus/metrics (its /metrics returns the router SD JSON, which fails to + # parse). This rule only fires for engine=trtllm; others keep /metrics. Mirrors + # the router's METRIC_PATH_BY_ENGINE. See docs/trtllm-launcher-impl-design §6.2. + - source_labels: [engine] + regex: trtllm + target_label: __metrics_path__ + replacement: /prometheus/metrics # Autoscaling/control-plane metrics from both dashboard backends. - job_name: llmops-backend From b44210bdccc36837cd3a93e1da2c673cc458ef49 Mon Sep 17 00:00:00 2001 From: max Date: Sat, 4 Jul 2026 20:59:40 +0800 Subject: [PATCH 08/20] feat(trtllm): reject on missing runtime (available() hook) + Grafana alerts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Route-agnostic Phase-1 hardening (independent of the engine-vs-pytorch backend decision): - Launcher.available() optional hook. TrtllmLauncher.available() checks shutil.which("trtllm-serve"); manager.start() rejects with EngineUnavailable (409) up-front when absent, instead of a trtllm model on the collapsed vLLM image crash-looping at spawn (§8.4). Other launchers omit the hook (getattr fallback = always available). Verified live: trtllm-serve present in the trtllm image, MISSING in the vLLM image. - deploy/grafana/provisioning/alerting/trtllm.yaml: mirror the vLLM alert set for trtllm_* (target-down, TTFT p95, KV utilization, requests-waiting). Uses the mixed-stack label scheme {job="engines", engine="trtllm"}. Verified: all 4 rules load in Grafana. - Tests: available() reflects trtllm-serve presence; start() raises EngineUnavailable when the launcher reports the runtime missing. 498 passed. Co-Authored-By: Claude Opus 4.8 --- apps/backend/app/api/models.py | 3 +- apps/backend/app/llmops/launchers.py | 15 ++ apps/backend/app/llmops/manager.py | 14 ++ apps/backend/tests/unit/test_ha_safety.py | 15 +- apps/backend/tests/unit/test_launchers.py | 9 + .../grafana/provisioning/alerting/trtllm.yaml | 188 ++++++++++++++++++ 6 files changed, 241 insertions(+), 3 deletions(-) create mode 100644 deploy/grafana/provisioning/alerting/trtllm.yaml diff --git a/apps/backend/app/api/models.py b/apps/backend/app/api/models.py index 6424ee7..913e546 100644 --- a/apps/backend/app/api/models.py +++ b/apps/backend/app/api/models.py @@ -14,6 +14,7 @@ from app.api.schemas import ModelView from app.core.auth import require_operator from app.llmops.manager import ( + EngineUnavailable, GpuUnavailable, LoraRuntimeError, ModelAlreadyRunning, @@ -213,7 +214,7 @@ async def start_model(key: str, force: bool = False, manager: ModelManager = Dep raise HTTPException(status.HTTP_404_NOT_FOUND, f"unknown model: {key}") except ModelAlreadyRunning: raise HTTPException(status.HTTP_409_CONFLICT, f"model already running: {key}") - except (VRAMInsufficient, GpuUnavailable) as e: + except (VRAMInsufficient, GpuUnavailable, EngineUnavailable) as e: raise HTTPException(status.HTTP_409_CONFLICT, str(e)) diff --git a/apps/backend/app/llmops/launchers.py b/apps/backend/app/llmops/launchers.py index eb91a6c..2714d95 100644 --- a/apps/backend/app/llmops/launchers.py +++ b/apps/backend/app/llmops/launchers.py @@ -14,6 +14,7 @@ import json import logging import os +import shutil import sys import tempfile from typing import Protocol @@ -176,6 +177,14 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: """Resolve the LaunchSpec for one key.""" ... + def available(self) -> bool: + """Whether this engine's runtime is present in the running image. Optional + hook — callers treat a missing method as always-available, so only engines + that can be absent (TensorRT-LLM, whose trtllm-serve ships only in its own + image) need implement it. Lets the collapsed / non-trtllm backend reject a + trtllm model up-front with a clear error instead of crashing at spawn (§8.4).""" + ... + class VllmLauncher: kind = ModelKind.LLM @@ -744,6 +753,12 @@ class TrtllmLauncher: "--tokenizer Qwen/Qwen3-0.6B --served_model_name qwen3-06b-trt " "--host 0.0.0.0 --port 8050") + def available(self) -> bool: + # trtllm-serve ships only in the dedicated TensorRT-LLM image; on the + # collapsed vLLM image which(...) is None, so start() rejects trtllm models + # up-front instead of crash-looping at spawn. See §8.4. + return shutil.which("trtllm-serve") is not None + def keys(self, config) -> list[str]: out: list[str] = [] for model_tag, engine in config.LLM_engines.items(): diff --git a/apps/backend/app/llmops/manager.py b/apps/backend/app/llmops/manager.py index e5f2303..2ffc7ca 100644 --- a/apps/backend/app/llmops/manager.py +++ b/apps/backend/app/llmops/manager.py @@ -65,6 +65,12 @@ class GpuUnavailable(RuntimeError): """The pinned cuda_device doesn't exist on this host (pre-flight guard).""" +class EngineUnavailable(RuntimeError): + """This image lacks the engine's runtime binary (e.g. a trtllm model on the + collapsed vLLM image, which has no trtllm-serve). Rejected up-front instead of + crash-looping at spawn.""" + + class LoraRuntimeError(RuntimeError): """A runtime LoRA load/unload against a vLLM instance failed.""" @@ -628,6 +634,14 @@ async def start(self, key: str, force: bool = False, reset_restart: bool = True) if key in await self.foreign_assignments(): return await self._defer_to_owner(inst, Desired.RUNNING) launcher = self._launcher_for(inst) + # Fail loud if this image lacks the engine's runtime (a trtllm model on the + # collapsed vLLM image has no trtllm-serve) — reject up-front instead of + # crash-looping at spawn. Optional launcher hook; absent = always available. + if not getattr(launcher, "available", lambda: True)(): + raise EngineUnavailable( + f"engine '{inst.engine}' runtime is not installed in this image; " + f"cannot start {key} — deploy it on the {inst.engine} backend image/profile." + ) # Re-resolve the spec (config may have changed) outside the lock so the # GPU pre-flight's nvidia-smi call never extends the critical section. diff --git a/apps/backend/tests/unit/test_ha_safety.py b/apps/backend/tests/unit/test_ha_safety.py index 919a5d0..846846c 100644 --- a/apps/backend/tests/unit/test_ha_safety.py +++ b/apps/backend/tests/unit/test_ha_safety.py @@ -5,8 +5,8 @@ from app.core.settings import BackendSettings from app.llmops.launchers import EmbeddingLauncher, VllmLauncher -from app.llmops.manager import (ModelAlreadyRunning, ModelConflict, ModelManager, - build_registry) +from app.llmops.manager import (EngineUnavailable, ModelAlreadyRunning, + ModelConflict, ModelManager, build_registry) from app.llmops.state import Desired, ModelState from schema import load_config @@ -156,6 +156,17 @@ async def test_start_rejected_while_stopping(tmp_path): await mgr.start("Qwen3-0.6B::a") +async def test_start_rejected_when_engine_runtime_missing(tmp_path): + """start() must reject up-front (not crash at spawn) when the launcher reports + its engine runtime is absent from this image (e.g. a trtllm model on the + collapsed vLLM image whose trtllm-serve is missing).""" + mgr, _ = _manager(tmp_path) + launcher = mgr._launcher_for(mgr.registry.get("Qwen3-0.6B::a")) + launcher.available = lambda: False # simulate the runtime not being installed + with pytest.raises(EngineUnavailable): + await mgr.start("Qwen3-0.6B::a") + + async def test_stop_cas_finalize_does_not_clobber_new_process(tmp_path): """stop()'s finalizer must not reset state to STOPPED when a newer process (different proc handle) has taken the slot during the drain window.""" diff --git a/apps/backend/tests/unit/test_launchers.py b/apps/backend/tests/unit/test_launchers.py index e51ae4b..a130c13 100644 --- a/apps/backend/tests/unit/test_launchers.py +++ b/apps/backend/tests/unit/test_launchers.py @@ -699,3 +699,12 @@ def test_trtllm_build_spec_command_and_capabilities(tmp_path, monkeypatch): def test_trtllm_launcher_keys_filters_by_engine(): keys = TrtllmLauncher().keys(_trtllm_config()) assert keys == ["T::a"] + + +def test_trtllm_available_reflects_trtllm_serve_presence(monkeypatch): + import app.llmops.launchers as L + monkeypatch.setattr(L.shutil, "which", lambda name: "/usr/bin/trtllm-serve" + if name == "trtllm-serve" else None) + assert TrtllmLauncher().available() is True + monkeypatch.setattr(L.shutil, "which", lambda name: None) + assert TrtllmLauncher().available() is False diff --git a/deploy/grafana/provisioning/alerting/trtllm.yaml b/deploy/grafana/provisioning/alerting/trtllm.yaml new file mode 100644 index 0000000..6b5ebcc --- /dev/null +++ b/deploy/grafana/provisioning/alerting/trtllm.yaml @@ -0,0 +1,188 @@ +# Provisioned Grafana alert rules for the TensorRT-LLM fleet. +# +# Mirrors deploy/grafana/provisioning/alerting/vllm.yaml (A = Prometheus range +# query, B = reduce(last), C = threshold), but against trtllm_* metrics scraped +# from /prometheus/metrics. In the mixed stack every engine is scraped under the +# single `engines` job and distinguished by the `engine` label, so the target-down +# rule filters on {job="engines", engine="trtllm"} (not job="trtllm"). Thresholds +# are starting points — tune them to your models/SLA. Wire a contact point under +# Alerting -> Contact points to actually get notified. +apiVersion: 1 + +groups: + - orgId: 1 + name: TensorRT-LLM + folder: Alerts + interval: 1m + rules: + - uid: trtllm-target-down + title: TensorRT-LLM target down + condition: C + for: 1m + labels: + severity: critical + annotations: + summary: "A TensorRT-LLM instance has stopped responding to Prometheus scrapes." + noDataState: NoData + execErrState: Error + data: + - refId: A + relativeTimeRange: { from: 300, to: 0 } + datasourceUid: prometheus + model: + datasource: { type: prometheus, uid: prometheus } + editorMode: code + expr: 'up{job="engines", engine="trtllm"}' + instant: false + range: true + intervalMs: 15000 + maxDataPoints: 43200 + refId: A + - refId: B + relativeTimeRange: { from: 300, to: 0 } + datasourceUid: __expr__ + model: + type: reduce + datasource: { type: __expr__, uid: __expr__ } + expression: A + reducer: last + refId: B + - refId: C + relativeTimeRange: { from: 300, to: 0 } + datasourceUid: __expr__ + model: + type: threshold + datasource: { type: __expr__, uid: __expr__ } + expression: B + conditions: + - evaluator: { type: lt, params: [1] } + refId: C + + - uid: trtllm-ttft-p95-high + title: TensorRT-LLM TTFT p95 high + condition: C + for: 5m + labels: + severity: warning + annotations: + summary: "Time-to-first-token p95 is above 2s — users are waiting too long for the first token." + noDataState: NoData + execErrState: Error + data: + - refId: A + relativeTimeRange: { from: 600, to: 0 } + datasourceUid: prometheus + model: + datasource: { type: prometheus, uid: prometheus } + editorMode: code + expr: 'histogram_quantile(0.95, sum by (le) (rate(trtllm_time_to_first_token_seconds_bucket{engine="trtllm"}[5m])))' + instant: false + range: true + intervalMs: 15000 + maxDataPoints: 43200 + refId: A + - refId: B + relativeTimeRange: { from: 600, to: 0 } + datasourceUid: __expr__ + model: + type: reduce + datasource: { type: __expr__, uid: __expr__ } + expression: A + reducer: last + refId: B + - refId: C + relativeTimeRange: { from: 600, to: 0 } + datasourceUid: __expr__ + model: + type: threshold + datasource: { type: __expr__, uid: __expr__ } + expression: B + conditions: + - evaluator: { type: gt, params: [2] } + refId: C + + - uid: trtllm-kv-cache-high + title: TensorRT-LLM KV cache near full + condition: C + for: 5m + labels: + severity: warning + annotations: + summary: "KV cache utilization is above 90% — the engine is close to capacity and may start queueing/preempting." + noDataState: NoData + execErrState: Error + data: + - refId: A + relativeTimeRange: { from: 300, to: 0 } + datasourceUid: prometheus + model: + datasource: { type: prometheus, uid: prometheus } + editorMode: code + expr: 'max(trtllm_kv_cache_utilization{engine="trtllm"})' + instant: false + range: true + intervalMs: 15000 + maxDataPoints: 43200 + refId: A + - refId: B + relativeTimeRange: { from: 300, to: 0 } + datasourceUid: __expr__ + model: + type: reduce + datasource: { type: __expr__, uid: __expr__ } + expression: A + reducer: last + refId: B + - refId: C + relativeTimeRange: { from: 300, to: 0 } + datasourceUid: __expr__ + model: + type: threshold + datasource: { type: __expr__, uid: __expr__ } + expression: B + conditions: + - evaluator: { type: gt, params: [0.9] } + refId: C + + - uid: trtllm-requests-waiting + title: TensorRT-LLM requests queueing + condition: C + for: 10m + labels: + severity: warning + annotations: + summary: "Requests have been waiting in the queue for 10m — the fleet is under-provisioned for current load." + noDataState: NoData + execErrState: Error + data: + - refId: A + relativeTimeRange: { from: 600, to: 0 } + datasourceUid: prometheus + model: + datasource: { type: prometheus, uid: prometheus } + editorMode: code + expr: 'sum(trtllm_num_requests_waiting{engine="trtllm"})' + instant: false + range: true + intervalMs: 15000 + maxDataPoints: 43200 + refId: A + - refId: B + relativeTimeRange: { from: 600, to: 0 } + datasourceUid: __expr__ + model: + type: reduce + datasource: { type: __expr__, uid: __expr__ } + expression: A + reducer: last + refId: B + - refId: C + relativeTimeRange: { from: 600, to: 0 } + datasourceUid: __expr__ + model: + type: threshold + datasource: { type: __expr__, uid: __expr__ } + expression: B + conditions: + - evaluator: { type: gt, params: [0] } + refId: C From 01bc4e2bf75d7033d3c8d50046157658cd3a5d95 Mon Sep 17 00:00:00 2001 From: max Date: Sat, 4 Jul 2026 21:01:05 +0800 Subject: [PATCH 09/20] docs(config): ship a 4-engine example config (vLLM + SGLang + llama.cpp + TensorRT-LLM) The tracked example config only had vLLM groups (+ embedding/rerank). Add one representative group per non-vLLM engine so the template demonstrates every supported backend end-to-end: - SGLang: Qwen3-0.6B-sglang, Qwen2.5-0.5B-Instruct-sglang (engine: sglang) - llama.cpp: Qwen2.5-0.5B / SmolLM2-360M / Qwen3-0.6B GGUF (engine: llamacpp) - TensorRT-LLM: Qwen3-0.6B-trt (engine: trtllm, BYO pre-built engine_dir) Values are tuned for a single 8GB card; adjust ports/memory/model_tag per host. Non-vLLM groups only run on their matching backend image (make up-mixed). Co-Authored-By: Claude Opus 4.8 --- packages/config-schema/config.yaml | 112 +++++++++++++++++++++++++++++ 1 file changed, 112 insertions(+) diff --git a/packages/config-schema/config.yaml b/packages/config-schema/config.yaml index 807ef4f..e356a70 100644 --- a/packages/config-schema/config.yaml +++ b/packages/config-schema/config.yaml @@ -53,6 +53,20 @@ LLM_engines: # Allow runtime (hot) load/unload from the dashboard — sets # VLLM_ALLOW_RUNTIME_LORA_UPDATING. Without it the LoRA set is fixed at launch. allow_runtime_lora: true + quantization: bitsandbytes + enforce_eager: true + # Sleep-mode warm-standby tier (Phase 0 autoscaling): launches with + # --enable-sleep-mode + VLLM_SERVER_DEV_MODE=1 so /sleep, /wake_up and + # /is_sleeping are available. Level-1 sleep frees VRAM but keeps a fast wake. + enable_sleep_mode: true + # Autoscaling (Phase 2): the autoscaler owns this group's instances — keeps + # min_ready warm, scales up on queue depth to max_ready, sleeps/stops when idle. + # Timings left at their production defaults; tune from the dashboard if needed. + autoscale: + enabled: true + min_ready: 1 + max_ready: 2 + # --- Single-instance group --- Qwen2.5-0.5B-Instruct: instances: @@ -209,6 +223,104 @@ LLM_engines: dtype: "float16" gpu_memory_utilization: 0.20 + # --- SGLang-engine groups (engine: sglang) ------------------------------- + # These run on a backend built from the SGLang image (deploy/engine-sglang.Dockerfile) + # — e.g. `make up-mixed` (a SGLang backend advertises LLMOPS_NODE_ENGINES=sglang and + # the scheduler places these here). On the plain vLLM `make up` they register but + # can't start (no sglang in that image). The launcher translates the engine-neutral + # params: max_model_len -> --context-length, gpu_memory_utilization -> + # --mem-fraction-static, tensor_parallel_size -> --tp-size. Keep vLLM-only flags + # (quantization/enforce_eager/tool parsers/sleep) OUT of sglang groups. + Qwen3-0.6B-sglang: + instances: + - id: "sgl-qwen3" + host: "localhost" + port: 8030 + cuda_device: 0 + model_config: + model_tag: "Qwen/Qwen3-0.6B" + engine: sglang + dtype: "bfloat16" + max_model_len: 4096 + gpu_memory_utilization: 0.30 + tensor_parallel_size: 1 + + Qwen2.5-0.5B-Instruct-sglang: + instances: + - id: "sgl-q25-05b" + host: "localhost" + port: 8031 + cuda_device: 0 + model_config: + model_tag: "Qwen/Qwen2.5-0.5B-Instruct" + engine: sglang + dtype: "bfloat16" + max_model_len: 4096 + gpu_memory_utilization: 0.30 + tensor_parallel_size: 1 + + # llama.cpp (GGUF) models — served by llama-server on the llamacpp backend + # (make up-mixed). model_tag is a HF GGUF repo pulled at launch via -hf; gguf_quant + # picks the quant file; n_gpu_layers=99 offloads all layers to GPU (set 0 for CPU). + # No dtype / gpu_memory_utilization / tensor_parallel_size — those don't apply to + # llama.cpp (the launcher drops them). See docs/mixed-engine-deployment.md. + Qwen2.5-0.5B-Instruct-gguf: + instances: + - id: "cpp-q25-05b" + host: "localhost" + port: 8040 + cuda_device: 0 + model_config: + model_tag: "Qwen/Qwen2.5-0.5B-Instruct-GGUF" + engine: llamacpp + gguf_quant: "Q4_K_M" + max_model_len: 4096 + n_gpu_layers: 99 + + SmolLM2-360M-Instruct-gguf: + instances: + - id: "cpp-smol360" + host: "localhost" + port: 8041 + cuda_device: 0 + model_config: + model_tag: "HuggingFaceTB/SmolLM2-360M-Instruct-GGUF" + engine: llamacpp + gguf_quant: "Q8_0" + max_model_len: 4096 + n_gpu_layers: 99 + + Qwen3-0.6B-gguf: + instances: + - id: "cpp-qwen3" + host: "localhost" + port: 8042 + cuda_device: 0 + model_config: + model_tag: "Qwen/Qwen3-0.6B-GGUF" + engine: llamacpp + gguf_quant: "Q8_0" + max_model_len: 4096 + n_gpu_layers: 99 + + # TensorRT-LLM (engine: trtllm) — serves a PRE-BUILT TRT engine (Phase 1 BYO). + # engine_dir is inside the container (mixed compose mounts ~/trtllm-test/engines + # -> /engines). Built for max_seq_len 2048 (see docs/trtllm-conversion-validation). + Qwen3-0.6B-trt: + instances: + - id: "trt-qwen3" + host: "localhost" + port: 8055 + cuda_device: 0 + model_config: + model_tag: "Qwen/Qwen3-0.6B" + engine: trtllm + engine_dir: "/engines/qwen3-06b_tp1_fp16_isl1024_osl1024_bs4" + tokenizer: "Qwen/Qwen3-0.6B" + served_model_name: "qwen3-06b-trt" + max_model_len: 2048 + gpu_memory_utilization: 0.4 + embedding_server: host: "localhost" port: 8005 From 5659620e891a0a23b411ecb5c437ecb3719f8d13 Mon Sep 17 00:00:00 2001 From: max Date: Sun, 5 Jul 2026 00:17:29 +0800 Subject: [PATCH 10/20] =?UTF-8?q?feat(trtllm):=20dual=20backend=20?= =?UTF-8?q?=E2=80=94=20engine=5Fdir=20->=20tensorrt,=20model=5Ftag=20->=20?= =?UTF-8?q?pytorch?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit trtllm-serve's tensorrt backend (pre-built engine) is being deprecated after v1.3.0rc20 and its pytorch backend loads HF weights directly with no build step. Benchmarked on Qwen3-0.6B (single stream): pytorch matches tensorrt on decode / throughput, only ~18ms slower TTFT. So support BOTH, chosen by config: - engine_dir set -> serve --backend tensorrt (lowest TTFT) - engine_dir unset -> serve --backend pytorch (no build; any TRT-LLM-supported arch, monitoring identical) Monitoring is identical across backends (same trtllm_* at /prometheus/metrics, same Grafana dashboard/alerts, zero router change) — BUT the perf YAML differs and build_spec now writes it per backend (both verified live): - tensorrt: return_perf_metrics only (adding enable_iter_perf_stats aborts load: "_TrtLLM got invalid argument: enable_iter_perf_stats") - pytorch: return_perf_metrics + enable_iter_perf_stats (the latter is REQUIRED for the iteration gauges num_requests_running/waiting + kv_cache_utilization that the router and dashboard depend on; without it only request-level metrics appear) Verified end-to-end: a config group with engine: trtllm and no engine_dir starts via the dashboard, serves through the router, and all 5 router-critical metrics land in Prometheus with engine=trtllm. config.yaml ships both a tensorrt and a pytorch example. 500 backend tests pass. Co-Authored-By: Claude Opus 4.8 --- apps/backend/app/llmops/launchers.py | 58 ++++++++++++++++------- apps/backend/tests/unit/test_launchers.py | 52 ++++++++++++++++++-- docs/trtllm-launcher-impl-design_zh-TW.md | 24 ++++++++++ packages/config-schema/config.yaml | 27 +++++++++-- packages/config-schema/schema.py | 10 ++-- 5 files changed, 141 insertions(+), 30 deletions(-) diff --git a/apps/backend/app/llmops/launchers.py b/apps/backend/app/llmops/launchers.py index 2714d95..7a97b78 100644 --- a/apps/backend/app/llmops/launchers.py +++ b/apps/backend/app/llmops/launchers.py @@ -686,30 +686,39 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: def build_trtllm_cli_args(model_cfg: dict, *, perf_yaml_path: str) -> list[str]: """dict -> ``trtllm-serve`` CLI args (the part after ``trtllm-serve``). - Phase 1 serves a pre-built TensorRT engine directory (``engine_dir``, required) - with ``--backend tensorrt``. ``MODEL`` is the positional engine dir. The tokenizer - is not inside the engine dir, so ``--tokenizer`` is always emitted (``tokenizer`` - override, else ``model_tag``). ``--served_model_name`` makes /v1/models advertise a - stable name (router forward_name + boot adopt identity). + Two launch modes, chosen by whether ``engine_dir`` is set: + - ``engine_dir`` present -> ``serve --backend tensorrt`` (BYO a + pre-built TensorRT engine; lowest TTFT, but engine is GPU-arch/version-bound). + - ``engine_dir`` absent -> ``serve --backend pytorch`` (load HF + weights directly, no build step). ``model_tag`` is the positional. + The tokenizer isn't inside a tensorrt engine dir, so ``--tokenizer`` is always + emitted (``tokenizer`` override, else ``model_tag``); for pytorch it's a harmless + restatement of the model's own tokenizer. ``--served_model_name`` makes /v1/models + advertise a stable name (router forward_name + boot adopt identity). Engine-neutral keys map via ``_TRTLLM_PARAM_MAP`` (``max_model_len`` -> ``--max_seq_len`` etc.); a small whitelist of trtllm-native flags passes through; everything else is dropped (trtllm-serve rejects unknown args). ``--extra_llm_api_options`` points at a per-instance YAML carrying ``return_perf_metrics: true`` — WITHOUT it - the server exposes no ``/prometheus/metrics``. + the server exposes no ``/prometheus/metrics`` (verified true for BOTH backends). """ + model_tag = model_cfg.get("model_tag") engine_dir = model_cfg.get("engine_dir") - if not engine_dir: + if engine_dir: + positional, backend = str(engine_dir), "tensorrt" + elif model_tag: + positional, backend = str(model_tag), "pytorch" + else: raise ValueError( - "trtllm model_config must provide 'engine_dir' (a pre-built TRT engine " - "directory inside the container); Phase 1 is bring-your-own-engine") - model_tag = model_cfg.get("model_tag") + "trtllm model_config needs either 'engine_dir' (a pre-built TRT engine " + "directory -> --backend tensorrt) or 'model_tag' (HF weights -> " + "--backend pytorch)") served = model_cfg.get("served_model_name") or model_tag tokenizer = model_cfg.get("tokenizer") or model_tag host = model_cfg.get("host") or "localhost" port = model_cfg.get("port") - args: list[str] = ["serve", str(engine_dir), "--backend", "tensorrt", + args: list[str] = ["serve", positional, "--backend", backend, "--host", str(host)] if port is not None: args += ["--port", str(port)] @@ -740,11 +749,14 @@ def build_trtllm_cli_args(model_cfg: dict, *, perf_yaml_path: str) -> list[str]: class TrtllmLauncher: kind = ModelKind.LLM engine = "trtllm" - # Serves a pre-built TRT engine over OpenAI /v1. Metrics only (Prometheus trtllm_* - # at /prometheus/metrics). NO sleep (/release_memory is AsyncLLM/pytorch-only — - # 500 on the engine backend), NO runtime LoRA (no /v1/load_lora_adapter), NO - # kv_transfer (that's disaggregated). Static LoRA needs a YAML lora_config — not - # done in Phase 1. See docs/trtllm-launcher-impl-design_zh-TW.md §7. + # Serves either a pre-built TRT engine (engine_dir -> --backend tensorrt) or HF + # weights directly (model_tag only -> --backend pytorch) over OpenAI /v1. Both + # expose the same Prometheus trtllm_* metrics at /prometheus/metrics, so monitoring + # is identical. Capabilities are advertised conservatively for BOTH modes: metrics + # only — NO runtime LoRA (no /v1/load_lora_adapter) and NO kv_transfer. (Sleep is + # pytorch-backend-only and left as a follow-up so the dashboard doesn't offer a + # button that 500s on tensorrt-mode instances.) Static LoRA needs a YAML + # lora_config. See docs/trtllm-launcher-impl-design_zh-TW.md §7. capabilities = frozenset({CAP_METRICS_TRTLLM}) lora_endpoint_prefix = "" metric_prefix = "trtllm" @@ -793,13 +805,23 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: cli_cfg = {**merged, "host": bind_host} if bind_host else merged # Per-instance YAML enabling Prometheus metrics. WITHOUT this the server has no - # /prometheus/metrics. (enable_iter_perf_stats is deliberately NOT set — the - # TensorRT backend rejects it.) + # /prometheus/metrics. The required keys differ by backend (both verified live): + # - tensorrt: `return_perf_metrics` ALONE gives the full trtllm_* set. Adding + # enable_iter_perf_stats makes the tensorrt backend abort at load + # ("_TrtLLM got invalid argument: enable_iter_perf_stats"). + # - pytorch: `return_perf_metrics` alone only yields request-level metrics + # (prompt/generation tokens, latency histograms); the iteration-level GAUGES + # the router + dashboard rely on (num_requests_running/waiting, + # kv_cache_utilization) need enable_iter_perf_stats too. + # engine_dir present => tensorrt mode; absent => pytorch mode (see cli builder). + is_pytorch = not merged.get("engine_dir") perf_yaml_path = os.path.join(LOG_DIR, f"{model_tag}__{instance_id}.trtllm-perf.yaml") try: os.makedirs(LOG_DIR, exist_ok=True) with open(perf_yaml_path, "w", encoding="utf-8") as f: f.write("return_perf_metrics: true\n") + if is_pytorch: + f.write("enable_iter_perf_stats: true\n") except OSError: logger.warning("trtllm: could not write perf YAML at %s", perf_yaml_path) diff --git a/apps/backend/tests/unit/test_launchers.py b/apps/backend/tests/unit/test_launchers.py index a130c13..ed822f4 100644 --- a/apps/backend/tests/unit/test_launchers.py +++ b/apps/backend/tests/unit/test_launchers.py @@ -671,9 +671,23 @@ def test_trtllm_args_param_map_and_passthrough_and_drops(): assert dropped not in args -def test_trtllm_args_requires_engine_dir(): - with pytest.raises(ValueError, match="engine_dir"): - build_trtllm_cli_args({"model_tag": "m"}, perf_yaml_path="/p.yaml") +def test_trtllm_args_pytorch_backend_from_model_tag(): + # No engine_dir -> serve the HF model_tag with --backend pytorch. + args = build_trtllm_cli_args( + {"model_tag": "Qwen/Qwen3-0.6B", "port": 8061, "host": "0.0.0.0", + "max_model_len": 2048}, + perf_yaml_path="/logs/T__a.trtllm-perf.yaml") + assert args[:4] == ["serve", "Qwen/Qwen3-0.6B", "--backend", "pytorch"] + assert args[args.index("--tokenizer") + 1] == "Qwen/Qwen3-0.6B" + assert args[args.index("--served_model_name") + 1] == "Qwen/Qwen3-0.6B" + assert args[args.index("--max_seq_len") + 1] == "2048" + # metrics YAML still passed -> monitoring identical to tensorrt mode + assert args[args.index("--extra_llm_api_options") + 1] == "/logs/T__a.trtllm-perf.yaml" + + +def test_trtllm_args_requires_engine_dir_or_model_tag(): + with pytest.raises(ValueError, match="engine_dir.*model_tag|model_tag"): + build_trtllm_cli_args({}, perf_yaml_path="/p.yaml") def test_trtllm_build_spec_command_and_capabilities(tmp_path, monkeypatch): @@ -690,10 +704,38 @@ def test_trtllm_build_spec_command_and_capabilities(tmp_path, monkeypatch): assert CAP_METRICS_TRTLLM in spec.capabilities assert CAP_SLEEP not in spec.capabilities and CAP_RUNTIME_LORA not in spec.capabilities assert spec.env.get("CUDA_VISIBLE_DEVICES") == "0" - # perf YAML was written and referenced + # perf YAML was written and referenced; tensorrt mode must NOT set + # enable_iter_perf_stats (the tensorrt backend aborts on it). + perf = spec.command[spec.command.index("--extra_llm_api_options") + 1] + with open(perf) as f: + content = f.read() + assert "return_perf_metrics: true" in content + assert "enable_iter_perf_stats" not in content + + +def test_trtllm_build_spec_pytorch_when_no_engine_dir(tmp_path, monkeypatch): + import app.llmops.launchers as L + monkeypatch.setattr(L, "LOG_DIR", str(tmp_path)) + # A trtllm group with NO engine_dir -> pytorch backend serving the HF model_tag. + cfg = RootConfig.model_validate({ + "server": {"host": "0.0.0.0", "port": 8887}, + "LLM_engines": {"T": { + "instances": [{"id": "a", "host": "localhost", "port": 8061, "cuda_device": 0}], + "model_config": {"model_tag": "Qwen/Qwen3-0.6B", "engine": "trtllm", + "served_model_name": "qwen3-pt", "max_model_len": 2048}, + }}, + }) + spec = TrtllmLauncher().build_spec(cfg, "config.yaml", "T::a") + assert spec.command[1:5] == ["serve", "Qwen/Qwen3-0.6B", "--backend", "pytorch"] + assert spec.served_name == "qwen3-pt" + assert CAP_METRICS_TRTLLM in spec.capabilities + # pytorch mode needs BOTH keys so the iteration-level gauges (running/waiting/ + # kv_cache_utilization) the router + dashboard use are exposed. perf = spec.command[spec.command.index("--extra_llm_api_options") + 1] with open(perf) as f: - assert "return_perf_metrics: true" in f.read() + content = f.read() + assert "return_perf_metrics: true" in content + assert "enable_iter_perf_stats: true" in content def test_trtllm_launcher_keys_filters_by_engine(): diff --git a/docs/trtllm-launcher-impl-design_zh-TW.md b/docs/trtllm-launcher-impl-design_zh-TW.md index 2084d4e..a4327b4 100644 --- a/docs/trtllm-launcher-impl-design_zh-TW.md +++ b/docs/trtllm-launcher-impl-design_zh-TW.md @@ -43,6 +43,30 @@ vLLM / SGLang / llama.cpp 都符合「spawn-and-probe 單行程」契約,加它 > 本文件 §2–§7、§9–§11 是 **Phase 1** 的完整實作;§8 是 **Phase 2** 的設計。 +### Phase 1b —「雙 backend」(已做):tensorrt engine + pytorch HF weights + +`--backend tensorrt` 是 NVIDIA 正在淘汰的路徑(`v1.3.0rc20` 是最後支援版,見 §14),而 `--backend +pytorch` 直接吃 HF 權重、免 build、還解鎖 sleep/wake。實測小模型單流:pytorch 的 decode/吞吐與 +tensorrt 打平,只 TTFT 略慢(22ms→40ms)。所以 launcher **同時支援兩者**,由 config 的 `engine_dir` +決定,使用者「放一般 HF 或放預編 engine 都能起」: + +| config | backend | 命令 | +|---|---|---| +| 有 `engine_dir` | tensorrt | `serve --backend tensorrt` | +| 無 `engine_dir`(只給 `model_tag`) | pytorch | `serve --backend pytorch` | + +**監控在兩個 backend 上一致**(同 `/prometheus/metrics`、同 `trtllm_*`、同 Grafana dashboard/告警), +但**開 metrics 的 perf YAML key 依 backend 不同**(兩者皆實測): + +| backend | perf YAML | 說明 | +|---|---|---| +| tensorrt | `return_perf_metrics: true` | 這樣就有全套。**加** `enable_iter_perf_stats` 會讓 tensorrt 在載入時 abort(`_TrtLLM got invalid argument`)。 | +| pytorch | `return_perf_metrics: true` + `enable_iter_perf_stats: true` | 只有前者的話,**只出 request 級指標**(prompt/generation tokens、latency histogram);router + dashboard 依賴的 **iteration gauge**(`num_requests_running`/`waiting`、`kv_cache_utilization`)要靠後者。 | + +`build_spec` 依 `engine_dir` 有無寫對應的 YAML(見 [launchers.py](../apps/backend/app/llmops/launchers.py) +`TrtllmLauncher.build_spec`)。capabilities 兩 mode 都保守只給 `CAP_METRICS_TRTLLM`(sleep 雖 pytorch 可用, +先不宣告,避免 dashboard 對 tensorrt-mode 實例顯示會 500 的 sleep 按鈕——列為後續)。 + --- ## 2. 檔案改動清單(對照 [adding-a-new-engine §1](adding-a-new-engine_zh-TW.md)) diff --git a/packages/config-schema/config.yaml b/packages/config-schema/config.yaml index e356a70..a18f10a 100644 --- a/packages/config-schema/config.yaml +++ b/packages/config-schema/config.yaml @@ -303,9 +303,12 @@ LLM_engines: max_model_len: 4096 n_gpu_layers: 99 - # TensorRT-LLM (engine: trtllm) — serves a PRE-BUILT TRT engine (Phase 1 BYO). - # engine_dir is inside the container (mixed compose mounts ~/trtllm-test/engines - # -> /engines). Built for max_seq_len 2048 (see docs/trtllm-conversion-validation). + # TensorRT-LLM has TWO launch modes on the same engine, chosen by engine_dir: + # + # (a) engine_dir SET -> --backend tensorrt: serves a PRE-BUILT TRT engine (lowest + # TTFT; engine is GPU-arch/version-bound). engine_dir is the in-container path + # (mixed compose mounts ~/trtllm-test/engines -> /engines). See + # docs/trtllm-conversion-validation for how to build it. Qwen3-0.6B-trt: instances: - id: "trt-qwen3" @@ -321,6 +324,24 @@ LLM_engines: max_model_len: 2048 gpu_memory_utilization: 0.4 + # (b) engine_dir UNSET -> --backend pytorch: loads the HF model_tag directly, NO + # build step. Monitoring is identical (same trtllm_* metrics / dashboard). Use + # this to run any HF checkpoint under TensorRT-LLM without pre-building an engine. + # Needs the dual-backend launcher (rebuild the engine images first, else an older + # image aborts on config load because its trtllm launcher requires engine_dir). + Qwen3-0.6B-trt-pytorch: + instances: + - id: "trt-pt-qwen3" + host: "localhost" + port: 8056 + cuda_device: 0 + model_config: + model_tag: "Qwen/Qwen3-0.6B" + engine: trtllm + served_model_name: "qwen3-06b-trt-pt" + max_model_len: 2048 + gpu_memory_utilization: 0.4 + embedding_server: host: "localhost" port: 8005 diff --git a/packages/config-schema/schema.py b/packages/config-schema/schema.py index 65d6065..2eb3ee3 100644 --- a/packages/config-schema/schema.py +++ b/packages/config-schema/schema.py @@ -76,10 +76,12 @@ class EngineModelConfig(BaseModel): max_model_len: Optional[int] = None gpu_memory_utilization: Optional[float] = None tensor_parallel_size: int = 1 - # TensorRT-LLM (engine: trtllm) only, Phase 1 "bring-your-own-engine": the path to - # a pre-built TRT engine directory (inside the container) that trtllm-serve serves - # with --backend tensorrt. `tokenizer` overrides the tokenizer source (a HF repo or - # dir); defaults to model_tag. Both ride the trtllm launcher; other engines ignore + # TensorRT-LLM (engine: trtllm) only. `engine_dir` selects the backend: set it to a + # pre-built TRT engine directory (inside the container) and trtllm-serve runs it with + # --backend tensorrt (lowest TTFT, engine is GPU-arch/version-bound); leave it unset + # and trtllm-serve loads `model_tag` (HF weights) with --backend pytorch (no build + # step, monitoring identical). `tokenizer` overrides the tokenizer source (a HF repo + # or dir); defaults to model_tag. Both ride the trtllm launcher; other engines ignore # them. See docs/trtllm-launcher-impl-design_zh-TW.md. engine_dir: Optional[str] = None tokenizer: Optional[str] = None From fee0e92e44b793562ea97178d672cf183198d6dd Mon Sep 17 00:00:00 2001 From: max Date: Sun, 5 Jul 2026 00:36:55 +0800 Subject: [PATCH 11/20] =?UTF-8?q?feat(trtllm):=20convert-to-TRT=20backend?= =?UTF-8?q?=20=E2=80=94=20build=20engines=20from=20HF=20via=20trtllm-bench?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model-library "convert to TRT" capability (backend + API; frontend to follow), mirroring the LoRA GGUF convert service. Builds a pre-built TensorRT engine from an HF model so a trtllm group can serve it with --backend tensorrt (lowest TTFT). Key: the builder is `trtllm-bench -m -w build ...` — the UNIFIED, architecture-agnostic engine builder (auto-detects the arch), so ONE code path covers every architecture TensorRT-LLM supports, not a per-arch convert_checkpoint.py matrix. Verified live: built Qwen3-0.6B end-to-end via POST /api/trt/convert. - app/services/trt_convert.py: BuildParams (+ cache_key over model/tp/pp/seq/batch/ tokens/quant/GPU-compute-cap/trtllm-version — an engine is arch+version bound), build() runs trtllm-bench into a temp workspace then moves the engine to // with a manifest; idempotent (reuses a valid build). TrtConvertManager async jobs mirror LoraConvertManager. bench_available() gates on trtllm-bench so the feature is unavailable off the TensorRT-LLM image. - app/api/trt.py: GET /trt/engines (built engines + availability + quantization options), GET /trt/conversions (jobs), POST /trt/convert (params: model_tag, tp/pp, max_seq_len, max_batch_size, max_num_tokens, quantization). require_operator on POST. - Wired managers in main.py + test conftest. Verified live on the trtllm backend: available=true, build completed (~130s), engine at /engines// with rank0.engine + manifest; listed by /trt/engines; vllm backend correctly reports available=false. 514 backend tests pass. Co-Authored-By: Claude Opus 4.8 --- apps/backend/app/api/trt.py | 76 ++++++ apps/backend/app/main.py | 4 + apps/backend/app/services/trt_convert.py | 253 ++++++++++++++++++++ apps/backend/tests/api/test_trt_routes.py | 65 +++++ apps/backend/tests/conftest.py | 2 + apps/backend/tests/unit/test_trt_convert.py | 142 +++++++++++ 6 files changed, 542 insertions(+) create mode 100644 apps/backend/app/api/trt.py create mode 100644 apps/backend/app/services/trt_convert.py create mode 100644 apps/backend/tests/api/test_trt_routes.py create mode 100644 apps/backend/tests/unit/test_trt_convert.py diff --git a/apps/backend/app/api/trt.py b/apps/backend/app/api/trt.py new file mode 100644 index 0000000..ba3e1cf --- /dev/null +++ b/apps/backend/app/api/trt.py @@ -0,0 +1,76 @@ +"""TensorRT-LLM engine build ("convert to TRT") endpoints. Mirrors api/lora.py. + +Builds a pre-built TRT engine from an HF model via trtllm-bench (unified, +architecture-agnostic). A completed build lands under the engines root and is usable +by a trtllm group with engine_dir pointing at it (--backend tensorrt, lowest TTFT). +Only available on the TensorRT-LLM backend image (trtllm-bench present). +""" +from __future__ import annotations + +import asyncio +from typing import Optional + +from fastapi import APIRouter, Depends, HTTPException, Request, status +from pydantic import BaseModel, Field + +from app.core.auth import require_operator +from app.services import trt_convert +from app.services.trt_convert import BuildParams + +router = APIRouter(tags=["trt"]) + + +def _mgr(request: Request): + return request.app.state.trt_convert_manager + + +class TrtBuildRequest(BaseModel): + model_tag: str = Field(min_length=1) + tp_size: int = Field(default=1, ge=1) + pp_size: int = Field(default=1, ge=1) + max_seq_len: int = Field(default=2048, ge=1) + max_batch_size: int = Field(default=4, ge=1) + max_num_tokens: int = Field(default=8192, ge=1) + quantization: Optional[str] = None # one of trt_convert.QUANTIZATIONS, or null + + +@router.get("/trt/engines") +async def list_trt_engines(request: Request): + """Built engines on disk + whether the build tooling is available here + the + quantization options the UI can offer.""" + mgr = _mgr(request) + loop = asyncio.get_event_loop() + engines = await loop.run_in_executor(None, mgr.list_engines) + return { + "available": mgr.available(), + "root": trt_convert.engines_root(), + "quantizations": list(trt_convert.QUANTIZATIONS), + "engines": engines, + } + + +@router.get("/trt/conversions") +async def list_trt_conversions(request: Request): + mgr = _mgr(request) + return {"available": mgr.available(), "jobs": mgr.list()} + + +@router.post("/trt/convert", status_code=status.HTTP_202_ACCEPTED, + dependencies=[Depends(require_operator)]) +async def start_trt_convert(body: TrtBuildRequest, request: Request): + """Kick off an async HF->TRT-engine build. Returns the job (poll /trt/conversions).""" + mgr = _mgr(request) + if not mgr.available(): + raise HTTPException( + status.HTTP_503_SERVICE_UNAVAILABLE, + "TensorRT-LLM engine build tooling (trtllm-bench) isn't available on this backend", + ) + try: + params = BuildParams( + model_tag=body.model_tag, tp_size=body.tp_size, pp_size=body.pp_size, + max_seq_len=body.max_seq_len, max_batch_size=body.max_batch_size, + max_num_tokens=body.max_num_tokens, quantization=body.quantization, + ) + return mgr.start(params) + except ValueError as e: + raise HTTPException(status.HTTP_400_BAD_REQUEST, str(e)) diff --git a/apps/backend/app/main.py b/apps/backend/app/main.py index c82a7b3..8468f59 100644 --- a/apps/backend/app/main.py +++ b/apps/backend/app/main.py @@ -27,6 +27,7 @@ from app.api import eval as eval_routes from app.api import lora as lora_routes from app.api import metrics as metrics_routes +from app.api import trt as trt_routes from app.api import models as model_routes from app.api import sso as sso_routes from app.api import perf as perf_routes @@ -51,6 +52,7 @@ from app.services.dataset_downloads import DatasetDownloadManager from app.services.downloads import DownloadManager from app.services.lora_convert import LoraConvertManager +from app.services.trt_convert import TrtConvertManager from app.services.lora_downloads import LoraDownloadManager from app.services.gpu_service import get_gpu_processes_with_info from app.services.overlay import build_merged_config, hydrate_overlay_from_store, overlay_path @@ -154,6 +156,7 @@ async def lifespan(app: FastAPI): app.state.dataset_download_manager = DatasetDownloadManager() app.state.lora_download_manager = LoraDownloadManager() app.state.lora_convert_manager = LoraConvertManager() + app.state.trt_convert_manager = TrtConvertManager() perf_root = os.path.join(os.path.dirname(store.db_path), "perf") app.state.perf_manager = PerfManager(store, manager, settings, perf_root, router_url) eval_root = os.path.join(os.path.dirname(store.db_path), "eval") @@ -308,6 +311,7 @@ def create_app() -> FastAPI: install_config_version_middleware(app) app.include_router(model_routes.router, prefix="/api") app.include_router(engine_routes.router, prefix="/api") + app.include_router(trt_routes.router, prefix="/api") app.include_router(system_routes.router, prefix="/api") app.include_router(config_routes.router, prefix="/api") app.include_router(observability_routes.router, prefix="/api") diff --git a/apps/backend/app/services/trt_convert.py b/apps/backend/app/services/trt_convert.py new file mode 100644 index 0000000..f998735 --- /dev/null +++ b/apps/backend/app/services/trt_convert.py @@ -0,0 +1,253 @@ +"""Build a TensorRT-LLM engine from an HF model, for the trtllm engine's tensorrt +backend (engine_dir mode). The model library's "convert to TRT" action. + +Mirrors lora_convert: an async job manager wrapping a blocking subprocess. Here the +subprocess is ``trtllm-bench -m -w build ...`` — the UNIFIED, +architecture-agnostic engine builder. It auto-detects the model architecture (unlike +the legacy per-arch ``convert_checkpoint.py`` scripts), so ONE code path covers every +architecture TensorRT-LLM supports. The pre-built engine gives the lowest TTFT, but is +bound to the GPU architecture + TRT-LLM version, so the cache key + manifest record +both and a mismatch forces a rebuild. + +Only the TensorRT-LLM image ships ``trtllm-bench``; ``bench_available()`` gates on it, +so on other backends the feature is simply unavailable (like GGUF conversion is). + +Engines land under the shared engines root (``LLMOPS_TRT_ENGINES_DIR`` -> the /engines +mount the trtllm launcher reads ``engine_dir`` from), so a completed build is +immediately usable by a trtllm group whose ``engine_dir`` points at it. +""" +from __future__ import annotations + +import asyncio +import glob +import json +import logging +import os +import re +import shutil +import subprocess +import time +from dataclasses import asdict, dataclass, field +from typing import Optional + +logger = logging.getLogger(__name__) + +# Quantization algorithms `trtllm-bench build -q` accepts. None = no quantization. +QUANTIZATIONS = ( + "W8A16", "W4A16", "W4A16_AWQ", "W4A8_AWQ", "W4A16_GPTQ", "FP8", "INT8", "NVFP4", +) +# Cap the build subprocess; large models build for minutes, but never hang forever. +BUILD_TIMEOUT_S = 3600.0 + + +def engines_root() -> str: + return os.environ.get("LLMOPS_TRT_ENGINES_DIR", "/engines") + + +def bench_available() -> bool: + """Whether trtllm-bench is present (only the TensorRT-LLM image ships it).""" + return shutil.which("trtllm-bench") is not None + + +def _trtllm_version() -> str: + try: + from importlib.metadata import version + return version("tensorrt_llm") + except Exception: + return "unknown" + + +def _gpu_compute_cap() -> str: + """SM compute capability of GPU 0 as digits (e.g. '86'). Empty if undeterminable. + Part of the cache key because a serialized engine can't deserialize on a different + GPU architecture.""" + try: + out = subprocess.run( + ["nvidia-smi", "--query-gpu=compute_cap", "--format=csv,noheader"], + capture_output=True, text=True, timeout=10, + ) + first = (out.stdout or "").strip().splitlines()[0].strip() + return first.replace(".", "") # "8.6" -> "86" + except Exception: + return "" + + +def _sanitize(s: str) -> str: + return re.sub(r"[^A-Za-z0-9._-]", "_", s.strip()) + + +@dataclass(frozen=True) +class BuildParams: + """The knobs that determine (and identify) a built engine. Anything that changes + the engine bytes belongs in the cache key.""" + model_tag: str + tp_size: int = 1 + pp_size: int = 1 + max_seq_len: int = 2048 + max_batch_size: int = 4 + max_num_tokens: int = 8192 + quantization: Optional[str] = None # one of QUANTIZATIONS, or None + + def cache_key(self) -> str: + q = (self.quantization or "none").lower() + return "_".join([ + _sanitize(self.model_tag), + f"tp{self.tp_size}pp{self.pp_size}", + f"seq{self.max_seq_len}", f"bs{self.max_batch_size}", f"nt{self.max_num_tokens}", + q, + f"sm{_gpu_compute_cap() or 'x'}", + f"trt{_sanitize(_trtllm_version())}", + ]) + + +def engine_dir_for(params: BuildParams) -> str: + return os.path.join(engines_root(), params.cache_key()) + + +def is_built(params: BuildParams) -> bool: + d = engine_dir_for(params) + return os.path.isfile(os.path.join(d, "manifest.json")) and bool( + glob.glob(os.path.join(d, "rank*.engine"))) + + +def _validate(params: BuildParams) -> None: + if not (params.model_tag or "").strip(): + raise ValueError("model_tag is required") + if params.quantization is not None and params.quantization not in QUANTIZATIONS: + raise ValueError( + f"unknown quantization '{params.quantization}'; expected one of {list(QUANTIZATIONS)}") + for name in ("tp_size", "pp_size", "max_seq_len", "max_batch_size", "max_num_tokens"): + if int(getattr(params, name)) < 1: + raise ValueError(f"{name} must be >= 1") + + +def build(params: BuildParams) -> str: + """Build (or reuse) the engine for ``params``; return its engine_dir. Blocking — + call via run_in_executor. + + Runs ``trtllm-bench build`` into a temp workspace, moves the produced engine dir + (the one holding ``rank*.engine``) to a flat ``//`` and + writes ``manifest.json``. Idempotent: a valid existing build is returned as-is. + + trtllm-bench inherits the process env; the backend is launched via a login shell + (compose ``command: bash -lc``), so LD_LIBRARY_PATH for the TRT libs is already set + — same reason the launcher can spawn trtllm-serve directly. + """ + if not bench_available(): + raise RuntimeError( + "trtllm-bench is not available on this backend (only the TensorRT-LLM image ships it)") + _validate(params) + final_dir = engine_dir_for(params) + if is_built(params): + logger.info("trt engine already built, reusing: %s", final_dir) + return final_dir + + ws = os.path.join(engines_root(), ".trt-build", params.cache_key()) + shutil.rmtree(ws, ignore_errors=True) + os.makedirs(ws, exist_ok=True) + cmd = [ + "trtllm-bench", "-m", params.model_tag, "-w", ws, "build", + "--tp_size", str(params.tp_size), "--pp_size", str(params.pp_size), + "--max_seq_len", str(params.max_seq_len), + "--max_batch_size", str(params.max_batch_size), + "--max_num_tokens", str(params.max_num_tokens), + ] + if params.quantization: + cmd += ["--quantization", params.quantization] + + logger.info("Building TRT engine [%s]: %s", params.cache_key(), " ".join(cmd)) + proc = subprocess.run(cmd, capture_output=True, text=True, timeout=BUILD_TIMEOUT_S) + if proc.returncode != 0: + shutil.rmtree(ws, ignore_errors=True) + tail = (proc.stderr or proc.stdout or "").strip()[-800:] + raise RuntimeError(f"trtllm-bench build failed (rc={proc.returncode}): {tail}") + + produced = glob.glob(os.path.join(ws, "**", "rank0.engine"), recursive=True) + if not produced: + shutil.rmtree(ws, ignore_errors=True) + raise RuntimeError("trtllm-bench build reported success but produced no engine") + engine_src = os.path.dirname(produced[0]) + + shutil.rmtree(final_dir, ignore_errors=True) + os.makedirs(os.path.dirname(final_dir), exist_ok=True) + shutil.move(engine_src, final_dir) + shutil.rmtree(ws, ignore_errors=True) + + manifest = { + "model_tag": params.model_tag, + "trtllm_version": _trtllm_version(), + "compute_capability": _gpu_compute_cap(), + "tp_size": params.tp_size, "pp_size": params.pp_size, + "max_seq_len": params.max_seq_len, "max_batch_size": params.max_batch_size, + "max_num_tokens": params.max_num_tokens, "quantization": params.quantization, + "cache_key": params.cache_key(), "built_at": time.time(), + } + with open(os.path.join(final_dir, "manifest.json"), "w", encoding="utf-8") as f: + json.dump(manifest, f, indent=2) + logger.info("Built TRT engine -> %s", final_dir) + return final_dir + + +@dataclass +class TrtBuildJob: + key: str # cache key (also the engine dir name) + params: dict # BuildParams as a dict, for the UI + state: str = "pending" # pending | building | completed | failed + engine_dir: Optional[str] = None + error: Optional[str] = None + started_at: float = field(default_factory=time.time) + updated_at: float = field(default_factory=time.time) + + +class TrtConvertManager: + """In-memory HF->TRT-engine build jobs, keyed by cache key. Mirrors LoraConvertManager.""" + + def __init__(self) -> None: + self._jobs: dict[str, TrtBuildJob] = {} + self._tasks: dict[str, asyncio.Task] = {} + + def available(self) -> bool: + return bench_available() + + def list(self) -> list[dict]: + return [asdict(j) for j in sorted(self._jobs.values(), key=lambda j: j.started_at, reverse=True)] + + def list_engines(self) -> list[dict]: + """Built engines on disk (a manifest.json under the engines root). Blocking.""" + out: list[dict] = [] + for mf in glob.glob(os.path.join(engines_root(), "*", "manifest.json")): + try: + with open(mf, encoding="utf-8") as f: + m = json.load(f) + except (OSError, json.JSONDecodeError): + continue + m["engine_dir"] = os.path.dirname(mf) + out.append(m) + return sorted(out, key=lambda m: m.get("built_at", 0), reverse=True) + + def start(self, params: BuildParams) -> dict: + _validate(params) + key = params.cache_key() + existing = self._jobs.get(key) + if existing and existing.state in ("pending", "building"): + return asdict(existing) # already in flight — idempotent + job = TrtBuildJob(key=key, params=asdict(params)) + self._jobs[key] = job + self._tasks[key] = asyncio.create_task(self._run(job, params)) + return asdict(job) + + async def _run(self, job: TrtBuildJob, params: BuildParams) -> None: + loop = asyncio.get_event_loop() + job.state = "building" + job.updated_at = time.time() + try: + job.engine_dir = await loop.run_in_executor(None, build, params) + job.state = "completed" + logger.info("Built TRT engine %s -> %s", job.key, job.engine_dir) + except Exception as e: + job.state = "failed" + job.error = str(e) + logger.warning("TRT engine build failed for %s: %s", job.key, e) + finally: + job.updated_at = time.time() + self._tasks.pop(job.key, None) diff --git a/apps/backend/tests/api/test_trt_routes.py b/apps/backend/tests/api/test_trt_routes.py new file mode 100644 index 0000000..ae1beb8 --- /dev/null +++ b/apps/backend/tests/api/test_trt_routes.py @@ -0,0 +1,65 @@ +"""TRT engine build API (app/api/trt.py).""" +import pytest + +from app.services import trt_convert as T + +pytestmark = pytest.mark.api + + +def test_list_engines_reports_availability_and_quantizations(client, monkeypatch): + monkeypatch.setattr(T, "bench_available", lambda: True) + r = client.get("/api/trt/engines") + assert r.status_code == 200 + body = r.json() + assert body["available"] is True + assert "FP8" in body["quantizations"] + assert isinstance(body["engines"], list) + + +def test_convert_unavailable_returns_503(auth_client, monkeypatch): + monkeypatch.setattr(T, "bench_available", lambda: False) + r = auth_client.post("/api/trt/convert", json={"model_tag": "Qwen/Qwen3-0.6B"}, + headers={"Authorization": "Bearer secret-admin"}) + assert r.status_code == 503 + + +def test_convert_requires_operator(auth_client): + # Auth is on (auth_client sets admin_token) and no token is sent -> the + # require_operator dependency rejects before reaching the manager. + r = auth_client.post("/api/trt/convert", json={"model_tag": "Qwen/Qwen3-0.6B"}) + assert r.status_code in (401, 403) + + +def test_convert_starts_job_and_passes_params(auth_client, monkeypatch): + monkeypatch.setattr(T, "bench_available", lambda: True) + monkeypatch.setattr(T, "_gpu_compute_cap", lambda: "86") + monkeypatch.setattr(T, "_trtllm_version", lambda: "1.3.0rc20") + captured = {} + + def fake_start(self, params): + captured["params"] = params + return {"key": params.cache_key(), "params": {}, "state": "pending"} + + monkeypatch.setattr(T.TrtConvertManager, "start", fake_start) + r = auth_client.post( + "/api/trt/convert", + json={"model_tag": "Qwen/Qwen3-0.6B", "max_batch_size": 8, + "max_num_tokens": 16384, "quantization": "FP8"}, + headers={"Authorization": "Bearer secret-admin"}, + ) + assert r.status_code == 202 + p = captured["params"] + assert p.model_tag == "Qwen/Qwen3-0.6B" and p.max_batch_size == 8 + assert p.max_num_tokens == 16384 and p.quantization == "FP8" + + +def test_convert_rejects_bad_quantization(auth_client, monkeypatch): + monkeypatch.setattr(T, "bench_available", lambda: True) + monkeypatch.setattr(T, "_gpu_compute_cap", lambda: "86") + monkeypatch.setattr(T, "_trtllm_version", lambda: "1.3.0rc20") + r = auth_client.post( + "/api/trt/convert", + json={"model_tag": "Qwen/Qwen3-0.6B", "quantization": "BOGUS"}, + headers={"Authorization": "Bearer secret-admin"}, + ) + assert r.status_code == 400 diff --git a/apps/backend/tests/conftest.py b/apps/backend/tests/conftest.py index c4e115f..4739f1e 100644 --- a/apps/backend/tests/conftest.py +++ b/apps/backend/tests/conftest.py @@ -572,6 +572,8 @@ def app(monkeypatch): application.state.lora_download_manager = LoraDownloadManager() from app.services.lora_convert import LoraConvertManager application.state.lora_convert_manager = LoraConvertManager() + from app.services.trt_convert import TrtConvertManager + application.state.trt_convert_manager = TrtConvertManager() application.state.perf_manager = PerfManager( store, application.state.manager, settings, str(BACKEND_ROOT), "http://127.0.0.1:8887" ) diff --git a/apps/backend/tests/unit/test_trt_convert.py b/apps/backend/tests/unit/test_trt_convert.py new file mode 100644 index 0000000..9e65351 --- /dev/null +++ b/apps/backend/tests/unit/test_trt_convert.py @@ -0,0 +1,142 @@ +"""HF -> TensorRT-LLM engine build service (app/services/trt_convert.py). + +The build subprocess (trtllm-bench) is mocked; these assert the wiring: cache key, +validation, the build's move-and-manifest, idempotent reuse, and engine listing. +""" +import json +import os + +import pytest + +from app.services import trt_convert as T +from app.services.trt_convert import BuildParams, TrtConvertManager + +pytestmark = pytest.mark.unit + + +@pytest.fixture +def env(tmp_path, monkeypatch): + monkeypatch.setenv("LLMOPS_TRT_ENGINES_DIR", str(tmp_path)) + # Deterministic cache key (no real GPU / package lookup). + monkeypatch.setattr(T, "_gpu_compute_cap", lambda: "86") + monkeypatch.setattr(T, "_trtllm_version", lambda: "1.3.0rc20") + monkeypatch.setattr(T, "bench_available", lambda: True) + return tmp_path + + +def test_cache_key_encodes_every_build_param(env): + k = BuildParams("Qwen/Qwen3-0.6B", tp_size=2, max_seq_len=4096, max_batch_size=8, + max_num_tokens=16384, quantization="FP8").cache_key() + # model sanitized (no slash), and each shape/quant/arch/version present + assert "Qwen_Qwen3-0.6B" in k and "tp2pp1" in k and "seq4096" in k + assert "bs8" in k and "nt16384" in k and "fp8" in k and "sm86" in k and "trt1.3.0rc20" in k + # a different knob -> a different key (no cache collision) + assert BuildParams("Qwen/Qwen3-0.6B", max_seq_len=2048).cache_key() != \ + BuildParams("Qwen/Qwen3-0.6B", max_seq_len=4096).cache_key() + + +def test_validate_rejects_bad_quant_and_dims(env): + with pytest.raises(ValueError, match="quantization"): + T._validate(BuildParams("m", quantization="BOGUS")) + with pytest.raises(ValueError, match="max_seq_len"): + T._validate(BuildParams("m", max_seq_len=0)) + with pytest.raises(ValueError, match="model_tag"): + T._validate(BuildParams(" ")) + + +def test_build_unavailable_raises(env, monkeypatch): + monkeypatch.setattr(T, "bench_available", lambda: False) + with pytest.raises(RuntimeError, match="trtllm-bench is not available"): + T.build(BuildParams("Qwen/Qwen3-0.6B")) + + +def _fake_bench(engine_leaf="rank0.engine"): + """Return a subprocess.run stand-in that plants a produced engine under the -w + workspace (mimicking trtllm-bench's //tp_1_pp_1/rank0.engine).""" + def run(cmd, **kwargs): + ws = cmd[cmd.index("-w") + 1] + out = os.path.join(ws, "Model", "tp_1_pp_1") + os.makedirs(out, exist_ok=True) + open(os.path.join(out, engine_leaf), "w").close() + open(os.path.join(out, "config.json"), "w").close() + + class R: + returncode = 0 + stdout = "ENGINE SAVED" + stderr = "" + return R() + return run + + +def test_build_moves_engine_and_writes_manifest(env, monkeypatch): + monkeypatch.setattr(T.subprocess, "run", _fake_bench()) + params = BuildParams("Qwen/Qwen3-0.6B", max_seq_len=2048) + engine_dir = T.build(params) + + assert engine_dir == T.engine_dir_for(params) + assert os.path.isfile(os.path.join(engine_dir, "rank0.engine")) + with open(os.path.join(engine_dir, "manifest.json")) as f: + m = json.load(f) + assert m["model_tag"] == "Qwen/Qwen3-0.6B" + assert m["compute_capability"] == "86" and m["trtllm_version"] == "1.3.0rc20" + assert m["max_seq_len"] == 2048 and m["cache_key"] == params.cache_key() + # the temp build workspace was cleaned up + assert not os.path.isdir(os.path.join(str(env), ".trt-build", params.cache_key())) + + +def test_build_is_idempotent_reuses_existing(env, monkeypatch): + calls = {"n": 0} + real = _fake_bench() + + def counting(cmd, **kwargs): + calls["n"] += 1 + return real(cmd, **kwargs) + + monkeypatch.setattr(T.subprocess, "run", counting) + params = BuildParams("Qwen/Qwen3-0.6B") + T.build(params) + T.build(params) # second call must hit the cache, not rebuild + assert calls["n"] == 1 + + +def test_build_failure_surfaces_stderr(env, monkeypatch): + def failing(cmd, **kwargs): + class R: + returncode = 1 + stdout = "" + stderr = "boom: unsupported architecture" + return R() + monkeypatch.setattr(T.subprocess, "run", failing) + with pytest.raises(RuntimeError, match="unsupported architecture"): + T.build(BuildParams("Weird/Model")) + + +def test_list_engines_reads_manifests(env, monkeypatch): + monkeypatch.setattr(T.subprocess, "run", _fake_bench()) + T.build(BuildParams("Qwen/Qwen3-0.6B")) + engines = TrtConvertManager().list_engines() + assert len(engines) == 1 + assert engines[0]["model_tag"] == "Qwen/Qwen3-0.6B" + assert engines[0]["engine_dir"] == T.engine_dir_for(BuildParams("Qwen/Qwen3-0.6B")) + + +async def test_manager_start_runs_build_and_completes(env, monkeypatch): + monkeypatch.setattr(T.subprocess, "run", _fake_bench()) + mgr = TrtConvertManager() + job = mgr.start(BuildParams("Qwen/Qwen3-0.6B")) + assert job["state"] in ("pending", "building") + await mgr._tasks[job["key"]] # await the build task + done = mgr.list()[0] + assert done["state"] == "completed" + assert done["engine_dir"] == T.engine_dir_for(BuildParams("Qwen/Qwen3-0.6B")) + + +async def test_manager_start_idempotent_while_in_flight(env, monkeypatch): + monkeypatch.setattr(T.subprocess, "run", _fake_bench()) + mgr = TrtConvertManager() + p = BuildParams("Qwen/Qwen3-0.6B") + j1 = mgr.start(p) + j2 = mgr.start(p) # same key, still in flight -> same job, no duplicate task + assert j1["key"] == j2["key"] + assert len(mgr.list()) == 1 + await mgr._tasks[j1["key"]] From 93f9fa126f01cb34b2ea8896197e67746f212baf Mon Sep 17 00:00:00 2001 From: max Date: Sun, 5 Jul 2026 00:46:04 +0800 Subject: [PATCH 12/20] feat(trtllm): convert-to-TRT UI in the model library MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Frontend for the convert-to-TRT backend (fee0e92). Each cached HF model in the Model Library gets a "Convert to TRT engine" action that builds a TensorRT-LLM engine for it, mirroring the LoRA GGUF convert UX. - lib/api.ts + types/api.ts: listTrtEngines / listTrtConversions / convertToTrt and the TrtBuildJob / TrtEngine / TrtLibraryInfo types. - LibraryView.vue: a Cpu-icon button per LLM card (gated on trt.available and the model being an LLM) opens a params dialog — max_seq_len, max_batch_size, max_num_tokens, tensor-parallel, and quantization (options from the API). Active builds show a spinner + status line; already-built engines show a "TRT · seq · bs" badge. Polls conversions alongside downloads. - i18n en + zh-TW. Verified live end-to-end (dashboard on trtllm-backend): the button + built-engine badge render for a cached model, and POST via nginx builds/reuses an engine (cache hit completes in ~2s). Gating verified: on a non-trtllm dashboard backend /api/trt/engines returns available=false and the button hides. Frontend typecheck (vue-tsc) passes. Note: the action is only available when the dashboard API is served by the TensorRT-LLM backend (it ships trtllm-bench) — i.e. DASHBOARD_BACKEND=trtllm-backend in the mixed stack, same as GGUF convert only working on the llama.cpp-capable image. Co-Authored-By: Claude Opus 4.8 --- apps/frontend_llmops/src/i18n/locales/en.ts | 14 ++ .../frontend_llmops/src/i18n/locales/zh-TW.ts | 14 ++ apps/frontend_llmops/src/lib/api.ts | 20 ++ apps/frontend_llmops/src/types/api.ts | 33 +++ .../frontend_llmops/src/views/LibraryView.vue | 189 ++++++++++++++++-- 5 files changed, 258 insertions(+), 12 deletions(-) diff --git a/apps/frontend_llmops/src/i18n/locales/en.ts b/apps/frontend_llmops/src/i18n/locales/en.ts index d4a60f7..79fcc78 100644 --- a/apps/frontend_llmops/src/i18n/locales/en.ts +++ b/apps/frontend_llmops/src/i18n/locales/en.ts @@ -522,6 +522,20 @@ export default { downloading: 'Downloading', downloadComplete: 'Done', downloadFailedBadge: 'Failed', + convertToTrt: 'Convert to TRT engine', + trtDialogHint: 'Build a TensorRT-LLM engine from this model for the lowest TTFT. The engine is bound to this GPU + TRT-LLM version; a build can take minutes.', + trtMaxSeqLen: 'Max sequence length', + trtMaxBatchSize: 'Max batch size', + trtMaxNumTokens: 'Max batched tokens', + trtTpSize: 'Tensor parallel', + trtQuantization: 'Quantization', + trtQuantNone: 'None (fp16)', + trtStartBuild: 'Build engine', + trtStarted: 'TRT build started for {repo}', + trtStartedDesc: 'Building the engine — this runs in the background and can take minutes.', + trtFailed: 'Failed to start TRT build', + trtBuilding: 'Building TRT engine…', + trtBuildFailed: 'TRT build failed', }, // ---- LoRA Library ---- diff --git a/apps/frontend_llmops/src/i18n/locales/zh-TW.ts b/apps/frontend_llmops/src/i18n/locales/zh-TW.ts index 6b855db..1113da5 100644 --- a/apps/frontend_llmops/src/i18n/locales/zh-TW.ts +++ b/apps/frontend_llmops/src/i18n/locales/zh-TW.ts @@ -510,6 +510,20 @@ export default { downloading: '下載中', downloadComplete: '完成', downloadFailedBadge: '失敗', + convertToTrt: '轉成 TRT engine', + trtDialogHint: '從這個模型建置 TensorRT-LLM engine 以取得最低 TTFT。engine 綁定此 GPU + TRT-LLM 版本;建置可能需要數分鐘。', + trtMaxSeqLen: '最大序列長度', + trtMaxBatchSize: '最大批次大小', + trtMaxNumTokens: '最大批次 token 數', + trtTpSize: '張量平行', + trtQuantization: '量化', + trtQuantNone: '無 (fp16)', + trtStartBuild: '開始建置', + trtStarted: '已開始建置 {repo} 的 TRT engine', + trtStartedDesc: '正在背景建置 engine,可能需要數分鐘。', + trtFailed: '無法開始 TRT 建置', + trtBuilding: '建置 TRT engine 中…', + trtBuildFailed: 'TRT 建置失敗', }, loraLibrary: { diff --git a/apps/frontend_llmops/src/lib/api.ts b/apps/frontend_llmops/src/lib/api.ts index bcd77d5..0bf1aa8 100644 --- a/apps/frontend_llmops/src/lib/api.ts +++ b/apps/frontend_llmops/src/lib/api.ts @@ -41,6 +41,8 @@ import type { LoraLibraryInfo, ModelStartupMetrics, ModelView, + TrtBuildJob, + TrtLibraryInfo, OpenAIModelList, ParsedModel, PerfRequest, @@ -430,6 +432,24 @@ export const api = { body: JSON.stringify({ base_model: baseModel }), }), + // ---- TensorRT-LLM engine build ("convert to TRT") ------------------------- + listTrtEngines: () => request(API_BASE, '/api/trt/engines'), + listTrtConversions: () => + request<{ available: boolean; jobs: TrtBuildJob[] }>(API_BASE, '/api/trt/conversions'), + convertToTrt: (body: { + model_tag: string + tp_size?: number + pp_size?: number + max_seq_len?: number + max_batch_size?: number + max_num_tokens?: number + quantization?: string | null + }) => + request(API_BASE, '/api/trt/convert', { + method: 'POST', + body: JSON.stringify(body), + }), + // ---- Benchmark datasets (ModelScope cache) -------------------------------- getDatasets: () => request(API_BASE, '/api/datasets'), listDatasetDownloads: () => request(API_BASE, '/api/datasets/downloads'), diff --git a/apps/frontend_llmops/src/types/api.ts b/apps/frontend_llmops/src/types/api.ts index ba5dc52..426bb5f 100644 --- a/apps/frontend_llmops/src/types/api.ts +++ b/apps/frontend_llmops/src/types/api.ts @@ -451,6 +451,39 @@ export interface CacheInfo { models: CachedModel[] } +// ---- TensorRT-LLM engine build ("convert to TRT") -------------------------- +export interface TrtBuildParams { + model_tag: string + tp_size: number + pp_size: number + max_seq_len: number + max_batch_size: number + max_num_tokens: number + quantization: string | null +} +export interface TrtBuildJob { + key: string + params: TrtBuildParams + state: 'pending' | 'building' | 'completed' | 'failed' + engine_dir: string | null + error: string | null + started_at: number + updated_at: number +} +export interface TrtEngine extends TrtBuildParams { + trtllm_version: string + compute_capability: string + cache_key: string + built_at: number + engine_dir: string +} +export interface TrtLibraryInfo { + available: boolean + root: string + quantizations: string[] + engines: TrtEngine[] +} + // ---- Benchmark datasets (ModelScope cache) ---- export interface DatasetEntry { key: string diff --git a/apps/frontend_llmops/src/views/LibraryView.vue b/apps/frontend_llmops/src/views/LibraryView.vue index 83b3b3f..7ad2451 100644 --- a/apps/frontend_llmops/src/views/LibraryView.vue +++ b/apps/frontend_llmops/src/views/LibraryView.vue @@ -1,17 +1,18 @@ + + diff --git a/apps/frontend_llmops/src/components/ui/Select.vue b/apps/frontend_llmops/src/components/ui/Select.vue new file mode 100644 index 0000000..20acf76 --- /dev/null +++ b/apps/frontend_llmops/src/components/ui/Select.vue @@ -0,0 +1,26 @@ + + + diff --git a/apps/frontend_llmops/src/components/ui/Sheet.vue b/apps/frontend_llmops/src/components/ui/Sheet.vue index 9b102cc..b57b9e3 100644 --- a/apps/frontend_llmops/src/components/ui/Sheet.vue +++ b/apps/frontend_llmops/src/components/ui/Sheet.vue @@ -25,7 +25,8 @@ const open = defineModel('open', { default: false })
{{ title }} diff --git a/apps/frontend_llmops/src/components/ui/Skeleton.vue b/apps/frontend_llmops/src/components/ui/Skeleton.vue new file mode 100644 index 0000000..5cecacc --- /dev/null +++ b/apps/frontend_llmops/src/components/ui/Skeleton.vue @@ -0,0 +1,10 @@ + + + diff --git a/apps/frontend_llmops/src/components/ui/variants.ts b/apps/frontend_llmops/src/components/ui/variants.ts index 47bfa73..2bbe8e8 100644 --- a/apps/frontend_llmops/src/components/ui/variants.ts +++ b/apps/frontend_llmops/src/components/ui/variants.ts @@ -1,7 +1,7 @@ import { cva, type VariantProps } from 'class-variance-authority' export const buttonVariants = cva( - "inline-flex items-center justify-center gap-2 whitespace-nowrap rounded-md text-sm font-medium transition-all focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-1 focus-visible:ring-offset-background disabled:pointer-events-none disabled:opacity-50 cursor-pointer [&_svg]:pointer-events-none [&_svg]:size-4 [&_svg]:shrink-0", + "inline-flex items-center justify-center gap-2 whitespace-nowrap rounded-md text-sm font-medium transition-all duration-150 ease-out focus-visible:outline-none focus-visible:ring-2 focus-visible:ring-ring focus-visible:ring-offset-1 focus-visible:ring-offset-background disabled:pointer-events-none disabled:opacity-50 cursor-pointer motion-safe:active:scale-[0.98] [&_svg]:pointer-events-none [&_svg]:size-4 [&_svg]:shrink-0", { variants: { variant: { diff --git a/apps/frontend_llmops/src/i18n/locales/en.ts b/apps/frontend_llmops/src/i18n/locales/en.ts index 93aa759..936b912 100644 --- a/apps/frontend_llmops/src/i18n/locales/en.ts +++ b/apps/frontend_llmops/src/i18n/locales/en.ts @@ -18,6 +18,7 @@ export default { stop: 'Stop', clear: 'Clear', clearAll: 'Clear all', + clearFilters: 'Clear filters', search: 'Search', all: 'All', none: 'None', @@ -150,6 +151,7 @@ export default { noMatch: 'No matching models found.', clearFilter: 'Try clearing the filter.', noConfig: 'No models configured in config.yaml.', + noConfigHint: 'Add your first model to start serving — pick a downloaded checkpoint or a Hugging Face repo id.', }, // ---- Traffic ---- @@ -206,6 +208,7 @@ export default { // ---- Monitoring ---- monitoring: { + selectGroup: 'Model group', overview: 'Overview', autoscaling: 'Autoscaling', vllmCapacity: 'vLLM Capacity', @@ -507,6 +510,7 @@ export default { repoId: 'Hugging Face repo id', repoPlaceholder: 'e.g. Qwen/Qwen3-0.6B', noCachedModels: 'No cached models.', + noCachedModelsHint: 'Download a model from Hugging Face above to get started.', files: 'files', size: 'Size', updated: 'Updated', @@ -540,6 +544,7 @@ export default { ggufModels: 'GGUF (llama.cpp)', trtEnginesTitle: 'TensorRT-LLM engines', noTrtEngines: 'No built engines yet — use the chip icon on a model to build one.', + noTrtEnginesHint: 'Convert a cached model above to build a TensorRT-LLM engine.', trtDelete: 'Delete engine', trtDeleteConfirm: 'Delete the TensorRT engine built from {model}?', trtDeleted: 'Engine deleted', @@ -623,6 +628,7 @@ export default { requestCount: 'requests', lastUsed: 'Last used: ', noKeys: 'No keys yet.', + noKeysHint: 'Create a key above to give clients authenticated access to the router.', revokeTitle: 'Revoke key', keyCreated: 'Key created', copyImmediate: 'Copy this key now — it will not be shown again after closing.', @@ -713,6 +719,7 @@ export default { hint: 'Slack/Discord get formatted messages; webhook gets raw JSON. The URL is stored but only shown masked.', sinks: 'Destinations', none: 'No destinations. Add one above, or set LLMOPS_ALERT_* env vars.', + noneHint: 'Alerts for model failures and state changes will be delivered to every destination at or above its severity.', envTitle: 'Configured via environment — not editable here', testAll: 'Test all', testOne: 'Send test', diff --git a/apps/frontend_llmops/src/i18n/locales/zh-TW.ts b/apps/frontend_llmops/src/i18n/locales/zh-TW.ts index 53ee859..72eb7e2 100644 --- a/apps/frontend_llmops/src/i18n/locales/zh-TW.ts +++ b/apps/frontend_llmops/src/i18n/locales/zh-TW.ts @@ -17,6 +17,7 @@ export default { stop: '停止', clear: '清除', clearAll: '清除全部', + clearFilters: '清除篩選', search: '搜尋', all: '全部', none: '無', @@ -145,6 +146,7 @@ export default { noMatch: '找不到符合的模型。', clearFilter: '試著清除篩選條件。', noConfig: 'config.yaml 中尚未設定模型。', + noConfigHint: '新增第一個模型即可開始服務 —— 可挑選已下載的模型或輸入 Hugging Face repo id。', }, traffic: { @@ -198,6 +200,7 @@ export default { }, monitoring: { + selectGroup: '模型群組', overview: '總覽', autoscaling: '自動擴縮', vllmCapacity: 'vLLM 容量', @@ -495,6 +498,7 @@ export default { repoId: 'Hugging Face repo id', repoPlaceholder: '例如:Qwen/Qwen3-0.6B', noCachedModels: '快取中尚無模型。', + noCachedModelsHint: '在上方輸入 Hugging Face repo id 下載模型即可開始。', files: '檔', size: '大小', updated: '更新', @@ -528,6 +532,7 @@ export default { ggufModels: 'GGUF (llama.cpp)', trtEnginesTitle: 'TensorRT-LLM 引擎', noTrtEngines: '尚無已建置的引擎 —— 在模型上點晶片圖示即可建置。', + noTrtEnginesHint: '將上方已快取的模型轉換為 TensorRT-LLM 引擎。', trtDelete: '刪除引擎', trtDeleteConfirm: '確定刪除從 {model} 建置的 TensorRT 引擎?', trtDeleted: '已刪除引擎', @@ -608,6 +613,7 @@ export default { requestCount: '次', lastUsed: '最後使用:', noKeys: '尚無金鑰。', + noKeysHint: '在上方建立金鑰,讓用戶端以驗證方式存取路由器。', revokeTitle: '撤銷金鑰', keyCreated: '金鑰已建立', copyImmediate: '請立即複製此金鑰,關閉後將無法再次顯示。', @@ -698,6 +704,7 @@ export default { hint: 'Slack/Discord 收格式化訊息;webhook 收原始 JSON。URL 會儲存但只顯示遮罩版。', sinks: '推送目標', none: '尚無推送目標。在上方新增,或設定 LLMOPS_ALERT_* 環境變數。', + noneHint: '模型故障與狀態變更的警報,會送到達到嚴重度門檻的每個目標。', envTitle: '由環境變數設定 —— 此處不可編輯', testAll: '全部測試', testOne: '送測試', diff --git a/apps/frontend_llmops/src/views/ActivityView.vue b/apps/frontend_llmops/src/views/ActivityView.vue index b2f6609..a5d949d 100644 --- a/apps/frontend_llmops/src/views/ActivityView.vue +++ b/apps/frontend_llmops/src/views/ActivityView.vue @@ -1,11 +1,13 @@