-
{{ group }}
+
+ {{ group }}
+ {{ engine }}
+
{{ subtitle }}
diff --git a/apps/frontend_llmops/src/i18n/locales/en.ts b/apps/frontend_llmops/src/i18n/locales/en.ts
index 737fb3d..8a64664 100644
--- a/apps/frontend_llmops/src/i18n/locales/en.ts
+++ b/apps/frontend_llmops/src/i18n/locales/en.ts
@@ -211,6 +211,9 @@ export default {
vllmCapacity: 'vLLM Capacity',
vllmPerf: 'vLLM Performance',
vllmQuery: 'vLLM Queries',
+ sglang: 'SGLang',
+ sglangDashboard: 'Dashboard',
+ shared: 'Shared',
gpu: 'GPU',
host: 'Host',
openGrafana: 'Open in Grafana',
@@ -864,8 +867,10 @@ export default {
addModel: {
createTitle: 'Add Model',
editTitle: 'Edit Model',
- pasteCommand: 'Paste vLLM launch command',
+ pasteCommand: 'Paste launch command',
parseCommand: 'Parse command',
+ or: 'or',
+ manualEntry: 'Manual entry',
parseFailed: 'Failed to parse command',
groupLabel: 'Group',
groupExists: 'Group already exists — will add as a new replica.',
@@ -877,6 +882,8 @@ export default {
gpuLabel: 'GPU (cuda_device)',
gpuAuto: 'None / Auto',
modelTagLabel: 'Model tag',
+ engineLabel: 'Inference engine',
+ engineHint: 'Each engine uses different launch flags (the launcher translates); needs a backend that runs that engine to start',
routingLabel: 'Routing strategy (load balancing)',
routingDefault: 'Follow global default',
routingHint: 'Traffic distribution policy for this group; empty inherits the global setting (switchable on the Traffic page). Only effective with multiple replicas.',
@@ -884,7 +891,7 @@ export default {
kvShareDesc: 'Replicas within the group share a KV store (/kv_cache) to reuse computed KV across instances. Saves re-computation for matching prefixes. Best with multiple replicas + high prefix overlap (fixed system prompts / RAG / multi-turn chat); off = each instance has independent KV.',
sleepModeLabel: 'Enable Sleep Mode (warm standby)',
sleepModeDesc: 'Launch with --enable-sleep-mode so an idle instance can level-1 sleep: VRAM is freed but wake stays seconds-fast (not a minutes-long cold start). Foundation for the autoscaling warm-standby tier; off = only running / stopped.',
- groupSharedWarn: 'vLLM parameters are group-shared — this change applies to all {n} replicas in group {group}.',
+ groupSharedWarn: 'Engine parameters are group-shared — this change applies to all {n} replicas in group {group}.',
weightsCached: 'Weights cached',
weightsDownloading: 'Downloading weights…',
weightsDownloadFailed: 'Download failed: ',
@@ -899,7 +906,7 @@ export default {
ngramSpec: 'N-gram speculative decoding (reduces single-request latency at low QPS)',
offloadHint: 'Offloads weights to CPU RAM for "it runs" (slower), not acceleration.',
partialPrefillHint: 'With mixed short + long prompts: set partial >1 and long small so short requests can cut in line.',
- vllmParams: 'vLLM parameters (model_config)',
+ vllmParams: 'Engine parameters (model_config)',
addParam: 'Add',
noExtraParams: 'No extra parameters.',
flagPlaceholder: 'Flag (snake_case)',
@@ -956,6 +963,13 @@ export default {
hashXxhash: 'xxhash (faster, not cryptographically secure)',
qwen3Thinking: 'Qwen3 (with thinking)',
toolCallingDocRef: 'See full reference in docs/vllm_auto_tool_reference.md. No matching parser = don\'t add it.',
+ // --- SGLang variants (the two engines support different things) ---
+ accelTitleSglang: '⚡ Acceleration settings (SGLang inference params)',
+ accelHintSglang: 'Empty = SGLang default. Changes require model restart. Radix cache (prefix cache) is on by default.',
+ sglSchedLpm: 'longest-prefix first',
+ toolAuto: 'Auto-detect (auto)',
+ toolCallingDescSglang: 'SGLang enables tool calling just by setting tool_call_parser=
(no enable_auto_tool_choice); reasoning models also need reasoning_parser. Use auto to detect from the chat template. Parser must match the model output format.',
+ toolCallingDocRefSglang: 'Parser list from sglang.launch_server --tool-call-parser / --reasoning-parser. No matching parser = don\'t add it.',
},
// ---- ModelDetailDrawer ----
@@ -987,7 +1001,7 @@ export default {
gpuMemUtilDesc: 'Newer vLLM includes CUDA graph memory in this budget. You set {current}; after deducting, the effective KV cache space equals {effective} in older versions. To increase KV cache (higher concurrency / longer context), edit to {suggested} after stopping — but VRAM headroom decreases and OOM risk increases (especially on small GPUs).',
servedModels: 'Served models',
routingPolicy: 'Routing policy (load balancing)',
- vllmParams: 'vLLM parameters (model_config)',
+ vllmParams: 'Engine parameters (model_config)',
loraAdapters: 'LoRA Adapters',
hotLoadEnabled: 'Hot-load enabled',
hotLoad: 'Load',
@@ -1021,7 +1035,7 @@ export default {
stopLabel: 'Stop',
terminateLabel: 'Terminate',
abortLabel: 'Abort startup',
- editParams: 'Edit parameters (must stop first; vLLM params are group-shared)',
+ editParams: 'Edit parameters (must stop first; engine params are group-shared)',
removeModel: 'Remove model (dynamic models only)',
removeSuccess: 'Removed {key}',
removeFailed: 'Failed to remove model',
@@ -1172,7 +1186,7 @@ export default {
addInstance: {
title: 'Add Instance',
sharedSettingsTitle: 'Inherited shared settings from this group',
- sharedSettingsHint: 'All instances in this group share vLLM parameters; only set the location for this instance here.',
+ sharedSettingsHint: 'All instances in this group share engine parameters; only set the location for this instance here.',
instanceIdLabel: 'Instance ID',
instanceIdPlaceholder: 'e.g. qwen3-5',
idConflict: 'This ID already exists in the group.',
diff --git a/apps/frontend_llmops/src/i18n/locales/zh-TW.ts b/apps/frontend_llmops/src/i18n/locales/zh-TW.ts
index 98e1f30..212b8da 100644
--- a/apps/frontend_llmops/src/i18n/locales/zh-TW.ts
+++ b/apps/frontend_llmops/src/i18n/locales/zh-TW.ts
@@ -203,6 +203,9 @@ export default {
vllmCapacity: 'vLLM 容量',
vllmPerf: 'vLLM 效能',
vllmQuery: 'vLLM 請求',
+ sglang: 'SGLang',
+ sglangDashboard: '儀表板',
+ shared: '共用',
gpu: 'GPU',
host: '主機',
openGrafana: '在 Grafana 開啟',
@@ -843,8 +846,10 @@ export default {
addModel: {
createTitle: '新增模型',
editTitle: '編輯模型',
- pasteCommand: '貼上 vLLM 啟動指令',
+ pasteCommand: '貼上啟動指令',
parseCommand: '解析指令',
+ or: '或',
+ manualEntry: '手動填寫',
parseFailed: '無法解析指令',
groupLabel: '群組',
groupExists: '已存在群組 — 將新增為新副本。',
@@ -856,6 +861,8 @@ export default {
gpuLabel: 'GPU(cuda_device)',
gpuAuto: '無 / 自動',
modelTagLabel: '模型標籤',
+ engineLabel: '推理引擎',
+ engineHint: '不同引擎用不同啟動參數(由 launcher 自動翻譯);需有對應引擎的 backend 才能啟動',
routingLabel: '路由策略(負載平衡)',
routingDefault: '跟隨全域預設',
routingHint: '此群組請求的分流方式;留空則跟隨全域設定(可在「流量」頁切換)。多副本才有效。',
@@ -863,7 +870,7 @@ export default {
kvShareDesc: '同群組各副本透過共享 store(/kv_cache)重用彼此算過的 KV,相同前綴不必重算。多副本 + 高前綴重複率(固定 system prompt / RAG / 多輪對話)效益最大;關閉則各副本各自獨立 KV。',
sleepModeLabel: '啟用 Sleep Mode(暖待命)',
sleepModeDesc: '以 --enable-sleep-mode 啟動:閒置時可讓實例進入 level-1 睡眠,釋放 VRAM 但保留快速喚醒(秒級,非數分鐘冷啟)。為自動擴縮的暖待命階奠基;關閉則只有 running / stopped 兩態。',
- groupSharedWarn: 'vLLM 參數為群組共用 — 此變更將套用至群組 {group} 的全部 {n} 個副本。',
+ groupSharedWarn: '引擎參數為群組共用 — 此變更將套用至群組 {group} 的全部 {n} 個副本。',
weightsCached: '權重已快取',
weightsDownloading: '下載權重中…',
weightsDownloadFailed: '下載失敗:',
@@ -878,7 +885,7 @@ export default {
ngramSpec: 'N-gram 推測解碼(低 QPS 降單請求延遲)',
offloadHint: '把權重 offload 到 CPU RAM 換「跑得起來」(會變慢),非加速。',
partialPrefillHint: '同時有短問題+長 prompt 時:把 partial 設 >1 且 long 設小,短請求可插隊不被長的卡住。',
- vllmParams: 'vLLM 參數(model_config)',
+ vllmParams: '引擎參數(model_config)',
addParam: '新增',
noExtraParams: '無額外參數。',
flagPlaceholder: '旗標(snake_case)',
@@ -935,6 +942,13 @@ export default {
hashXxhash: 'xxhash(較快,非密碼安全)',
qwen3Thinking: 'Qwen3(含 thinking)',
toolCallingDocRef: '完整對照見 docs/vllm_auto_tool_整理.md。沒有對應 parser 的模型請勿亂加。',
+ // --- SGLang 變體(兩引擎支援情況不同,分開呈現)---
+ accelTitleSglang: '⚡ 加速設定(SGLang 推理參數)',
+ accelHintSglang: '空白=用 SGLang 預設。改了需重啟模型生效。radix cache(前綴快取)預設開。',
+ sglSchedLpm: '最長前綴優先',
+ toolAuto: '自動偵測(auto)',
+ toolCallingDescSglang: 'SGLang 只需設 tool_call_parser= 即可啟用工具調用(不需 enable_auto_tool_choice);reasoning 模型再加 reasoning_parser。可用 auto 讓它從 chat template 自動偵測。parser 要對得上模型輸出格式。',
+ toolCallingDocRefSglang: 'parser 清單來自 sglang.launch_server 的 --tool-call-parser / --reasoning-parser。沒有對應 parser 的模型請勿亂加。',
},
modelDetail: {
@@ -965,7 +979,7 @@ export default {
gpuMemUtilDesc: '新版 vLLM 把 CUDA graph 記憶體也算進這個額度。你設 {current},扣掉後實際給 KV cache 的空間只等於舊版的 {effective}。想要更大 KV cache(更高並發 / 更長 context)可在停止後於「編輯參數」提到 {suggested},但顯存餘裕會變小、OOM 風險上升(小顯卡尤其要保守)。',
servedModels: '服務的模型',
routingPolicy: '路由策略(負載平衡)',
- vllmParams: 'vLLM 參數(model_config)',
+ vllmParams: '引擎參數(model_config)',
loraAdapters: 'LoRA Adapters',
hotLoadEnabled: '熱加載已啟用',
hotLoad: '載入',
@@ -999,7 +1013,7 @@ export default {
stopLabel: '停止',
terminateLabel: '終止',
abortLabel: '中止啟動',
- editParams: '編輯參數(需先停止;vLLM 參數為群組共用)',
+ editParams: '編輯參數(需先停止;引擎參數為群組共用)',
removeModel: '移除模型(僅限動態新增的模型)',
removeSuccess: '已移除 {key}',
removeFailed: '無法移除模型',
@@ -1144,7 +1158,7 @@ export default {
addInstance: {
title: '新增實例',
sharedSettingsTitle: '沿用此群組的共用設定',
- sharedSettingsHint: '同群組的所有實例共用 vLLM 參數,這裡只需設定本實例的位置。',
+ sharedSettingsHint: '同群組的所有實例共用引擎參數,這裡只需設定本實例的位置。',
instanceIdLabel: '實例 ID',
instanceIdPlaceholder: '例如:qwen3-5',
idConflict: '此 ID 已存在於群組中。',
diff --git a/apps/frontend_llmops/src/lib/api.ts b/apps/frontend_llmops/src/lib/api.ts
index 10380c6..01c88ce 100644
--- a/apps/frontend_llmops/src/lib/api.ts
+++ b/apps/frontend_llmops/src/lib/api.ts
@@ -170,10 +170,10 @@ export const api = {
`/api/models/${enc(group)}/autoscale`,
{ method: 'PUT', body: JSON.stringify(body) },
),
- parseCommand: (command: string) =>
+ parseCommand: (command: string, engine?: string) =>
request(API_BASE, '/api/models/parse', {
method: 'POST',
- body: JSON.stringify({ command }),
+ body: JSON.stringify({ command, engine }),
}),
createModel: (payload: CreateModelPayload) =>
request(API_BASE, '/api/models', { method: 'POST', body: JSON.stringify(payload) }),
diff --git a/apps/frontend_llmops/src/stores/models.ts b/apps/frontend_llmops/src/stores/models.ts
index 0a7596b..66ae4c3 100644
--- a/apps/frontend_llmops/src/stores/models.ts
+++ b/apps/frontend_llmops/src/stores/models.ts
@@ -5,6 +5,16 @@ import type { ConfigSummary, ModelKind, ModelState, ModelView } from '@/types/ap
type ConnState = 'connecting' | 'live' | 'polling' | 'error'
+/** Whether a model's observed state has caught up to its desired intent — i.e. the
+ * pending start/stop/sleep has landed (failed is terminal for any intent). */
+function settledForDesired(m: ModelView): boolean {
+ if (m.state === 'failed') return true
+ if (m.desired === 'running') return m.state === 'ready'
+ if (m.desired === 'stopped') return m.state === 'stopped'
+ if (m.desired === 'asleep') return m.state === 'sleeping'
+ return m.state !== 'starting' && m.state !== 'stopping'
+}
+
/**
* Single source of truth for model lifecycle state. Subscribes to the backend
* SSE stream (`/api/stream/models`) for push updates and falls back to polling
@@ -46,14 +56,24 @@ export const useModelsStore = defineStore('models', () => {
const hasFailures = computed(() => counts.value.failed > 0)
function applySnapshot(next: ModelView[]) {
- models.value = next
- lastUpdated.value = Date.now()
- // Clear pending markers once the stream reflects a settled state.
+ // With HA deferred actuation a stop/start is async: right after a stop request
+ // the owning node may still report `ready` for a beat before its reconcile loop
+ // converges (and vice-versa for start). Showing that stale state made the row
+ // flicker (e.g. stop -> ready -> stopped). While a key is pending, clear it once
+ // the observed state has caught up to the *desired* intent; until then, hold a
+ // transitional display (stopping/starting) so it never shows the stale state.
for (const m of next) {
- if (pending.value.has(m.key) && m.state !== 'starting' && m.state !== 'stopping') {
+ if (!pending.value.has(m.key)) continue
+ if (settledForDesired(m)) {
pending.value.delete(m.key)
+ } else if (m.desired === 'stopped') {
+ m.state = 'stopping'
+ } else if (m.desired === 'running' && m.state !== 'starting') {
+ m.state = 'starting'
}
}
+ models.value = next
+ lastUpdated.value = Date.now()
}
/** Look up the static engine config (cuda_device, max_model_len…) for a key. */
diff --git a/apps/frontend_llmops/src/types/api.ts b/apps/frontend_llmops/src/types/api.ts
index b4fbb8b..d64c193 100644
--- a/apps/frontend_llmops/src/types/api.ts
+++ b/apps/frontend_llmops/src/types/api.ts
@@ -28,6 +28,7 @@ export interface ModelStartupMetrics {
export interface ModelView {
key: string
kind: ModelKind
+ engine?: string
model_tag: string | null
host: string
port: number
diff --git a/apps/frontend_llmops/src/views/ModelsView.vue b/apps/frontend_llmops/src/views/ModelsView.vue
index 4b727ce..4cdfb06 100644
--- a/apps/frontend_llmops/src/views/ModelsView.vue
+++ b/apps/frontend_llmops/src/views/ModelsView.vue
@@ -1,5 +1,5 @@