From e5bcda78c4d6abf2a3f5a588b92a493f5afb7f7c Mon Sep 17 00:00:00 2001 From: max Date: Tue, 30 Jun 2026 18:22:15 +0800 Subject: [PATCH 01/29] feat(engine): multi-backend abstraction layer (engine + capabilities) Generalize the launcher layer from "LLM == vLLM" to dispatch by (kind, engine), so a future SGLang/llama.cpp/TensorRT-LLM launcher slots in without touching the router (an engine-agnostic OpenAI proxy) or the reconciler. Collapsed-first: the new `engine` field defaults to "vllm", so all existing config and behaviour are byte-for-byte unchanged. - schema: EngineModelConfig gains `engine: Literal[...] = "vllm"` (orthogonal to the router-facing `kind`); engine-native flags still ride extra="allow". - launchers: Launcher Protocol gains `engine` + `capabilities` (CAP_SLEEP / RUNTIME_LORA / LORA_MODULES / KV_TRANSFER / METRICS_VLLM); VllmLauncher.keys() filters by engine; EmbeddingLauncher registers under ENGINE_DEFAULT. - manager: _launchers keyed on (kind, engine) + _launcher_for(inst); create/update resolve the launcher by the group's engine and reject unregistered engines cleanly (no KeyError -> 500). - capability gating: runtime LoRA requires CAP_RUNTIME_LORA (not just the config flag); autoscaler sleep tier already keys on sleep_enabled (vLLM-only) so a non-vLLM group degrades to ready<->stopped automatically. - engine threads through LaunchSpec/ModelInstance/ModelView/observed_dict, so the fleet view + HA store carry it. - design doc + SGLang v0.5.14 research notes; per-engine image packaging decision (FROM lmsysorg/sglang), typed-param normalization, autoscaling deferred. Tests: backend 386, router 119, schema 5 green; live docker smoke test (vLLM model launches via the new path -> READY -> inference) verified. Co-Authored-By: Claude Opus 4.8 --- apps/backend/app/api/schemas.py | 3 + apps/backend/app/llmops/instance.py | 9 + apps/backend/app/llmops/launchers.py | 42 +- apps/backend/app/llmops/manager.py | 52 +- apps/backend/tests/unit/test_launchers.py | 61 +- apps/backend/tests/unit/test_lora_hotload.py | 5 +- .../backend/tests/unit/test_manager_engine.py | 178 ++++ docs/multi-backend-engine-design_zh-CN.md | 311 ++++++ docs/sglang_related_info.md | 882 ++++++++++++++++++ packages/config-schema/schema.py | 9 +- 10 files changed, 1536 insertions(+), 16 deletions(-) create mode 100644 apps/backend/tests/unit/test_manager_engine.py create mode 100644 docs/multi-backend-engine-design_zh-CN.md create mode 100644 docs/sglang_related_info.md diff --git a/apps/backend/app/api/schemas.py b/apps/backend/app/api/schemas.py index d5e78db..d8d06c4 100644 --- a/apps/backend/app/api/schemas.py +++ b/apps/backend/app/api/schemas.py @@ -14,6 +14,8 @@ class ModelView(BaseModel): key: str kind: ModelKind + # Which inference engine backs this instance ("vllm" / "sglang" / …). + engine: str = "vllm" model_tag: Optional[str] = None host: str port: int @@ -35,6 +37,7 @@ def from_instance(cls, inst: ModelInstance) -> "ModelView": return cls( key=inst.key, kind=inst.kind, + engine=inst.engine, model_tag=inst.model_tag, host=inst.host, port=inst.port, diff --git a/apps/backend/app/llmops/instance.py b/apps/backend/app/llmops/instance.py index 75935e8..d20086b 100644 --- a/apps/backend/app/llmops/instance.py +++ b/apps/backend/app/llmops/instance.py @@ -29,6 +29,11 @@ class LaunchSpec: host: str port: int probe_url: str + # Which inference engine this spec launches ("vllm" / "sglang" / …) and the + # optional features it supports. Callers gate on capabilities, not the engine + # name. See launchers.CAP_* and docs/multi-backend-engine-design_zh-CN.md. + engine: str = "vllm" + capabilities: frozenset = field(default_factory=frozenset) model_tag: Optional[str] = None # True when launched with --enable-sleep-mode + VLLM_SERVER_DEV_MODE=1, so the # /sleep, /wake_up and /is_sleeping dev endpoints are available. @@ -51,6 +56,9 @@ class ModelInstance: port: int spec: LaunchSpec model_tag: Optional[str] = None + # Inference engine backing this instance (mirrors spec.engine); surfaced in the + # fleet view + persisted to the shared store (HA) so any replica can render it. + engine: str = "vllm" desired: Desired = Desired.STOPPED state: ModelState = ModelState.STOPPED @@ -85,6 +93,7 @@ def observed_dict(self) -> dict: return { "key": self.key, "kind": self.kind.value, + "engine": self.engine, "model_tag": self.model_tag, "host": self.host, "port": self.port, diff --git a/apps/backend/app/llmops/launchers.py b/apps/backend/app/llmops/launchers.py index a20c687..b61bb00 100644 --- a/apps/backend/app/llmops/launchers.py +++ b/apps/backend/app/llmops/launchers.py @@ -55,8 +55,9 @@ def _write_effective_config(config) -> str: _LORA_RUNTIME_KEY = "allow_runtime_lora" # Router-only knobs that ride the shared model_config (EngineModelConfig is # extra="allow") but belong to the router, not vLLM — never pass them to -# `vllm serve` or it errors on an unknown argument. -_ROUTER_ONLY_KEYS = frozenset({"routing_strategy", "kind"}) +# `vllm serve` or it errors on an unknown argument. `engine` is launcher-meta +# (which backend to run); it too must never reach the vLLM CLI. +_ROUTER_ONLY_KEYS = frozenset({"routing_strategy", "kind", "engine"}) # Everything build_vllm_cli_args must skip (model_tag is the positional arg). _SKIP_CLI_KEYS = frozenset({"model_tag", _LORA_RUNTIME_KEY}) | _ROUTER_ONLY_KEYS @@ -126,11 +127,31 @@ def build_vllm_cli_args(model_cfg: dict) -> list[str]: return cli_args +# Engine capability flags. Callers gate optional features on *capabilities*, never +# on the engine name (`if "sleep" in caps`, never `if engine == "vllm"`), so adding +# an engine is just declaring its capability set. See docs/multi-backend-engine-design_zh-CN.md §4. +CAP_SLEEP = "sleep" # /sleep + /wake_up (warm standby, frees VRAM, process stays up) +CAP_RUNTIME_LORA = "runtime_lora" # runtime LoRA load/unload endpoints +CAP_LORA_MODULES = "lora_modules" # static --lora-modules at launch +CAP_KV_TRANSFER = "kv_transfer" # cross-instance KV cache sharing +CAP_METRICS_VLLM = "metrics_vllm" # exposes vLLM-format Prometheus metrics (waiting queue, …) + +# Sentinel engine name for non-LLM launchers (embedding server): they aren't +# selected by an engine choice, so they register under one fixed value. +ENGINE_DEFAULT = "default" + + class Launcher(Protocol): kind: ModelKind + # Which engine this launcher serves; dispatch is keyed on (kind, engine). + engine: str + # Optional features this engine supports (see CAP_* above). Threaded onto the + # LaunchSpec so callers (autoscaler, sleep/LoRA APIs, metrics) can gate without + # re-checking the engine name. + capabilities: frozenset[str] def keys(self, config) -> list[str]: - """All instance keys this launcher's kind defines in the config.""" + """All instance keys this launcher defines in the config (its engine only).""" ... def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: @@ -140,10 +161,19 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: class VllmLauncher: kind = ModelKind.LLM + engine = "vllm" + capabilities = frozenset({ + CAP_SLEEP, CAP_RUNTIME_LORA, CAP_LORA_MODULES, CAP_KV_TRANSFER, CAP_METRICS_VLLM, + }) def keys(self, config) -> list[str]: out: list[str] = [] for model_tag, engine in config.LLM_engines.items(): + # Only claim groups configured for this engine. `engine` defaults to + # "vllm" (EngineModelConfig), so a config with no engine field is all + # vLLM = today's behaviour. + if getattr(engine.settings, "engine", "vllm") != self.engine: + continue for inst in engine.instances: out.append(f"{model_tag}::{inst.id}") return out @@ -204,6 +234,8 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: return LaunchSpec( key=key, kind=self.kind, + engine=self.engine, + capabilities=self.capabilities, command=command, env=env, log_path=log_path, @@ -217,6 +249,8 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: class EmbeddingLauncher: kind = ModelKind.EMBEDDING + engine = ENGINE_DEFAULT # not engine-selectable; one launcher for the embedding server + capabilities = frozenset() def keys(self, config) -> list[str]: emb = config.embedding_server @@ -246,6 +280,8 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: return LaunchSpec( key=key, kind=self.kind, + engine=self.engine, + capabilities=self.capabilities, command=command, env=env, log_path=log_path, diff --git a/apps/backend/app/llmops/manager.py b/apps/backend/app/llmops/manager.py index 73d31c7..14170b8 100644 --- a/apps/backend/app/llmops/manager.py +++ b/apps/backend/app/llmops/manager.py @@ -16,7 +16,7 @@ from app.core.settings import BackendSettings from app.llmops.events import emit_transition from app.llmops.instance import ModelInstance -from app.llmops.launchers import Launcher +from app.llmops.launchers import CAP_RUNTIME_LORA, CAP_SLEEP, Launcher from app.llmops.process import spawn_process, terminate_process_group from app.llmops.registry import ModelRegistry from app.llmops.state import Desired, ModelKind, ModelState @@ -68,6 +68,7 @@ def build_registry(config, config_path: str, launchers: list[Launcher]) -> Model ModelInstance( key=key, kind=launcher.kind, + engine=spec.engine, host=spec.host, port=spec.port, spec=spec, @@ -93,7 +94,11 @@ def __init__( notifier=None, ) -> None: self.registry = registry - self._launchers: dict[ModelKind, Launcher] = {l.kind: l for l in launchers} + # Dispatch is keyed on (kind, engine): vLLM and SGLang are both ModelKind.LLM + # but distinct launchers. Embedding registers under ENGINE_DEFAULT. + self._launchers: dict[tuple[ModelKind, str], Launcher] = { + (l.kind, l.engine): l for l in launchers + } self.http_client = http_client self.config = config self.config_path = config_path @@ -111,6 +116,22 @@ def _require(self, key: str) -> ModelInstance: raise ModelNotFound(key) return inst + def _launcher_for(self, inst: ModelInstance) -> Launcher: + """The launcher that owns an instance, by its (kind, engine).""" + return self._launchers[(inst.kind, inst.engine)] + + def _llm_engine_capabilities(self, group: str) -> frozenset: + """Capabilities of the engine an LLM group is configured for. Callers gate + optional features (sleep, runtime LoRA, …) on these rather than the engine + name, so a new engine only needs to declare its capability set. Empty if the + group / its engine's launcher is unknown.""" + engine = self.config.LLM_engines.get(group) + if engine is None: + return frozenset() + engine_name = getattr(engine.settings, "engine", "vllm") + launcher = self._launchers.get((ModelKind.LLM, engine_name)) + return launcher.capabilities if launcher else frozenset() + async def _record(self, inst, from_state, to_state, detail=None) -> None: """Persist a state transition + dispatch any alert, via the shared funnel. Best-effort: telemetry/alerts never break ops.""" @@ -357,7 +378,7 @@ async def _gpu_exists_preflight(self, key: str, spec) -> None: async def start(self, key: str, force: bool = False, reset_restart: bool = True) -> ModelInstance: inst = self._require(key) - launcher = self._launchers[inst.kind] + launcher = self._launcher_for(inst) # Re-resolve the spec (config may have changed) outside the lock so the # GPU pre-flight's nvidia-smi call never extends the critical section. @@ -617,13 +638,17 @@ async def create_overlay_model(self, group: str, instance: dict, model_config: d save_overlay(overlay, self.overlay_path) self.config = new_config - launcher = self._launchers[ModelKind.LLM] + group_engine = getattr(new_config.LLM_engines[group].settings, "engine", "vllm") + launcher = self._launchers.get((ModelKind.LLM, group_engine)) + if launcher is None: + raise ModelConflict(f"unsupported engine '{group_engine}' (no launcher registered)") spec = launcher.build_spec(self.config, self.config_path, key) async with self.registry.lock: self.registry.add( ModelInstance( key=key, kind=ModelKind.LLM, + engine=spec.engine, host=spec.host, port=spec.port, spec=spec, @@ -679,10 +704,15 @@ async def update_overlay_model(self, key: str, instance: dict, model_config: dic save_overlay(overlay, self.overlay_path) self.config = new_config - # Re-resolve the spec so the next start uses the edited values. - launcher = self._launchers[ModelKind.LLM] + # Re-resolve the spec so the next start uses the edited values. The edit may + # have changed the group's engine, so pick the launcher by the new config. + group_engine = getattr(new_config.LLM_engines[group].settings, "engine", "vllm") + launcher = self._launchers.get((ModelKind.LLM, group_engine)) + if launcher is None: + raise ModelConflict(f"unsupported engine '{group_engine}' (no launcher registered)") spec = launcher.build_spec(self.config, self.config_path, key) async with self.registry.lock: + inst.engine = spec.engine inst.host = spec.host inst.port = spec.port inst.spec = spec @@ -752,6 +782,11 @@ async def load_lora(self, group: str, name: str, path: str, engine = self.config.LLM_engines.get(group) if engine is None: raise ModelNotFound(group) + if CAP_RUNTIME_LORA not in self._llm_engine_capabilities(group): + engine_name = getattr(engine.settings, "engine", "vllm") + raise ModelConflict( + f"{group}'s engine ({engine_name}) does not support runtime LoRA" + ) if not getattr(engine.settings, "allow_runtime_lora", False): raise ModelConflict( f"{group} was not started with runtime LoRA updating — enable " @@ -897,13 +932,14 @@ def resync_registry(self, new_config) -> dict[str, list[str]]: for key in added: kind, spec = desired[key] self.registry.add(ModelInstance( - key=key, kind=kind, host=spec.host, port=spec.port, spec=spec, - model_tag=spec.model_tag, log_path=spec.log_path, + key=key, kind=kind, engine=spec.engine, host=spec.host, port=spec.port, + spec=spec, model_tag=spec.model_tag, log_path=spec.log_path, )) for key in changed: kind, spec = desired[key] inst = self.registry.get(key) inst.spec, inst.host, inst.port = spec, spec.host, spec.port + inst.engine = spec.engine inst.model_tag, inst.log_path = spec.model_tag, spec.log_path return {"added": added, "removed": removed, "changed": changed} diff --git a/apps/backend/tests/unit/test_launchers.py b/apps/backend/tests/unit/test_launchers.py index 4441ef2..faf4d26 100644 --- a/apps/backend/tests/unit/test_launchers.py +++ b/apps/backend/tests/unit/test_launchers.py @@ -2,9 +2,9 @@ import pytest -from app.llmops.launchers import (EMBEDDING_KEY, EmbeddingLauncher, - VllmLauncher, _write_effective_config, - build_vllm_cli_args) +from app.llmops.launchers import (CAP_SLEEP, EMBEDDING_KEY, ENGINE_DEFAULT, + EmbeddingLauncher, VllmLauncher, + _write_effective_config, build_vllm_cli_args) from app.llmops.state import ModelKind from schema import RootConfig from tests.conftest import FAKE_CONFIG @@ -12,6 +12,20 @@ pytestmark = pytest.mark.unit +def _engine_config(engine: str) -> RootConfig: + """A one-group LLM config whose engine is `engine` (no field = default).""" + mc = {"model_tag": "org/m"} + if engine is not None: + mc["engine"] = engine + return RootConfig.model_validate({ + "server": {"host": "0.0.0.0", "port": 8887}, + "LLM_engines": {"G": { + "instances": [{"id": "a", "host": "localhost", "port": 8000}], + "model_config": mc, + }}, + }) + + def _lora_config() -> RootConfig: return RootConfig.model_validate( { @@ -245,3 +259,44 @@ def test_embedding_launcher_keys_and_spec(): assert spec.env["CUDA_VISIBLE_DEVICES"] == "1" assert "PYTHONPATH" in spec.env assert spec.probe_url == "http://localhost:8005/health" + + +# ---- multi-engine abstraction (docs/multi-backend-engine-design_zh-CN.md) ---- + +def test_engine_defaults_to_vllm_when_unset(): + """A config with no `engine` field is all vLLM = historical behaviour.""" + cfg = _engine_config(engine=None) + assert cfg.LLM_engines["G"].settings.engine == "vllm" + + +def test_vllm_launcher_claims_only_its_engine(): + """keys() returns groups for this launcher's engine; a non-vllm group is + skipped so a future SGLang launcher can claim it instead.""" + v = VllmLauncher() + assert v.engine == "vllm" + assert v.keys(_engine_config(None)) == ["G::a"] # default + assert v.keys(_engine_config("vllm")) == ["G::a"] # explicit + assert v.keys(_engine_config("sglang")) == [] # not mine + + +def test_vllm_launcher_declares_capabilities_on_spec(): + v = VllmLauncher() + spec = v.build_spec(_engine_config("vllm"), "config.yaml", "G::a") + assert spec.engine == "vllm" + assert CAP_SLEEP in spec.capabilities + assert spec.capabilities == v.capabilities + + +def test_embedding_launcher_registers_under_default_engine(): + e = EmbeddingLauncher() + assert e.engine == ENGINE_DEFAULT + assert e.capabilities == frozenset() + spec = e.build_spec(FAKE_CONFIG, "config.yaml", EMBEDDING_KEY) + assert spec.engine == ENGINE_DEFAULT + + +def test_engine_is_never_passed_to_vllm_cli(): + """`engine` is launcher-meta; it must not reach `vllm serve` (unknown arg).""" + args = build_vllm_cli_args({"model_tag": "org/m", "engine": "vllm", "kind": "chat"}) + assert "--engine" not in args + assert "--kind" not in args diff --git a/apps/backend/tests/unit/test_lora_hotload.py b/apps/backend/tests/unit/test_lora_hotload.py index fe052b0..145493f 100644 --- a/apps/backend/tests/unit/test_lora_hotload.py +++ b/apps/backend/tests/unit/test_lora_hotload.py @@ -6,6 +6,7 @@ import pytest import yaml +from app.llmops.launchers import VllmLauncher from app.llmops.manager import LoraRuntimeError, ModelConflict, ModelManager, ModelNotFound from app.llmops.registry import ModelRegistry from app.llmops.state import ModelKind, ModelState @@ -64,7 +65,9 @@ def _ready_instances(ports): def _mgr(config, client, registry): - return ModelManager(registry, [], client, config, "config.yaml", + # Register the vLLM launcher so runtime-LoRA capability resolves (the gate + # checks the group's engine supports it). + return ModelManager(registry, [VllmLauncher()], client, config, "config.yaml", BackendSettings(), overlay_path="overlay.json") diff --git a/apps/backend/tests/unit/test_manager_engine.py b/apps/backend/tests/unit/test_manager_engine.py new file mode 100644 index 0000000..683f271 --- /dev/null +++ b/apps/backend/tests/unit/test_manager_engine.py @@ -0,0 +1,178 @@ +"""Multi-engine dispatch: the manager picks a launcher by (kind, engine), and the +engine threads from config -> registry -> instance. See +docs/multi-backend-engine-design_zh-CN.md.""" +import pytest + +from app.core.settings import BackendSettings +from app.llmops.instance import LaunchSpec +from app.llmops.launchers import (CAP_RUNTIME_LORA, CAP_SLEEP, EmbeddingLauncher, + VllmLauncher) +from app.llmops.manager import ModelConflict, ModelManager, build_registry +from app.llmops.state import ModelKind +from schema import load_config + +pytestmark = pytest.mark.unit + + +class _FakeLauncher: + """A minimal second LLM engine with no optional capabilities — stands in for a + future engine to exercise capability gating + dispatch. Uses the "sglang" + engine slot (a valid schema Literal) but declares no capabilities of its own.""" + kind = ModelKind.LLM + engine = "sglang" + capabilities = frozenset() + + def keys(self, config): + out = [] + for tag, eng in config.LLM_engines.items(): + if getattr(eng.settings, "engine", "vllm") != self.engine: + continue + out.extend(f"{tag}::{i.id}" for i in eng.instances) + return out + + def build_spec(self, config, config_path, key): + tag, _, iid = key.partition("::") + inst = next(i for i in config.LLM_engines[tag].instances if i.id == iid) + return LaunchSpec( + key=key, kind=self.kind, engine=self.engine, capabilities=self.capabilities, + command=["fake-serve"], env={}, log_path="x.log", + host=inst.host, port=inst.port, probe_url=f"http://{inst.host}:{inst.port}/health", + model_tag=config.LLM_engines[tag].settings.model_tag, + ) + +# Two groups: one default (vllm), one explicitly configured for a not-yet-built +# engine (sglang). Only VllmLauncher is registered, so the sglang group is simply +# not claimed — proving keys() filters by engine rather than erroring. +CONFIG_YAML = """ +server: + port: 8887 +LLM_engines: + Qwen3-0.6B: + instances: + - id: a + host: localhost + port: 8002 + model_config: + model_tag: Qwen/Qwen3-0.6B + FutureModel: + instances: + - id: a + host: localhost + port: 8010 + model_config: + model_tag: org/future + engine: sglang +""" + + +def _manager(tmp_path): + cfg_path = tmp_path / "config.yaml" + cfg_path.write_text(CONFIG_YAML, encoding="utf-8") + config = load_config(str(cfg_path)) + launchers = [VllmLauncher(), EmbeddingLauncher()] + registry = build_registry(config, str(cfg_path), launchers) + mgr = ModelManager( + registry, launchers, None, config, str(cfg_path), + BackendSettings(), store=None, overlay_path=str(tmp_path / "overlay.json"), + ) + return mgr + + +def test_build_registry_threads_engine_and_filters_unclaimed(tmp_path): + mgr = _manager(tmp_path) + # vLLM group is registered with engine="vllm" on the instance. + vllm_inst = mgr.registry.get("Qwen3-0.6B::a") + assert vllm_inst is not None + assert vllm_inst.engine == "vllm" + assert vllm_inst.spec.engine == "vllm" + # The sglang group has no launcher registered for it -> not in the registry. + assert mgr.registry.get("FutureModel::a") is None + + +def test_launcher_for_dispatches_by_kind_and_engine(tmp_path): + mgr = _manager(tmp_path) + inst = mgr.registry.get("Qwen3-0.6B::a") + launcher = mgr._launcher_for(inst) + assert isinstance(launcher, VllmLauncher) + assert (launcher.kind, launcher.engine) == (ModelKind.LLM, "vllm") + + +def test_observed_dict_includes_engine(tmp_path): + mgr = _manager(tmp_path) + inst = mgr.registry.get("Qwen3-0.6B::a") + assert inst.observed_dict()["engine"] == "vllm" + + +async def test_create_overlay_model_records_engine(tmp_path): + """A dashboard-added model (default engine) registers with engine='vllm'.""" + mgr = _manager(tmp_path) + inst = await mgr.create_overlay_model( + "NewGroup", + {"id": "x", "host": "localhost", "port": 8020}, + {"model_tag": "org/new"}, + ) + assert inst.engine == "vllm" + assert mgr.registry.get("NewGroup::x").engine == "vllm" + + +# ---- capability gating ------------------------------------------------------ + +FAKE_ENGINE_YAML = """ +server: + port: 8887 +LLM_engines: + Plain: + instances: + - id: a + host: localhost + port: 8030 + model_config: + model_tag: org/plain + engine: sglang + allow_runtime_lora: true +""" + + +def _manager_with_fake(tmp_path): + cfg_path = tmp_path / "config.yaml" + cfg_path.write_text(FAKE_ENGINE_YAML, encoding="utf-8") + config = load_config(str(cfg_path)) + launchers = [VllmLauncher(), _FakeLauncher(), EmbeddingLauncher()] + registry = build_registry(config, str(cfg_path), launchers) + return ModelManager( + registry, launchers, None, config, str(cfg_path), + BackendSettings(), store=None, overlay_path=str(tmp_path / "overlay.json"), + ) + + +def test_capabilities_resolved_per_engine(tmp_path): + mgr = _manager_with_fake(tmp_path) + # fake engine declares no capabilities; vllm would declare sleep + lora. + assert mgr._llm_engine_capabilities("Plain") == frozenset() + assert CAP_SLEEP not in mgr._llm_engine_capabilities("Plain") + + +def test_fake_engine_group_is_dispatched_to_its_launcher(tmp_path): + mgr = _manager_with_fake(tmp_path) + inst = mgr.registry.get("Plain::a") + assert inst is not None and inst.engine == "sglang" + assert isinstance(mgr._launcher_for(inst), _FakeLauncher) + + +async def test_load_lora_rejected_when_engine_lacks_capability(tmp_path): + """Even with allow_runtime_lora set in config, an engine without the + runtime_lora capability must be refused (gate on capability, not the flag).""" + mgr = _manager_with_fake(tmp_path) + with pytest.raises(ModelConflict, match="runtime LoRA"): + await mgr.load_lora("Plain", "adapter", "repo/adapter") + + +async def test_create_overlay_model_rejects_unregistered_engine(tmp_path): + # trtllm is a valid engine name in the schema, but no launcher is registered + # for it here — the manager must refuse cleanly, not KeyError into a 500. + mgr = _manager_with_fake(tmp_path) + with pytest.raises(ModelConflict, match="unsupported engine"): + await mgr.create_overlay_model( + "Brand", {"id": "z", "host": "localhost", "port": 8040}, + {"model_tag": "org/brand", "engine": "trtllm"}, + ) diff --git a/docs/multi-backend-engine-design_zh-CN.md b/docs/multi-backend-engine-design_zh-CN.md new file mode 100644 index 0000000..6c1caae --- /dev/null +++ b/docs/multi-backend-engine-design_zh-CN.md @@ -0,0 +1,311 @@ +# 多推理後端抽象層設計(vLLM / SGLang / llama.cpp / TensorRT-LLM) + +> 目標:讓系統能在**同一個 group / 不同 group** 用不同的推理引擎(vLLM、SGLang、llama.cpp、 +> TensorRT-LLM…),而不是寫死 vLLM。做法**不是重寫**,而是把現有那層已經很乾淨的 `Launcher` +> 抽象「補完」:新增一個 `engine` 維度、把 launcher 分派從「依 kind」改成「依 engine」、並引入 +> **capabilities(能力旗標)** 讓只有 vLLM 才有的功能(sleep mode / runtime LoRA / KV transfer / +> 指標格式)對其他引擎自動退化。 +> +> ⚠️ **設計原則:collapsed-first / 零行為變更。** `engine` 預設 `"vllm"`,所有現有 config 不動、 +> 行為 byte-for-byte 不變,現有測試全綠 —— 與 HA 那套同樣的哲學。新引擎是**增量加上去的 launcher**, +> 不碰既有路徑。 +> +> 本文是**抽象層藍圖**;各引擎的實際 CLI/啟動細節(SGLang 等)由各自的 launcher 實作,另行補上。 + +## 0. 為什麼這件事比想像中容易 —— 現有架構已經幫你做了一半 + +兩個既成事實讓「多後端」幾乎只剩「加 launcher」: + +1. **Router 對後端完全是 OpenAI-compatible HTTP 反向代理** + ([router.py](../apps/router-server/src/llm_router/router.py) proxy `/v1/chat/completions`、 + `/v1/completions`、`/v1/embeddings`、`/v1/rerank`…)。它**不知道也不在乎** `host:port` 後面跑的是 + 哪個引擎 —— 只要那個 port 開的是 OpenAI 相容 server。SGLang、llama.cpp(`llama-server`)、 + TensorRT-LLM(`trtllm-serve`)都提供 OpenAI 相容端點。 + **結論:推理流量這條路徑零修改。** + +2. **`Launcher` 已經是後端無關的抽象**([launchers.py](../apps/backend/app/llmops/launchers.py)): + + ``` + Launcher.keys(config) # 此 launcher 在 config 裡定義了哪些 instance key + Launcher.build_spec(config, key) # config → LaunchSpec(command/env/probe_url/host/port…) + ``` + + `build_registry` 只是 `for launcher in launchers: for key in launcher.keys(...)` + ([manager.py](../apps/backend/app/llmops/manager.py))。`LaunchSpec` + ([instance.py](../apps/backend/app/llmops/instance.py))已經是引擎無關的「一份 command + env + + probe_url」。`process.py` 拿到 spec 就 spawn,**完全不認識 vLLM**。 + + **結論:加一個引擎 = 寫一個 launcher 產對應的 command,spawn / 健康探測 / 重啟 / fleet 視圖 / HA + 全部沿用。** + +## 1. 現況的耦合點:vLLM 假設藏在哪 + +抽象層要拆掉的、目前「LLM 就等於 vLLM」的硬寫死: + +| # | 位置 | 現況 | 問題 | +|---|---|---|---| +| C1 | [state.py](../apps/backend/app/llmops/state.py) `ModelKind`(`llm`/`embedding`)+ [manager.py](../apps/backend/app/llmops/manager.py) `_launchers: dict[ModelKind, Launcher]` | launcher 依 `ModelKind` 分派,**一個 kind 只能對一個 launcher** | vLLM 和 SGLang 都是 "llm",kind 不足以選 launcher | +| C2 | [manager.py](../apps/backend/app/llmops/manager.py) `self._launchers[ModelKind.LLM]`(建立 / 編輯 / LoRA 等多處) | 硬寫「LLM 就是那唯一一個 launcher」 | 多引擎共存時選錯 launcher | +| C3 | [schema.py](../packages/config-schema/schema.py) `EngineModelConfig` 只有 `kind`(chat/embed/rerank,**路由端點類型**,非引擎類型) | 沒有「用哪個引擎」的欄位 | config 無法表達 engine 選擇 | +| C4 | [launchers.py](../apps/backend/app/llmops/launchers.py) `build_vllm_cli_args` / `VllmLauncher` | vLLM 專屬 CLI 慣例(`serve `、`--no-` bool、kebab-case、`--lora-modules` 多值) | 各引擎 flag 名稱與慣例不同 | +| C5 | 功能耦合:sleep mode([launchers.py](../apps/backend/app/llmops/launchers.py) `--enable-sleep-mode`+`VLLM_SERVER_DEV_MODE`)、runtime LoRA([manager.py](../apps/backend/app/llmops/manager.py))、KV transfer、autoscaler 吃的 vLLM Prometheus 指標([metrics_poller](../apps/router-server/src/llm_router/metrics_poller.py)) | 假設每個 LLM 都支援 | 其他引擎沒有 → 功能會壞或誤判 | + +C1–C4 是「接線」,直接;**C5 是真風險(功能對等)**,用 capabilities 解(見 §4)。 + +## 2. 資料模型變更:`engine` 維度 + +### 2.1 config schema(C3) + +在 [schema.py](../packages/config-schema/schema.py) `EngineModelConfig` 新增一個欄位: + +```python +class EngineModelConfig(BaseModel): + model_config = ConfigDict(extra="allow", protected_namespaces=()) + model_tag: str + # 用哪個推理引擎啟動這個 group。預設 vllm = 現有行為不變。 + # 注意:這跟 `kind`(chat/embed/rerank,路由端點類型)是兩個正交維度。 + engine: Literal["vllm", "sglang", "llamacpp", "trtllm"] = "vllm" + kind: Literal["chat", "embed", "rerank"] = "chat" + ... +``` + +- `extra="allow"` 已經讓**任意引擎 flag 透傳** → 各引擎專屬旗標**不需要動 schema**,直接寫在 + `model_config` 底下,由該引擎的 arg builder 解讀。 +- `engine` 預設 `"vllm"` → 現有 config 不寫此欄位 = vLLM = 零變更。 + +> `engine` 屬於 group 層級(`model_config`)而非 instance 層級:同一 group 的所有 instance 用同一個引擎 +> (它們是同一個模型的多副本)。要混引擎就開不同 group。 + +### 2.2 ModelKind vs engine 的關係 + +- `ModelKind`(`llm`/`embedding`)維持原意:**路由 + 探針的大分類**,決定 router 怎麼對待它。 +- `engine` 是**「llm 這類用什麼程式去起」**的子維度。 +- `embedding` 目前只有一個 launcher(router-server 的 embedding server),暫不引入 engine 選擇; + 未來若要 SGLang/其他做 embedding 再說。**本次抽象聚焦 `kind=llm` 底下的多引擎。** + +### 2.3 通用參數正規化:typed 共用欄位,各 launcher 翻譯成自家旗標 + +**決定:常見概念用一組 engine 無關的 typed 欄位表達,由各 launcher 翻成自家 CLI 旗標。** 因為不同引擎 +對同一概念的旗標名不同(下表),若讓使用者直接寫引擎旗標,dashboard UI 就無法統一渲染、換引擎就要重學。 + +| 概念(EngineModelConfig typed 欄位) | vLLM 旗標 | SGLang 旗標 | 備註 | +|---|---|---|---| +| `max_model_len` | `--max-model-len` | `--context-length` | 直接對應 | +| `gpu_memory_utilization` | `--gpu-memory-utilization` | `--mem-fraction-static` | **語意非逐字等價**(SGLang 是「靜態配給權重+KV pool 的顯存比例」);UI 要註明 | +| `tensor_parallel_size` | `--tensor-parallel-size` | `--tp-size`(alias `--tensor-parallel-size`) | 直接對應 | +| `dtype` | `--dtype` | `--dtype` | 同名 | +| `model_tag`(served name) | 位置參數 | `--model-path` + `--served-model-name ` | SGLang 預設 served name = model_path,顯式帶 model_tag 讓 router forward_name 穩定 | + +實作:每個 launcher 有自己的 **arg builder**(取代 `build_vllm_cli_args` 對所有引擎通用的假設)。builder 的 +職責 = 「翻譯這張表的 typed 欄位 + 透傳 `extra="allow"` 的引擎原生旗標 + 套用該引擎的 bool 慣例」。 +vLLM 的 builder 對這些 typed 欄位是 **identity 翻譯**(欄位名 kebab-case 後就是旗標),所以**現有行為不變**。 + +> bool 慣例各引擎不同:vLLM 是 `BooleanOptionalAction`(`--flag` / `--no-flag`);SGLang 是 `store_true` +> (只有 `--flag`,且很多旗標名本身就是負向如 `--disable-radix-cache`)。**不可**跨引擎沿用同一套 bool 邏輯。 + +## 3. Launcher 分派重構(C1 + C2) + +### 3.1 Launcher Protocol 加上 `engine` 與 `capabilities` + +```python +class Launcher(Protocol): + kind: ModelKind # 既有:路由/探針大分類 + engine: str # 新增:"vllm" / "sglang" / "llamacpp" / "trtllm" + capabilities: frozenset[str] # 新增:見 §4 + + def keys(self, config) -> list[str]: ... + def build_spec(self, config, config_path, key) -> LaunchSpec: ... +``` + +### 3.2 註冊表從「依 kind」改成「依 (kind, engine)」 + +現在([manager.py](../apps/backend/app/llmops/manager.py)): + +```python +self._launchers: dict[ModelKind, Launcher] = {l.kind: l for l in launchers} +... +launcher = self._launchers[ModelKind.LLM] # C2:寫死 +``` + +改成: + +```python +# (kind, engine) → launcher。embedding 用哨兵 engine 名("default")保持單一。 +self._launchers: dict[tuple[ModelKind, str], Launcher] = { + (l.kind, l.engine): l for l in launchers +} + +def _launcher_for(self, inst) -> Launcher: + return self._launchers[(inst.kind, inst.engine)] +``` + +- `build_registry` 在列舉 key 時,launcher 自己知道自己負責哪些 group(`keys()` 只回傳 + `model_config.engine == self.engine` 的 group)→ **不同引擎的 launcher 自然分工,不重疊**。 +- C2 那幾處 `self._launchers[ModelKind.LLM]` 改成 `self._launcher_for(inst)`。 + +### 3.3 `keys()` 依 engine 過濾 + +每個 LLM launcher 只認領 `engine` 等於自己的 group: + +```python +class VllmLauncher: + kind = ModelKind.LLM + engine = "vllm" + def keys(self, config): + out = [] + for model_tag, eng in config.LLM_engines.items(): + if getattr(eng.settings, "engine", "vllm") != self.engine: + continue # 不是我的引擎,跳過 + for inst in eng.instances: + out.append(f"{model_tag}::{inst.id}") + return out +``` + +→ collapsed 情況(全 vLLM):`VllmLauncher` 認領全部、其他 launcher 認領 0 個 = 今天的行為。 + +### 3.4 ModelInstance 帶上 engine + +[instance.py](../apps/backend/app/llmops/instance.py) `ModelInstance` 加 `engine: str = "vllm"`, +`observed_dict()` 一併輸出(讓 dashboard / fleet 視圖顯示引擎、HA 入庫帶上)。建 instance 處 +([manager.py](../apps/backend/app/llmops/manager.py))從 `engine.settings.engine` 帶入。 + +## 4. Capabilities:功能對等的關鍵設計(C5) + +**這是讓多引擎乾淨共存、避免一堆 `if engine == "vllm"` 散落各處的核心。** 把「只有某些引擎支援的功能」 +宣告成 launcher 的能力集,呼叫端依**能力**而非**引擎名**做 gate。 + +### 4.1 能力清單(初版) + +| capability | 意義 | 誰在乎 | +|---|---|---| +| `sleep` | 支援 `/sleep`+`/wake_up`(暖待命,VRAM 釋放但 process 存活) | autoscaler 的暖待命層、sleep API | +| `runtime_lora` | 支援執行期 LoRA 掛載/卸載端點 | LoRA 管理 UI / API([manager.py](../apps/backend/app/llmops/manager.py)) | +| `kv_transfer` | 支援跨實例 KV cache 共享 | KV transfer 設定 | +| `metrics_vllm` | 暴露 vLLM 格式 Prometheus 指標(waiting queue 等) | autoscaler 的擴縮訊號、[metrics_poller](../apps/router-server/src/llm_router/metrics_poller.py) | +| `lora_modules` | 啟動時可帶 `--lora-modules` 靜態 adapter | launcher arg 組裝 | + +```python +class VllmLauncher: + capabilities = frozenset({"sleep", "runtime_lora", "kv_transfer", "metrics_vllm", "lora_modules"}) + +class SglangLauncher: + capabilities = frozenset({"metrics_sglang"}) # 例:不同指標格式;無 sleep/lora +``` + +### 4.2 呼叫端如何 gate + +- **autoscaler**([autoscaler.py](../apps/backend/app/llmops/autoscaler.py)):暖待命層(ready→asleep→ + stopped)只在 group 的引擎有 `sleep` 能力時啟用;否則自動退化成 `ready ↔ stopped` 直接擴縮。 +- **sleep API / LoRA API / KV transfer 設定**:目標 group 引擎缺對應 capability 時,API 回 409/友善錯誤, + dashboard 對該 group 隱藏/灰掉這些動作。 +- **metrics_poller / autoscaler 訊號**:依 `metrics_*` 能力選對應的指標解析器;無相容指標的引擎, + 擴縮退化成「只看 router 端可觀測的訊號(例如 in-flight / 佇列)」或維持固定副本。 +- **LaunchSpec 帶 capabilities**:reconciler / autoscaler 不必回查 launcher,直接看 spec。 + +> 設計準則:**呼叫端永遠問「這個實例有沒有 X 能力」,絕不問「它是不是 vLLM」。** 新增引擎時只需宣告 +> 它的能力集,所有 gate 自動正確。 + +## 5. 封裝模型:對稱式 per-engine image + engine 變成 node 能力 + +**決定:每個 engine 一顆 backend image,結構與現有 vLLM 對稱。** 因為 vLLM / SGLang 各自死釘特定 +torch/CUDA/flashinfer,塞同一顆 image 極易版本打架;且 launcher 是**在 backend 容器內 spawn 子行程** +([process.py](../apps/backend/app/llmops/process.py)),所以「backend 能起哪些 engine」= 它容器裡裝了什麼。 + +``` +engine.Dockerfile FROM vllm/vllm-openai:latest + backend code → 能起 vllm +engine-sglang.Dockerfile FROM lmsysorg/sglang:latest + backend code → 能起 sglang +``` + +兩顆 image 都含**同一份 backend FastAPI 程式**,只是 base image(= 可用的 engine CLI)不同。單機 collapsed +仍是今天的樣子(跑 vLLM image);要混 engine 就**多跑一個用 sglang image 的 backend 副本**。 + +這把 `engine` 從「group 屬性」自然提升成 **node 能力**: + +- node-agent 心跳時宣告自己能跑哪些 engine(由 image 決定,例:`engines=["sglang"]`)。 +- scheduler placement 時,只把某 engine 的 group 排到**宣告支援該 engine 的 node**。 +- 這牽涉 [node_agent.py](../apps/backend/app/llmops/node_agent.py) / [scheduler.py](../apps/backend/app/llmops/scheduler.py) / + `nodes` 表,屬於 [ha-per-node-actuation-design](ha-per-node-actuation-design_zh-CN.md) 那塊;**本階段先不做多節點排程**, + 單機驗證「sglang image 能起 sglang 模型 + router 能路由」即可。 + +### 5.1 各引擎 launcher 概要 + +| 引擎 | 啟動(容器內) | base image | OpenAI 相容 | 探針 | 難度 | +|---|---|---|---|---|---| +| **vLLM**(現有) | `vllm serve --flags` | `vllm/vllm-openai` | ✅ | `/health` | — | +| **SGLang** | `python3 -m sglang.launch_server --model-path …` | `lmsysorg/sglang` | ✅ | `/health` | 低(第一個) | +| **llama.cpp** | `llama-server -m …` | `ghcr.io/ggml-org/llama.cpp` | ✅ | `/health` | 低 | +| **TensorRT-LLM** | 先離線 `trtllm-build` 再 `trtllm-serve` | `nvcr.io/.../tritonserver` | ✅ | — | 高(暫緩) | + +每個 launcher 就是一個 class:`kind=LLM`、`engine=…`、`capabilities=…`、`keys()`(依 engine 過濾)、 +`build_spec()`(產該引擎的 command + 設對的 `probe_url` + env)+ **自己的 arg builder**。 + +### 5.2 SGLang 具體規格(依 [sglang_related_info.md](sglang_related_info.md),v0.5.14) + +- **command(容器內)**:`python3 -m sglang.launch_server`(對齊官方容器 entrypoint;host 端 `sglang serve` 等價)。 +- **arg builder**(SGLang 專屬,見 §2.3 翻譯表): + - `--model-path `、`--served-model-name `、`--host`、`--port` + - typed 翻譯:`max_model_len→--context-length`、`gpu_memory_utilization→--mem-fraction-static`、 + `tensor_parallel_size→--tp-size`、`dtype→--dtype` + - bool = `store_true`(只 `--flag`,**無** `--no-`);其餘 `extra="allow"` 旗標 kebab-case 後透傳 + - 並行旗標用 canonical `--tp-size` / `--dp-size`(不要假設 `--tp`/`--dp` 存在) +- **probe_url**:`/health`(SGLang 預設 `/health` 會做 1-token 生成檢查,Starting 時回 503 → 正好當 readiness)。 + 例外:若部署設了 `SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION=false`,readiness 改用 `/health_generate`。 +- **GPU**:沿用 `CUDA_VISIBLE_DEVICES`(同 vLLM 路徑);同機多實例切片再配 `--base-gpu-id`/`--gpu-id-step`。 +- **capabilities** = `{runtime_lora, lora_modules, metrics_sglang}`: + - `runtime_lora` ✅:`POST /load_lora_adapter` / `/unload_lora_adapter`(啟動需 `--enable-lora`,建議帶 + `--max-lora-rank` / `--lora-target-modules`) + - `sleep` ❌:SGLang 無 vLLM 式 `/sleep`/`/wake_up`(只有 `--sleep-on-idle`,降 CPU 非釋放 VRAM)→ + autoscaler 暖待命層對 SGLang group 自動退化成 `stop` + - `kv_transfer` ❌ +- **容器需求**:`--ipc=host` 或大 `--shm-size`(SGLang 對 shared memory 敏感);掛 HF cache。 +- **router 零修改**:chat/completions、completions、streaming、final usage chunk(`choices=[]` + usage)、 + `/v1/models`、`/tokenize`、`/detokenize` 全有 → 計費(靠 final usage chunk)可接。 + +> **autoscaling 首版不含**(決定):SGLang 指標自成一格(`sglang:num_queue_reqs` / `num_running_reqs` …, +> 或 `/v1/loads` JSON)。首版只做 launch + route + lifecycle + runtime LoRA;autoscaler 對 SGLang group +> 維持固定副本(退化)。`metrics_sglang` 解析器留作後續階段(屆時才宣告該 capability 真正生效)。 + +## 6. 落地步驟(每步可獨立 commit、跑全測確保零行為變更) + +1. ✅ **抽象層骨架(不新增引擎)** — 已完成、已驗證 + - schema 加 `engine`(預設 vllm);`Launcher` Protocol 加 `engine`/`capabilities`; + `VllmLauncher`/`EmbeddingLauncher` 標上 `engine`/`capabilities`。 + - `_launchers` 改 `(kind, engine)` keyed;C2 各處改 `_launcher_for(inst)`;`keys()` 依 engine 過濾。 + - `ModelInstance`/`LaunchSpec`/`ModelView` 帶 `engine`,fleet 視圖 + HA 入庫一併輸出。 + - **驗收:全 vLLM,backend 386 / router 119 / schema 5 全綠;docker live 跑 vLLM 模型 → READY → 推理 OK。** +2. ✅ **capability gating** — 已完成 + - LoRA 依 `CAP_RUNTIME_LORA` gate;create/update 對未註冊 launcher 的 engine 回乾淨錯誤; + autoscaler sleep 層 / sleep API 既有的 `sleep_enabled` 判斷 → 非 vLLM 自動退化成 `stop`。 + - 測試:一個無能力的假 launcher,確認 gate 正確拒絕 + dispatch 正確。 +3. **SGLang image**:新增 `engine-sglang.Dockerfile`(`FROM lmsysorg/sglang` + backend code), + compose/k8s 加一個用該 image 的 backend 服務範例(含 `--ipc=host` / shm + HF cache 掛載)。 + *(先單機驗證 image 能起 sglang 模型,再進 launcher。)* +4. **`SglangLauncher`**(試金石,§5.2):arg builder + `/health` probe + capabilities。docker 實測: + 同時跑一個 vLLM group + 一個 SGLang group,router 對兩者都能 proxy chat/completions。 +5. **capability 套用回歸**:確認 SGLang group 在 dashboard 不顯示 sleep、autoscaler 對它走退化(固定副本)、 + runtime LoRA 可用。 +6.(後續)**SGLang autoscaling**:新增 `metrics_sglang` 解析器(`sglang:*` 或 `/v1/loads`),宣告該 capability。 +7.(可選)**llama.cpp** / (可選/暫緩)**TensorRT-LLM**。 + +## 7. 風險與不做什麼 + +- **不重寫 router**:它已是引擎無關的 OpenAI proxy。唯一要測的是各引擎對 `/v1/models`、`/tokenize`、 + `/detokenize`、streaming `usage` 的支援差異(router 已有 fallback,逐一驗證即可)。 +- **不在 embedding 引入 engine 維度**(本次):聚焦 `kind=llm`。 +- **不假設功能對等**:任何 vLLM 專屬功能一律走 capabilities,缺能力就退化,**絕不**讓非 vLLM group 因為 + 缺 sleep/lora 端點而當機或被 autoscaler 誤判。 +- **混引擎只在 group 之間**:同 group 同引擎(同模型多副本),避免 instance 級別混搭的複雜度。 + +## 8. 相關檔案索引 + +- [launchers.py](../apps/backend/app/llmops/launchers.py) — Launcher Protocol + Vllm/Embedding 實作(主戰場) +- [manager.py](../apps/backend/app/llmops/manager.py) — `_launchers` 分派、建立/編輯/LoRA(C2) +- [instance.py](../apps/backend/app/llmops/instance.py) — `LaunchSpec` / `ModelInstance`(加 engine/capabilities) +- [state.py](../apps/backend/app/llmops/state.py) — `ModelKind` +- [schema.py](../packages/config-schema/schema.py) — `EngineModelConfig`(加 engine 欄位) +- [autoscaler.py](../apps/backend/app/llmops/autoscaler.py) — sleep 層 capability gate +- [metrics_poller.py](../apps/router-server/src/llm_router/metrics_poller.py) — 指標解析依引擎 +- [router.py](../apps/router-server/src/llm_router/router.py) — OpenAI proxy(預期零修改) +- [process.py](../apps/backend/app/llmops/process.py) — spawn 子行程(决定 engine 必須在 backend 容器內) +- [engine.Dockerfile](../deploy/engine.Dockerfile) — vLLM image;SGLang 對稱新增 `engine-sglang.Dockerfile` +- [sglang_related_info.md](sglang_related_info.md) — SGLang v0.5.14 啟動/旗標/端點/capability 調研(SglangLauncher 依據) diff --git a/docs/sglang_related_info.md b/docs/sglang_related_info.md new file mode 100644 index 0000000..ba8c83b --- /dev/null +++ b/docs/sglang_related_info.md @@ -0,0 +1,882 @@ +# SGLang 最新 Server 啟動整理 + +更新日期: 2026-06-30 +整理基準: + +- 最新穩定版: `v0.5.14` +- GitHub latest release 發布時間: `2026-06-26` +- 官方建議 CLI: `sglang serve` +- 官方 Docker / compose 範例目前仍多使用 `python3 -m sglang.launch_server` + +--- + +## 1. 快速結論 + +如果你是要做 launcher / router / autoscaler 對接,先看這段就夠: + +- 最新版正式推薦入口是 `sglang serve --model-path ...`。 +- `python -m sglang.launch_server ...` 仍然支援,但程式本身已明確警告這是舊但仍相容的入口。 +- 官方 Docker 與 `docker/compose.yaml` 目前仍直接用 `python3 -m sglang.launch_server`,所以容器內沿用這種寫法也沒問題。 +- 指定模型旗標是 `--model-path`,也支援 alias `--model`;可接 Hugging Face repo ID,也可接本地路徑。 +- `host` / `port` 旗標就是 `--host` / `--port`。 +- Tensor parallel 的 canonical 旗標是 `--tp-size`,alias 是 `--tensor-parallel-size`。 +- Data parallel 的 canonical 旗標是 `--dp-size`,alias 是 `--data-parallel-size`。 +- 官方文件範例還常寫 `--tp` / `--dp`,但最新版 `server_args.py` 裡我沒有看到它們被明確註冊成 CLI alias。若你要寫 arg builder,建議以 `--tp-size` / `--dp-size` 為準,不要假設 `--tp` / `--dp` 一定存在。 +- GPU 指定同時牽涉 `CUDA_VISIBLE_DEVICES` 與 SGLang 自己的 `--base-gpu-id` / `--gpu-id-step`。SGLang 內部也會依可見 GPU 重新設 `CUDA_VISIBLE_DEVICES`。 +- `vLLM --max-model-len` 對應 SGLang 的 `--context-length`。 +- `vLLM --gpu-memory-utilization` 沒有完全同名對應;最接近的是 `--mem-fraction-static`,但語意是「靜態配置給權重 + KV cache pool 的顯存比例」,不是逐字等價。 +- Bool 旗標慣例是 `store_true`,也就是通常只有 `--flag`,沒有通用的 `--no-flag` 自動對偶。 +- 但 SGLang 本身有很多「負向命名的正向旗標」,例如 `--disable-radix-cache`、`--skip-server-warmup`。所以不能直接沿用 vLLM 那種 `--no-xxx` builder 邏輯。 +- 旗標命名以 `kebab-case` 為主,例如 `--model-path`、`--context-length`、`--mem-fraction-static`。 +- OpenAI 相容端點有 `/v1/chat/completions`、`/v1/completions`、`/v1/models`、`/v1/tokenize`、`/v1/detokenize`,streaming 支援。 +- `stream_options.include_usage=true` 時,SGLang 會在串流最後補一個 `choices=[]` 的 final usage chunk。 +- 若 `stream_options.continuous_usage_stats=true`,則每個 chunk 都可能帶 `usage`;若 `false`,通常只有最後那個 usage chunk。 +- readiness probe 最準的是 `/health` 或 `/health_generate`。最新版預設下 `/health` 就會做真正的 1-token 生成健康檢查,不只是單純 200。 +- Prometheus metrics 端點是 `/metrics`,需啟動 `--enable-metrics`。 +- 與 autoscaler 最相關的 queue / running 指標已有內建: + - `sglang:num_running_reqs` + - `sglang:num_queue_reqs` + - `sglang:num_grammar_queue_reqs` + - `sglang:num_used_tokens` + - `sglang:token_usage` + - `sglang:gen_throughput` +- 若你不想 parse Prometheus,還可用 `/v1/loads` 直接拿 per-DP rank 的 JSON load 資訊。 +- 沒查到 vLLM 那種 `/sleep` / `/wake_up` 類型、可釋放 GPU 顯存待命的 LLM server API。SGLang 目前只有 `--sleep-on-idle`,它是降低 CPU idle 使用率,不是釋放 VRAM。 +- Runtime LoRA 熱掛載/卸載是有的: `/load_lora_adapter`、`/unload_lora_adapter`。 + +--- + +## 2. 版本與最新狀態 + +### 2.1 最新穩定版 + +- GitHub latest release: [`v0.5.14`](https://github.com/sgl-project/sglang/releases/tag/v0.5.14) +- 發布時間: `2026-06-26` + +### 2.2 文件與 release 的一個小落差 + +官方安裝頁的 source 安裝範例目前還能看到舊 tag 範例,但 latest release 已是 `v0.5.14`。 +如果你要 pin 版本,應優先跟著 release / Docker tag 走,不要照舊文件裡的舊 tag。 + +--- + +## 3. 正式啟動方式 + +### 3.1 最新推薦入口 + +最新版 Python package 有 console script: + +```bash +sglang serve --model-path [options] +``` + +這是目前官方推薦入口。 + +原因: + +- `python/pyproject.toml` 有註冊 `sglang = "sglang.cli.main:main"` +- `python/sglang/cli/main.py` 裡有 `serve` subcommand +- `python/sglang/launch_server.py` 在 `__main__` 中會直接 warning: + - `python -m sglang.launch_server` still supported + - `sglang serve` is the recommended entrypoint + +### 3.2 舊入口是否還能用 + +可以,仍支援: + +```bash +python -m sglang.launch_server --model-path [options] +``` + +而且目前官方 Docker 範例與 `docker/compose.yaml` 也還是這樣啟動。 + +### 3.3 我該選哪個 + +建議: + +- 你自己的 host-side launcher: 用 `sglang serve` +- 你要對齊官方 Docker / compose / 既有腳本: 用 `python -m sglang.launch_server` + +兩者在 LLM server 路徑最後都會進到同一套 server args 與 `run_server(...)` 邏輯。 + +--- + +## 4. Docker 最新整理 + +### 4.1 官方 image + +官方 Docker Hub: + +- [`lmsysorg/sglang`](https://hub.docker.com/r/lmsysorg/sglang/tags) + +官方文件說明: + +- 預設是 CUDA 13 環境 +- 若要 CUDA 12 系列,使用 `-cu12` 或 `-cu129` 後綴 +- 有 nightly tags +- production 建議可考慮 `runtime` 變體 + +### 4.2 目前可確認到的新版 tag + +從 2026-06-30 查到的 tags 可見至少包含: + +- `latest` +- `v0.5.14` +- `latest-cu129` +- `v0.5.14-cu129` +- `latest-cu130` +- `v0.5.14-cu130` +- `latest-cu129-runtime` +- `v0.5.14-cu129-runtime` +- 多組 `nightly-*` + +說明: + +- `latest` 目前對應到 `v0.5.14` +- docs 明確提到 `latest-runtime` +- Docker Hub tag 頁可看到 runtime / nightly / CUDA suffix 系列 + +### 4.3 官方 `docker run` 範例 + +一般版: + +```bash +docker run --gpus all \ + --shm-size 32g \ + -p 30000:30000 \ + -v ~/.cache/huggingface:/root/.cache/huggingface \ + --env "HF_TOKEN=" \ + --ipc=host \ + lmsysorg/sglang:latest \ + python3 -m sglang.launch_server \ + --model-path meta-llama/Llama-3.1-8B-Instruct \ + --host 0.0.0.0 \ + --port 30000 +``` + +runtime 版: + +```bash +docker run --gpus all \ + --shm-size 32g \ + -p 30000:30000 \ + -v ~/.cache/huggingface:/root/.cache/huggingface \ + --env "HF_TOKEN=" \ + --ipc=host \ + lmsysorg/sglang:latest-runtime \ + python3 -m sglang.launch_server \ + --model-path meta-llama/Llama-3.1-8B-Instruct \ + --host 0.0.0.0 \ + --port 30000 +``` + +### 4.4 Docker 啟動注意事項 + +- 建議帶 `--ipc=host` 或足夠大的 `--shm-size` +- 官方文件特別提醒 Docker / Kubernetes 要注意 shared memory +- 通常會掛載 Hugging Face cache: + +```bash +-v ~/.cache/huggingface:/root/.cache/huggingface +``` + +- 若使用 gated repo,通常要帶: + +```bash +--env "HF_TOKEN=" +``` + +### 4.5 官方 docker compose + +官方 `docker/compose.yaml` 目前重點如下: + +- `image: lmsysorg/sglang:latest` +- `entrypoint: python3 -m sglang.launch_server` +- `command: --model-path ... --host 0.0.0.0 --port 30000` +- `network_mode: host` +- `privileged: true` +- `ipc: host` +- `healthcheck: curl -f http://localhost:30000/health || exit 1` +- GPU reservation 使用 NVIDIA device reservation + +如果你要做 K8s / compose 對接,官方健康檢查目前就是打 `/health`。 + +--- + +## 5. 核心啟動指令範本 + +### 5.1 單卡 + +```bash +sglang serve \ + --model-path meta-llama/Meta-Llama-3.1-8B-Instruct \ + --host 0.0.0.0 \ + --port 30000 +``` + +### 5.2 指定 dtype / context / 顯存比例 + +```bash +sglang serve \ + --model-path meta-llama/Meta-Llama-3.1-8B-Instruct \ + --dtype bfloat16 \ + --context-length 32768 \ + --mem-fraction-static 0.8 \ + --host 0.0.0.0 \ + --port 30000 +``` + +### 5.3 Tensor parallel + +建議用 canonical flag: + +```bash +sglang serve \ + --model-path meta-llama/Meta-Llama-3.1-8B-Instruct \ + --tp-size 2 +``` + +也可用 alias: + +```bash +sglang serve \ + --model-path meta-llama/Meta-Llama-3.1-8B-Instruct \ + --tensor-parallel-size 2 +``` + +### 5.4 Data parallel + +```bash +sglang serve \ + --model-path meta-llama/Meta-Llama-3.1-8B-Instruct \ + --dp-size 2 +``` + +或: + +```bash +sglang serve \ + --model-path meta-llama/Meta-Llama-3.1-8B-Instruct \ + --data-parallel-size 2 +``` + +### 5.5 多節點 TP + +```bash +sglang serve \ + --model-path meta-llama/Meta-Llama-3.1-8B-Instruct \ + --tp-size 4 \ + --dist-init-addr sgl-dev-0:50000 \ + --nnodes 2 \ + --node-rank 0 +``` + +第二台把 `--node-rank` 改成 `1`。 + +--- + +## 6. 參數命名與 vLLM 對照 + +## 6.1 高優先級映射表 + +| 需求 | SGLang 最新旗標 | 備註 | +| --- | --- | --- | +| 模型路徑 / HF repo | `--model-path` | alias: `--model` | +| host | `--host` | 預設 `127.0.0.1` | +| port | `--port` | 預設 `30000` | +| dtype | `--dtype` | `auto/half/float16/bfloat16/float/float32` | +| 最大長度 | `--context-length` | 對應 vLLM `--max-model-len` | +| 顯存比例 | `--mem-fraction-static` | 接近 vLLM `--gpu-memory-utilization`,但不是完全同義 | +| Tensor parallel | `--tp-size` | alias: `--tensor-parallel-size` | +| Data parallel | `--dp-size` | alias: `--data-parallel-size` | +| NCCL init addr | `--dist-init-addr` | alias: `--nccl-init-addr` | +| 多節點數 | `--nnodes` | | +| 節點 rank | `--node-rank` | | +| API key | `--api-key` | OpenAI compatible server 也會使用 | +| admin API key | `--admin-api-key` | 管理端點保護 | +| OpenAI `/v1/models` model name | `--served-model-name` | 不設時預設為 `model_path` | +| metrics | `--enable-metrics` | Prometheus `/metrics` | +| MFU metrics | `--enable-mfu-metrics` | 需與 `--enable-metrics` 一起用 | + +## 6.2 關於 `--tp` / `--dp` + +官方文件的 launch examples 仍大量寫: + +- `--tp 2` +- `--dp 2` + +但以最新版 `python/sglang/srt/server_args.py` 來看: + +- `tp_size` 自動生成的是 `--tp-size` +- alias 是 `--tensor-parallel-size` +- `dp_size` 自動生成的是 `--dp-size` +- alias 是 `--data-parallel-size` + +我沒有在最新版 parser 定義裡看到 `--tp` / `--dp` 被明確註冊成正式 alias。 +因此如果你正在做 launcher 的 arg builder,建議: + +- 寫出 `--tp-size` +- 寫出 `--dp-size` +- 不要把 `--tp` / `--dp` 當成唯一可信介面 + +## 6.3 Bool 旗標慣例 + +SGLang 的 dataclass CLI 產生器對 `bool` 使用的是 `store_true`。 + +這代表: + +- 一般情況是 `--flag` 表示設成 `true` +- 沒有通用的 `--no-flag` +- 不能直接套用 vLLM 那種 `BooleanOptionalAction` / `--no-xxx` builder + +但要注意: + +- 很多 SGLang 旗標本身名字就是負向,例如 `--disable-radix-cache` +- 這不是 `--no-radix-cache` 的自動對偶,而是獨立存在的旗標名 + +實務建議: + +- 對 SGLang 寫專屬 bool builder +- 不要假設所有 bool 都能從 `--foo` 推導 `--no-foo` + +## 6.4 命名風格 + +CLI 旗標是 `kebab-case`: + +- `--model-path` +- `--context-length` +- `--mem-fraction-static` +- `--stream-response-default-include-usage` + +對應 dataclass 欄位通常是 snake_case: + +- `model_path` +- `context_length` +- `mem_fraction_static` + +YAML config 檔也建議使用 kebab-case key。 + +--- + +## 7. GPU 指定方式 + +SGLang 不是只有單純吃 `CUDA_VISIBLE_DEVICES`。 + +最新版可以確認到兩層機制: + +### 7.1 外部環境變數 + +PyTorch / CUDA 層面仍然會吃: + +```bash +CUDA_VISIBLE_DEVICES=0,1 +``` + +而 SGLang source 也有明確處理 `CUDA_VISIBLE_DEVICES` 與 logical-to-physical device 映射。 + +### 7.2 SGLang 自己的 GPU 選擇旗標 + +- `--base-gpu-id` +- `--gpu-id-step` + +用途: + +- `--base-gpu-id 2` 可從 GPU 2 開始選 +- `--gpu-id-step 2` 可選 0,2,4 這類間隔 GPU + +### 7.3 對 launcher 的建議 + +如果你有既有 vLLM launcher: + +- 最穩妥方式: 先用 `CUDA_VISIBLE_DEVICES` 限縮可見卡 +- 若還要在同機多實例切片,再配合 `--base-gpu-id` / `--gpu-id-step` + +--- + +## 8. OpenAI 相容性 + +## 8.1 已確認存在的端點 + +最新版 LLM server 有: + +- `/v1/chat/completions` +- `/v1/completions` +- `/v1/embeddings` +- `/v1/models` +- `/v1/models/{model}` +- `/v1/tokenize` +- `/v1/detokenize` +- `/v1/responses` +- `/v1/audio/transcriptions` +- `/v1/realtime` + +也保留部分 legacy / short routes: + +- `/tokenize` +- `/detokenize` + +## 8.2 Streaming 支援 + +支援。 + +`/v1/chat/completions` 和 `/v1/completions` 都有完整 stream 路徑。 + +## 8.3 `stream_options.include_usage` + +這點對 router 計費很重要,結論是: + +- 有支援 `stream_options.include_usage` +- 會在串流最後補一個 usage chunk +- 這個 chunk 的 `choices` 會是空陣列 + +chat stream 實作可看到: + +- 最後會額外組一個 `ChatCompletionStreamResponse(... choices=[], usage=usage)` + +completion stream 也同樣: + +- 最後會額外組一個 `CompletionStreamResponse(... choices=[], usage=usage)` + +所以如果你的 router 是靠 final usage chunk 計費,SGLang 這條路是可接的。 + +## 8.4 `continuous_usage_stats` + +最新版還多了一個能力: + +```json +"stream_options": { + "include_usage": true, + "continuous_usage_stats": true +} +``` + +行為: + +- `continuous_usage_stats=true`: 每個串流 chunk 都可能附 `usage` +- `continuous_usage_stats=false`: 通常只會有最後那個 final usage chunk + +另外還有 server-level 預設旗標: + +- `--stream-response-default-include-usage` + +用途: + +- 就算 client 沒帶 `stream_options.include_usage`,server 也可預設在 streaming response 裡帶 usage + +## 8.5 `/v1/models` 回傳 model name 格式 + +預設: + +- `served_model_name == model_path` + +如果你有帶: + +```bash +--served-model-name my-router-name +``` + +則 `/v1/models` 會回這個名字。 + +而 `/v1/models` 會列出: + +- base model +- 已載入的 LoRA adapters + +base model `ModelCard` 會包含: + +- `id` +- `root` +- `max_model_len` + +LoRA model card 會附帶: + +- `parent` 指向 base model + +## 8.6 `/tokenize` / `/detokenize` + +有。 + +同時提供: + +- `/v1/tokenize` +- `/tokenize` +- `/v1/detokenize` +- `/detokenize` + +如果你的 router 有 tokenizer sidecar 或 prompt 預處理需求,這點可直接接。 + +--- + +## 9. 健康檢查與 readiness + +## 9.1 端點總覽 + +最新版可確認到: + +- `/health` +- `/health_generate` +- `/model_info` +- `/get_model_info` (deprecated) +- `/server_info` +- `/get_server_info` (deprecated) +- `/v1/loads` +- `/ping` (SageMaker health) + +## 9.2 `/health` 與 `/health_generate` + +兩者目前走同一個 handler。 + +而且這個 handler不是單純回 200,它的邏輯是: + +- 若 server 正在 `Starting`,回 `503` +- 否則送一個特殊 request 做 1-token generate / embedding 檢查 +- 只要在 timeout 內收到 scheduler / detokenizer 回應,就算健康 + +但有一個例外: + +- 若環境變數 `SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION=false` +- 且你打的是 `/health` +- 那 `/health` 會直接回 `200` +- 此時 `/health_generate` 才是「真的做生成檢查」的 probe + +最新版預設值是: + +```text +SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION = true +``` + +所以預設情況下: + +- `/health` 就是最準的 readiness probe + +## 9.3 `/model_info` 能不能拿來當 ready + +可以當「HTTP server 起來了」的訊號,但不是最準的「可生成服務」探針。 + +原因: + +- server 內建 warmup 邏輯會先輪詢 `/model_info` +- 確認能取到 model info 後,才再送 warmup generate / encode request + +也就是說: + +- `/model_info` 比較像 control plane / metadata ready +- `/health` 或 `/health_generate` 比較像 data plane ready + +## 9.4 建議 probe 策略 + +如果你是做 orchestrator / reconciler: + +- liveness: + - 可用 `/model_info` + - 或直接沿用官方 `/health` +- readiness: + - 優先用 `/health` + - 若你刻意把 `SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION=false`,則改用 `/health_generate` + +## 9.5 從啟動到 ready 的典型耗時 + +官方沒有給單一固定值,因為差異非常大,取決於: + +- 模型大小 +- 權重是否已快取 +- 是否首次下載 +- 是否會觸發 kernel JIT / warmup +- TP / DP / 多節點設定 + +但從最新版程式行為可推得: + +- 啟動 warmup 會先最多輪詢 `/model_info` 約 `120s` +- `/health` 單次健康檢查 timeout 預設是 `20s` + +實務上可這樣抓量級: + +- 已快取的小模型 / 7B 級: 常見是數十秒 +- 70B 級或首次拉權重 / 首次 JIT: 常見是數分鐘 +- 多節點 / 大量 kernel 預熱: 可能更久 + +這段是依程式邏輯與部署經驗推論,不是官方 SLA。 + +--- + +## 10. LoRA 能力 + +## 10.1 啟動時靜態掛載 + +啟動旗標: + +- `--enable-lora` +- `--lora-paths` +- `--max-lora-rank` +- `--lora-target-modules` +- `--max-loras-per-batch` +- `--max-loaded-loras` +- `--lora-eviction-policy` +- `--enable-lora-overlap-loading` + +`--lora-paths` 支援: + +- `` +- `=` +- JSON 物件格式 + +## 10.2 Runtime 動態載入 / 卸載 + +最新版 LLM server 有: + +- `POST /load_lora_adapter` +- `POST /load_lora_adapter_from_tensors` +- `POST /unload_lora_adapter` + +官方文件與測試都顯示可以在 server 執行中熱掛載。 + +### 10.2.1 載入範例 + +```json +POST /load_lora_adapter +{ + "lora_name": "lora0", + "lora_path": "algoprog/fact-generation-llama-3.1-8b-instruct-lora" +} +``` + +### 10.2.2 卸載範例 + +```json +POST /unload_lora_adapter +{ + "lora_name": "lora0" +} +``` + +### 10.2.3 對 capability 設計的結論 + +- `runtime_lora`: 有 +- `lora_modules`: 有,且建議啟動時顯式指定 `--max-lora-rank` 與 `--lora-target-modules` + +如果不顯式指定,SGLang 可能從初始 `--lora-paths` 推論,之後動態載入的 adapter 就要遵守相容形狀限制。 + +--- + +## 11. Sleep / 待命能力 + +我沒有在最新版 LLM server source 裡查到 vLLM 類似的: + +- `/sleep` +- `/wake_up` +- `--enable-sleep-mode` + +可確認到的只有: + +- `--sleep-on-idle` + +但這個旗標的說明是: + +- `Reduce CPU usage when sglang is idle` + +也就是: + +- 它是 idle 時降低 CPU 使用率 +- 不是釋放 GPU VRAM 的 sleep mode +- 也不是帶 API 的可喚醒待命機制 + +### 對 capability 設計的結論 + +- vLLM 式 `sleep capability`: 目前看起來沒有對等能力 +- autoscaler 若要做 warm standby,不能假設有 `/sleep` / `/wake_up` + +--- + +## 12. Metrics / Autoscaler 訊號 + +## 12.1 啟用方式 + +啟動時加: + +```bash +--enable-metrics +``` + +如需 MFU 類估算指標,再加: + +```bash +--enable-mfu-metrics +``` + +## 12.2 Prometheus 端點 + +端點: + +```text +/metrics +``` + +官方 docs 與測試都以 `http://localhost:30000/metrics` 為例。 + +## 12.3 與 autoscaler 最相關的核心指標 + +最新版 source / docs / tests 可確認至少有: + +- `sglang:num_running_reqs` +- `sglang:num_queue_reqs` +- `sglang:num_grammar_queue_reqs` +- `sglang:num_used_tokens` +- `sglang:token_usage` +- `sglang:gen_throughput` +- `sglang:cache_hit_rate` +- `sglang:prompt_tokens_total` +- `sglang:generation_tokens_total` +- `sglang:cached_tokens_total` +- `sglang:num_requests_total` +- `sglang:time_to_first_token_seconds` +- `sglang:inter_token_latency_seconds` +- `sglang:e2e_request_latency_seconds` +- `sglang:http_requests_active` + +### 對你最關鍵的幾個 + +如果你要做擴縮: + +- 等待中請求數 / queue depth: + - `sglang:num_queue_reqs` + - 另外還有 grammar queue: + - `sglang:num_grammar_queue_reqs` +- 執行中請求數: + - `sglang:num_running_reqs` +- 顯存 / token pool 壓力: + - `sglang:num_used_tokens` + - `sglang:token_usage` +- 吞吐: + - `sglang:gen_throughput` + +## 12.4 `/v1/loads` 可當另一條 autoscaler 訊號來源 + +最新版另有: + +```text +GET /v1/loads +``` + +這個端點可以回 JSON,也可 `format=prometheus`。 + +它的 per-DP rank 欄位包含: + +- `dp_rank` +- `timestamp` +- `num_running_reqs` +- `num_waiting_reqs` +- `num_waiting_uncached_tokens` +- `num_used_tokens` +- `num_total_tokens` +- `max_total_num_tokens` +- `token_usage` +- `gen_throughput` +- `cache_hit_rate` +- `utilization` +- `max_running_requests` + +如果你不想自己 scrape `/metrics` 再 parse family/sample,`/v1/loads` 會更像 control-plane 友善 API。 + +### 建議 + +對 autoscaler: + +- 若你原本已有 Prometheus pipeline,優先吃 `/metrics` +- 若你要在 controller 內即時拉取、且想要結構化 JSON,考慮用 `/v1/loads` + +--- + +## 13. 給 SglangLauncher / Router 的實作建議 + +## 13.1 啟動命令 + +建議分兩層: + +- host / bare-metal launcher: + - `sglang serve` +- Docker 內部 entrypoint: + - `python3 -m sglang.launch_server` + +原因是這樣最貼近最新版官方定位與官方容器實作。 + +## 13.2 Arg builder + +建議你把 SGLang 視為獨立 schema,不要直接重用 vLLM schema。 + +最低限度至少要分開處理: + +- `--context-length` vs vLLM `--max-model-len` +- `--mem-fraction-static` vs vLLM `--gpu-memory-utilization` +- bool flag 不支援通用 `--no-foo` +- canonical parallel flags 建議用 `--tp-size` / `--dp-size` + +## 13.3 Probe URL + +建議: + +- 預設 readiness: `/health` +- 若環境有把 `SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION=false`: + - readiness 改用 `/health_generate` +- 若你還想做 cheap liveness: + - `/model_info` + +## 13.4 OpenAI router 零修改風險 + +大致可接,但有幾點要注意: + +- `/v1/chat/completions` 與 `/v1/completions`: 有 +- streaming: 有 +- final usage chunk: 有 +- `/v1/models`: 有 +- `/tokenize` / `/detokenize`: 有 +- model name 預設等於 `model_path`,若你要 forward_name 穩定,建議顯式傳 `--served-model-name` + +## 13.5 Sleep capability + +目前不要宣稱有 vLLM 對等 sleep/wake 能力。 + +## 13.6 Runtime LoRA capability + +可以宣稱有,但前提是: + +- server 要用 `--enable-lora` +- 最好同時設 `--max-lora-rank` +- 最好同時設 `--lora-target-modules` + +## 13.7 Autoscaler metrics parser + +你需要新增 SGLang-specific parser,至少抓: + +- `sglang:num_running_reqs` +- `sglang:num_queue_reqs` +- `sglang:num_grammar_queue_reqs` +- `sglang:num_used_tokens` +- `sglang:token_usage` +- `sglang:gen_throughput` + +不要直接用 vLLM 指標名去猜。 + +--- + +## 14. 參考來源 + +官方文件 / 官方原始碼 / 官方發布頁: + +- SGLang 安裝與 Docker 文件: [docs.sglang.io/docs/get-started/install](https://docs.sglang.io/docs/get-started/install) +- SGLang server arguments: [docs.sglang.io/docs/advanced_features/server_arguments.html](https://docs.sglang.io/docs/advanced_features/server_arguments.html) +- SGLang production metrics: [docs.sglang.io/docs/references/production_metrics.html](https://docs.sglang.io/docs/references/production_metrics.html) +- GitHub latest release `v0.5.14`: [github.com/sgl-project/sglang/releases/tag/v0.5.14](https://github.com/sgl-project/sglang/releases/tag/v0.5.14) +- Docker Hub tags: [hub.docker.com/r/lmsysorg/sglang/tags](https://hub.docker.com/r/lmsysorg/sglang/tags) +- `python/pyproject.toml`: script entrypoints (`sglang`) +- `python/sglang/cli/main.py`: `serve` subcommand +- `python/sglang/cli/serve.py`: `sglang serve` dispatch logic +- `python/sglang/launch_server.py`: 推薦改用 `sglang serve` 的 warning +- `python/sglang/srt/server_args.py`: 最新 CLI schema +- `python/sglang/srt/entrypoints/http_server.py`: health / OpenAI routes / LoRA routes +- `python/sglang/srt/entrypoints/v1_loads.py`: `/v1/loads` +- `python/sglang/srt/entrypoints/openai/serving_chat.py` +- `python/sglang/srt/entrypoints/openai/serving_completions.py` +- `python/sglang/srt/observability/metrics_collector.py` +- `docs_new/docs/advanced_features/lora.mdx` +- 官方 `docker/compose.yaml` + diff --git a/packages/config-schema/schema.py b/packages/config-schema/schema.py index 6aaca3e..da6774b 100644 --- a/packages/config-schema/schema.py +++ b/packages/config-schema/schema.py @@ -53,7 +53,14 @@ class EngineModelConfig(BaseModel): # silences the spurious "model_" field warnings. model_config = ConfigDict(extra="allow", protected_namespaces=()) model_tag: str - # Router-facing endpoint kind (NOT a vLLM flag — the launcher skips it): + # Which inference engine runs this group. Picks the launcher (and its + # capabilities); orthogonal to `kind` below. Default "vllm" = historical + # behaviour, byte-for-byte unchanged. Engine-specific flags ride extra="allow" + # and are interpreted by that engine's arg builder. A group is single-engine + # (all its instances are replicas of one model); mix engines across groups. + # See docs/multi-backend-engine-design_zh-CN.md. + engine: Literal["vllm", "sglang", "llamacpp", "trtllm"] = "vllm" + # Router-facing endpoint kind (NOT an engine flag — the launcher skips it): # chat -> /v1/chat/completions + /v1/completions (a generate model) # embed -> /v1/embeddings (a vLLM pooling embedding model) # rerank -> /v1/rerank + /v1/score (a vLLM cross-encoder/classify model) From 95cddea3866ef318b6929a096e610bd0e7e69cc9 Mon Sep 17 00:00:00 2001 From: max Date: Tue, 30 Jun 2026 18:23:53 +0800 Subject: [PATCH 02/29] docs(engine): note SGLang Prometheus/Grafana monitoring path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Prometheus can scrape SGLang directly (needs --enable-metrics + scrape target), but the existing vLLM Grafana dashboard queries vllm:* names and stays empty for SGLang's sglang:* metrics — a parallel SGLang dashboard is needed. Folded into the deferred metrics/autoscaling phase (§5.3, §6 step 6). Co-Authored-By: Claude Opus 4.8 --- docs/multi-backend-engine-design_zh-CN.md | 25 +++++++++++++++++++---- 1 file changed, 21 insertions(+), 4 deletions(-) diff --git a/docs/multi-backend-engine-design_zh-CN.md b/docs/multi-backend-engine-design_zh-CN.md index 6c1caae..0aeda06 100644 --- a/docs/multi-backend-engine-design_zh-CN.md +++ b/docs/multi-backend-engine-design_zh-CN.md @@ -261,9 +261,23 @@ engine-sglang.Dockerfile FROM lmsysorg/sglang:latest + backend code → - **router 零修改**:chat/completions、completions、streaming、final usage chunk(`choices=[]` + usage)、 `/v1/models`、`/tokenize`、`/detokenize` 全有 → 計費(靠 final usage chunk)可接。 -> **autoscaling 首版不含**(決定):SGLang 指標自成一格(`sglang:num_queue_reqs` / `num_running_reqs` …, -> 或 `/v1/loads` JSON)。首版只做 launch + route + lifecycle + runtime LoRA;autoscaler 對 SGLang group -> 維持固定副本(退化)。`metrics_sglang` 解析器留作後續階段(屆時才宣告該 capability 真正生效)。 +### 5.3 監控(Prometheus / Grafana):指標進得去,現有 vLLM dashboard 不會亮 + +- **Prometheus 抓取 ✅**(需兩個前提):SGLang 啟動要帶 **`--enable-metrics`** 才有 `/metrics`(標準 + Prometheus 格式);且要把 SGLang 實例**寫進 scrape targets**(現在 backend 的 file_sd 只寫 vLLM 實例, + 見 [metrics_poller](../apps/router-server/src/llm_router/metrics_poller.py) `write_prometheus_targets`)。 +- **現有 vLLM Grafana dashboard ❌ 不能直接用**:panel query 全是 `vllm:*`,SGLang 吐的是 `sglang:*` + (名稱不同),所以 SGLang 實例的數據進得了 Prometheus、但現有 dashboard 的圖是空的。概念 1:1 對得上 + (queue=`sglang:num_queue_reqs`、running=`sglang:num_running_reqs`、throughput=`sglang:gen_throughput`、 + TTFT=`sglang:time_to_first_token_seconds`、cache=`sglang:cache_hit_rate`),所以解法是**另做一份並列的 + SGLang dashboard**(或加 dashboard variable 切引擎)。 +- 這跟 autoscaler 訊號是**同一塊**(都吃 `sglang:*`),所以綁在一起當後續階段做。 + +> **autoscaling + 監控 首版不含**(決定):SGLang 指標自成一格(`sglang:num_queue_reqs` / +> `num_running_reqs` …,或 `/v1/loads` JSON)。首版只做 launch + route + lifecycle + runtime LoRA; +> autoscaler 對 SGLang group 維持固定副本(退化),Grafana SGLang 圖暫時空著(不影響起模型/路由/推理)。 +> `metrics_sglang` 解析器 + scrape target(`--enable-metrics`)+ SGLang Grafana dashboard 一起留作後續階段 +> (屆時才宣告 `metrics_sglang` capability 真正生效)。 ## 6. 落地步驟(每步可獨立 commit、跑全測確保零行為變更) @@ -284,7 +298,10 @@ engine-sglang.Dockerfile FROM lmsysorg/sglang:latest + backend code → 同時跑一個 vLLM group + 一個 SGLang group,router 對兩者都能 proxy chat/completions。 5. **capability 套用回歸**:確認 SGLang group 在 dashboard 不顯示 sleep、autoscaler 對它走退化(固定副本)、 runtime LoRA 可用。 -6.(後續)**SGLang autoscaling**:新增 `metrics_sglang` 解析器(`sglang:*` 或 `/v1/loads`),宣告該 capability。 +6.(後續)**SGLang 監控 + autoscaling**(同一塊,都吃 `sglang:*`): + - launcher 帶 `--enable-metrics`;把 SGLang 實例寫進 Prometheus file_sd scrape targets。 + - 新增 `metrics_sglang` 解析器(`sglang:*` 或 `/v1/loads` JSON),餵 autoscaler;宣告 `metrics_sglang` capability。 + - 另做一份並列的 **SGLang Grafana dashboard**(panel query 換成 `sglang:*`)。 7.(可選)**llama.cpp** / (可選/暫緩)**TensorRT-LLM**。 ## 7. 風險與不做什麼 From 9332288452e6edc778a43fb9c95544c59a96bc74 Mon Sep 17 00:00:00 2001 From: max Date: Tue, 30 Jun 2026 19:01:18 +0800 Subject: [PATCH 03/29] feat(engine): SGLang launcher + engine image (launch/route/infer) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Add SGLang as the first second engine on the multi-backend abstraction. - engine-sglang.Dockerfile: symmetric to engine.Dockerfile but FROM lmsysorg/sglang — same backend code, drops vllm (not in this base), keeps sglang/torch from the base untouched. A backend on this image launches SGLang; a backend on the vLLM image launches vLLM (engine == what's in the image). - SglangLauncher + build_sglang_cli_args: --model-path + --served-model-name (stable forward_name), translates the engine-neutral typed params to SGLang's names (max_model_len->--context-length, gpu_memory_utilization-> --mem-fraction-static, tensor_parallel_size->--tp-size), store_true bools (no --no- dual), /health probe. capabilities=frozenset() for now — SGLang has runtime LoRA + sglang:* metrics but those need extra wiring (per-engine LoRA endpoint path, a metrics parser), declared in a follow-up; no sleep, so the autoscaler degrades to ready<->stopped. - main.py registers SglangLauncher alongside Vllm/Embedding. Tests: 8 new launcher tests (arg translation, store_true bool, keys filter, capabilities, build_spec); backend 394 green. Live e2e verified on docker with a Postgres store (proving store is decoupled from the engine): reconciler spawns sglang via the launcher -> READY -> router proxies inference (non-stream + stream final usage chunk) -> /sleep correctly rejected (409, no sleep capability). Co-Authored-By: Claude Opus 4.8 --- apps/backend/app/llmops/launchers.py | 115 ++++++++++++++++++++++ apps/backend/app/main.py | 4 +- apps/backend/tests/unit/test_launchers.py | 93 ++++++++++++++++- deploy/engine-sglang.Dockerfile | 35 +++++++ docs/multi-backend-engine-design_zh-CN.md | 17 ++-- 5 files changed, 253 insertions(+), 11 deletions(-) create mode 100644 deploy/engine-sglang.Dockerfile diff --git a/apps/backend/app/llmops/launchers.py b/apps/backend/app/llmops/launchers.py index b61bb00..c2ba5fd 100644 --- a/apps/backend/app/llmops/launchers.py +++ b/apps/backend/app/llmops/launchers.py @@ -289,3 +289,118 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: port=emb.port, probe_url=f"http://{emb.host}:{emb.port}/health", ) + + +# ---- SGLang --------------------------------------------------------------- + +# Engine-neutral typed params (EngineModelConfig) -> SGLang flag names. These three +# have different names from vLLM; everything else falls through as a kebab-cased +# -- (engine-native passthrough via extra="allow"). See §2.3 of the design doc. +_SGLANG_PARAM_MAP = { + "max_model_len": "context-length", + "gpu_memory_utilization": "mem-fraction-static", + "tensor_parallel_size": "tp-size", +} +# Keys consumed specially (model_tag/served_model_name handled up front; id is the +# key; cuda_device becomes CUDA_VISIBLE_DEVICES) plus the router-only knobs. +_SGLANG_SKIP_CLI_KEYS = frozenset( + {"model_tag", "served_model_name", "id", "cuda_device", _LORA_RUNTIME_KEY} +) | _ROUTER_ONLY_KEYS + + +def build_sglang_cli_args(model_cfg: dict) -> list[str]: + """dict -> ``python -m sglang.launch_server`` CLI args. + + Unlike vLLM: the model is ``--model-path`` (not a positional), and we always + emit ``--served-model-name`` so ``/v1/models`` (and the router's forward_name) + is the stable ``model_tag`` rather than SGLang's default of the raw path. + + Bools are SGLang's ``store_true``: a True value emits ``--flag``; a False value + is *omitted* — there is NO ``--no-flag`` dual (vLLM's BooleanOptionalAction), + and many SGLang flags are themselves negative (``--disable-radix-cache``), so we + must not synthesise ``--no-`` forms. The three common params above are + translated; the rest pass through kebab-cased. + """ + model_tag = model_cfg.get("model_tag") + if not model_tag: + raise ValueError("model_config must provide 'model_tag'") + served = model_cfg.get("served_model_name") or model_tag + + args = ["--model-path", str(model_tag), "--served-model-name", str(served)] + for key, value in model_cfg.items(): + if key in _SGLANG_SKIP_CLI_KEYS or value is None: + continue + flag = "--" + _SGLANG_PARAM_MAP.get(key, key).replace("_", "-") + if isinstance(value, bool): + if value: # store_true: True -> present; False -> omit + args.append(flag) + elif isinstance(value, list): + # SGLang multi-value flags take space-separated values (e.g. --lora-paths). + args.append(flag) + args.extend(str(v) for v in value) + elif isinstance(value, dict): + args.append(flag) + args.append(json.dumps(value, ensure_ascii=False)) + else: + args.append(flag) + args.append(str(value)) + return args + + +class SglangLauncher: + kind = ModelKind.LLM + engine = "sglang" + # First cut covers launch + route + lifecycle. SGLang *does* have runtime LoRA + # (POST /load_lora_adapter — note: no /v1 prefix, unlike vLLM) and sglang:* Prometheus + # metrics, but both need extra wiring (a per-engine LoRA endpoint path + a + # sglang metrics parser) — declared in a follow-up so we never advertise a + # capability that isn't end-to-end wired. No sleep (SGLang has no /sleep+/wake_up; + # `--sleep-on-idle` only lowers CPU, doesn't free VRAM) -> autoscaler degrades to + # ready<->stopped. See docs/multi-backend-engine-design_zh-CN.md §5.2. + capabilities = frozenset() + + def keys(self, config) -> list[str]: + out: list[str] = [] + for model_tag, engine in config.LLM_engines.items(): + if getattr(engine.settings, "engine", "vllm") != self.engine: + continue + for inst in engine.instances: + out.append(f"{model_tag}::{inst.id}") + return out + + def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: + model_tag, _, instance_id = key.partition("::") + engine = config.LLM_engines.get(model_tag) + if engine is None: + raise KeyError(f"model group '{model_tag}' not in config") + inst = next((i for i in engine.instances if i.id == instance_id), None) + if inst is None: + raise KeyError(f"instance '{instance_id}' not in group '{model_tag}'") + + # Merge shared model_config with instance overrides, mirroring VllmLauncher: + # single-GPU cuda_device -> CUDA_VISIBLE_DEVICES; the `id` field is dropped. + merged: dict = engine.settings.model_dump(by_alias=False) + merged.update(inst.model_dump()) + + env: dict[str, str] = {} + if merged.get("tensor_parallel_size", 1) == 1: + cuda_device = merged.pop("cuda_device", None) + if cuda_device is not None: + env["CUDA_VISIBLE_DEVICES"] = str(cuda_device) + merged.pop("id", None) + + command = [sys.executable, "-m", "sglang.launch_server"] + build_sglang_cli_args(merged) + log_path = os.path.join(LOG_DIR, f"{model_tag}__{instance_id}.log") + return LaunchSpec( + key=key, + kind=self.kind, + engine=self.engine, + capabilities=self.capabilities, + command=command, + env=env, + log_path=log_path, + host=inst.host, + port=inst.port, + probe_url=f"http://{inst.host}:{inst.port}/health", + model_tag=engine.settings.model_tag, + ) diff --git a/apps/backend/app/main.py b/apps/backend/app/main.py index 22ec514..6506af7 100644 --- a/apps/backend/app/main.py +++ b/apps/backend/app/main.py @@ -36,7 +36,7 @@ from app.core.logging import setup_logging from app.core.settings import BackendSettings from app.core.store import LLMOpsStore -from app.llmops.launchers import EmbeddingLauncher, VllmLauncher +from app.llmops.launchers import EmbeddingLauncher, SglangLauncher, VllmLauncher from app.llmops.manager import ModelManager, build_registry from app.llmops.autoscaler import autoscaler_loop from app.llmops.scheduler import Scheduler @@ -113,7 +113,7 @@ async def lifespan(app: FastAPI): # Base config.yaml + dynamically-added models (overlay), merged into one view. config = build_merged_config(config_path) - launchers = [VllmLauncher(), EmbeddingLauncher()] + launchers = [VllmLauncher(), SglangLauncher(), EmbeddingLauncher()] registry = build_registry(config, config_path, launchers) # `or` (not get's default) so an env var set-but-empty (as the compose env # passes it) still falls back instead of yielding "" — an empty router_url diff --git a/apps/backend/tests/unit/test_launchers.py b/apps/backend/tests/unit/test_launchers.py index faf4d26..ea30315 100644 --- a/apps/backend/tests/unit/test_launchers.py +++ b/apps/backend/tests/unit/test_launchers.py @@ -3,8 +3,9 @@ import pytest from app.llmops.launchers import (CAP_SLEEP, EMBEDDING_KEY, ENGINE_DEFAULT, - EmbeddingLauncher, VllmLauncher, - _write_effective_config, build_vllm_cli_args) + EmbeddingLauncher, SglangLauncher, VllmLauncher, + _write_effective_config, build_sglang_cli_args, + build_vllm_cli_args) from app.llmops.state import ModelKind from schema import RootConfig from tests.conftest import FAKE_CONFIG @@ -300,3 +301,91 @@ def test_engine_is_never_passed_to_vllm_cli(): args = build_vllm_cli_args({"model_tag": "org/m", "engine": "vllm", "kind": "chat"}) assert "--engine" not in args assert "--kind" not in args + + +# ---- SGLang launcher (docs/multi-backend-engine-design_zh-CN.md §5.2) -------- + +def _sglang_config(extra: dict | None = None) -> RootConfig: + mc = {"model_tag": "Qwen/Qwen3-0.6B", "engine": "sglang"} + if extra: + mc.update(extra) + return RootConfig.model_validate({ + "server": {"host": "0.0.0.0", "port": 8887}, + "LLM_engines": {"S": { + "instances": [{"id": "a", "host": "localhost", "port": 8100, "cuda_device": 2}], + "model_config": mc, + }}, + }) + + +def test_sglang_args_model_path_and_served_name(): + args = build_sglang_cli_args({"model_tag": "Qwen/Qwen3-0.6B"}) + assert args[:2] == ["--model-path", "Qwen/Qwen3-0.6B"] + # served-model-name defaults to model_tag so /v1/models + forward_name are stable. + assert "--served-model-name" in args + assert args[args.index("--served-model-name") + 1] == "Qwen/Qwen3-0.6B" + + +def test_sglang_args_translate_typed_params(): + # The three common params have different SGLang flag names. + args = build_sglang_cli_args({ + "model_tag": "org/m", "max_model_len": 4096, + "gpu_memory_utilization": 0.45, "tensor_parallel_size": 2, + }) + assert args[args.index("--context-length") + 1] == "4096" + assert args[args.index("--mem-fraction-static") + 1] == "0.45" + assert args[args.index("--tp-size") + 1] == "2" + # vLLM names must NOT appear. + assert "--max-model-len" not in args + assert "--gpu-memory-utilization" not in args + + +def test_sglang_args_bool_is_store_true_no_dual(): + args = build_sglang_cli_args({ + "model_tag": "org/m", "disable_radix_cache": True, "enable_metrics": False, + }) + assert "--disable-radix-cache" in args # True -> present + assert "--enable-metrics" not in args # False -> omitted + assert "--no-enable-metrics" not in args # never synthesise a --no- dual + + +def test_sglang_args_skip_router_only_keys(): + args = build_sglang_cli_args( + {"model_tag": "org/m", "engine": "sglang", "kind": "chat", + "routing_strategy": "session_affinity", "dtype": "bfloat16"}) + assert "--engine" not in args and "--kind" not in args + assert "--routing-strategy" not in args + assert args[args.index("--dtype") + 1] == "bfloat16" # real flags still pass through + + +def test_sglang_args_explicit_served_name_wins(): + args = build_sglang_cli_args({"model_tag": "org/m", "served_model_name": "my-name"}) + assert args[args.index("--served-model-name") + 1] == "my-name" + + +def test_sglang_launcher_claims_only_sglang_engine(): + s = SglangLauncher() + assert s.kind == ModelKind.LLM and s.engine == "sglang" + assert s.keys(_sglang_config()) == ["S::a"] + assert s.keys(FAKE_CONFIG) == [] # FAKE_CONFIG is all vLLM + assert VllmLauncher().keys(_sglang_config()) == [] # vLLM doesn't claim sglang group + + +def test_sglang_launcher_capabilities_no_sleep(): + # First cut: no capabilities advertised (sleep absent in SGLang; LoRA/metrics deferred). + assert SglangLauncher().capabilities == frozenset() + assert CAP_SLEEP not in SglangLauncher().capabilities + + +def test_sglang_build_spec(): + spec = SglangLauncher().build_spec(_sglang_config({"max_model_len": 4096}), "config.yaml", "S::a") + assert spec.engine == "sglang" + assert spec.command[1:3] == ["-m", "sglang.launch_server"] + assert "--model-path" in spec.command + assert spec.command[spec.command.index("--context-length") + 1] == "4096" + # single-GPU cuda_device -> env, not a CLI flag; id dropped. + assert spec.env["CUDA_VISIBLE_DEVICES"] == "2" + assert "--cuda-device" not in spec.command and "--id" not in spec.command + assert spec.probe_url == "http://localhost:8100/health" + assert spec.host == "localhost" and spec.port == 8100 + assert "--host" in spec.command and spec.command[spec.command.index("--host") + 1] == "localhost" diff --git a/deploy/engine-sglang.Dockerfile b/deploy/engine-sglang.Dockerfile new file mode 100644 index 0000000..ee17367 --- /dev/null +++ b/deploy/engine-sglang.Dockerfile @@ -0,0 +1,35 @@ +# syntax=docker/dockerfile:1 +# +# SGLang variant of the engine image (see engine.Dockerfile for the vLLM one). +# Multi-backend design: each inference engine gets its own backend image, built +# FROM that engine's official base, because vLLM / SGLang pin conflicting +# torch/CUDA/flashinfer and the launcher spawns the engine as a subprocess *inside +# this container* (apps/backend/app/llmops/process.py). So "which engines a backend +# can launch" == "what's installed in its image". +# +# This image runs the control-plane backend (which shells out to +# `python -m sglang.launch_server`). It is byte-for-byte the same backend code as +# the vLLM image — only the base (= the engine CLI available) differs. +# See docs/multi-backend-engine-design_zh-CN.md §5. +FROM lmsysorg/sglang:latest + +WORKDIR /app + +COPY apps/backend/requirements.txt /tmp/backend-req.txt +COPY apps/router-server/requirements.txt /tmp/router-req.txt +# vllm is not in this base and we don't want it (this image only launches SGLang); +# sglang is already in the base; pytest* are dev-only. Drop them so we don't pull a +# multi-GB vllm wheel or reinstall sglang. +RUN sed -i -E '/^(vllm|sglang|pytest.*)$/d' /tmp/router-req.txt /tmp/backend-req.txt \ + && pip install --no-cache-dir -r /tmp/backend-req.txt -r /tmp/router-req.txt + +# App code + shared packages — same layout as the vLLM image so the in-code +# sys.path bootstrap and default config/overlay/db paths resolve to /app. +COPY apps/backend ./apps/backend +COPY apps/router-server ./apps/router-server +COPY packages ./packages + +# The SGLang base image sets its own ENTRYPOINT; clear it so the compose +# `command:` (uvicorn for the backend) runs verbatim. +ENTRYPOINT [] +CMD ["bash"] diff --git a/docs/multi-backend-engine-design_zh-CN.md b/docs/multi-backend-engine-design_zh-CN.md index 0aeda06..a9b340f 100644 --- a/docs/multi-backend-engine-design_zh-CN.md +++ b/docs/multi-backend-engine-design_zh-CN.md @@ -291,13 +291,16 @@ engine-sglang.Dockerfile FROM lmsysorg/sglang:latest + backend code → - LoRA 依 `CAP_RUNTIME_LORA` gate;create/update 對未註冊 launcher 的 engine 回乾淨錯誤; autoscaler sleep 層 / sleep API 既有的 `sleep_enabled` 判斷 → 非 vLLM 自動退化成 `stop`。 - 測試:一個無能力的假 launcher,確認 gate 正確拒絕 + dispatch 正確。 -3. **SGLang image**:新增 `engine-sglang.Dockerfile`(`FROM lmsysorg/sglang` + backend code), - compose/k8s 加一個用該 image 的 backend 服務範例(含 `--ipc=host` / shm + HF cache 掛載)。 - *(先單機驗證 image 能起 sglang 模型,再進 launcher。)* -4. **`SglangLauncher`**(試金石,§5.2):arg builder + `/health` probe + capabilities。docker 實測: - 同時跑一個 vLLM group + 一個 SGLang group,router 對兩者都能 proxy chat/completions。 -5. **capability 套用回歸**:確認 SGLang group 在 dashboard 不顯示 sleep、autoscaler 對它走退化(固定副本)、 - runtime LoRA 可用。 +3. ✅ **SGLang image** — 已完成:[engine-sglang.Dockerfile](../deploy/engine-sglang.Dockerfile) + (`FROM lmsysorg/sglang` + 同一份 backend code;砍掉 vllm,sglang/torch 來自 base 不動)。 + 驗證:image 不含 vllm、backend FastAPI 在其中正常 import/boot。 +4. ✅ **`SglangLauncher`**(§5.2)— 已完成、已 live 驗證:arg builder(typed 翻譯 + store_true bool + + `--served-model-name`)+ `/health` probe + `capabilities=frozenset()`(LoRA/metrics 列後續)。 + docker 實測(**store 用 Postgres**,證明 store 與引擎解耦):reconciler 經 SglangLauncher spawn + `python -m sglang.launch_server` → READY → router proxy 推理(非串流 + 串流 final usage chunk)OK。 + 單元測試 8 條(arg 翻譯/bool/keys 過濾/capabilities/build_spec),全測 394 綠。 +5. ✅ **capability 回歸** — 已驗證:SGLang group 的 `/sleep` 被擋(409,無 sleep 能力)、API 顯示 + `engine=sglang`、autoscaler 對它退化(固定副本)。runtime LoRA 屬後續(步驟 6)。 6.(後續)**SGLang 監控 + autoscaling**(同一塊,都吃 `sglang:*`): - launcher 帶 `--enable-metrics`;把 SGLang 實例寫進 Prometheus file_sd scrape targets。 - 新增 `metrics_sglang` 解析器(`sglang:*` 或 `/v1/loads` JSON),餵 autoscaler;宣告 `metrics_sglang` capability。 From ae25e9692cde8a0dcf41378cf2100a8b5b11107b Mon Sep 17 00:00:00 2001 From: max Date: Tue, 30 Jun 2026 19:41:54 +0800 Subject: [PATCH 04/29] feat(engine): SGLang runtime LoRA (per-engine endpoint + --enable-lora) Make the SGLang LoRA capability real instead of deferred. - SglangLauncher capabilities = {runtime_lora, lora_modules}; the arg builder emits --enable-lora when any LoRA usage is configured (static modules, enable_lora, or the allow_runtime_lora runtime toggle) and static adapters as SGLang's --lora-paths NAME=PATH (not vLLM's --lora-modules JSON). Other LoRA knobs (--max-lora-rank, --lora-target-modules) pass through. - manager._post_lora picks the endpoint path by engine: SGLang serves /_lora_adapter (no /v1), vLLM serves /v1/_lora_adapter; same body. Tests: 4 new (sglang lora args, runtime toggle enables --enable-lora, per-engine endpoint path); backend 398 green. Live e2e: hot-loaded a real Qwen3-0.6B adapter through our API -> appears in sglang /v1/models -> adapter inference works -> unload removes it. Co-Authored-By: Claude Opus 4.8 --- apps/backend/app/llmops/launchers.py | 37 +++++++++++---- apps/backend/app/llmops/manager.py | 7 ++- apps/backend/tests/unit/test_launchers.py | 46 +++++++++++++++---- .../backend/tests/unit/test_manager_engine.py | 32 +++++++++++++ docs/multi-backend-engine-design_zh-CN.md | 23 ++++++---- 5 files changed, 116 insertions(+), 29 deletions(-) diff --git a/apps/backend/app/llmops/launchers.py b/apps/backend/app/llmops/launchers.py index c2ba5fd..fa72a11 100644 --- a/apps/backend/app/llmops/launchers.py +++ b/apps/backend/app/llmops/launchers.py @@ -302,9 +302,11 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: "tensor_parallel_size": "tp-size", } # Keys consumed specially (model_tag/served_model_name handled up front; id is the -# key; cuda_device becomes CUDA_VISIBLE_DEVICES) plus the router-only knobs. +# key; cuda_device becomes CUDA_VISIBLE_DEVICES; enable_lora/lora_modules drive the +# LoRA flags below; allow_runtime_lora is an enable_lora trigger) + router-only knobs. _SGLANG_SKIP_CLI_KEYS = frozenset( - {"model_tag", "served_model_name", "id", "cuda_device", _LORA_RUNTIME_KEY} + {"model_tag", "served_model_name", "id", "cuda_device", + "enable_lora", "lora_modules", _LORA_RUNTIME_KEY} ) | _ROUTER_ONLY_KEYS @@ -327,6 +329,22 @@ def build_sglang_cli_args(model_cfg: dict) -> list[str]: served = model_cfg.get("served_model_name") or model_tag args = ["--model-path", str(model_tag), "--served-model-name", str(served)] + + # LoRA: SGLang needs --enable-lora at launch to accept either static adapters + # (--lora-paths NAME=PATH …) or runtime ones (POST /load_lora_adapter). Turn it + # on if any LoRA usage is configured (static modules, enable_lora, or our runtime + # toggle). Static modules use SGLang's NAME=PATH form (not vLLM's --lora-modules + # JSON). Other LoRA knobs (max_lora_rank, lora_target_modules, …) pass through + # below as plain kebab-cased flags. + lora_modules = model_cfg.get("lora_modules") or [] + if model_cfg.get("enable_lora") or model_cfg.get(_LORA_RUNTIME_KEY) or lora_modules: + args.append("--enable-lora") + paths = [f"{m['name']}={m['path']}" for m in lora_modules + if m.get("name") and m.get("path")] + if paths: + args.append("--lora-paths") + args.extend(paths) + for key, value in model_cfg.items(): if key in _SGLANG_SKIP_CLI_KEYS or value is None: continue @@ -350,14 +368,13 @@ def build_sglang_cli_args(model_cfg: dict) -> list[str]: class SglangLauncher: kind = ModelKind.LLM engine = "sglang" - # First cut covers launch + route + lifecycle. SGLang *does* have runtime LoRA - # (POST /load_lora_adapter — note: no /v1 prefix, unlike vLLM) and sglang:* Prometheus - # metrics, but both need extra wiring (a per-engine LoRA endpoint path + a - # sglang metrics parser) — declared in a follow-up so we never advertise a - # capability that isn't end-to-end wired. No sleep (SGLang has no /sleep+/wake_up; - # `--sleep-on-idle` only lowers CPU, doesn't free VRAM) -> autoscaler degrades to - # ready<->stopped. See docs/multi-backend-engine-design_zh-CN.md §5.2. - capabilities = frozenset() + # Runtime + static LoRA are wired (--enable-lora / --lora-paths at launch; + # POST /load_lora_adapter — no /v1 prefix, unlike vLLM — for hot load/unload). + # No sleep (SGLang has no /sleep+/wake_up; `--sleep-on-idle` only lowers CPU, + # doesn't free VRAM) -> autoscaler degrades to ready<->stopped. sglang:* metrics + # exist but need a parser, so CAP_METRICS_* is left for a follow-up. + # See docs/multi-backend-engine-design_zh-CN.md §5.2. + capabilities = frozenset({CAP_RUNTIME_LORA, CAP_LORA_MODULES}) def keys(self, config) -> list[str]: out: list[str] = [] diff --git a/apps/backend/app/llmops/manager.py b/apps/backend/app/llmops/manager.py index 14170b8..9b50799 100644 --- a/apps/backend/app/llmops/manager.py +++ b/apps/backend/app/llmops/manager.py @@ -736,8 +736,11 @@ async def _ready_lora_targets(self, group: str) -> list[ModelInstance]: ] async def _post_lora(self, inst: ModelInstance, action: str, payload: dict) -> None: - """POST /v1/{load,unload}_lora_adapter to one instance; raise on failure.""" - url = f"http://{inst.host}:{inst.port}/v1/{action}_lora_adapter" + """POST {load,unload}_lora_adapter to one instance; raise on failure. The + path differs by engine: vLLM serves /v1/_lora_adapter, SGLang serves + /_lora_adapter (no /v1). The JSON body is the same for both.""" + prefix = "" if getattr(inst, "engine", "vllm") == "sglang" else "/v1" + url = f"http://{inst.host}:{inst.port}{prefix}/{action}_lora_adapter" resp = await self.http_client.post(url, json=payload, timeout=120.0) if resp.status_code >= 400: try: diff --git a/apps/backend/tests/unit/test_launchers.py b/apps/backend/tests/unit/test_launchers.py index ea30315..d609592 100644 --- a/apps/backend/tests/unit/test_launchers.py +++ b/apps/backend/tests/unit/test_launchers.py @@ -2,10 +2,10 @@ import pytest -from app.llmops.launchers import (CAP_SLEEP, EMBEDDING_KEY, ENGINE_DEFAULT, - EmbeddingLauncher, SglangLauncher, VllmLauncher, - _write_effective_config, build_sglang_cli_args, - build_vllm_cli_args) +from app.llmops.launchers import (CAP_LORA_MODULES, CAP_RUNTIME_LORA, CAP_SLEEP, + EMBEDDING_KEY, ENGINE_DEFAULT, EmbeddingLauncher, + SglangLauncher, VllmLauncher, _write_effective_config, + build_sglang_cli_args, build_vllm_cli_args) from app.llmops.state import ModelKind from schema import RootConfig from tests.conftest import FAKE_CONFIG @@ -371,10 +371,40 @@ def test_sglang_launcher_claims_only_sglang_engine(): assert VllmLauncher().keys(_sglang_config()) == [] # vLLM doesn't claim sglang group -def test_sglang_launcher_capabilities_no_sleep(): - # First cut: no capabilities advertised (sleep absent in SGLang; LoRA/metrics deferred). - assert SglangLauncher().capabilities == frozenset() - assert CAP_SLEEP not in SglangLauncher().capabilities +def test_sglang_launcher_capabilities(): + # Runtime + static LoRA are wired; sleep is absent in SGLang (degrades), and + # metrics need a parser so CAP_METRICS_* is not advertised yet. + caps = SglangLauncher().capabilities + assert CAP_RUNTIME_LORA in caps and CAP_LORA_MODULES in caps + assert CAP_SLEEP not in caps + + +def test_sglang_args_lora_enable_and_paths(): + # enable_lora -> --enable-lora; static lora_modules -> SGLang NAME=PATH form. + args = build_sglang_cli_args({ + "model_tag": "org/m", "enable_lora": True, + "lora_modules": [{"name": "sql", "path": "/lora/sql"}, + {"name": "math", "path": "/lora/math"}], + "max_lora_rank": 16, + }) + assert "--enable-lora" in args + i = args.index("--lora-paths") + assert args[i + 1] == "sql=/lora/sql" and args[i + 2] == "math=/lora/math" + assert args[args.index("--max-lora-rank") + 1] == "16" # other lora knobs pass through + assert "--lora-modules" not in args # not vLLM's JSON form + + +def test_sglang_args_runtime_lora_toggle_enables_lora(): + # Our runtime toggle (allow_runtime_lora) must turn on --enable-lora so SGLang + # accepts POST /load_lora_adapter, even with no static modules. + args = build_sglang_cli_args({"model_tag": "org/m", "allow_runtime_lora": True}) + assert "--enable-lora" in args + assert "--allow-runtime-lora" not in args # the toggle is not a flag + + +def test_sglang_args_no_lora_by_default(): + args = build_sglang_cli_args({"model_tag": "org/m"}) + assert "--enable-lora" not in args and "--lora-paths" not in args def test_sglang_build_spec(): diff --git a/apps/backend/tests/unit/test_manager_engine.py b/apps/backend/tests/unit/test_manager_engine.py index 683f271..414a563 100644 --- a/apps/backend/tests/unit/test_manager_engine.py +++ b/apps/backend/tests/unit/test_manager_engine.py @@ -176,3 +176,35 @@ async def test_create_overlay_model_rejects_unregistered_engine(tmp_path): "Brand", {"id": "z", "host": "localhost", "port": 8040}, {"model_tag": "org/brand", "engine": "trtllm"}, ) + + +# ---- per-engine runtime LoRA endpoint path ---------------------------------- + +def _inst(engine: str, port: int): + from app.llmops.instance import LaunchSpec, ModelInstance + spec = LaunchSpec(key=f"G::a", kind=ModelKind.LLM, engine=engine, capabilities=frozenset(), + command=[], env={}, log_path="x", host="localhost", port=port, + probe_url=f"http://localhost:{port}/health") + return ModelInstance(key="G::a", kind=ModelKind.LLM, engine=engine, host="localhost", + port=port, spec=spec) + + +async def test_post_lora_endpoint_path_per_engine(tmp_path): + from tests.conftest import FakeHTTPClient + client = FakeHTTPClient() + mgr = ModelManager(build_registry(load_config(_write_min_cfg(tmp_path)), str(tmp_path/'c.yaml'), + [VllmLauncher()]), + [VllmLauncher()], client, load_config(_write_min_cfg(tmp_path)), + str(tmp_path/'c.yaml'), BackendSettings(), store=None, + overlay_path=str(tmp_path/'o.json')) + await mgr._post_lora(_inst("vllm", 8002), "load", {"lora_name": "x", "lora_path": "/p"}) + await mgr._post_lora(_inst("sglang", 8100), "load", {"lora_name": "x", "lora_path": "/p"}) + urls = [u for u, _ in client.posts] + assert urls[0] == "http://localhost:8002/v1/load_lora_adapter" # vLLM: /v1 prefix + assert urls[1] == "http://localhost:8100/load_lora_adapter" # SGLang: no /v1 + + +def _write_min_cfg(tmp_path): + p = tmp_path / "c.yaml" + p.write_text("server:\n port: 8887\nLLM_engines: {}\n", encoding="utf-8") + return str(p) diff --git a/docs/multi-backend-engine-design_zh-CN.md b/docs/multi-backend-engine-design_zh-CN.md index a9b340f..b13a8a5 100644 --- a/docs/multi-backend-engine-design_zh-CN.md +++ b/docs/multi-backend-engine-design_zh-CN.md @@ -251,12 +251,13 @@ engine-sglang.Dockerfile FROM lmsysorg/sglang:latest + backend code → - **probe_url**:`/health`(SGLang 預設 `/health` 會做 1-token 生成檢查,Starting 時回 503 → 正好當 readiness)。 例外:若部署設了 `SGLANG_ENABLE_HEALTH_ENDPOINT_GENERATION=false`,readiness 改用 `/health_generate`。 - **GPU**:沿用 `CUDA_VISIBLE_DEVICES`(同 vLLM 路徑);同機多實例切片再配 `--base-gpu-id`/`--gpu-id-step`。 -- **capabilities** = `{runtime_lora, lora_modules, metrics_sglang}`: - - `runtime_lora` ✅:`POST /load_lora_adapter` / `/unload_lora_adapter`(啟動需 `--enable-lora`,建議帶 - `--max-lora-rank` / `--lora-target-modules`) +- **capabilities** = `{runtime_lora, lora_modules}`(已實作);`metrics_sglang` 待 parser 完成才宣告: + - `runtime_lora` ✅(已 live 驗證):`POST /load_lora_adapter` / `/unload_lora_adapter`(啟動需 + `--enable-lora`,建議帶 `--max-lora-rank` / `--lora-target-modules`);`_post_lora` 依引擎選端點路徑 + - `lora_modules` ✅:靜態 `--lora-paths NAME=PATH`(非 vLLM 的 `--lora-modules` JSON 形式) - `sleep` ❌:SGLang 無 vLLM 式 `/sleep`/`/wake_up`(只有 `--sleep-on-idle`,降 CPU 非釋放 VRAM)→ autoscaler 暖待命層對 SGLang group 自動退化成 `stop` - - `kv_transfer` ❌ + - `kv_transfer` ❌;`metrics_sglang` 待後續(步驟 6) - **容器需求**:`--ipc=host` 或大 `--shm-size`(SGLang 對 shared memory 敏感);掛 HF cache。 - **router 零修改**:chat/completions、completions、streaming、final usage chunk(`choices=[]` + usage)、 `/v1/models`、`/tokenize`、`/detokenize` 全有 → 計費(靠 final usage chunk)可接。 @@ -295,12 +296,16 @@ engine-sglang.Dockerfile FROM lmsysorg/sglang:latest + backend code → (`FROM lmsysorg/sglang` + 同一份 backend code;砍掉 vllm,sglang/torch 來自 base 不動)。 驗證:image 不含 vllm、backend FastAPI 在其中正常 import/boot。 4. ✅ **`SglangLauncher`**(§5.2)— 已完成、已 live 驗證:arg builder(typed 翻譯 + store_true bool + - `--served-model-name`)+ `/health` probe + `capabilities=frozenset()`(LoRA/metrics 列後續)。 - docker 實測(**store 用 Postgres**,證明 store 與引擎解耦):reconciler 經 SglangLauncher spawn - `python -m sglang.launch_server` → READY → router proxy 推理(非串流 + 串流 final usage chunk)OK。 - 單元測試 8 條(arg 翻譯/bool/keys 過濾/capabilities/build_spec),全測 394 綠。 + `--served-model-name`)+ `/health` probe。docker 實測(**store 用 Postgres**,證明 store 與引擎解耦): + reconciler 經 SglangLauncher spawn `python -m sglang.launch_server` → READY → router proxy 推理 + (非串流 + 串流 final usage chunk)OK。 5. ✅ **capability 回歸** — 已驗證:SGLang group 的 `/sleep` 被擋(409,無 sleep 能力)、API 顯示 - `engine=sglang`、autoscaler 對它退化(固定副本)。runtime LoRA 屬後續(步驟 6)。 + `engine=sglang`、autoscaler 對它退化(固定副本)。 +5b. ✅ **SGLang runtime LoRA** — 已完成、已 live 驗證:`capabilities={runtime_lora, lora_modules}`; + arg builder 在有 LoRA 設定時帶 `--enable-lora` + 靜態 `--lora-paths NAME=PATH`;`_post_lora` 依引擎 + 選端點(SGLang `/load_lora_adapter`,無 `/v1`)。docker 實測:經我們的 API 熱掛載真實 adapter + (qwen3-test-lora)→ sglang `/v1/models` 出現該 adapter → 對 adapter 推理 OK → 卸載後消失。 + 全測 398 綠。 6.(後續)**SGLang 監控 + autoscaling**(同一塊,都吃 `sglang:*`): - launcher 帶 `--enable-metrics`;把 SGLang 實例寫進 Prometheus file_sd scrape targets。 - 新增 `metrics_sglang` 解析器(`sglang:*` 或 `/v1/loads` JSON),餵 autoscaler;宣告 `metrics_sglang` capability。 From 232d0e175131aa3ebf185eb8061d1194b7d7434f Mon Sep 17 00:00:00 2001 From: max Date: Tue, 30 Jun 2026 19:55:31 +0800 Subject: [PATCH 05/29] feat(engine): SGLang metrics + autoscaling (engine-aware parser) Make the autoscaler scale SGLang groups, and add a SGLang Grafana dashboard. - router metrics client is now engine-aware (METRIC_NAMES_BY_ENGINE): it parses sglang:num_queue_reqs / num_running_reqs / token_usage into the SAME normalized {waiting, running, kv} shape, so load-monitor / autoscaler / load-aware routing stay engine-agnostic and need no change. The poller picks the parser by each group's engine; fetch_many accepts (url, engine) tuples (bare url -> vllm). - SglangLauncher always emits --enable-metrics (vLLM exposes /metrics by default; SGLang needs the flag) and declares CAP_METRICS_SGLANG. - Prometheus file_sd targets gain an `engine` label so dashboards can filter vllm:* vs sglang:* panels. - New SGLang Grafana dashboard (deploy/grafana/dashboards/sglang/overview.json): running/queue/token-usage/cache-hit/throughput/TTFT from sglang:*. Tests: backend 399, router 122 green. Live: under 24 concurrent requests the router /metrics shows running=24 and token_usage 0.09->0.25, parsed from sglang:* into the normalized shape the autoscaler consumes. Co-Authored-By: Claude Opus 4.8 --- apps/backend/app/llmops/launchers.py | 14 ++- .../app/services/prometheus_targets.py | 3 + apps/backend/tests/unit/test_launchers.py | 25 ++-- .../tests/unit/test_prometheus_targets.py | 1 + .../src/llm_router/metrics_poller.py | 9 +- .../src/llm_router/vllm_metrics_client.py | 72 +++++++----- .../tests/unit/test_vllm_metrics_client.py | 33 ++++++ .../grafana/dashboards/sglang/overview.json | 108 ++++++++++++++++++ docs/multi-backend-engine-design_zh-CN.md | 31 +++-- 9 files changed, 243 insertions(+), 53 deletions(-) create mode 100644 deploy/grafana/dashboards/sglang/overview.json diff --git a/apps/backend/app/llmops/launchers.py b/apps/backend/app/llmops/launchers.py index fa72a11..dcbfc6f 100644 --- a/apps/backend/app/llmops/launchers.py +++ b/apps/backend/app/llmops/launchers.py @@ -135,6 +135,7 @@ def build_vllm_cli_args(model_cfg: dict) -> list[str]: CAP_LORA_MODULES = "lora_modules" # static --lora-modules at launch CAP_KV_TRANSFER = "kv_transfer" # cross-instance KV cache sharing CAP_METRICS_VLLM = "metrics_vllm" # exposes vLLM-format Prometheus metrics (waiting queue, …) +CAP_METRICS_SGLANG = "metrics_sglang" # exposes sglang:* Prometheus metrics (the router parses these) # Sentinel engine name for non-LLM launchers (embedding server): they aren't # selected by an engine choice, so they register under one fixed value. @@ -306,7 +307,7 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: # LoRA flags below; allow_runtime_lora is an enable_lora trigger) + router-only knobs. _SGLANG_SKIP_CLI_KEYS = frozenset( {"model_tag", "served_model_name", "id", "cuda_device", - "enable_lora", "lora_modules", _LORA_RUNTIME_KEY} + "enable_lora", "lora_modules", "enable_metrics", _LORA_RUNTIME_KEY} ) | _ROUTER_ONLY_KEYS @@ -330,6 +331,10 @@ def build_sglang_cli_args(model_cfg: dict) -> list[str]: args = ["--model-path", str(model_tag), "--served-model-name", str(served)] + # Always expose Prometheus /metrics (vLLM does by default; SGLang needs the + # flag). The router scrapes sglang:* from it for the autoscaler signal. + args.append("--enable-metrics") + # LoRA: SGLang needs --enable-lora at launch to accept either static adapters # (--lora-paths NAME=PATH …) or runtime ones (POST /load_lora_adapter). Turn it # on if any LoRA usage is configured (static modules, enable_lora, or our runtime @@ -371,10 +376,11 @@ class SglangLauncher: # Runtime + static LoRA are wired (--enable-lora / --lora-paths at launch; # POST /load_lora_adapter — no /v1 prefix, unlike vLLM — for hot load/unload). # No sleep (SGLang has no /sleep+/wake_up; `--sleep-on-idle` only lowers CPU, - # doesn't free VRAM) -> autoscaler degrades to ready<->stopped. sglang:* metrics - # exist but need a parser, so CAP_METRICS_* is left for a follow-up. + # doesn't free VRAM) -> autoscaler degrades to ready<->stopped. metrics_sglang: + # launches with --enable-metrics and the router parses sglang:* into the same + # normalized load shape, so the autoscaler scales SGLang groups too. # See docs/multi-backend-engine-design_zh-CN.md §5.2. - capabilities = frozenset({CAP_RUNTIME_LORA, CAP_LORA_MODULES}) + capabilities = frozenset({CAP_RUNTIME_LORA, CAP_LORA_MODULES, CAP_METRICS_SGLANG}) def keys(self, config) -> list[str]: out: list[str] = [] diff --git a/apps/backend/app/services/prometheus_targets.py b/apps/backend/app/services/prometheus_targets.py index 48d09b3..b0b35b9 100644 --- a/apps/backend/app/services/prometheus_targets.py +++ b/apps/backend/app/services/prometheus_targets.py @@ -48,6 +48,9 @@ def build_targets(instances: Iterable[ModelInstance]) -> list[dict]: "group": group, "instance_id": instance_id, "model_tag": inst.model_tag or "", + # Engine lets dashboards filter vLLM (vllm:*) vs SGLang (sglang:*) + # panels, since their metric names differ. + "engine": getattr(inst, "engine", "vllm"), }, } ) diff --git a/apps/backend/tests/unit/test_launchers.py b/apps/backend/tests/unit/test_launchers.py index d609592..0ded359 100644 --- a/apps/backend/tests/unit/test_launchers.py +++ b/apps/backend/tests/unit/test_launchers.py @@ -2,9 +2,10 @@ import pytest -from app.llmops.launchers import (CAP_LORA_MODULES, CAP_RUNTIME_LORA, CAP_SLEEP, - EMBEDDING_KEY, ENGINE_DEFAULT, EmbeddingLauncher, - SglangLauncher, VllmLauncher, _write_effective_config, +from app.llmops.launchers import (CAP_LORA_MODULES, CAP_METRICS_SGLANG, + CAP_RUNTIME_LORA, CAP_SLEEP, EMBEDDING_KEY, + ENGINE_DEFAULT, EmbeddingLauncher, SglangLauncher, + VllmLauncher, _write_effective_config, build_sglang_cli_args, build_vllm_cli_args) from app.llmops.state import ModelKind from schema import RootConfig @@ -342,11 +343,11 @@ def test_sglang_args_translate_typed_params(): def test_sglang_args_bool_is_store_true_no_dual(): args = build_sglang_cli_args({ - "model_tag": "org/m", "disable_radix_cache": True, "enable_metrics": False, + "model_tag": "org/m", "disable_radix_cache": True, "skip_server_warmup": False, }) assert "--disable-radix-cache" in args # True -> present - assert "--enable-metrics" not in args # False -> omitted - assert "--no-enable-metrics" not in args # never synthesise a --no- dual + assert "--skip-server-warmup" not in args # False -> omitted + assert "--no-skip-server-warmup" not in args # never synthesise a --no- dual def test_sglang_args_skip_router_only_keys(): @@ -372,13 +373,21 @@ def test_sglang_launcher_claims_only_sglang_engine(): def test_sglang_launcher_capabilities(): - # Runtime + static LoRA are wired; sleep is absent in SGLang (degrades), and - # metrics need a parser so CAP_METRICS_* is not advertised yet. + # Runtime + static LoRA and sglang:* metrics are wired; sleep is absent in + # SGLang (autoscaler degrades to ready<->stopped). caps = SglangLauncher().capabilities assert CAP_RUNTIME_LORA in caps and CAP_LORA_MODULES in caps + assert CAP_METRICS_SGLANG in caps assert CAP_SLEEP not in caps +def test_sglang_args_always_enable_metrics(): + # /metrics must be on (vLLM exposes it by default; SGLang needs the flag) so the + # router can scrape sglang:* for the autoscaler. + args = build_sglang_cli_args({"model_tag": "org/m"}) + assert "--enable-metrics" in args + + def test_sglang_args_lora_enable_and_paths(): # enable_lora -> --enable-lora; static lora_modules -> SGLang NAME=PATH form. args = build_sglang_cli_args({ diff --git a/apps/backend/tests/unit/test_prometheus_targets.py b/apps/backend/tests/unit/test_prometheus_targets.py index 1e8732b..e6a5428 100644 --- a/apps/backend/tests/unit/test_prometheus_targets.py +++ b/apps/backend/tests/unit/test_prometheus_targets.py @@ -33,6 +33,7 @@ def test_build_targets_only_includes_ready_llm(): assert entry["labels"]["group"] == "Qwen3-0.6B" assert entry["labels"]["instance_id"] == "qwen3" assert entry["labels"]["model_tag"] == "Qwen/Qwen3-0.6B" + assert entry["labels"]["engine"] == "vllm" # default; lets dashboards filter by engine def test_build_targets_excludes_embedding_server(): diff --git a/apps/router-server/src/llm_router/metrics_poller.py b/apps/router-server/src/llm_router/metrics_poller.py index c41837a..980004f 100644 --- a/apps/router-server/src/llm_router/metrics_poller.py +++ b/apps/router-server/src/llm_router/metrics_poller.py @@ -56,8 +56,11 @@ async def poll_metrics_forever(app, interval: float = 1.0): app.state.live_addrs = live_addrs index = [] # (model_key, instance_id, composite_key) - backends: Dict[str, str] = {} + backends: Dict[str, tuple] = {} # composite -> (url, engine) for model_key, model_cfg in llm_engines.items(): + # Pick the engine's metric parser (sglang exposes sglang:* names, + # not vllm:*). Default vllm so existing configs are unchanged. + engine = (model_cfg.get("model_config") or {}).get("engine", "vllm") for instance in model_cfg.get("instances", []): composite = f"{model_key}\x00{instance['id']}" host, port = live_addrs.get( @@ -66,7 +69,7 @@ async def poll_metrics_forever(app, interval: float = 1.0): ) url = f"http://{host}:{port}" index.append((model_key, instance["id"], composite)) - backends[composite] = url + backends[composite] = (url, engine) flat = await metrics_client.fetch_many(backends) if backends else {} @@ -82,7 +85,7 @@ async def poll_metrics_forever(app, interval: float = 1.0): for instance in model_cfg.get("instances", []): composite = f"{model_key}\x00{instance['id']}" if composite in backends: - sleep_backends[composite] = backends[composite] + sleep_backends[composite] = backends[composite][0] # url only sleeping = ( await _probe_sleeping(app.state.http_client, sleep_backends) if sleep_backends else {} diff --git a/apps/router-server/src/llm_router/vllm_metrics_client.py b/apps/router-server/src/llm_router/vllm_metrics_client.py index 1339e78..ff827cf 100644 --- a/apps/router-server/src/llm_router/vllm_metrics_client.py +++ b/apps/router-server/src/llm_router/vllm_metrics_client.py @@ -52,48 +52,63 @@ def to_dict(self): } -class VLLMMetricsClient: - METRIC_NAMES = { +# Each engine exposes the same concepts under different Prometheus metric names. +# Parsing normalises both into the identical VLLMInstanceMetrics shape, so the +# downstream load-monitor / autoscaler / routing stay engine-agnostic. SGLang's +# token_usage (0..1 of the token pool) is the closest analog to vLLM's KV usage. +METRIC_NAMES_BY_ENGINE = { + "vllm": { "running": "vllm:num_requests_running", "waiting": "vllm:num_requests_waiting", "kv_cache_usage_perc": "vllm:kv_cache_usage_perc", "prompt_tokens": "vllm:prompt_tokens", "generation_tokens": "vllm:generation_tokens", - } - + }, + "sglang": { + "running": "sglang:num_running_reqs", + "waiting": "sglang:num_queue_reqs", + "kv_cache_usage_perc": "sglang:token_usage", + "prompt_tokens": "sglang:prompt_tokens_total", + "generation_tokens": "sglang:generation_tokens_total", + }, +} + + +class VLLMMetricsClient: + # Default (vLLM) names; engine-specific lookups use METRIC_NAMES_BY_ENGINE. + METRIC_NAMES = METRIC_NAMES_BY_ENGINE["vllm"] + def __init__(self, http_client: httpx.AsyncClient, timeout: float = 2.0) -> None: self.http_client = http_client self.timeout = timeout - - async def fetch(self, base_url: str) -> Optional[VLLMInstanceMetrics]: + + async def fetch(self, base_url: str, engine: str = "vllm") -> Optional[VLLMInstanceMetrics]: metrics_url = base_url.rstrip("/") + "/metrics" resp = await self.http_client.get(metrics_url, timeout=self.timeout) resp.raise_for_status() - + parsed = self.parse_metrics(resp.text) - + names = METRIC_NAMES_BY_ENGINE.get(engine, self.METRIC_NAMES) + return VLLMInstanceMetrics( base_url=base_url, - running=parsed.get(self.METRIC_NAMES["running"], 0.0), - waiting=parsed.get(self.METRIC_NAMES["waiting"], 0.0), - kv_cache_usage_perc=parsed.get( - self.METRIC_NAMES["kv_cache_usage_perc"], 0.0 - ), - prompt_tokens=parsed.get(self.METRIC_NAMES["prompt_tokens"], 0.0), - generation_tokens=parsed.get( - self.METRIC_NAMES["generation_tokens"], 0.0 - ), + running=parsed.get(names["running"], 0.0), + waiting=parsed.get(names["waiting"], 0.0), + kv_cache_usage_perc=parsed.get(names["kv_cache_usage_perc"], 0.0), + prompt_tokens=parsed.get(names["prompt_tokens"], 0.0), + generation_tokens=parsed.get(names["generation_tokens"], 0.0), raw_metrics=resp.text, ) - + async def _safe_fetch( self, backend_name: str, base_url: str, + engine: str = "vllm", ) -> tuple[str, VLLMInstanceMetrics]: try: - metrics = await self.fetch(base_url) + metrics = await self.fetch(base_url, engine) return backend_name, metrics except Exception: return backend_name, VLLMInstanceMetrics( @@ -110,26 +125,27 @@ async def _safe_fetch( async def fetch_many( self, - backends: Dict[str, str], + backends: Dict[str, object], ) -> Dict[str, VLLMInstanceMetrics]: """ Fetch metrics for many backends. Args: - backends: mapping like + backends: mapping of name -> base_url, or name -> (base_url, engine) to + scrape multi-engine fleets. A bare string defaults engine to "vllm": { "qwen14b-a": "http://127.0.0.1:8001", - "qwen14b-b": "http://127.0.0.1:8002", + "sgl-b": ("http://127.0.0.1:8100", "sglang"), } Returns: Dict[str, VLLMInstanceMetrics] - """ - - tasks = [ - self._safe_fetch(backend_name, base_url) for backend_name, base_url in backends.items() - ] - + """ + tasks = [] + for backend_name, value in backends.items(): + base_url, engine = value if isinstance(value, tuple) else (value, "vllm") + tasks.append(self._safe_fetch(backend_name, base_url, engine)) + pairs = await asyncio.gather(*tasks) return dict(pairs) diff --git a/apps/router-server/tests/unit/test_vllm_metrics_client.py b/apps/router-server/tests/unit/test_vllm_metrics_client.py index 9990ada..1f48bc0 100644 --- a/apps/router-server/tests/unit/test_vllm_metrics_client.py +++ b/apps/router-server/tests/unit/test_vllm_metrics_client.py @@ -71,3 +71,36 @@ async def get(self, url, timeout=None): async def test_to_dict_keeps_real_kv_cache_value(): m = VLLMInstanceMetrics("x", running=1, waiting=0, kv_cache_usage_perc=0.42) assert m.to_dict()["kv_cache_usage_perc"] == 0.42 + + +async def test_fetch_parses_sglang_names_into_normalized_shape(): + # SGLang exposes sglang:* names; the client must normalise them to the same + # running/waiting/kv fields the autoscaler reads, so downstream is engine-agnostic. + text = ( + "sglang:num_running_reqs 3\n" + "sglang:num_queue_reqs 5\n" + "sglang:token_usage 0.4\n" + ) + client = VLLMMetricsClient(http_client=_FakeClient(text)) + m = await client.fetch("http://localhost:8100", engine="sglang") + assert m.running == 3.0 + assert m.waiting == 5.0 + assert m.kv_cache_usage_perc == 0.4 + + +async def test_fetch_vllm_engine_ignores_sglang_names(): + # A vLLM scrape of sglang:* (wrong engine) yields zeros, never crashes. + text = "sglang:num_queue_reqs 7\n" + client = VLLMMetricsClient(http_client=_FakeClient(text)) + m = await client.fetch("http://localhost:8002", engine="vllm") + assert m.waiting == 0.0 + + +async def test_fetch_many_accepts_url_and_url_engine_tuple(): + client = VLLMMetricsClient(http_client=_FakeClient("sglang:num_queue_reqs 2\n")) + out = await client.fetch_many({ + "v": "http://localhost:8002", # bare url -> vllm + "s": ("http://localhost:8100", "sglang"), # (url, engine) + }) + assert out["v"].waiting == 0.0 # vllm parser doesn't see sglang:* + assert out["s"].waiting == 2.0 # sglang parser does diff --git a/deploy/grafana/dashboards/sglang/overview.json b/deploy/grafana/dashboards/sglang/overview.json new file mode 100644 index 0000000..309a9d5 --- /dev/null +++ b/deploy/grafana/dashboards/sglang/overview.json @@ -0,0 +1,108 @@ +{ + "uid": "sglang-overview", + "title": "SGLang Overview", + "tags": ["sglang", "llm"], + "schemaVersion": 39, + "version": 1, + "editable": true, + "refresh": "10s", + "time": { "from": "now-1h", "to": "now" }, + "templating": { + "list": [ + { + "name": "ds", + "label": "Datasource", + "type": "datasource", + "query": "prometheus", + "current": {}, + "hide": 0, + "refresh": 1 + }, + { + "name": "model_name", + "label": "Model", + "type": "query", + "datasource": { "type": "prometheus", "uid": "${ds}" }, + "definition": "label_values(sglang:num_running_reqs, model_name)", + "query": { "qryType": 1, "query": "label_values(sglang:num_running_reqs, model_name)", "refId": "var" }, + "includeAll": true, + "multi": true, + "current": { "text": "All", "value": "$__all" }, + "refresh": 2 + } + ] + }, + "annotations": { "list": [] }, + "panels": [ + { + "id": 1, "type": "stat", "title": "Running requests", + "datasource": { "type": "prometheus", "uid": "${ds}" }, + "gridPos": { "h": 4, "w": 6, "x": 0, "y": 0 }, + "fieldConfig": { "defaults": { "unit": "short", "decimals": 0 }, "overrides": [] }, + "options": { "reduceOptions": { "calcs": ["lastNotNull"] }, "colorMode": "value", "graphMode": "area" }, + "targets": [ { "datasource": { "type": "prometheus", "uid": "${ds}" }, "editorMode": "code", "expr": "sum(sglang:num_running_reqs{model_name=~\"$model_name\"})", "range": true, "refId": "A", "legendFormat": "running" } ] + }, + { + "id": 2, "type": "stat", "title": "Queue depth (waiting)", + "datasource": { "type": "prometheus", "uid": "${ds}" }, + "gridPos": { "h": 4, "w": 6, "x": 6, "y": 0 }, + "fieldConfig": { "defaults": { "unit": "short", "decimals": 0, "thresholds": { "mode": "absolute", "steps": [ { "color": "green", "value": null }, { "color": "orange", "value": 4 }, { "color": "red", "value": 16 } ] } }, "overrides": [] }, + "options": { "reduceOptions": { "calcs": ["lastNotNull"] }, "colorMode": "value", "graphMode": "area" }, + "targets": [ { "datasource": { "type": "prometheus", "uid": "${ds}" }, "editorMode": "code", "expr": "sum(sglang:num_queue_reqs{model_name=~\"$model_name\"})", "range": true, "refId": "A", "legendFormat": "queue" } ] + }, + { + "id": 3, "type": "stat", "title": "Token pool usage", + "datasource": { "type": "prometheus", "uid": "${ds}" }, + "gridPos": { "h": 4, "w": 6, "x": 12, "y": 0 }, + "fieldConfig": { "defaults": { "unit": "percentunit", "decimals": 2, "min": 0, "max": 1 }, "overrides": [] }, + "options": { "reduceOptions": { "calcs": ["lastNotNull"] }, "colorMode": "value", "graphMode": "area" }, + "targets": [ { "datasource": { "type": "prometheus", "uid": "${ds}" }, "editorMode": "code", "expr": "avg(sglang:token_usage{model_name=~\"$model_name\"})", "range": true, "refId": "A", "legendFormat": "token_usage" } ] + }, + { + "id": 4, "type": "stat", "title": "Cache hit rate", + "datasource": { "type": "prometheus", "uid": "${ds}" }, + "gridPos": { "h": 4, "w": 6, "x": 18, "y": 0 }, + "fieldConfig": { "defaults": { "unit": "percentunit", "decimals": 2, "min": 0, "max": 1 }, "overrides": [] }, + "options": { "reduceOptions": { "calcs": ["lastNotNull"] }, "colorMode": "value", "graphMode": "area" }, + "targets": [ { "datasource": { "type": "prometheus", "uid": "${ds}" }, "editorMode": "code", "expr": "avg(sglang:cache_hit_rate{model_name=~\"$model_name\"})", "range": true, "refId": "A", "legendFormat": "cache_hit" } ] + }, + { + "id": 5, "type": "timeseries", "title": "Running vs Queue", + "datasource": { "type": "prometheus", "uid": "${ds}" }, + "gridPos": { "h": 8, "w": 12, "x": 0, "y": 4 }, + "fieldConfig": { "defaults": { "unit": "short", "custom": { "drawStyle": "line", "fillOpacity": 10 } }, "overrides": [] }, + "options": { "legend": { "displayMode": "list", "placement": "bottom" } }, + "targets": [ + { "datasource": { "type": "prometheus", "uid": "${ds}" }, "editorMode": "code", "expr": "sum(sglang:num_running_reqs{model_name=~\"$model_name\"})", "range": true, "refId": "A", "legendFormat": "running" }, + { "datasource": { "type": "prometheus", "uid": "${ds}" }, "editorMode": "code", "expr": "sum(sglang:num_queue_reqs{model_name=~\"$model_name\"})", "range": true, "refId": "B", "legendFormat": "queue (waiting)" } + ] + }, + { + "id": 6, "type": "timeseries", "title": "Generation throughput (tok/s)", + "datasource": { "type": "prometheus", "uid": "${ds}" }, + "gridPos": { "h": 8, "w": 12, "x": 12, "y": 4 }, + "fieldConfig": { "defaults": { "unit": "short", "custom": { "drawStyle": "line", "fillOpacity": 10 } }, "overrides": [] }, + "options": { "legend": { "displayMode": "list", "placement": "bottom" } }, + "targets": [ { "datasource": { "type": "prometheus", "uid": "${ds}" }, "editorMode": "code", "expr": "sum(sglang:gen_throughput{model_name=~\"$model_name\"})", "range": true, "refId": "A", "legendFormat": "gen_throughput" } ] + }, + { + "id": 7, "type": "timeseries", "title": "Time to first token (p95)", + "datasource": { "type": "prometheus", "uid": "${ds}" }, + "gridPos": { "h": 8, "w": 12, "x": 0, "y": 12 }, + "fieldConfig": { "defaults": { "unit": "s", "custom": { "drawStyle": "line", "fillOpacity": 10 } }, "overrides": [] }, + "options": { "legend": { "displayMode": "list", "placement": "bottom" } }, + "targets": [ { "datasource": { "type": "prometheus", "uid": "${ds}" }, "editorMode": "code", "expr": "histogram_quantile(0.95, sum(rate(sglang:time_to_first_token_seconds_bucket{model_name=~\"$model_name\"}[5m])) by (le))", "range": true, "refId": "A", "legendFormat": "ttft p95" } ] + }, + { + "id": 8, "type": "timeseries", "title": "Token throughput (rate)", + "datasource": { "type": "prometheus", "uid": "${ds}" }, + "gridPos": { "h": 8, "w": 12, "x": 12, "y": 12 }, + "fieldConfig": { "defaults": { "unit": "short", "custom": { "drawStyle": "line", "fillOpacity": 10 } }, "overrides": [] }, + "options": { "legend": { "displayMode": "list", "placement": "bottom" } }, + "targets": [ + { "datasource": { "type": "prometheus", "uid": "${ds}" }, "editorMode": "code", "expr": "sum(rate(sglang:prompt_tokens_total{model_name=~\"$model_name\"}[1m]))", "range": true, "refId": "A", "legendFormat": "prompt tok/s" }, + { "datasource": { "type": "prometheus", "uid": "${ds}" }, "editorMode": "code", "expr": "sum(rate(sglang:generation_tokens_total{model_name=~\"$model_name\"}[1m]))", "range": true, "refId": "B", "legendFormat": "gen tok/s" } + ] + } + ] +} diff --git a/docs/multi-backend-engine-design_zh-CN.md b/docs/multi-backend-engine-design_zh-CN.md index b13a8a5..6165a7e 100644 --- a/docs/multi-backend-engine-design_zh-CN.md +++ b/docs/multi-backend-engine-design_zh-CN.md @@ -257,7 +257,8 @@ engine-sglang.Dockerfile FROM lmsysorg/sglang:latest + backend code → - `lora_modules` ✅:靜態 `--lora-paths NAME=PATH`(非 vLLM 的 `--lora-modules` JSON 形式) - `sleep` ❌:SGLang 無 vLLM 式 `/sleep`/`/wake_up`(只有 `--sleep-on-idle`,降 CPU 非釋放 VRAM)→ autoscaler 暖待命層對 SGLang group 自動退化成 `stop` - - `kv_transfer` ❌;`metrics_sglang` 待後續(步驟 6) + - `metrics_sglang` ✅(已 live 驗證):`--enable-metrics` + router 依引擎解析 `sglang:*` → autoscaler 可擴縮 + - `kv_transfer` ❌ - **容器需求**:`--ipc=host` 或大 `--shm-size`(SGLang 對 shared memory 敏感);掛 HF cache。 - **router 零修改**:chat/completions、completions、streaming、final usage chunk(`choices=[]` + usage)、 `/v1/models`、`/tokenize`、`/detokenize` 全有 → 計費(靠 final usage chunk)可接。 @@ -272,12 +273,15 @@ engine-sglang.Dockerfile FROM lmsysorg/sglang:latest + backend code → (queue=`sglang:num_queue_reqs`、running=`sglang:num_running_reqs`、throughput=`sglang:gen_throughput`、 TTFT=`sglang:time_to_first_token_seconds`、cache=`sglang:cache_hit_rate`),所以解法是**另做一份並列的 SGLang dashboard**(或加 dashboard variable 切引擎)。 -- 這跟 autoscaler 訊號是**同一塊**(都吃 `sglang:*`),所以綁在一起當後續階段做。 +- 這跟 autoscaler 訊號是**同一塊**(都吃 `sglang:*`),所以一起做(見步驟 6,已完成)。 -> **autoscaling + 監控 首版不含**(決定):SGLang 指標自成一格(`sglang:num_queue_reqs` / -> `num_running_reqs` …,或 `/v1/loads` JSON)。首版只做 launch + route + lifecycle + runtime LoRA; -> autoscaler 對 SGLang group 維持固定副本(退化),Grafana SGLang 圖暫時空著(不影響起模型/路由/推理)。 -> `metrics_sglang` 解析器 + scrape target(`--enable-metrics`)+ SGLang Grafana dashboard 一起留作後續階段 +> **已完成(步驟 6)**:autoscaler 訊號靠 router 的指標 client 依引擎解析 `sglang:*` → 正規化成同一 +> `{waiting,running,kv}`,所以 load-monitor / autoscaler **零改**就能擴縮 SGLang。並附一份 SGLang Grafana +> dashboard。下面這段保留為「當初的權衡記錄」: +> +> ~~autoscaling + 監控 首版不含~~(後來在步驟 6 補上):SGLang 指標自成一格(`sglang:num_queue_reqs` / +> `num_running_reqs` …,或 `/v1/loads` JSON)。`metrics_sglang` 解析器 + scrape target(`--enable-metrics`) +> + SGLang Grafana dashboard > (屆時才宣告 `metrics_sglang` capability 真正生效)。 ## 6. 落地步驟(每步可獨立 commit、跑全測確保零行為變更) @@ -306,10 +310,17 @@ engine-sglang.Dockerfile FROM lmsysorg/sglang:latest + backend code → 選端點(SGLang `/load_lora_adapter`,無 `/v1`)。docker 實測:經我們的 API 熱掛載真實 adapter (qwen3-test-lora)→ sglang `/v1/models` 出現該 adapter → 對 adapter 推理 OK → 卸載後消失。 全測 398 綠。 -6.(後續)**SGLang 監控 + autoscaling**(同一塊,都吃 `sglang:*`): - - launcher 帶 `--enable-metrics`;把 SGLang 實例寫進 Prometheus file_sd scrape targets。 - - 新增 `metrics_sglang` 解析器(`sglang:*` 或 `/v1/loads` JSON),餵 autoscaler;宣告 `metrics_sglang` capability。 - - 另做一份並列的 **SGLang Grafana dashboard**(panel query 換成 `sglang:*`)。 +6. ✅ **SGLang 監控 + autoscaling** — 已完成、已 live 驗證(都吃 `sglang:*`): + - launcher 一律帶 `--enable-metrics`(vLLM 預設就有,SGLang 要旗標)。 + - router 的指標 client 改成**依引擎解析**([vllm_metrics_client.py](../apps/router-server/src/llm_router/vllm_metrics_client.py) + `METRIC_NAMES_BY_ENGINE`):把 `sglang:num_queue_reqs`/`num_running_reqs`/`token_usage` 正規化成 + **同一個 `{waiting,running,kv}` 形狀**,所以 load-monitor / autoscaler / routing **零改**就能擴縮 SGLang。 + poller 依 group 的 engine 選 parser。宣告 `metrics_sglang` capability。 + - Prometheus file_sd target 加上 `engine` label(供 dashboard 篩選);SGLang 實例本來就會被寫入(engine 無關)。 + - 並列的 **SGLang Grafana dashboard**([deploy/grafana/dashboards/sglang/overview.json](../deploy/grafana/dashboards/sglang/overview.json), + 8 panel,query 用 `sglang:*`)。 + - **Live 驗證**:24 並發請求時 router `/metrics` 顯示 `running=24`、`token_usage` 0.09→0.25(由 `sglang:*` + 正規化而來)→ autoscaler 訊號成立。 7.(可選)**llama.cpp** / (可選/暫緩)**TensorRT-LLM**。 ## 7. 風險與不做什麼 From c14b05b08509d4de22e457551535062c41402074 Mon Sep 17 00:00:00 2001 From: max Date: Tue, 30 Jun 2026 20:05:14 +0800 Subject: [PATCH 06/29] =?UTF-8?q?feat(ha):=20per-node=20actuation=20core?= =?UTF-8?q?=20=E2=80=94=20converge=5Fdesired=20+=20un-leader-gate=20reconc?= =?UTF-8?q?ile?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit HA Phase 7 sub-phases A+B: move actuation from "the leader spawns everything" to "each node converges the instances assigned to it" — the groundwork for a real multi-node (and mixed-engine) fleet. Collapsed single host is byte-for-byte unchanged: the one node is the leader and owns everything, so it converges the whole fleet exactly as before. - reconciler.converge_desired(): generalises boot-time replay_desired into a per-pass convergence over *owned* (non-foreign) instances — desired=running & STOPPED -> start desired=stopped & live -> stop desired=asleep & READY -> sleep desired=running & SLEEPING -> wake FAILED+desired-running recovery stays with _process_restarts (budget/backoff), so this never fights the crash-loop guard. Wired into reconcile_once, reusing the foreign set already computed there. - main.py loop split: reconcile/actuation + gpu-poll now run on EVERY replica (lifespan), since actuation must happen on the GPU-holding host and each node reports its own GPU. Scheduler / autoscaler / load-monitor / pruning stay leader-only singletons. Tests: 8 new converge_desired tests (each transition, foreign exclusion, FAILED left alone, unknown key); backend 407 green. Live: collapsed boot still actuates a model to READY via the per-node reconcile loop. Remaining Phase 7: C (API/autoscaler write intent), D (scheduler reject feedback), E (real multi-GPU live) — see docs/ha-per-node-actuation-design_zh-CN.md. Co-Authored-By: Claude Opus 4.8 --- apps/backend/app/llmops/reconciler.py | 56 +++++++++++ apps/backend/app/main.py | 27 +++++- .../tests/unit/test_converge_desired.py | 95 +++++++++++++++++++ docs/ha-per-node-actuation-design_zh-CN.md | 21 ++-- 4 files changed, 188 insertions(+), 11 deletions(-) create mode 100644 apps/backend/tests/unit/test_converge_desired.py diff --git a/apps/backend/app/llmops/reconciler.py b/apps/backend/app/llmops/reconciler.py index 5dc01dc..73049da 100644 --- a/apps/backend/app/llmops/reconciler.py +++ b/apps/backend/app/llmops/reconciler.py @@ -262,6 +262,58 @@ async def _backfill_node_state( logger.debug("prune_instance_observed failed", exc_info=True) +async def converge_desired( + registry: ModelRegistry, settings: BackendSettings, store, manager, foreign: set[str], +) -> None: + """Drive every *owned* instance toward its persisted desired state — the per-node + actuation step (HA Phase 7). Generalises boot-time `replay_desired` into a + continuous convergence so a node actuates whatever is assigned to it, no matter + which replica received the API/autoscaler write: + + desired=running & STOPPED -> start (a STOPPED-but-wanted instance, e.g. + just assigned to this node) + desired=stopped & live -> stop (ready/starting/sleeping) + desired=asleep & READY -> sleep + desired=running & SLEEPING -> wake + + FAILED + desired=running recovery is intentionally left to `_process_restarts` + (restart budget + backoff), so this never fights the crash-loop guard. `foreign` + keys (owned by another live node) are skipped. Collapsed single host: the one + node owns everything, so this converges the whole fleet — same end state as the + sync API path, just driven by the loop. Best-effort per key.""" + if store is None or manager is None or not hasattr(store, "list_instance_desired"): + return + try: + desired = await store.list_instance_desired() + except Exception: + logger.debug("converge_desired: failed to read desired", exc_info=True) + return + + # Snapshot (key, state) under the lock; act outside it (start/stop take the lock). + async with registry.lock: + states = { + inst.key: inst.state for inst in registry.values() + } + + for key, want in desired.items(): + if key in foreign or key not in states: + continue + state = states[key] + try: + if want == Desired.RUNNING.value and state == ModelState.STOPPED: + await manager.start(key) + elif want == Desired.STOPPED.value and state in ( + ModelState.READY, ModelState.STARTING, ModelState.SLEEPING + ): + await manager.stop(key) + elif want == Desired.ASLEEP.value and state == ModelState.READY: + await manager.sleep(key) + elif want == Desired.RUNNING.value and state == ModelState.SLEEPING: + await manager.wake(key) + except Exception: + logger.debug("converge_desired: action for %s (want=%s) failed", key, want, exc_info=True) + + async def reconcile_once( registry: ModelRegistry, http_client, settings: BackendSettings, store=None, manager=None, notifier=None, @@ -300,6 +352,10 @@ async def reconcile_once( await manager.write_prometheus_targets() if manager is not None and settings.auto_restart: await _process_restarts(registry, settings, store, manager, foreign) + # Per-node actuation (HA Phase 7): converge owned instances to their persisted + # desired state. Reuses the `foreign` set computed above. Collapsed = no-op + # beyond what the sync API already did (idempotent). + await converge_desired(registry, settings, store, manager, foreign) async def adopt_running( diff --git a/apps/backend/app/main.py b/apps/backend/app/main.py index 6506af7..3adaaf9 100644 --- a/apps/backend/app/main.py +++ b/apps/backend/app/main.py @@ -186,12 +186,16 @@ async def lifespan(app: FastAPI): async def _on_acquire() -> None: # Restore desired state on the leader only (a follower must not also start - # models). Skips anything adopt already found alive. + # models). Skips anything adopt already found alive. The per-node reconcile + # loop's converge_desired also restores it, but this gives the leader an + # immediate boot replay rather than waiting a poll interval. if settings.replay_desired: await manager.replay_desired() + # SINGLETON loops — global decisions, one place only: scheduler (placement), + # autoscaler (desired counts), load-monitor (global router view), pruning. + # The per-node loops (reconcile/actuation + gpu-poll) run on every replica; + # see below. leader_loops[:] = [ - asyncio.create_task(reconcile_loop(registry, http_client, settings, store, manager, notifier)), - asyncio.create_task(_gpu_poll_loop(app, settings.gpu_poll_interval)), asyncio.create_task( load_monitor_loop(app, registry, http_client, router_url, settings.load_poll_interval) ), @@ -202,7 +206,7 @@ async def _on_acquire() -> None: asyncio.create_task(_audit_prune_loop(store, settings.audit_max_rows)), asyncio.create_task(_config_versions_prune_loop(store, settings.config_versions_max)), ] - logger.info("Control loops started (leader)") + logger.info("Singleton control loops started (leader)") async def _on_release() -> None: for t in leader_loops: @@ -226,9 +230,24 @@ async def _on_release() -> None: app.state.node_agent = node_agent node_agent_task = asyncio.create_task(node_agent.run()) + # PER-NODE loops (HA Phase 7): reconcile/actuation + GPU polling run on EVERY + # replica, not just the leader — actuation must happen on the host that holds + # the GPU, and each node reports its own GPU. They converge only the instances + # assigned to this node (foreign_assignments gates ownership), so collapsed + # single-host is unchanged (one node owns + converges everything). + node_loops = [ + asyncio.create_task( + reconcile_loop(registry, http_client, settings, store, manager, notifier) + ), + asyncio.create_task(_gpu_poll_loop(app, settings.gpu_poll_interval)), + ] + try: yield finally: + for t in node_loops: + t.cancel() + await asyncio.gather(*node_loops, return_exceptions=True) node_agent_task.cancel() try: await node_agent_task diff --git a/apps/backend/tests/unit/test_converge_desired.py b/apps/backend/tests/unit/test_converge_desired.py new file mode 100644 index 0000000..77073ea --- /dev/null +++ b/apps/backend/tests/unit/test_converge_desired.py @@ -0,0 +1,95 @@ +"""Per-node actuation (HA Phase 7): converge_desired drives owned instances toward +their persisted desired state, skipping foreign (other-node-owned) keys.""" +import pytest + +from app.core.settings import BackendSettings +from app.llmops.launchers import EmbeddingLauncher, VllmLauncher +from app.llmops.manager import build_registry +from app.llmops.reconciler import converge_desired +from app.llmops.state import Desired, ModelState +from tests.conftest import FAKE_CONFIG + +pytestmark = pytest.mark.unit + +A = "Qwen3-0.6B::qwen3" +B = "Qwen3-0.6B::qwen3-2" + + +def _registry(): + return build_registry(FAKE_CONFIG, "config.yaml", [VllmLauncher(), EmbeddingLauncher()]) + + +class _DesiredStore: + def __init__(self, desired: dict[str, str]): + self._desired = desired + + async def list_instance_desired(self) -> dict[str, str]: + return dict(self._desired) + + +class _RecordingManager: + """Records the actuation calls converge_desired makes.""" + def __init__(self): + self.calls: list[tuple[str, str]] = [] + + async def start(self, key, **kw): self.calls.append(("start", key)) + async def stop(self, key, **kw): self.calls.append(("stop", key)) + async def sleep(self, key, **kw): self.calls.append(("sleep", key)) + async def wake(self, key, **kw): self.calls.append(("wake", key)) + + +async def _run(states: dict[str, ModelState], desired: dict[str, str], foreign=frozenset()): + reg = _registry() + for key, st in states.items(): + reg.get(key).state = st + mgr = _RecordingManager() + await converge_desired(reg, BackendSettings(), _DesiredStore(desired), mgr, set(foreign)) + return mgr.calls + + +async def test_starts_stopped_but_wanted(): + calls = await _run({A: ModelState.STOPPED}, {A: Desired.RUNNING.value}) + assert calls == [("start", A)] + + +async def test_stops_live_but_unwanted(): + calls = await _run({A: ModelState.READY}, {A: Desired.STOPPED.value}) + assert calls == [("stop", A)] + + +async def test_sleeps_ready_when_asleep_wanted(): + calls = await _run({A: ModelState.READY}, {A: Desired.ASLEEP.value}) + assert calls == [("sleep", A)] + + +async def test_wakes_sleeping_when_running_wanted(): + calls = await _run({A: ModelState.SLEEPING}, {A: Desired.RUNNING.value}) + assert calls == [("wake", A)] + + +async def test_failed_is_left_to_restart_logic(): + # desired=running + FAILED must NOT be started here (that is _process_restarts' + # job, with budget/backoff) — only STOPPED-but-wanted is converged. + calls = await _run({A: ModelState.FAILED}, {A: Desired.RUNNING.value}) + assert calls == [] + + +async def test_already_converged_is_noop(): + calls = await _run({A: ModelState.READY}, {A: Desired.RUNNING.value}) + assert calls == [] + + +async def test_foreign_keys_skipped(): + # A owned by another live node -> not actuated here; B (ours) is. + calls = await _run( + {A: ModelState.STOPPED, B: ModelState.STOPPED}, + {A: Desired.RUNNING.value, B: Desired.RUNNING.value}, + foreign={A}, + ) + assert calls == [("start", B)] + + +async def test_unknown_key_skipped(): + # A desired entry with no registry instance is ignored (no crash). + calls = await _run({}, {"Ghost::x": Desired.RUNNING.value}) + assert calls == [] diff --git a/docs/ha-per-node-actuation-design_zh-CN.md b/docs/ha-per-node-actuation-design_zh-CN.md index 48d7ff9..8fdd774 100644 --- a/docs/ha-per-node-actuation-design_zh-CN.md +++ b/docs/ha-per-node-actuation-design_zh-CN.md @@ -100,13 +100,20 @@ ## 5. 分階段執行(collapsed-first、每步單機 0 行為改變) -| 子階段 | 產出 | 單機 collapsed? | 風險 | -|---|---|---|---| -| **A** 把 reconcile 的「desired→observed 收斂(start/stop/sleep/wake)」抽成一個函式 | 邏輯就位,仍 leader 跑、行為不變 | ✅ | 中 | -| **B** un-leader-gate reconcile(每 node 跑)、ownership-scoped | follower 開始收斂自己擁有的(單機沒有 follower → 不變) | ✅ | 中高(改執行模型) | -| **C** API/autoscaler 改寫 desired/assignment(非同步收斂)+ 保留軟預檢 | 寫入與執行解耦 | ✅(owning=self) | 高(API 時序/預檢語意) | -| **D** scheduler 加 node 拒絕回饋 + 重指派 | 放置自我修正 | — | 中 | -| **E** 真多 GPU 主機 live 驗:並行起模型、node failover 接管 | 真並行多節點 | — | 需實體多機 | +| 子階段 | 產出 | 單機 collapsed? | 風險 | 狀態 | +|---|---|---|---|---| +| **A** 把 reconcile 的「desired→observed 收斂(start/stop/sleep/wake)」抽成一個函式 | 邏輯就位,仍 leader 跑、行為不變 | ✅ | 中 | ✅ 已完成 | +| **B** un-leader-gate reconcile(每 node 跑)、ownership-scoped | follower 開始收斂自己擁有的(單機沒有 follower → 不變) | ✅ | 中高(改執行模型) | ✅ 已完成 | +| **C** API/autoscaler 改寫 desired/assignment(非同步收斂)+ 保留軟預檢 | 寫入與執行解耦 | ✅(owning=self) | 高(API 時序/預檢語意) | ⬜ | +| **D** scheduler 加 node 拒絕回饋 + 重指派 | 放置自我修正 | — | 中 | ⬜ | +| **E** 真多 GPU 主機 live 驗:並行起模型、node failover 接管 | 真並行多節點 | — | 需實體多機 | ⬜ | + +> **A+B 已完成**([reconciler.py](../apps/backend/app/llmops/reconciler.py) `converge_desired` + +> [main.py](../apps/backend/app/main.py) 迴圈拆分):reconcile/actuation + gpu-poll 移到 lifespan(每 node 跑); +> scheduler / autoscaler / load-monitor / prune 留 leader-only。`converge_desired` 對 owned(非 foreign)實例做 +> desired→observed 收斂(STOPPED→start、live→stop、ready→sleep、sleeping→wake;FAILED 留給 `_process_restarts`)。 +> 單機 collapsed 行為不變(唯一 node=leader=全擁有);全測 407 綠 + live collapsed 起模型→READY 驗過。 +> 多 node 收斂以 fake 多 node unit test(foreign 排除等)覆蓋。剩 C/D(寫意圖解耦 + 排程拒絕回饋)與 E(實體多機)。 > 每一步:**單機先過既有全套測試 0 退化**、SQLite + Postgres 雙驗;多 node 邏輯以 fake 多 node 寫 unit test。 > A→D 都能在單機完成且行為不變;**E 一定要有第二台 GPU 主機**才驗得起來。 From 3421c2ca196423acc03716e53de957ef6b500280 Mon Sep 17 00:00:00 2001 From: max Date: Tue, 30 Jun 2026 20:15:10 +0800 Subject: [PATCH 07/29] feat(engine): SGLang routable bind-host + single-host mixed-fleet validation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - SglangLauncher honours LLMOPS_VLLM_BIND_HOST (shared with vLLM): binds SGLang to 0.0.0.0 so a router in another container/host reaches it via the advertised LLMOPS_NODE_HOST (instances_live), while the local probe + recorded host stay localhost. Enables cross-container routing for split / mixed-engine deploys. - Tests: sglang bind-host override + default (backend 409 green). Live-validated the full mixed-engine HA fleet on a single host (2 backend containers — vLLM image + SGLang image — sharing one Postgres, one router, one GPU): each model is actuated on its matching node, backfilled to instances_live (vllm-node / sglang-node), and routed by the one router (2+2 via vLLM, 3+3 via SGLang). The vLLM model's first start timed out under concurrent load and the follower node's own reconcile loop auto-restarted it to READY — confirming Phase 7B (per-node, un-leader-gated actuation). Shows the per-node-actuation "E" correctness is verifiable single-host; true multi-GPU parallelism still needs real hardware. Co-Authored-By: Claude Opus 4.8 --- apps/backend/app/llmops/launchers.py | 10 +++++++++- apps/backend/tests/unit/test_launchers.py | 15 +++++++++++++++ docs/ha-per-node-actuation-design_zh-CN.md | 13 +++++++++++-- 3 files changed, 35 insertions(+), 3 deletions(-) diff --git a/apps/backend/app/llmops/launchers.py b/apps/backend/app/llmops/launchers.py index dcbfc6f..a98cc3e 100644 --- a/apps/backend/app/llmops/launchers.py +++ b/apps/backend/app/llmops/launchers.py @@ -412,7 +412,15 @@ def build_spec(self, config, config_path: str, key: str) -> LaunchSpec: env["CUDA_VISIBLE_DEVICES"] = str(cuda_device) merged.pop("id", None) - command = [sys.executable, "-m", "sglang.launch_server"] + build_sglang_cli_args(merged) + # HA split deploys: optionally bind SGLang to a routable interface (0.0.0.0) + # so a router in another container/host reaches it via the advertised + # LLMOPS_NODE_HOST (instances_live). Only the bind --host changes; the local + # probe + recorded host stay localhost (which a 0.0.0.0 bind also serves). + # Empty (default) = bind the configured host = today's localhost-only. Shares + # the env var with vLLM so one setting governs both engines. + bind_host = os.environ.get("LLMOPS_VLLM_BIND_HOST", "").strip() + cli_cfg = {**merged, "host": bind_host} if bind_host else merged + command = [sys.executable, "-m", "sglang.launch_server"] + build_sglang_cli_args(cli_cfg) log_path = os.path.join(LOG_DIR, f"{model_tag}__{instance_id}.log") return LaunchSpec( key=key, diff --git a/apps/backend/tests/unit/test_launchers.py b/apps/backend/tests/unit/test_launchers.py index 0ded359..fc59640 100644 --- a/apps/backend/tests/unit/test_launchers.py +++ b/apps/backend/tests/unit/test_launchers.py @@ -428,3 +428,18 @@ def test_sglang_build_spec(): assert spec.probe_url == "http://localhost:8100/health" assert spec.host == "localhost" and spec.port == 8100 assert "--host" in spec.command and spec.command[spec.command.index("--host") + 1] == "localhost" + + +def test_sglang_bind_host_env_overrides_only_the_bind_address(monkeypatch): + # Cross-container HA: LLMOPS_VLLM_BIND_HOST binds sglang to 0.0.0.0 (--host), but + # the probe + recorded host stay localhost; routers reach it via NODE_HOST. + monkeypatch.setenv("LLMOPS_VLLM_BIND_HOST", "0.0.0.0") + spec = SglangLauncher().build_spec(_sglang_config(), "config.yaml", "S::a") + assert spec.command[spec.command.index("--host") + 1] == "0.0.0.0" # binds all + assert spec.host == "localhost" # record unchanged + assert spec.probe_url == "http://localhost:8100/health" # local probe unchanged + + +def test_sglang_binds_configured_host_by_default(): + spec = SglangLauncher().build_spec(_sglang_config(), "config.yaml", "S::a") + assert spec.command[spec.command.index("--host") + 1] == "localhost" diff --git a/docs/ha-per-node-actuation-design_zh-CN.md b/docs/ha-per-node-actuation-design_zh-CN.md index 8fdd774..b0dc89d 100644 --- a/docs/ha-per-node-actuation-design_zh-CN.md +++ b/docs/ha-per-node-actuation-design_zh-CN.md @@ -112,8 +112,17 @@ > [main.py](../apps/backend/app/main.py) 迴圈拆分):reconcile/actuation + gpu-poll 移到 lifespan(每 node 跑); > scheduler / autoscaler / load-monitor / prune 留 leader-only。`converge_desired` 對 owned(非 foreign)實例做 > desired→observed 收斂(STOPPED→start、live→stop、ready→sleep、sleeping→wake;FAILED 留給 `_process_restarts`)。 -> 單機 collapsed 行為不變(唯一 node=leader=全擁有);全測 407 綠 + live collapsed 起模型→READY 驗過。 -> 多 node 收斂以 fake 多 node unit test(foreign 排除等)覆蓋。剩 C/D(寫意圖解耦 + 排程拒絕回饋)與 E(實體多機)。 +> 單機 collapsed 行為不變(唯一 node=leader=全擁有);全測 409 綠 + live collapsed 起模型→READY 驗過。 +> 多 node 收斂以 fake 多 node unit test(foreign 排除等)覆蓋。剩 C/D(寫意圖解耦 + 排程拒絕回饋)。 +> +> **E 大部分可在單機驗(已做)**:原以為要實體多機,實際上**單機跑 2 個 backend 容器(vLLM image + +> SGLang image)共享一顆 Postgres + 一個 router + 一張 GPU** 就能驗證 per-node ownership / 跨容器路由 / +> follower 自我修復。Live 驗證結果:兩個不同引擎的模型各自被「對的 node」起起來、各自 backfill 到 +> `instances_live`(`mixed-vllm-backend:8002`/vllm-node、`mixed-sglang-backend:8100`/sglang-node)、**同一個 +> router** 路由到兩者(`2+2=4` 走 vLLM、`3+3=6` 走 SGLang)。期間 vLLM 首次啟動因並發競爭逾時 FAILED, +> **由 follower(vllm-node)自己的 reconcile loop auto-restart 回 READY** —— 直接證明 B 解除 leader 綁定後 +> follower 真的會 actuate 自己那份。SGLang 跨容器路由靠新加的 `LLMOPS_VLLM_BIND_HOST` bind-host 支援。 +> 真「並行多 GPU 加速」仍需實體多卡,但**功能正確性已單機驗完**。 > 每一步:**單機先過既有全套測試 0 退化**、SQLite + Postgres 雙驗;多 node 邏輯以 fake 多 node 寫 unit test。 > A→D 都能在單機完成且行為不變;**E 一定要有第二台 GPU 主機**才驗得起來。 From c29b23168c91b1057547b54c426e1fe68d2da14a Mon Sep 17 00:00:00 2001 From: max Date: Tue, 30 Jun 2026 20:32:01 +0800 Subject: [PATCH 08/29] feat(ha): engine-aware scheduling + Phase 7C write-intent (auto-placed mixed fleet) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Make a mixed-engine fleet self-organize: the scheduler places each model on a node that can actually run its engine, and an API call to a node that can't run the engine writes intent instead of failing locally. Collapsed single host unchanged (a node with no advertised engines runs anything). - nodes table gains an `engines` column (JSON list; NULL = runs any); upsert_node / list_nodes carry it; idempotent migration for existing DBs. - LLMOPS_NODE_ENGINES setting (set per engine image) -> node_agent advertises which engines this node can run. - scheduler: node_supports() + engine-aware place() — assign each desired model to the emptiest live node that supports its engine; move one sitting on a non-matching node; leave unassigned if no node can run it. Scheduler takes the registry to resolve key->engine. - Phase 7C: manager.start/stop/sleep/wake defer (write desired, no local spawn) when this node can't run the model's engine — the engine-matching node converges it. Tests: node_agent engines advertisement, engine-aware placement (match / no-capable / move-off-wrong / no-churn), manager defer + _node_can_run; backend 420, store 39. Co-Authored-By: Claude Opus 4.8 --- apps/backend/app/core/settings.py | 10 ++- apps/backend/app/llmops/manager.py | 43 +++++++++++++ apps/backend/app/llmops/node_agent.py | 10 ++- apps/backend/app/llmops/scheduler.py | 64 ++++++++++++++----- apps/backend/app/main.py | 2 +- .../backend/tests/unit/test_manager_engine.py | 61 ++++++++++++++++++ apps/backend/tests/unit/test_node_agent.py | 23 ++++++- apps/backend/tests/unit/test_scheduler.py | 51 +++++++++++++++ packages/llmops-store/llmops_store.py | 20 +++--- 9 files changed, 256 insertions(+), 28 deletions(-) diff --git a/apps/backend/app/core/settings.py b/apps/backend/app/core/settings.py index de8d311..855caa0 100644 --- a/apps/backend/app/core/settings.py +++ b/apps/backend/app/core/settings.py @@ -8,7 +8,7 @@ import os import socket -from dataclasses import dataclass +from dataclasses import dataclass, field from typing import Optional @@ -132,6 +132,11 @@ class BackendSettings: # The live address is heartbeated each reconcile pass; live_ttl is its lease. node_host: str = "" live_ttl: float = 30.0 + # HA Phase 7: which engines this node can actually run (determined by its image — + # the vLLM image runs vllm, the SGLang image runs sglang). Advertised in the + # node registry so the scheduler places each model on an engine-matching node. + # Empty/unset = unspecified => runs any engine (collapsed single host: unchanged). + node_engines: list[str] = field(default_factory=list) # HA Phase 3b: node-agent heartbeat. node_id reuses instance_id (the same id # used for the leader lease and live-address node_id). The agent re-registers @@ -198,6 +203,9 @@ def from_env(cls) -> "BackendSettings": or f"{socket.gethostname()}:{os.getpid()}", leader_lease_ttl=_env_float("LLMOPS_LEADER_LEASE_TTL", 15.0), node_host=os.environ.get("LLMOPS_NODE_HOST", "").strip(), + node_engines=[ + e.strip() for e in os.environ.get("LLMOPS_NODE_ENGINES", "").split(",") if e.strip() + ], live_ttl=_env_float("LLMOPS_LIVE_TTL", 30.0), node_heartbeat_interval=_env_float("LLMOPS_NODE_HEARTBEAT_INTERVAL", 10.0), node_ttl=_env_float("LLMOPS_NODE_TTL", 30.0), diff --git a/apps/backend/app/llmops/manager.py b/apps/backend/app/llmops/manager.py index 9b50799..d77f0b0 100644 --- a/apps/backend/app/llmops/manager.py +++ b/apps/backend/app/llmops/manager.py @@ -120,6 +120,38 @@ def _launcher_for(self, inst: ModelInstance) -> Launcher: """The launcher that owns an instance, by its (kind, engine).""" return self._launchers[(inst.kind, inst.engine)] + def _node_can_run(self, engine: str) -> bool: + """Whether THIS node can run an engine (its image has it). Empty + node_engines = unspecified = runs any (collapsed single host / single-engine + deploys: always True, so the sync actuation path below is unchanged).""" + ne = self.settings.node_engines + return not ne or engine in ne + + async def _defer_to_owner(self, inst: ModelInstance, desired: Desired) -> ModelInstance: + """HA Phase 7C: this node can't run `inst`'s engine, so don't actuate locally — + just record the intent and let the scheduler place it on an engine-matching + node, whose reconcile loop converges it. Returns the instance (state unchanged + here; the dashboard tracks progress via observed state from the owning node).""" + async with self.registry.lock: + inst.desired = desired + inst.touch() + if self.store is not None: + try: + await self.store.set_instance_desired(inst.key, desired.value) + except Exception: + logger.warning("defer: failed to persist desired for %s", inst.key, exc_info=True) + # Clear any assignment so the engine-aware scheduler places it fresh on a + # node that can actually run it (it would reassign anyway, but this avoids + # a transient wrong-node attempt). + if desired != Desired.STOPPED and hasattr(self.store, "delete_assignment"): + try: + await self.store.delete_assignment(inst.key) + except Exception: + logger.debug("defer: clear assignment failed for %s", inst.key, exc_info=True) + logger.info("Deferred %s (engine=%s) to an engine-matching node (desired=%s)", + inst.key, inst.engine, desired.value) + return inst + def _llm_engine_capabilities(self, group: str) -> frozenset: """Capabilities of the engine an LLM group is configured for. Callers gate optional features (sleep, runtime LoRA, …) on these rather than the engine @@ -378,6 +410,10 @@ async def _gpu_exists_preflight(self, key: str, spec) -> None: async def start(self, key: str, force: bool = False, reset_restart: bool = True) -> ModelInstance: inst = self._require(key) + # HA Phase 7C: if this node can't run the engine, write intent and let an + # engine-matching node actuate it (collapsed/single-engine: always can-run). + if not self._node_can_run(inst.engine): + return await self._defer_to_owner(inst, Desired.RUNNING) launcher = self._launcher_for(inst) # Re-resolve the spec (config may have changed) outside the lock so the @@ -432,6 +468,9 @@ async def start(self, key: str, force: bool = False, reset_restart: bool = True) async def stop(self, key: str) -> ModelInstance: inst = self._require(key) + # HA Phase 7C: not our engine -> record desired=stopped; the owning node stops it. + if not self._node_can_run(inst.engine): + return await self._defer_to_owner(inst, Desired.STOPPED) async with self.registry.lock: prev = inst.state @@ -483,6 +522,8 @@ async def sleep(self, key: str, level: int = 1) -> ModelInstance: The HTTP call runs outside the registry lock; on failure the desired intent is reverted so the reconciler doesn't fight a half-applied state.""" inst = self._require(key) + if not self._node_can_run(inst.engine): # HA Phase 7C: owning node sleeps it + return await self._defer_to_owner(inst, Desired.ASLEEP) if not getattr(inst.spec, "sleep_enabled", False): raise ModelConflict( f"{key} was not launched with sleep mode " @@ -522,6 +563,8 @@ async def wake(self, key: str) -> ModelInstance: """Wake a SLEEPING instance back to READY (vLLM /wake_up reloads weights to GPU). Seconds, not a cold start.""" inst = self._require(key) + if not self._node_can_run(inst.engine): # HA Phase 7C: owning node wakes it + return await self._defer_to_owner(inst, Desired.RUNNING) async with self.registry.lock: if inst.state != ModelState.SLEEPING: raise ModelConflict( diff --git a/apps/backend/app/llmops/node_agent.py b/apps/backend/app/llmops/node_agent.py index 6c5a90d..cfc4ef1 100644 --- a/apps/backend/app/llmops/node_agent.py +++ b/apps/backend/app/llmops/node_agent.py @@ -51,12 +51,20 @@ def _capacity(self) -> Optional[str]: ] return json.dumps(slim) + def _engines(self) -> Optional[str]: + """Engines this node can run, as JSON, or None when unspecified (runs any). + Set per engine image via LLMOPS_NODE_ENGINES so the scheduler places each + model on an engine-matching node.""" + engines = self.settings.node_engines + return json.dumps(engines) if engines else None + async def heartbeat_once(self) -> None: """Register/refresh this node. No-op if the store can't track nodes.""" if self.store is None or not hasattr(self.store, "upsert_node"): return await self.store.upsert_node( - self.node_id, self.hostname, self._capacity(), ttl=self.settings.node_ttl + self.node_id, self.hostname, self._capacity(), + ttl=self.settings.node_ttl, engines=self._engines(), ) # Housekeeping so a vanished peer's row doesn't linger past its lease. if hasattr(self.store, "prune_nodes"): diff --git a/apps/backend/app/llmops/scheduler.py b/apps/backend/app/llmops/scheduler.py index 8ed8589..a8ba0e3 100644 --- a/apps/backend/app/llmops/scheduler.py +++ b/apps/backend/app/llmops/scheduler.py @@ -30,32 +30,66 @@ def node_free_vram(node: dict) -> int: return sum(max(0, g.get("memory_total", 0) - g.get("memory_used", 0)) for g in gpus) +def node_supports(node: dict, engine: str) -> bool: + """Whether a node can run a given engine. A node that doesn't advertise engines + (engines NULL/empty) is unspecified and accepts any engine — so collapsed single + host and pre-Phase-7 deploys behave exactly as before.""" + raw = node.get("engines") + if not raw: + return True + try: + return engine in json.loads(raw) + except (json.JSONDecodeError, TypeError): + return True + + def place( desired: set[str], nodes: list[dict], assignments: dict[str, str], + key_engines: Optional[dict[str, str]] = None, ) -> dict[str, str]: - """Decide assignments to (re)write. For every desired-running instance whose - current assignment is missing or points to a node not in `nodes` (dead), pick - the live node with the most free VRAM. Returns only the *changes* {key: node}; - instances already on a live node are left where they are (no churn). + """Decide assignments to (re)write. For every desired-running instance, pick the + emptiest live node that can run the instance's engine — unless it's already on a + live, engine-matching node (no churn). Returns only the *changes* {key: node}. - `nodes` is the set of currently-alive nodes (each a row with node_id+capacity). + `key_engines` maps key -> engine (default "vllm"). A key whose engine no live + node supports is left unassigned until a matching node appears — and one sitting + on a non-matching node (e.g. started on the wrong backend) is moved to a matching + one. `nodes` is the set of alive nodes (node_id + capacity + engines). """ - alive = {n["node_id"] for n in nodes} - if not alive: + if not nodes: return {} # nowhere to place; leave as-is until a node appears - # Greedy: prefer the emptiest node. Stable tiebreak on node_id for determinism. - ranked = sorted(nodes, key=lambda n: (-node_free_vram(n), n["node_id"])) - target = ranked[0]["node_id"] + key_engines = key_engines or {} + by_id = {n["node_id"]: n for n in nodes} changes: dict[str, str] = {} for key in desired: - cur = assignments.get(key) - if cur is None or cur not in alive: - changes[key] = target + engine = key_engines.get(key, "vllm") + candidates = [n for n in nodes if node_supports(n, engine)] + if not candidates: + continue # no live node can run this engine; leave unassigned + cur_node = by_id.get(assignments.get(key)) + # Keep the current placement only if its node is alive AND engine-matching. + if cur_node is not None and node_supports(cur_node, engine): + continue + # Greedy: emptiest matching node; stable tiebreak on node_id. + target = sorted(candidates, key=lambda n: (-node_free_vram(n), n["node_id"]))[0] + changes[key] = target["node_id"] return changes class Scheduler: - """Leader-only loop: keep desired instances placed on live nodes.""" + """Leader-only loop: keep desired instances placed on live, engine-matching nodes. + + `registry` (optional) resolves each instance's engine so placement can match it + to a node that can run it. Without it, engines default to "vllm" — fine for a + single-engine fleet.""" + + def __init__(self, registry=None) -> None: + self.registry = registry + + def _key_engines(self) -> dict[str, str]: + if self.registry is None: + return {} + return {inst.key: getattr(inst, "engine", "vllm") for inst in self.registry.values()} async def reschedule_once(self, store, settings) -> dict[str, str]: """One placement pass. Reads desired-running instances, live nodes and @@ -73,7 +107,7 @@ async def reschedule_once(self, store, settings) -> dict[str, str]: from app.llmops.state import Desired desired = {k for k, v in desired_map.items() if v == Desired.RUNNING.value} - changes = place(desired, nodes, assignments) + changes = place(desired, nodes, assignments, self._key_engines()) for key, node_id in changes.items(): try: await store.set_assignment(key, node_id) diff --git a/apps/backend/app/main.py b/apps/backend/app/main.py index 3adaaf9..37562d5 100644 --- a/apps/backend/app/main.py +++ b/apps/backend/app/main.py @@ -201,7 +201,7 @@ async def _on_acquire() -> None: ), asyncio.create_task(autoscaler_loop(app, manager, settings.autoscale_interval)), asyncio.create_task( - Scheduler().run(store, settings, settings.schedule_interval) + Scheduler(registry).run(store, settings, settings.schedule_interval) ), asyncio.create_task(_audit_prune_loop(store, settings.audit_max_rows)), asyncio.create_task(_config_versions_prune_loop(store, settings.config_versions_max)), diff --git a/apps/backend/tests/unit/test_manager_engine.py b/apps/backend/tests/unit/test_manager_engine.py index 414a563..98a333d 100644 --- a/apps/backend/tests/unit/test_manager_engine.py +++ b/apps/backend/tests/unit/test_manager_engine.py @@ -208,3 +208,64 @@ def _write_min_cfg(tmp_path): p = tmp_path / "c.yaml" p.write_text("server:\n port: 8887\nLLM_engines: {}\n", encoding="utf-8") return str(p) + + +# ---- HA Phase 7C: API write-intent when this node can't run the engine ------- + +from app.llmops.launchers import SglangLauncher # noqa: E402 +from app.llmops.state import Desired, ModelState # noqa: E402 + + +class _DesiredRecordingStore: + def __init__(self): + self.desired: dict[str, str] = {} + self.deleted_assignments: list[str] = [] + + async def set_instance_desired(self, key, val): self.desired[key] = val + async def delete_assignment(self, key): self.deleted_assignments.append(key) + + +SGLANG_ONLY_YAML = """ +server: + port: 8887 +LLM_engines: + S: + instances: [{ id: a, host: localhost, port: 8100 }] + model_config: { model_tag: org/s, engine: sglang } +""" + + +def _mgr_node_engines(tmp_path, node_engines): + cfg = tmp_path / "config.yaml" + cfg.write_text(SGLANG_ONLY_YAML, encoding="utf-8") + config = load_config(str(cfg)) + launchers = [VllmLauncher(), SglangLauncher(), EmbeddingLauncher()] + registry = build_registry(config, str(cfg), launchers) + store = _DesiredRecordingStore() + mgr = ModelManager(registry, launchers, None, config, str(cfg), + BackendSettings(node_engines=node_engines), store=store, + overlay_path=str(tmp_path / "o.json")) + return mgr, store + + +async def test_start_defers_when_node_cannot_run_engine(tmp_path): + # A vLLM-only node asked to start a SGLang model: write intent, do NOT spawn. + mgr, store = _mgr_node_engines(tmp_path, ["vllm"]) + inst = await mgr.start("S::a") + assert inst.state == ModelState.STOPPED # never spawned locally + assert inst.desired == Desired.RUNNING # intent recorded + assert store.desired["S::a"] == "running" + assert "S::a" in store.deleted_assignments # cleared so scheduler re-places + + +async def test_node_can_run_all_when_unspecified(tmp_path): + # No node_engines = runs any engine -> _node_can_run True (collapsed unchanged). + mgr, _ = _mgr_node_engines(tmp_path, []) + assert mgr._node_can_run("sglang") is True + assert mgr._node_can_run("vllm") is True + + +async def test_node_can_run_respects_advertised(tmp_path): + mgr, _ = _mgr_node_engines(tmp_path, ["vllm"]) + assert mgr._node_can_run("vllm") is True + assert mgr._node_can_run("sglang") is False diff --git a/apps/backend/tests/unit/test_node_agent.py b/apps/backend/tests/unit/test_node_agent.py index ac6ca07..79e9374 100644 --- a/apps/backend/tests/unit/test_node_agent.py +++ b/apps/backend/tests/unit/test_node_agent.py @@ -16,8 +16,9 @@ def __init__(self): self.nodes = {} self.pruned = 0 - async def upsert_node(self, node_id, hostname, capacity, ttl, ts=None): - self.nodes[node_id] = {"hostname": hostname, "capacity": capacity, "ttl": ttl} + async def upsert_node(self, node_id, hostname, capacity, ttl, ts=None, engines=None): + self.nodes[node_id] = {"hostname": hostname, "capacity": capacity, "ttl": ttl, + "engines": engines} async def prune_nodes(self, ts=None): self.pruned += 1 @@ -72,3 +73,21 @@ def boom(): agent = NodeAgent(store, _settings()) await agent.heartbeat_once() assert store.nodes["node-A"]["capacity"] is None + + +async def test_heartbeat_advertises_engines(monkeypatch): + import app.llmops.node_agent as na + monkeypatch.setattr(na, "get_gpu_info", lambda: []) + store = FakeNodeStore() + agent = NodeAgent(store, BackendSettings(instance_id="node-A", node_engines=["sglang"])) + await agent.heartbeat_once() + assert store.nodes["node-A"]["engines"] == '["sglang"]' + + +async def test_heartbeat_engines_null_when_unspecified(monkeypatch): + import app.llmops.node_agent as na + monkeypatch.setattr(na, "get_gpu_info", lambda: []) + store = FakeNodeStore() + agent = NodeAgent(store, BackendSettings(instance_id="node-A")) # no node_engines + await agent.heartbeat_once() + assert store.nodes["node-A"]["engines"] is None diff --git a/apps/backend/tests/unit/test_scheduler.py b/apps/backend/tests/unit/test_scheduler.py index edc0755..ab3fa14 100644 --- a/apps/backend/tests/unit/test_scheduler.py +++ b/apps/backend/tests/unit/test_scheduler.py @@ -76,3 +76,54 @@ async def set_assignment(self, key, node_id, ts=None): async def test_reschedule_noop_without_store(): assert await Scheduler().reschedule_once(None, BackendSettings()) == {} + + +# ---- engine-aware placement (HA Phase 7) ------------------------------------ + +from app.llmops.scheduler import node_supports # noqa: E402 + + +def _enode(node_id, engines, *gpus): + n = _node(node_id, *gpus) + n["engines"] = json.dumps(engines) if engines is not None else None + return n + + +def test_node_supports_unspecified_runs_any(): + assert node_supports({"engines": None}, "sglang") is True + assert node_supports({}, "vllm") is True + + +def test_node_supports_matches_advertised(): + n = _enode("n", ["sglang"]) + assert node_supports(n, "sglang") is True + assert node_supports(n, "vllm") is False + + +def test_place_matches_engine_to_capable_node(): + nodes = [_enode("vllm-node", ["vllm"], (8192, 0)), + _enode("sglang-node", ["sglang"], (8192, 0))] + changes = place({"G::v", "G::s"}, nodes, {}, + key_engines={"G::v": "vllm", "G::s": "sglang"}) + assert changes["G::v"] == "vllm-node" + assert changes["G::s"] == "sglang-node" + + +def test_place_leaves_unassigned_when_no_capable_node(): + nodes = [_enode("vllm-node", ["vllm"], (8192, 0))] + changes = place({"G::s"}, nodes, {}, key_engines={"G::s": "sglang"}) + assert "G::s" not in changes # no sglang node -> stays unassigned + + +def test_place_moves_model_off_wrong_engine_node(): + # Started on the vllm node by mistake; scheduler moves it to the sglang node. + nodes = [_enode("vllm-node", ["vllm"], (8192, 0)), + _enode("sglang-node", ["sglang"], (8192, 0))] + changes = place({"G::s"}, nodes, {"G::s": "vllm-node"}, key_engines={"G::s": "sglang"}) + assert changes["G::s"] == "sglang-node" + + +def test_place_keeps_well_placed_engine_match(): + nodes = [_enode("sglang-node", ["sglang"], (8192, 0))] + changes = place({"G::s"}, nodes, {"G::s": "sglang-node"}, key_engines={"G::s": "sglang"}) + assert changes == {} # already on a matching live node -> no churn diff --git a/packages/llmops-store/llmops_store.py b/packages/llmops-store/llmops_store.py index 9852a92..b4a16a5 100644 --- a/packages/llmops-store/llmops_store.py +++ b/packages/llmops-store/llmops_store.py @@ -209,6 +209,8 @@ def get_db_url() -> Optional[str]: node_id TEXT PRIMARY KEY, -- agent/replica id (== leader holder id) hostname TEXT, capacity TEXT, -- JSON: GPU inventory (index/name/mem) + engines TEXT, -- JSON list of engines this node can run + -- (NULL = unspecified => runs any engine) expires_at REAL NOT NULL -- epoch; lapsed => node considered down ); @@ -243,6 +245,7 @@ def get_db_url() -> Optional[str]: ("api_keys", "rpm_limit", "INTEGER"), ("api_keys", "token_quota", "INTEGER"), ("api_keys", "quota_period", "TEXT"), + ("nodes", "engines", "TEXT"), # HA Phase 7: engines a node can run (mixed-engine fleets) ] @@ -865,19 +868,20 @@ async def list_instances_live(self, ts: Optional[float] = None) -> list[dict]: async def upsert_node( self, node_id: str, hostname: Optional[str], capacity: Optional[str], - ttl: float, ts: Optional[float] = None, + ttl: float, ts: Optional[float] = None, engines: Optional[str] = None, ) -> None: - """Register/refresh a node-agent's heartbeat + capacity (JSON).""" + """Register/refresh a node-agent's heartbeat + capacity (JSON). `engines` is + a JSON list of the engines this node can run (NULL = any).""" import time now = ts if ts is not None else time.time() await self._db.execute( - "INSERT INTO nodes (node_id, hostname, capacity, expires_at) " - "VALUES (?, ?, ?, ?) " + "INSERT INTO nodes (node_id, hostname, capacity, engines, expires_at) " + "VALUES (?, ?, ?, ?, ?) " "ON CONFLICT(node_id) DO UPDATE SET " "hostname = excluded.hostname, capacity = excluded.capacity, " - "expires_at = excluded.expires_at", - (node_id, hostname, capacity, now + ttl), + "engines = excluded.engines, expires_at = excluded.expires_at", + (node_id, hostname, capacity, engines, now + ttl), ) await self._db.commit() @@ -887,12 +891,12 @@ async def list_nodes(self, ts: Optional[float] = None) -> list[dict]: now = ts if ts is not None else time.time() cur = await self._db.execute( - "SELECT node_id, hostname, capacity, expires_at FROM nodes WHERE expires_at > ?", + "SELECT node_id, hostname, capacity, engines, expires_at FROM nodes WHERE expires_at > ?", (now,), ) return [ {"node_id": r["node_id"], "hostname": r["hostname"], - "capacity": r["capacity"], "expires_at": r["expires_at"]} + "capacity": r["capacity"], "engines": r["engines"], "expires_at": r["expires_at"]} for r in await cur.fetchall() ] From 3277997bfa97e60d28a020f45031e500609a22b7 Mon Sep 17 00:00:00 2001 From: max Date: Tue, 30 Jun 2026 20:38:05 +0800 Subject: [PATCH 09/29] feat(frontend): engine selector + badge + capability-gated sleep - AddModelDialog: an "Inference engine" dropdown (vllm / sglang / llamacpp / trtllm) writes settings.engine; loaded back when editing/parsing; the Sleep Mode toggle is hidden for engines without sleep (only vLLM has it). - ModelGroupCard: an engine badge next to the group name for non-vLLM groups. - ModelView type gains `engine`; zh-TW + en strings for the new field. Type-check clean. The dropdown lets users pick SGLang without hand-typing an advanced param; it still needs a backend running that engine's image to start. Co-Authored-By: Claude Opus 4.8 --- .../src/components/AddModelDialog.vue | 36 ++++++++++++++++--- .../src/components/ModelGroupCard.vue | 11 +++++- apps/frontend_llmops/src/i18n/locales/en.ts | 2 ++ .../frontend_llmops/src/i18n/locales/zh-TW.ts | 2 ++ apps/frontend_llmops/src/types/api.ts | 1 + 5 files changed, 47 insertions(+), 5 deletions(-) diff --git a/apps/frontend_llmops/src/components/AddModelDialog.vue b/apps/frontend_llmops/src/components/AddModelDialog.vue index 161ecbb..03bd8f6 100644 --- a/apps/frontend_llmops/src/components/AddModelDialog.vue +++ b/apps/frontend_llmops/src/components/AddModelDialog.vue @@ -40,6 +40,12 @@ const host = ref('localhost') const port = ref(8000) const cudaDevice = ref(null) const modelTag = ref('') +// Inference engine for the whole group. Different engines map the same concepts to +// different launch flags; the launcher translates. vLLM is the default. +const engine = ref('vllm') +const ENGINE_OPTIONS = ['vllm', 'sglang', 'llamacpp', 'trtllm'] +// Capability gating: only vLLM supports sleep mode; hide the toggle otherwise. +const engineHasSleep = computed(() => engine.value === 'vllm') const params = ref<{ key: string; value: string }[]>([]) // Router-only load-balancing policy for the group. Lives in model_config but is // NOT a vLLM flag, so it's edited as its own field and kept out of the raw param @@ -164,6 +170,7 @@ function reset() { port.value = 8000 cudaDevice.value = null modelTag.value = '' + engine.value = 'vllm' params.value = [] routingStrategy.value = '' kvShared.value = false @@ -209,6 +216,7 @@ function prefillForEdit() { port.value = cfg.port cudaDevice.value = cfg.cuda_device ?? null modelTag.value = String(cfg.settings.model_tag ?? '') + engine.value = String(cfg.settings.engine ?? 'vllm') routingStrategy.value = String(cfg.settings.routing_strategy ?? '') kvShared.value = isKvShared(cfg.settings) sleepMode.value = !!cfg.settings.enable_sleep_mode @@ -216,6 +224,7 @@ function prefillForEdit() { Object.entries(cfg.settings).filter( ([k2]) => k2 !== 'model_tag' && + k2 !== 'engine' && k2 !== 'routing_strategy' && k2 !== 'kv_transfer_config' && k2 !== 'enable_sleep_mode', @@ -252,6 +261,7 @@ async function parse() { port.value = p.instance.port cudaDevice.value = p.instance.cuda_device modelTag.value = String(p.model_config.model_tag ?? '') + engine.value = String((p.model_config as Record).engine ?? 'vllm') routingStrategy.value = String( (p.model_config as Record).routing_strategy ?? '', ) @@ -261,6 +271,7 @@ async function parse() { Object.entries(p.model_config).filter( ([k]) => k !== 'model_tag' && + k !== 'engine' && k !== 'routing_strategy' && k !== 'kv_transfer_config' && k !== 'enable_sleep_mode', @@ -421,12 +432,15 @@ async function submit() { lora_modules?: LoraModule[] kv_transfer_config?: KvTransferConfig enable_sleep_mode?: boolean + engine?: string } = { model_tag: modelTag.value, + // Engine for the whole group; default vllm keeps existing configs unchanged. + engine: engine.value, } for (const { key: k, value } of params.value) { - // `lora_modules`, `routing_strategy` and `kv_transfer_config` have dedicated - // editors below — never let a raw param (e.g. a stray "" from a null, or a + // `lora_modules`, `routing_strategy`, `kv_transfer_config`, `engine` have + // dedicated editors — never let a raw param (e.g. a stray "" from a null, or a // leaked key) stomp them via the generic param list. const kk = k.trim() if ( @@ -434,7 +448,8 @@ async function submit() { kk !== 'lora_modules' && kk !== 'routing_strategy' && kk !== 'kv_transfer_config' && - kk !== 'enable_sleep_mode' + kk !== 'enable_sleep_mode' && + kk !== 'engine' ) settings[kk] = coerce(value) } @@ -446,7 +461,7 @@ async function submit() { if (kvShared.value) settings.kv_transfer_config = KV_SHARE_PRESET // Sleep-mode warm-standby tier: launch with --enable-sleep-mode + dev mode so // the instance can be slept/woken (see docs/autoscaling-design_zh-CN.md). - if (sleepMode.value) settings.enable_sleep_mode = true + if (sleepMode.value && engineHasSleep.value) settings.enable_sleep_mode = true // Mounted adapters: keep only filled rows; drop the empty base_model_name field. const cleanLoras = loras.value .filter((l) => l.name.trim() && l.path.trim()) @@ -580,6 +595,18 @@ async function submit() { {{ $t('addModel.modelTagLabel') }} * +