From 96d376ed6a6abe823332d2406609cc3fa66daf28 Mon Sep 17 00:00:00 2001 From: Erik LaBianca Date: Thu, 23 Jul 2026 16:15:10 -0400 Subject: [PATCH 1/8] =?UTF-8?q?feat(lucebox):=20hub=20CLI=20=E2=80=94=20in?= =?UTF-8?q?stall,=20serve,=20config,=20models?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit User-facing host wrapper + in-container Python CLI for launching and configuring the lucebox-hub image. Completes the docker-stack (#334) follow-up that intentionally shipped images without the CLI package. Host surface: - install.sh bootstrap + lucebox.sh wrapper (probe host, systemd unit, serve/pull/update/check/logs) - config.toml as system of record (env > file > defaults) In-container package (lucebox/): - check, pull, print-run, print-serve-argv - config {get,set,unset} - models {list,download} with VRAM-tier recommend_preset - VRAM-tier DFLASH_* heuristic (runtime_from_host); seeds config.toml on first models download --activate so 24 GB hosts get 98k/tq3_0 instead of the 16k class default - prefix_cache stays off by default (agent tool-prompt regression) Image/workspace: - COPY lucebox into CUDA and ROCm images; uv workspace member - entrypoint host_info missing-dir guard + pure-bash trim Tests/CI: lucebox pytest, wrapper sandbox scripts, lefthook. Deferred to follow-ups: autotune sweep/profiles, profile/smoke, agent-client harness adapters. --- .github/workflows/ci.yml | 29 + Dockerfile | 13 +- Dockerfile.rocm | 4 + install.sh | 138 +++ lefthook.yml | 59 + lucebox.sh | 1285 ++++++++++++++++++++++ lucebox/.gitignore | 3 + lucebox/README.md | 18 + lucebox/pyproject.toml | 54 + lucebox/src/lucebox/__init__.py | 16 + lucebox/src/lucebox/__main__.py | 6 + lucebox/src/lucebox/autotune.py | 80 ++ lucebox/src/lucebox/cli.py | 356 ++++++ lucebox/src/lucebox/config.py | 482 ++++++++ lucebox/src/lucebox/docker_run.py | 232 ++++ lucebox/src/lucebox/download.py | 515 +++++++++ lucebox/src/lucebox/host_check.py | 232 ++++ lucebox/src/lucebox/host_facts.py | 58 + lucebox/src/lucebox/py.typed | 1 + lucebox/src/lucebox/types.py | 140 +++ lucebox/tests/test_autotune.py | 48 + lucebox/tests/test_check.py | 118 ++ lucebox/tests/test_cli.py | 102 ++ lucebox/tests/test_config.py | 210 ++++ lucebox/tests/test_config_cli.py | 127 +++ lucebox/tests/test_docker_run.py | 254 +++++ lucebox/tests/test_download.py | 323 ++++++ lucebox/tests/test_models_cli.py | 142 +++ pyproject.toml | 10 +- scripts/check_lucebox_wrapper_sandbox.sh | 242 ++++ scripts/test_lucebox_sh.sh | 1130 +++++++++++++++++++ server/scripts/entrypoint.sh | 32 +- uv.lock | 32 + 33 files changed, 6474 insertions(+), 17 deletions(-) create mode 100755 install.sh create mode 100644 lefthook.yml create mode 100755 lucebox.sh create mode 100644 lucebox/.gitignore create mode 100644 lucebox/README.md create mode 100644 lucebox/pyproject.toml create mode 100644 lucebox/src/lucebox/__init__.py create mode 100644 lucebox/src/lucebox/__main__.py create mode 100644 lucebox/src/lucebox/autotune.py create mode 100644 lucebox/src/lucebox/cli.py create mode 100644 lucebox/src/lucebox/config.py create mode 100644 lucebox/src/lucebox/docker_run.py create mode 100644 lucebox/src/lucebox/download.py create mode 100644 lucebox/src/lucebox/host_check.py create mode 100644 lucebox/src/lucebox/host_facts.py create mode 100644 lucebox/src/lucebox/py.typed create mode 100644 lucebox/src/lucebox/types.py create mode 100644 lucebox/tests/test_autotune.py create mode 100644 lucebox/tests/test_check.py create mode 100644 lucebox/tests/test_cli.py create mode 100644 lucebox/tests/test_config.py create mode 100644 lucebox/tests/test_config_cli.py create mode 100644 lucebox/tests/test_docker_run.py create mode 100644 lucebox/tests/test_download.py create mode 100644 lucebox/tests/test_models_cli.py create mode 100755 scripts/check_lucebox_wrapper_sandbox.sh create mode 100755 scripts/test_lucebox_sh.sh diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 8c01e7685..c918ae922 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -34,6 +34,35 @@ jobs: - name: Lint Python surfaces touched by lucebox tooling run: uv run --frozen --extra dev ruff check . + - name: Install shellcheck (for bash test runner) + # ubuntu-latest typically ships shellcheck pre-installed, but pin + # the dependency explicitly so the bash test runner can always rely + # on `command -v shellcheck` succeeding. + run: | + if ! command -v shellcheck >/dev/null 2>&1; then + sudo apt-get update + sudo apt-get install -y shellcheck + fi + shellcheck --version | head -3 + + - name: Typecheck lucebox CLI + run: uv run --frozen --extra dev python -m mypy --package lucebox + + - name: Unit-test lucebox CLI + # The fast workspace sync above is enough: the suite mocks the + # docker / HTTP surfaces, so no torch wheel or GPU is needed. + # Keeps the lucebox Python honest on every push. + run: uv run --frozen --extra dev pytest lucebox -q + + - name: Smoke-test lucebox.sh wrapper + # Catches `set -u` regressions, syntax errors, and stale dispatch + # handlers in the host-side wrapper + the in-container entrypoint. + # Runs shellcheck --severity=error across every shipped .sh file, + # exercises every subcommand dispatch under `set -u`, and drives the + # entrypoint's draft-resolution block through every family-glob + # branch — all on the bare runner without docker/nvidia/systemd. + run: bash scripts/test_lucebox_sh.sh + build: name: Build (cmake + uv sync --extra megakernel) runs-on: ubuntu-latest diff --git a/Dockerfile b/Dockerfile index 9ad9177be..26df32c78 100644 --- a/Dockerfile +++ b/Dockerfile @@ -117,14 +117,15 @@ RUN cd /src/server/build \ # of these reuses the cached CUDA layers above and only re-runs the # runtime stage's uv sync (~70s) instead of the full ~25-minute build. # -# Host-side Python tooling (lucebox/, harness/) is intentionally not copied -# here: this image is the server. Such tooling can layer on top later via a -# follow-up COPY directive or a runtime bind-mount during dev. +# The lucebox CLI is a workspace member (root pyproject + uv.lock), so its +# source must be present for `uv sync --frozen` in the runtime stage. It runs +# inside the container — the host wrapper `docker exec`s into it. COPY pyproject.toml uv.lock README.md /src/ COPY server/pyproject.toml server/README.md /src/server/ COPY server/scripts /src/server/scripts COPY optimizations/pflash /src/optimizations/pflash COPY optimizations/megakernel /src/optimizations/megakernel +COPY lucebox /src/lucebox # ─── Stage 2: runtime ─────────────────────────────────────────────────────── # Runtime image: ships nvidia driver libs but no nvcc / dev headers. Matches @@ -179,9 +180,9 @@ COPY --from=builder /src/optimizations/megakernel/pyproject.toml \ /src/optimizations/megakernel/README.md \ /opt/lucebox-hub/optimizations/megakernel/ -# Host-side Python tooling (lucebox/, harness/) is intentionally absent -# here: this image is the server base layer. Such tooling can layer on top -# later via a follow-up COPY directive or a runtime bind-mount during dev. +# The lucebox CLI package (a workspace member) — installed by the uv sync +# below and invoked in-container by the host wrapper via `docker exec`. +COPY --from=builder /src/lucebox /opt/lucebox-hub/lucebox # server: ship the entrypoint/benchmark scripts, the pyproject + README that uv # resolves against, and the pruned build tree (binaries + .so files from the diff --git a/Dockerfile.rocm b/Dockerfile.rocm index b2581857f..ca9fb1b65 100644 --- a/Dockerfile.rocm +++ b/Dockerfile.rocm @@ -122,6 +122,7 @@ COPY server/pyproject.toml server/README.md /src/server/ COPY server/scripts /src/server/scripts COPY optimizations/pflash /src/optimizations/pflash COPY optimizations/megakernel /src/optimizations/megakernel +COPY lucebox /src/lucebox # ─── Stage 2: runtime ─────────────────────────────────────────────────────── # Runtime reuses the ROCm base so the HIP runtime libs (libamdhip64, @@ -171,6 +172,9 @@ COPY --from=builder /src/optimizations/megakernel/pyproject.toml \ /src/optimizations/megakernel/README.md \ /opt/lucebox-hub/optimizations/megakernel/ +# The lucebox CLI is a workspace member and must be present before uv sync. +COPY --from=builder /src/lucebox /opt/lucebox-hub/lucebox + COPY --from=builder /src/server/scripts /opt/lucebox-hub/server/scripts COPY --from=builder /src/server/pyproject.toml /src/server/README.md \ /opt/lucebox-hub/server/ diff --git a/install.sh b/install.sh new file mode 100755 index 000000000..cff54a02e --- /dev/null +++ b/install.sh @@ -0,0 +1,138 @@ +#!/usr/bin/env bash +# install.sh — Bootstrap installer for the lucebox host wrapper. +# +# Canonical install (Luce-Org main, stable channel): +# +# curl -fsSL https://raw.githubusercontent.com/Luce-Org/lucebox-hub/main/install.sh | bash +# +# Install from a different fork / branch (dev channel). Note the env var +# is on the `bash` side of the pipe — `VAR=val curl … | bash` would attach +# it to the `curl` process, leaving `bash` with the canonical default: +# +# curl -fsSL https://raw.githubusercontent.com/easel/lucebox-hub/feat/lucebox-docker/install.sh | \ +# LUCEBOX_INSTALL_URL=https://raw.githubusercontent.com/easel/lucebox-hub/feat/lucebox-docker/lucebox.sh bash +# +# The installer bakes the source URL into the installed `lucebox.sh` as +# `LUCEBOX_INSTALLED_FROM=...`, so `lucebox update` later re-pulls from the +# same channel without the user having to remember which fork they used. +# +# Override the install destination via $LUCEBOX_INSTALL_DEST (default +# $HOME/.local/bin/lucebox). This is what `lucebox update` uses to replace +# the file in place. + +set -euo pipefail + +LUCEBOX_INSTALL_URL="${LUCEBOX_INSTALL_URL:-https://raw.githubusercontent.com/Luce-Org/lucebox-hub/main/lucebox.sh}" +DEST="${LUCEBOX_INSTALL_DEST:-$HOME/.local/bin/lucebox}" + +# ── helpers ─────────────────────────────────────────────────────────────── +C_OK=$'\033[1;32m' ; C_ERR=$'\033[1;31m' ; C_DIM=$'\033[2m' ; C_RST=$'\033[0m' +if [ ! -t 1 ] || [ "${NO_COLOR:-}" ]; then + C_OK="" ; C_ERR="" ; C_DIM="" ; C_RST="" +fi +info() { printf '%s[install]%s %s\n' "$C_DIM" "$C_RST" "$*"; } +ok() { printf '%s[install] ✓%s %s\n' "$C_OK" "$C_RST" "$*"; } +die() { printf '%s[install] ✗%s %s\n' "$C_ERR" "$C_RST" "$*" >&2; exit 1; } + +command -v curl >/dev/null 2>&1 || die "curl is required (apt-get install curl)" + +# ── fetch ───────────────────────────────────────────────────────────────── +tmp=$(mktemp -t lucebox.XXXXXX) || die "couldn't create temp file" +# shellcheck disable=SC2064 # we want $tmp expanded now, not at trap time +trap "rm -f '$tmp' '$tmp.bak'" EXIT +info "fetching $LUCEBOX_INSTALL_URL" +curl -fsSL "$LUCEBOX_INSTALL_URL" -o "$tmp" \ + || die "download failed from $LUCEBOX_INSTALL_URL" + +# ── sanity check ────────────────────────────────────────────────────────── +# Refuse to install something that isn't recognizably lucebox.sh. Catches +# 404 pages, redirects to HTML, and accidental URL typos. +head -1 "$tmp" | grep -q '^#!/usr/bin/env bash$' \ + || die "downloaded file does not look like a bash script (got: $(head -1 "$tmp"))" +grep -q '^VERSION=' "$tmp" \ + || die "downloaded file is missing VERSION marker — not lucebox.sh?" + +# ── decide what gets baked in as the persisted channel ─────────────────── +# `lucebox update` reads LUCEBOX_INSTALLED_FROM from the installed copy and +# re-fetches from it. Persisting a SHA-pinned URL is a footgun — every +# future update would re-install the same frozen SHA forever, defeating +# the point of `update`. So: +# +# 1. If $LUCEBOX_INSTALL_CHANNEL is set, that's the persisted URL +# (caller takes responsibility for picking a real branch URL). +# 2. Else if LUCEBOX_INSTALL_URL has a 40-char hex SHA segment, refuse +# to persist it — tell the user to set LUCEBOX_INSTALL_CHANNEL. +# Common case: someone curl'd from /raw// to bypass a stale CDN +# cache during dev; they meant for updates to track the branch. +# 3. Else persist LUCEBOX_INSTALL_URL as-is (branch or canonical main). +channel_url="${LUCEBOX_INSTALL_CHANNEL:-}" +if [ -z "$channel_url" ]; then + # Match a full 40-char hex SHA in the URL path, not the broader + # {7,40} range — a 7-39 char hex segment is more likely a branch + # name shaped like a short SHA (e.g. `feat/abc1234-hotfix`) than an + # actual SHA-pin. Keeping the gate at exactly 40 chars matches what + # `git rev-parse HEAD` emits and what `/raw//` URLs from + # GitHub's CDN actually carry. + if [[ "$LUCEBOX_INSTALL_URL" =~ /[0-9a-fA-F]{40}/[^/]+\.sh$ ]]; then + die "$(cat </install.sh | \\ + LUCEBOX_INSTALL_URL=/lucebox.sh \\ + LUCEBOX_INSTALL_CHANNEL=https://raw.githubusercontent.com////lucebox.sh \\ + bash +EOM +)" + fi + channel_url="$LUCEBOX_INSTALL_URL" +fi + +# Bake the channel URL into the file. Use a `|` delimiter since URLs +# contain `/`. The line is expected to exist in lucebox.sh with a `:-` +# default; we rewrite the whole assignment. +# +# The URL ends up inside a bash double-quoted literal in the installed +# script, so any of $ ` " \ in `channel_url` would break the installed +# file (or worse, allow command substitution to run at next sourcing). +# Validate that the URL is plain http(s)+ASCII-URL-safe characters; we +# don't expect arbitrary content here, only an upstream raw.github URL +# (or a forked equivalent). Escape the sed metachars (\&|) separately so +# the substitution itself round-trips. +case "$channel_url" in + *['"$`\']*) die "channel URL contains unsafe characters: $channel_url" ;; +esac +escaped_url=$(printf '%s' "$channel_url" | sed 's/[\\&|]/\\&/g') +sed "s|^LUCEBOX_INSTALLED_FROM=.*|LUCEBOX_INSTALLED_FROM=\"$escaped_url\"|" "$tmp" > "$tmp.baked" +mv "$tmp.baked" "$tmp" +grep -q "^LUCEBOX_INSTALLED_FROM=\"$escaped_url\"$" "$tmp" \ + || die "failed to bake install source into the downloaded script" + +# ── install ─────────────────────────────────────────────────────────────── +mkdir -p "$(dirname "$DEST")" +chmod +x "$tmp" +mv "$tmp" "$DEST" +trap - EXIT +ok "installed lucebox → $DEST" +info " fetched from: $LUCEBOX_INSTALL_URL" +info " update channel: $channel_url" +if [ "$LUCEBOX_INSTALL_URL" != "$channel_url" ]; then + info " (lucebox update will track the channel URL, not the fetch URL)" +fi + +# ── PATH hint ───────────────────────────────────────────────────────────── +case ":${PATH:-}:" in + *":$(dirname "$DEST"):"*) ;; + *) info " hint: add $(dirname "$DEST") to PATH so 'lucebox' is on the path" ;; +esac + +cat </dev/null || realpath "$0" 2>/dev/null || echo "$0")" +SCRIPT_NAME="$(basename "$SCRIPT_PATH")" + +# ── tunables / env overrides ─────────────────────────────────────────────── +# Host-side scalars (image registry+variant, port, container name, models +# dir). Resolution order, applied uniformly via _lucebox_resolve below: +# 1. $LUCEBOX_ per-invocation env override +# 2. config.toml
. persisted user choice (system of record) +# 3. derived / canonical default +# This keeps the wrapper and the in-container Python CLI agreeing on +# effective values — config.toml is the single source of truth, both +# sides read it. +UNIT_NAME="lucebox.service" +UNIT_PATH="${XDG_CONFIG_HOME:-$HOME/.config}/systemd/user/$UNIT_NAME" + +# CUDA driver floor for the prebuilt CUDA 12 image. +# shellcheck disable=SC2034 +MIN_DRIVER_CUDA12=525 + +# Canonical source of `lucebox.sh`. The bootstrap installer (`install.sh`) +# rewrites this line at install time to record which URL the user actually +# installed from — `lucebox update` then re-pulls from the same channel +# without losing track of forks. Falls back to the Luce-Org main branch +# when nothing was baked in (e.g. someone curl'd the script directly). +LUCEBOX_INSTALLED_FROM="${LUCEBOX_INSTALLED_FROM:-https://raw.githubusercontent.com/Luce-Org/lucebox-hub/main/lucebox.sh}" + +# Path to the persisted config.toml. Mirrors +# lucebox.config.default_config_path: $LUCEBOX_HOME/config.toml if set, +# else $HOME/.lucebox/config.toml. Read-only from this wrapper — the +# Python CLI is the writer. +_lucebox_config_path() { + if [ -n "${LUCEBOX_HOME:-}" ]; then + printf '%s/config.toml' "$LUCEBOX_HOME" + return + fi + printf '%s/.lucebox/config.toml' "$HOME" +} + +# Read a `
.` value from config.toml. Returns empty if the +# file is missing, the section/key is absent, or the value is empty. +# Handles the subset of TOML that lucebox writes: +# [section] +# key = "string" # surrounding double-quotes are stripped +# key = 8080 # bare scalars passed through verbatim +# key = true # same +# Inline `# comment` is honored. Arrays / inline tables / multi-line +# strings aren't written by the Python persister, so we don't parse them. +_lucebox_config_get() { + local dotted="$1" cfg + cfg="$(_lucebox_config_path)" + [ -f "$cfg" ] || return 0 + local section="${dotted%.*}" + local key="${dotted##*.}" + [ "$section" = "$dotted" ] && section="" + awk -v want_section="$section" -v want_key="$key" ' + BEGIN { current = "" } + /^[[:space:]]*\[/ { + t = $0 + sub(/^[[:space:]]*\[[[:space:]]*/, "", t) + sub(/[[:space:]]*\][[:space:]]*$/, "", t) + current = t + next + } + /^[[:space:]]*#/ { next } + /=/ { + if (current != want_section) next + line = $0 + sub(/#.*$/, "", line) + eq = index(line, "=") + if (eq == 0) next + k = substr(line, 1, eq - 1) + v = substr(line, eq + 1) + gsub(/^[[:space:]]+|[[:space:]]+$/, "", k) + gsub(/^[[:space:]]+|[[:space:]]+$/, "", v) + if (k != want_key) next + if (length(v) >= 2 && substr(v, 1, 1) == "\"" && substr(v, length(v), 1) == "\"") + v = substr(v, 2, length(v) - 2) + print v + exit + } + ' "$cfg" +} + +# Resolve a scalar through the precedence ladder. env_value comes from +# the caller (typically `"${LUCEBOX_FOO:-}"` — the `:-` matters under +# `set -u`). +_lucebox_resolve() { + local env_value="$1" toml_key="$2" default="$3" v + if [ -n "$env_value" ]; then + printf '%s' "$env_value" + return + fi + v="$(_lucebox_config_get "$toml_key")" + if [ -n "$v" ]; then + printf '%s' "$v" + return + fi + printf '%s' "$default" +} + +# Derive the default image URL from the install source so a fork install +# (e.g. easel/lucebox-hub) gets the fork's GHCR image automatically when +# config.toml hasn't pinned one yet. Pattern: +# https://raw.githubusercontent.com////lucebox.sh +# → ghcr.io// +# GHCR rejects mixed-case org paths so the org segment is lowercased; the +# repo name is preserved as-is. Falls back to the canonical Luce-Org image +# when the URL doesn't match the raw.githubusercontent.com pattern. +_lucebox_derive_image() { + # The ref segment can contain slashes (e.g. `feat/lucebox-docker`), so + # the middle `.+` greedily eats everything up to the trailing + # `/lucebox.sh`. The first two `[^/]+` capture org + repo, which are + # never slash-containing on GitHub. + local url="$1" org repo + if [[ "$url" =~ ^https?://raw\.githubusercontent\.com/([^/]+)/([^/]+)/.+/lucebox\.sh$ ]]; then + org=$(printf '%s' "${BASH_REMATCH[1]}" | tr '[:upper:]' '[:lower:]') + repo="${BASH_REMATCH[2]}" + printf 'ghcr.io/%s/%s' "$org" "$repo" + return + fi + printf 'ghcr.io/luce-org/lucebox-hub' +} + +# Effective scalars, env > config.toml > default. +CONTAINER_NAME=$(_lucebox_resolve "${LUCEBOX_CONTAINER:-}" runtime.container_name "lucebox") +DEFAULT_PORT=$(_lucebox_resolve "${LUCEBOX_PORT:-}" runtime.port "8080") +DEFAULT_MODELS_DIR=$(_lucebox_resolve "${LUCEBOX_MODELS:-}" paths.models "${XDG_DATA_HOME:-$HOME/.local/share}/lucebox/models") +IMAGE_BASE=$(_lucebox_resolve "${LUCEBOX_IMAGE:-}" image.registry "$(_lucebox_derive_image "$LUCEBOX_INSTALLED_FROM")") + +# ── LUCEBOX_HOST_* safe defaults (belt-and-suspenders) ──────────────────── +# `set -u` makes any unbound LUCEBOX_HOST_* read fatal. Historically this has +# been the #1 source of regressions in this wrapper: someone adds a code path +# that touches a LUCEBOX_HOST_* var before probe_host has run, the call sites +# that DO pre-probe still work, and the bug ships. To make the bug literally +# unrepresentable we seed every LUCEBOX_HOST_* with an explicit safe default +# at script-load time (these mirror probe_host's "nothing detected" state). +# probe_host then overwrites them with real values. Any future read — pre- or +# post-probe — is now well-defined. +: "${LUCEBOX_HOST_NPROC:=1}" +: "${LUCEBOX_HOST_RAM_GB:=0}" +: "${LUCEBOX_HOST_GPU_VENDOR:=none}" +: "${LUCEBOX_HOST_GPU_NAME:=}" +: "${LUCEBOX_HOST_GPU_COUNT:=0}" +: "${LUCEBOX_HOST_VRAM_GB:=0}" +: "${LUCEBOX_HOST_GPU_SM:=}" +: "${LUCEBOX_HOST_DRIVER_VERSION:=}" +: "${LUCEBOX_HOST_DRIVER_MAJOR:=0}" +: "${LUCEBOX_HOST_HAS_SYSTEMD:=0}" +: "${LUCEBOX_HOST_IS_WSL:=0}" +: "${LUCEBOX_HOST_HAS_DOCKER:=0}" +: "${LUCEBOX_HOST_DOCKER_VERSION:=}" +: "${LUCEBOX_HOST_HAS_CTK:=none}" +# Host-identity facts (item 1 — host-identity capture). These ride along +# the existing LUCEBOX_HOST_* convoy into the container so /opt/lucebox-hub/ +# HOST_INFO can be written without re-probing inside the container (where +# /proc and nvidia-smi see the container's view, not the rig's). +: "${LUCEBOX_HOST_OS_PRETTY:=}" +: "${LUCEBOX_HOST_KERNEL:=}" +: "${LUCEBOX_HOST_WSL_VERSION:=}" +: "${LUCEBOX_HOST_NVIDIA_CTK_VERSION:=}" +: "${LUCEBOX_HOST_CPU_MODEL:=}" +: "${LUCEBOX_HOST_GPU_LIST_CSV:=}" +: "${LUCEBOX_HOST_CUDA_VISIBLE_DEVICES:=}" +# Tracks whether probe_host has actually run; pieces of the code that need +# fresh host facts (e.g. cmd_check, cmd_serve) gate on this. Default 0. +: "${_LUCEBOX_HOST_PROBED:=0}" + +# ── output helpers ──────────────────────────────────────────────────────── +if [ -t 1 ] && [ -z "${NO_COLOR:-}" ]; then + C_INFO='\033[1;34m'; C_OK='\033[1;32m'; C_WARN='\033[1;33m' + C_ERR='\033[1;31m'; C_DIM='\033[2m'; C_RST='\033[0m' +else + C_INFO=''; C_OK=''; C_WARN=''; C_ERR=''; C_DIM=''; C_RST='' +fi + +info() { printf '%b[INFO]%b %s\n' "$C_INFO" "$C_RST" "$*"; } +ok() { printf '%b[OK]%b %s\n' "$C_OK" "$C_RST" "$*"; } +warn() { printf '%b[WARN]%b %s\n' "$C_WARN" "$C_RST" "$*"; } +err() { printf '%b[ERROR]%b %s\n' "$C_ERR" "$C_RST" "$*" >&2; } +hint() { printf ' %b%s%b\n' "$C_DIM" "$*" "$C_RST"; } +die() { err "$*"; exit 1; } + +# ── host probing ────────────────────────────────────────────────────────── +# Sets the LUCEBOX_HOST_* variables consumed by the in-container Python CLI +# (passed through with -e). The Python side trusts these and doesn't reprobe +# — it can't see the host's /proc anyway, only the container's. + +probe_host() { + LUCEBOX_HOST_NPROC=$(nproc 2>/dev/null || echo 1) + # RAM: try Linux /proc/meminfo first, then macOS/BSD sysctl, else 0. + LUCEBOX_HOST_RAM_GB=0 + if [ -r /proc/meminfo ]; then + LUCEBOX_HOST_RAM_GB=$(awk '/MemTotal/{printf "%.0f", $2/1024/1024}' /proc/meminfo 2>/dev/null || echo 0) + elif command -v sysctl &>/dev/null; then + mem_bytes=$(sysctl -n hw.memsize 2>/dev/null || echo 0) + LUCEBOX_HOST_RAM_GB=$(( mem_bytes / 1024 / 1024 / 1024 )) + fi + LUCEBOX_HOST_GPU_VENDOR="none" + LUCEBOX_HOST_GPU_NAME="" + LUCEBOX_HOST_GPU_COUNT=0 + LUCEBOX_HOST_VRAM_GB=0 + LUCEBOX_HOST_GPU_SM="" + LUCEBOX_HOST_DRIVER_VERSION="" + LUCEBOX_HOST_DRIVER_MAJOR=0 + + if command -v nvidia-smi &>/dev/null; then + local q + if q=$(nvidia-smi --query-gpu=name,memory.total,driver_version,compute_cap \ + --format=csv,noheader,nounits 2>/dev/null) && [ -n "$q" ]; then + LUCEBOX_HOST_GPU_VENDOR="nvidia" + LUCEBOX_HOST_GPU_NAME=$(printf '%s\n' "$q" | head -1 | awk -F', ' '{print $1}') + local mem_mib + mem_mib=$(printf '%s\n' "$q" | head -1 | awk -F', ' '{print $2}') + LUCEBOX_HOST_VRAM_GB=$((mem_mib / 1024)) + LUCEBOX_HOST_DRIVER_VERSION=$(printf '%s\n' "$q" | head -1 | awk -F', ' '{print $3}') + LUCEBOX_HOST_DRIVER_MAJOR=${LUCEBOX_HOST_DRIVER_VERSION%%.*} + local cc + cc=$(printf '%s\n' "$q" | head -1 | awk -F', ' '{print $4}') + LUCEBOX_HOST_GPU_SM="${cc//./}" + LUCEBOX_HOST_GPU_COUNT=$(printf '%s\n' "$q" | wc -l) + fi + # Multi-GPU enumeration for /props.host. The single-GPU vars + # above (GPU_NAME / GPU_SM / VRAM_GB / DRIVER_VERSION) keep + # describing GPU 0 for back-compat with cmd_check + autotune; + # the full per-GPU CSV rides along separately so HOST_INFO can + # emit the whole array. + LUCEBOX_HOST_GPU_LIST_CSV=$(nvidia-smi \ + --query-gpu=index,uuid,pci.bus_id,name,compute_cap,memory.total,power.limit \ + --format=csv,noheader 2>/dev/null || echo "") + fi + # CUDA_VISIBLE_DEVICES from the caller's env (empty default = "all GPUs"). + LUCEBOX_HOST_CUDA_VISIBLE_DEVICES="${CUDA_VISIBLE_DEVICES:-}" + + # OS / kernel identity. /etc/os-release is the freedesktop spec for + # "what distro is this?" and we keep PRETTY_NAME verbatim (it already + # includes the version, e.g. "Ubuntu 22.04.3 LTS"). + LUCEBOX_HOST_OS_PRETTY="" + if [ -r /etc/os-release ]; then + # shellcheck source=/dev/null + LUCEBOX_HOST_OS_PRETTY=$(. /etc/os-release 2>/dev/null && printf '%s' "${PRETTY_NAME:-}") + fi + LUCEBOX_HOST_KERNEL=$(uname -r 2>/dev/null || echo "") + + # WSL version detection. "wsl2" matches the kernel-side string the + # MS-shipped WSL2 kernel embeds; "wsl1" is what the legacy translation + # layer writes. Anything else stays empty (= not WSL). + LUCEBOX_HOST_WSL_VERSION="" + if [ -r /proc/version ]; then + if grep -q "microsoft-standard-WSL2" /proc/version 2>/dev/null; then + LUCEBOX_HOST_WSL_VERSION="wsl2" + elif grep -qi "Microsoft" /proc/version 2>/dev/null; then + LUCEBOX_HOST_WSL_VERSION="wsl1" + fi + fi + + # CPU model — first "model name" hit in /proc/cpuinfo. Cheaper than + # lscpu and keeps the bash side dep-free. + LUCEBOX_HOST_CPU_MODEL="" + if [ -r /proc/cpuinfo ]; then + LUCEBOX_HOST_CPU_MODEL=$(awk -F': ' '/^model name/{print $2; exit}' /proc/cpuinfo 2>/dev/null || echo "") + fi + + LUCEBOX_HOST_HAS_SYSTEMD=0 + if command -v systemctl &>/dev/null && systemctl --user show-environment &>/dev/null; then + LUCEBOX_HOST_HAS_SYSTEMD=1 + fi + + LUCEBOX_HOST_IS_WSL=0 + if grep -qi microsoft /proc/version 2>/dev/null \ + || [ -e /proc/sys/fs/binfmt_misc/WSLInterop ]; then + LUCEBOX_HOST_IS_WSL=1 + fi + + LUCEBOX_HOST_HAS_DOCKER=0 + LUCEBOX_HOST_DOCKER_VERSION="" + if command -v docker &>/dev/null && docker ps &>/dev/null; then + LUCEBOX_HOST_HAS_DOCKER=1 + LUCEBOX_HOST_DOCKER_VERSION=$(timeout 5 docker version --format '{{.Server.Version}}' 2>/dev/null || echo "") + fi + + LUCEBOX_HOST_HAS_CTK="none" + if [ "$LUCEBOX_HOST_HAS_DOCKER" = "1" ]; then + if command -v nvidia-container-runtime &>/dev/null; then + LUCEBOX_HOST_HAS_CTK="runtime" + elif command -v nvidia-ctk &>/dev/null \ + && nvidia-ctk cdi list 2>/dev/null | grep -q 'nvidia.com/gpu'; then + LUCEBOX_HOST_HAS_CTK="cdi" + elif command -v nvidia-ctk &>/dev/null; then + LUCEBOX_HOST_HAS_CTK="installed-unwired" + fi + fi + + # NVIDIA Container Toolkit version (best-effort; empty when nvidia-ctk + # is not installed). nvidia-ctk --version prints "NVIDIA Container + # Toolkit CLI version 1.16.2" on a single line — extract the trailing + # token so the host-info JSON carries just the version, not the banner. + LUCEBOX_HOST_NVIDIA_CTK_VERSION="" + if command -v nvidia-ctk &>/dev/null; then + LUCEBOX_HOST_NVIDIA_CTK_VERSION=$(nvidia-ctk --version 2>/dev/null \ + | awk '/version/{print $NF; exit}' \ + || echo "") + fi + + export LUCEBOX_HOST_NPROC LUCEBOX_HOST_RAM_GB LUCEBOX_HOST_GPU_VENDOR + export LUCEBOX_HOST_GPU_NAME LUCEBOX_HOST_GPU_COUNT LUCEBOX_HOST_VRAM_GB + export LUCEBOX_HOST_GPU_SM LUCEBOX_HOST_DRIVER_VERSION LUCEBOX_HOST_DRIVER_MAJOR + export LUCEBOX_HOST_HAS_SYSTEMD LUCEBOX_HOST_IS_WSL + export LUCEBOX_HOST_HAS_DOCKER LUCEBOX_HOST_DOCKER_VERSION + export LUCEBOX_HOST_HAS_CTK + export LUCEBOX_HOST_OS_PRETTY LUCEBOX_HOST_KERNEL LUCEBOX_HOST_WSL_VERSION + export LUCEBOX_HOST_NVIDIA_CTK_VERSION LUCEBOX_HOST_CPU_MODEL + export LUCEBOX_HOST_GPU_LIST_CSV LUCEBOX_HOST_CUDA_VISIBLE_DEVICES + _LUCEBOX_HOST_PROBED=1 +} + +# Cheap idempotency wrapper. Anything that needs real host facts (vs the safe +# defaults seeded at script-load) calls this. Subcommands that go straight to +# `systemctl`/`journalctl` no longer need to remember to call probe_host. +ensure_probed() { + [ "$_LUCEBOX_HOST_PROBED" = "1" ] || probe_host +} + +pick_variant() { + # CUDA 12.8 is the supported image variant for this branch. Effective + # value goes through the same env > config.toml > default ladder as + # everything else so `config set image.variant=...` propagates. + _lucebox_resolve "${LUCEBOX_VARIANT:-}" image.variant "cuda12" +} + +# ── prereq checks (host-only) ───────────────────────────────────────────── +# Print-and-exit on anything that needs root to install. The Python CLI does +# the richer reporting; this is the bare minimum to make `docker run` viable. + +require_host_prereqs() { + local missing=0 + if ! command -v docker &>/dev/null; then + err "docker is not installed" + hint "Install: https://docs.docker.com/engine/install/" + missing=1 + elif ! docker ps &>/dev/null; then + err "docker daemon not reachable" + hint "sudo systemctl start docker (or: add your user to the 'docker' group, then re-login)" + missing=1 + fi + + if ! command -v nvidia-smi &>/dev/null; then + err "nvidia-smi not found — no NVIDIA driver detected" + hint "Install the NVIDIA driver: https://www.nvidia.com/Download/index.aspx" + missing=1 + elif ! nvidia-smi --query-gpu=name --format=csv,noheader &>/dev/null; then + err "nvidia-smi present but NVML calls fail — likely a driver/library mismatch" + hint "Reboot, or reinstall the matching NVIDIA driver package" + missing=1 + fi + + [ "$missing" = "0" ] || exit 1 +} + +require_ctk() { + case "$LUCEBOX_HOST_HAS_CTK" in + runtime|cdi) return 0 ;; + installed-unwired) + err "NVIDIA Container Toolkit installed but not wired into docker" + hint "sudo nvidia-ctk runtime configure --runtime=docker && sudo systemctl restart docker" + hint " or generate a CDI spec: sudo nvidia-ctk cdi generate --output=/etc/cdi/nvidia.yaml" + exit 1 ;; + none|*) + err "NVIDIA Container Toolkit not installed" + hint "Install: https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html" + hint "Then register with docker:" + hint " sudo nvidia-ctk runtime configure --runtime=docker && sudo systemctl restart docker" + exit 1 ;; + esac +} + +require_systemd() { + # Earlier versions of this wrapper had `start`/`stop`/`logs`/etc. drop + # straight into cmd_systemctl_passthrough without probing first, which + # tripped `set -u` on the reference below. Two layers of defence now: + # 1) top-of-script seeds LUCEBOX_HOST_HAS_SYSTEMD=0 unconditionally, so + # no read can be unbound even if probe_host is bypassed entirely. + # 2) ensure_probed runs probe_host on first call so we still get the + # real answer for the require_systemd error path. + ensure_probed + if [ "$LUCEBOX_HOST_HAS_SYSTEMD" != "1" ]; then + err "user systemd is not available — required for $1" + hint "On WSL: set 'systemd=true' under [boot] in /etc/wsl.conf, then 'wsl --shutdown'." + hint "Otherwise: install systemd, or run '$SCRIPT_NAME serve' to run in the foreground without systemd." + exit 1 + fi +} + +# ── docker run construction ─────────────────────────────────────────────── +# All the Python-CLI subcommands share the same docker run incantation: +# mount the host docker socket (so the in-container CLI can spawn server / +# bench containers on the host daemon), mount $HOME at the same path (so +# paths look identical in and out), and pass host facts via env. When an +# NVIDIA GPU is detected we also pass --gpus all so the orchestrator can +# call nvidia-smi during profile snapshot export; without it nvidia_smi_csv (and +# any downstream power/utilization fields) come back empty. + +DOCKER_SOCK_PATH="${DOCKER_HOST:-/var/run/docker.sock}" +DOCKER_SOCK_PATH="${DOCKER_SOCK_PATH#unix://}" + +# Append `-e LUCEBOX_HOST_=` for every exported host fact onto the +# named docker-argv array (bash 4.3+ nameref). The Python side reads these +# instead of reprobing — see build_orchestrator_argv / cmd_exec_in_container. +_append_host_env() { + # shellcheck disable=SC2178 # nameref to a caller's array, not a string + local -n _arr="$1" + local var + for var in $(compgen -e | grep '^LUCEBOX_HOST_' || true); do + _arr+=(-e "$var=${!var}") + done +} + +# Append the LUCEBOX_* scalar overrides (image/variant/port/container/models) +# plus the optional HF_TOKEN guard onto the named docker-argv array. Shared +# by the docker-run (build_orchestrator_argv) and docker-exec +# (cmd_exec_in_container) paths so both forward an identical env subset. +_append_scalar_env() { + # shellcheck disable=SC2178 # nameref to a caller's array, not a string + local -n _arr="$1" + local variant="$2" + _arr+=(-e "LUCEBOX_IMAGE=$IMAGE_BASE") + _arr+=(-e "LUCEBOX_VARIANT=$variant") + _arr+=(-e "LUCEBOX_PORT=$DEFAULT_PORT") + _arr+=(-e "LUCEBOX_CONTAINER=$CONTAINER_NAME") + _arr+=(-e "LUCEBOX_MODELS=$DEFAULT_MODELS_DIR") + [ -n "${HF_TOKEN:-}" ] && _arr+=(-e "HF_TOKEN=$HF_TOKEN") + return 0 +} + +# Pick docker's interactive flags: -it on a real tty, -i otherwise. +# Writes into a caller-supplied array via nameref. This MUST run in the +# caller's scope (not a subshell or `< <(...)` process substitution): the +# `[ -t 1 ]` test inspects fd 1, and inside a process substitution fd 1 is +# the pipe to the consumer, not the terminal — which would force -i even on +# a real tty and break the interactive client TUIs (lucebox claude, etc.). +_set_tty_flags() { # usage: _set_tty_flags arrayname + # shellcheck disable=SC2178 + local -n _a="$1" + if [ -t 0 ] && [ -t 1 ]; then + _a=(-it) + else + _a=(-i) + fi +} + +build_orchestrator_argv() { + local variant="$1"; shift + local tty=() + _set_tty_flags tty + local argv=(docker run --rm "${tty[@]}") + if [ "${LUCEBOX_HOST_GPU_VENDOR:-none}" = "nvidia" ]; then + argv+=(--gpus all) + fi + argv+=(--name "${CONTAINER_NAME}-cli-$$") + argv+=(--user "$(id -u):$(id -g)") + # Only bind-mount the docker socket when DOCKER_HOST actually points + # at a unix socket on this host. With DOCKER_HOST=tcp://… or ssh://… + # the path we'd construct is `tcp` or empty, and `docker run -v` would + # bark with an "invalid mount" error before the orchestrator even + # starts. The orchestrator-in-container relies on docker access only + # when actually needed; pulling that mount when the host talks to + # docker over TCP/SSH is fine. + if [ -S "$DOCKER_SOCK_PATH" ]; then + argv+=(--group-add "$(stat -c '%g' "$DOCKER_SOCK_PATH")") + argv+=(-v "$DOCKER_SOCK_PATH:/var/run/docker.sock") + fi + argv+=(-v "$HOME:$HOME") + # Bind-mount the XDG models dir explicitly (host = container path) so + # paths line up in/out. The $HOME mount above already covers it when + # XDG_DATA_HOME is unset, but an explicit -v is required when the user + # points XDG_DATA_HOME outside $HOME. + mkdir -p "$DEFAULT_MODELS_DIR" + argv+=(-v "$DEFAULT_MODELS_DIR:$DEFAULT_MODELS_DIR") + argv+=(-w "$PWD") + argv+=(-e "HOME=$HOME") + # Host facts — Python side reads these instead of reprobing. + _append_host_env argv + # User overrides for image/port/container/models scalars + HF_TOKEN. + # Always exports the resolved models dir so the in-container CLI sees + # the same path the wrapper mounts (the XDG default flows through too). + _append_scalar_env argv "$variant" + + argv+=("${IMAGE_BASE}:${variant}") + # `lucebox` is the entrypoint subcommand handled by server/scripts/entrypoint.sh + # — it execs `python -m lucebox` with whatever args we pass on. + argv+=(lucebox "$@") + printf '%s\n' "${argv[@]}" +} + +# ── subcommand implementations ──────────────────────────────────────────── + +cmd_serve() { + # Long-running foreground server. Also what systemd's ExecStart= calls. + # + # Two-stage so config.toml takes effect: + # 1. Run an ephemeral orchestrator container that emits the canonical + # server docker-run argv from .lucebox/config.toml (one arg per + # line on stdout). + # 2. Exec that argv. + # + # If stage 1 fails (image not pulled yet, no config), fall back to a + # conservative docker run — the container's own VRAM-tiered autotune + # picks reasonable defaults from there. + require_host_prereqs + ensure_probed + require_ctk + local variant + variant=$(pick_variant) + + # Pre-flight: refuse to stomp on something that's already serving this + # slot. Three states to distinguish, because silently `docker rm -f`-ing + # whatever is there hides real bugs (e.g. the user forgot they had a + # systemd unit up, and we'd happily race two servers on the same port): + # + # 1. systemd unit active → refuse, redirect to `logs`/`stop` + # 2. container running (no systemd)→ refuse, redirect to `docker logs` + # 3. container present but stopped → orphan from a SIGKILLed previous + # run (docker run --rm only cleans up on clean exit). Remove it, + # but TELL the user — they need to know their last run died dirty. + # CRITICAL: when systemd invokes US as the unit's ExecStart, is-active + # returns true *because of us* — refusing here would deadlock the unit + # in a restart loop (and historically did — commit a30dbe5 shipped this + # bug). systemd sets $INVOCATION_ID in every service exec, so its + # presence is the unambiguous "I am running as the systemd ExecStart" + # signal. Skip the unit-active check in that case; the container-state + # check below still catches a stale container holding the slot. + if [ -z "${INVOCATION_ID:-}" ] \ + && systemctl --user is-active --quiet "$UNIT_NAME" 2>/dev/null; then + err "${UNIT_NAME} is already running under systemd." + hint " $SCRIPT_NAME logs # follow the journal" + hint " $SCRIPT_NAME restart # bounce the service" + hint " $SCRIPT_NAME stop # stop the service" + exit 1 + fi + local container_state + container_state=$(docker inspect --format '{{.State.Status}}' "$CONTAINER_NAME" 2>/dev/null || echo absent) + case "$container_state" in + absent) + ;; + running|restarting) + err "Container '$CONTAINER_NAME' is already running (outside systemd)." + hint " docker logs -f $CONTAINER_NAME # follow output" + hint " $SCRIPT_NAME stop # stop it" + exit 1 + ;; + exited|created|paused|dead) + info "Removing stale '$CONTAINER_NAME' container (state=$container_state, likely from a previous unclean exit)" + docker rm -f "$CONTAINER_NAME" >/dev/null + ;; + *) + warn "Container '$CONTAINER_NAME' is in unexpected state '$container_state' — removing" + docker rm -f "$CONTAINER_NAME" >/dev/null + ;; + esac + + local orch_argv server_argv + mapfile -t orch_argv < <(build_orchestrator_argv "$variant" print-serve-argv) + + if mapfile -t server_argv < <("${orch_argv[@]}" 2>/dev/null) \ + && [ "${#server_argv[@]}" -gt 0 ] \ + && [ "${server_argv[0]}" = "docker" ]; then + info "Starting lucebox server (variant=$variant, from config.toml)" + _serve_and_track "${server_argv[@]}" + return $? + fi + + warn "Couldn't fetch server argv from container (image not pulled?) — using fallback" + info "Starting lucebox server (variant=$variant, port=$DEFAULT_PORT, defaults only)" + local fallback_models="$DEFAULT_MODELS_DIR" + mkdir -p "$fallback_models" + # Forward host facts even on the fallback path so the in-container + # entrypoint can still write /opt/lucebox-hub/HOST_INFO from the host's + # view of the rig. Matches the orchestrator path (see + # build_orchestrator_argv) — without it, HOST_INFO would be written + # with "source: unknown" any time print-serve-argv fails. + local fallback_argv=(docker run --rm + --name "$CONTAINER_NAME" + --gpus all + -p "$DEFAULT_PORT:8080" + -v "$HOME:$HOME" + -v "$fallback_models:/opt/lucebox-hub/server/models") + _append_host_env fallback_argv + fallback_argv+=("${IMAGE_BASE}:${variant}") + _serve_and_track "${fallback_argv[@]}" +} + +# Foreground server runner with controlling-process lifetime semantics: +# the docker daemon owns containers independently of the CLI, so a bare +# `exec docker run` leaves the container alive after the wrapper's parent +# (a terminal, a systemd unit, anything) goes away. `docker run --rm` only +# cleans up on the container's own clean exit, not on our death. +# +# Fix: run docker as a child, install signal traps that issue `docker stop` +# before exiting. Now `lucebox serve` behaves like a normal foreground +# program — close the terminal, kill the wrapper, send SIGTERM from +# systemd, the container goes down with it. +# +# Stops also from EXIT so even a `set -e` propagation cleans up. +_serve_and_track() { + "$@" & + local docker_pid=$! + # shellcheck disable=SC2317 # called via trap, not "unreachable" + _serve_stop() { + trap - HUP INT TERM EXIT + # Best-effort: container may already be exiting / never started. + # `docker stop` blocks up to -t seconds for graceful shutdown + # (server handles SIGTERM), then SIGKILLs. 10s is enough for the + # in-flight request to finish on a typical decode. + docker stop -t 10 "$CONTAINER_NAME" >/dev/null 2>&1 || true + # docker_pid may be out of scope when this fires as the EXIT trap + # after _serve_and_track has unwound (e.g. the server fast-failed); + # guard so `set -u` doesn't turn cleanup into its own error. + [ -n "${docker_pid:-}" ] && wait "$docker_pid" 2>/dev/null || true + } + trap _serve_stop HUP INT TERM EXIT + wait "$docker_pid" + local rc=$? + trap - HUP INT TERM EXIT + return $rc +} + +cmd_systemd_install() { + require_host_prereqs + ensure_probed + require_systemd "service install" + local docker_bin + docker_bin=$(command -v docker) + + mkdir -p "$(dirname "$UNIT_PATH")" + # Capture the user's resolved env at install time so the unit launches + # with the same image/variant/port/models the user expected when they + # ran `lucebox install`. Systemd's user-session env is sparse — without + # this block, the wrapper inside the unit would fall back to the + # in-script defaults and silently pick a different image or models + # directory than the user's interactive session uses. + # + # ExecStartPre cleans up any orphaned container with the target name + # left behind by a previous crash (docker's `--rm` only fires on clean + # exit — a SIGKILL or daemon restart leaves the name claimed, and the + # next ExecStart would die with "name already in use" while systemd + # reports a useless "exit code 125"). + cat > "$UNIT_PATH" </dev/null | awk -F= '/^Linger=/{print $2}') + if [ "$linger" != "yes" ]; then + warn "Linger is off for $USER — the service will stop when you log out" + hint "To enable (requires sudo): sudo loginctl enable-linger \"$USER\"" + fi + + printf '\nNext:\n' + hint " $SCRIPT_NAME start # start now" + hint " $SCRIPT_NAME enable # start at every login" + hint " $SCRIPT_NAME logs # follow the journal" +} + +cmd_systemd_uninstall() { + require_systemd "service uninstall" + if systemctl --user is-active --quiet "$UNIT_NAME" 2>/dev/null; then + info "Stopping $UNIT_NAME" + systemctl --user stop "$UNIT_NAME" || true + fi + if systemctl --user is-enabled --quiet "$UNIT_NAME" 2>/dev/null; then + info "Disabling $UNIT_NAME" + systemctl --user disable "$UNIT_NAME" || true + fi + if [ -f "$UNIT_PATH" ]; then + rm -f "$UNIT_PATH" + ok "Removed $UNIT_PATH" + else + info "No unit at $UNIT_PATH — nothing to remove" + fi + systemctl --user daemon-reload + hint "Config and models are left in place. Remove them by hand if you want." +} + +cmd_systemctl_passthrough() { + local action="$1" + require_systemd "$action" + if [ ! -f "$UNIT_PATH" ]; then + err "$UNIT_NAME is not installed — run '$SCRIPT_NAME install' first" + exit 1 + fi + case "$action" in + start|restart) + # `systemctl start` is fire-and-forget for Type=exec: it returns + # success as soon as execve() completes, even if the wrapper + # exits 1 a millisecond later. That gave us the worst possible + # UX — `lucebox start` reports no error but no container ever + # binds port 8080. Poll is-active for a few seconds and dump + # status + recent journal lines so the user sees the real cause. + local current + current=$(systemctl --user is-active "$UNIT_NAME" 2>/dev/null || true) + # `start` against an already-active unit: systemctl returns 0 + # silently. That's polite for scripts but confusing for humans + # — say so explicitly. For `restart` always run through. + if [ "$action" = "start" ] && [ "$current" = "active" ]; then + ok "$UNIT_NAME is already active" + hint "logs: $SCRIPT_NAME logs" + hint "smoke: curl -s http://localhost:$DEFAULT_PORT/v1/models" + hint "(use \`$SCRIPT_NAME restart\` to bounce, \`$SCRIPT_NAME stop\` to halt)" + return 0 + fi + # `start` against a unit stuck in restart-loop ("activating") is + # the symptom of a broken ExecStart — calling start would just + # block waiting for active that never comes. Surface this + # specifically so the user goes to `lucebox logs` to find the + # exit reason rather than waiting for the poll to give up. + if [ "$action" = "start" ] && [ "$current" = "activating" ]; then + err "$UNIT_NAME is in restart-loop (state=activating)" + hint "the unit is failing and being auto-restarted by systemd" + hint " $SCRIPT_NAME stop # halt the loop first" + hint " $SCRIPT_NAME logs # find the exit reason" + exit 1 + fi + info "$action $UNIT_NAME" + if ! systemctl --user "$action" "$UNIT_NAME"; then + err "systemctl --user $action $UNIT_NAME failed" + systemctl --user status "$UNIT_NAME" --no-pager -n 30 || true + exit 1 + fi + local i state + for i in 1 2 3 4 5 6 7 8 9 10; do + state=$(systemctl --user is-active "$UNIT_NAME" 2>/dev/null || true) + case "$state" in + active) break ;; # already up — no need to keep polling + activating) ;; # still booting; keep waiting + *) break ;; # failed / inactive — fall through to error path + esac + sleep 1 + done + state=$(systemctl --user is-active "$UNIT_NAME" 2>/dev/null || true) + if [ "$state" != "active" ]; then + err "$UNIT_NAME did not reach active state (current: ${state:-unknown})" + if [ "$state" = "activating" ]; then + hint "the unit is in a restart loop — \`$SCRIPT_NAME stop\` to halt it" + fi + hint "status:" + systemctl --user status "$UNIT_NAME" --no-pager -n 30 || true + hint "recent journal:" + journalctl --user -u "$UNIT_NAME" -n 30 --no-pager || true + exit 1 + fi + ok "$UNIT_NAME is active" + hint "logs: $SCRIPT_NAME logs" + hint "smoke: curl -s http://localhost:$DEFAULT_PORT/v1/models" + ;; + stop|enable|disable) + exec systemctl --user "$action" "$UNIT_NAME" ;; + status) + exec systemctl --user status "$UNIT_NAME" --no-pager ;; + *) + die "unknown systemctl passthrough: $action" ;; + esac +} + +cmd_logs() { + require_systemd "logs" + # Pure passthrough: any flags the user wants (-f, -n, --since, ...) go + # straight to journalctl. Default is follow. + if [ $# -eq 0 ]; then + exec journalctl --user -u "$UNIT_NAME" -f + fi + exec journalctl --user -u "$UNIT_NAME" "$@" +} + +cmd_pull() { + # Pull has to run on the host. Delegating this into the container creates a + # stale-image trap: docker may start an old local tag before the fresh tag + # has been pulled. + require_host_prereqs + local variant + variant=$(pick_variant) + info "Pulling ${IMAGE_BASE}:${variant}" + exec docker pull "${IMAGE_BASE}:${variant}" +} + +cmd_update() { + # Re-run the bootstrap installer against the channel we were installed + # from. The installer is the source of truth for "how do you install + # lucebox correctly" — chmod, atomic mv, validation, baking the source + # URL back into the new copy so the channel is preserved across + # upgrades. Keeping the logic in install.sh means it can evolve + # independently (sha verify, signature check, etc.) and the installed + # `lucebox update` picks those changes up on the next run. + # + # The installer URL is derived from LUCEBOX_INSTALLED_FROM by swapping + # `lucebox.sh` → `install.sh` in the same directory, so forks don't + # need a separate registration. Override the source channel via + # $LUCEBOX_INSTALL_URL (e.g. to switch from canonical to a dev fork). + local source_url installer_url target + source_url="${LUCEBOX_INSTALL_URL:-$LUCEBOX_INSTALLED_FROM}" + if [[ "$source_url" != */lucebox.sh ]]; then + die "LUCEBOX_INSTALLED_FROM doesn't end in /lucebox.sh: $source_url" + fi + installer_url="${source_url%/lucebox.sh}/install.sh" + target=$(realpath "$SCRIPT_PATH") + + info "Updating lucebox via $installer_url" + info " source: $source_url" + info " target: $target" + + # Pass the URLs through to install.sh via env. The installer reads + # $LUCEBOX_INSTALL_URL (which we set to source_url) and + # $LUCEBOX_INSTALL_DEST (the realpath of *this* file, so a symlinked + # install replaces the actual file behind the link). + LUCEBOX_INSTALL_URL="$source_url" \ + LUCEBOX_INSTALL_DEST="$target" \ + bash -c "$(curl -fsSL "$installer_url")" \ + || die "update failed (installer exited non-zero)" +} + +cmd_completion() { + # Print shell completion script for bash / zsh / fish. Usage: + # + # # bash (in ~/.bashrc): + # source <(lucebox completion bash) + # + # # zsh (in ~/.zshrc, before `compinit`): + # source <(lucebox completion zsh) + # + # # fish: + # lucebox completion fish | source + # + # Keep this in sync with the dispatch table in main() and the sub-app + # verbs (config get/set/unset, models list/download). Adding a new + # top-level command means adding it here too. + local shell="${1:-}" + case "$shell" in + bash) + cat <<'BASH' +# lucebox bash completion. Source from ~/.bashrc: +# source <(lucebox completion bash) +_lucebox_complete() { + local cur prev cmds config_verbs models_verbs completion_shells + COMPREPLY=() + cur="${COMP_WORDS[COMP_CWORD]}" + prev="${COMP_WORDS[COMP_CWORD-1]}" + cmds="install uninstall start stop restart enable disable status logs \ + serve pull update check completion config models \ + print-run help version" + config_verbs="get set unset" + models_verbs="list download" + completion_shells="bash zsh fish" + + # Sub-app verbs / shell args. + case "$prev" in + config) COMPREPLY=( $(compgen -W "$config_verbs" -- "$cur") ); return ;; + models) COMPREPLY=( $(compgen -W "$models_verbs" -- "$cur") ); return ;; + completion) COMPREPLY=( $(compgen -W "$completion_shells" -- "$cur") ); return ;; + esac + + # Top-level command. + if [ "$COMP_CWORD" = 1 ]; then + COMPREPLY=( $(compgen -W "$cmds" -- "$cur") ) + return + fi +} +complete -F _lucebox_complete lucebox lucebox.sh +BASH + ;; + zsh) + # Bash-compat shim: zsh sources our bash completion through + # bashcompinit. Users who prefer native zsh _arguments-style + # completion can write their own; this gets `` working + # in two lines for free. + cat <<'ZSH' +# lucebox zsh completion. Source from ~/.zshrc (after compinit): +# source <(lucebox completion zsh) +autoload -Uz compinit bashcompinit +compinit +bashcompinit +ZSH + cmd_completion bash + ;; + fish) + cat <<'FISH' +# lucebox fish completion. Source from ~/.config/fish/config.fish: +# lucebox completion fish | source +complete -c lucebox -f +set -l __lucebox_cmds install uninstall start stop restart enable disable \ + status logs serve pull update check completion config models \ + print-run help version +for cmd in $__lucebox_cmds + complete -c lucebox -n "not __fish_seen_subcommand_from $__lucebox_cmds" -a $cmd +end +complete -c lucebox -n "__fish_seen_subcommand_from config" -a "get set unset" +complete -c lucebox -n "__fish_seen_subcommand_from models" -a "list download" +complete -c lucebox -n "__fish_seen_subcommand_from completion" -a "bash zsh fish" +FISH + ;; + ""|--help|-h) + cat </dev/null; then + _row 0 "docker daemon" "installed but unreachable — start the daemon or add user to 'docker' group" + else + _row 0 "docker daemon" "not installed — https://docs.docker.com/engine/install/" + fi + + # nvidia container toolkit + case "$LUCEBOX_HOST_HAS_CTK" in + runtime) _row 1 "nvidia ctk" "wired into docker (runtime)" ;; + cdi) _row 1 "nvidia ctk" "wired via CDI (nvidia.com/gpu)" ;; + installed-unwired) _row warn "nvidia ctk" "installed but not registered with docker — sudo nvidia-ctk runtime configure --runtime=docker && sudo systemctl restart docker" ;; + none|*) _row 0 "nvidia ctk" "not installed — https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html" ;; + esac + + # nvidia-smi + driver + if [ "$LUCEBOX_HOST_GPU_VENDOR" = "nvidia" ]; then + if [ "$LUCEBOX_HOST_DRIVER_MAJOR" -ge "$MIN_DRIVER_CUDA12" ]; then + _row 1 "nvidia driver" "$LUCEBOX_HOST_DRIVER_VERSION (≥ $MIN_DRIVER_CUDA12 required for cuda12)" + else + _row 0 "nvidia driver" "$LUCEBOX_HOST_DRIVER_VERSION (< $MIN_DRIVER_CUDA12 — cuda12 image will fail)" + fi + elif command -v nvidia-smi &>/dev/null; then + _row 0 "nvidia driver" "nvidia-smi present but NVML calls fail — driver/library mismatch, try reboot" + else + _row 0 "nvidia driver" "nvidia-smi not found — install the NVIDIA driver" + fi + + # GPU detail + if [ "$LUCEBOX_HOST_GPU_VENDOR" = "nvidia" ]; then + _row 1 "gpu" "$LUCEBOX_HOST_GPU_NAME × $LUCEBOX_HOST_GPU_COUNT (sm_$LUCEBOX_HOST_GPU_SM, ${LUCEBOX_HOST_VRAM_GB} GB VRAM)" + # cuda12 image arch coverage: sm_75;80;86;89;90;120 (see docker-bake.hcl) + case "$LUCEBOX_HOST_GPU_SM" in + 75|80|86|89|90|120) _row 1 "cuda12 arch" "sm_$LUCEBOX_HOST_GPU_SM covered by image" ;; + "") _row warn "cuda12 arch" "compute_cap not detected" ;; + *) _row warn "cuda12 arch" "sm_$LUCEBOX_HOST_GPU_SM not in image arch list (75;80;86;89;90;120)" ;; + esac + fi + + # systemd + if [ "$LUCEBOX_HOST_HAS_SYSTEMD" = "1" ]; then + _row 1 "user systemd" "available (needed for '$SCRIPT_NAME install')" + elif [ "$LUCEBOX_HOST_IS_WSL" = "1" ]; then + _row warn "user systemd" "WSL detected — set 'systemd=true' under [boot] in /etc/wsl.conf, then 'wsl --shutdown'" + else + _row warn "user systemd" "not available — '$SCRIPT_NAME install' (service unit) won't work; '$SCRIPT_NAME serve' (foreground) will" + fi + + # image we'd pull — marked ✗ when the host clearly can't run cuda12 + # (no nvidia driver, or no CTK wired into docker). It's still useful + # to print the line so the user knows what would be pulled, but a + # green ✓ would be misleading. + if [ "$LUCEBOX_HOST_GPU_VENDOR" != "nvidia" ]; then + _row 0 "image" "${IMAGE_BASE}:${variant} — requires NVIDIA driver" + elif [ "$LUCEBOX_HOST_HAS_CTK" = "none" ] || [ "$LUCEBOX_HOST_HAS_CTK" = "installed-unwired" ]; then + _row 0 "image" "${IMAGE_BASE}:${variant} — needs NVIDIA Container Toolkit wired into docker" + else + _row 1 "image" "${IMAGE_BASE}:${variant}" + fi + # RAM / cores (informational) + _row 1 "host" "${LUCEBOX_HOST_NPROC} cpus, ${LUCEBOX_HOST_RAM_GB} GB RAM" +} + +cmd_in_container() { + # Generic dispatcher: anything that isn't a systemd action goes here. + # Runs the in-container Python CLI with the supplied argv. + require_host_prereqs + ensure_probed + # CTK isn't strictly required for every subcommand (e.g. `config get` + # or `autotune` only touch local files), but the server-spawning + # subcommands need it. + # Letting docker error its own way is fine for the no-CTK case. + local variant + variant=$(pick_variant) + local argv + mapfile -t argv < <(build_orchestrator_argv "$variant" "$@") + exec "${argv[@]}" +} + +# Is the long-running lucebox container currently up? Used by the dispatcher +# to decide between `docker exec` into it (cheap, shares the running server's +# network namespace so localhost:8080 reaches the server) vs. `docker run` +# (cold start, isolated network — can't reach the live server). +# +# `docker ps -q -f name=^$` prints the container id when running, +# empty otherwise. The anchored regex avoids matching `lucebox-cli-12345` +# style ephemeral siblings. +_lucebox_container_running() { + # No docker on PATH → definitely not running. Don't even probe. + command -v docker >/dev/null 2>&1 || return 1 + local id + id=$(docker ps -q -f "name=^${CONTAINER_NAME}\$" 2>/dev/null || true) + [ -n "$id" ] +} + +# `docker exec` variant of cmd_in_container. Same calling convention, but: +# - shares the running container's network namespace (localhost:8080 → the +# server), filesystem, and mounts — no bind mounts needed. +# - skips the ~1-3s cold-start cost of a fresh `docker run --rm`. +# - only safe for steady-state / read-only / config-only subcommands. Any +# command that restarts the lucebox service (autotune --sweep, serve) +# would kill the very container the exec is in — caller must route those +# to cmd_in_container instead. +# +# Pass through the same env-var subset the run path uses so the in-container +# CLI sees consistent overrides whichever route it took: HOME, every +# LUCEBOX_HOST_*, the image/port/container/models scalars, and HF_TOKEN. +cmd_exec_in_container() { + require_host_prereqs + ensure_probed + local variant + variant=$(pick_variant) + local tty=() + _set_tty_flags tty + local argv=(docker exec "${tty[@]}") + argv+=(--user "$(id -u):$(id -g)") + argv+=(-w "$PWD") + argv+=(-e "HOME=$HOME") + _append_host_env argv + _append_scalar_env argv "$variant" + # The image has no top-level `lucebox` binary on PATH — that name only + # works as the first arg to /opt/lucebox-hub/server/scripts/entrypoint.sh, + # which then `exec uv run ... python -m lucebox`s. docker exec bypasses + # the image's ENTRYPOINT, so we invoke the entrypoint shim explicitly + # with `lucebox` as its SUBCMD and the user's argv tail. Keeps the + # exec path bit-for-bit equivalent to what docker run does on the + # SUBCMD=lucebox branch. + argv+=("$CONTAINER_NAME" /opt/lucebox-hub/server/scripts/entrypoint.sh lucebox "$@") + exec "${argv[@]}" +} + +# Decide whether a given (subcommand, argv) pair is safe to run via +# `docker exec` into the live container. Returns 0 (yes, prefer exec) or 1 +# (no, must use docker run / host-side). +# +# The safe-to-exec set is exactly the steady-state / read-only / hits-the- +# running-server subcommands. Anything that restarts the service, mutates +# images, or is itself the long-running service must stay on cmd_in_container. +# +_lucebox_prefer_exec() { + local cmd="$1"; shift + case "$cmd" in + config|models|check|print-run|print-serve-argv) + return 0 + ;; + *) + return 1 + ;; + esac +} + +# Top-level routing for the in-container Python CLI. Picks between exec +# (cheap, shares the live server's namespace) and run (cold start, isolated). +# +# Decision tree: +# 1. LUCEBOX_NO_EXEC=1 / --no-exec was set → always run, never exec. +# Useful for debugging the wrapper or when the in-container Python is +# stale relative to the image. +# 2. cmd is not in the prefer-exec list → run (sweep, service mutators). +# 3. container is running → exec (the fast path, hits the live server). +# 4. container is not running → run (fall back so first-run / pre-install +# flows still work without a live service). +cmd_route_to_container() { + local cmd="$1"; shift + if [ "${LUCEBOX_NO_EXEC:-0}" = "1" ]; then + cmd_in_container "$cmd" "$@" + return + fi + if _lucebox_prefer_exec "$cmd" "$@" && _lucebox_container_running; then + cmd_exec_in_container "$cmd" "$@" + return + fi + cmd_in_container "$cmd" "$@" +} + +usage() { + cat < print shell completion script (bash / zsh / fish) + models list / download / activate model presets + config read / write keys in .lucebox/config.toml + print-run print the docker-run command for the server + +Misc: + help, --help, -h this message + version, --version print version + +Environment overrides: + LUCEBOX_IMAGE image name without tag (default: ghcr.io/luce-org/lucebox-hub) + LUCEBOX_VARIANT image tag to pull/run (default: cuda12) + LUCEBOX_PORT host port for the server (default: 8080) + LUCEBOX_CONTAINER server container name (default: lucebox) + LUCEBOX_MODELS host model directory (default: \$XDG_DATA_HOME/lucebox/models + LUCEBOX_NO_EXEC=1 force docker-run for in-container subcommands even + when the container is up (equivalent to --no-exec) + HF_TOKEN propagated to \`models download\` for gated HF repos + +Container routing: + When the long-running '$CONTAINER_NAME' container is up, steady-state + subcommands (config, models, check, print-run, print-serve-argv) + 'docker exec' into it instead of starting a fresh container. This avoids + the ~1-3s docker-run cold-start AND shares the live server's network + namespace so localhost:\$LUCEBOX_PORT reaches the server. Service-restarting + commands (serve, pull, update, install, etc.) stay on the host-side / + docker-run path. Pass --no-exec (or LUCEBOX_NO_EXEC=1) to force docker-run. +EOF +} + +# ── dispatch ────────────────────────────────────────────────────────────── + +main() { + # Global flag pass: `--no-exec` anywhere before the subcommand forces the + # docker-run path even if the container is up. Equivalent to + # `LUCEBOX_NO_EXEC=1 lucebox ...`. We pop it out of argv up-front so the + # rest of dispatch doesn't have to know about it. + local args=() + while [ $# -gt 0 ]; do + case "$1" in + --no-exec) export LUCEBOX_NO_EXEC=1; shift ;; + *) args+=("$1"); shift ;; + esac + done + set -- "${args[@]}" + + local cmd="${1:-help}" + [ $# -gt 0 ] && shift + case "$cmd" in + # Systemd surface + install) cmd_systemd_install "$@" ;; + uninstall) cmd_systemd_uninstall "$@" ;; + start|stop|restart|enable|disable|status) + cmd_systemctl_passthrough "$cmd" "$@" ;; + logs) cmd_logs "$@" ;; + + # Direct server + serve) cmd_serve "$@" ;; + pull) cmd_pull "$@" ;; + + # Self-update — re-runs the bootstrap installer against the channel + # this script was installed from (LUCEBOX_INSTALLED_FROM). + update) cmd_update "$@" ;; + + # Host-only readiness check — pure shell, never enters the container. + check) cmd_check "$@" ;; + + # Shell completion — print a script the user sources into their rc + # file. Bash and zsh share the bash-style emitter (zsh users add a + # `bashcompinit; complete` shim); fish is native. + completion) cmd_completion "$@" ;; + + # Help / version + help|--help|-h) usage ;; + version|--version) printf '%s\n' "$VERSION" ;; + + # Everything else → in-container Python CLI. cmd_route_to_container + # picks between `docker exec` into the live container (cheap, shares + # the running server's network namespace) and `docker run` (cold, + # isolated) based on container state + the safe-to-exec command set. + *) cmd_route_to_container "$cmd" "$@" ;; + esac +} + +main "$@" diff --git a/lucebox/.gitignore b/lucebox/.gitignore new file mode 100644 index 000000000..15f95f0e7 --- /dev/null +++ b/lucebox/.gitignore @@ -0,0 +1,3 @@ + +# Generated by hatch-vcs at build time from git tags. +src/lucebox/_version.py diff --git a/lucebox/README.md b/lucebox/README.md new file mode 100644 index 000000000..747a49eef --- /dev/null +++ b/lucebox/README.md @@ -0,0 +1,18 @@ +# lucebox — host CLI for the lucebox-hub container + +This package ships *inside* the `ghcr.io/luce-org/lucebox-hub` Docker image +and is invoked from the host via the [`lucebox.sh`](../lucebox.sh) wrapper: + + lucebox.sh check # `docker run … lucebox check` + lucebox.sh config get + lucebox.sh print-run + +The wrapper is the only thing that runs on the host; everything else (host +checks, TOML config, docker daemon calls, model download) is Python in the +container. Host facts (driver, GPU, RAM, VRAM, systemd availability) are +passed in via `LUCEBOX_HOST_*` environment variables so the Python side +doesn't reprobe. The autotune sweep, profiling, and agent-client launchers +land in follow-up PRs. + +Subcommands are defined in [`lucebox/cli.py`](src/lucebox/cli.py). See the +top-level [README.md](../README.md) for the user-facing flow. diff --git a/lucebox/pyproject.toml b/lucebox/pyproject.toml new file mode 100644 index 000000000..5277b268d --- /dev/null +++ b/lucebox/pyproject.toml @@ -0,0 +1,54 @@ +[project] +name = "lucebox" +# Version is derived from git tags via hatch-vcs (see [tool.hatch.version] +# below). Tag `lucebox-v0.2.1` → release version `0.2.1`. Commits past a +# tag get a `.devN+g` suffix so dev installs are visibly distinct +# from releases. Single source of truth: the git tag. +dynamic = ["version"] +description = "Host-side CLI for the lucebox-hub container: launch, config, model download" +readme = "README.md" +requires-python = ">=3.11" +authors = [{ name = "Lucebox" }] +license = { text = "Apache-2.0" } + +# Kept intentionally narrow. typer pulls click+rich; tomli-w gives us TOML +# writes (stdlib tomllib only reads). httpx for the smoke + readiness probes. +# huggingface_hub for download-models — used directly (not via subprocess) +# so we can drive a Rich progress bar + verify sha256 against the repo +# metadata before re-fetching multi-GB GGUFs. +dependencies = [ + "typer>=0.12", + "rich>=13", + "httpx>=0.27", + "tomli-w>=1.0", + "huggingface_hub>=0.27", + # luce-bench is consumed lazily by the autotune sweep scorer + # (agent_replay_pass_rate in sweep.py does a function-local + # `from lucebench.areas.agent_recorded import ...` wrapped in try/except). + # It's deliberately NOT a hard dep here because the workspace can't lock + # against it until #337 (luce-bench in-tree) lands. Install with + # `uv pip install luce-bench` on the host running the scorer. +] + +[project.scripts] +lucebox = "lucebox.cli:app" + +[build-system] +requires = ["hatchling", "hatch-vcs"] +build-backend = "hatchling.build" + +[tool.hatch.version] +source = "vcs" +# Untagged checkouts (e.g. fresh clone before tagging lucebox-v0.2.1) +# resolve to this rather than 0.0.0.dev0. +fallback-version = "0.2.1.dev0" +raw-options.tag_regex = '''^lucebox-v(?P\d+\.\d+\.\d+)$''' + +[tool.hatch.build.hooks.vcs] +# Build hook writes the resolved version into src/lucebox/_version.py +# so `__init__.py` can `from lucebox._version import __version__`. +# Generated file — see lucebox/.gitignore. +version-file = "src/lucebox/_version.py" + +[tool.hatch.build.targets.wheel] +packages = ["src/lucebox"] diff --git a/lucebox/src/lucebox/__init__.py b/lucebox/src/lucebox/__init__.py new file mode 100644 index 000000000..8ca821024 --- /dev/null +++ b/lucebox/src/lucebox/__init__.py @@ -0,0 +1,16 @@ +"""lucebox — host-side CLI for the lucebox-hub container. + +Runs inside the container; the host wrapper at ../lucebox.sh handles `docker +run` plumbing and systemd integration. This package owns: TOML config, the +host-derived DFLASH_* serve heuristic, docker daemon calls (via the mounted +socket), and model download. The empirical autotune sweep, profiling, and +agent-client launchers land in follow-up PRs. +""" + +# Version is generated by hatch-vcs at build time into _version.py. +# Fresh source-tree checkouts before any build will not yet have the +# file — fall back to a dev marker so imports don't break. +try: + from lucebox._version import __version__ +except ImportError: + __version__ = "0.0.0.dev0+unbuilt" diff --git a/lucebox/src/lucebox/__main__.py b/lucebox/src/lucebox/__main__.py new file mode 100644 index 000000000..128e2ca87 --- /dev/null +++ b/lucebox/src/lucebox/__main__.py @@ -0,0 +1,6 @@ +"""Entry point for `python -m lucebox`.""" + +from lucebox.cli import app + +if __name__ == "__main__": + app() diff --git a/lucebox/src/lucebox/autotune.py b/lucebox/src/lucebox/autotune.py new file mode 100644 index 000000000..51b4f8424 --- /dev/null +++ b/lucebox/src/lucebox/autotune.py @@ -0,0 +1,80 @@ +"""Heuristic autotune: VRAM tier → DflashRuntime defaults. + +The recommended runtime is computed from HostFacts (VRAM, is_wsl) — stateless: +it takes HostFacts in and returns a fresh DflashRuntime. ``config.live_config`` +applies it so ``lucebox print-serve-argv`` / ``docker_run`` bake conservative +DFLASH_* defaults into the serve command for the detected VRAM tier. + +The empirical sweep + per-workload profiles (``lucebox autotune --sweep``) +live in a follow-up PR; this module keeps only the host-derived heuristic +that the serve path depends on. +""" + +from __future__ import annotations + +from lucebox.types import DflashRuntime, HostFacts + + +def runtime_from_host(host: HostFacts) -> DflashRuntime: + """Pick a conservative DflashRuntime that 'should work' on this VRAM tier. + + Tiers (NVIDIA, baseline = Qwen3.6-27B Q4_K_M ~18 GB total): + <12 GB — too small for 27B; pick min ctx as a floor so a fallback + start at least gets an error from the daemon rather than + a silent OOM. + 12-21 — fits but tight; cap ctx. + 22-31 — 24 GB-class consumer flagships (3090/4090/5090/5090-Laptop). + 98 K with tq3_0 KV (~2 GB KV + ~18 GB model ≈ 20 GB). + Confirmed on bragi (RTX 5090 Laptop, 23 GB VRAM) 2026-05-31. + 32-47 — RTX 6000 Ada / A100 40 GB. Full 128 K. + ≥48 — A100 80 GB / H100 / RTX 6000 Pro. Full 128 K. + + Prefix cache remains an explicit sweep tunable, but the automatic + baseline keeps it off because tool prompts currently exercise a daemon + snapshot path that is not reliable with prefix slots enabled. + Empirically confirmed on bragi 2026-05-31: prefix_cache_slots=32 + caused -19pp regression on agent_recorded (23.1% vs 42.3% baseline). + 5 previously-passing cases regressed; 0 new cases unlocked. See + docs/experiments/qwen3.6-27b-prefix-cache-regression-bragi-2026-05-31.md. + + On `lazy`: the C++ server requires `--prefill-drafter` (and `--draft`) + to be set for `--lazy-draft` to take effect, and silently ignores it + otherwise (`--lazy-draft ignored: requires both --prefill-drafter and + --draft`). Since the heuristic path does NOT set `prefill_drafter`, + we default `lazy=False` here — "what we say" matches "what runs". + Users who explicitly opt in via config.toml will be warned at server + startup that the flag is being dropped (see entrypoint.sh). + """ + if host.vram_gb <= 0: + return DflashRuntime() # no VRAM signal — stick with class defaults + + if host.vram_gb < 12: + return DflashRuntime(max_ctx=4096) + if host.vram_gb < 22: + return DflashRuntime(max_ctx=32768) + if host.vram_gb < 32: + # 22-31 GB cards. tq3_0 KV is required at 98K: model (~18-19 GB) + + # q8_0 KV at 98K (~5-6 GB) = 24-25 GB → OOM, while tq3_0 KV (~2 GB) + # leaves ~3 GB headroom. Confirmed on bragi (RTX 5090 Laptop, 23 GB + # VRAM) 2026-05-30 — q8_0 timed out on every 98K cell; all tq3_0 cells + # passed. Preset-size-aware capping (large models → 32K) lives with the + # autotune sweep in a follow-up PR. + if host.is_wsl: + # Bumped from max_ctx=65536 → 98304 on 2026-05-30 after the + # coding-agent-loop sweep on sindri proved 98K serves real + # 90K-token agentic prompts with ~3 GB VRAM headroom and no + # CUDA VMM failures. See + # docs/experiments/gemma4-26b-coding-agent-loop-sweep-2026-05-30.md. + # The original 65K cap cited unverified VMM failures — + # bisect history showed no commit reproducing them. + return DflashRuntime( + budget=16, max_ctx=98304, + cache_type_k="tq3_0", cache_type_v="tq3_0", + ) + return DflashRuntime( + max_ctx=98304, + cache_type_k="tq3_0", cache_type_v="tq3_0", + ) + if host.vram_gb < 48: + return DflashRuntime(max_ctx=131072) + return DflashRuntime(max_ctx=131072) diff --git a/lucebox/src/lucebox/cli.py b/lucebox/src/lucebox/cli.py new file mode 100644 index 000000000..87cc29366 --- /dev/null +++ b/lucebox/src/lucebox/cli.py @@ -0,0 +1,356 @@ +"""Typer app — the user-facing subcommands. + +Layout follows the host wrapper's dispatch table. Anything `lucebox` +doesn't intercept (everything outside the systemd surface) ends up here. + +Subcommand inventory: + check — readiness report + config get/set/unset — read / write a single key in config.toml + pull — docker pull the cuda12 image + print-run — emit the docker-run command for the server + print-serve-argv — same, raw argv lines (consumed by `lucebox serve`) + models — list / download presets, activate one +""" + +from __future__ import annotations + +import os +import sys +from dataclasses import replace +from pathlib import Path +from typing import Annotated + +import typer +from rich.console import Console +from rich.table import Table + +import lucebox.config as config_mod +import lucebox.docker_run as docker_run +import lucebox.download as download_mod +import lucebox.host_check as host_check +from lucebox import __version__ +from lucebox.config import config_get, config_set, config_unset, live_config +from lucebox.host_facts import from_env + +app = typer.Typer( + name="lucebox", + help="Host CLI for the lucebox-hub container. Invoked by lucebox.sh.", + no_args_is_help=True, + add_completion=False, +) +console = Console() + + +# ── helpers ──────────────────────────────────────────────────────────────── + + +def _load_or_build() -> config_mod.Config: # type: ignore[name-defined] + """env > config.toml > dataclass defaults — the canonical precedence. + + Without the env-overlay step below, `config_mod.load()` returned the + persisted config verbatim and `LUCEBOX_IMAGE` / `LUCEBOX_VARIANT` / + `LUCEBOX_PORT` / `LUCEBOX_CONTAINER` / `LUCEBOX_MODELS` from the + systemd unit's `Environment=` (or any one-shot shell export) were + silently dropped. That contradicted the precedence lucebox.sh + documents and applies — and bit sindri when its config.toml had + `[image]` without `registry`, so the dataclass default + `ghcr.io/luce-org/lucebox-hub` won over the unit's + `LUCEBOX_IMAGE=ghcr.io/easel/lucebox-hub`. + + Fix: overlay env on top of the loaded config (or the live_config + fallback when config.toml is absent). Only the five top-level + scalars have env hooks — dflash/host/model don't, by design. + """ + cfg = config_mod.load() + if cfg is None: + cfg = live_config() + # Overlay live host facts. When ``config.toml`` exists without a + # ``[host]`` block (the common case — operators don't hand-edit + # host facts), ``cfg.host`` defaults to a zero-filled ``HostFacts`` + # and the DFLASH_* serve heuristic silently falls through to the + # "no VRAM signal" path. Re-probe from env so the wrapper-exported + # LUCEBOX_HOST_* facts always win over the persisted (possibly + # absent) snapshot. + live_host = from_env() + host = live_host if live_host.vram_gb > 0 or live_host.nproc > 0 else cfg.host + return replace( + cfg, + variant=os.environ.get("LUCEBOX_VARIANT", cfg.variant), + image=os.environ.get("LUCEBOX_IMAGE", cfg.image), + container_name=os.environ.get("LUCEBOX_CONTAINER", cfg.container_name), + port=int(os.environ.get("LUCEBOX_PORT", str(cfg.port))), + models_dir=Path(os.environ.get("LUCEBOX_MODELS", str(cfg.models_dir))), + host=host, + ) + + +# ── subcommands ──────────────────────────────────────────────────────────── + + +@app.command() +def check() -> None: + """Print a readiness report (driver, docker, CTK, RAM, VRAM, systemd).""" + host = from_env() + results = host_check.run_checks(host) + worst = host_check.render(console, host, results) + if worst == "fail": + raise typer.Exit(code=1) + + +@app.command() +def pull() -> None: + """`docker pull` the image variant from config.toml.""" + cfg = _load_or_build() + tag = f"{cfg.image}:{cfg.variant}" + console.print(f"[bold]Pulling {tag}[/bold] (~14 GB; takes a while)…") + rc = docker_run.docker_pull(tag) + if rc != 0: + raise typer.Exit(code=rc) + + +@app.command("print-run") +def print_run() -> None: + """Print the docker-run command for the server (copy-pasteable).""" + cfg = _load_or_build() + spec = docker_run.server_run_spec(cfg) + print(spec.printable()) + + +@app.command("print-serve-argv") +def print_serve_argv() -> None: + """Emit the server docker-run argv, one token per line. + + Consumed by lucebox.sh's `serve` subcommand and the systemd unit. Kept as + a separate command from `print-run` so the bash side has a guaranteed + machine-readable contract that's independent of the pretty formatter. + """ + cfg = _load_or_build() + spec = docker_run.server_run_spec(cfg) + for tok in spec.argv(): + print(tok) + + +# ── config sub-app ───────────────────────────────────────────────────────── + + +config_app = typer.Typer(no_args_is_help=True, help="Read/write keys in config.toml.") +app.add_typer(config_app, name="config") + + +@config_app.command("get") +def config_get_cmd( + key: Annotated[str, typer.Argument(help="Dotted key (omit to list every key).")] = "", +) -> None: + """Print a single key (or every reachable key) with its origin annotation.""" + try: + entries = config_get(key or None) + except KeyError as exc: + console.print(f"[red]{exc}[/red]") + raise typer.Exit(code=2) from exc + for k, (value, origin) in entries.items(): + console.print(f"{k} = {value!r} ([dim]from {origin}[/dim])") + + +@config_app.command("set") +def config_set_cmd( + kv: Annotated[str, typer.Argument(help='"key=value" pair (e.g. "model.preset=qwen3.6-27b")')], +) -> None: + """Set one dotted key. Auto-creates config.toml when missing. + + Only the named key is written — other on-disk keys are preserved + untouched, unset keys stay implicit. Use `lucebox config unset` to + remove a key (next read falls back to the live default). + """ + if "=" not in kv: + console.print("[red]argument must be key=value[/red]") + raise typer.Exit(code=2) + key, _, value = kv.partition("=") + key = key.strip() + value = value.strip() + try: + config_set(key, value) + except (KeyError, ValueError) as exc: + console.print(f"[red]{exc}[/red]") + raise typer.Exit(code=2) from exc + console.print(f"[green]Set[/green] {key} = {value}") + + +@config_app.command("unset") +def config_unset_cmd( + key: Annotated[str, typer.Argument(help="Dotted key to remove from config.toml.")], +) -> None: + """Remove a key from config.toml. Next read uses the live default.""" + try: + changed = config_unset(key) + except KeyError as exc: + console.print(f"[red]{exc}[/red]") + raise typer.Exit(code=2) from exc + if changed: + console.print(f"[green]Unset[/green] {key}") + else: + console.print(f"[dim]{key} was not in config.toml; nothing to do[/dim]") + + +# ── models sub-app ───────────────────────────────────────────────────────── + + +models_app = typer.Typer( + no_args_is_help=False, help="Manage local model presets (list, download, activate)." +) +app.add_typer(models_app, name="models") + + +def _print_installed_presets() -> None: + cfg = _load_or_build() + installed = download_mod.installed_presets(cfg) + active = cfg.model.preset + console.print(f"Models dir: [bold]{cfg.models_dir}[/bold]") + if not installed: + console.print("[dim]No presets installed yet — try `lucebox models download`.[/dim]") + return + table = Table() + table.add_column("preset") + table.add_column("status") + table.add_column("size (GB)") + for pres in installed: + marker = "* " if pres.name == active else " " + size_gb = download_mod.installed_size_gb(cfg, pres) + table.add_row(f"{marker}{pres.name}", "installed", f"{size_gb:.1f}") + console.print(table) + total = sum(download_mod.installed_size_gb(cfg, p) for p in installed) + console.print(f"[dim]Total disk usage: {total:.1f} GB[/dim]") + + +@models_app.callback(invoke_without_command=True) +def models_default(ctx: typer.Context) -> None: + """Default action: list installed presets, mark active with `*`.""" + if ctx.invoked_subcommand is None: + _print_installed_presets() + + +@models_app.command("list") +def models_list() -> None: + """Show every registered preset (installed or not) with status + size.""" + cfg = _load_or_build() + active = cfg.model.preset + table = Table() + table.add_column("preset") + table.add_column("status") + table.add_column("size (GB)") + table.add_column("description") + for name in sorted(download_mod.PRESETS): + pres = download_mod.PRESETS[name] + marker = "* " if name == active else " " + status = download_mod.installed_status(cfg, pres) + size = download_mod.installed_size_gb(cfg, pres) + size_text = f"{size:.1f}" if size > 0 else f"~{pres.approx_total_gb}*" + table.add_row(f"{marker}{name}", status, size_text, pres.description or "") + console.print(table) + + +@models_app.command("download") +def models_download( + preset: Annotated[str, typer.Argument(help="Preset name (empty = recommend)")] = "", + activate: Annotated[ + bool, typer.Option("--activate", help="Also set as active preset (model.preset).") + ] = False, +) -> None: + """Fetch a preset's GGUFs into the models dir. + + With no argument and no preset configured, recommends one for this + host's VRAM tier and auto-activates it (the first-install path). + Otherwise the named preset is downloaded; pass ``--activate`` to + also flip `model.preset` to it. + """ + cfg = _load_or_build() + if not preset: + if cfg.model.preset: + console.print( + "[yellow]No preset specified and one is already active. " + "Pass an explicit preset name (or use --activate to switch).[/yellow]" + ) + raise typer.Exit(code=2) + recommended = download_mod.recommend_preset(cfg.host) + if recommended is None: + console.print( + "[red]Cannot recommend a preset for this host. " + "Run `lucebox models list` and pick one explicitly.[/red]" + ) + raise typer.Exit(code=2) + preset = recommended + activate = True + console.print( + f"[bold]Recommended preset: {preset}[/bold] " + "(no preset configured; auto-activating after download)" + ) + + try: + pres = download_mod.resolve_preset(preset) + except KeyError as exc: + console.print(f"[red]{exc}[/red]") + raise typer.Exit(code=2) from exc + + current = download_mod.status(cfg, pres) + console.print(f"Models dir: [bold]{cfg.models_dir}[/bold]") + console.print(f"Preset: [bold]{pres.name}[/bold]") + console.print( + f" target ({pres.target_repo}/{pres.target_file}):" + f" {'present' if current['target_present'] else 'will download'}" + ) + if pres.has_draft: + console.print( + f" draft ({pres.draft_repo}/{pres.draft_file}):" + f" {'present' if current['draft_present'] else 'will download'}" + ) + else: + console.print(" draft [dim](none — target-only preset)[/dim]") + + if current["target_present"] and current["draft_present"]: + console.print("[green]Already present.[/green]") + else: + console.print(f"[bold]Downloading[/bold] (~{pres.approx_total_gb} GB total)…") + rc = download_mod.download_preset(cfg, pres) + if rc != 0: + raise typer.Exit(code=rc) + console.print("[green]Done.[/green]") + + if activate: + config_set("model.preset", preset) + if pres.target_file: + config_set("model.target_file", pres.target_file) + if pres.has_draft and pres.draft_file: + config_set("model.draft_file", pres.draft_file) + else: + # Drop any stale draft_file from a previous activation; the + # active preset has no draft. + config_unset("model.draft_file") + console.print(f"[green]Activated:[/green] model.preset = {preset}") + # First-time setup: bake the VRAM-tier DFLASH_* heuristic into + # config.toml so `lucebox serve` is auto-tuned to this host instead + # of falling back to the conservative class defaults. Never clobbers + # an existing [dflash] section. + if config_mod.seed_dflash_from_host(cfg.host): + max_ctx = config_get("dflash.max_ctx")["dflash.max_ctx"][0] + console.print( + f"[green]Auto-tuned:[/green] VRAM-tier DFLASH_* defaults " + f"(max_ctx={max_ctx}) written to config.toml" + ) + + +@app.command() +def version() -> None: + """Print lucebox version.""" + print(__version__) + + +def main() -> None: + """Module entrypoint — `python -m lucebox`.""" + try: + app() + except KeyboardInterrupt: + console.print("\n[dim]interrupted[/dim]") + sys.exit(130) + + +if __name__ == "__main__": + main() diff --git a/lucebox/src/lucebox/config.py b/lucebox/src/lucebox/config.py new file mode 100644 index 000000000..b3f63eb0d --- /dev/null +++ b/lucebox/src/lucebox/config.py @@ -0,0 +1,482 @@ +"""Sparse TOML persistence for .lucebox/config.toml. + +Single source of truth for user-overridden configuration. We track which +dotted keys were explicitly set by the user (or by commands acting on +their behalf) and serialize ONLY those keys back to disk — defaults +stay implicit, so `config.toml` reads like a diff against live defaults +and upgrades that add new fields don't gratuitously rewrite every file. + +The dotted-key surface area is small and flat: + model.preset, model.target_file, model.draft_file + port, models_dir, variant, image, container_name + dflash. for each of the 11 DflashRuntime knobs + think_max + +Load resolves the TOML file → ``Config`` object, with anything absent +filled from ``Config()`` defaults. Save writes back only the keys that +appear in the TOML doc (tracked on ``Config._user_set``). The TOML doc +itself is a plain ``dict[str, Any]`` carrying only the set keys. +""" + +from __future__ import annotations + +import os +import re +import tomllib +from collections.abc import Callable +from dataclasses import asdict, replace +from datetime import UTC +from pathlib import Path +from typing import Any + +import tomli_w + +from lucebox.types import ( + Config, + DflashRuntime, + HostFacts, + ModelMeta, + Variant, + default_models_dir, +) + + +def default_config_path() -> Path: + """Where .lucebox/config.toml lives. + + Convention: under $LUCEBOX_HOME if set, otherwise $HOME/.lucebox. Lives in + the bind-mounted host home dir so the config survives container teardown + and is editable from the host. + """ + base = os.environ.get("LUCEBOX_HOME") + if base: + return Path(base) / "config.toml" + return Path.home() / ".lucebox" / "config.toml" + + +# ── dotted-key registry ──────────────────────────────────────────────────── + +def _cast_prefill_mode(v: Any) -> str: + s = str(v) + if s not in {"off", "auto", "always"}: + raise ValueError(f"prefill_mode must be off/auto/always, got {s!r}") + return s + + +def _cast_bool(v: Any) -> bool: + """Strict-ish boolean coercion for config values. + + - Native booleans pass through. + - Strings: 1/true/yes/on → True; 0/false/no/off/"" → False (case-insensitive). + - Anything else raises ``ValueError`` rather than silently coercing, + because that's what bit ``dflash.debug_thinking_logits`` — the + built-in ``bool`` caster turned ``"false"`` into ``True``. + """ + if isinstance(v, bool): + return v + if isinstance(v, str): + s = v.strip().lower() + if s in ("1", "true", "yes", "on"): + return True + if s in ("0", "false", "no", "off", ""): + return False + raise ValueError(f"cannot parse boolean: {v!r}") + if isinstance(v, int): + return bool(v) + raise ValueError(f"cannot parse boolean: {v!r}") + + +# Each entry: dotted-key → (toml_path, type_caster, default_getter). +# ``toml_path`` is the (section, field) pair on disk; ``"_root"`` means the +# key lives at the top level (no [section]). ``default_getter`` returns the +# in-memory default so ``config get`` can annotate origin. +KEY_REGISTRY: dict[str, tuple[tuple[str, str], Callable[[Any], Any]]] = { + "variant": (("image", "variant"), str), + "image": (("image", "registry"), str), + "container_name": (("runtime", "container_name"), str), + "port": (("runtime", "port"), int), + "models_dir": (("paths", "models"), str), + "model.preset": (("model", "preset"), str), + "model.target_file": (("model", "target_file"), str), + "model.draft_file": (("model", "draft_file"), str), + "dflash.budget": (("dflash", "budget"), int), + "dflash.max_ctx": (("dflash", "max_ctx"), int), + "dflash.lazy": (("dflash", "lazy"), _cast_bool), + "dflash.prefix_cache_slots": (("dflash", "prefix_cache_slots"), int), + "dflash.prefill_cache_slots": (("dflash", "prefill_cache_slots"), int), + "dflash.cache_type_k": (("dflash", "cache_type_k"), str), + "dflash.cache_type_v": (("dflash", "cache_type_v"), str), + "dflash.prefill_mode": (("dflash", "prefill_mode"), _cast_prefill_mode), + "dflash.prefill_keep_ratio": (("dflash", "prefill_keep_ratio"), float), + "dflash.prefill_threshold": (("dflash", "prefill_threshold"), int), + "dflash.prefill_drafter": (("dflash", "prefill_drafter"), str), + "dflash.think_max": (("dflash", "think_max"), int), + "dflash.fa_window": (("dflash", "fa_window"), int), + "dflash.think_soft_close_min_ratio": ( + ("dflash", "think_soft_close_min_ratio"), float), + "dflash.debug_thinking_logits": ( + ("dflash", "debug_thinking_logits"), _cast_bool), +} + + +def _doc_get(doc: dict[str, Any], section: str, field: str) -> Any: + if section == "_root": + return doc.get(field) + sub = doc.get(section) + if isinstance(sub, dict): + return sub.get(field) + return None + + +def _doc_set(doc: dict[str, Any], section: str, field: str, value: Any) -> None: + if section == "_root": + doc[field] = value + return + doc.setdefault(section, {})[field] = value + + +def _doc_unset(doc: dict[str, Any], section: str, field: str) -> bool: + """Remove a dotted key from the doc. Returns True iff something was removed.""" + if section == "_root": + if field in doc: + del doc[field] + return True + return False + sub = doc.get(section) + if isinstance(sub, dict) and field in sub: + del sub[field] + if not sub: + del doc[section] + return True + return False + + +# ── load ─────────────────────────────────────────────────────────────────── + + +def load(path: Path | None = None) -> Config | None: + """Load config.toml, or return None if missing. + + If a legacy `.env` sits next to it (or in place of it), migrate that + first and write back as TOML. + """ + path = path or default_config_path() + if path.exists(): + return _load_toml(path) + + legacy = path.with_suffix(".env") + if legacy.exists(): + cfg, doc = _load_legacy_env(legacy) + save(cfg, path, doc=doc) + return cfg + + return None + + +def _load_toml(path: Path) -> Config: + raw = tomllib.loads(path.read_text()) + return _from_dict(raw) + + +def load_doc(path: Path | None = None) -> dict[str, Any]: + """Return the raw TOML doc (a dict). Empty when no file or empty file.""" + path = path or default_config_path() + if not path.exists(): + return {} + return tomllib.loads(path.read_text()) + + +_LEGACY_KEY_MAP: dict[str, tuple[str, str, Callable[[str], Any]]] = { + "DFLASH_BUDGET": ("dflash", "budget", int), + "DFLASH_MAX_CTX": ("dflash", "max_ctx", int), + "DFLASH_LAZY": ("dflash", "lazy", + lambda v: str(v).strip().lower() in ("1", "true", "yes", "on")), + "DFLASH_PREFIX_CACHE_SLOTS": ("dflash", "prefix_cache_slots", int), + "DFLASH_PORT": ("runtime", "port", int), + "LUCEBOX_VARIANT": ("image", "variant", str), + "LUCEBOX_IMAGE": ("image", "registry", str), + "LUCEBOX_MODELS": ("paths", "models", str), +} + + +def _load_legacy_env(path: Path) -> tuple[Config, dict[str, Any]]: + """Best-effort migration from the bash-era .lucebox/config.env.""" + raw: dict[str, Any] = {} + line_re = re.compile(r"^([A-Z_][A-Z0-9_]*)=(.*)$") + for line in path.read_text().splitlines(): + line = line.strip() + if not line or line.startswith("#"): + continue + m = line_re.match(line) + if not m: + continue + key, val = m.group(1), m.group(2).strip().strip('"').strip("'") + if key not in _LEGACY_KEY_MAP: + continue + section, field, cast_fn = _LEGACY_KEY_MAP[key] + try: + raw.setdefault(section, {})[field] = cast_fn(val) + except (TypeError, ValueError): + continue + return _from_dict(raw), raw + + +def _from_dict(raw: dict[str, Any]) -> Config: + img = raw.get("image", {}) + variant: Variant = str(img.get("variant", "cuda12")) + registry = img.get("registry", "ghcr.io/luce-org/lucebox-hub") + + runtime = raw.get("runtime", {}) + port = int(runtime.get("port", 8080)) + container_name = str(runtime.get("container_name", "lucebox")) + + paths = raw.get("paths", {}) + models_dir = Path(paths.get("models", str(default_models_dir()))) + + df = raw.get("dflash", {}) + dflash = DflashRuntime( + budget=int(df.get("budget", 22)), + max_ctx=int(df.get("max_ctx", 16384)), + lazy=bool(df.get("lazy", False)), + prefix_cache_slots=int(df.get("prefix_cache_slots", 0)), + prefill_cache_slots=int(df.get("prefill_cache_slots", 0)), + cache_type_k=str(df.get("cache_type_k", "")), + cache_type_v=str(df.get("cache_type_v", "")), + prefill_mode=df.get("prefill_mode", "off"), + prefill_keep_ratio=float(df.get("prefill_keep_ratio", 0.05)), + prefill_threshold=int(df.get("prefill_threshold", 32000)), + prefill_drafter=str(df.get("prefill_drafter", "")), + think_max=int(df.get("think_max", 15488)), + fa_window=int(df.get("fa_window", 0)), + think_soft_close_min_ratio=float( + df.get("think_soft_close_min_ratio", 0.0)), + debug_thinking_logits=bool(df.get("debug_thinking_logits", False)), + ) + + host_raw = raw.get("host", {}) + host = HostFacts( + nproc=int(host_raw.get("nproc", 0)), + ram_gb=int(host_raw.get("ram_gb", 0)), + gpu_vendor=host_raw.get("gpu_vendor", "none"), + gpu_name=str(host_raw.get("gpu_name", "")), + gpu_count=int(host_raw.get("gpu_count", 0)), + vram_gb=int(host_raw.get("vram_gb", 0)), + gpu_sm=str(host_raw.get("gpu_sm", "")), + driver_version=str(host_raw.get("driver_version", "")), + driver_major=int(host_raw.get("driver_major", 0)), + has_systemd=bool(host_raw.get("has_systemd", False)), + is_wsl=bool(host_raw.get("is_wsl", False)), + has_docker=bool(host_raw.get("has_docker", False)), + docker_version=str(host_raw.get("docker_version", "")), + ctk=host_raw.get("ctk", "none"), + ) + + # `[model]` is optional — legacy configs (pre-multi-model) carry no + # such section and we want them to keep working unchanged. If + # `preset` is set but `target_file` / `draft_file` isn't, derive + # them from the registry so users only have to write one key. + mdl = raw.get("model", {}) + preset_name = str(mdl.get("preset", "")) + target_file = str(mdl.get("target_file", "")) + draft_file = str(mdl.get("draft_file", "")) + if preset_name and (not target_file or not draft_file): + from lucebox.download import PRESETS + + if preset_name in PRESETS: + pres = PRESETS[preset_name] + if not target_file: + target_file = pres.target_file + if not draft_file and pres.has_draft and pres.draft_file: + draft_file = pres.draft_file + model = ModelMeta(preset=preset_name, target_file=target_file, draft_file=draft_file) + + return Config( + variant=variant, + image=registry, + container_name=container_name, + port=port, + models_dir=models_dir, + dflash=dflash, + host=host, + model=model, + ) + + +# ── save ─────────────────────────────────────────────────────────────────── + + +def _atomic_write_doc(path: Path, doc: dict[str, Any]) -> None: + """Serialize ``doc`` to TOML and write it to ``path`` atomically. + + Write to a sibling ``.toml.tmp`` then ``replace`` so a crash mid-write + never leaves a truncated config.toml. Caller ensures ``path.parent`` exists. + """ + tmp = path.with_suffix(".toml.tmp") + tmp.write_bytes(tomli_w.dumps(doc).encode("utf-8")) + tmp.replace(path) + + +def save(cfg: Config, path: Path | None = None, *, doc: dict[str, Any] | None = None) -> Path: + """Persist a Config to ``path``. Only keys present in ``doc`` are written. + + ``doc`` is the raw TOML mapping returned by ``load_doc`` — it carries + exactly the keys the user (or a command on their behalf) has set. When + ``doc=None`` and the file exists we re-use the on-disk doc; when both + are absent we write an empty file. + """ + path = path or default_config_path() + path.parent.mkdir(parents=True, exist_ok=True) + if doc is None: + doc = load_doc(path) + _atomic_write_doc(path, doc) + # Silence unused-arg: cfg is the on-disk representation's source of + # truth for callers that want to round-trip through a Config object, + # but the sparse write never re-derives keys from it. + del cfg + return path + + +def seed_dflash_from_host(host: HostFacts, *, path: Path | None = None) -> bool: + """Persist the VRAM-tier DFLASH_* heuristic to config.toml on first setup. + + Returns True when it wrote. No-op (returns False) when a ``[dflash]`` + section already exists, so it never clobbers values a prior tune or the + user set. Called when a preset is first activated: without it a fresh + install serves at the conservative ``DflashRuntime`` class defaults + (``load()`` returns those for a config.toml that has no ``[dflash]``), + ignoring the host's VRAM tier. The ``live_config`` heuristic only fires + when there is no config.toml at all, which stops being true the moment a + model is activated — so the heuristic is persisted here instead. + """ + from datetime import datetime + + import lucebox.autotune as autotune_mod + + path = path or default_config_path() + doc = load_doc(path) + if "dflash" in doc: + return False + runtime = autotune_mod.runtime_from_host(host) + for field, value in asdict(runtime).items(): + _doc_set(doc, "dflash", field, _value_to_toml(value)) + _doc_set(doc, "autotune", "source", "heuristic") + _doc_set( + doc, "autotune", "timestamp", + datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ"), + ) + path.parent.mkdir(parents=True, exist_ok=True) + _atomic_write_doc(path, doc) + return True + + +# ── dotted-key API ───────────────────────────────────────────────────────── + + +def _value_to_toml(value: Any) -> Any: + """Make a Python value safe for tomli_w (no None, Path→str).""" + if isinstance(value, Path): + return str(value) + return value + + +def _live_default(key: str) -> Any: + """Return the in-memory default for ``key`` (from a fresh Config()).""" + cfg = Config() + section_field = KEY_REGISTRY[key][0] + section, field = section_field + if section == "image": + return {"variant": cfg.variant, "registry": cfg.image}[field] + if section == "runtime": + return {"port": cfg.port, "container_name": cfg.container_name}[field] + if section == "paths": + return str(cfg.models_dir) if field == "models" else None + if section == "dflash": + return getattr(cfg.dflash, field) + if section == "model": + return getattr(cfg.model, field) + return None + + +def config_set(key: str, value: Any, *, path: Path | None = None) -> None: + """Set one dotted key and write the file. Auto-creates a missing file.""" + if key not in KEY_REGISTRY: + raise KeyError(f"unknown config key {key!r}; known: {sorted(KEY_REGISTRY)}") + section_field, caster = KEY_REGISTRY[key] + section, field = section_field + try: + cast_value = caster(value) + except (TypeError, ValueError) as exc: + raise ValueError(f"cannot coerce {value!r} for {key}: {exc}") from exc + path = path or default_config_path() + doc = load_doc(path) if path.exists() else {} + _doc_set(doc, section, field, _value_to_toml(cast_value)) + path.parent.mkdir(parents=True, exist_ok=True) + _atomic_write_doc(path, doc) + + +def config_unset(key: str, *, path: Path | None = None) -> bool: + """Remove a dotted key from the file. Returns True if something changed.""" + if key not in KEY_REGISTRY: + raise KeyError(f"unknown config key {key!r}; known: {sorted(KEY_REGISTRY)}") + section_field, _ = KEY_REGISTRY[key] + section, field = section_field + path = path or default_config_path() + if not path.exists(): + return False + doc = load_doc(path) + changed = _doc_unset(doc, section, field) + if changed: + # Leave the file in place even when empty — `config set` will + # repopulate; deleting would surprise users who expect their + # config dir to exist. + _atomic_write_doc(path, doc) + return changed + + +def config_get(key: str | None = None, *, path: Path | None = None) -> dict[str, tuple[Any, str]]: + """Return ``{key: (value, origin)}``. ``origin`` is ``"file"`` or ``"default"``. + + When ``key`` is None or empty, every registered key is returned. + Otherwise just that one key (still as a single-item dict, for caller + uniformity). + """ + path = path or default_config_path() + doc = load_doc(path) if path.exists() else {} + keys = [key] if key else list(KEY_REGISTRY) + out: dict[str, tuple[Any, str]] = {} + for k in keys: + if k not in KEY_REGISTRY: + raise KeyError(f"unknown config key {k!r}; known: {sorted(KEY_REGISTRY)}") + section_field, _ = KEY_REGISTRY[k] + section, field = section_field + in_file = _doc_get(doc, section, field) + if in_file is not None: + out[k] = (in_file, "file") + else: + out[k] = (_live_default(k), "default") + return out + + +def live_config() -> Config: + """Build a fresh Config from current host facts + the DFLASH_* heuristic. + + Used as the no-config fallback in ``cli._load_or_build`` and reused by + the ``models`` sub-app, so the host probe + heuristic + env-override + logic lives in one place rather than being duplicated per caller. + """ + # Lazy import to avoid the autotune ↔ config import cycle the importer + # would hit if this moved to module scope. + import lucebox.autotune as autotune_mod + from lucebox.host_facts import from_env + + host = from_env() + default = Config() + return replace( + default, + variant=os.environ.get("LUCEBOX_VARIANT", "cuda12"), + image=os.environ.get("LUCEBOX_IMAGE", default.image), + container_name=os.environ.get("LUCEBOX_CONTAINER", default.container_name), + port=int(os.environ.get("LUCEBOX_PORT", str(default.port))), + models_dir=Path(os.environ.get("LUCEBOX_MODELS", str(default.models_dir))), + dflash=autotune_mod.runtime_from_host(host), + host=host, + ) diff --git a/lucebox/src/lucebox/docker_run.py b/lucebox/src/lucebox/docker_run.py new file mode 100644 index 000000000..ff3615b26 --- /dev/null +++ b/lucebox/src/lucebox/docker_run.py @@ -0,0 +1,232 @@ +"""Build and execute `docker run` argv for the server and download containers. + +We shell out to the `docker` CLI rather than using the docker SDK because +(a) the CLI is the user-visible contract — errors look the same whether +issued by lucebox or the user; (b) zero import cost; (c) trivially mockable +via subprocess in tests. Wrap everything in one module so swapping to the +SDK later is a single-file change. +""" + +from __future__ import annotations + +import os +import shlex +import subprocess +from dataclasses import dataclass +from pathlib import Path + +from lucebox.types import Config + + +def _host_facts_env() -> list[tuple[str, str]]: + """Forward LUCEBOX_HOST_* from the orchestrator's env into the server. + + lucebox.sh's probe_host() exports every host-identity fact (OS, + kernel, GPU list CSV, CTK version, …) before invoking ``docker run`` + on the orchestrator. The orchestrator inherits them and we pass + them through verbatim so the server entrypoint can write + /opt/lucebox-hub/HOST_INFO without re-probing inside the container + (where /proc and nvidia-smi see the container's view, not the + rig's). See entrypoint.sh::write_host_info and http_server.cpp's + /props.host block. + """ + out: list[tuple[str, str]] = [] + for key, value in sorted(os.environ.items()): + if key.startswith("LUCEBOX_HOST_"): + out.append((key, value)) + return out + + +def _resolve_model_files(cfg: Config) -> tuple[str, str, str]: + """Return (target_file, draft_file, draft_dir) for DFLASH_TARGET / DFLASH_DRAFT. + + Resolution order — first non-empty wins per field: + 1. cfg.model.target_file / draft_file (explicit override in config.toml) + 2. PRESETS[cfg.model.preset].target_file / draft_file / speculator_dir (registry) + 3. "" (entrypoint autodetect path runs unchanged). + + ``draft_dir`` is a directory name under ``models/draft/`` holding a + safetensors speculator (e.g. ``laguna-xs2-speculator``). It is only set + when the preset declares one AND the directory exists on disk; otherwise + it is empty. When non-empty, docker_run_spec uses it as DFLASH_DRAFT + (a directory path) instead of the GGUF-file path, allowing the entrypoint + to discover the safetensors file inside it. + + Imported lazily to avoid the lucebox.types ↔ lucebox.download circular + import that surfaces when this module is imported from ``__init__``. + """ + target = cfg.model.target_file + draft = cfg.model.draft_file + draft_dir = "" + if (not target or not draft) and cfg.model.preset: + from lucebox.download import PRESETS + + pres = PRESETS.get(cfg.model.preset) + if pres is not None: + if not target: + target = pres.target_file + if not draft and pres.has_draft and pres.draft_file: + draft = pres.draft_file + if not draft and pres.speculator_dir: + spec_path = cfg.models_dir / "draft" / pres.speculator_dir + if spec_path.is_dir(): + draft_dir = pres.speculator_dir + return target, draft, draft_dir + + +def _runtime_volumes(cfg: Config) -> tuple[tuple[str, str], ...]: + """Mount models plus $HOME so absolute symlink targets remain valid.""" + home = str(Path.home()) + models = str(cfg.models_dir) + volumes = [(models, "/opt/lucebox-hub/server/models")] + if home != models: + volumes.append((home, home)) + return tuple(volumes) + + +@dataclass(frozen=True, slots=True) +class DockerRunSpec: + """Pre-render of a docker-run command. Render via `argv()` or `printable()`.""" + + image: str + name: str + gpus: bool = True + detach: bool = False + remove: bool = True + port_publish: tuple[int, int] | None = None # (host, container) + volumes: tuple[tuple[str, str], ...] = () + env: tuple[tuple[str, str], ...] = () + entrypoint_args: tuple[str, ...] = () + extra: tuple[str, ...] = () + + def argv(self) -> list[str]: + out = ["docker", "run"] + if self.remove: + out.append("--rm") + if self.detach: + out.append("-d") + out += ["--name", self.name] + if self.gpus: + out += ["--gpus", "all"] + if self.port_publish is not None: + host, container = self.port_publish + out += ["-p", f"{host}:{container}"] + for host_path, container_path in self.volumes: + out += ["-v", f"{host_path}:{container_path}"] + for k, v in self.env: + out += ["-e", f"{k}={v}"] + out += list(self.extra) + out.append(self.image) + out += list(self.entrypoint_args) + return out + + def printable(self) -> str: + """Human-readable, one-flag-per-line docker run. Copy-pasteable.""" + argv = self.argv() + if not argv: + return "" + out = argv[0] + i = 1 + while i < len(argv): + tok = argv[i] + out += " \\\n " + tok + # Glue value-taking flags onto the same line. + if tok in { + "-p", + "-v", + "-e", + "--name", + "--gpus", + "--env", + "--volume", + "--publish", + "--entrypoint", + } and i + 1 < len(argv): + i += 1 + out += " " + shlex.quote(argv[i]) + i += 1 + return out + + +# ── server argv from Config ──────────────────────────────────────────────── + + +def server_run_spec(cfg: Config) -> DockerRunSpec: + """Long-running OpenAI-compatible server. Foreground (systemd manages + lifecycle), --gpus all, models bind-mounted, DFLASH_* propagated. + """ + # LUCEBOX_HOST_* first so they ride out front in the rendered argv, + # making it obvious in `print-run` output what host facts get forwarded. + env: list[tuple[str, str]] = list(_host_facts_env()) + env += [ + ("DFLASH_BUDGET", str(cfg.dflash.budget)), + ("DFLASH_MAX_CTX", str(cfg.dflash.max_ctx)), + ("DFLASH_PREFIX_CACHE_SLOTS", str(cfg.dflash.prefix_cache_slots)), + ("DFLASH_PREFILL_CACHE_SLOTS", str(cfg.dflash.prefill_cache_slots)), + ("DFLASH_THINK_MAX", str(cfg.dflash.think_max)), + ("DFLASH_PORT", "8080"), + ] + # Resolve target/draft GGUFs in priority order: + # 1. cfg.model.target_file / draft_file (explicit override in config.toml) + # 2. PRESETS[cfg.model.preset].target_file / draft_file / speculator_dir (registry) + # 3. unset — entrypoint's autodetect path runs unchanged. + # Container view of the models dir is /opt/lucebox-hub/server/models + # (see _runtime_volumes); the entrypoint reads DFLASH_TARGET / DFLASH_DRAFT. + # draft_dir is a subdirectory of models/draft/ holding a safetensors speculator; + # it takes effect only when draft_file is empty and the directory exists on disk. + target_file, draft_file, draft_dir = _resolve_model_files(cfg) + if target_file: + env.append(("DFLASH_TARGET", f"/opt/lucebox-hub/server/models/{target_file}")) + if draft_file: + env.append(("DFLASH_DRAFT", f"/opt/lucebox-hub/server/models/draft/{draft_file}")) + elif draft_dir: + env.append(("DFLASH_DRAFT", f"/opt/lucebox-hub/server/models/draft/{draft_dir}")) + if cfg.dflash.lazy: + env.append(("DFLASH_LAZY", "1")) + if cfg.dflash.cache_type_k: + env.append(("DFLASH_CACHE_TYPE_K", cfg.dflash.cache_type_k)) + if cfg.dflash.cache_type_v: + env.append(("DFLASH_CACHE_TYPE_V", cfg.dflash.cache_type_v)) + if cfg.dflash.prefill_mode != "off": + env += [ + ("DFLASH_PREFILL_MODE", cfg.dflash.prefill_mode), + ("DFLASH_PREFILL_KEEP", str(cfg.dflash.prefill_keep_ratio)), + ("DFLASH_PREFILL_THRESHOLD", str(cfg.dflash.prefill_threshold)), + ] + if cfg.dflash.prefill_drafter: + env.append(("DFLASH_PREFILL_DRAFTER", cfg.dflash.prefill_drafter)) + # fa_window=0 is the server's own default (full attention); only emit + # the env when the operator has selected a sparse decode window. The + # entrypoint mirrors this guard so an unset env reproduces the + # server's stock behavior. + if cfg.dflash.fa_window > 0: + env.append(("DFLASH_FA_WINDOW", str(cfg.dflash.fa_window))) + # Soft-close ratio: 0.0 is server-side disabled (byte-identical + # to pre-PR-#326 behavior). Emit only when nonzero to keep the + # docker env minimal and mirror the entrypoint's `case` guard. + if cfg.dflash.think_soft_close_min_ratio > 0.0: + env.append(( + "DFLASH_THINK_SOFT_CLOSE_MIN_RATIO", + f"{cfg.dflash.think_soft_close_min_ratio:g}", + )) + if cfg.dflash.debug_thinking_logits: + env.append(("DFLASH_DEBUG_THINKING_LOGITS", "1")) + + return DockerRunSpec( + image=f"{cfg.image}:{cfg.variant}", + name=cfg.container_name, + gpus=True, + remove=True, + detach=False, + port_publish=(cfg.port, 8080), + volumes=_runtime_volumes(cfg), + env=tuple(env), + ) + + +# ── subprocess helpers ───────────────────────────────────────────────────── + + +def docker_pull(image_tag: str) -> int: + """Pull an image, streaming progress. Returns docker's exit code.""" + return subprocess.call(["docker", "pull", image_tag]) diff --git a/lucebox/src/lucebox/download.py b/lucebox/src/lucebox/download.py new file mode 100644 index 000000000..df1d5c96f --- /dev/null +++ b/lucebox/src/lucebox/download.py @@ -0,0 +1,515 @@ +"""Model download orchestration. + +Runs *inside* the orchestrator container. Uses `huggingface_hub` directly +(no subprocess) so we can: + + * drive a Rich progress bar based on real byte counts (the previous + `uvx hf download` subprocess produced no visible progress inside the + container — hf-xet's TTY detection misfires there), + * verify each candidate file's size and sha256 against the repo + metadata BEFORE downloading, so a re-run on a host that already has + the target GGUF (e.g. previous download into the same models_dir) + skips the multi-GB fetch entirely. + +The :data:`PRESETS` registry encodes the canonical (target_repo, +target_file, draft_repo, draft_file) tuple per model — selectable via +``lucebox models download ``. ``DEFAULT_PRESET`` stays pinned to +Qwen3.6-27B for back-compat with callers that pre-date the registry. +Drafts are optional: presets that have no published DFlash draft +(e.g. Laguna's speculator is safetensors, not GGUF) carry +``draft_repo=None`` and run target-only. +""" + +from __future__ import annotations + +import hashlib +import os +import threading +import time +from dataclasses import dataclass +from pathlib import Path + +# hf-xet (huggingface_hub ≥ 1.16) streams the entire file in one final +# burst — the polling-based progress bar sits at 0% for ~14 minutes +# then snaps to 100% on a 17 GB GGUF. Force the chunked Python +# downloader instead so bytes grow continuously and the Rich bar tracks +# reality. Set before importing hf_hub_download so the import picks +# the env up. `setdefault` lets a user override on the command line. +os.environ.setdefault("HF_HUB_DISABLE_XET", "1") + +from huggingface_hub import HfApi, hf_hub_download # noqa: E402 +from huggingface_hub._local_folder import get_local_download_paths # noqa: E402 +from rich.console import Console +from rich.progress import ( + BarColumn, + DownloadColumn, + Progress, + TextColumn, + TimeRemainingColumn, + TransferSpeedColumn, +) + +from lucebox.types import Config, HostFacts + + +@dataclass(frozen=True, slots=True) +class ModelPreset: + """Canonical (target, draft) repo+filename pair for a supported model. + + ``draft_repo`` and ``draft_file`` may both be ``None`` for models + where no GGUF DFlash draft is published (e.g. Laguna's safetensors + speculator). In that case the entrypoint runs target-only — DFlash + speculative decoding is disabled but the server still works. + + ``speculator_dir`` names a directory under ``models/draft/`` that holds + a safetensors-format speculator (e.g. ``model.safetensors``). When + present on disk the server launch sets ``DFLASH_DRAFT`` to that + directory; absent, the server runs target-only. Unlike ``draft_file`` + (which marks the preset as incomplete when missing), ``speculator_dir`` + is optional supplementary hardware and doesn't affect installed_status. + """ + + name: str + target_repo: str + target_file: str + draft_repo: str | None + draft_file: str | None + approx_total_gb: int + description: str = "" + speculator_dir: str | None = None + + @property + def has_draft(self) -> bool: + return bool(self.draft_repo and self.draft_file) + + +# Registry of supported models. Keyed by preset name; the CLI surface +# exposes these via `lucebox models download ` and the +# `lucebox models list` table. The values come straight from the model +# cards under share/model_cards/ — keep them in sync. +PRESETS: dict[str, ModelPreset] = { + "qwen3.6-27b": ModelPreset( + name="qwen3.6-27b", + target_repo="unsloth/Qwen3.6-27B-GGUF", + target_file="Qwen3.6-27B-Q4_K_M.gguf", + draft_repo="spiritbuun/Qwen3.6-27B-DFlash-GGUF", + draft_file="dflash-draft-3.6-q4_k_m.gguf", + approx_total_gb=17, + description="Qwen3.6 27B dense (Q4_K_M) + Qwen3.6 DFlash draft. Lucebox default.", + ), + "gemma-4-26b": ModelPreset( + name="gemma-4-26b", + target_repo="bartowski/google_gemma-4-26B-A4B-it-GGUF", + target_file="google_gemma-4-26B-A4B-it-Q4_K_M.gguf", + draft_repo="Lucebox/gemma-4-26B-A4B-it-DFlash-GGUF", + draft_file="gemma-4-26B-A4B-it-DFlash-q8_0.gguf", + approx_total_gb=18, + description="Gemma 4 26B-A4B IT MoE (Q4_K_M) + Lucebox DFlash q8_0 draft.", + ), + "gemma-4-31b": ModelPreset( + name="gemma-4-31b", + target_repo="bartowski/google_gemma-4-31B-it-GGUF", + target_file="google_gemma-4-31B-it-Q4_K_M.gguf", + draft_repo="Lucebox/gemma-4-31B-it-DFlash-GGUF", + draft_file="gemma-4-31B-it-DFlash-q8_0.gguf", + approx_total_gb=21, + description="Gemma 4 31B IT dense (Q4_K_M) + Lucebox DFlash q8_0 draft.", + ), + "laguna-xs.2": ModelPreset( + name="laguna-xs.2", + target_repo="Lucebox/Laguna-XS.2-GGUF", + target_file="laguna-xs2-Q4_K_M.gguf", + # Laguna's DFlash speculator is safetensors-format + # (poolside/Laguna-XS.2-speculator.dflash), downloaded manually + # into models/draft/laguna-xs2-speculator/. The download command + # doesn't fetch it automatically — it's opt-in. When present, + # speculator_dir wires it into DFLASH_DRAFT at server launch. + draft_repo=None, + draft_file=None, + speculator_dir="laguna-xs2-speculator", + approx_total_gb=20, + description=( + "Laguna-XS.2 MoE code model (Q4_K_M). " + "DFlash safetensors speculator in draft/laguna-xs2-speculator/ " + "is used automatically when present." + ), + ), + "qwen3.6-moe": ModelPreset( + name="qwen3.6-moe", + target_repo="unsloth/Qwen3.6-35B-A3B-GGUF", + # Unsloth's MoE repo publishes both a "UD" (dynamic) and a plain + # Q4_K_M family. Verified 2026-05-28 via HfApi.repo_info: the + # `-UD-Q4_K_M.gguf` variant (22.1 GB) is the canonical Q4_K_M + # release — there is no plain `Q4_K_M.gguf` on the MoE repo. + target_file="Qwen3.6-35B-A3B-UD-Q4_K_M.gguf", + # No DFlash draft GGUF has been published for the MoE variant + # (probed Lucebox/* and spiritbuun/* repos 2026-05-28 — none + # exist). Target-only, mirroring laguna-xs.2's wiring. The + # lucebox C++ server speaks the `qwen35moe` arch natively + # (server/src/qwen35moe/) so this runs without a draft. + draft_repo=None, + draft_file=None, + approx_total_gb=22, + description=( + "Qwen3.6 35B-A3B MoE (3B active per token), Q4_K_M unsloth " + "dynamic quant. Target-only — no DFlash MoE draft published " + "yet. Uses lucebox's qwen35moe arch backend." + ), + ), +} + +DEFAULT_PRESET = PRESETS["qwen3.6-27b"] + + +def resolve_preset(name: str | None) -> ModelPreset: + """Look up a preset by name, with a friendly error on typos. + + ``None`` (or empty string) resolves to :data:`DEFAULT_PRESET` so + callers and the CLI default both flow through one code path. + """ + if not name: + return DEFAULT_PRESET + if name in PRESETS: + return PRESETS[name] + # Build a suggestion list — show every known preset; the user's + # search space is small (4 entries today) so listing them all is + # cheaper and clearer than a fuzzy-match heuristic. + known = ", ".join(sorted(PRESETS.keys())) + raise KeyError(f"unknown preset {name!r}. Known presets: {known}") + + +def _file_meta(api: HfApi, repo_id: str, filename: str) -> tuple[int, str | None]: + """Return (expected_size, lfs_sha256_or_None) for filename in repo_id.""" + info = api.model_info(repo_id, files_metadata=True) + for sib in info.siblings or []: + if sib.rfilename == filename: + sha = getattr(sib.lfs, "sha256", None) if sib.lfs else None + return int(sib.size or 0), sha + raise FileNotFoundError(f"{filename} not present in repo {repo_id}") + + +def _sha256(path: Path, chunk_mb: int = 16) -> str: + h = hashlib.sha256() + chunk = chunk_mb * 1024 * 1024 + with path.open("rb") as f: + while buf := f.read(chunk): + h.update(buf) + return h.hexdigest() + + +def _local_matches(path: Path, size: int, sha256: str | None, console: Console) -> bool: + """True iff a local file at `path` matches the expected size + sha256. + + Size mismatch shortcircuits (cheap). Sha256 is verified for LFS files + (multi-GB GGUFs always carry one) and skipped when the repo doesn't + expose a hash. Hashing 17 GB takes ~30s on a fast SSD — worth it to + avoid a multi-GB re-download on rate-limited / metered links. + """ + if not path.exists(): + return False + actual_size = path.stat().st_size + if actual_size != size: + console.print( + f" [yellow]✗[/yellow] {path.name} present but size {actual_size:,} != " + f"expected {size:,} — will re-download" + ) + return False + if sha256: + console.print(f" [dim]verifying sha256 of {path.name} ({actual_size / 1e9:.1f} GB)…[/dim]") + actual_sha = _sha256(path) + if actual_sha != sha256: + console.print( + f" [yellow]✗[/yellow] {path.name} sha256 {actual_sha[:12]}… != " + f"expected {sha256[:12]}… — will re-download" + ) + return False + return True + + +def _incomplete_path_candidates(local_dir: Path, filename: str, etag: str | None) -> list[Path]: + """Return likely paths of the partial file currently being written. + + huggingface_hub 1.x (with hf-xet) stages downloads under + ``{local_dir}/.cache/huggingface/download/`` using a *hashed* name — + ``{short_hash(metadata_filename)}.{etag}.incomplete`` — so a naive + ``{filename}.incomplete`` poll never sees any growth and the + progress bar sits at 0 % for the whole multi-GB transfer. + + We get the *exact* expected staging path from + ``get_local_download_paths().incomplete_path(etag)`` when we already + know the LFS sha256 (which acts as the etag for Xet downloads), and + fall back to globbing every ``*.incomplete`` in the staging dir + otherwise. The legacy non-Xet downloader writes a ``.incomplete`` + next to the destination blob in ``~/.cache/huggingface/hub`` — but + when ``local_dir`` is set hf-hub always uses the local staging dir, + so the two candidates above cover every code path we hit. + """ + paths = get_local_download_paths(local_dir, filename) + candidates: list[Path] = [] + if etag: + candidates.append(paths.incomplete_path(etag)) + # Fallback: every .incomplete file in the staging dir. This is what + # rescues us when sha256 is unknown (non-LFS file) or when hf-hub + # changes the etag derivation again in some future release. + candidates.append(paths.metadata_path.parent) # sentinel: glob this dir + return candidates + + +def _current_bytes(target: Path, candidates: list[Path]) -> int: + """Best-effort byte count of the file currently being written.""" + if target.exists(): + try: + return target.stat().st_size + except OSError: + pass + for c in candidates: + if c.is_dir(): + # Glob every .incomplete in the staging dir; return the + # largest (there's typically only one in-flight transfer). + largest = 0 + try: + for p in c.glob("*.incomplete"): + try: + largest = max(largest, p.stat().st_size) + except OSError: + continue + except OSError: + continue + if largest: + return largest + else: + try: + if c.exists(): + return c.stat().st_size + except OSError: + continue + return 0 + + +def _download_with_progress( + repo_id: str, + filename: str, + local_dir: Path, + expected_size: int, + console: Console, + etag: str | None = None, +) -> Path: + """Download a single HF file with a Rich progress bar. + + Runs hf_hub_download in a worker thread; the main thread polls the + growing file size and updates the Rich progress bar. The polled + target is computed via ``get_local_download_paths`` so we hit the + actual hf-xet staging path (a hashed filename under + ``.cache/huggingface/download/``), not a guess. + """ + local_dir.mkdir(parents=True, exist_ok=True) + target = local_dir / filename + candidates = _incomplete_path_candidates(local_dir, filename, etag) + + result: list[str | None] = [None] + error: list[BaseException | None] = [None] + + def _worker() -> None: + try: + result[0] = hf_hub_download( + repo_id=repo_id, + filename=filename, + local_dir=str(local_dir), + ) + except BaseException as exc: # propagate to main thread + error[0] = exc + + t = threading.Thread(target=_worker, daemon=True) + t.start() + + with Progress( + TextColumn("[cyan]{task.description}"), + BarColumn(bar_width=40), + DownloadColumn(), + TransferSpeedColumn(), + TimeRemainingColumn(), + console=console, + transient=False, + ) as progress: + task = progress.add_task(filename, total=expected_size or 1) + while t.is_alive(): + current = _current_bytes(target, candidates) + # Always tick the bar — even at 0 bytes — so Rich repaints + # the spinner/ETA and the user sees the UI is alive within + # the first poll tick rather than a blank "Downloading…" line. + progress.update(task, completed=min(current, expected_size or current or 1)) + time.sleep(0.5) + # Final tick after the worker finishes so the bar paints 100%. + if target.exists(): + progress.update(task, completed=target.stat().st_size) + + t.join(timeout=5) + if error[0] is not None: + raise error[0] + if result[0] is None: + raise RuntimeError(f"hf_hub_download returned no path for {filename}") + return Path(result[0]) + + +def _fetch( + api: HfApi, + repo_id: str, + filename: str, + local_dir: Path, + console: Console, +) -> Path: + """Verify-or-download a single file. Skips when the local copy matches.""" + size, sha = _file_meta(api, repo_id, filename) + target = local_dir / filename + if _local_matches(target, size, sha, console): + console.print(f" [green]✓[/green] {filename} already present (size + sha256 match)") + return target + # `sha` doubles as the etag for hf-xet's staging path + # ({local_dir}/.cache/huggingface/download/{hash}.{etag}.incomplete); + # passing it through is what makes the Rich progress bar see real + # byte counts during the multi-GB transfer. + return _download_with_progress(repo_id, filename, local_dir, size, console, etag=sha) + + +def download_preset(cfg: Config, preset: ModelPreset | None = None) -> int: + """Fetch the target GGUF + (optional) DFlash draft into cfg.models_dir. + + Returns 0 on success, non-zero on failure. Verifies each file's size + and (LFS) sha256 against the repo metadata before downloading, so a + repeat run with the files already on disk is a no-op + sha256 walk. + + ``preset=None`` resolves to :data:`DEFAULT_PRESET` for back-compat; + presets with ``has_draft=False`` (e.g. Laguna) skip the draft fetch + entirely and let the server run target-only. + """ + preset = preset or DEFAULT_PRESET + console = Console() + api = HfApi() + models = cfg.models_dir + models.mkdir(parents=True, exist_ok=True) + draft = models / "draft" + draft.mkdir(exist_ok=True) + + try: + _fetch(api, preset.target_repo, preset.target_file, models, console) + if preset.has_draft: + # Narrow the optionals for the type-checker — has_draft is + # exactly the predicate that proves these aren't None. + assert preset.draft_repo is not None and preset.draft_file is not None + _fetch(api, preset.draft_repo, preset.draft_file, draft, console) + else: + console.print( + f" [dim]no DFlash draft published for {preset.name} — running target-only[/dim]" + ) + except Exception as exc: + console.print(f"[red]download failed:[/red] {exc}") + return 1 + return 0 + + +def _local_target_path(cfg: Config, preset: ModelPreset) -> Path: + return cfg.models_dir / preset.target_file + + +def _local_draft_path(cfg: Config, preset: ModelPreset) -> Path | None: + if not (preset.has_draft and preset.draft_file): + return None + return cfg.models_dir / "draft" / preset.draft_file + + +def installed_status(cfg: Config, preset: ModelPreset) -> str: + """Return ``"installed"`` / ``"partial"`` / ``"absent"`` for a preset. + + Size-only — doesn't hash. ``"installed"`` requires the target (and + draft when one is published) to exist on disk; ``"partial"`` means + at least one of the two is present but the set is incomplete. + """ + target_exists = _local_target_path(cfg, preset).exists() + draft_path = _local_draft_path(cfg, preset) + if draft_path is None: + return "installed" if target_exists else "absent" + draft_exists = draft_path.exists() + if target_exists and draft_exists: + return "installed" + if target_exists or draft_exists: + return "partial" + return "absent" + + +def installed_size_gb(cfg: Config, preset: ModelPreset) -> float: + """Sum of on-disk byte sizes for the preset's files, in GB (binary 1e9).""" + total = 0 + target = _local_target_path(cfg, preset) + if target.exists(): + try: + total += target.stat().st_size + except OSError: + pass + draft = _local_draft_path(cfg, preset) + if draft is not None and draft.exists(): + try: + total += draft.stat().st_size + except OSError: + pass + return total / 1e9 + + +def installed_presets(cfg: Config) -> list[ModelPreset]: + """Return every preset whose files are currently present in cfg.models_dir. + + "Present" follows ``installed_status`` — fully installed only. + Partial states (target without draft, etc.) are excluded so the + default ``lucebox models`` view stays uncluttered. + """ + out: list[ModelPreset] = [] + for name in sorted(PRESETS): + pres = PRESETS[name] + if installed_status(cfg, pres) == "installed": + out.append(pres) + return out + + +def status(cfg: Config, preset: ModelPreset | None = None) -> dict[str, bool]: + """Quick presence check — what's already on disk? Size-only, no sha256. + + For presets without a published DFlash draft, ``draft_present`` is + reported as ``True`` (nothing to fetch → nothing missing). That + keeps the "all present, nothing to do" UX path uniform whether or + not a draft exists. + """ + preset = preset or DEFAULT_PRESET + api = HfApi() + out: dict[str, bool] = {} + try: + size, _ = _file_meta(api, preset.target_repo, preset.target_file) + local = cfg.models_dir / preset.target_file + out["target_present"] = local.exists() and local.stat().st_size == size + except Exception: + out["target_present"] = False + + if preset.has_draft: + assert preset.draft_repo is not None and preset.draft_file is not None + try: + size, _ = _file_meta(api, preset.draft_repo, preset.draft_file) + local = cfg.models_dir / "draft" / preset.draft_file + out["draft_present"] = local.exists() and local.stat().st_size == size + except Exception: + out["draft_present"] = False + else: + out["draft_present"] = True + return out + + +def recommend_preset(host: HostFacts) -> str | None: + """Pick a default preset for first-run install. None = ask the user. + + Tiers follow the model size catalog: 22 GB+ → Qwen3.6-27B (the + Lucebox default), 16-21 GB → Laguna-XS.2 (small target-only). Below + 16 GB we punt and let the user pick explicitly — the registered + presets all need at least 16 GB to run usefully. + """ + if host.vram_gb >= 22: + return "qwen3.6-27b" + if host.vram_gb >= 16: + return "laguna-xs.2" + return None diff --git a/lucebox/src/lucebox/host_check.py b/lucebox/src/lucebox/host_check.py new file mode 100644 index 000000000..2ce8d3889 --- /dev/null +++ b/lucebox/src/lucebox/host_check.py @@ -0,0 +1,232 @@ +"""Readiness check: aggregate HostFacts (provided by lucebox.sh) with the +docker-daemon checks we can do from inside the container via the mounted +socket. Prints a status report and returns an aggregate severity. +""" + +from __future__ import annotations + +import os +from dataclasses import dataclass +from typing import Literal + +from rich.console import Console + +from lucebox.types import HostFacts + +Severity = Literal["ok", "warn", "fail"] +_SEVERITY_ORDER: dict[Severity, int] = {"ok": 0, "warn": 1, "fail": 2} + + +@dataclass(frozen=True, slots=True) +class CheckResult: + name: str + severity: Severity + message: str + hint: str | None = None + + +def run_checks(host: HostFacts) -> list[CheckResult]: + return [ + _check_docker(host), + _check_nvidia_driver(host), + _check_ctk(host), + _check_ram(host), + _check_vram(host), + _check_systemd(host), + ] + + +def _check_docker(host: HostFacts) -> CheckResult: + if not host.has_docker: + return CheckResult( + "docker", + "fail", + "docker daemon unreachable", + "sudo systemctl start docker, or add your user to the 'docker' group", + ) + return CheckResult("docker", "ok", f"daemon reachable ({host.docker_version})") + + +def _check_nvidia_driver(host: HostFacts) -> CheckResult: + if host.gpu_vendor != "nvidia": + if host.gpu_vendor == "amd": + return CheckResult( + "gpu", + "fail", + "AMD GPU detected — prebuilt images are NVIDIA-only", + "Build dflash from source with HIP; see dflash/README.md", + ) + return CheckResult("gpu", "fail", "no NVIDIA GPU detected") + if not host.driver_version: + return CheckResult( + "driver", + "warn", + "nvidia-smi present but NVML query failed (likely driver/library mismatch)", + "reboot, or reinstall the matching NVIDIA driver", + ) + if host.driver_major < 525: + return CheckResult( + "driver", + "fail", + f"driver r{host.driver_major} too old (need r525+ for cuda12)", + "upgrade the NVIDIA driver", + ) + return CheckResult("driver", "ok", f"nvidia r{host.driver_major} ({host.driver_version})") + + +def _check_ctk(host: HostFacts) -> CheckResult: + match host.ctk: + case "runtime": + return CheckResult("ctk", "ok", "NVIDIA Container Toolkit registered as docker runtime") + case "cdi": + return CheckResult("ctk", "ok", "NVIDIA Container Toolkit available via CDI") + case "installed-unwired": + return CheckResult( + "ctk", + "warn", + "NVIDIA Container Toolkit installed but not wired into docker", + "sudo nvidia-ctk runtime configure --runtime=docker && " + "sudo systemctl restart docker", + ) + case _: + return CheckResult( + "ctk", + "fail", + "NVIDIA Container Toolkit not installed", + "https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html", + ) + + +def _check_ram(host: HostFacts) -> CheckResult: + if host.ram_gb == 0: + return CheckResult("ram", "warn", "RAM unknown") + if host.ram_gb < 16: + return CheckResult("ram", "warn", f"{host.ram_gb} GB RAM — model load may swap") + return CheckResult("ram", "ok", f"{host.ram_gb} GB RAM") + + +def _check_vram(host: HostFacts) -> CheckResult: + if host.vram_gb == 0: + return CheckResult("vram", "warn", "VRAM unknown") + if host.vram_gb < 12: + return CheckResult( + "vram", + "fail", + f"VRAM {host.vram_gb} GB < 12 GB — 27B target won't fit", + "use a smaller model preset or larger GPU", + ) + if host.vram_gb < 22: + return CheckResult( + "vram", + "warn", + f"VRAM {host.vram_gb} GB — 27B fits but max_ctx will be capped near 32K", + ) + return CheckResult("vram", "ok", f"VRAM {host.vram_gb} GB ({host.gpu_name})") + + +def _check_systemd(host: HostFacts) -> CheckResult: + if not host.has_systemd: + return CheckResult( + "systemd", + "warn", + "user systemd not available", + "WSL: enable systemd in /etc/wsl.conf; otherwise 'lucebox serve' " + "still works in the foreground", + ) + return CheckResult("systemd", "ok", "user systemd available") + + +def aggregate(results: list[CheckResult]) -> Severity: + worst: Severity = "ok" + for r in results: + if _SEVERITY_ORDER[r.severity] > _SEVERITY_ORDER[worst]: + worst = r.severity + return worst + + +def render(console: Console, host: HostFacts, results: list[CheckResult]) -> Severity: + """Print a status block, return the worst severity.""" + summary = f"[bold]Host:[/bold] {host.nproc} CPUs · {host.ram_gb} GB RAM" + if host.gpu_vendor == "nvidia" and host.gpu_name: + summary += f" · {host.gpu_name} · {host.vram_gb} GB VRAM" + ( + f" (sm_{host.gpu_sm})" if host.gpu_sm else "" + ) + if host.is_wsl: + summary += " · WSL2" + console.print(summary) + console.print() + + sev_style = { + "ok": "[green]OK[/green]", + "warn": "[yellow]WARN[/yellow]", + "fail": "[red]FAIL[/red]", + } + for r in results: + console.print(f" {sev_style[r.severity]:<22} {r.name:<8} {r.message}") + if r.hint: + console.print(f" {'':<22} {'':<8} [dim]{r.hint}[/dim]") + + render_host_facts(console) + + worst = aggregate(results) + console.print() + if worst == "ok": + console.print("[green]All checks passed.[/green]") + elif worst == "warn": + console.print("[yellow]Checks passed with warnings.[/yellow]") + else: + console.print( + "[red]Critical checks failed — fix the issues above before 'lucebox start'.[/red]" + ) + return worst + + +def render_host_facts(console: Console) -> None: + """Print a pretty 'Host facts' section sourced from LUCEBOX_HOST_*. + + Same data that ends up in /opt/lucebox-hub/HOST_INFO inside the + container — printed here so the operator can sanity-check the + rig classification BEFORE starting a long bench run, and so the + CI exit-code gate (the pass/fail checks above) stays orthogonal + to the informational host facts. + + Reads from the same LUCEBOX_HOST_* env the host wrapper exports + (see lucebox.sh::probe_host). Quiet — emits the section header + even when most facts are unset, since "no host facts probed at + all" is itself a useful signal. + """ + console.print() + console.print("[bold]Host facts[/bold] (LUCEBOX_HOST_*, surfaced as /props.host)") + facts = [ + ("os", os.environ.get("LUCEBOX_HOST_OS_PRETTY", "")), + ("kernel", os.environ.get("LUCEBOX_HOST_KERNEL", "")), + ("wsl_version", os.environ.get("LUCEBOX_HOST_WSL_VERSION", "")), + ("docker", os.environ.get("LUCEBOX_HOST_DOCKER_VERSION", "")), + ("nvidia_driver", os.environ.get("LUCEBOX_HOST_DRIVER_VERSION", "")), + ("nvidia_ctk", os.environ.get("LUCEBOX_HOST_NVIDIA_CTK_VERSION", "")), + ("cpu", os.environ.get("LUCEBOX_HOST_CPU_MODEL", "")), + ("cuda_visible_devices", os.environ.get("LUCEBOX_HOST_CUDA_VISIBLE_DEVICES", "")), + ] + for key, value in facts: + display = value if value else "[dim](unset)[/dim]" + console.print(f" {key:<22} {display}") + + # Multi-GPU table — one line per device. LUCEBOX_HOST_GPU_LIST_CSV + # carries the verbatim nvidia-smi CSV the host wrapper probed. + csv = os.environ.get("LUCEBOX_HOST_GPU_LIST_CSV", "") + if csv: + console.print(" gpus:") + for line in csv.splitlines(): + line = line.strip() + if not line: + continue + parts = [c.strip() for c in line.split(",")] + if len(parts) >= 7: + idx, _uuid, _pci, name, sm, mem, plimit = parts[:7] + console.print( + f" [{idx}] {name} (sm_{sm}, {mem}, {plimit})" + ) + else: + console.print(f" {line}") + else: + console.print(" gpus [dim](none — nvidia-smi unavailable)[/dim]") diff --git a/lucebox/src/lucebox/host_facts.py b/lucebox/src/lucebox/host_facts.py new file mode 100644 index 000000000..5deb6721a --- /dev/null +++ b/lucebox/src/lucebox/host_facts.py @@ -0,0 +1,58 @@ +"""Read HostFacts from the LUCEBOX_HOST_* env vars that lucebox.sh exports. + +We deliberately don't try to detect anything ourselves on the Python side — +inside the container, /proc/meminfo reports the container's view, not the +host's, and nvidia-smi may or may not be available depending on how the +caller invoked us. The host wrapper is the only thing that can see the +truth, and it's already paid for the probe. +""" + +from __future__ import annotations + +import os +from typing import cast + +from lucebox.types import CtkStatus, GpuVendor, HostFacts + + +def _env_int(key: str, default: int = 0) -> int: + raw = os.environ.get(key, "").strip() + if not raw: + return default + try: + return int(raw) + except ValueError: + return default + + +def _env_bool(key: str) -> bool: + return os.environ.get(key, "").strip() in {"1", "true", "yes", "on"} + + +def from_env() -> HostFacts: + vendor: GpuVendor = "none" + raw_vendor = os.environ.get("LUCEBOX_HOST_GPU_VENDOR", "none") + if raw_vendor in {"nvidia", "amd", "none"}: + vendor = cast(GpuVendor, raw_vendor) + + ctk: CtkStatus = "none" + raw_ctk = os.environ.get("LUCEBOX_HOST_HAS_CTK", "none") + if raw_ctk in {"runtime", "cdi", "installed-unwired", "none"}: + ctk = cast(CtkStatus, raw_ctk) + + return HostFacts( + nproc=_env_int("LUCEBOX_HOST_NPROC"), + ram_gb=_env_int("LUCEBOX_HOST_RAM_GB"), + gpu_vendor=vendor, + gpu_name=os.environ.get("LUCEBOX_HOST_GPU_NAME", ""), + gpu_count=_env_int("LUCEBOX_HOST_GPU_COUNT"), + vram_gb=_env_int("LUCEBOX_HOST_VRAM_GB"), + gpu_sm=os.environ.get("LUCEBOX_HOST_GPU_SM", ""), + driver_version=os.environ.get("LUCEBOX_HOST_DRIVER_VERSION", ""), + driver_major=_env_int("LUCEBOX_HOST_DRIVER_MAJOR"), + has_systemd=_env_bool("LUCEBOX_HOST_HAS_SYSTEMD"), + is_wsl=_env_bool("LUCEBOX_HOST_IS_WSL"), + has_docker=_env_bool("LUCEBOX_HOST_HAS_DOCKER"), + docker_version=os.environ.get("LUCEBOX_HOST_DOCKER_VERSION", ""), + ctk=ctk, + ) diff --git a/lucebox/src/lucebox/py.typed b/lucebox/src/lucebox/py.typed new file mode 100644 index 000000000..8b1378917 --- /dev/null +++ b/lucebox/src/lucebox/py.typed @@ -0,0 +1 @@ + diff --git a/lucebox/src/lucebox/types.py b/lucebox/src/lucebox/types.py new file mode 100644 index 000000000..e1d3620d7 --- /dev/null +++ b/lucebox/src/lucebox/types.py @@ -0,0 +1,140 @@ +"""Shared dataclasses passed between modules. + +HostFacts is populated from the LUCEBOX_HOST_* env vars set by lucebox.sh. +Config is what we serialize to/from .lucebox/config.toml. Both are frozen so +mistakes (e.g. mutating a config after autotune wrote it) fail loudly. +""" + +from __future__ import annotations + +import os +from dataclasses import dataclass, field +from pathlib import Path +from typing import Literal + +Variant = str +CtkStatus = Literal["runtime", "cdi", "installed-unwired", "none"] + + +def default_models_dir() -> Path: + """Resolve the default models directory under the XDG Base Directory spec. + + $XDG_DATA_HOME (default ~/.local/share) is the conventional location for + user-specific data files on Linux + macOS. Lucebox nests its model store + under that so downloads live alongside other per-user app data instead + of cluttering $HOME directly. The host wrapper bind-mounts this path + into the container so paths line up in and out of the image. + """ + base = os.environ.get("XDG_DATA_HOME") or str(Path.home() / ".local" / "share") + return Path(base) / "lucebox" / "models" + + +GpuVendor = Literal["nvidia", "amd", "none"] + + +@dataclass(frozen=True, slots=True) +class HostFacts: + """Probed once by lucebox.sh, passed in via env vars. Single source of + truth on the Python side — we never reprobe (we can't see host /proc).""" + + nproc: int = 0 + ram_gb: int = 0 + gpu_vendor: GpuVendor = "none" + gpu_name: str = "" + gpu_count: int = 0 + vram_gb: int = 0 + gpu_sm: str = "" # e.g. "120" — matches docker-bake arch lists + driver_version: str = "" # e.g. "595.71.05" + driver_major: int = 0 + has_systemd: bool = False + is_wsl: bool = False + has_docker: bool = False + docker_version: str = "" + ctk: CtkStatus = "none" + + +@dataclass(frozen=True, slots=True) +class DflashRuntime: + """The DFLASH_* knobs as typed values. Serialized under [dflash] in TOML + and emitted as -e DFLASH_FOO=bar args to docker run. + + The 11 fields below (budget through prefill_drafter) form the strict + allowlist mirrored by lucebench's snapshot config.json — keep both + in lockstep. ``think_max`` is a separate phase-1 thinking cap that + isn't part of the runtime snapshot allowlist (it's per-request, not + per-server). + """ + + budget: int = 22 + max_ctx: int = 16384 + lazy: bool = False + prefix_cache_slots: int = 0 + prefill_cache_slots: int = 0 + cache_type_k: str = "" + cache_type_v: str = "" + prefill_mode: Literal["off", "auto", "always"] = "off" + prefill_keep_ratio: float = 0.05 + prefill_threshold: int = 32000 + prefill_drafter: str = "" + # Phase-1 (thinking) cap when a request opts into thinking. Default mirrors + # antirez/ds4 ds4_eval.c: think_max_tokens = max_tokens - hard_limit_reply + # budget = 16000 - 512 = 15488. The server's own hardcoded default is 10000. + think_max: int = 15488 + # Flash-attention sliding-window on full-attention layers. 0 = full + # attention (server default). On gemma4's hybrid iSWA the full-attn + # layers grow KV linearly with max_ctx; a sparse fa_window keeps + # decode compute bounded on long prompts without changing the KV + # footprint. Q: passed through to the server's `--fa-window ` + # flag (see server/src/server/server_main.cpp). + fa_window: int = 0 + # Soft-close thinking termination dial (PR #326 in lucebox-hub). + # Lets the AR loop force early when the close-token logit + # comes within this probability ratio of the chosen-token logit. + # Range [0.0, 1.0]; 0.0 = disabled (byte-identical to pre-change + # behaviour). 0.5 = close when close-token prob >= 0.5 * chosen-token + # prob; 0.9 = aggressive. Qwen3.5/3.6 AR path only in v1. Surfaced + # to the server via DFLASH_THINK_SOFT_CLOSE_MIN_RATIO → + # --think-soft-close-min-ratio. + think_soft_close_min_ratio: float = 0.0 + # Diagnostic: when True, surface --debug-thinking-logits to the + # server CLI via DFLASH_DEBUG_THINKING_LOGITS=1, producing one + # stderr line per thinking AR step recording the close-vs-chosen + # logit gap. Used to fit a sliding-ratio curve from real trajectory + # data. Heavy stderr (one line per thinking token across all + # in-flight requests); leave off in production. + debug_thinking_logits: bool = False + + +@dataclass(frozen=True, slots=True) +class ModelMeta: + """Which preset the operator picked at configure/download time. + + Persisted under ``[model]`` in config.toml so `lucebox serve` can + pass ``DFLASH_TARGET=/opt/lucebox-hub/server/models/`` and + ``DFLASH_DRAFT`` for the draft GGUF (when one is published for the + preset). The entrypoint's "multiple candidate GGUFs" branch never + has to guess which one to load. + + ``target_file`` and ``draft_file`` are advanced overrides — when set + they win over the preset's registry default. Empty strings mean + "fall back to the registry value for [model] preset, then to the + entrypoint's autodetect". + """ + + preset: str = "" + target_file: str = "" + draft_file: str = "" + + +@dataclass(frozen=True, slots=True) +class Config: + """The whole config.toml, materialized.""" + + variant: Variant = "cuda12" + image: str = "ghcr.io/luce-org/lucebox-hub" + container_name: str = "lucebox" + port: int = 8080 + models_dir: Path = field(default_factory=default_models_dir) + dflash: DflashRuntime = field(default_factory=DflashRuntime) + host: HostFacts = field(default_factory=HostFacts) + model: ModelMeta = field(default_factory=ModelMeta) diff --git a/lucebox/tests/test_autotune.py b/lucebox/tests/test_autotune.py new file mode 100644 index 000000000..c47c17c9e --- /dev/null +++ b/lucebox/tests/test_autotune.py @@ -0,0 +1,48 @@ +from lucebox.autotune import runtime_from_host +from lucebox.types import HostFacts + + +def test_wsl_24gb_defaults_leave_cuda_headroom() -> None: + runtime = runtime_from_host(HostFacts(vram_gb=24, is_wsl=True)) + + assert runtime.budget == 16 + # Bumped 65536 → 98304 on 2026-05-30 after the gemma4-26b coding- + # agent-loop sweep proved 98K serves 90K-token agentic prompts + # with ~3 GB VRAM headroom and no CUDA VMM failures on the 3090 Ti + # WSL configuration (see + # docs/experiments/gemma4-26b-coding-agent-loop-sweep-2026-05-30.md). + assert runtime.max_ctx == 98304 + # lazy is False because the heuristic path does NOT set prefill_drafter, + # and the C++ server silently ignores --lazy-draft without it. Flipping + # to False makes the host config match runtime behaviour. See the + # `entrypoint.sh` warning emitted when the two are out-of-sync. + assert runtime.lazy is False + assert runtime.prefix_cache_slots == 0 + + +def test_native_24gb_caps_context_below_vmm_failure_boundary() -> None: + runtime = runtime_from_host(HostFacts(vram_gb=24, is_wsl=False)) + + assert runtime.budget == 22 + assert runtime.max_ctx == 98304 + assert runtime.lazy is False # see WSL test above + assert runtime.prefix_cache_slots == 0 + + +def test_no_heuristic_tier_sets_lazy_without_prefill_drafter() -> None: + """Regression for the `--lazy-draft ignored` silent no-op. + + The C++ dflash_server drops `--lazy-draft` unless `--prefill-drafter` + is also passed. The heuristic doesn't set `prefill_drafter`, so any + tier that sets `lazy=True` would produce a host config that doesn't + match what actually ran — exactly the mismatch the sindri decode + sweep tripped over (every docker.stderr contained the warning). + """ + for vram in (0, 8, 16, 24, 40, 80): + for is_wsl in (False, True): + rt = runtime_from_host(HostFacts(vram_gb=vram, is_wsl=is_wsl)) + if rt.lazy: + assert rt.prefill_drafter, ( + f"vram={vram} is_wsl={is_wsl}: lazy=True without " + f"prefill_drafter → silent no-op on the C++ server" + ) diff --git a/lucebox/tests/test_check.py b/lucebox/tests/test_check.py new file mode 100644 index 000000000..3fdd469d9 --- /dev/null +++ b/lucebox/tests/test_check.py @@ -0,0 +1,118 @@ +"""Tests for ``lucebox check`` — readiness report. + +The check command has two surfaces that must stay independent: + + * pass/fail checks → drive the exit code, so the command is usable + as a CI exit-code gate; + * Host facts section → informational, prints the LUCEBOX_HOST_* + convoy that gets baked into /opt/lucebox-hub/HOST_INFO inside + the container. +""" + +from __future__ import annotations + +import pytest +from lucebox.cli import app +from lucebox.types import HostFacts +from rich.console import Console +from typer.testing import CliRunner + +from lucebox import host_check + + +def test_check_prints_host_facts_section(monkeypatch: pytest.MonkeyPatch) -> None: + """`lucebox check` includes a Host facts block sourced from LUCEBOX_HOST_*.""" + monkeypatch.setenv("LUCEBOX_HOST_OS_PRETTY", "Ubuntu 22.04.3 LTS") + monkeypatch.setenv("LUCEBOX_HOST_KERNEL", "6.6.87.2-microsoft-standard-WSL2") + monkeypatch.setenv("LUCEBOX_HOST_WSL_VERSION", "wsl2") + monkeypatch.setenv("LUCEBOX_HOST_DOCKER_VERSION", "29.1.3") + monkeypatch.setenv("LUCEBOX_HOST_DRIVER_VERSION", "596.36") + monkeypatch.setenv("LUCEBOX_HOST_NVIDIA_CTK_VERSION", "1.16.2") + monkeypatch.setenv("LUCEBOX_HOST_CPU_MODEL", "Intel Test CPU") + monkeypatch.setenv( + "LUCEBOX_HOST_GPU_LIST_CSV", + "0, GPU-abc, 00000000:01:00.0, NVIDIA RTX 5090, 12.0, 24576 MiB, 175.00 W", + ) + # Stub HostFacts so the pass/fail checks succeed at least minimally. + # `cli.check` imports `from_env` into its module namespace, so patch + # both names. + def stub() -> HostFacts: + return HostFacts( + nproc=24, + ram_gb=64, + gpu_vendor="nvidia", + gpu_name="NVIDIA RTX 5090", + gpu_count=1, + vram_gb=24, + gpu_sm="120", + driver_version="596.36", + driver_major=596, + has_systemd=True, + is_wsl=True, + has_docker=True, + docker_version="29.1.3", + ctk="runtime", + ) + monkeypatch.setattr("lucebox.host_facts.from_env", stub) + monkeypatch.setattr("lucebox.cli.from_env", stub) + result = CliRunner().invoke(app, ["check"]) + # The pass/fail half of `check` should still exit 0 on this stubbed host. + assert result.exit_code == 0, result.stdout + assert "Host facts" in result.stdout + assert "Ubuntu 22.04.3 LTS" in result.stdout + assert "wsl2" in result.stdout + assert "1.16.2" in result.stdout + assert "Intel Test CPU" in result.stdout + # Multi-GPU table line. + assert "NVIDIA RTX 5090" in result.stdout + + +def test_render_host_facts_unset_env_shows_placeholders( + monkeypatch: pytest.MonkeyPatch, capsys: pytest.CaptureFixture[str] +) -> None: + """All LUCEBOX_HOST_* unset → section still renders with explicit (unset) markers.""" + for k in list(__import__("os").environ): + if k.startswith("LUCEBOX_HOST_"): + monkeypatch.delenv(k, raising=False) + console = Console(force_terminal=False, no_color=True, record=True) + host_check.render_host_facts(console) + text = console.export_text() + assert "Host facts" in text + # Multi-line section renders even when no env was passed in. + assert "(unset)" in text + assert "gpus" in text + + +def test_check_exit_code_independent_of_host_facts( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """Host facts section must not change the exit-code semantics of check. + + Drives the pass/fail logic through a known-fail HostFacts (no docker) + and asserts the exit code is still 1, regardless of what the Host + facts block prints. + """ + monkeypatch.setenv("LUCEBOX_HOST_OS_PRETTY", "Bare Linux") + def stub() -> HostFacts: + return HostFacts( + nproc=8, + ram_gb=16, + gpu_vendor="nvidia", + gpu_name="X", + gpu_count=1, + vram_gb=24, + gpu_sm="86", + driver_version="555.00", + driver_major=555, + has_systemd=False, + is_wsl=False, + has_docker=False, # → fail + docker_version="", + ctk="none", # also fail + ) + monkeypatch.setattr("lucebox.host_facts.from_env", stub) + monkeypatch.setattr("lucebox.cli.from_env", stub) + result = CliRunner().invoke(app, ["check"]) + assert result.exit_code == 1 + # Host facts block still printed despite the failure. + assert "Host facts" in result.stdout diff --git a/lucebox/tests/test_cli.py b/lucebox/tests/test_cli.py new file mode 100644 index 000000000..f7628e8b7 --- /dev/null +++ b/lucebox/tests/test_cli.py @@ -0,0 +1,102 @@ +"""Tests for the top-level Typer surface.""" + +from __future__ import annotations + +import os + +import pytest +from lucebox.cli import app +from typer.testing import CliRunner + + +def test_config_subcommand_is_registered() -> None: + result = CliRunner().invoke(app, ["config", "--help"]) + assert result.exit_code == 0 + assert "get" in result.output + assert "set" in result.output + assert "unset" in result.output + + +def test_models_subcommand_is_registered() -> None: + result = CliRunner().invoke(app, ["models", "--help"]) + assert result.exit_code == 0 + assert "list" in result.output + assert "download" in result.output + + +@pytest.mark.parametrize( + "verb", + [ + "autotune", + "sweep", + "profile", + "smoke", + "claude", + "codex", + "opencode", + "hermes", + "pi", + "openclaw", + ], +) +def test_deferred_verbs_are_not_registered(verb: str) -> None: + """autotune/sweep, profile/smoke and the client launchers are deferred to + follow-up PRs — this core CLI (launch / serve / install / download) must + not expose them.""" + result = CliRunner().invoke(app, [verb, "--help"]) + assert result.exit_code != 0 + + +def test_core_verbs_present_in_app() -> None: + """The core launch/serve surface stays wired into the Typer command table.""" + registered = { + c.name or (c.callback.__name__ if c.callback else "") + for c in app.registered_commands + } + for verb in ("check", "pull", "print-run", "print-serve-argv", "version"): + assert verb in registered + + +def test_legacy_subcommands_are_removed() -> None: + """`configure` and `download-models` were folded into config/models.""" + cfg = CliRunner().invoke(app, ["configure", "--help"]) + assert cfg.exit_code != 0 + dl = CliRunner().invoke(app, ["download-models", "--help"]) + assert dl.exit_code != 0 + + +def test_server_run_spec_forwards_lucebox_host_env(monkeypatch) -> None: + """server_run_spec carries LUCEBOX_HOST_* from the orchestrator into the server. + + lucebox.sh exports the LUCEBOX_HOST_* convoy before `docker run` on the + orchestrator; the orchestrator inherits them and we forward each one + as ``-e KEY=VALUE`` to the server container so entrypoint.sh's + write_host_info() can populate /opt/lucebox-hub/HOST_INFO. + """ + import lucebox.docker_run as docker_run + from lucebox.config import live_config + + # Scrub any pre-existing LUCEBOX_HOST_* env so the test sees only what we set. + for k in list(os.environ): + if k.startswith("LUCEBOX_HOST_"): + monkeypatch.delenv(k, raising=False) + monkeypatch.setenv("LUCEBOX_HOST_OS_PRETTY", "Ubuntu 22.04.3 LTS") + monkeypatch.setenv("LUCEBOX_HOST_KERNEL", "6.6.87.2-microsoft-standard-WSL2") + monkeypatch.setenv("LUCEBOX_HOST_WSL_VERSION", "wsl2") + monkeypatch.setenv( + "LUCEBOX_HOST_GPU_LIST_CSV", + "0, GPU-x, 00000000:01:00.0, NVIDIA RTX 5090, 12.0, 24576 MiB, 175.00 W", + ) + + cfg = live_config() + spec = docker_run.server_run_spec(cfg) + env_keys = {k for k, _ in spec.env} + assert "LUCEBOX_HOST_OS_PRETTY" in env_keys + assert "LUCEBOX_HOST_KERNEL" in env_keys + assert "LUCEBOX_HOST_WSL_VERSION" in env_keys + assert "LUCEBOX_HOST_GPU_LIST_CSV" in env_keys + # DFLASH_* still present. + assert "DFLASH_BUDGET" in env_keys + # Values surface verbatim. + env_map = dict(spec.env) + assert env_map["LUCEBOX_HOST_OS_PRETTY"] == "Ubuntu 22.04.3 LTS" diff --git a/lucebox/tests/test_config.py b/lucebox/tests/test_config.py new file mode 100644 index 000000000..2ec795a12 --- /dev/null +++ b/lucebox/tests/test_config.py @@ -0,0 +1,210 @@ +"""Tests for the sparse TOML config persistence layer.""" + +from __future__ import annotations + +from pathlib import Path + +import pytest +from lucebox.config import config_get, config_set, config_unset + +from lucebox import config + + +def test_legacy_env_migration_skips_invalid_values(tmp_path: Path) -> None: + legacy = tmp_path / "config.env" + legacy.write_text("DFLASH_BUDGET=not-an-int\nDFLASH_MAX_CTX=65536\nDFLASH_LAZY=true\n") + + cfg, _doc = config._load_legacy_env(legacy) + + assert cfg.dflash.budget == 22 + assert cfg.dflash.max_ctx == 65536 + assert cfg.dflash.lazy is True + + +def test_image_variant_round_trips_from_toml(tmp_path: Path) -> None: + path = tmp_path / "config.toml" + path.write_text( + "[image]\n" + 'registry = "ghcr.io/luce-org/lucebox-hub"\n' + 'variant = "integration-props-uv-squared-clean-cuda12"\n' + ) + + cfg = config._load_toml(path) + + assert cfg.image == "ghcr.io/luce-org/lucebox-hub" + assert cfg.variant == "integration-props-uv-squared-clean-cuda12" + + +def test_model_preset_round_trips_through_set_and_load(tmp_path: Path) -> None: + """Setting model.preset writes a sparse TOML doc that loads back correctly.""" + path = tmp_path / "config.toml" + config_set("model.preset", "gemma-4-26b", path=path) + config_set("model.target_file", "google_gemma-4-26B-A4B-it-Q4_K_M.gguf", path=path) + + cfg = config._load_toml(path) + assert cfg.model.preset == "gemma-4-26b" + assert cfg.model.target_file == "google_gemma-4-26B-A4B-it-Q4_K_M.gguf" + + +def test_legacy_config_without_model_section_stays_unpinned(tmp_path: Path) -> None: + """Legacy configs (no [model] section) must NOT silently pin to qwen.""" + path = tmp_path / "config.toml" + path.write_text('[image]\nvariant = "cuda12"\n') + + cfg = config._load_toml(path) + + assert cfg.model.preset == "" + assert cfg.model.target_file == "" + assert cfg.model.draft_file == "" + + +def test_model_section_picks_target_file_from_registry(tmp_path: Path) -> None: + """A bare [model] preset="..." entry pulls target_file from the registry.""" + path = tmp_path / "config.toml" + path.write_text('[model]\npreset = "gemma-4-31b"\n') + + cfg = config._load_toml(path) + + assert cfg.model.preset == "gemma-4-31b" + assert cfg.model.target_file == "google_gemma-4-31B-it-Q4_K_M.gguf" + + +def test_model_section_picks_draft_file_from_registry(tmp_path: Path) -> None: + """When preset has a published draft GGUF, [model] preset="..." picks draft_file too.""" + path = tmp_path / "config.toml" + path.write_text('[model]\npreset = "qwen3.6-27b"\n') + + cfg = config._load_toml(path) + assert cfg.model.preset == "qwen3.6-27b" + assert cfg.model.draft_file == "dflash-draft-3.6-q4_k_m.gguf" + + +def test_config_set_writes_only_named_key(tmp_path: Path) -> None: + """Sparse persistence: setting one key does NOT serialize every default.""" + path = tmp_path / "config.toml" + config_set("dflash.budget", 16, path=path) + body = path.read_text() + # The only [dflash] field that should appear is budget — none of the others. + assert "[dflash]" in body + assert "budget = 16" in body + assert "max_ctx" not in body # not user-set, must not appear + assert "lazy" not in body + assert "[host]" not in body # whole section absent + assert "[image]" not in body # not touched either + + +def test_config_set_preserves_existing_keys(tmp_path: Path) -> None: + """Setting a new key leaves previously-set keys intact.""" + path = tmp_path / "config.toml" + config_set("dflash.budget", 16, path=path) + config_set("model.preset", "qwen3.6-27b", path=path) + body = path.read_text() + assert "budget = 16" in body + assert 'preset = "qwen3.6-27b"' in body + + +def test_config_unset_removes_one_key(tmp_path: Path) -> None: + """Unset removes the named key and leaves siblings alone.""" + path = tmp_path / "config.toml" + config_set("dflash.budget", 16, path=path) + config_set("dflash.max_ctx", 65536, path=path) + changed = config_unset("dflash.budget", path=path) + assert changed is True + body = path.read_text() + assert "budget" not in body + assert "max_ctx = 65536" in body + + +def test_config_unset_drops_empty_section(tmp_path: Path) -> None: + """Unsetting the last key in a section drops the empty section.""" + path = tmp_path / "config.toml" + config_set("dflash.budget", 16, path=path) + config_unset("dflash.budget", path=path) + body = path.read_text() + # The section may still exist as an empty table but `[dflash]` shouldn't. + assert "[dflash]" not in body + + +def test_config_get_reports_origin(tmp_path: Path) -> None: + """Each key carries an origin label — `file` when overridden, `default` otherwise.""" + path = tmp_path / "config.toml" + config_set("dflash.budget", 9, path=path) + entries = config_get(path=path) + assert entries["dflash.budget"] == (9, "file") + # max_ctx wasn't set so should report the live default. + value, origin = entries["dflash.max_ctx"] + assert origin == "default" + assert value == 16384 # DflashRuntime.max_ctx default + + +def test_config_get_rejects_unknown_key(tmp_path: Path) -> None: + path = tmp_path / "config.toml" + with pytest.raises(KeyError): + config_get("not.a.key", path=path) + + +def test_config_set_rejects_unknown_key(tmp_path: Path) -> None: + path = tmp_path / "config.toml" + with pytest.raises(KeyError): + config_set("not.a.key", 1, path=path) + + +def test_config_set_auto_creates_file(tmp_path: Path) -> None: + """`config set` creates a missing config.toml on first write.""" + path = tmp_path / "config.toml" + assert not path.exists() + config_set("port", 9090, path=path) + assert path.exists() + assert "port = 9090" in path.read_text() + + +def test_save_writes_sparse_doc(tmp_path: Path) -> None: + """`save` writes whatever doc is handed in — no defaults serialized.""" + path = tmp_path / "config.toml" + cfg = config._from_dict({}) + config.save(cfg, path, doc={"dflash": {"budget": 9}}) + body = path.read_text() + assert "budget = 9" in body + assert "max_ctx" not in body + + +def test_live_config_uses_recommend_preset_indirectly(tmp_path: Path) -> None: + """``live_config()`` returns a Config — no implicit preset when none given.""" + # The function probes the env-provided HostFacts; with no preset arg + # we must NOT silently pin one (that would surprise legacy installs). + cfg = config.live_config() + assert cfg.model.preset == "" + + +def test_seed_dflash_writes_heuristic_when_absent(tmp_path: Path) -> None: + """First-time activate seeds the VRAM-tier heuristic into config.toml. + + A config.toml with only [model] (what `models download --activate` + writes) would otherwise load with DflashRuntime class defaults + (max_ctx=16384), ignoring the host tier. Seeding writes the heuristic. + """ + from lucebox.types import HostFacts + + path = tmp_path / "config.toml" + path.write_text('[model]\npreset = "qwen3.6-27b"\n') + wrote = config.seed_dflash_from_host(HostFacts(vram_gb=24, is_wsl=True), path=path) + assert wrote is True + loaded = config.load(path) + assert loaded is not None + # 24 GB tier heuristic → 98304, not the 16384 class default. + assert loaded.dflash.max_ctx == 98304 + # Provenance recorded; [model] preserved. + doc = config.load_doc(path) + assert doc["autotune"]["source"] == "heuristic" + assert doc["model"]["preset"] == "qwen3.6-27b" + + +def test_seed_dflash_is_noop_when_dflash_present(tmp_path: Path) -> None: + """Never clobber a [dflash] the user or a prior tune already wrote.""" + from lucebox.types import HostFacts + + path = tmp_path / "config.toml" + path.write_text("[dflash]\nmax_ctx = 4096\n") + wrote = config.seed_dflash_from_host(HostFacts(vram_gb=80), path=path) + assert wrote is False + assert config.load(path).dflash.max_ctx == 4096 diff --git a/lucebox/tests/test_config_cli.py b/lucebox/tests/test_config_cli.py new file mode 100644 index 000000000..446ab41b6 --- /dev/null +++ b/lucebox/tests/test_config_cli.py @@ -0,0 +1,127 @@ +"""Tests for the ``lucebox config`` sub-app CLI.""" + +from __future__ import annotations + +from pathlib import Path + +import pytest +from lucebox.cli import app +from typer.testing import CliRunner + + +def _set_config_path(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: + monkeypatch.setenv("LUCEBOX_HOME", str(tmp_path)) + return tmp_path / "config.toml" + + +def test_config_set_then_get_round_trip( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + cfg_path = _set_config_path(tmp_path, monkeypatch) + set_result = CliRunner().invoke(app, ["config", "set", "dflash.budget=12"]) + assert set_result.exit_code == 0 + assert cfg_path.exists() + get_result = CliRunner().invoke(app, ["config", "get", "dflash.budget"]) + assert get_result.exit_code == 0 + assert "12" in get_result.stdout + assert "from file" in get_result.stdout + + +def test_config_get_with_no_key_lists_every_registered_key( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + _set_config_path(tmp_path, monkeypatch) + result = CliRunner().invoke(app, ["config", "get"]) + assert result.exit_code == 0 + # Every registered dotted key shows up at least once. + for key in ("model.preset", "dflash.budget", "port"): + assert key in result.stdout + + +def test_config_unset_drops_key( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + cfg_path = _set_config_path(tmp_path, monkeypatch) + CliRunner().invoke(app, ["config", "set", "dflash.budget=9"]) + assert "budget = 9" in cfg_path.read_text() + unset_result = CliRunner().invoke(app, ["config", "unset", "dflash.budget"]) + assert unset_result.exit_code == 0 + body = cfg_path.read_text() + assert "budget" not in body + + +def test_config_set_unknown_key_errors( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + _set_config_path(tmp_path, monkeypatch) + result = CliRunner().invoke(app, ["config", "set", "totally.unknown=1"]) + assert result.exit_code == 2 + + +def test_config_set_rejects_missing_equals( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + _set_config_path(tmp_path, monkeypatch) + result = CliRunner().invoke(app, ["config", "set", "dflash.budget"]) + assert result.exit_code == 2 + + +def test_config_set_creates_file_when_missing( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + cfg_path = _set_config_path(tmp_path, monkeypatch) + assert not cfg_path.exists() + CliRunner().invoke(app, ["config", "set", "port=9090"]) + assert cfg_path.exists() + assert "port = 9090" in cfg_path.read_text() + + +def test_load_or_build_env_overrides_persisted_config( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """LUCEBOX_* env vars must win over config.toml. + + Regression test for the precedence bug fixed in this commit: prior + to the fix, `_load_or_build()` returned `config_mod.load()`'s result + verbatim when config.toml existed, so the systemd unit's + `Environment=LUCEBOX_IMAGE=...` was silently ignored. Sindri's + config.toml had `[image]` without `registry`, which made the + dataclass default `ghcr.io/luce-org/lucebox-hub` win over the + intended easel image. + """ + from lucebox.cli import _load_or_build + + cfg_path = _set_config_path(tmp_path, monkeypatch) + # Write a config.toml WITHOUT an image.registry line — the + # bug-trigger shape on sindri. + cfg_path.write_text( + '[image]\nvariant = "cuda12"\n[runtime]\nport = 9090\n' + '[dflash]\nbudget = 22\n' + ) + # Env should override what config.toml says (and what dataclass + # defaults fill in for missing keys). + monkeypatch.setenv("LUCEBOX_IMAGE", "ghcr.io/myfork/lucebox-hub") + monkeypatch.setenv("LUCEBOX_PORT", "7777") + monkeypatch.setenv("LUCEBOX_CONTAINER", "lucebox-test") + cfg = _load_or_build() + assert cfg.image == "ghcr.io/myfork/lucebox-hub" # env beats dataclass default + assert cfg.port == 7777 # env beats config.toml + assert cfg.container_name == "lucebox-test" # env applied + # variant is in config.toml — config.toml value (no env override). + assert cfg.variant == "cuda12" + # dflash IS persisted in config.toml — env doesn't touch it (no DFLASH_* + # env hooks at this layer). + assert cfg.dflash.budget == 22 + + +def test_load_or_build_no_toml_env_overrides_defaults( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """When config.toml is absent, env must still override defaults.""" + from lucebox.cli import _load_or_build + + _set_config_path(tmp_path, monkeypatch) + # Don't write a config.toml — exercise the live_config() fallback. + monkeypatch.setenv("LUCEBOX_IMAGE", "ghcr.io/myfork/lucebox-hub") + cfg = _load_or_build() + assert cfg.image == "ghcr.io/myfork/lucebox-hub" diff --git a/lucebox/tests/test_docker_run.py b/lucebox/tests/test_docker_run.py new file mode 100644 index 000000000..bb888514f --- /dev/null +++ b/lucebox/tests/test_docker_run.py @@ -0,0 +1,254 @@ +"""Tests for the docker-run serve-argv builder. + +This is the core's whole job: turn a Config into the exact `docker run` +command (and DFLASH_* env) that launches the server. The argv contract is +what `lucebox serve` / the systemd unit / `print-run` all consume, so it is +pinned field-by-field here rather than only smoke-tested. +""" + +from __future__ import annotations + +from pathlib import Path + +from lucebox.download import PRESETS +from lucebox.types import Config, DflashRuntime, ModelMeta + +from lucebox import docker_run + + +def _env(spec) -> dict[str, str]: + return dict(spec.env) + + +# ── DockerRunSpec.argv ─────────────────────────────────────────────────────── + + +def test_argv_minimal_defaults() -> None: + spec = docker_run.DockerRunSpec(image="img:tag", name="box") + argv = spec.argv() + assert argv[:2] == ["docker", "run"] + assert "--rm" in argv # remove defaults True + assert ["--name", "box"] == argv[argv.index("--name") : argv.index("--name") + 2] + assert ["--gpus", "all"] == argv[argv.index("--gpus") : argv.index("--gpus") + 2] + # image is the last positional (no entrypoint_args here) + assert argv[-1] == "img:tag" + assert "-d" not in argv # detach defaults False + + +def test_argv_flags_and_ordering() -> None: + spec = docker_run.DockerRunSpec( + image="img:tag", + name="box", + gpus=False, + detach=True, + remove=False, + port_publish=(8080, 8080), + volumes=(("/host/models", "/opt/lucebox-hub/server/models"),), + env=(("DFLASH_BUDGET", "22"),), + entrypoint_args=("serve",), + extra=("--shm-size", "1g"), + ) + argv = spec.argv() + assert "--rm" not in argv # remove=False + assert "-d" in argv # detach + assert "--gpus" not in argv # gpus=False + assert ["-p", "8080:8080"] == argv[argv.index("-p") : argv.index("-p") + 2] + assert ["-v", "/host/models:/opt/lucebox-hub/server/models"] == argv[ + argv.index("-v") : argv.index("-v") + 2 + ] + assert ["-e", "DFLASH_BUDGET=22"] == argv[argv.index("-e") : argv.index("-e") + 2] + # extra flags precede the image; entrypoint_args follow it. + assert argv[-1] == "serve" + assert argv[-2] == "img:tag" + assert argv.index("--shm-size") < argv.index("img:tag") + + +def test_printable_glues_value_taking_flags() -> None: + spec = docker_run.DockerRunSpec( + image="img:tag", + name="box", + port_publish=(8080, 8080), + env=(("K", "v"),), + ) + out = spec.printable() + # one flag per line, continued with backslash-newline + assert out.startswith("docker \\\n run") + # value-taking flags keep their value on the same line + assert "--name box" in out + assert "--gpus all" in out + assert "-p 8080:8080" in out + assert "-e K=v" in out + + +# ── _runtime_volumes ───────────────────────────────────────────────────────── + + +def test_runtime_volumes_mounts_models_and_home(tmp_path: Path) -> None: + cfg = Config(models_dir=tmp_path / "models") + vols = docker_run._runtime_volumes(cfg) + assert (str(tmp_path / "models"), "/opt/lucebox-hub/server/models") in vols + # $HOME is also mounted so absolute symlink targets resolve in-container. + assert any(host == str(Path.home()) for host, _ in vols) + + +def test_runtime_volumes_dedupes_when_models_is_home(monkeypatch, tmp_path: Path) -> None: + monkeypatch.setattr(Path, "home", staticmethod(lambda: tmp_path)) + cfg = Config(models_dir=tmp_path) + vols = docker_run._runtime_volumes(cfg) + # models_dir == home → only the models mount, no duplicate home mount. + assert len(vols) == 1 + + +# ── _resolve_model_files ───────────────────────────────────────────────────── + + +def test_resolve_model_files_explicit_override_wins(tmp_path: Path) -> None: + cfg = Config( + models_dir=tmp_path, + model=ModelMeta(preset="qwen3.6-27b", target_file="custom.gguf", draft_file="d.gguf"), + ) + target, draft, draft_dir = docker_run._resolve_model_files(cfg) + assert target == "custom.gguf" + assert draft == "d.gguf" + assert draft_dir == "" + + +def test_resolve_model_files_falls_back_to_preset_registry(tmp_path: Path) -> None: + pres = PRESETS["qwen3.6-27b"] + cfg = Config(models_dir=tmp_path, model=ModelMeta(preset="qwen3.6-27b")) + target, draft, draft_dir = docker_run._resolve_model_files(cfg) + assert target == pres.target_file + assert draft == (pres.draft_file or "") + assert draft_dir == "" # no speculator dir on disk + + +def test_resolve_model_files_no_preset_no_override(tmp_path: Path) -> None: + cfg = Config(models_dir=tmp_path) # ModelMeta() defaults: all empty + assert docker_run._resolve_model_files(cfg) == ("", "", "") + + +# ── server_run_spec ────────────────────────────────────────────────────────── + + +def test_server_run_spec_top_level_shape(tmp_path: Path) -> None: + cfg = Config( + image="ghcr.io/x/lucebox-hub", + variant="cuda12", + container_name="lucebox", + port=9000, + models_dir=tmp_path, + ) + spec = docker_run.server_run_spec(cfg) + assert spec.image == "ghcr.io/x/lucebox-hub:cuda12" + assert spec.name == "lucebox" + assert spec.gpus is True + assert spec.remove is True + assert spec.detach is False + assert spec.port_publish == (9000, 8080) + assert (str(tmp_path), "/opt/lucebox-hub/server/models") in spec.volumes + + +def test_server_run_spec_always_emits_core_dflash_env(tmp_path: Path) -> None: + cfg = Config(models_dir=tmp_path, dflash=DflashRuntime(budget=22, max_ctx=32768)) + env = _env(docker_run.server_run_spec(cfg)) + assert env["DFLASH_BUDGET"] == "22" + assert env["DFLASH_MAX_CTX"] == "32768" + assert env["DFLASH_PREFIX_CACHE_SLOTS"] == "0" + assert env["DFLASH_PREFILL_CACHE_SLOTS"] == "0" + assert env["DFLASH_THINK_MAX"] == "15488" + assert env["DFLASH_PORT"] == "8080" + + +def test_server_run_spec_optional_env_off_by_default(tmp_path: Path) -> None: + env = _env(docker_run.server_run_spec(Config(models_dir=tmp_path))) + for absent in ( + "DFLASH_LAZY", + "DFLASH_CACHE_TYPE_K", + "DFLASH_CACHE_TYPE_V", + "DFLASH_PREFILL_MODE", + "DFLASH_FA_WINDOW", + "DFLASH_THINK_SOFT_CLOSE_MIN_RATIO", + "DFLASH_DEBUG_THINKING_LOGITS", + "DFLASH_TARGET", + "DFLASH_DRAFT", + ): + assert absent not in env + + +def test_server_run_spec_optional_env_emitted_when_set(tmp_path: Path) -> None: + cfg = Config( + models_dir=tmp_path, + dflash=DflashRuntime( + lazy=True, + cache_type_k="tq3_0", + cache_type_v="tq3_0", + prefill_mode="auto", + prefill_keep_ratio=0.1, + prefill_threshold=20000, + prefill_drafter="drafter.gguf", + fa_window=512, + think_soft_close_min_ratio=0.5, + debug_thinking_logits=True, + ), + ) + env = _env(docker_run.server_run_spec(cfg)) + assert env["DFLASH_LAZY"] == "1" + assert env["DFLASH_CACHE_TYPE_K"] == "tq3_0" + assert env["DFLASH_CACHE_TYPE_V"] == "tq3_0" + assert env["DFLASH_PREFILL_MODE"] == "auto" + assert env["DFLASH_PREFILL_KEEP"] == "0.1" + assert env["DFLASH_PREFILL_THRESHOLD"] == "20000" + assert env["DFLASH_PREFILL_DRAFTER"] == "drafter.gguf" + assert env["DFLASH_FA_WINDOW"] == "512" + assert env["DFLASH_THINK_SOFT_CLOSE_MIN_RATIO"] == "0.5" + assert env["DFLASH_DEBUG_THINKING_LOGITS"] == "1" + + +def test_server_run_spec_resolves_target_and_draft_paths(tmp_path: Path) -> None: + pres = PRESETS["qwen3.6-27b"] + cfg = Config(models_dir=tmp_path, model=ModelMeta(preset="qwen3.6-27b")) + env = _env(docker_run.server_run_spec(cfg)) + assert env["DFLASH_TARGET"] == f"/opt/lucebox-hub/server/models/{pres.target_file}" + if pres.draft_file: + assert env["DFLASH_DRAFT"] == ( + f"/opt/lucebox-hub/server/models/draft/{pres.draft_file}" + ) + + +def test_server_run_spec_forwards_host_env(monkeypatch, tmp_path: Path) -> None: + monkeypatch.setenv("LUCEBOX_HOST_OS_PRETTY", "Ubuntu 22.04") + monkeypatch.setenv("LUCEBOX_HOST_GPU_NAME", "RTX 5090") + env = _env(docker_run.server_run_spec(Config(models_dir=tmp_path))) + assert env["LUCEBOX_HOST_OS_PRETTY"] == "Ubuntu 22.04" + assert env["LUCEBOX_HOST_GPU_NAME"] == "RTX 5090" + + +def test_large_preset_serves_at_safe_default_ctx(tmp_path: Path) -> None: + """Regression guard for the preset-cap analysis (#5). + + Activating a preset writes only [model], never [dflash], so a loaded + Config keeps the conservative DflashRuntime() floor (max_ctx=16384). + The VRAM-tier heuristic's higher caps only apply via `autotune --apply` + (which threads cfg.model.preset and is a separate PR). This test pins + that a large preset does NOT silently serve at a high, OOM-prone ctx + through the default serve path. + """ + cfg = Config(models_dir=tmp_path, model=ModelMeta(preset="qwen3.6-27b")) + env = _env(docker_run.server_run_spec(cfg)) + assert env["DFLASH_MAX_CTX"] == "16384" + + +# ── docker_pull ────────────────────────────────────────────────────────────── + + +def test_docker_pull_shells_out_and_returns_code(monkeypatch) -> None: + seen: dict[str, list[str]] = {} + + def fake_call(argv: list[str]) -> int: + seen["argv"] = argv + return 7 + + monkeypatch.setattr(docker_run.subprocess, "call", fake_call) + rc = docker_run.docker_pull("img:tag") + assert rc == 7 + assert seen["argv"] == ["docker", "pull", "img:tag"] diff --git a/lucebox/tests/test_download.py b/lucebox/tests/test_download.py new file mode 100644 index 000000000..8b69e96b8 --- /dev/null +++ b/lucebox/tests/test_download.py @@ -0,0 +1,323 @@ +"""Tests for the model-download orchestration. + +The downloader now drives `huggingface_hub.hf_hub_download` directly +(no subprocess) and verifies size + sha256 against the repo metadata +before re-fetching. The tests stub out the network calls so the +behavior contract — what gets requested, when downloads are skipped — +stays pinned without actually talking to the Hub. +""" + +from pathlib import Path +from types import SimpleNamespace + +import pytest +from lucebox.download import ( + DEFAULT_PRESET, + PRESETS, + recommend_preset, + resolve_preset, + status, +) +from lucebox.types import HostFacts + +from lucebox import download + + +def test_default_preset_uses_quantized_gguf_draft(): + assert DEFAULT_PRESET.draft_repo == "spiritbuun/Qwen3.6-27B-DFlash-GGUF" + assert DEFAULT_PRESET.draft_file == "dflash-draft-3.6-q4_k_m.gguf" + + +def test_default_preset_is_registered_under_qwen_name(): + assert DEFAULT_PRESET is PRESETS["qwen3.6-27b"] + assert DEFAULT_PRESET.name == "qwen3.6-27b" + + +def test_resolve_preset_returns_default_on_none(): + assert resolve_preset(None) is DEFAULT_PRESET + assert resolve_preset("") is DEFAULT_PRESET + + +def test_resolve_preset_picks_gemma_target_and_draft(): + pres = resolve_preset("gemma-4-26b") + assert pres.name == "gemma-4-26b" + assert pres.target_repo == "bartowski/google_gemma-4-26B-A4B-it-GGUF" + assert pres.target_file == "google_gemma-4-26B-A4B-it-Q4_K_M.gguf" + assert pres.draft_repo == "Lucebox/gemma-4-26B-A4B-it-DFlash-GGUF" + assert pres.draft_file == "gemma-4-26B-A4B-it-DFlash-q8_0.gguf" + assert pres.has_draft + + +def test_resolve_preset_supports_target_only_laguna(): + pres = resolve_preset("laguna-xs.2") + assert pres.target_repo == "Lucebox/Laguna-XS.2-GGUF" + assert pres.draft_repo is None + assert not pres.has_draft + + +def test_resolve_preset_picks_qwen36_moe_target_only(): + """Qwen3.6 MoE preset routes to unsloth's UD-Q4_K_M file, no draft. + + The MoE variant has no published DFlash draft GGUF (verified against + HfApi.repo_info 2026-05-28), so it runs target-only like Laguna. The + file stem is `Qwen3.6-35B-A3B-UD-Q4_K_M.gguf` — the unsloth repo only + publishes the UD ("unsloth dynamic") family at Q4_K_M, not a plain + `Q4_K_M.gguf`. + """ + pres = resolve_preset("qwen3.6-moe") + assert pres.name == "qwen3.6-moe" + assert pres.target_repo == "unsloth/Qwen3.6-35B-A3B-GGUF" + assert pres.target_file == "Qwen3.6-35B-A3B-UD-Q4_K_M.gguf" + assert pres.draft_repo is None + assert pres.draft_file is None + assert not pres.has_draft + + +def test_download_preset_target_only_qwen36_moe_skips_draft(tmp_path, monkeypatch): + """qwen3.6-moe behaves identically to laguna-xs.2: target only, no draft fetch.""" + cfg = SimpleNamespace(models_dir=tmp_path) + pres = resolve_preset("qwen3.6-moe") + assert not pres.has_draft + fetches: list[tuple[str, str]] = [] + + def _meta(_api, repo_id: str, filename: str) -> tuple[int, None]: + return 10, None + + def _stub_fetch(api, repo_id, filename, local_dir, console): # noqa: ARG001 + fetches.append((repo_id, filename)) + out = local_dir / filename + out.parent.mkdir(parents=True, exist_ok=True) + with out.open("wb") as f: + f.truncate(10) + return out + + monkeypatch.setattr(download, "_file_meta", _meta) + monkeypatch.setattr(download, "_fetch", _stub_fetch) + + assert download.download_preset(cfg, pres) == 0 + # Only the target — no draft attempt at all. + assert fetches == [(pres.target_repo, pres.target_file)] + + +def test_status_qwen36_moe_reports_draft_present_when_target_only(tmp_path, monkeypatch): + """No published draft → status reports draft_present=True (nothing to fetch).""" + cfg = SimpleNamespace(models_dir=tmp_path) + pres = resolve_preset("qwen3.6-moe") + + def _meta(_api, repo_id: str, filename: str) -> tuple[int, None]: + return 22 * 10**9, None + + monkeypatch.setattr(download, "_file_meta", _meta) + # Target absent → target_present False, draft_present True (no draft). + assert status(cfg, pres) == {"target_present": False, "draft_present": True} + + +def test_resolve_preset_unknown_name_lists_known_options(): + with pytest.raises(KeyError) as exc_info: + resolve_preset("qwen-99b") + msg = str(exc_info.value) + # Every registered preset must appear in the suggestion list so the + # user can copy-paste the right name. + for name in PRESETS: + assert name in msg + + +def _stub_file_meta(target_size: int, draft_size: int): + """Build a `_file_meta` replacement that returns (size, None) per repo+file. + + sha256 is left None so tests don't need to compute real hashes; the + real metadata path is exercised by the live `models download` + invocation, not the unit tests. + """ + + def _meta(_api, repo_id: str, filename: str) -> tuple[int, None]: + if repo_id == DEFAULT_PRESET.target_repo and filename == DEFAULT_PRESET.target_file: + return target_size, None + if repo_id == DEFAULT_PRESET.draft_repo and filename == DEFAULT_PRESET.draft_file: + return draft_size, None + raise FileNotFoundError(f"unexpected ({repo_id}, {filename})") + + return _meta + + +def test_status_checks_default_draft_gguf(tmp_path, monkeypatch): + cfg = SimpleNamespace(models_dir=tmp_path) + draft_dir = tmp_path / "draft" + draft_dir.mkdir() + target = tmp_path / DEFAULT_PRESET.target_file + draft = draft_dir / DEFAULT_PRESET.draft_file + + monkeypatch.setattr(download, "_file_meta", _stub_file_meta(target_size=1024, draft_size=512)) + + # Neither file exists yet. + assert status(cfg) == {"target_present": False, "draft_present": False} + + # Write files at the expected sizes. + with target.open("wb") as f: + f.truncate(1024) + with draft.open("wb") as f: + f.truncate(512) + assert status(cfg) == {"target_present": True, "draft_present": True} + + +def test_status_rejects_partial_model_files(tmp_path, monkeypatch): + cfg = SimpleNamespace(models_dir=tmp_path) + draft_dir = tmp_path / "draft" + draft_dir.mkdir() + target = tmp_path / DEFAULT_PRESET.target_file + draft = draft_dir / DEFAULT_PRESET.draft_file + target.write_bytes(b"partial") + draft.write_bytes(b"partial") + + # Repo says the target is 1 GB; a 7-byte file is partial, not present. + monkeypatch.setattr( + download, "_file_meta", _stub_file_meta(target_size=10**9, draft_size=10**6) + ) + assert status(cfg) == {"target_present": False, "draft_present": False} + + +def test_current_bytes_reads_xet_staging_path(tmp_path): + """Regression: progress polling must see hf-xet's hashed staging file. + + huggingface_hub 1.x writes partial Xet downloads to + ``{local_dir}/.cache/huggingface/download/{short_hash}.{etag}.incomplete`` + — NOT to ``{local_dir}/{filename}.incomplete``. Before the fix the + polling code only checked the latter (which never appears) so the + Rich progress bar sat at 0 bytes for the entire transfer. + """ + filename = "model.gguf" + etag = "abc123" + candidates = download._incomplete_path_candidates(tmp_path, filename, etag) + # The first candidate must point at the actual hf-xet staging path. + xet_path: Path = candidates[0] + assert xet_path.parent == tmp_path / ".cache" / "huggingface" / "download" + assert xet_path.name.endswith(f".{etag}.incomplete") + + # Now: writing to that path must be observed by _current_bytes. + xet_path.parent.mkdir(parents=True, exist_ok=True) + xet_path.write_bytes(b"x" * 4096) + target = tmp_path / filename + assert download._current_bytes(target, candidates) == 4096 + + +def test_current_bytes_falls_back_to_glob_without_etag(tmp_path): + """When sha256 is unknown we still find growing .incomplete files.""" + filename = "model.gguf" + candidates = download._incomplete_path_candidates(tmp_path, filename, etag=None) + target = tmp_path / filename + + staging = tmp_path / ".cache" / "huggingface" / "download" + staging.mkdir(parents=True, exist_ok=True) + (staging / "deadbeef.deadbeef.incomplete").write_bytes(b"x" * 8192) + assert download._current_bytes(target, candidates) == 8192 + + +def test_current_bytes_prefers_final_target_when_complete(tmp_path): + filename = "model.gguf" + candidates = download._incomplete_path_candidates(tmp_path, filename, etag="abc") + target = tmp_path / filename + target.write_bytes(b"x" * 1234) + assert download._current_bytes(target, candidates) == 1234 + + +def test_download_preset_fetches_exact_draft_file(tmp_path, monkeypatch): + cfg = SimpleNamespace(models_dir=tmp_path) + fetches: list[tuple[str, str, str]] = [] + + monkeypatch.setattr(download, "_file_meta", _stub_file_meta(target_size=10, draft_size=10)) + + # Stub the actual download to record what was requested + create a stub + # file of the expected size so `_local_matches` would pass on a re-run. + def _stub_fetch(api, repo_id, filename, local_dir, console): # noqa: ARG001 + fetches.append((repo_id, filename, str(local_dir))) + target = local_dir / filename + target.parent.mkdir(parents=True, exist_ok=True) + with target.open("wb") as f: + f.truncate(10) + return target + + monkeypatch.setattr(download, "_fetch", _stub_fetch) + + assert download.download_preset(cfg) == 0 + assert (DEFAULT_PRESET.target_repo, DEFAULT_PRESET.target_file, str(tmp_path)) in fetches + assert ( + DEFAULT_PRESET.draft_repo, + DEFAULT_PRESET.draft_file, + str(tmp_path / "draft"), + ) in fetches + + +def test_download_preset_routes_gemma_preset_to_correct_repos(tmp_path, monkeypatch): + cfg = SimpleNamespace(models_dir=tmp_path) + pres = resolve_preset("gemma-4-26b") + fetches: list[tuple[str, str, str]] = [] + + def _meta(_api, repo_id: str, filename: str) -> tuple[int, None]: + return 10, None + + def _stub_fetch(api, repo_id, filename, local_dir, console): # noqa: ARG001 + fetches.append((repo_id, filename, str(local_dir))) + out = local_dir / filename + out.parent.mkdir(parents=True, exist_ok=True) + with out.open("wb") as f: + f.truncate(10) + return out + + monkeypatch.setattr(download, "_file_meta", _meta) + monkeypatch.setattr(download, "_fetch", _stub_fetch) + + assert download.download_preset(cfg, pres) == 0 + assert (pres.target_repo, pres.target_file, str(tmp_path)) in fetches + assert (pres.draft_repo, pres.draft_file, str(tmp_path / "draft")) in fetches + + +def test_download_preset_target_only_skips_draft_fetch(tmp_path, monkeypatch): + cfg = SimpleNamespace(models_dir=tmp_path) + pres = resolve_preset("laguna-xs.2") + assert not pres.has_draft + fetches: list[tuple[str, str]] = [] + + def _meta(_api, repo_id: str, filename: str) -> tuple[int, None]: + return 10, None + + def _stub_fetch(api, repo_id, filename, local_dir, console): # noqa: ARG001 + fetches.append((repo_id, filename)) + out = local_dir / filename + out.parent.mkdir(parents=True, exist_ok=True) + with out.open("wb") as f: + f.truncate(10) + return out + + monkeypatch.setattr(download, "_file_meta", _meta) + monkeypatch.setattr(download, "_fetch", _stub_fetch) + + assert download.download_preset(cfg, pres) == 0 + # Target fetched, no draft fetch attempted at all. + assert fetches == [(pres.target_repo, pres.target_file)] + + +def test_status_target_only_preset_reports_draft_as_present(tmp_path, monkeypatch): + cfg = SimpleNamespace(models_dir=tmp_path) + pres = resolve_preset("laguna-xs.2") + + def _meta(_api, repo_id: str, filename: str) -> tuple[int, None]: + return 1024, None + + monkeypatch.setattr(download, "_file_meta", _meta) + # Target absent → target_present False, draft_present True (nothing to download). + assert status(cfg, pres) == {"target_present": False, "draft_present": True} + + +def test_recommend_preset_tiers() -> None: + """First-run preset recommendation is a pure VRAM-tier function. + + 22 GB+ → the Lucebox default (qwen3.6-27b); 16-21 GB → laguna-xs.2; + below 16 GB → None (the registered presets need ≥16 GB, so we punt to + an explicit choice rather than recommend something that can't run). + """ + assert recommend_preset(HostFacts(vram_gb=24)) == "qwen3.6-27b" + assert recommend_preset(HostFacts(vram_gb=22)) == "qwen3.6-27b" + assert recommend_preset(HostFacts(vram_gb=20)) == "laguna-xs.2" + assert recommend_preset(HostFacts(vram_gb=16)) == "laguna-xs.2" + assert recommend_preset(HostFacts(vram_gb=12)) is None + assert recommend_preset(HostFacts(vram_gb=0)) is None diff --git a/lucebox/tests/test_models_cli.py b/lucebox/tests/test_models_cli.py new file mode 100644 index 000000000..f44583044 --- /dev/null +++ b/lucebox/tests/test_models_cli.py @@ -0,0 +1,142 @@ +"""Tests for the ``lucebox models`` sub-app.""" + +from __future__ import annotations + +from pathlib import Path + +import pytest +from lucebox.cli import app +from lucebox.download import PRESETS +from lucebox.types import HostFacts +from typer.testing import CliRunner + +from lucebox import config as config_mod +from lucebox import download as download_mod + + +def _set_config_path(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> Path: + monkeypatch.setenv("LUCEBOX_HOME", str(tmp_path)) + monkeypatch.setenv("LUCEBOX_MODELS", str(tmp_path / "models")) + return tmp_path / "config.toml" + + +def _stub_host(monkeypatch: pytest.MonkeyPatch, vram_gb: int) -> None: + monkeypatch.setattr("lucebox.host_facts.from_env", lambda: HostFacts(vram_gb=vram_gb)) + monkeypatch.setattr("lucebox.cli.from_env", lambda: HostFacts(vram_gb=vram_gb)) + + +def test_models_list_shows_every_registered_preset( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + _set_config_path(tmp_path, monkeypatch) + _stub_host(monkeypatch, vram_gb=24) + result = CliRunner().invoke(app, ["models", "list"]) + assert result.exit_code == 0 + for name in PRESETS: + assert name in result.stdout + + +def test_models_default_view_lists_only_installed( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + _set_config_path(tmp_path, monkeypatch) + _stub_host(monkeypatch, vram_gb=24) + # No models on disk → default view says "no presets installed". + result = CliRunner().invoke(app, ["models"]) + assert result.exit_code == 0 + assert "No presets installed" in result.stdout or "Models dir" in result.stdout + + +def test_models_download_recommends_when_empty( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """No preset configured + nothing on argv → auto-recommend + auto-activate.""" + cfg_path = _set_config_path(tmp_path, monkeypatch) + _stub_host(monkeypatch, vram_gb=24) + + # Stub the network calls so the test doesn't try to talk to HF. + monkeypatch.setattr(download_mod, "download_preset", lambda cfg, pres: 0) + monkeypatch.setattr( + download_mod, + "status", + lambda cfg, pres: {"target_present": True, "draft_present": True}, + ) + + result = CliRunner().invoke(app, ["models", "download"]) + assert result.exit_code == 0 + assert "Recommended preset" in result.stdout + assert cfg_path.exists() + # The active preset should now be model.preset = qwen3.6-27b. + entries = config_mod.config_get(path=cfg_path) + assert entries["model.preset"] == ("qwen3.6-27b", "file") + + +def test_models_download_refuses_silent_switch( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """When a preset is already active, `download` with no arg refuses.""" + cfg_path = _set_config_path(tmp_path, monkeypatch) + _stub_host(monkeypatch, vram_gb=24) + config_mod.config_set("model.preset", "qwen3.6-27b", path=cfg_path) + + result = CliRunner().invoke(app, ["models", "download"]) + assert result.exit_code == 2 + assert "already active" in result.stdout.lower() + + +def test_models_download_explicit_preset_no_activate( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """Passing a preset without --activate downloads but doesn't flip model.preset.""" + cfg_path = _set_config_path(tmp_path, monkeypatch) + _stub_host(monkeypatch, vram_gb=24) + monkeypatch.setattr(download_mod, "download_preset", lambda cfg, pres: 0) + monkeypatch.setattr( + download_mod, + "status", + lambda cfg, pres: {"target_present": False, "draft_present": False}, + ) + + result = CliRunner().invoke(app, ["models", "download", "gemma-4-26b"]) + assert result.exit_code == 0 + if cfg_path.exists(): + entries = config_mod.config_get(path=cfg_path) + assert entries["model.preset"] == ("", "default") + + +def test_models_download_explicit_preset_with_activate( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + cfg_path = _set_config_path(tmp_path, monkeypatch) + _stub_host(monkeypatch, vram_gb=24) + monkeypatch.setattr(download_mod, "download_preset", lambda cfg, pres: 0) + monkeypatch.setattr( + download_mod, + "status", + lambda cfg, pres: {"target_present": False, "draft_present": False}, + ) + + result = CliRunner().invoke(app, ["models", "download", "gemma-4-26b", "--activate"]) + assert result.exit_code == 0 + entries = config_mod.config_get(path=cfg_path) + assert entries["model.preset"] == ("gemma-4-26b", "file") + + +def test_installed_helpers_track_presence( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """``installed_status`` / ``installed_size_gb`` reflect on-disk byte counts.""" + _set_config_path(tmp_path, monkeypatch) + _stub_host(monkeypatch, vram_gb=24) + from lucebox.config import live_config + + cfg = live_config() + cfg.models_dir.mkdir(parents=True, exist_ok=True) + laguna = PRESETS["laguna-xs.2"] + assert download_mod.installed_status(cfg, laguna) == "absent" + + target = cfg.models_dir / laguna.target_file + target.parent.mkdir(parents=True, exist_ok=True) + target.write_bytes(b"x" * (5 * 10**9)) + assert download_mod.installed_status(cfg, laguna) == "installed" + assert download_mod.installed_size_gb(cfg, laguna) == pytest.approx(5.0, rel=0.01) diff --git a/pyproject.toml b/pyproject.toml index 56ae2bf4f..520838041 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -8,6 +8,7 @@ license = { text = "Apache-2.0" } authors = [{ name = "Lucebox" }] dependencies = [ + "lucebox", "lucebox-dflash", "pflash", ] @@ -23,7 +24,7 @@ line-length = 100 # server-internal and optimization Python (server/scripts, optimizations/*) # carries pre-existing style debt and is added to `include` as it is cleaned # up. Vendored deps stay excluded permanently (extend-exclude below). -include = ["harness/**/*.py", "scripts/**/*.py"] +include = ["harness/**/*.py", "scripts/**/*.py", "lucebox/**/*.py"] extend-exclude = [ "dflash/deps", "megakernel", @@ -51,11 +52,12 @@ package = false no-build-isolation-package = ["qwen35-megakernel-bf16"] [tool.uv.workspace] -# Workspace members. Keeping the list to the packages that live in this -# repo lets `uv lock --check` / `uv sync --frozen` pass. -members = ["server", "optimizations/megakernel", "optimizations/pflash"] +# Workspace members. PR adds the lucebox/ package alongside the existing +# server / megakernel / pflash members. +members = ["lucebox", "server", "optimizations/megakernel", "optimizations/pflash"] [tool.uv.sources] +lucebox = { workspace = true } lucebox-dflash = { workspace = true } pflash = { workspace = true } qwen35-megakernel-bf16 = { workspace = true } diff --git a/scripts/check_lucebox_wrapper_sandbox.sh b/scripts/check_lucebox_wrapper_sandbox.sh new file mode 100755 index 000000000..df2b2b9bc --- /dev/null +++ b/scripts/check_lucebox_wrapper_sandbox.sh @@ -0,0 +1,242 @@ +#!/usr/bin/env bash +# Exercise the host-side lucebox.sh installer/wrapper from an isolated prefix. +# +# The script intentionally runs from a throwaway HOME, XDG_CONFIG_HOME, +# LUCEBOX_HOME, model directory, and working directory. That catches accidental +# dependencies on the checkout or the user's real ~/.lucebox while keeping the +# test reproducible enough to paste into a bug report. + +set -euo pipefail + +IMAGE="${LUCEBOX_TEST_IMAGE:-ghcr.io/easel/lucebox-hub}" +VARIANT="${LUCEBOX_TEST_VARIANT:-integration-props-uv-squared-clean-cuda12}" +WRAPPER_SOURCE="${LUCEBOX_TEST_WRAPPER_SOURCE:-local}" +RUN_PULL="${LUCEBOX_TEST_RUN_PULL:-1}" +RUN_CONTAINER_CLI="${LUCEBOX_TEST_RUN_CONTAINER_CLI:-1}" +KEEP_SANDBOX="${LUCEBOX_TEST_KEEP_SANDBOX:-0}" + +ROOT="" +LOG="" + +usage() { + cat <&2; usage >&2; exit 2 ;; + esac +done + +die() { + echo "[FAIL] $*" >&2 + if [ -n "$LOG" ] && [ -f "$LOG" ]; then + echo "[FAIL] transcript: $LOG" >&2 + fi + exit 1 +} + +note() { + printf '[INFO] %s\n' "$*" +} + +pass() { + printf '[PASS] %s\n' "$*" +} + +assert_file() { + [ -f "$1" ] || die "missing file: $1" + pass "file exists: $1" +} + +assert_contains() { + local file="$1" + local pattern="$2" + if ! grep -Fq "$pattern" "$file"; then + echo "----- $file -----" >&2 + sed -n '1,220p' "$file" >&2 || true + echo "-----------------" >&2 + die "expected '$pattern' in $file" + fi + pass "$file contains: $pattern" +} + +run_logged() { + note "run: $*" + { + printf '\n===== %s =====\n' "$*" + "$@" + printf '===== exit=0 =====\n' + } 2>&1 | tee -a "$LOG" +} + +run_logged_capture() { + local out="$1" + shift + note "run: $* > $out" + { + printf '\n===== %s > %s =====\n' "$*" "$out" + "$@" + local rc=$? + printf '===== exit=%s =====\n' "$rc" + return "$rc" + } 2>&1 | tee "$out" | tee -a "$LOG" >/dev/null +} + +cleanup() { + if [ -n "$ROOT" ] && [ "$KEEP_SANDBOX" != "1" ]; then + rm -rf "$ROOT" + elif [ -n "$ROOT" ]; then + note "kept sandbox: $ROOT" + note "transcript: $LOG" + fi +} +trap cleanup EXIT + +REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +ROOT="$(mktemp -d "${TMPDIR:-/tmp}/lucebox-wrapper-sandbox.XXXXXX")" +LOG="$ROOT/transcript.log" + +HOME_DIR="$ROOT/home" +BIN_DIR="$ROOT/bin" +XDG_DIR="$ROOT/xdg" +MODELS_DIR="$ROOT/models" +WORK_DIR="$ROOT/work" +mkdir -p "$HOME_DIR" "$BIN_DIR" "$XDG_DIR" "$MODELS_DIR" "$WORK_DIR" + +note "sandbox: $ROOT" +note "transcript: $LOG" + +case "$WRAPPER_SOURCE" in + local) + cp "$REPO_ROOT/lucebox.sh" "$BIN_DIR/lucebox" + ;; + http://*|https://*) + curl -fsSL "$WRAPPER_SOURCE" -o "$BIN_DIR/lucebox" + ;; + *) + cp "$WRAPPER_SOURCE" "$BIN_DIR/lucebox" + ;; +esac +chmod +x "$BIN_DIR/lucebox" + +FIRST_LINE="$(head -n 1 "$BIN_DIR/lucebox")" +[ "$FIRST_LINE" = "#!/usr/bin/env bash" ] || die "unexpected shebang: $FIRST_LINE" +pass "wrapper has expected shebang" + +export HOME="$HOME_DIR" +export XDG_CONFIG_HOME="$XDG_DIR" +export LUCEBOX_HOME="$HOME_DIR/.lucebox" +export LUCEBOX_MODELS="$MODELS_DIR" +export LUCEBOX_IMAGE="$IMAGE" +export LUCEBOX_VARIANT="$VARIANT" +export LUCEBOX_CONTAINER="lucebox-sandbox" +export LUCEBOX_PORT="18080" +export PATH="$BIN_DIR:$PATH" + +cd "$WORK_DIR" +[ "$PWD" = "$WORK_DIR" ] || die "failed to enter sandbox workdir" +pass "working directory isolated: $PWD" + +run_logged_capture "$ROOT/version.out" lucebox version +assert_contains "$ROOT/version.out" "0.2.0" + +run_logged_capture "$ROOT/help.out" lucebox help +assert_contains "$ROOT/help.out" "LUCEBOX_VARIANT" +assert_contains "$ROOT/help.out" "LUCEBOX_IMAGE" + +docker manifest inspect "${IMAGE}:${VARIANT}" >/dev/null +pass "image manifest exists: ${IMAGE}:${VARIANT}" + +if [ "$RUN_PULL" = "1" ]; then + run_logged_capture "$ROOT/pull.out" lucebox pull + assert_contains "$ROOT/pull.out" "${IMAGE}:${VARIANT}" +fi + +if [ "$RUN_CONTAINER_CLI" = "1" ]; then + run_logged_capture "$ROOT/check.out" lucebox check + # Sparse persistence: `config set` creates config.toml with only the + # named key. Replaces the old `configure --overwrite` path. + run_logged_capture "$ROOT/config-image.out" lucebox config set "image=$IMAGE" + run_logged_capture "$ROOT/config-variant.out" lucebox config set "variant=$VARIANT" + assert_file "$LUCEBOX_HOME/config.toml" + [ "$(stat -c '%u' "$LUCEBOX_HOME/config.toml")" = "$(id -u)" ] \ + || die "config.toml is not owned by the invoking user" + pass "config.toml ownership matches invoking user" + assert_contains "$LUCEBOX_HOME/config.toml" "registry = \"$IMAGE\"" + assert_contains "$LUCEBOX_HOME/config.toml" "variant = \"$VARIANT\"" + + run_logged_capture "$ROOT/print-run.out" lucebox print-run + assert_contains "$ROOT/print-run.out" "${IMAGE}:${VARIANT}" + assert_contains "$ROOT/print-run.out" "$MODELS_DIR:/opt/lucebox-hub/dflash/models" + if grep -Fq "$REPO_ROOT" "$ROOT/print-run.out"; then + die "print-run leaked repository path: $REPO_ROOT" + fi + pass "print-run did not reference repository checkout" +fi + +# Exercise `lucebox install` without allowing it to call real systemctl, +# loginctl, docker, or nvidia-smi. The generated user unit must land under the +# sandbox XDG_CONFIG_HOME and point ExecStart at the sandbox-installed wrapper. +SHIM_DIR="$ROOT/shims" +mkdir -p "$SHIM_DIR" +cat > "$SHIM_DIR/docker" <<'EOF' +#!/usr/bin/env bash +case "${1:-}" in + info) exit 0 ;; + version) echo "25.0.0"; exit 0 ;; + stop) exit 0 ;; + *) echo "docker shim: $*" >&2; exit 0 ;; +esac +EOF +cat > "$SHIM_DIR/nvidia-smi" <<'EOF' +#!/usr/bin/env bash +case "$*" in + *"--query-gpu=name,memory.total,driver_version,compute_cap"*) + echo "Fake GPU, 24576, 555.42.01, 8.6"; exit 0 ;; + *"--query-gpu=name"*) + echo "Fake GPU"; exit 0 ;; + *) echo "Fake GPU"; exit 0 ;; +esac +EOF +cat > "$SHIM_DIR/systemctl" <<'EOF' +#!/usr/bin/env bash +if [ "$1" = "--user" ] && [ "$2" = "show-environment" ]; then exit 0; fi +if [ "$1" = "--user" ] && [ "$2" = "daemon-reload" ]; then exit 0; fi +echo "systemctl shim: $*" >&2 +exit 0 +EOF +cat > "$SHIM_DIR/loginctl" <<'EOF' +#!/usr/bin/env bash +echo "Linger=no" +EOF +chmod +x "$SHIM_DIR/docker" "$SHIM_DIR/nvidia-smi" "$SHIM_DIR/systemctl" "$SHIM_DIR/loginctl" + +PATH="$SHIM_DIR:$BIN_DIR:$PATH" run_logged_capture "$ROOT/install.out" lucebox install +UNIT="$XDG_CONFIG_HOME/systemd/user/lucebox.service" +assert_file "$UNIT" +assert_contains "$UNIT" "ExecStart=$BIN_DIR/lucebox serve" +assert_contains "$UNIT" "ExecStop=$SHIM_DIR/docker stop -t 30 lucebox-sandbox" +assert_contains "$ROOT/install.out" "Installed $UNIT" + +pass "sandbox wrapper check completed" +note "summary: image=${IMAGE}:${VARIANT} wrapper_source=${WRAPPER_SOURCE}" diff --git a/scripts/test_lucebox_sh.sh b/scripts/test_lucebox_sh.sh new file mode 100755 index 000000000..3960760f6 --- /dev/null +++ b/scripts/test_lucebox_sh.sh @@ -0,0 +1,1130 @@ +#!/usr/bin/env bash +# scripts/test_lucebox_sh.sh — smoke tests for the host-side wrapper + +# every other bash script we ship. +# +# Catches regressions like: +# * syntax errors (bash -n) +# * shellcheck error-level findings across every shipped bash script +# * `set -u` violations in command paths that don't need docker/nvidia — +# each subcommand dispatch is exercised in isolation to verify no +# LUCEBOX_HOST_* or DFLASH_* read fires before the helper that should +# populate it has run. +# * missing dispatch handlers (help, version, check, usage) +# * stale references to subcommands removed from main's case +# +# The wrapper is shell + has zero non-coreutils deps for the host-only +# commands, so this script doesn't need docker/nvidia/systemd present — +# probe_host degrades cleanly when those aren't found, and the +# formatter must render fine for the "everything is missing" case too. +# +# Run from anywhere: scripts/test_lucebox_sh.sh + +set -euo pipefail + +# Resolve repo root + script under test. +ROOT="$(git rev-parse --show-toplevel 2>/dev/null || (cd "$(dirname "$0")/.." && pwd))" +SCRIPT="$ROOT/lucebox.sh" +ENTRYPOINT="$ROOT/server/scripts/entrypoint.sh" +INSTALLER="$ROOT/install.sh" + +if [ ! -f "$SCRIPT" ]; then + echo "FAIL: lucebox.sh not found at $SCRIPT" >&2 + exit 1 +fi + +# entrypoint.sh ships with the docker-stack PR (#334). When it's absent +# (e.g. on the lucebox-cli branch in isolation), skip the entire suite — +# every section below either references $ENTRYPOINT in shellcheck targets, +# parses it with `bash -n`, or sources/dispatches into it directly. The +# host-only lucebox.sh wrapper itself is covered by lucebox.sh's own unit +# tests; this script's value is the wrapper↔entrypoint contract. +if [ ! -f "$ENTRYPOINT" ]; then + echo "Skipping entrypoint tests: server/scripts/entrypoint.sh not present (provided by #334 docker-stack)" + exit 0 +fi + +fail=0 +pass=0 +report() { + if [ "$1" = "ok" ]; then + printf ' \033[1;32m✓\033[0m %s\n' "$2" + pass=$((pass + 1)) + else + printf ' \033[1;31m✗\033[0m %s\n' "$2" + if [ -n "${3:-}" ]; then + printf ' %s\n' "$3" + fi + fail=$((fail + 1)) + fi +} + +# Helper: run the wrapper with strict bash, capture stdout+stderr, check for +# (a) zero exit code, (b) substring match. NO_COLOR is set so colour codes +# don't pollute substring matches. +assert_runs() { + local label="$1" cmd="$2" expect="${3:-}" + local out rc + out=$(NO_COLOR=1 bash -c "$cmd" 2>&1) + rc=$? + if [ "$rc" -ne 0 ]; then + report fail "$label" "exit $rc; output: $(printf '%s' "$out" | head -3)" + return + fi + if [ -n "$expect" ] && ! grep -qF "$expect" <<<"$out"; then + report fail "$label" "missing expected substring '$expect'; got: $(printf '%s' "$out" | head -3)" + return + fi + report ok "$label" +} + +# Helper: run a subcommand whose successful completion would normally need +# docker / nvidia / systemd. We only care that the bash dispatch up to the +# point of the missing dependency does NOT trip `set -u`. Exit code is +# allowed to be non-zero; what we forbid is a raw "unbound variable" / +# "syntax error" / "line N:" leak in the captured output. +# +# Wrapped in `timeout` so subcommands that exec into a follow-style binary +# (logs → journalctl -f, status when systemd is healthy, etc.) don't hang +# the test runner on a dev box where the underlying tools succeed. +assert_no_set_u_leak() { + local label="$1" + shift + local out + out=$(NO_COLOR=1 timeout 5 bash "$@" 2>&1 || true) + # The "line N:" pattern is anchored to a script-path prefix to avoid + # false positives from journalctl output ("systemd[1385106]:") which + # contains a similar shape but isn't a bash error. Bash always emits + # the source filename before the line number, e.g. + # /tmp/lbh-flat/lucebox.sh: line 200: VAR: unbound variable + if grep -qE 'unbound variable|syntax error|\.sh: line [0-9]+:' <<<"$out"; then + report fail "$label" "raw bash error leaked: $(head -3 <<<"$out")" + else + report ok "$label" + fi +} + +echo "[test_lucebox_sh] running against $SCRIPT" + +# ── 1. shellcheck ───────────────────────────────────────────────────────── +# Run shellcheck across every bash script we ship (the wrapper, the +# in-container entrypoint, and every helper under scripts/). Error-level +# findings fail the build; warnings are informational only — those have +# been triaged and the SC2034/SC2155/SC2164 hits in sweep_ds4_2case.sh +# aren't user-visible bugs. +SHELLCHECK_TARGETS=( + "$SCRIPT" + "$ENTRYPOINT" + "$INSTALLER" +) +# Add every scripts/*.sh except this one (don't recurse into our own tests). +while IFS= read -r -d '' f; do + [ "$f" = "${BASH_SOURCE[0]}" ] && continue + SHELLCHECK_TARGETS+=("$f") +done < <(find "$ROOT/scripts" -maxdepth 1 -name '*.sh' -type f -print0 2>/dev/null) +SHELLCHECK_TARGETS+=("${BASH_SOURCE[0]}") + +if command -v shellcheck >/dev/null 2>&1; then + sc_out=$(shellcheck --severity=error "${SHELLCHECK_TARGETS[@]}" 2>&1) || sc_rc=$? + sc_rc="${sc_rc:-0}" + if [ "$sc_rc" -eq 0 ]; then + report ok "shellcheck --severity=error (${#SHELLCHECK_TARGETS[@]} files)" + else + report fail "shellcheck --severity=error" "$(printf '%s' "$sc_out" | head -10)" + fi +else + report fail "shellcheck not installed" "install via 'apt-get install -y shellcheck' (Ubuntu) or 'brew install shellcheck'" +fi + +# ── 2. Syntax / parse ───────────────────────────────────────────────────── +if bash -n "$SCRIPT"; then report ok "bash -n lucebox.sh parses cleanly" +else report fail "bash -n lucebox.sh"; fi +if bash -n "$ENTRYPOINT"; then report ok "bash -n entrypoint.sh parses cleanly" +else report fail "bash -n entrypoint.sh"; fi + +# ── 3. Trivial subcommands (zero-exit expected) ─────────────────────────── +assert_runs "help" "bash '$SCRIPT' help" "host-side wrapper" +assert_runs "--help" "bash '$SCRIPT' --help" "host-side wrapper" +assert_runs "-h" "bash '$SCRIPT' -h" "host-side wrapper" +assert_runs "version" "bash '$SCRIPT' version" "" +assert_runs "--version" "bash '$SCRIPT' --version" "" + +# ── 4. check — host-only, must run to completion even without docker/nvidia. +# This is the path that broke last time (multi-byte glyph + set -u). +assert_runs "check" "bash '$SCRIPT' check" "host readiness report" + +# ── 5. systemd-surface subcommands — every one of these used to crash with +# `LUCEBOX_HOST_HAS_SYSTEMD: unbound variable` because cmd_systemctl_passthrough +# / cmd_logs / cmd_systemd_uninstall reached require_systemd without first +# calling probe_host. The fix routes through require_systemd → probe_host +# when the var is unset; these tests pin that invariant. +# +# On the bare runner there is no user systemd, no installed unit, and no +# docker — so every command is expected to exit non-zero with a CLEAN error +# message. What we forbid is a raw bash "unbound variable" leak. +for sub in start stop restart enable disable status install uninstall; do + assert_no_set_u_leak "$sub dispatch (no set -u leak)" "$SCRIPT" "$sub" +done +# `logs` is special: it execs `journalctl -f` which streams every historical +# journal record for the unit. On a dev box where the lucebox service has +# actually run, that stream contains every past error — including the very +# bugs this test exists to prevent — and we'd false-positive on them. Pass +# `-n 0 --no-pager` so we only see new entries (none, in the test window). +assert_no_set_u_leak "logs dispatch (no set -u leak)" "$SCRIPT" logs -n 0 --no-pager + +# ── 6. server-spawning subcommands — exercise the dispatch up to where +# the missing docker daemon stops them. `serve` is intentionally skipped +# because on a host with a working docker + the cuda12 image already +# pulled, it would actually exec into the container — at which point +# we'd be testing the image's entrypoint, not the wrapper. `pull` just +# execs `docker pull`, so we still smoke its host-side dispatch. +assert_no_set_u_leak "pull dispatch (no set -u leak)" "$SCRIPT" pull + +# ── 7. Unknown subcommand → cmd_in_container fallback path. Same rule: +# clean error, no raw bash leak. +assert_no_set_u_leak "unknown subcommand dispatch" "$SCRIPT" no-such-subcommand + +# ── 8. Pre-populated LUCEBOX_HOST_* env (simulates an already-probed host +# whose vars are passed in from a parent process). Useful in CI matrices +# where we want to mock a "good host" without nvidia-smi/docker on PATH. +out=$( + NO_COLOR=1 \ + LUCEBOX_HOST_HAS_SYSTEMD=0 \ + LUCEBOX_HOST_HAS_DOCKER=0 \ + LUCEBOX_HOST_HAS_CTK=none \ + LUCEBOX_HOST_GPU_VENDOR=none \ + LUCEBOX_HOST_GPU_NAME="" \ + LUCEBOX_HOST_GPU_COUNT=0 \ + LUCEBOX_HOST_VRAM_GB=0 \ + LUCEBOX_HOST_GPU_SM="" \ + LUCEBOX_HOST_DRIVER_VERSION="" \ + LUCEBOX_HOST_DRIVER_MAJOR=0 \ + LUCEBOX_HOST_NPROC=1 \ + LUCEBOX_HOST_RAM_GB=0 \ + LUCEBOX_HOST_IS_WSL=0 \ + LUCEBOX_HOST_DOCKER_VERSION="" \ + timeout 5 bash "$SCRIPT" start 2>&1 || true +) +if grep -qE 'unbound variable|syntax error' <<<"$out"; then + report fail "start with pre-populated LUCEBOX_HOST_* env" "leak: $(head -3 <<<"$out")" +else + report ok "start with pre-populated LUCEBOX_HOST_* env" +fi + +# ── 8b. PIN the top-of-script LUCEBOX_HOST_* safe-default seeds. Even with +# probe_host short-circuited to a no-op (the worst-case bug recurrence: a +# future refactor accidentally deletes the call from a dispatch path) the +# wrapper must not leak `unbound variable` on `start`. We achieve "probe_host +# is a no-op" by exporting `_LUCEBOX_HOST_PROBED=1` so ensure_probed skips +# the real probe — equivalent to a future refactor that calls ensure_probed +# but mis-implements the gate. +out=$( + NO_COLOR=1 \ + _LUCEBOX_HOST_PROBED=1 \ + timeout 5 bash "$SCRIPT" start 2>&1 || true +) +if grep -qE 'unbound variable|syntax error' <<<"$out"; then + report fail "start with probe_host bypassed (seed defaults must catch this)" "leak: $(head -3 <<<"$out")" +else + report ok "start with probe_host bypassed (seed defaults intact)" +fi + +# Same for every other systemd-surface subcommand, since the seed defaults +# are the only thing keeping these safe under `set -u` if probe_host is ever +# bypassed. +for sub in stop restart enable disable status install uninstall logs; do + out=$( + NO_COLOR=1 \ + _LUCEBOX_HOST_PROBED=1 \ + timeout 5 bash "$SCRIPT" "$sub" -n 0 --no-pager 2>&1 || true + ) + if grep -qE 'unbound variable|syntax error' <<<"$out"; then + report fail "$sub with probe_host bypassed" "leak: $(head -3 <<<"$out")" + else + report ok "$sub with probe_host bypassed" + fi +done + +# ── 8c. Install path writes a robust unit file. Use a sandbox HOME so we +# don't clobber the developer's real ~/.config/systemd/user/lucebox.service, +# and verify the generated unit contains the Environment= / ExecStartPre= +# hardening that Bug 2 ("systemctl start succeeds but no container") added. +# The install runs in a host with no real systemd (the sandbox doesn't have +# `systemctl --user`), so we pre-seed LUCEBOX_HOST_HAS_SYSTEMD=1 to slip past +# the require_systemd gate, then stub out the `systemctl` binary itself so +# daemon-reload is a no-op. +test_install_writes_robust_unit() { + local label="install writes hardened unit file" + local sandbox shim_dir + sandbox=$(mktemp -d) + shim_dir="$sandbox/bin" + mkdir -p "$shim_dir" + # Stub systemctl + docker + nvidia-smi + loginctl so the install's + # require_host_prereqs and daemon-reload calls all succeed. + for binname in systemctl docker nvidia-smi loginctl; do + cat > "$shim_dir/$binname" <<'STUB' +#!/usr/bin/env bash +case "$1" in + ps|version) exit 0 ;; + show-user) echo "Linger=no" ;; + --query-gpu=*) echo "Fake, 24576, 550.00, 8.9" ;; +esac +exit 0 +STUB + chmod +x "$shim_dir/$binname" + done + local out rc unit_path + unit_path="$sandbox/.config/systemd/user/lucebox.service" + out=$( + set +e + HOME="$sandbox" \ + XDG_CONFIG_HOME="$sandbox/.config" \ + XDG_DATA_HOME="$sandbox/.local/share" \ + PATH="$shim_dir:$PATH" \ + LUCEBOX_HOST_HAS_SYSTEMD=1 \ + LUCEBOX_HOST_HAS_DOCKER=1 \ + LUCEBOX_HOST_HAS_CTK=runtime \ + LUCEBOX_HOST_GPU_VENDOR=nvidia \ + _LUCEBOX_HOST_PROBED=1 \ + NO_COLOR=1 \ + timeout 10 bash "$SCRIPT" install 2>&1 + echo "RC=$?" + ) + rc=$(grep -oE 'RC=[0-9]+$' <<<"$out" | tail -1 | sed 's/^RC=//') + rc="${rc:-99}" + if [ "$rc" != "0" ]; then + report fail "$label" "exit $rc; output: $(head -10 <<<"$out")" + rm -rf "$sandbox" + return + fi + if [ ! -f "$unit_path" ]; then + report fail "$label" "unit file not written at $unit_path" + rm -rf "$sandbox" + return + fi + # Required hardening — each line is a Bug-2 root-cause defence: + # ExecStartPre=…docker rm -f … → clear orphaned container name + # Environment=PATH=… → systemd user-session PATH is sparse + # Environment=LUCEBOX_IMAGE=… → pin the image the user installed against + # SuccessExitStatus=143 → `serve` exits 143 on SIGTERM; a normal + # `systemctl stop` must land "inactive", not "failed" + local missing="" + for needle in \ + "ExecStartPre=" \ + "Environment=PATH=" \ + "Environment=LUCEBOX_IMAGE=" \ + "Environment=LUCEBOX_VARIANT=" \ + "Environment=LUCEBOX_PORT=" \ + "Environment=LUCEBOX_MODELS=" \ + "SuccessExitStatus=143" \ + ; do + grep -qF "$needle" "$unit_path" || missing="$missing $needle" + done + if [ -n "$missing" ]; then + report fail "$label" "unit missing required directives:$missing" + rm -rf "$sandbox" + return + fi + report ok "$label" + rm -rf "$sandbox" +} +test_install_writes_robust_unit + +# ── 9. entrypoint.sh dispatch — confirm the in-container dispatch routes +# trivial subcommands (shell, an unknown passthrough) without firing +# `set -u` on DFLASH_* / DRAFT_* vars that only get assigned on the +# serve path. We can't fully exec the serve path here (it needs nvidia +# and the compiled binary) but we can confirm the early dispatch is clean. +# +# Each `exec` would actually try to run the underlying binary, which we +# don't have — so we shim it by overriding `exec` via a wrapper script. +# Easier: just confirm `bash -n` parses and run a tiny subset. +out=$(NO_COLOR=1 SUBCMD=help bash -c " + cd '$ROOT' + # Simulate 'docker run ... lucebox-hub:cuda12 shell echo ok' — entrypoint + # gets SUBCMD=shell and execs /bin/bash with the rest of argv. We replace + # exec via PATH so we don't actually exec. + tmpdir=\$(mktemp -d) + trap 'rm -rf \$tmpdir' EXIT + cat > \$tmpdir/uv <<'STUB' +#!/usr/bin/env bash +echo \"uv stub: \$*\" +exit 0 +STUB + chmod +x \$tmpdir/uv + PATH=\$tmpdir:\$PATH bash $ENTRYPOINT shell -c 'echo entrypoint-shell-dispatched' +" 2>&1 || true) +if grep -qE 'unbound variable|syntax error' <<<"$out"; then + report fail "entrypoint shell dispatch (no set -u leak)" "leak: $(head -5 <<<"$out")" +else + report ok "entrypoint shell dispatch (no set -u leak)" +fi + +# ── 10. entrypoint.sh serve-path under `set -u` — drive the REAL +# server/scripts/entrypoint.sh through its full draft-resolution block by +# sandboxing it with a synthetic DFLASH_DIR layout and a `dflash_server` +# shim that captures argv instead of execing the native binary. The +# `DRAFT_FAMILY_GLOB: unbound variable` bug fired precisely here — the +# previous version of this test inlined the block instead of sourcing +# the real file, and silently passed even when the shipped script was +# broken. So this test invokes server/scripts/entrypoint.sh directly. +# Build the shared entrypoint-serve sandbox: a synthetic DFLASH_DIR layout +# plus the `dflash_server` + `nvidia-smi` shims used by the three serve-path +# tests below. Assigns sandbox/models_dir/draft_dir/bin_dir/shim_dir into the +# CALLER'S scope (bash dynamic scoping) — the caller must `local`-declare +# them first. Mirrors the _make_docker_shim factoring above. +_make_entrypoint_sandbox() { + sandbox=$(mktemp -d) + models_dir="$sandbox/models" + draft_dir="$models_dir/draft" + bin_dir="$sandbox/build" + shim_dir="$sandbox/bin" + mkdir -p "$draft_dir" "$bin_dir" "$shim_dir" + # `dflash_server` shim — print argv and exit 0 instead of running. + cat > "$bin_dir/dflash_server" <<'STUB' +#!/usr/bin/env bash +printf '[shim] dflash_server' +for a in "$@"; do printf ' %q' "$a"; done +printf '\n' +exit 0 +STUB + chmod +x "$bin_dir/dflash_server" + # `nvidia-smi` shim — pretend we have a 24 GB GPU so the autotune + # block runs but doesn't pick the under-12-GB warn tier. + cat > "$shim_dir/nvidia-smi" <<'STUB' +#!/usr/bin/env bash +case "$*" in + *"--query-gpu=memory.total"*) echo 24576 ;; + -L|*-L*) echo "GPU 0: Fake (UUID: 0)" ;; + *) echo "ok" ;; +esac +exit 0 +STUB + chmod +x "$shim_dir/nvidia-smi" +} + +test_entrypoint_serve_path() { + local label="$1" target_name="$2" draft_file="$3" + local sandbox draft_dir models_dir bin_dir shim_dir + _make_entrypoint_sandbox + # Synthetic target (must be a real file at least 5 GB to pass the + # auto-detect block, OR we set DFLASH_TARGET explicitly to skip it). + touch "$models_dir/$target_name" + touch "$draft_dir/$draft_file" + + local out rc + out=$( + set +e + PATH="$shim_dir:$PATH" \ + DFLASH_DIR="$sandbox" \ + DFLASH_SERVER_BIN="$bin_dir/dflash_server" \ + DFLASH_TARGET="$models_dir/$target_name" \ + DFLASH_DRAFT="$draft_dir" \ + timeout 10 bash "$ENTRYPOINT" serve 2>&1 + echo "RC=$?" + ) + rc=$(grep -oE 'RC=[0-9]+$' <<<"$out" | tail -1 | sed 's/^RC=//') + rc="${rc:-99}" + rm -rf "$sandbox" + if grep -qE 'unbound variable|syntax error' <<<"$out"; then + report fail "$label" "leak: $(head -5 <<<"$out")" + elif [ "$rc" != "0" ]; then + report fail "$label" "exit $rc; output: $(head -5 <<<"$out")" + elif ! grep -qF "[shim] dflash_server" <<<"$out"; then + report fail "$label" "shim never executed; output: $(head -5 <<<"$out")" + else + report ok "$label" + fi +} + +# Exercise three branches of the family-glob logic: qwen3.6 + gemma-4 (the +# two families with family-specific globs) and an unknown target that +# triggers the empty-FAMILY_GLOBS fallback to the generic glob list. +test_entrypoint_serve_path "entrypoint serve: qwen3.6 family match" \ + "Qwen3.6-27B-Q4_K_M.gguf" "dflash-draft-3.6-test.gguf" +test_entrypoint_serve_path "entrypoint serve: gemma-4-31b family match" \ + "gemma-4-31B-it-Q8_0.gguf" "gemma-4-31b-dflash-q8.gguf" +test_entrypoint_serve_path "entrypoint serve: generic fallback" \ + "Mystery-Model-7B.gguf" "model.gguf" + +# ── 11. entrypoint.sh serve-path with MULTIPLE target-sized GGUFs in +# models/. The single-candidate fixture in test 10 doesn't exercise the +# auto-detect path that picks "first alphabetically" when more than one +# target ≥5 GB lives in the models dir — that path is what the sindri +# decode sweep tripped over after the user added the qwen3.6-moe preset +# (commit 4b6bced) alongside the existing Qwen3.6-27B target. The crash +# manifested as `DRAFT_FAMILY_GLOB: unbound variable`, and the partial +# fix in a87bb93 didn't survive a recurrence. +# +# Uses sparse files (`truncate -s 6G`) so the test stays cheap on disk — +# the 6 GB virtual size is enough to clear the find ... -size +5G filter +# without consuming actual blocks. Skip if truncate is missing (e.g. +# minimal busybox CI image). +test_entrypoint_multi_target() { + local label="$1" + shift + if ! command -v truncate &>/dev/null; then + report ok "$label (skipped: truncate not available)" + return + fi + local sandbox draft_dir models_dir bin_dir shim_dir + _make_entrypoint_sandbox + # Two qwen3.6-shaped targets ≥5 GB each — exactly the layout that + # broke on sindri (Qwen3.6-27B + Qwen3.6-35B-A3B-UD-Q4_K_M). + truncate -s 6G "$models_dir/Qwen3.6-27B-Q4_K_M.gguf" + truncate -s 6G "$models_dir/Qwen3.6-35B-A3B-UD-Q4_K_M.gguf" + touch "$draft_dir/dflash-draft-3.6-test.gguf" + + local out rc + out=$( + set +e + # NOTE: deliberately NOT setting DFLASH_TARGET — the test must + # exercise the auto-detect block (line ~151). The explicit-config + # workaround from the bug report would skip the bug entirely. + PATH="$shim_dir:$PATH" \ + DFLASH_DIR="$sandbox" \ + DFLASH_SERVER_BIN="$bin_dir/dflash_server" \ + DFLASH_DRAFT="$draft_dir" \ + timeout 10 bash "$ENTRYPOINT" serve 2>&1 + echo "RC=$?" + ) + rc=$(grep -oE 'RC=[0-9]+$' <<<"$out" | tail -1 | sed 's/^RC=//') + rc="${rc:-99}" + rm -rf "$sandbox" + # The auto-detect block is entered (so any `set -u` regression on + # DRAFT_FAMILY_GLOB will trip) and then the entrypoint refuses to + # auto-pick — the deliberate safety added in PR #334's cubic round. + # We require: no set-u leak, the refuse warn fired, a non-zero exit + # (so a future regression that logs the warning but still returns 0 + # cannot slip past — the container MUST fail to start, not silently + # auto-pick a stale GGUF), and the shim was NOT exec'd. + if grep -qE 'unbound variable|syntax error' <<<"$out"; then + report fail "$label" "leak: $(grep -E 'unbound variable|syntax error' <<<"$out" | head -3)" + elif ! grep -qF "Refusing to auto-select" <<<"$out"; then + report fail "$label" "refuse-to-auto-pick warn missing — did the auto-detect block fire? rc=$rc output: $(head -5 <<<"$out")" + elif [ "$rc" = "0" ]; then + report fail "$label" "refuse warn fired but rc=0 — entrypoint must exit non-zero on multi-target refuse" + elif grep -qF "[shim] dflash_server" <<<"$out"; then + report fail "$label" "shim was exec'd despite multi-target refuse" + else + report ok "$label" + fi +} + +# Drive the regression: the sindri layout that broke (post-moe-preset). +test_entrypoint_multi_target "entrypoint serve: multi-target auto-detect (no DRAFT_FAMILY_GLOB leak)" + +# Also drive the DFLASH_DRAFT-is-a-file path. The init at entrypoint.sh:257 +# sits inside `if [ -d "$DFLASH_DRAFT" ]; then` — when DRAFT is a file the +# block is skipped, and any future read of DRAFT_FAMILY_GLOB outside the +# block would trip set -u. The defensive `:-` guard at the read site is +# meant to survive that refactor; this test guarantees it. +test_entrypoint_draft_is_file() { + local label="$1" + local sandbox draft_dir models_dir bin_dir shim_dir + _make_entrypoint_sandbox + touch "$models_dir/Qwen3.6-27B-Q4_K_M.gguf" + # DFLASH_DRAFT points at a FILE (not a directory). + touch "$draft_dir/dflash-draft-3.6-test.gguf" + + local out rc + out=$( + set +e + PATH="$shim_dir:$PATH" \ + DFLASH_DIR="$sandbox" \ + DFLASH_SERVER_BIN="$bin_dir/dflash_server" \ + DFLASH_TARGET="$models_dir/Qwen3.6-27B-Q4_K_M.gguf" \ + DFLASH_DRAFT="$draft_dir/dflash-draft-3.6-test.gguf" \ + timeout 10 bash "$ENTRYPOINT" serve 2>&1 + echo "RC=$?" + ) + rc=$(grep -oE 'RC=[0-9]+$' <<<"$out" | tail -1 | sed 's/^RC=//') + rc="${rc:-99}" + rm -rf "$sandbox" + if grep -qE 'unbound variable|syntax error' <<<"$out"; then + report fail "$label" "leak: $(grep -E 'unbound variable|syntax error' <<<"$out" | head -3)" + elif [ "$rc" != "0" ]; then + report fail "$label" "exit $rc; output: $(head -5 <<<"$out")" + else + report ok "$label" + fi +} +test_entrypoint_draft_is_file "entrypoint serve: DFLASH_DRAFT is a file (no DRAFT_FAMILY_GLOB leak)" + +# ── 12. entrypoint.sh writes HOST_INFO atomically on the serve path. The +# C++ server reads /opt/lucebox-hub/HOST_INFO into ServerConfig.host_info +# and surfaces it under /props.host. We can't write to /opt/lucebox-hub +# from the test runner, so override the path by sourcing the helpers and +# calling _build_host_info_json directly. The full entrypoint runs in +# test 10/11 already; this test pins the JSON shape independently. +test_entrypoint_host_info_json() { + local label="$1" + # Source the helper functions from the real entrypoint.sh. + # shellcheck disable=SC1090 + source <(awk '/^_json_escape\(\) \{/,/^\}/' "$ENTRYPOINT") + # shellcheck disable=SC1090 + source <(awk '/^_json_str_or_null\(\) \{/,/^\}/' "$ENTRYPOINT") + # shellcheck disable=SC1090 + source <(awk '/^_json_int_or_null\(\) \{/,/^\}/' "$ENTRYPOINT") + # shellcheck disable=SC1090 + source <(awk '/^_trim\(\) \{/,/^\}/' "$ENTRYPOINT") + # shellcheck disable=SC1090 + source <(awk '/^_emit_gpu_array\(\) \{/,/^\}/' "$ENTRYPOINT") + # shellcheck disable=SC1090 + source <(awk '/^_build_host_info_json\(\) \{/,/^\}/' "$ENTRYPOINT") + + local out + LUCEBOX_HOST_OS_PRETTY="Ubuntu 22.04.3 LTS" \ + LUCEBOX_HOST_KERNEL="6.6.87.2-microsoft-standard-WSL2" \ + LUCEBOX_HOST_WSL_VERSION="wsl2" \ + LUCEBOX_HOST_DOCKER_VERSION="29.1.3" \ + LUCEBOX_HOST_DRIVER_VERSION="596.36" \ + LUCEBOX_HOST_NVIDIA_CTK_VERSION="1.16.2" \ + LUCEBOX_HOST_CPU_MODEL='Intel(R) Core(TM) Ultra 9 275HX' \ + LUCEBOX_HOST_NPROC=24 \ + LUCEBOX_HOST_RAM_GB=64 \ + LUCEBOX_HOST_GPU_LIST_CSV="0, GPU-abc, 00000000:01:00.0, NVIDIA RTX 5090, 12.0, 24576 MiB, 175.00 W" \ + LUCEBOX_HOST_CUDA_VISIBLE_DEVICES="0" \ + out=$(_build_host_info_json "lucebox.sh" "lucebox.sh" "2026-05-28T20:31:42Z") + if ! python3 -c "import json,sys; d=json.loads(sys.argv[1]); assert d['os_pretty']=='Ubuntu 22.04.3 LTS'; assert d['wsl_version']=='wsl2'; assert d['nvidia_ctk_version']=='1.16.2'; assert d['source']=='lucebox.sh'; assert d['gpus'][0]['vram_gb']==24; assert d['gpus'][0]['name']=='NVIDIA RTX 5090'" "$out" >/dev/null 2>&1; then + report fail "$label (populated)" "JSON shape mismatch: $out" + return + fi + # Now drive the unknown path: every LUCEBOX_HOST_* unset → nulls and source=unknown. + out=$(env -i bash -c " + set -u + $(declare -f _json_escape _json_str_or_null _json_int_or_null _emit_gpu_array _build_host_info_json) + _build_host_info_json 'unknown' 'entrypoint.sh' '2026-05-28T20:31:42Z' + ") + if ! python3 -c "import json,sys; d=json.loads(sys.argv[1]); assert d['source']=='unknown'; assert d['gpus']==[]; assert d['os_pretty'] is None" "$out" >/dev/null 2>&1; then + report fail "$label (unknown)" "JSON shape mismatch: $out" + return + fi + report ok "$label" +} +test_entrypoint_host_info_json "entrypoint HOST_INFO JSON shape (populated + unknown)" + +# ── install.sh end-to-end ───────────────────────────────────────────────── +# Drive install.sh against a file:// URL pointing at a fixture lucebox.sh, +# verify the installed copy has LUCEBOX_INSTALLED_FROM rewritten to the +# fetched URL — that's the contract that `lucebox update` depends on to +# preserve the user's channel across upgrades. +test_install_sh_bakes_source_url() { + local label="$1" + local tmp dest_dir dest_path src_url out rc + tmp=$(mktemp -d -t lucebox-install.XXXXXX) + # Use the real lucebox.sh as the "remote" file — `file://` works with + # curl out of the box and exercises the same install.sh code path as + # an https fetch would. + src_url="file://$SCRIPT" + dest_dir="$tmp/bin" + dest_path="$dest_dir/lucebox" + out=$(LUCEBOX_INSTALL_URL="$src_url" LUCEBOX_INSTALL_DEST="$dest_path" \ + NO_COLOR=1 bash "$INSTALLER" 2>&1) || rc=$? + rc="${rc:-0}" + if [ "$rc" -ne 0 ]; then + rm -rf "$tmp" + report fail "$label" "installer exited $rc; output: $(printf '%s' "$out" | head -3)" + return + fi + if [ ! -x "$dest_path" ]; then + rm -rf "$tmp" + report fail "$label" "installed file missing or not executable at $dest_path" + return + fi + if ! grep -q "^LUCEBOX_INSTALLED_FROM=\"$src_url\"$" "$dest_path"; then + rm -rf "$tmp" + report fail "$label" "LUCEBOX_INSTALLED_FROM not rewritten in installed copy" + return + fi + rm -rf "$tmp" + report ok "$label" +} +test_install_sh_bakes_source_url "install.sh bakes LUCEBOX_INSTALLED_FROM into installed copy" + +# ── update dispatch ─────────────────────────────────────────────────────── +# `lucebox update` must dispatch to cmd_update — verify it's wired in the +# main case statement and appears in --help. We can't actually run the +# update (it'd curl + replace this very script) so the test is parse-level. +test_update_subcommand_wired() { + local label="$1" + local out + out=$(LUCEBOX_HOST_HAS_SYSTEMD=0 "$SCRIPT" --help 2>&1) + if ! grep -q '^ update ' <<<"$out"; then + report fail "$label" "update command missing from --help output" + return + fi + if ! grep -q '^[[:space:]]*update)[[:space:]]*cmd_update' "$SCRIPT"; then + report fail "$label" "update) → cmd_update dispatch not wired" + return + fi + report ok "$label" +} +test_update_subcommand_wired "lucebox update subcommand is wired" + +# ── IMAGE_BASE derived from install source ──────────────────────────────── +# Source lucebox.sh in a subshell with LUCEBOX_INSTALLED_FROM pointing at +# various URLs, then check that IMAGE_BASE comes out right. Uses +# `set -e; return` early so we don't actually run the wrapper's main(). +test_image_base_derives_from_install_url() { + local label="$1" url expected got + for case in \ + "https://raw.githubusercontent.com/easel/lucebox-hub/feat/lucebox-docker/lucebox.sh|ghcr.io/easel/lucebox-hub" \ + "https://raw.githubusercontent.com/Luce-Org/lucebox-hub/main/lucebox.sh|ghcr.io/luce-org/lucebox-hub" \ + "https://raw.githubusercontent.com/easel/lucebox-hub/601ab52/lucebox.sh|ghcr.io/easel/lucebox-hub" \ + "https://example.com/bogus|ghcr.io/luce-org/lucebox-hub" + do + url="${case%%|*}" + expected="${case##*|}" + # Extract the derivation function from the script and run it in + # isolation — sourcing the whole script triggers main() and side + # effects we don't want under a test harness. + got=$(bash -c ' + '"$(sed -n "/^_lucebox_derive_image()/,/^}/p" "$SCRIPT")"' + _lucebox_derive_image "$1" + ' bash "$url") + if [ "$got" != "$expected" ]; then + report fail "$label" "url=$url expected=$expected got=$got" + return + fi + done + report ok "$label" +} +test_image_base_derives_from_install_url "IMAGE_BASE derived from LUCEBOX_INSTALLED_FROM (4 URL shapes)" + +# ── config.toml reader + resolver ───────────────────────────────────────── +# Drive _lucebox_config_get + _lucebox_resolve against a fixture +# config.toml in a tmp $LUCEBOX_HOME. Verifies the wrapper agrees with +# the Python CLI on every scalar that lives in [image]/[runtime]/[paths]. +test_config_toml_reader_and_resolve() { + local label="$1" tmp got + tmp=$(mktemp -d -t lucebox-cfg.XXXXXX) + cat > "$tmp/config.toml" <<'TOML' +[image] +variant = "cuda13" +registry = "ghcr.io/myorg/forkedhub" + +[runtime] +port = 9090 +container_name = "luce-test" + +[paths] +models = "/srv/models" + +[dflash] +budget = 22 +lazy = false +TOML + + # Exercise both helpers + the resolver via a subshell that sources + # the relevant snippets out of lucebox.sh. Each case is a triple: + # env_value | toml_key | default | expected + local cases=( + "|image.registry|ghcr.io/luce-org/lucebox-hub|ghcr.io/myorg/forkedhub" + "|image.variant|cuda12|cuda13" + "|runtime.port|8080|9090" + "|runtime.container_name|lucebox|luce-test" + "|paths.models|/var/lib/lucebox|/srv/models" + "OVERRIDE|image.registry|ghcr.io/luce-org/lucebox-hub|OVERRIDE" + "|missing.key|fallback-default|fallback-default" + ) + local case env_value toml_key default expected + for case in "${cases[@]}"; do + IFS='|' read -r env_value toml_key default expected <<<"$case" + got=$(LUCEBOX_HOME="$tmp" bash -c ' + '"$(sed -n "/^_lucebox_config_path()/,/^}/p" "$SCRIPT")"' + '"$(sed -n "/^_lucebox_config_get()/,/^}/p" "$SCRIPT")"' + '"$(sed -n "/^_lucebox_resolve()/,/^}/p" "$SCRIPT")"' + _lucebox_resolve "$1" "$2" "$3" + ' bash "$env_value" "$toml_key" "$default") + if [ "$got" != "$expected" ]; then + rm -rf "$tmp" + report fail "$label" "env=$env_value key=$toml_key default=$default expected=$expected got=$got" + return + fi + done + rm -rf "$tmp" + report ok "$label" +} +test_config_toml_reader_and_resolve "config.toml reader + env > toml > default resolution (7 cases)" + +# ── cmd_serve under systemd: INVOCATION_ID short-circuits is-active ────── +# When systemd invokes the wrapper as a unit's ExecStart, it sets +# $INVOCATION_ID. The wrapper must NOT then refuse "already running under +# systemd" — that's a self-defeating check that turns into a restart loop. +# Verify the guard is present in the source (the actual behavior requires +# a running systemd unit to test end-to-end, which the harness can't do). +test_cmd_serve_invocation_id_guard() { + local label="$1" + if ! grep -q 'INVOCATION_ID' "$SCRIPT"; then + report fail "$label" "INVOCATION_ID guard missing from cmd_serve preflight" + return + fi + # The guard must be the AND-condition gating the is-active check. + # If grep finds the is-active line WITHOUT INVOCATION_ID nearby, + # the guard isn't wired correctly. + if ! awk ' + /INVOCATION_ID/ { saw_guard = NR } + /is-active --quiet "\$UNIT_NAME"/ { + if (saw_guard && NR - saw_guard <= 3) found = 1 + } + END { exit (found ? 0 : 1) } + ' "$SCRIPT"; then + report fail "$label" "INVOCATION_ID not adjacent to is-active check (guard not wired)" + return + fi + report ok "$label" +} +test_cmd_serve_invocation_id_guard "cmd_serve has INVOCATION_ID guard on systemd is-active check" + +# ── cmd_systemctl_passthrough: smart start ─────────────────────────────── +# Verify the source has the "already active" + "restart loop" short +# circuits for the start action. Behavior-level testing requires a real +# unit; this is a source-level guarantee that the branches exist. +test_cmd_start_already_active_shortcircuit() { + local label="$1" + if ! grep -q 'is already active' "$SCRIPT"; then + report fail "$label" "already-active short-circuit missing" + return + fi + if ! grep -q 'is in restart-loop' "$SCRIPT"; then + report fail "$label" "restart-loop short-circuit missing" + return + fi + report ok "$label" +} +test_cmd_start_already_active_shortcircuit "lucebox start has already-active + restart-loop short-circuits" + +# ── install.sh SHA-pin refusal + CHANNEL override ──────────────────────── +# A SHA-pinned LUCEBOX_INSTALL_URL with no LUCEBOX_INSTALL_CHANNEL must +# refuse — otherwise `lucebox update` would re-fetch that frozen SHA +# forever. With CHANNEL set, the bake-in uses the channel URL, not the +# fetch URL. +test_install_sha_pin_refusal_and_channel_override() { + local label="$1" tmp got rc + tmp=$(mktemp -d -t lucebox-sha.XXXXXX) + + # Case 1: SHA-pinned URL without CHANNEL → must refuse + LUCEBOX_INSTALL_URL="https://raw.githubusercontent.com/easel/lucebox-hub/abc1234567/lucebox.sh" \ + LUCEBOX_INSTALL_DEST="$tmp/lucebox1" \ + NO_COLOR=1 \ + bash "$INSTALLER" >/dev/null 2>&1 && rc=0 || rc=$? + if [ "$rc" -eq 0 ]; then + rm -rf "$tmp" + report fail "$label" "SHA-pinned URL without CHANNEL should have refused (rc=$rc, got success)" + return + fi + if [ -f "$tmp/lucebox1" ]; then + rm -rf "$tmp" + report fail "$label" "SHA-pinned URL refusal still wrote $tmp/lucebox1" + return + fi + + # Case 2: SHA-pinned URL WITH CHANNEL → installs, bakes CHANNEL + LUCEBOX_INSTALL_URL="file://$SCRIPT" \ + LUCEBOX_INSTALL_CHANNEL="https://raw.githubusercontent.com/easel/lucebox-hub/feat/lucebox-docker/lucebox.sh" \ + LUCEBOX_INSTALL_DEST="$tmp/lucebox2" \ + NO_COLOR=1 \ + bash "$INSTALLER" >/dev/null 2>&1 || rc=$? + got=$(grep '^LUCEBOX_INSTALLED_FROM=' "$tmp/lucebox2" 2>/dev/null || echo missing) + if [ "$got" != 'LUCEBOX_INSTALLED_FROM="https://raw.githubusercontent.com/easel/lucebox-hub/feat/lucebox-docker/lucebox.sh"' ]; then + rm -rf "$tmp" + report fail "$label" "CHANNEL not baked; got: $got" + return + fi + + rm -rf "$tmp" + report ok "$label" +} +test_install_sha_pin_refusal_and_channel_override "install.sh refuses SHA-pin without CHANNEL + honors CHANNEL override" + +# ── lucebox completion ─────────────────────────────────────────────────── +# The completion script must source cleanly and complete a known prefix. +test_completion_bash() { + local label="$1" out + out=$(LUCEBOX_HOST_HAS_SYSTEMD=0 bash -c ' + source <("$1" completion bash 2>/dev/null) + COMP_WORDS=(lucebox conf) + COMP_CWORD=1 + _lucebox_complete + printf "%s\n" "${COMPREPLY[@]}" + ' bash "$SCRIPT") + if ! grep -qx 'config' <<<"$out"; then + report fail "$label" "completion didn't suggest 'config' for prefix 'conf'; got: $(printf '%s' "$out" | tr '\n' ' ')" + return + fi + report ok "$label" +} +test_completion_bash "lucebox completion bash completes a known prefix" + +# ── docker exec routing ─────────────────────────────────────────────────── +# When the lucebox container is running, steady-state subcommands must +# `docker exec` into it (cheap + shares the live server's net namespace) and +# service-restarting subcommands (serve, pull, ...) must stay on +# `docker run`. We mock docker via a PATH shim that: +# - on `docker ps -q -f name=^lucebox$` prints a fake container id +# (signals "container is running") iff DOCKER_FAKE_RUNNING=1. +# - on any other call (run, exec, pull, ...) echoes its argv on stdout and +# exits 0. The test then asserts on the captured first-token (run vs exec) +# and trailing argv. +# +# nvidia-smi is stubbed too so probe_host doesn't barf, but the captured argv +# we care about is the docker invocation downstream of dispatch. +_make_docker_shim() { + local sandbox="$1" running="$2" + local shim_dir="$sandbox/bin" + mkdir -p "$shim_dir" + # docker shim: dispatch on first arg. Important: ps -q -f name=^lucebox$ + # must print a fake id when DOCKER_FAKE_RUNNING=1 and nothing otherwise. + # All other invocations (run, exec, pull) print "DOCKER_INVOKED " + # on stdout so the caller can grep it. + cat > "$shim_dir/docker" < "$shim_dir/nvidia-smi" <<'STUB' +#!/usr/bin/env bash +case "$*" in + *"--query-gpu="*) echo "Fake GPU, 24576, 550.00, 8.9" ;; + *) echo "ok" ;; +esac +exit 0 +STUB + chmod +x "$shim_dir/nvidia-smi" +} + +# Drive the wrapper through the dispatch case under test and capture the +# docker invocation it would have exec'd. Because `cmd_in_container` / +# `cmd_exec_in_container` call `exec docker ...` we replace `exec` semantics +# by running the wrapper in a subshell — the docker shim prints what it was +# called with and the captured stdout is the proof. +_run_wrapper_capture_docker() { + local sandbox="$1"; shift + local shim_dir="$sandbox/bin" + set +e + HOME="$sandbox" \ + XDG_CONFIG_HOME="$sandbox/.config" \ + XDG_DATA_HOME="$sandbox/.local/share" \ + LUCEBOX_HOME="$sandbox/.lucebox" \ + PATH="$shim_dir:$PATH" \ + LUCEBOX_HOST_HAS_DOCKER=1 \ + LUCEBOX_HOST_HAS_CTK=runtime \ + LUCEBOX_HOST_GPU_VENDOR=nvidia \ + LUCEBOX_HOST_DRIVER_MAJOR=550 \ + LUCEBOX_HOST_DRIVER_VERSION="550.00" \ + LUCEBOX_HOST_GPU_NAME="Fake GPU" \ + LUCEBOX_HOST_GPU_COUNT=1 \ + LUCEBOX_HOST_VRAM_GB=24 \ + LUCEBOX_HOST_GPU_SM="89" \ + LUCEBOX_HOST_NPROC=8 \ + LUCEBOX_HOST_RAM_GB=64 \ + LUCEBOX_HOST_HAS_SYSTEMD=0 \ + LUCEBOX_HOST_IS_WSL=0 \ + LUCEBOX_HOST_DOCKER_VERSION="29.1.3" \ + _LUCEBOX_HOST_PROBED=1 \ + NO_COLOR=1 \ + timeout 10 bash "$SCRIPT" "$@" 2>&1 + set -e +} + +test_routes_to_exec_when_running() { + local label="$1" sandbox out + sandbox=$(mktemp -d -t lucebox-route.XXXXXX) + _make_docker_shim "$sandbox" 1 + out=$(_run_wrapper_capture_docker "$sandbox" config get model.preset || true) + rm -rf "$sandbox" + if ! grep -q '^DOCKER_INVOKED exec' <<<"$out"; then + report fail "$label" "expected 'docker exec' invocation; got: $(head -3 <<<"$out")" + return + fi + if grep -q '^DOCKER_INVOKED run' <<<"$out"; then + report fail "$label" "got 'docker run' when container is up — should have exec'd" + return + fi + # Sanity: the exec line ends with `lucebox config get model.preset`. + if ! grep -qE 'lucebox config get model.preset' <<<"$out"; then + report fail "$label" "exec argv missing tail 'lucebox config get model.preset'; got: $(head -3 <<<"$out")" + return + fi + # The exec path must forward the LUCEBOX_* scalar env subset (shared + # with the docker-run path via _append_scalar_env). Pin LUCEBOX_IMAGE= + # so a regression in that helper is caught here. + if ! grep -q 'LUCEBOX_IMAGE=' <<<"$out"; then + report fail "$label" "exec argv missing 'LUCEBOX_IMAGE=' scalar env; got: $(head -3 <<<"$out")" + return + fi + report ok "$label" +} +test_routes_to_exec_when_running "config get routes to docker exec when container running" + +test_routes_to_run_when_not_running() { + local label="$1" sandbox out + sandbox=$(mktemp -d -t lucebox-route.XXXXXX) + _make_docker_shim "$sandbox" 0 + out=$(_run_wrapper_capture_docker "$sandbox" config get model.preset || true) + rm -rf "$sandbox" + if ! grep -q '^DOCKER_INVOKED run' <<<"$out"; then + report fail "$label" "expected 'docker run' invocation (container not running); got: $(head -3 <<<"$out")" + return + fi + if grep -q '^DOCKER_INVOKED exec' <<<"$out"; then + report fail "$label" "got 'docker exec' but container is not running — should fall back to run" + return + fi + report ok "$label" +} +test_routes_to_run_when_not_running "config get falls back to docker run when container not running" + +test_no_exec_flag_forces_run() { + local label="$1" sandbox out + sandbox=$(mktemp -d -t lucebox-route.XXXXXX) + _make_docker_shim "$sandbox" 1 + # --no-exec must override the prefer-exec path even when container is up. + out=$(_run_wrapper_capture_docker "$sandbox" --no-exec config get model.preset || true) + rm -rf "$sandbox" + if grep -q '^DOCKER_INVOKED exec' <<<"$out"; then + report fail "$label" "--no-exec failed to force run path; got exec" + return + fi + if ! grep -q '^DOCKER_INVOKED run' <<<"$out"; then + report fail "$label" "expected 'docker run' under --no-exec; got: $(head -3 <<<"$out")" + return + fi + report ok "$label" +} +test_no_exec_flag_forces_run "--no-exec flag forces docker run even when container is up" + +test_no_exec_env_forces_run() { + local label="$1" sandbox out + sandbox=$(mktemp -d -t lucebox-route.XXXXXX) + _make_docker_shim "$sandbox" 1 + out=$( + LUCEBOX_NO_EXEC=1 _run_wrapper_capture_docker "$sandbox" config get model.preset || true + ) + rm -rf "$sandbox" + if grep -q '^DOCKER_INVOKED exec' <<<"$out"; then + report fail "$label" "LUCEBOX_NO_EXEC=1 failed to force run path; got exec" + return + fi + if ! grep -q '^DOCKER_INVOKED run' <<<"$out"; then + report fail "$label" "expected 'docker run' under LUCEBOX_NO_EXEC=1; got: $(head -3 <<<"$out")" + return + fi + report ok "$label" +} +test_no_exec_env_forces_run "LUCEBOX_NO_EXEC=1 env override forces docker run" + +test_models_routes_to_exec() { + local label="$1" sandbox out + sandbox=$(mktemp -d -t lucebox-route.XXXXXX) + _make_docker_shim "$sandbox" 1 + out=$(_run_wrapper_capture_docker "$sandbox" models list || true) + rm -rf "$sandbox" + if ! grep -q '^DOCKER_INVOKED exec' <<<"$out"; then + report fail "$label" "expected 'docker exec' for models when running; got: $(head -3 <<<"$out")" + return + fi + # Confirm the exec'd command tail is `lucebox models list` — the + # in-container CLI's argv must NOT be polluted with dispatcher bookkeeping. + if ! grep -qE 'lucebox models list' <<<"$out"; then + report fail "$label" "exec'd argv missing 'lucebox models list' tail" + return + fi + report ok "$label" +} +test_models_routes_to_exec "models list routes to docker exec when container running" + +# ── usage mentions exec-when-running ────────────────────────────────────── +test_usage_mentions_exec_routing() { + local label="$1" out + out=$(NO_COLOR=1 bash "$SCRIPT" --help 2>&1) + if ! grep -qi 'docker exec\|--no-exec' <<<"$out"; then + report fail "$label" "usage doesn't mention the exec routing / --no-exec flag" + return + fi + report ok "$label" +} +test_usage_mentions_exec_routing "usage documents docker exec routing + --no-exec flag" + +# ── TTY flag selection. Regression guard for the process-substitution bug: +# _set_tty_flags must run in the CALLER's scope so `[ -t 1 ]` inspects the +# real terminal. If it is ever moved back behind `< <(...)` or `$(...)`, +# fd 1 becomes a pipe and it emits -i even on a real tty, silently dropping +# docker's -t and breaking the interactive client TUIs (lucebox claude …). +# The rest of this suite runs non-tty, so only this test exercises the -it +# branch — via a real PTY allocated by python's pty.fork. +test_tty_flags_selection() { + local label="$1" fn out + fn=$(awk '/^_set_tty_flags\(\) \{/,/^\}/' "$SCRIPT") + + # (a) non-tty (stdin /dev/null, stdout a pipe) → -i + out=$(bash -c "$fn"$'\n''f=(); _set_tty_flags f; printf "%s" "${f[*]}"' /dev/null) + if [ "$out" != "-i" ]; then + report fail "$label" "non-tty expected -i, got '$out'" + return + fi + + # (b) real tty on fd0+fd1 (python pty.fork) → -it + out=$(python3 - "$SCRIPT" <<'PY' 2>/dev/null +import os, pty, re, sys +src = open(sys.argv[1]).read() +fn = re.search(r'^_set_tty_flags\(\) \{.*?^\}', src, re.S | re.M).group(0) +script = fn + '\nf=(); _set_tty_flags f; printf "TTYFLAG=%s\\n" "${f[*]}"\n' +pid, fd = pty.fork() +if pid == 0: + os.execvp("bash", ["bash", "-c", script]) +buf = b"" +try: + while True: + chunk = os.read(fd, 1024) + if not chunk: + break + buf += chunk +except OSError: + pass +os.waitpid(pid, 0) +m = re.search(rb"TTYFLAG=(\S+)", buf) +sys.stdout.write(m.group(1).decode() if m else "NONE") +PY +) + if [ "$out" != "-it" ]; then + report fail "$label" "real tty expected -it, got '$out'" + return + fi + report ok "$label" +} +test_tty_flags_selection "_set_tty_flags: -it on a real tty, -i otherwise" + +echo +if [ "$fail" -eq 0 ]; then + echo "[test_lucebox_sh] $pass passed, 0 failed" + exit 0 +else + echo "[test_lucebox_sh] $pass passed, $fail failed" >&2 + exit 1 +fi diff --git a/server/scripts/entrypoint.sh b/server/scripts/entrypoint.sh index 2602517f5..cd0b5184e 100755 --- a/server/scripts/entrypoint.sh +++ b/server/scripts/entrypoint.sh @@ -73,6 +73,15 @@ esac # write-failure (read-only FS, etc.) gets a warning and we continue. write_host_info() { local target="/opt/lucebox-hub/HOST_INFO" + # If the target dir doesn't exist (e.g. running the entrypoint outside + # the canonical container layout: unit tests, plain `docker run` without + # a bind mount), don't try to write — bash's own "No such file or + # directory" complaint on the `> "$tmp"` redirect below would leak to + # stderr regardless of `2>/dev/null` (that suppresses the command's + # stderr, not the redirect itself). HOST_INFO is informational. + if [ ! -d "$(dirname "$target")" ]; then + return 0 + fi local tmp="${target}.tmp.$$" local collected_at collected_at=$(date -u +%FT%TZ 2>/dev/null || echo "") @@ -158,6 +167,15 @@ _json_int_or_null() { # `nvidia-smi --query-gpu=index,uuid,pci.bus_id,name,compute_cap,memory.total,power.limit # --format=csv,noheader` produced on the host) into a JSON # array. Empty CSV → "[]". Each row becomes one object. +# Strip leading/trailing whitespace from a string. Pure bash (no sed fork) +# via prefix/suffix removal of the longest run of spaces or tabs. +_trim() { + local s="$1" + s="${s#"${s%%[![:space:]]*}"}" # leading + s="${s%"${s##*[![:space:]]}"}" # trailing + printf '%s' "$s" +} + _emit_gpu_array() { local csv="${LUCEBOX_HOST_GPU_LIST_CSV:-}" if [ -z "$csv" ]; then @@ -173,13 +191,13 @@ _emit_gpu_array() { # split on `,` alone and trim whitespace per field so both forms parse. local idx uuid pci name cc mem plimit IFS=',' read -r idx uuid pci name cc mem plimit <<<"$line" - idx=$(printf '%s' "$idx" | sed 's/^[[:space:]]*//; s/[[:space:]]*$//') - uuid=$(printf '%s' "$uuid" | sed 's/^[[:space:]]*//; s/[[:space:]]*$//') - pci=$(printf '%s' "$pci" | sed 's/^[[:space:]]*//; s/[[:space:]]*$//') - name=$(printf '%s' "$name" | sed 's/^[[:space:]]*//; s/[[:space:]]*$//') - cc=$(printf '%s' "$cc" | sed 's/^[[:space:]]*//; s/[[:space:]]*$//') - mem=$(printf '%s' "$mem" | sed 's/^[[:space:]]*//; s/[[:space:]]*$//') - plimit=$(printf '%s' "$plimit" | sed 's/^[[:space:]]*//; s/[[:space:]]*$//') + idx=$(_trim "$idx") + uuid=$(_trim "$uuid") + pci=$(_trim "$pci") + name=$(_trim "$name") + cc=$(_trim "$cc") + mem=$(_trim "$mem") + plimit=$(_trim "$plimit") # Strip units. "24576 MiB" → 24576; "175.00 W" → 175 (truncate). local mem_mib vram_gb power_w mem_mib=$(printf '%s' "$mem" | awk '{print $1+0}') diff --git a/uv.lock b/uv.lock index fee8de0df..ba16922d8 100644 --- a/uv.lock +++ b/uv.lock @@ -9,6 +9,7 @@ resolution-markers = [ [manifest] members = [ + "lucebox", "lucebox-dflash", "lucebox-hub", "pflash", @@ -429,6 +430,26 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/59/67/a6739ac96e28b7855808bdb0370e250606104a859750d209e5a0716fe7ab/librt-0.11.0-cp312-cp312-win_arm64.whl", hash = "sha256:2f10cf143e4a9bb0f4f5af568a00df94a2d69ef41c2579584454bb0fe5cc642c", size = 103470, upload-time = "2026-05-10T18:16:10.369Z" }, ] +[[package]] +name = "lucebox" +source = { editable = "lucebox" } +dependencies = [ + { name = "httpx" }, + { name = "huggingface-hub" }, + { name = "rich" }, + { name = "tomli-w" }, + { name = "typer" }, +] + +[package.metadata] +requires-dist = [ + { name = "httpx", specifier = ">=0.27" }, + { name = "huggingface-hub", specifier = ">=0.27" }, + { name = "rich", specifier = ">=13" }, + { name = "tomli-w", specifier = ">=1.0" }, + { name = "typer", specifier = ">=0.12" }, +] + [[package]] name = "lucebox-dflash" version = "0.1.0" @@ -466,6 +487,7 @@ name = "lucebox-hub" version = "0.0.0" source = { virtual = "." } dependencies = [ + { name = "lucebox" }, { name = "lucebox-dflash" }, { name = "pflash" }, ] @@ -482,6 +504,7 @@ megakernel = [ [package.metadata] requires-dist = [ + { name = "lucebox", editable = "lucebox" }, { name = "lucebox-dflash", virtual = "server" }, { name = "mypy", marker = "extra == 'dev'", specifier = ">=1.10,<2" }, { name = "pflash", editable = "optimizations/pflash" }, @@ -1124,6 +1147,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/72/f4/0de46cfa12cdcbcd464cc59fde36912af405696f687e53a091fb432f694c/tokenizers-0.22.2-cp39-abi3-win_arm64.whl", hash = "sha256:9ce725d22864a1e965217204946f830c37876eee3b2ba6fc6255e8e903d5fcbc", size = 2612133, upload-time = "2026-01-05T10:45:17.232Z" }, ] +[[package]] +name = "tomli-w" +version = "1.2.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/19/75/241269d1da26b624c0d5e110e8149093c759b7a286138f4efd61a60e75fe/tomli_w-1.2.0.tar.gz", hash = "sha256:2dd14fac5a47c27be9cd4c976af5a12d87fb1f0b4512f81d69cce3b35ae25021", size = 7184, upload-time = "2025-01-15T12:07:24.262Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/c7/18/c86eb8e0202e32dd3df50d43d7ff9854f8e0603945ff398974c1d91ac1ef/tomli_w-1.2.0-py3-none-any.whl", hash = "sha256:188306098d013b691fcadc011abd66727d3c414c571bb01b1a174ba8c983cf90", size = 6675, upload-time = "2025-01-15T12:07:22.074Z" }, +] + [[package]] name = "torch" version = "2.11.0+cu128" From c91d7f23304f400bf6b51600c4377efcf6d2f263 Mon Sep 17 00:00:00 2001 From: mrciffa <49000955+davide221@users.noreply.github.com> Date: Mon, 27 Jul 2026 11:55:05 +0200 Subject: [PATCH 2/8] feat(lucebox): support AMD and heterogeneous GPU builds --- install.sh | 2 +- lucebox.sh | 402 +++++++++++++++++++++++++----- lucebox/README.md | 7 +- lucebox/src/lucebox/autotune.py | 2 +- lucebox/src/lucebox/cli.py | 2 +- lucebox/src/lucebox/config.py | 8 +- lucebox/src/lucebox/docker_run.py | 36 ++- lucebox/src/lucebox/host_check.py | 75 ++++-- lucebox/src/lucebox/host_facts.py | 7 +- lucebox/src/lucebox/types.py | 12 +- lucebox/tests/test_check.py | 50 ++++ lucebox/tests/test_config.py | 18 ++ lucebox/tests/test_docker_run.py | 35 ++- lucebox/tests/test_models_cli.py | 5 +- scripts/test_lucebox_sh.sh | 142 +++++++++++ server/scripts/entrypoint.sh | 68 ++++- 16 files changed, 767 insertions(+), 104 deletions(-) diff --git a/install.sh b/install.sh index cff54a02e..c44f92712 100755 --- a/install.sh +++ b/install.sh @@ -131,7 +131,7 @@ esac cat </dev/null || echo 1) # RAM: try Linux /proc/meminfo first, then macOS/BSD sysctl, else 0. @@ -232,18 +300,30 @@ probe_host() { LUCEBOX_HOST_RAM_GB=$(( mem_bytes / 1024 / 1024 / 1024 )) fi LUCEBOX_HOST_GPU_VENDOR="none" + LUCEBOX_HOST_HAS_NVIDIA_GPU=0 + LUCEBOX_HOST_HAS_AMD_GPU=0 LUCEBOX_HOST_GPU_NAME="" LUCEBOX_HOST_GPU_COUNT=0 LUCEBOX_HOST_VRAM_GB=0 LUCEBOX_HOST_GPU_SM="" LUCEBOX_HOST_DRIVER_VERSION="" LUCEBOX_HOST_DRIVER_MAJOR=0 + LUCEBOX_HOST_GPU_LIST_CSV="" + LUCEBOX_HOST_ROCM_VERSION="" + LUCEBOX_HOST_HAS_KFD=0 + LUCEBOX_HOST_HAS_DRI=0 + LUCEBOX_HOST_AMD_GPU_NAME="" + LUCEBOX_HOST_AMD_GPU_COUNT=0 + LUCEBOX_HOST_AMD_VRAM_GB=0 + LUCEBOX_HOST_AMD_GPU_ARCH="" + LUCEBOX_HOST_AMD_GPU_LIST_CSV="" if command -v nvidia-smi &>/dev/null; then local q if q=$(nvidia-smi --query-gpu=name,memory.total,driver_version,compute_cap \ --format=csv,noheader,nounits 2>/dev/null) && [ -n "$q" ]; then LUCEBOX_HOST_GPU_VENDOR="nvidia" + LUCEBOX_HOST_HAS_NVIDIA_GPU=1 LUCEBOX_HOST_GPU_NAME=$(printf '%s\n' "$q" | head -1 | awk -F', ' '{print $1}') local mem_mib mem_mib=$(printf '%s\n' "$q" | head -1 | awk -F', ' '{print $2}') @@ -264,8 +344,96 @@ probe_host() { --query-gpu=index,uuid,pci.bus_id,name,compute_cap,memory.total,power.limit \ --format=csv,noheader 2>/dev/null || echo "") fi + + # Probe AMD independently even on mixed NVIDIA + Strix systems. A working + # NVIDIA GPU remains the default backend (RTX 3090 + Strix → cuda12), but + # recording the AMD companion prevents the APU from confusing readiness + # reporting and lets an explicit rocm variant remain possible. + local amd_csv="" amd_rows="" + if command -v amd-smi &>/dev/null; then + amd_csv=$(amd-smi static --asic --vram --csv 2>/dev/null || echo "") + if [ -n "$amd_csv" ]; then + amd_rows=$(printf '%s\n' "$amd_csv" | _parse_amd_smi_csv) + fi + fi + if [ -z "$amd_rows" ] && command -v rocm-smi &>/dev/null; then + amd_csv=$(rocm-smi --showproductname --showmeminfo vram --csv 2>/dev/null || echo "") + if [ -n "$amd_csv" ]; then + amd_rows=$(printf '%s\n' "$amd_csv" | _parse_rocm_smi_csv) + fi + fi + # Minimal fallback for ROCm installations without either SMI frontend. + if [ -z "$amd_rows" ] && [ -e /dev/kfd ] && command -v rocminfo &>/dev/null; then + local roc_arches + roc_arches=$(rocminfo 2>/dev/null \ + | awk '/^[[:space:]]*Name:[[:space:]]+gfx[0-9a-z]+/{print $2}' \ + | awk '!seen[$0]++' || echo "") + if [ -n "$roc_arches" ]; then + local amd_idx=0 arch + while IFS= read -r arch; do + amd_rows+="${amd_idx}|AMD GPU|${arch}|0"$'\n' + amd_idx=$((amd_idx + 1)) + done <<<"$roc_arches" + amd_rows=${amd_rows%$'\n'} + fi + fi + + if [ -n "$amd_rows" ]; then + LUCEBOX_HOST_HAS_AMD_GPU=1 + LUCEBOX_HOST_AMD_GPU_COUNT=$(printf '%s\n' "$amd_rows" | awk 'NF{n++} END{print n+0}') + local amd_primary + amd_primary=$(printf '%s\n' "$amd_rows" | sort -t'|' -k4,4nr | head -1) + local amd_idx amd_name amd_arch amd_mem_mib + IFS='|' read -r amd_idx amd_name amd_arch amd_mem_mib <<<"$amd_primary" + LUCEBOX_HOST_AMD_GPU_NAME="$amd_name" + LUCEBOX_HOST_AMD_GPU_ARCH="$amd_arch" + LUCEBOX_HOST_AMD_VRAM_GB=$((amd_mem_mib / 1024)) + LUCEBOX_HOST_AMD_GPU_LIST_CSV=$(printf '%s\n' "$amd_rows" \ + | awk -F'|' '{printf "%s, , , %s, %s, %s MiB,\n", $1, $2, $3, $4}') + # Strix Halo exposes most memory as unified system RAM, while SMI may + # report only a 512 MiB carve-out. Use host RAM as the effective model + # capacity on a Strix-only build; a discrete R9700 remains primary on + # the R9700 + Strix build because it has the largest physical VRAM. + if [ "$amd_arch" = "gfx1151" ] \ + && [ "$LUCEBOX_HOST_AMD_VRAM_GB" -lt 12 ] \ + && [ "$LUCEBOX_HOST_RAM_GB" -ge 32 ]; then + LUCEBOX_HOST_AMD_VRAM_GB=$LUCEBOX_HOST_RAM_GB + fi + + if command -v amd-smi &>/dev/null; then + LUCEBOX_HOST_ROCM_VERSION=$(amd-smi version 2>/dev/null \ + | sed -n 's/.*ROCm version: \([^ |]*\).*/\1/p' \ + | head -1 || echo "") + fi + if [ -z "$LUCEBOX_HOST_ROCM_VERSION" ] && command -v hipconfig &>/dev/null; then + LUCEBOX_HOST_ROCM_VERSION=$(hipconfig --version 2>/dev/null \ + | sed 's/-.*//' | head -1 || echo "") + fi + + if [ "$LUCEBOX_HOST_GPU_VENDOR" = "none" ]; then + LUCEBOX_HOST_GPU_VENDOR="amd" + LUCEBOX_HOST_GPU_NAME="$LUCEBOX_HOST_AMD_GPU_NAME" + LUCEBOX_HOST_GPU_COUNT=$LUCEBOX_HOST_AMD_GPU_COUNT + LUCEBOX_HOST_VRAM_GB=$LUCEBOX_HOST_AMD_VRAM_GB + LUCEBOX_HOST_GPU_SM="$LUCEBOX_HOST_AMD_GPU_ARCH" + LUCEBOX_HOST_GPU_LIST_CSV="$LUCEBOX_HOST_AMD_GPU_LIST_CSV" + fi + fi + + if [ -r /dev/kfd ] && [ -w /dev/kfd ]; then + LUCEBOX_HOST_HAS_KFD=1 + fi + local render_node + for render_node in /dev/dri/renderD*; do + [ -e "$render_node" ] || continue + if [ -r "$render_node" ] && [ -w "$render_node" ]; then + LUCEBOX_HOST_HAS_DRI=1 + break + fi + done # CUDA_VISIBLE_DEVICES from the caller's env (empty default = "all GPUs"). LUCEBOX_HOST_CUDA_VISIBLE_DEVICES="${CUDA_VISIBLE_DEVICES:-}" + LUCEBOX_HOST_HIP_VISIBLE_DEVICES="${HIP_VISIBLE_DEVICES:-}" # OS / kernel identity. /etc/os-release is the freedesktop spec for # "what distro is this?" and we keep PRETTY_NAME verbatim (it already @@ -338,14 +506,20 @@ probe_host() { fi export LUCEBOX_HOST_NPROC LUCEBOX_HOST_RAM_GB LUCEBOX_HOST_GPU_VENDOR + export LUCEBOX_HOST_HAS_NVIDIA_GPU LUCEBOX_HOST_HAS_AMD_GPU export LUCEBOX_HOST_GPU_NAME LUCEBOX_HOST_GPU_COUNT LUCEBOX_HOST_VRAM_GB export LUCEBOX_HOST_GPU_SM LUCEBOX_HOST_DRIVER_VERSION LUCEBOX_HOST_DRIVER_MAJOR export LUCEBOX_HOST_HAS_SYSTEMD LUCEBOX_HOST_IS_WSL export LUCEBOX_HOST_HAS_DOCKER LUCEBOX_HOST_DOCKER_VERSION export LUCEBOX_HOST_HAS_CTK + export LUCEBOX_HOST_ROCM_VERSION LUCEBOX_HOST_HAS_KFD LUCEBOX_HOST_HAS_DRI + export LUCEBOX_HOST_AMD_GPU_NAME LUCEBOX_HOST_AMD_GPU_COUNT + export LUCEBOX_HOST_AMD_VRAM_GB LUCEBOX_HOST_AMD_GPU_ARCH + export LUCEBOX_HOST_AMD_GPU_LIST_CSV export LUCEBOX_HOST_OS_PRETTY LUCEBOX_HOST_KERNEL LUCEBOX_HOST_WSL_VERSION export LUCEBOX_HOST_NVIDIA_CTK_VERSION LUCEBOX_HOST_CPU_MODEL export LUCEBOX_HOST_GPU_LIST_CSV LUCEBOX_HOST_CUDA_VISIBLE_DEVICES + export LUCEBOX_HOST_HIP_VISIBLE_DEVICES _LUCEBOX_HOST_PROBED=1 } @@ -357,10 +531,41 @@ ensure_probed() { } pick_variant() { - # CUDA 12.8 is the supported image variant for this branch. Effective - # value goes through the same env > config.toml > default ladder as - # everything else so `config set image.variant=...` propagates. - _lucebox_resolve "${LUCEBOX_VARIANT:-}" image.variant "cuda12" + # Explicit env/config always wins. On a fresh install choose the backend + # from hardware: a working NVIDIA GPU takes priority on RTX + Strix builds; + # otherwise an AMD GPU selects ROCm (R9700 + Strix and Strix-only builds). + local configured + if [ -n "${LUCEBOX_VARIANT:-}" ]; then + printf '%s' "$LUCEBOX_VARIANT" + return + fi + configured=$(_lucebox_config_get image.variant) + if [ -n "$configured" ]; then + printf '%s' "$configured" + return + fi + ensure_probed + if [ "$LUCEBOX_HOST_HAS_NVIDIA_GPU" = "1" ]; then + printf 'cuda12' + elif [ "$LUCEBOX_HOST_HAS_AMD_GPU" = "1" ]; then + printf 'rocm' + else + # Keep the historical default so `lucebox check` can still tell a + # GPU-less host what image it would otherwise use. + printf 'cuda12' + fi +} + +_variant_is_rocm() { + # Variant names are not limited to the moving `rocm` tag. Releases and + # CI produce tags such as `0.3.0-rocm` and `pr-335-rocm`; treat any tag + # containing the backend marker as ROCm. Spell out case-insensitivity + # instead of using Bash 4's `${value,,}` so host-only commands such as + # `check` keep working under the older Bash bundled with macOS too. + case "$1" in + *[Rr][Oo][Cc][Mm]*) return 0 ;; + *) return 1 ;; + esac } # ── prereq checks (host-only) ───────────────────────────────────────────── @@ -368,6 +573,9 @@ pick_variant() { # the richer reporting; this is the bare minimum to make `docker run` viable. require_host_prereqs() { + ensure_probed + local variant="${1:-}" + [ -n "$variant" ] || variant=$(pick_variant) local missing=0 if ! command -v docker &>/dev/null; then err "docker is not installed" @@ -379,20 +587,36 @@ require_host_prereqs() { missing=1 fi - if ! command -v nvidia-smi &>/dev/null; then - err "nvidia-smi not found — no NVIDIA driver detected" - hint "Install the NVIDIA driver: https://www.nvidia.com/Download/index.aspx" - missing=1 - elif ! nvidia-smi --query-gpu=name --format=csv,noheader &>/dev/null; then - err "nvidia-smi present but NVML calls fail — likely a driver/library mismatch" - hint "Reboot, or reinstall the matching NVIDIA driver package" - missing=1 + if _variant_is_rocm "$variant"; then + if [ "$LUCEBOX_HOST_HAS_AMD_GPU" != "1" ]; then + err "ROCm image selected but no working AMD GPU was detected" + hint "Install ROCm/amd-smi, or choose LUCEBOX_VARIANT=cuda12 on an NVIDIA build." + missing=1 + fi + if [ "$LUCEBOX_HOST_HAS_KFD" != "1" ]; then + err "/dev/kfd is missing or not accessible" + hint "Add the user to the render group, then re-login: sudo usermod -aG render \"$USER\"" + missing=1 + fi + if [ "$LUCEBOX_HOST_HAS_DRI" != "1" ]; then + err "no accessible /dev/dri/renderD* device was found" + hint "Add the user to the render and video groups, then re-login." + missing=1 + fi + else + if [ "$LUCEBOX_HOST_HAS_NVIDIA_GPU" != "1" ]; then + err "CUDA image selected but no working NVIDIA GPU was detected" + hint "Install the NVIDIA driver, or choose LUCEBOX_VARIANT=rocm on an AMD build." + missing=1 + fi fi [ "$missing" = "0" ] || exit 1 } require_ctk() { + local variant="${1:-$(pick_variant)}" + _variant_is_rocm "$variant" && return 0 case "$LUCEBOX_HOST_HAS_CTK" in runtime|cdi) return 0 ;; installed-unwired) @@ -430,10 +654,9 @@ require_systemd() { # All the Python-CLI subcommands share the same docker run incantation: # mount the host docker socket (so the in-container CLI can spawn server / # bench containers on the host daemon), mount $HOME at the same path (so -# paths look identical in and out), and pass host facts via env. When an -# NVIDIA GPU is detected we also pass --gpus all so the orchestrator can -# call nvidia-smi during profile snapshot export; without it nvidia_smi_csv (and -# any downstream power/utilization fields) come back empty. +# paths look identical in and out), and pass host facts via env. The selected +# image gets its native accelerator contract: --gpus all for +# CUDA; /dev/kfd + /dev/dri and the render/video groups for ROCm. DOCKER_SOCK_PATH="${DOCKER_HOST:-/var/run/docker.sock}" DOCKER_SOCK_PATH="${DOCKER_SOCK_PATH#unix://}" @@ -467,6 +690,27 @@ _append_scalar_env() { return 0 } +# Append the Docker accelerator contract for the chosen image variant. +# CUDA and ROCm use fundamentally different runtime flags; keeping this in +# one helper prevents the orchestrator, canonical server argv, and fallback +# server path from drifting apart. +_append_gpu_args() { # usage: _append_gpu_args arrayname variant + # shellcheck disable=SC2178 + local -n _gpu_arr="$1" + local variant="$2" + if _variant_is_rocm "$variant"; then + _gpu_arr+=( + --device /dev/kfd + --device /dev/dri + --group-add video + --group-add render + --security-opt seccomp=unconfined + ) + else + _gpu_arr+=(--gpus all) + fi +} + # Pick docker's interactive flags: -it on a real tty, -i otherwise. # Writes into a caller-supplied array via nameref. This MUST run in the # caller's scope (not a subshell or `< <(...)` process substitution): the @@ -488,9 +732,7 @@ build_orchestrator_argv() { local tty=() _set_tty_flags tty local argv=(docker run --rm "${tty[@]}") - if [ "${LUCEBOX_HOST_GPU_VENDOR:-none}" = "nvidia" ]; then - argv+=(--gpus all) - fi + _append_gpu_args argv "$variant" argv+=(--name "${CONTAINER_NAME}-cli-$$") argv+=(--user "$(id -u):$(id -g)") # Only bind-mount the docker socket when DOCKER_HOST actually points @@ -541,11 +783,11 @@ cmd_serve() { # If stage 1 fails (image not pulled yet, no config), fall back to a # conservative docker run — the container's own VRAM-tiered autotune # picks reasonable defaults from there. - require_host_prereqs ensure_probed - require_ctk local variant variant=$(pick_variant) + require_host_prereqs "$variant" + require_ctk "$variant" # Pre-flight: refuse to stomp on something that's already serving this # slot. Three states to distinguish, because silently `docker rm -f`-ing @@ -615,10 +857,10 @@ cmd_serve() { # with "source: unknown" any time print-serve-argv fails. local fallback_argv=(docker run --rm --name "$CONTAINER_NAME" - --gpus all -p "$DEFAULT_PORT:8080" -v "$HOME:$HOME" -v "$fallback_models:/opt/lucebox-hub/server/models") + _append_gpu_args fallback_argv "$variant" _append_host_env fallback_argv fallback_argv+=("${IMAGE_BASE}:${variant}") _serve_and_track "${fallback_argv[@]}" @@ -660,8 +902,10 @@ _serve_and_track() { } cmd_systemd_install() { - require_host_prereqs ensure_probed + local variant + variant=$(pick_variant) + require_host_prereqs "$variant" require_systemd "service install" local docker_bin docker_bin=$(command -v docker) @@ -696,7 +940,7 @@ RestartSec=10 SuccessExitStatus=143 SIGTERM Environment=PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin Environment=LUCEBOX_IMAGE=$IMAGE_BASE -Environment=LUCEBOX_VARIANT=$(pick_variant) +Environment=LUCEBOX_VARIANT=$variant Environment=LUCEBOX_PORT=$DEFAULT_PORT Environment=LUCEBOX_CONTAINER=$CONTAINER_NAME Environment=LUCEBOX_MODELS=$DEFAULT_MODELS_DIR @@ -839,9 +1083,10 @@ cmd_pull() { # Pull has to run on the host. Delegating this into the container creates a # stale-image trap: docker may start an old local tag before the fresh tag # has been pulled. - require_host_prereqs + ensure_probed local variant variant=$(pick_variant) + require_host_prereqs "$variant" info "Pulling ${IMAGE_BASE}:${variant}" exec docker pull "${IMAGE_BASE}:${variant}" } @@ -1011,36 +1256,67 @@ cmd_check() { _row 0 "docker daemon" "not installed — https://docs.docker.com/engine/install/" fi - # nvidia container toolkit - case "$LUCEBOX_HOST_HAS_CTK" in - runtime) _row 1 "nvidia ctk" "wired into docker (runtime)" ;; - cdi) _row 1 "nvidia ctk" "wired via CDI (nvidia.com/gpu)" ;; - installed-unwired) _row warn "nvidia ctk" "installed but not registered with docker — sudo nvidia-ctk runtime configure --runtime=docker && sudo systemctl restart docker" ;; - none|*) _row 0 "nvidia ctk" "not installed — https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html" ;; - esac + if [ "$LUCEBOX_HOST_HAS_NVIDIA_GPU" = "1" ]; then + # nvidia container toolkit + case "$LUCEBOX_HOST_HAS_CTK" in + runtime) _row 1 "nvidia ctk" "wired into docker (runtime)" ;; + cdi) _row 1 "nvidia ctk" "wired via CDI (nvidia.com/gpu)" ;; + installed-unwired) _row warn "nvidia ctk" "installed but not registered with docker — sudo nvidia-ctk runtime configure --runtime=docker && sudo systemctl restart docker" ;; + none|*) _row 0 "nvidia ctk" "not installed — https://docs.nvidia.com/datacenter/cloud-native/container-toolkit/latest/install-guide.html" ;; + esac - # nvidia-smi + driver - if [ "$LUCEBOX_HOST_GPU_VENDOR" = "nvidia" ]; then if [ "$LUCEBOX_HOST_DRIVER_MAJOR" -ge "$MIN_DRIVER_CUDA12" ]; then _row 1 "nvidia driver" "$LUCEBOX_HOST_DRIVER_VERSION (≥ $MIN_DRIVER_CUDA12 required for cuda12)" else _row 0 "nvidia driver" "$LUCEBOX_HOST_DRIVER_VERSION (< $MIN_DRIVER_CUDA12 — cuda12 image will fail)" fi - elif command -v nvidia-smi &>/dev/null; then - _row 0 "nvidia driver" "nvidia-smi present but NVML calls fail — driver/library mismatch, try reboot" - else - _row 0 "nvidia driver" "nvidia-smi not found — install the NVIDIA driver" - fi - - # GPU detail - if [ "$LUCEBOX_HOST_GPU_VENDOR" = "nvidia" ]; then - _row 1 "gpu" "$LUCEBOX_HOST_GPU_NAME × $LUCEBOX_HOST_GPU_COUNT (sm_$LUCEBOX_HOST_GPU_SM, ${LUCEBOX_HOST_VRAM_GB} GB VRAM)" + _row 1 "nvidia gpu" "$LUCEBOX_HOST_GPU_NAME × $LUCEBOX_HOST_GPU_COUNT (sm_$LUCEBOX_HOST_GPU_SM, ${LUCEBOX_HOST_VRAM_GB} GB VRAM)" # cuda12 image arch coverage: sm_75;80;86;89;90;120 (see docker-bake.hcl) case "$LUCEBOX_HOST_GPU_SM" in 75|80|86|89|90|120) _row 1 "cuda12 arch" "sm_$LUCEBOX_HOST_GPU_SM covered by image" ;; - "") _row warn "cuda12 arch" "compute_cap not detected" ;; + "") _row warn "cuda12 arch" "compute capability not detected" ;; *) _row warn "cuda12 arch" "sm_$LUCEBOX_HOST_GPU_SM not in image arch list (75;80;86;89;90;120)" ;; esac + elif command -v nvidia-smi &>/dev/null; then + _row 0 "nvidia driver" "nvidia-smi present but NVML calls fail — driver/library mismatch, try reboot" + fi + + if [ "$LUCEBOX_HOST_HAS_AMD_GPU" = "1" ]; then + if [ -n "$LUCEBOX_HOST_ROCM_VERSION" ]; then + _row 1 "rocm" "$LUCEBOX_HOST_ROCM_VERSION" + else + _row warn "rocm" "GPU detected, userspace version unavailable" + fi + _row 1 "amd gpu" "${LUCEBOX_HOST_AMD_GPU_COUNT} device(s); primary ${LUCEBOX_HOST_AMD_GPU_NAME} (${LUCEBOX_HOST_AMD_GPU_ARCH}, ${LUCEBOX_HOST_AMD_VRAM_GB} GB effective)" + local amd_line amd_device_idx amd_device_name amd_device_arch amd_device_mem + while IFS= read -r amd_line; do + [ -n "$amd_line" ] || continue + amd_device_idx=$(printf '%s' "$amd_line" | awk -F',' '{gsub(/^[[:space:]]+|[[:space:]]+$/, "", $1); print $1}') + amd_device_name=$(printf '%s' "$amd_line" | awk -F',' '{gsub(/^[[:space:]]+|[[:space:]]+$/, "", $4); print $4}') + amd_device_arch=$(printf '%s' "$amd_line" | awk -F',' '{gsub(/^[[:space:]]+|[[:space:]]+$/, "", $5); print $5}') + amd_device_mem=$(printf '%s' "$amd_line" | awk -F',' '{gsub(/^[[:space:]]+|[[:space:]]+$/, "", $6); print $6}') + _row 1 " amd[$amd_device_idx]" "$amd_device_name ($amd_device_arch, $amd_device_mem physical)" + done <<<"$LUCEBOX_HOST_AMD_GPU_LIST_CSV" + case "$LUCEBOX_HOST_AMD_GPU_ARCH" in + gfx1100|gfx1151|gfx1200|gfx1201) _row 1 "rocm arch" "$LUCEBOX_HOST_AMD_GPU_ARCH covered by published image" ;; + "") _row warn "rocm arch" "gfx architecture not detected" ;; + *) _row warn "rocm arch" "$LUCEBOX_HOST_AMD_GPU_ARCH not in published image arch list" ;; + esac + if [ "$LUCEBOX_HOST_HAS_KFD" = "1" ]; then + _row 1 "amd /dev/kfd" "accessible" + else + _row 0 "amd /dev/kfd" "missing or inaccessible — user needs the render group" + fi + if [ "$LUCEBOX_HOST_HAS_DRI" = "1" ]; then + _row 1 "amd /dev/dri" "render node accessible" + else + _row 0 "amd /dev/dri" "no accessible renderD* node — user needs render/video groups" + fi + fi + + if [ "$LUCEBOX_HOST_HAS_NVIDIA_GPU" != "1" ] \ + && [ "$LUCEBOX_HOST_HAS_AMD_GPU" != "1" ]; then + _row 0 "gpu" "no supported NVIDIA or AMD GPU detected" fi # systemd @@ -1052,16 +1328,24 @@ cmd_check() { _row warn "user systemd" "not available — '$SCRIPT_NAME install' (service unit) won't work; '$SCRIPT_NAME serve' (foreground) will" fi - # image we'd pull — marked ✗ when the host clearly can't run cuda12 - # (no nvidia driver, or no CTK wired into docker). It's still useful - # to print the line so the user knows what would be pulled, but a - # green ✓ would be misleading. - if [ "$LUCEBOX_HOST_GPU_VENDOR" != "nvidia" ]; then - _row 0 "image" "${IMAGE_BASE}:${variant} — requires NVIDIA driver" - elif [ "$LUCEBOX_HOST_HAS_CTK" = "none" ] || [ "$LUCEBOX_HOST_HAS_CTK" = "installed-unwired" ]; then - _row 0 "image" "${IMAGE_BASE}:${variant} — needs NVIDIA Container Toolkit wired into docker" + # Selected backend and image. On heterogeneous builds the unselected APU + # stays visible in the inventory above but does not change this decision. + if _variant_is_rocm "$variant"; then + if [ "$LUCEBOX_HOST_HAS_AMD_GPU" != "1" ]; then + _row 0 "image" "${IMAGE_BASE}:${variant} — requires an AMD GPU" + elif [ "$LUCEBOX_HOST_HAS_KFD" != "1" ] || [ "$LUCEBOX_HOST_HAS_DRI" != "1" ]; then + _row 0 "image" "${IMAGE_BASE}:${variant} — needs accessible /dev/kfd and /dev/dri" + else + _row 1 "image" "${IMAGE_BASE}:${variant} (AMD selected)" + fi else - _row 1 "image" "${IMAGE_BASE}:${variant}" + if [ "$LUCEBOX_HOST_HAS_NVIDIA_GPU" != "1" ]; then + _row 0 "image" "${IMAGE_BASE}:${variant} — requires an NVIDIA GPU" + elif [ "$LUCEBOX_HOST_HAS_CTK" = "none" ] || [ "$LUCEBOX_HOST_HAS_CTK" = "installed-unwired" ]; then + _row 0 "image" "${IMAGE_BASE}:${variant} — needs NVIDIA Container Toolkit wired into docker" + else + _row 1 "image" "${IMAGE_BASE}:${variant} (NVIDIA selected)" + fi fi # RAM / cores (informational) _row 1 "host" "${LUCEBOX_HOST_NPROC} cpus, ${LUCEBOX_HOST_RAM_GB} GB RAM" @@ -1070,7 +1354,6 @@ cmd_check() { cmd_in_container() { # Generic dispatcher: anything that isn't a systemd action goes here. # Runs the in-container Python CLI with the supplied argv. - require_host_prereqs ensure_probed # CTK isn't strictly required for every subcommand (e.g. `config get` # or `autotune` only touch local files), but the server-spawning @@ -1078,6 +1361,7 @@ cmd_in_container() { # Letting docker error its own way is fine for the no-CTK case. local variant variant=$(pick_variant) + require_host_prereqs "$variant" local argv mapfile -t argv < <(build_orchestrator_argv "$variant" "$@") exec "${argv[@]}" @@ -1112,10 +1396,10 @@ _lucebox_container_running() { # CLI sees consistent overrides whichever route it took: HOME, every # LUCEBOX_HOST_*, the image/port/container/models scalars, and HF_TOKEN. cmd_exec_in_container() { - require_host_prereqs ensure_probed local variant variant=$(pick_variant) + require_host_prereqs "$variant" local tty=() _set_tty_flags tty local argv=(docker exec "${tty[@]}") @@ -1196,7 +1480,7 @@ Direct server invocation (foreground, no systemd): Provisioning + workloads (delegated to the in-container Python CLI): check host + docker readiness report - pull docker pull the cuda12 image + pull docker pull the auto-selected CUDA or ROCm image update re-run the bootstrap installer to upgrade this script completion print shell completion script (bash / zsh / fish) models list / download / activate model presets @@ -1209,7 +1493,7 @@ Misc: Environment overrides: LUCEBOX_IMAGE image name without tag (default: ghcr.io/luce-org/lucebox-hub) - LUCEBOX_VARIANT image tag to pull/run (default: cuda12) + LUCEBOX_VARIANT image tag override (default: cuda12 on NVIDIA, rocm on AMD) LUCEBOX_PORT host port for the server (default: 8080) LUCEBOX_CONTAINER server container name (default: lucebox) LUCEBOX_MODELS host model directory (default: \$XDG_DATA_HOME/lucebox/models diff --git a/lucebox/README.md b/lucebox/README.md index 747a49eef..6af3d2a02 100644 --- a/lucebox/README.md +++ b/lucebox/README.md @@ -7,9 +7,12 @@ and is invoked from the host via the [`lucebox.sh`](../lucebox.sh) wrapper: lucebox.sh config get lucebox.sh print-run -The wrapper is the only thing that runs on the host; everything else (host +The wrapper is the only thing that runs on the host. It selects CUDA for a +working NVIDIA GPU (including RTX 3090 + Strix builds) and ROCm for AMD builds +(including R9700 + Strix), then applies the matching Docker device flags. +Everything else (host checks, TOML config, docker daemon calls, model download) is Python in the -container. Host facts (driver, GPU, RAM, VRAM, systemd availability) are +container. Host facts (driver/runtime, GPU, RAM, VRAM, systemd availability) are passed in via `LUCEBOX_HOST_*` environment variables so the Python side doesn't reprobe. The autotune sweep, profiling, and agent-client launchers land in follow-up PRs. diff --git a/lucebox/src/lucebox/autotune.py b/lucebox/src/lucebox/autotune.py index 51b4f8424..5541cf271 100644 --- a/lucebox/src/lucebox/autotune.py +++ b/lucebox/src/lucebox/autotune.py @@ -18,7 +18,7 @@ def runtime_from_host(host: HostFacts) -> DflashRuntime: """Pick a conservative DflashRuntime that 'should work' on this VRAM tier. - Tiers (NVIDIA, baseline = Qwen3.6-27B Q4_K_M ~18 GB total): + Tiers (selected CUDA/ROCm device, baseline = Qwen3.6-27B Q4_K_M ~18 GB total): <12 GB — too small for 27B; pick min ctx as a floor so a fallback start at least gets an error from the daemon rather than a silent OOM. diff --git a/lucebox/src/lucebox/cli.py b/lucebox/src/lucebox/cli.py index 87cc29366..e3c6101bd 100644 --- a/lucebox/src/lucebox/cli.py +++ b/lucebox/src/lucebox/cli.py @@ -6,7 +6,7 @@ Subcommand inventory: check — readiness report config get/set/unset — read / write a single key in config.toml - pull — docker pull the cuda12 image + pull — docker pull the selected CUDA or ROCm image print-run — emit the docker-run command for the server print-serve-argv — same, raw argv lines (consumed by `lucebox serve`) models — list / download presets, activate one diff --git a/lucebox/src/lucebox/config.py b/lucebox/src/lucebox/config.py index b3f63eb0d..6e3d113c8 100644 --- a/lucebox/src/lucebox/config.py +++ b/lucebox/src/lucebox/config.py @@ -257,12 +257,17 @@ def _from_dict(raw: dict[str, Any]) -> Config: nproc=int(host_raw.get("nproc", 0)), ram_gb=int(host_raw.get("ram_gb", 0)), gpu_vendor=host_raw.get("gpu_vendor", "none"), + has_nvidia_gpu=bool(host_raw.get("has_nvidia_gpu", False)), + has_amd_gpu=bool(host_raw.get("has_amd_gpu", False)), gpu_name=str(host_raw.get("gpu_name", "")), gpu_count=int(host_raw.get("gpu_count", 0)), vram_gb=int(host_raw.get("vram_gb", 0)), gpu_sm=str(host_raw.get("gpu_sm", "")), driver_version=str(host_raw.get("driver_version", "")), driver_major=int(host_raw.get("driver_major", 0)), + rocm_version=str(host_raw.get("rocm_version", "")), + has_kfd=bool(host_raw.get("has_kfd", False)), + has_dri=bool(host_raw.get("has_dri", False)), has_systemd=bool(host_raw.get("has_systemd", False)), is_wsl=bool(host_raw.get("is_wsl", False)), has_docker=bool(host_raw.get("has_docker", False)), @@ -470,9 +475,10 @@ def live_config() -> Config: host = from_env() default = Config() + default_variant = "rocm" if host.gpu_vendor == "amd" else "cuda12" return replace( default, - variant=os.environ.get("LUCEBOX_VARIANT", "cuda12"), + variant=os.environ.get("LUCEBOX_VARIANT", default_variant), image=os.environ.get("LUCEBOX_IMAGE", default.image), container_name=os.environ.get("LUCEBOX_CONTAINER", default.container_name), port=int(os.environ.get("LUCEBOX_PORT", str(default.port))), diff --git a/lucebox/src/lucebox/docker_run.py b/lucebox/src/lucebox/docker_run.py index ff3615b26..8e61e355f 100644 --- a/lucebox/src/lucebox/docker_run.py +++ b/lucebox/src/lucebox/docker_run.py @@ -15,7 +15,7 @@ from dataclasses import dataclass from pathlib import Path -from lucebox.types import Config +from lucebox.types import Config, GpuVendor def _host_facts_env() -> list[tuple[str, str]]: @@ -91,6 +91,7 @@ class DockerRunSpec: image: str name: str gpus: bool = True + gpu_vendor: GpuVendor = "nvidia" detach: bool = False remove: bool = True port_publish: tuple[int, int] | None = None # (host, container) @@ -106,8 +107,21 @@ def argv(self) -> list[str]: if self.detach: out.append("-d") out += ["--name", self.name] - if self.gpus: + if self.gpus and self.gpu_vendor == "nvidia": out += ["--gpus", "all"] + elif self.gpus and self.gpu_vendor == "amd": + out += [ + "--device", + "/dev/kfd", + "--device", + "/dev/dri", + "--group-add", + "video", + "--group-add", + "render", + "--security-opt", + "seccomp=unconfined", + ] if self.port_publish is not None: host, container = self.port_publish out += ["-p", f"{host}:{container}"] @@ -141,6 +155,9 @@ def printable(self) -> str: "--volume", "--publish", "--entrypoint", + "--device", + "--group-add", + "--security-opt", } and i + 1 < len(argv): i += 1 out += " " + shlex.quote(argv[i]) @@ -153,7 +170,7 @@ def printable(self) -> str: def server_run_spec(cfg: Config) -> DockerRunSpec: """Long-running OpenAI-compatible server. Foreground (systemd manages - lifecycle), --gpus all, models bind-mounted, DFLASH_* propagated. + lifecycle), vendor-specific GPU devices, models bind-mounted, DFLASH_* propagated. """ # LUCEBOX_HOST_* first so they ride out front in the rendered argv, # making it obvious in `print-run` output what host facts get forwarded. @@ -212,10 +229,23 @@ def server_run_spec(cfg: Config) -> DockerRunSpec: if cfg.dflash.debug_thinking_logits: env.append(("DFLASH_DEBUG_THINKING_LOGITS", "1")) + # The chosen image variant is the runtime contract. This matters on a + # heterogeneous RTX + Strix host: the probe records both vendors, while an + # explicit ``variant=rocm`` must still receive AMD device flags. Unknown + # custom variants fall back to the probe's selected/default vendor. + variant_lower = cfg.variant.lower() + if "rocm" in variant_lower: + gpu_vendor: GpuVendor = "amd" + elif "cuda" in variant_lower: + gpu_vendor = "nvidia" + else: + gpu_vendor = cfg.host.gpu_vendor if cfg.host.gpu_vendor != "none" else "nvidia" + return DockerRunSpec( image=f"{cfg.image}:{cfg.variant}", name=cfg.container_name, gpus=True, + gpu_vendor=gpu_vendor, remove=True, detach=False, port_publish=(cfg.port, 8080), diff --git a/lucebox/src/lucebox/host_check.py b/lucebox/src/lucebox/host_check.py index 2ce8d3889..8f2d8d8d8 100644 --- a/lucebox/src/lucebox/host_check.py +++ b/lucebox/src/lucebox/host_check.py @@ -26,14 +26,15 @@ class CheckResult: def run_checks(host: HostFacts) -> list[CheckResult]: - return [ - _check_docker(host), - _check_nvidia_driver(host), - _check_ctk(host), - _check_ram(host), - _check_vram(host), - _check_systemd(host), - ] + results = [_check_docker(host)] + if host.gpu_vendor == "nvidia": + results += [_check_nvidia_driver(host), _check_ctk(host)] + elif host.gpu_vendor == "amd": + results += [_check_amd_driver(host), _check_amd_devices(host)] + else: + results.append(CheckResult("gpu", "fail", "no supported NVIDIA or AMD GPU detected")) + results += [_check_ram(host), _check_vram(host), _check_systemd(host)] + return results def _check_docker(host: HostFacts) -> CheckResult: @@ -48,15 +49,6 @@ def _check_docker(host: HostFacts) -> CheckResult: def _check_nvidia_driver(host: HostFacts) -> CheckResult: - if host.gpu_vendor != "nvidia": - if host.gpu_vendor == "amd": - return CheckResult( - "gpu", - "fail", - "AMD GPU detected — prebuilt images are NVIDIA-only", - "Build dflash from source with HIP; see dflash/README.md", - ) - return CheckResult("gpu", "fail", "no NVIDIA GPU detected") if not host.driver_version: return CheckResult( "driver", @@ -74,6 +66,33 @@ def _check_nvidia_driver(host: HostFacts) -> CheckResult: return CheckResult("driver", "ok", f"nvidia r{host.driver_major} ({host.driver_version})") +def _check_amd_driver(host: HostFacts) -> CheckResult: + if not host.rocm_version: + return CheckResult( + "rocm", + "warn", + "AMD GPU detected but ROCm userspace version is unknown", + "install amd-smi or hipconfig so Lucebox can verify the ROCm runtime", + ) + return CheckResult("rocm", "ok", f"ROCm {host.rocm_version} ({host.gpu_sm or 'gfx?'})") + + +def _check_amd_devices(host: HostFacts) -> CheckResult: + missing: list[str] = [] + if not host.has_kfd: + missing.append("/dev/kfd") + if not host.has_dri: + missing.append("/dev/dri/renderD*") + if missing: + return CheckResult( + "devices", + "fail", + f"missing or inaccessible: {', '.join(missing)}", + "add the user to the render and video groups, then re-login", + ) + return CheckResult("devices", "ok", "/dev/kfd and /dev/dri are accessible") + + def _check_ctk(host: HostFacts) -> CheckResult: match host.ctk: case "runtime": @@ -147,10 +166,13 @@ def aggregate(results: list[CheckResult]) -> Severity: def render(console: Console, host: HostFacts, results: list[CheckResult]) -> Severity: """Print a status block, return the worst severity.""" summary = f"[bold]Host:[/bold] {host.nproc} CPUs · {host.ram_gb} GB RAM" - if host.gpu_vendor == "nvidia" and host.gpu_name: - summary += f" · {host.gpu_name} · {host.vram_gb} GB VRAM" + ( - f" (sm_{host.gpu_sm})" if host.gpu_sm else "" - ) + if host.gpu_name: + arch = host.gpu_sm + if arch and host.gpu_vendor == "nvidia": + arch = f"sm_{arch}" + summary += f" · {host.gpu_name} · {host.vram_gb} GB VRAM" + if arch: + summary += f" ({arch})" if host.is_wsl: summary += " · WSL2" console.print(summary) @@ -202,10 +224,13 @@ def render_host_facts(console: Console) -> None: ("kernel", os.environ.get("LUCEBOX_HOST_KERNEL", "")), ("wsl_version", os.environ.get("LUCEBOX_HOST_WSL_VERSION", "")), ("docker", os.environ.get("LUCEBOX_HOST_DOCKER_VERSION", "")), + ("gpu_vendor", os.environ.get("LUCEBOX_HOST_GPU_VENDOR", "")), ("nvidia_driver", os.environ.get("LUCEBOX_HOST_DRIVER_VERSION", "")), ("nvidia_ctk", os.environ.get("LUCEBOX_HOST_NVIDIA_CTK_VERSION", "")), + ("rocm", os.environ.get("LUCEBOX_HOST_ROCM_VERSION", "")), ("cpu", os.environ.get("LUCEBOX_HOST_CPU_MODEL", "")), ("cuda_visible_devices", os.environ.get("LUCEBOX_HOST_CUDA_VISIBLE_DEVICES", "")), + ("hip_visible_devices", os.environ.get("LUCEBOX_HOST_HIP_VISIBLE_DEVICES", "")), ] for key, value in facts: display = value if value else "[dim](unset)[/dim]" @@ -223,10 +248,10 @@ def render_host_facts(console: Console) -> None: parts = [c.strip() for c in line.split(",")] if len(parts) >= 7: idx, _uuid, _pci, name, sm, mem, plimit = parts[:7] - console.print( - f" [{idx}] {name} (sm_{sm}, {mem}, {plimit})" - ) + arch = sm if sm.startswith("gfx") else f"sm_{sm}" + detail = ", ".join(part for part in (arch, mem, plimit) if part) + console.print(f" [{idx}] {name} ({detail})") else: console.print(f" {line}") else: - console.print(" gpus [dim](none — nvidia-smi unavailable)[/dim]") + console.print(" gpus [dim](none — GPU probe unavailable)[/dim]") diff --git a/lucebox/src/lucebox/host_facts.py b/lucebox/src/lucebox/host_facts.py index 5deb6721a..c9556d246 100644 --- a/lucebox/src/lucebox/host_facts.py +++ b/lucebox/src/lucebox/host_facts.py @@ -2,7 +2,7 @@ We deliberately don't try to detect anything ourselves on the Python side — inside the container, /proc/meminfo reports the container's view, not the -host's, and nvidia-smi may or may not be available depending on how the +host's, and nvidia-smi/amd-smi may or may not be available depending on how the caller invoked us. The host wrapper is the only thing that can see the truth, and it's already paid for the probe. """ @@ -44,12 +44,17 @@ def from_env() -> HostFacts: nproc=_env_int("LUCEBOX_HOST_NPROC"), ram_gb=_env_int("LUCEBOX_HOST_RAM_GB"), gpu_vendor=vendor, + has_nvidia_gpu=_env_bool("LUCEBOX_HOST_HAS_NVIDIA_GPU"), + has_amd_gpu=_env_bool("LUCEBOX_HOST_HAS_AMD_GPU"), gpu_name=os.environ.get("LUCEBOX_HOST_GPU_NAME", ""), gpu_count=_env_int("LUCEBOX_HOST_GPU_COUNT"), vram_gb=_env_int("LUCEBOX_HOST_VRAM_GB"), gpu_sm=os.environ.get("LUCEBOX_HOST_GPU_SM", ""), driver_version=os.environ.get("LUCEBOX_HOST_DRIVER_VERSION", ""), driver_major=_env_int("LUCEBOX_HOST_DRIVER_MAJOR"), + rocm_version=os.environ.get("LUCEBOX_HOST_ROCM_VERSION", ""), + has_kfd=_env_bool("LUCEBOX_HOST_HAS_KFD"), + has_dri=_env_bool("LUCEBOX_HOST_HAS_DRI"), has_systemd=_env_bool("LUCEBOX_HOST_HAS_SYSTEMD"), is_wsl=_env_bool("LUCEBOX_HOST_IS_WSL"), has_docker=_env_bool("LUCEBOX_HOST_HAS_DOCKER"), diff --git a/lucebox/src/lucebox/types.py b/lucebox/src/lucebox/types.py index e1d3620d7..e9e2fa3ec 100644 --- a/lucebox/src/lucebox/types.py +++ b/lucebox/src/lucebox/types.py @@ -40,12 +40,20 @@ class HostFacts: nproc: int = 0 ram_gb: int = 0 gpu_vendor: GpuVendor = "none" + has_nvidia_gpu: bool = False + has_amd_gpu: bool = False gpu_name: str = "" gpu_count: int = 0 vram_gb: int = 0 - gpu_sm: str = "" # e.g. "120" — matches docker-bake arch lists - driver_version: str = "" # e.g. "595.71.05" + # NVIDIA compute capability without the dot ("120") or AMD gfx target + # ("gfx1151"). The historical ``gpu_sm`` name stays in the serialized + # contract for compatibility with HOST_INFO consumers. + gpu_sm: str = "" + driver_version: str = "" # NVIDIA driver, e.g. "595.71.05" driver_major: int = 0 + rocm_version: str = "" # AMD ROCm userspace, e.g. "7.2.4" + has_kfd: bool = False # /dev/kfd exists and is accessible to the user + has_dri: bool = False # at least one accessible /dev/dri/renderD* node has_systemd: bool = False is_wsl: bool = False has_docker: bool = False diff --git a/lucebox/tests/test_check.py b/lucebox/tests/test_check.py index 3fdd469d9..ba48b4d7d 100644 --- a/lucebox/tests/test_check.py +++ b/lucebox/tests/test_check.py @@ -116,3 +116,53 @@ def stub() -> HostFacts: assert result.exit_code == 1 # Host facts block still printed despite the failure. assert "Host facts" in result.stdout + + +def test_amd_rocm_host_passes_without_nvidia_ctk(monkeypatch: pytest.MonkeyPatch) -> None: + """ROCm readiness uses /dev/kfd + /dev/dri and never requires NVIDIA CTK.""" + + def stub() -> HostFacts: + return HostFacts( + nproc=32, + ram_gb=125, + gpu_vendor="amd", + has_amd_gpu=True, + gpu_name="AMD Radeon AI PRO R9700", + gpu_count=2, + vram_gb=31, + gpu_sm="gfx1201", + rocm_version="7.2.4", + has_kfd=True, + has_dri=True, + has_systemd=True, + has_docker=True, + docker_version="29.1.3", + ctk="none", + ) + + monkeypatch.setattr("lucebox.host_facts.from_env", stub) + monkeypatch.setattr("lucebox.cli.from_env", stub) + result = CliRunner().invoke(app, ["check"]) + assert result.exit_code == 0, result.stdout + assert "ROCm 7.2.4" in result.stdout + assert "gfx1201" in result.stdout + assert "/dev/kfd and /dev/dri are accessible" in result.stdout + assert "NVIDIA Container Toolkit" not in result.stdout + + +def test_amd_rocm_host_fails_when_devices_are_inaccessible() -> None: + host = HostFacts( + gpu_vendor="amd", + has_amd_gpu=True, + gpu_name="AMD Radeon Graphics", + vram_gb=125, + gpu_sm="gfx1151", + rocm_version="7.2.4", + has_docker=True, + has_kfd=False, + has_dri=True, + ) + results = host_check.run_checks(host) + devices = next(result for result in results if result.name == "devices") + assert devices.severity == "fail" + assert "/dev/kfd" in devices.message diff --git a/lucebox/tests/test_config.py b/lucebox/tests/test_config.py index 2ec795a12..9c2b913a3 100644 --- a/lucebox/tests/test_config.py +++ b/lucebox/tests/test_config.py @@ -176,6 +176,24 @@ def test_live_config_uses_recommend_preset_indirectly(tmp_path: Path) -> None: assert cfg.model.preset == "" +def test_live_config_selects_rocm_for_amd(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("LUCEBOX_HOST_GPU_VENDOR", "amd") + monkeypatch.delenv("LUCEBOX_VARIANT", raising=False) + + cfg = config.live_config() + + assert cfg.variant == "rocm" + + +def test_live_config_variant_override_wins_on_amd(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setenv("LUCEBOX_HOST_GPU_VENDOR", "amd") + monkeypatch.setenv("LUCEBOX_VARIANT", "test-cuda12") + + cfg = config.live_config() + + assert cfg.variant == "test-cuda12" + + def test_seed_dflash_writes_heuristic_when_absent(tmp_path: Path) -> None: """First-time activate seeds the VRAM-tier heuristic into config.toml. diff --git a/lucebox/tests/test_docker_run.py b/lucebox/tests/test_docker_run.py index bb888514f..0f8999ea1 100644 --- a/lucebox/tests/test_docker_run.py +++ b/lucebox/tests/test_docker_run.py @@ -11,7 +11,7 @@ from pathlib import Path from lucebox.download import PRESETS -from lucebox.types import Config, DflashRuntime, ModelMeta +from lucebox.types import Config, DflashRuntime, HostFacts, ModelMeta from lucebox import docker_run @@ -63,6 +63,23 @@ def test_argv_flags_and_ordering() -> None: assert argv.index("--shm-size") < argv.index("img:tag") +def test_argv_amd_uses_rocm_device_contract() -> None: + spec = docker_run.DockerRunSpec(image="img:rocm", name="box", gpu_vendor="amd") + argv = spec.argv() + assert "--gpus" not in argv + assert ["--device", "/dev/kfd"] == argv[ + argv.index("--device") : argv.index("--device") + 2 + ] + assert "/dev/dri" in argv + assert ["--group-add", "video"] == argv[ + argv.index("--group-add") : argv.index("--group-add") + 2 + ] + assert "render" in argv + assert ["--security-opt", "seccomp=unconfined"] == argv[ + argv.index("--security-opt") : argv.index("--security-opt") + 2 + ] + + def test_printable_glues_value_taking_flags() -> None: spec = docker_run.DockerRunSpec( image="img:tag", @@ -148,6 +165,22 @@ def test_server_run_spec_top_level_shape(tmp_path: Path) -> None: assert (str(tmp_path), "/opt/lucebox-hub/server/models") in spec.volumes +def test_server_run_spec_rocm_uses_amd_devices_on_heterogeneous_host(tmp_path: Path) -> None: + cfg = Config( + variant="rocm", + models_dir=tmp_path, + # The generic probe may select NVIDIA by default on RTX + Strix. The + # explicit image variant must still control Docker's device contract. + host=HostFacts(gpu_vendor="nvidia", has_nvidia_gpu=True, has_amd_gpu=True), + ) + spec = docker_run.server_run_spec(cfg) + assert spec.gpu_vendor == "amd" + argv = spec.argv() + assert "--gpus" not in argv + assert "/dev/kfd" in argv + assert "/dev/dri" in argv + + def test_server_run_spec_always_emits_core_dflash_env(tmp_path: Path) -> None: cfg = Config(models_dir=tmp_path, dflash=DflashRuntime(budget=22, max_ctx=32768)) env = _env(docker_run.server_run_spec(cfg)) diff --git a/lucebox/tests/test_models_cli.py b/lucebox/tests/test_models_cli.py index f44583044..744f40c29 100644 --- a/lucebox/tests/test_models_cli.py +++ b/lucebox/tests/test_models_cli.py @@ -137,6 +137,9 @@ def test_installed_helpers_track_presence( target = cfg.models_dir / laguna.target_file target.parent.mkdir(parents=True, exist_ok=True) - target.write_bytes(b"x" * (5 * 10**9)) + # Apparent size is what the CLI reports; a sparse file exercises the same + # stat path without allocating and writing a 5 GB Python byte string. + with target.open("wb") as sparse: + sparse.truncate(5 * 10**9) assert download_mod.installed_status(cfg, laguna) == "installed" assert download_mod.installed_size_gb(cfg, laguna) == pytest.approx(5.0, rel=0.01) diff --git a/scripts/test_lucebox_sh.sh b/scripts/test_lucebox_sh.sh index 3960760f6..f858690ba 100755 --- a/scripts/test_lucebox_sh.sh +++ b/scripts/test_lucebox_sh.sh @@ -32,6 +32,27 @@ if [ ! -f "$SCRIPT" ]; then exit 1 fi +# Make the whole suite safe on a contributor workstation. Several dispatch +# smoke tests intentionally invoke start/stop/install/uninstall/pull up to the +# missing-prerequisite boundary; without global shims those commands can touch +# a real user service or pull a multi-GB image on a fully provisioned host. +SUITE_SANDBOX=$(mktemp -d "${TMPDIR:-/tmp}/lucebox-shell-suite.XXXXXX") +SUITE_SHIMS="$SUITE_SANDBOX/shims" +mkdir -p "$SUITE_SHIMS" "$SUITE_SANDBOX/home" "$SUITE_SANDBOX/xdg" "$SUITE_SANDBOX/data" +for binname in docker systemctl journalctl loginctl; do + cat > "$SUITE_SHIMS/$binname" <<'STUB' +#!/usr/bin/env bash +exit 1 +STUB + chmod +x "$SUITE_SHIMS/$binname" +done +export HOME="$SUITE_SANDBOX/home" +export XDG_CONFIG_HOME="$SUITE_SANDBOX/xdg" +export XDG_DATA_HOME="$SUITE_SANDBOX/data" +export LUCEBOX_HOME="$SUITE_SANDBOX/home/.lucebox" +export PATH="$SUITE_SHIMS:$PATH" +trap 'rm -rf "$SUITE_SANDBOX"' EXIT + # entrypoint.sh ships with the docker-stack PR (#334). When it's absent # (e.g. on the lucebox-cli branch in isolation), skip the entire suite — # every section below either references $ENTRYPOINT in shellcheck targets, @@ -284,6 +305,7 @@ STUB LUCEBOX_HOST_HAS_DOCKER=1 \ LUCEBOX_HOST_HAS_CTK=runtime \ LUCEBOX_HOST_GPU_VENDOR=nvidia \ + LUCEBOX_HOST_HAS_NVIDIA_GPU=1 \ _LUCEBOX_HOST_PROBED=1 \ NO_COLOR=1 \ timeout 10 bash "$SCRIPT" install 2>&1 @@ -935,6 +957,7 @@ _run_wrapper_capture_docker() { LUCEBOX_HOST_HAS_DOCKER=1 \ LUCEBOX_HOST_HAS_CTK=runtime \ LUCEBOX_HOST_GPU_VENDOR=nvidia \ + LUCEBOX_HOST_HAS_NVIDIA_GPU=1 \ LUCEBOX_HOST_DRIVER_MAJOR=550 \ LUCEBOX_HOST_DRIVER_VERSION="550.00" \ LUCEBOX_HOST_GPU_NAME="Fake GPU" \ @@ -1071,6 +1094,125 @@ test_usage_mentions_exec_routing() { } test_usage_mentions_exec_routing "usage documents docker exec routing + --no-exec flag" +# ── cross-vendor GPU selection / Docker contract ────────────────────────── +test_amd_smi_parser() { + local label="$1" fn out + fn=$(awk '/^_parse_amd_smi_csv\(\) \{/,/^\}/' "$SCRIPT") + out=$(bash -c "$fn"$'\n''_parse_amd_smi_csv' <<'CSV' +gpu,market_name,vendor_id,vendor_name,subvendor_id,device_id,subsystem_id,rev_id,asic_serial,oam_id,num_compute_units,target_graphics_version,type,vendor,size,bit_width,max_bandwidth +0,AMD Radeon AI PRO R9700,0x1002,AMD,0xf111,0x7551,0x000a,0xc0,serial,N/A,64,gfx1201,GDDR6,SAMSUNG,32624,256,N/A +1,AMD Radeon Graphics,0x1002,AMD,0xf111,0x1586,0x000a,0xc1,serial,N/A,40,gfx1151,GDDR7,UNKNOWN,512,256,N/A +CSV +) + if ! grep -qF '0|AMD Radeon AI PRO R9700|gfx1201|32624' <<<"$out" \ + || ! grep -qF '1|AMD Radeon Graphics|gfx1151|512' <<<"$out"; then + report fail "$label" "unexpected normalized rows: $out" + return + fi + report ok "$label" +} +test_amd_smi_parser "amd-smi parser recognizes R9700 + Strix" + +test_variant_autoselection() { + local label="$1" fn common got + fn=$(awk '/^pick_variant\(\) \{/,/^\}/' "$SCRIPT") + common=$'_lucebox_config_get(){ :; }\nensure_probed(){ :; }\nLUCEBOX_VARIANT=""\n' + + got=$(bash -c "$fn"$'\n'"$common"$'LUCEBOX_HOST_HAS_NVIDIA_GPU=1\nLUCEBOX_HOST_HAS_AMD_GPU=1\npick_variant') + [ "$got" = "cuda12" ] || { report fail "$label" "mixed RTX + Strix chose $got"; return; } + + got=$(bash -c "$fn"$'\n'"$common"$'LUCEBOX_HOST_HAS_NVIDIA_GPU=0\nLUCEBOX_HOST_HAS_AMD_GPU=1\npick_variant') + [ "$got" = "rocm" ] || { report fail "$label" "R9700 + Strix chose $got"; return; } + + got=$(bash -c "$fn"$'\n'"$common"$'LUCEBOX_VARIANT=rocm\nLUCEBOX_HOST_HAS_NVIDIA_GPU=1\nLUCEBOX_HOST_HAS_AMD_GPU=1\npick_variant') + [ "$got" = "rocm" ] || { report fail "$label" "explicit override chose $got"; return; } + report ok "$label" +} +test_variant_autoselection "variant selection: RTX+Strix=cuda12, R9700+Strix=rocm" + +test_mixed_rtx_strix_probe_prefers_cuda() { + local label="$1" sandbox shim_dir out + sandbox=$(mktemp -d -t lucebox-mixed-gpu.XXXXXX) + shim_dir="$sandbox/bin" + mkdir -p "$shim_dir" + cat > "$shim_dir/nvidia-smi" <<'STUB' +#!/usr/bin/env bash +case "$*" in + *"name,memory.total,driver_version,compute_cap"*) + echo "NVIDIA GeForce RTX 3090, 24576, 550.00, 8.6" ;; + *"index,uuid,pci.bus_id,name,compute_cap,memory.total,power.limit"*) + echo "0, GPU-test, 00000000:01:00.0, NVIDIA GeForce RTX 3090, 8.6, 24576 MiB, 350.00 W" ;; + *"--query-gpu=name"*) echo "NVIDIA GeForce RTX 3090" ;; + "-L") echo "GPU 0: NVIDIA GeForce RTX 3090" ;; +esac +exit 0 +STUB + cat > "$shim_dir/amd-smi" <<'STUB' +#!/usr/bin/env bash +if [ "${1:-}" = "version" ]; then + echo "AMDSMI Tool: test | ROCm version: 7.2.4 | Platform: Linux Baremetal" + exit 0 +fi +cat <<'CSV' +gpu,market_name,vendor_id,vendor_name,subvendor_id,device_id,subsystem_id,rev_id,asic_serial,oam_id,num_compute_units,target_graphics_version,type,vendor,size,bit_width,max_bandwidth +0,AMD Radeon Graphics,0x1002,AMD,0xf111,0x1586,0x000a,0xc1,serial,N/A,40,gfx1151,GDDR7,UNKNOWN,512,256,N/A +CSV +STUB + cat > "$shim_dir/docker" <<'STUB' +#!/usr/bin/env bash +case "${1:-}" in + ps) exit 0 ;; + version) echo "29.1.3" ;; +esac +exit 0 +STUB + for binname in nvidia-container-runtime; do + printf '#!/usr/bin/env bash\nexit 0\n' > "$shim_dir/$binname" + done + chmod +x "$shim_dir"/* + out=$(HOME="$sandbox" LUCEBOX_HOME="$sandbox/.lucebox" \ + PATH="$shim_dir:$PATH" NO_COLOR=1 bash "$SCRIPT" check 2>&1) + rm -rf "$sandbox" + if ! grep -qF "NVIDIA GeForce RTX 3090" <<<"$out" \ + || ! grep -qF "AMD Radeon Graphics" <<<"$out" \ + || ! grep -qF ":cuda12 (NVIDIA selected)" <<<"$out"; then + report fail "$label" "mixed probe output: $(printf '%s' "$out" | head -20)" + return + fi + report ok "$label" +} +test_mixed_rtx_strix_probe_prefers_cuda "mixed RTX 3090 + Strix probe selects CUDA" + +test_cross_vendor_docker_args() { + local label="$1" helpers cuda_args rocm_args pinned_rocm_args + helpers=$(awk '/^_variant_is_rocm\(\) \{/,/^\}/' "$SCRIPT")$'\n' + helpers+=$(awk '/^_append_gpu_args\(\) \{/,/^\}/' "$SCRIPT") + cuda_args=$(bash -c "$helpers"$'\n''a=(); _append_gpu_args a cuda12; printf "%s\n" "${a[@]}"') + rocm_args=$(bash -c "$helpers"$'\n''a=(); _append_gpu_args a rocm; printf "%s\n" "${a[@]}"') + pinned_rocm_args=$(bash -c "$helpers"$'\n''a=(); _append_gpu_args a 0.3.0-rocm; printf "%s\n" "${a[@]}"') + if ! grep -qF -- '--gpus' <<<"$cuda_args" || ! grep -qF 'all' <<<"$cuda_args"; then + report fail "$label" "CUDA args missing --gpus all" + return + fi + for expected in /dev/kfd /dev/dri video render seccomp=unconfined; do + if ! grep -qF "$expected" <<<"$rocm_args"; then + report fail "$label" "ROCm args missing $expected" + return + fi + done + if grep -qF -- '--gpus' <<<"$rocm_args"; then + report fail "$label" "ROCm args incorrectly contain --gpus" + return + fi + if ! grep -qF '/dev/kfd' <<<"$pinned_rocm_args" \ + || grep -qF -- '--gpus' <<<"$pinned_rocm_args"; then + report fail "$label" "versioned ROCm tag did not select ROCm args" + return + fi + report ok "$label" +} +test_cross_vendor_docker_args "Docker args select CUDA or ROCm device contract" + # ── TTY flag selection. Regression guard for the process-substitution bug: # _set_tty_flags must run in the CALLER's scope so `[ -t 1 ]` inspects the # real terminal. If it is ever moved back behind `< <(...)` or `$(...)`, diff --git a/server/scripts/entrypoint.sh b/server/scripts/entrypoint.sh index cd0b5184e..78f96c60f 100755 --- a/server/scripts/entrypoint.sh +++ b/server/scripts/entrypoint.sh @@ -8,8 +8,8 @@ # Fallback path: a user runs the image directly (`docker run --gpus all # ghcr.io/luce-org/lucebox-hub:cuda12`) with no env-var prep. We then do a # minimal VRAM-tiered autotune — same tiers as `lucebox autotune`, kept in -# sync by hand. Anything more elaborate (driver-version probes, AMD paths, -# lspci fallbacks) belongs in the host CLI, not here. +# sync by hand. NVIDIA and AMD are both supported; elaborate driver/version +# diagnostics and heterogeneous-backend selection stay in the host CLI. set -euo pipefail @@ -227,13 +227,16 @@ _build_host_info_json() { printf '"kernel":%s,' "$(_json_str_or_null "${LUCEBOX_HOST_KERNEL:-}")" printf '"wsl_version":%s,' "$(_json_str_or_null "${LUCEBOX_HOST_WSL_VERSION:-}")" printf '"docker_version":%s,' "$(_json_str_or_null "${LUCEBOX_HOST_DOCKER_VERSION:-}")" + printf '"gpu_vendor":%s,' "$(_json_str_or_null "${LUCEBOX_HOST_GPU_VENDOR:-}")" printf '"nvidia_driver":%s,' "$(_json_str_or_null "${LUCEBOX_HOST_DRIVER_VERSION:-}")" printf '"nvidia_ctk_version":%s,' "$(_json_str_or_null "${LUCEBOX_HOST_NVIDIA_CTK_VERSION:-}")" + printf '"rocm_version":%s,' "$(_json_str_or_null "${LUCEBOX_HOST_ROCM_VERSION:-}")" printf '"cpu_model":%s,' "$(_json_str_or_null "${LUCEBOX_HOST_CPU_MODEL:-}")" printf '"nproc":%s,' "$(_json_int_or_null "${LUCEBOX_HOST_NPROC:-}")" printf '"ram_gb":%s,' "$(_json_int_or_null "${LUCEBOX_HOST_RAM_GB:-}")" printf '"gpus":%s,' "$(_emit_gpu_array)" printf '"cuda_visible_devices":%s,' "$(_json_str_or_null "${LUCEBOX_HOST_CUDA_VISIBLE_DEVICES:-}")" + printf '"hip_visible_devices":%s,' "$(_json_str_or_null "${LUCEBOX_HOST_HIP_VISIBLE_DEVICES:-}")" printf '"source":%s,' "$(_json_str_or_null "$source_tag")" printf '"collector":%s,' "$(_json_str_or_null "$collector_tag")" printf '"collected_at":%s' "$(_json_str_or_null "$collected_at")" @@ -243,17 +246,70 @@ _build_host_info_json() { write_host_info # ── detect ───────────────────────────────────────────────────────────────── -# nvidia-smi is always present here (--gpus all wires the driver in). +# The host wrapper wires either NVIDIA (--gpus all) or AMD (/dev/kfd + +# /dev/dri). Direct docker users get the same fallback detection here. GPU_VRAM_GB=0 +GPU_COUNT=0 if command -v nvidia-smi &>/dev/null; then if mem_mib=$(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits 2>/dev/null \ | head -1) && [ -n "$mem_mib" ]; then GPU_VRAM_GB=$((mem_mib / 1024)) fi -fi -GPU_COUNT=0 -if command -v nvidia-smi &>/dev/null; then GPU_COUNT=$(nvidia-smi -L 2>/dev/null | awk '/^GPU /{n++} END{print n+0}') || GPU_COUNT=0 +elif command -v amd-smi &>/dev/null; then + amd_stats=$(amd-smi static --asic --vram --csv 2>/dev/null | awk -F',' ' + NR == 1 { + for (i = 1; i <= NF; i++) { + key = $i; gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", key); col[key] = i + } + next + } + { + mem = $(col["size"]); arch = $(col["target_graphics_version"]) + gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", mem) + gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", arch) + if (mem ~ /^[0-9]+([.][0-9]+)?$/) { + n++ + if (mem > max) { max = mem; max_arch = arch } + } + } + END { if (n) printf "%d %d %s", max / 1024, n, max_arch } + ' || echo "") + if [ -n "$amd_stats" ]; then + read -r GPU_VRAM_GB GPU_COUNT GPU_ARCH <<<"$amd_stats" + fi +elif command -v rocm-smi &>/dev/null; then + amd_stats=$(rocm-smi --showproductname --showmeminfo vram --csv 2>/dev/null | awk -F',' ' + NR == 1 { + for (i = 1; i <= NF; i++) { + key = $i; gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", key); col[key] = i + } + next + } + { + bytes = $(col["VRAM Total Memory (B)"]); arch = $(col["GFX Version"]) + gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", bytes) + gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", arch) + if (bytes ~ /^[0-9]+$/) { + n++ + if (bytes > max) { max = bytes; max_arch = arch } + } + } + END { if (n) printf "%d %d %s", max / 1073741824, n, max_arch } + ' || echo "") + if [ -n "$amd_stats" ]; then + read -r GPU_VRAM_GB GPU_COUNT GPU_ARCH <<<"$amd_stats" + fi +fi + +# Strix Halo's unified memory is usable for model weights even when SMI only +# reports the small fixed VRAM carve-out. Mirror the host wrapper's effective +# capacity rule for direct `docker run` users. +if [ "${GPU_ARCH:-}" = "gfx1151" ] && [ "$GPU_VRAM_GB" -lt 12 ]; then + host_ram_gb=$(awk '/MemTotal/{printf "%.0f", $2/1024/1024}' /proc/meminfo 2>/dev/null || echo 0) + if [ "$host_ram_gb" -ge 32 ]; then + GPU_VRAM_GB=$host_ram_gb + fi fi # ── fallback autotune (only fills unset env) ─────────────────────────────── From 81d9206f3ea943c64228bd3dbafa2f9ab1ec4086 Mon Sep 17 00:00:00 2001 From: mrciffa <49000955+davide221@users.noreply.github.com> Date: Mon, 27 Jul 2026 13:12:29 +0200 Subject: [PATCH 3/8] fix(lucebox): harden config and target-only launches --- install.sh | 56 ++++++--------- lucebox.sh | 74 +++++++++++++++----- lucebox/src/lucebox/autotune.py | 5 +- lucebox/src/lucebox/cli.py | 19 ++--- lucebox/src/lucebox/config.py | 17 +++-- lucebox/src/lucebox/docker_run.py | 24 +++++-- lucebox/tests/test_config.py | 37 ++++++++++ lucebox/tests/test_config_cli.py | 13 ++++ lucebox/tests/test_docker_run.py | 33 ++++++++- scripts/test_lucebox_sh.sh | 112 +++++++++++++++++++++++++++--- server/scripts/entrypoint.sh | 6 +- 11 files changed, 305 insertions(+), 91 deletions(-) diff --git a/install.sh b/install.sh index c44f92712..c6f959bdc 100755 --- a/install.sh +++ b/install.sh @@ -36,43 +36,13 @@ die() { printf '%s[install] ✗%s %s\n' "$C_ERR" "$C_RST" "$*" >&2; exit 1; } command -v curl >/dev/null 2>&1 || die "curl is required (apt-get install curl)" -# ── fetch ───────────────────────────────────────────────────────────────── -tmp=$(mktemp -t lucebox.XXXXXX) || die "couldn't create temp file" -# shellcheck disable=SC2064 # we want $tmp expanded now, not at trap time -trap "rm -f '$tmp' '$tmp.bak'" EXIT -info "fetching $LUCEBOX_INSTALL_URL" -curl -fsSL "$LUCEBOX_INSTALL_URL" -o "$tmp" \ - || die "download failed from $LUCEBOX_INSTALL_URL" - -# ── sanity check ────────────────────────────────────────────────────────── -# Refuse to install something that isn't recognizably lucebox.sh. Catches -# 404 pages, redirects to HTML, and accidental URL typos. -head -1 "$tmp" | grep -q '^#!/usr/bin/env bash$' \ - || die "downloaded file does not look like a bash script (got: $(head -1 "$tmp"))" -grep -q '^VERSION=' "$tmp" \ - || die "downloaded file is missing VERSION marker — not lucebox.sh?" - # ── decide what gets baked in as the persisted channel ─────────────────── -# `lucebox update` reads LUCEBOX_INSTALLED_FROM from the installed copy and -# re-fetches from it. Persisting a SHA-pinned URL is a footgun — every -# future update would re-install the same frozen SHA forever, defeating -# the point of `update`. So: -# -# 1. If $LUCEBOX_INSTALL_CHANNEL is set, that's the persisted URL -# (caller takes responsibility for picking a real branch URL). -# 2. Else if LUCEBOX_INSTALL_URL has a 40-char hex SHA segment, refuse -# to persist it — tell the user to set LUCEBOX_INSTALL_CHANNEL. -# Common case: someone curl'd from /raw// to bypass a stale CDN -# cache during dev; they meant for updates to track the branch. -# 3. Else persist LUCEBOX_INSTALL_URL as-is (branch or canonical main). +# Do this before fetching so a SHA-pinned URL is refused for the intended +# reason even when that remote object is unavailable. channel_url="${LUCEBOX_INSTALL_CHANNEL:-}" if [ -z "$channel_url" ]; then - # Match a full 40-char hex SHA in the URL path, not the broader - # {7,40} range — a 7-39 char hex segment is more likely a branch - # name shaped like a short SHA (e.g. `feat/abc1234-hotfix`) than an - # actual SHA-pin. Keeping the gate at exactly 40 chars matches what - # `git rev-parse HEAD` emits and what `/raw//` URLs from - # GitHub's CDN actually carry. + # A full 40-char SHA is what `git rev-parse HEAD` and GitHub raw URLs use. + # Shorter hex-like path segments may be branch names, so don't reject them. if [[ "$LUCEBOX_INSTALL_URL" =~ /[0-9a-fA-F]{40}/[^/]+\.sh$ ]]; then die "$(cat < "$tmp.baked" mv "$tmp.baked" "$tmp" -grep -q "^LUCEBOX_INSTALLED_FROM=\"$escaped_url\"$" "$tmp" \ +grep -Fqx "LUCEBOX_INSTALLED_FROM=\"$channel_url\"" "$tmp" \ || die "failed to bake install source into the downloaded script" # ── install ─────────────────────────────────────────────────────────────── diff --git a/lucebox.sh b/lucebox.sh index ebdbb1139..d392064fd 100755 --- a/lucebox.sh +++ b/lucebox.sh @@ -101,7 +101,29 @@ _lucebox_config_get() { /=/ { if (current != want_section) next line = $0 - sub(/#.*$/, "", line) + # Strip a TOML comment only when # is outside a quoted string. + # Paths and image tags may legitimately contain #. + in_quote = 0 + escaped = 0 + for (i = 1; i <= length(line); i++) { + ch = substr(line, i, 1) + if (escaped) { + escaped = 0 + continue + } + if (in_quote && ch == "\\") { + escaped = 1 + continue + } + if (ch == "\"") { + in_quote = !in_quote + continue + } + if (ch == "#" && !in_quote) { + line = substr(line, 1, i - 1) + break + } + } eq = index(line, "=") if (eq == 0) next k = substr(line, 1, eq - 1) @@ -162,6 +184,7 @@ CONTAINER_NAME=$(_lucebox_resolve "${LUCEBOX_CONTAINER:-}" runtime.container_nam DEFAULT_PORT=$(_lucebox_resolve "${LUCEBOX_PORT:-}" runtime.port "8080") DEFAULT_MODELS_DIR=$(_lucebox_resolve "${LUCEBOX_MODELS:-}" paths.models "${XDG_DATA_HOME:-$HOME/.local/share}/lucebox/models") IMAGE_BASE=$(_lucebox_resolve "${LUCEBOX_IMAGE:-}" image.registry "$(_lucebox_derive_image "$LUCEBOX_INSTALLED_FROM")") +CONFIG_HOME="${LUCEBOX_HOME:-$HOME/.lucebox}" # ── LUCEBOX_HOST_* safe defaults (belt-and-suspenders) ──────────────────── # `set -u` makes any unbound LUCEBOX_HOST_* read fatal. Historically this has @@ -239,7 +262,7 @@ _parse_amd_smi_csv() { awk -F',' ' NR == 1 { for (i = 1; i <= NF; i++) { - key = $i + key = tolower($i) gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", key) col[key] = i } @@ -266,7 +289,7 @@ _parse_rocm_smi_csv() { awk -F',' ' NR == 1 { for (i = 1; i <= NF; i++) { - key = $i + key = tolower($i) gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", key) col[key] = i } @@ -274,9 +297,9 @@ _parse_rocm_smi_csv() { } { dev = $(col["device"]) - bytes = $(col["VRAM Total Memory (B)"]) - name = $(col["Card Series"]) - arch = $(col["GFX Version"]) + bytes = $(col["vram total memory (b)"]) + name = $(col["card series"]) + arch = $(col["gfx version"]) gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", dev) gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", bytes) gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", name) @@ -559,9 +582,9 @@ pick_variant() { _variant_is_rocm() { # Variant names are not limited to the moving `rocm` tag. Releases and # CI produce tags such as `0.3.0-rocm` and `pr-335-rocm`; treat any tag - # containing the backend marker as ROCm. Spell out case-insensitivity - # instead of using Bash 4's `${value,,}` so host-only commands such as - # `check` keep working under the older Bash bundled with macOS too. + # containing the backend marker as ROCm. Spell out case-insensitivity so + # host-only commands such as `check` reach no Bash-4-only expansion. Full + # container dispatch still requires Bash 4.3+ for the argv namerefs below. case "$1" in *[Rr][Oo][Cc][Mm]*) return 0 ;; *) return 1 ;; @@ -686,6 +709,7 @@ _append_scalar_env() { _arr+=(-e "LUCEBOX_PORT=$DEFAULT_PORT") _arr+=(-e "LUCEBOX_CONTAINER=$CONTAINER_NAME") _arr+=(-e "LUCEBOX_MODELS=$DEFAULT_MODELS_DIR") + _arr+=(-e "LUCEBOX_HOME=$CONFIG_HOME") [ -n "${HF_TOKEN:-}" ] && _arr+=(-e "HF_TOKEN=$HF_TOKEN") return 0 } @@ -728,9 +752,9 @@ _set_tty_flags() { # usage: _set_tty_flags arrayname } build_orchestrator_argv() { - local variant="$1"; shift - local tty=() - _set_tty_flags tty + local variant="$1" caller_has_tty="$2"; shift 2 + local tty=(-i) + [ "$caller_has_tty" = "1" ] && tty=(-it) local argv=(docker run --rm "${tty[@]}") _append_gpu_args argv "$variant" argv+=(--name "${CONTAINER_NAME}-cli-$$") @@ -743,7 +767,11 @@ build_orchestrator_argv() { # when actually needed; pulling that mount when the host talks to # docker over TCP/SSH is fine. if [ -S "$DOCKER_SOCK_PATH" ]; then - argv+=(--group-add "$(stat -c '%g' "$DOCKER_SOCK_PATH")") + local socket_gid + socket_gid=$(stat -c '%g' "$DOCKER_SOCK_PATH" 2>/dev/null \ + || stat -f '%g' "$DOCKER_SOCK_PATH" 2>/dev/null \ + || echo "") + [ -n "$socket_gid" ] && argv+=(--group-add "$socket_gid") argv+=(-v "$DOCKER_SOCK_PATH:/var/run/docker.sock") fi argv+=(-v "$HOME:$HOME") @@ -753,7 +781,12 @@ build_orchestrator_argv() { # points XDG_DATA_HOME outside $HOME. mkdir -p "$DEFAULT_MODELS_DIR" argv+=(-v "$DEFAULT_MODELS_DIR:$DEFAULT_MODELS_DIR") - argv+=(-w "$PWD") + # A custom LUCEBOX_HOME may sit outside $HOME, so mount it explicitly. + # Use an image-owned working directory: callers may invoke lucebox from + # /tmp or another path that is not bind-mounted into the container. + mkdir -p "$CONFIG_HOME" + argv+=(-v "$CONFIG_HOME:$CONFIG_HOME") + argv+=(-w /opt/lucebox-hub) argv+=(-e "HOME=$HOME") # Host facts — Python side reads these instead of reprobing. _append_host_env argv @@ -836,7 +869,7 @@ cmd_serve() { esac local orch_argv server_argv - mapfile -t orch_argv < <(build_orchestrator_argv "$variant" print-serve-argv) + mapfile -t orch_argv < <(build_orchestrator_argv "$variant" 0 print-serve-argv) if mapfile -t server_argv < <("${orch_argv[@]}" 2>/dev/null) \ && [ "${#server_argv[@]}" -gt 0 ] \ @@ -859,6 +892,7 @@ cmd_serve() { --name "$CONTAINER_NAME" -p "$DEFAULT_PORT:8080" -v "$HOME:$HOME" + -v "$CONFIG_HOME:$CONFIG_HOME" -v "$fallback_models:/opt/lucebox-hub/server/models") _append_gpu_args fallback_argv "$variant" _append_host_env fallback_argv @@ -944,6 +978,7 @@ Environment=LUCEBOX_VARIANT=$variant Environment=LUCEBOX_PORT=$DEFAULT_PORT Environment=LUCEBOX_CONTAINER=$CONTAINER_NAME Environment=LUCEBOX_MODELS=$DEFAULT_MODELS_DIR +Environment=LUCEBOX_HOME=$CONFIG_HOME ExecStartPre=-$docker_bin rm -f $CONTAINER_NAME ExecStart=$SCRIPT_PATH serve ExecStop=$docker_bin stop -t 30 $CONTAINER_NAME @@ -1362,8 +1397,11 @@ cmd_in_container() { local variant variant=$(pick_variant) require_host_prereqs "$variant" - local argv - mapfile -t argv < <(build_orchestrator_argv "$variant" "$@") + local caller_has_tty=0 argv + if [ -t 0 ] && [ -t 1 ]; then + caller_has_tty=1 + fi + mapfile -t argv < <(build_orchestrator_argv "$variant" "$caller_has_tty" "$@") exec "${argv[@]}" } @@ -1404,7 +1442,7 @@ cmd_exec_in_container() { _set_tty_flags tty local argv=(docker exec "${tty[@]}") argv+=(--user "$(id -u):$(id -g)") - argv+=(-w "$PWD") + argv+=(-w /opt/lucebox-hub) argv+=(-e "HOME=$HOME") _append_host_env argv _append_scalar_env argv "$variant" diff --git a/lucebox/src/lucebox/autotune.py b/lucebox/src/lucebox/autotune.py index 5541cf271..0baf0fe91 100644 --- a/lucebox/src/lucebox/autotune.py +++ b/lucebox/src/lucebox/autotune.py @@ -26,8 +26,7 @@ def runtime_from_host(host: HostFacts) -> DflashRuntime: 22-31 — 24 GB-class consumer flagships (3090/4090/5090/5090-Laptop). 98 K with tq3_0 KV (~2 GB KV + ~18 GB model ≈ 20 GB). Confirmed on bragi (RTX 5090 Laptop, 23 GB VRAM) 2026-05-31. - 32-47 — RTX 6000 Ada / A100 40 GB. Full 128 K. - ≥48 — A100 80 GB / H100 / RTX 6000 Pro. Full 128 K. + ≥32 — RTX 6000 Ada / A100 / H100 class. Full 128 K. Prefix cache remains an explicit sweep tunable, but the automatic baseline keeps it off because tool prompts currently exercise a daemon @@ -75,6 +74,4 @@ def runtime_from_host(host: HostFacts) -> DflashRuntime: max_ctx=98304, cache_type_k="tq3_0", cache_type_v="tq3_0", ) - if host.vram_gb < 48: - return DflashRuntime(max_ctx=131072) return DflashRuntime(max_ctx=131072) diff --git a/lucebox/src/lucebox/cli.py b/lucebox/src/lucebox/cli.py index e3c6101bd..9c0e1809c 100644 --- a/lucebox/src/lucebox/cli.py +++ b/lucebox/src/lucebox/cli.py @@ -22,6 +22,7 @@ import typer from rich.console import Console +from rich.markup import escape from rich.table import Table import lucebox.config as config_mod @@ -102,7 +103,7 @@ def pull() -> None: """`docker pull` the image variant from config.toml.""" cfg = _load_or_build() tag = f"{cfg.image}:{cfg.variant}" - console.print(f"[bold]Pulling {tag}[/bold] (~14 GB; takes a while)…") + console.print(f"[bold]Pulling {escape(tag)}[/bold] (~14 GB; takes a while)…") rc = docker_run.docker_pull(tag) if rc != 0: raise typer.Exit(code=rc) @@ -145,10 +146,10 @@ def config_get_cmd( try: entries = config_get(key or None) except KeyError as exc: - console.print(f"[red]{exc}[/red]") + console.print(f"[red]{escape(str(exc))}[/red]") raise typer.Exit(code=2) from exc for k, (value, origin) in entries.items(): - console.print(f"{k} = {value!r} ([dim]from {origin}[/dim])") + console.print(f"{k} = {escape(repr(value))} ([dim]from {origin}[/dim])") @config_app.command("set") @@ -170,9 +171,9 @@ def config_set_cmd( try: config_set(key, value) except (KeyError, ValueError) as exc: - console.print(f"[red]{exc}[/red]") + console.print(f"[red]{escape(str(exc))}[/red]") raise typer.Exit(code=2) from exc - console.print(f"[green]Set[/green] {key} = {value}") + console.print(f"[green]Set[/green] {escape(key)} = {escape(value)}") @config_app.command("unset") @@ -183,12 +184,12 @@ def config_unset_cmd( try: changed = config_unset(key) except KeyError as exc: - console.print(f"[red]{exc}[/red]") + console.print(f"[red]{escape(str(exc))}[/red]") raise typer.Exit(code=2) from exc if changed: - console.print(f"[green]Unset[/green] {key}") + console.print(f"[green]Unset[/green] {escape(key)}") else: - console.print(f"[dim]{key} was not in config.toml; nothing to do[/dim]") + console.print(f"[dim]{escape(key)} was not in config.toml; nothing to do[/dim]") # ── models sub-app ───────────────────────────────────────────────────────── @@ -287,7 +288,7 @@ def models_download( try: pres = download_mod.resolve_preset(preset) except KeyError as exc: - console.print(f"[red]{exc}[/red]") + console.print(f"[red]{escape(str(exc))}[/red]") raise typer.Exit(code=2) from exc current = download_mod.status(cfg, pres) diff --git a/lucebox/src/lucebox/config.py b/lucebox/src/lucebox/config.py index 6e3d113c8..7ca5d2445 100644 --- a/lucebox/src/lucebox/config.py +++ b/lucebox/src/lucebox/config.py @@ -26,7 +26,7 @@ from dataclasses import asdict, replace from datetime import UTC from pathlib import Path -from typing import Any +from typing import Any, Literal, cast import tomli_w @@ -55,11 +55,11 @@ def default_config_path() -> Path: # ── dotted-key registry ──────────────────────────────────────────────────── -def _cast_prefill_mode(v: Any) -> str: +def _cast_prefill_mode(v: Any) -> Literal["off", "auto", "always"]: s = str(v) if s not in {"off", "auto", "always"}: raise ValueError(f"prefill_mode must be off/auto/always, got {s!r}") - return s + return cast(Literal["off", "auto", "always"], s) def _cast_bool(v: Any) -> bool: @@ -236,12 +236,12 @@ def _from_dict(raw: dict[str, Any]) -> Config: dflash = DflashRuntime( budget=int(df.get("budget", 22)), max_ctx=int(df.get("max_ctx", 16384)), - lazy=bool(df.get("lazy", False)), + lazy=_cast_bool(df.get("lazy", False)), prefix_cache_slots=int(df.get("prefix_cache_slots", 0)), prefill_cache_slots=int(df.get("prefill_cache_slots", 0)), cache_type_k=str(df.get("cache_type_k", "")), cache_type_v=str(df.get("cache_type_v", "")), - prefill_mode=df.get("prefill_mode", "off"), + prefill_mode=_cast_prefill_mode(df.get("prefill_mode", "off")), prefill_keep_ratio=float(df.get("prefill_keep_ratio", 0.05)), prefill_threshold=int(df.get("prefill_threshold", 32000)), prefill_drafter=str(df.get("prefill_drafter", "")), @@ -249,7 +249,7 @@ def _from_dict(raw: dict[str, Any]) -> Config: fa_window=int(df.get("fa_window", 0)), think_soft_close_min_ratio=float( df.get("think_soft_close_min_ratio", 0.0)), - debug_thinking_logits=bool(df.get("debug_thinking_logits", False)), + debug_thinking_logits=_cast_bool(df.get("debug_thinking_logits", False)), ) host_raw = raw.get("host", {}) @@ -357,6 +357,11 @@ def seed_dflash_from_host(host: HostFacts, *, path: Path | None = None) -> bool: import lucebox.autotune as autotune_mod path = path or default_config_path() + # Preserve the normal load() migration contract even when this helper is + # called directly. Otherwise creating a fresh config.toml here would make + # an adjacent legacy config.env invisible to every future load. + if not path.exists() and path.with_suffix(".env").exists(): + load(path) doc = load_doc(path) if "dflash" in doc: return False diff --git a/lucebox/src/lucebox/docker_run.py b/lucebox/src/lucebox/docker_run.py index 8e61e355f..f8b688247 100644 --- a/lucebox/src/lucebox/docker_run.py +++ b/lucebox/src/lucebox/docker_run.py @@ -52,8 +52,8 @@ def _resolve_model_files(cfg: Config) -> tuple[str, str, str]: (a directory path) instead of the GGUF-file path, allowing the entrypoint to discover the safetensors file inside it. - Imported lazily to avoid the lucebox.types ↔ lucebox.download circular - import that surfaces when this module is imported from ``__init__``. + The preset registry is imported lazily so constructing a minimal Docker + spec does not import the Hugging Face download surface unnecessarily. """ target = cfg.model.target_file draft = cfg.model.draft_file @@ -75,12 +75,18 @@ def _resolve_model_files(cfg: Config) -> tuple[str, str, str]: def _runtime_volumes(cfg: Config) -> tuple[tuple[str, str], ...]: - """Mount models plus $HOME so absolute symlink targets remain valid.""" - home = str(Path.home()) + """Mount models, $HOME, and any config dir outside the home mount.""" + home_path = Path.home() + home = str(home_path) models = str(cfg.models_dir) volumes = [(models, "/opt/lucebox-hub/server/models")] if home != models: volumes.append((home, home)) + config_home = Path(os.environ.get("LUCEBOX_HOME", home_path / ".lucebox")).absolute() + # The same-path home mount normally covers config. If models_dir == HOME, + # that path is mounted at /opt/... instead, so add config explicitly too. + if home == models or not config_home.is_relative_to(home_path): + volumes.append((str(config_home), str(config_home))) return tuple(volumes) @@ -175,7 +181,11 @@ def server_run_spec(cfg: Config) -> DockerRunSpec: # LUCEBOX_HOST_* first so they ride out front in the rendered argv, # making it obvious in `print-run` output what host facts get forwarded. env: list[tuple[str, str]] = list(_host_facts_env()) + config_home = str( + Path(os.environ.get("LUCEBOX_HOME", Path.home() / ".lucebox")).absolute() + ) env += [ + ("LUCEBOX_HOME", config_home), ("DFLASH_BUDGET", str(cfg.dflash.budget)), ("DFLASH_MAX_CTX", str(cfg.dflash.max_ctx)), ("DFLASH_PREFIX_CACHE_SLOTS", str(cfg.dflash.prefix_cache_slots)), @@ -198,6 +208,12 @@ def server_run_spec(cfg: Config) -> DockerRunSpec: env.append(("DFLASH_DRAFT", f"/opt/lucebox-hub/server/models/draft/{draft_file}")) elif draft_dir: env.append(("DFLASH_DRAFT", f"/opt/lucebox-hub/server/models/draft/{draft_dir}")) + elif cfg.model.preset: + # An active target-only preset is an explicit choice, not permission + # for entrypoint.sh to scan models/draft and attach an unrelated stale + # draft left by a previously active model. The entrypoint treats this + # guaranteed-missing path as "run target-only". + env.append(("DFLASH_DRAFT", "/opt/lucebox-hub/server/models/.lucebox-no-draft")) if cfg.dflash.lazy: env.append(("DFLASH_LAZY", "1")) if cfg.dflash.cache_type_k: diff --git a/lucebox/tests/test_config.py b/lucebox/tests/test_config.py index 9c2b913a3..afedc7d91 100644 --- a/lucebox/tests/test_config.py +++ b/lucebox/tests/test_config.py @@ -35,6 +35,28 @@ def test_image_variant_round_trips_from_toml(tmp_path: Path) -> None: assert cfg.variant == "integration-props-uv-squared-clean-cuda12" +def test_toml_load_uses_strict_bool_and_prefill_casters(tmp_path: Path) -> None: + path = tmp_path / "config.toml" + path.write_text( + '[dflash]\nlazy = "false"\ndebug_thinking_logits = "false"\n' + 'prefill_mode = "auto"\n' + ) + + cfg = config._load_toml(path) + + assert cfg.dflash.lazy is False + assert cfg.dflash.debug_thinking_logits is False + assert cfg.dflash.prefill_mode == "auto" + + +def test_toml_load_rejects_invalid_prefill_mode(tmp_path: Path) -> None: + path = tmp_path / "config.toml" + path.write_text('[dflash]\nprefill_mode = "sometimes"\n') + + with pytest.raises(ValueError, match="prefill_mode"): + config._load_toml(path) + + def test_model_preset_round_trips_through_set_and_load(tmp_path: Path) -> None: """Setting model.preset writes a sparse TOML doc that loads back correctly.""" path = tmp_path / "config.toml" @@ -226,3 +248,18 @@ def test_seed_dflash_is_noop_when_dflash_present(tmp_path: Path) -> None: wrote = config.seed_dflash_from_host(HostFacts(vram_gb=80), path=path) assert wrote is False assert config.load(path).dflash.max_ctx == 4096 + + +def test_seed_dflash_migrates_legacy_config_before_writing(tmp_path: Path) -> None: + from lucebox.types import HostFacts + + path = tmp_path / "config.toml" + path.with_suffix(".env").write_text("DFLASH_PORT=9090\n") + + wrote = config.seed_dflash_from_host(HostFacts(vram_gb=24), path=path) + + assert wrote is True + loaded = config.load(path) + assert loaded is not None + assert loaded.port == 9090 + assert loaded.dflash.max_ctx == 98304 diff --git a/lucebox/tests/test_config_cli.py b/lucebox/tests/test_config_cli.py index 446ab41b6..6525ea319 100644 --- a/lucebox/tests/test_config_cli.py +++ b/lucebox/tests/test_config_cli.py @@ -76,6 +76,19 @@ def test_config_set_creates_file_when_missing( assert "port = 9090" in cfg_path.read_text() +def test_config_markup_characters_do_not_crash_output( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + _set_config_path(tmp_path, monkeypatch) + + set_result = CliRunner().invoke(app, ["config", "set", "image=registry/[dev]"]) + get_result = CliRunner().invoke(app, ["config", "get", "image"]) + + assert set_result.exit_code == 0, set_result.stdout + assert get_result.exit_code == 0, get_result.stdout + assert "registry/[dev]" in get_result.stdout + + def test_load_or_build_env_overrides_persisted_config( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: diff --git a/lucebox/tests/test_docker_run.py b/lucebox/tests/test_docker_run.py index 0f8999ea1..d4652cc5c 100644 --- a/lucebox/tests/test_docker_run.py +++ b/lucebox/tests/test_docker_run.py @@ -110,10 +110,26 @@ def test_runtime_volumes_mounts_models_and_home(tmp_path: Path) -> None: def test_runtime_volumes_dedupes_when_models_is_home(monkeypatch, tmp_path: Path) -> None: monkeypatch.setattr(Path, "home", staticmethod(lambda: tmp_path)) + monkeypatch.delenv("LUCEBOX_HOME", raising=False) cfg = Config(models_dir=tmp_path) vols = docker_run._runtime_volumes(cfg) - # models_dir == home → only the models mount, no duplicate home mount. - assert len(vols) == 1 + # models_dir == home is mounted at /opt/..., not at its host path, so the + # nested config directory still needs an explicit same-path mount. + assert len(vols) == 2 + assert (str(tmp_path / ".lucebox"), str(tmp_path / ".lucebox")) in vols + + +def test_runtime_volumes_mounts_custom_config_home( + monkeypatch, tmp_path: Path +) -> None: + home = tmp_path / "home" + config_home = tmp_path / "config-outside-home" + monkeypatch.setattr(Path, "home", staticmethod(lambda: home)) + monkeypatch.setenv("LUCEBOX_HOME", str(config_home)) + + vols = docker_run._runtime_volumes(Config(models_dir=tmp_path / "models")) + + assert (str(config_home), str(config_home)) in vols # ── _resolve_model_files ───────────────────────────────────────────────────── @@ -190,6 +206,19 @@ def test_server_run_spec_always_emits_core_dflash_env(tmp_path: Path) -> None: assert env["DFLASH_PREFILL_CACHE_SLOTS"] == "0" assert env["DFLASH_THINK_MAX"] == "15488" assert env["DFLASH_PORT"] == "8080" + assert env["LUCEBOX_HOME"] + + +def test_server_run_spec_target_only_preset_disables_stale_draft(tmp_path: Path) -> None: + cfg = Config( + models_dir=tmp_path, + model=ModelMeta(preset="qwen3.6-moe"), + ) + + env = _env(docker_run.server_run_spec(cfg)) + + assert env["DFLASH_TARGET"].endswith("Qwen3.6-35B-A3B-UD-Q4_K_M.gguf") + assert env["DFLASH_DRAFT"].endswith("/.lucebox-no-draft") def test_server_run_spec_optional_env_off_by_default(tmp_path: Path) -> None: diff --git a/scripts/test_lucebox_sh.sh b/scripts/test_lucebox_sh.sh index f858690ba..80125e8f0 100755 --- a/scripts/test_lucebox_sh.sh +++ b/scripts/test_lucebox_sh.sh @@ -85,8 +85,11 @@ report() { assert_runs() { local label="$1" cmd="$2" expect="${3:-}" local out rc - out=$(NO_COLOR=1 bash -c "$cmd" 2>&1) - rc=$? + if out=$(NO_COLOR=1 bash -c "$cmd" 2>&1); then + rc=0 + else + rc=$? + fi if [ "$rc" -ne 0 ]; then report fail "$label" "exit $rc; output: $(printf '%s' "$out" | head -3)" return @@ -337,6 +340,7 @@ STUB "Environment=LUCEBOX_VARIANT=" \ "Environment=LUCEBOX_PORT=" \ "Environment=LUCEBOX_MODELS=" \ + "Environment=LUCEBOX_HOME=" \ "SuccessExitStatus=143" \ ; do grep -qF "$needle" "$unit_path" || missing="$missing $needle" @@ -730,7 +734,7 @@ port = 9090 container_name = "luce-test" [paths] -models = "/srv/models" +models = "/srv/models#fast" # an actual TOML comment [dflash] budget = 22 @@ -745,7 +749,7 @@ TOML "|image.variant|cuda12|cuda13" "|runtime.port|8080|9090" "|runtime.container_name|lucebox|luce-test" - "|paths.models|/var/lib/lucebox|/srv/models" + "|paths.models|/var/lib/lucebox|/srv/models#fast" "OVERRIDE|image.registry|ghcr.io/luce-org/lucebox-hub|OVERRIDE" "|missing.key|fallback-default|fallback-default" ) @@ -822,14 +826,14 @@ test_cmd_start_already_active_shortcircuit "lucebox start has already-active + r # forever. With CHANNEL set, the bake-in uses the channel URL, not the # fetch URL. test_install_sha_pin_refusal_and_channel_override() { - local label="$1" tmp got rc + local label="$1" tmp got out rc tmp=$(mktemp -d -t lucebox-sha.XXXXXX) # Case 1: SHA-pinned URL without CHANNEL → must refuse - LUCEBOX_INSTALL_URL="https://raw.githubusercontent.com/easel/lucebox-hub/abc1234567/lucebox.sh" \ - LUCEBOX_INSTALL_DEST="$tmp/lucebox1" \ - NO_COLOR=1 \ - bash "$INSTALLER" >/dev/null 2>&1 && rc=0 || rc=$? + out=$(LUCEBOX_INSTALL_URL="https://raw.githubusercontent.com/easel/lucebox-hub/0123456789abcdef0123456789abcdef01234567/lucebox.sh" \ + LUCEBOX_INSTALL_DEST="$tmp/lucebox1" \ + NO_COLOR=1 \ + bash "$INSTALLER" 2>&1) && rc=0 || rc=$? if [ "$rc" -eq 0 ]; then rm -rf "$tmp" report fail "$label" "SHA-pinned URL without CHANNEL should have refused (rc=$rc, got success)" @@ -840,6 +844,11 @@ test_install_sha_pin_refusal_and_channel_override() { report fail "$label" "SHA-pinned URL refusal still wrote $tmp/lucebox1" return fi + if ! grep -qF 'is SHA-pinned' <<<"$out"; then + rm -rf "$tmp" + report fail "$label" "did not reach SHA-pin refusal branch: $(head -3 <<<"$out")" + return + fi # Case 2: SHA-pinned URL WITH CHANNEL → installs, bakes CHANNEL LUCEBOX_INSTALL_URL="file://$SCRIPT" \ @@ -952,7 +961,7 @@ _run_wrapper_capture_docker() { HOME="$sandbox" \ XDG_CONFIG_HOME="$sandbox/.config" \ XDG_DATA_HOME="$sandbox/.local/share" \ - LUCEBOX_HOME="$sandbox/.lucebox" \ + LUCEBOX_HOME="${TEST_LUCEBOX_HOME:-$sandbox/.lucebox}" \ PATH="$shim_dir:$PATH" \ LUCEBOX_HOST_HAS_DOCKER=1 \ LUCEBOX_HOST_HAS_CTK=runtime \ @@ -1001,6 +1010,10 @@ test_routes_to_exec_when_running() { report fail "$label" "exec argv missing 'LUCEBOX_IMAGE=' scalar env; got: $(head -3 <<<"$out")" return fi + if ! grep -q -- '-w /opt/lucebox-hub' <<<"$out"; then + report fail "$label" "exec argv uses a caller-dependent working directory: $(head -3 <<<"$out")" + return + fi report ok "$label" } test_routes_to_exec_when_running "config get routes to docker exec when container running" @@ -1023,6 +1036,85 @@ test_routes_to_run_when_not_running() { } test_routes_to_run_when_not_running "config get falls back to docker run when container not running" +test_custom_lucebox_home_is_mounted_and_forwarded() { + local label="$1" sandbox config_home out + sandbox=$(mktemp -d -t lucebox-route.XXXXXX) + config_home=$(mktemp -d -t lucebox-config.XXXXXX) + _make_docker_shim "$sandbox" 0 + out=$(TEST_LUCEBOX_HOME="$config_home" \ + _run_wrapper_capture_docker "$sandbox" config get model.preset || true) + rm -rf "$sandbox" "$config_home" + if ! grep -qF -- "-v $config_home:$config_home" <<<"$out"; then + report fail "$label" "custom config dir was not mounted: $(head -3 <<<"$out")" + return + fi + if ! grep -qF "LUCEBOX_HOME=$config_home" <<<"$out"; then + report fail "$label" "custom config dir was not forwarded: $(head -3 <<<"$out")" + return + fi + if ! grep -q -- '-w /opt/lucebox-hub' <<<"$out"; then + report fail "$label" "orchestrator uses a caller-dependent working directory" + return + fi + report ok "$label" +} +test_custom_lucebox_home_is_mounted_and_forwarded \ + "custom LUCEBOX_HOME is mounted + forwarded to orchestrator" + +test_run_route_preserves_tty() { + local label="$1" sandbox out + sandbox=$(mktemp -d -t lucebox-route-tty.XXXXXX) + _make_docker_shim "$sandbox" 0 + out=$(python3 - "$SCRIPT" "$sandbox" <<'PY' 2>/dev/null +import os +import pty +import sys + +script, sandbox = sys.argv[1:] +env = os.environ.copy() +env.update({ + "HOME": sandbox, + "XDG_CONFIG_HOME": sandbox + "/.config", + "XDG_DATA_HOME": sandbox + "/.local/share", + "LUCEBOX_HOME": sandbox + "/.lucebox", + "PATH": sandbox + "/bin:" + env["PATH"], + "LUCEBOX_HOST_HAS_DOCKER": "1", + "LUCEBOX_HOST_HAS_CTK": "runtime", + "LUCEBOX_HOST_GPU_VENDOR": "nvidia", + "LUCEBOX_HOST_HAS_NVIDIA_GPU": "1", + "LUCEBOX_HOST_DRIVER_MAJOR": "550", + "LUCEBOX_HOST_GPU_NAME": "Fake GPU", + "LUCEBOX_HOST_GPU_COUNT": "1", + "LUCEBOX_HOST_VRAM_GB": "24", + "LUCEBOX_HOST_GPU_SM": "89", + "_LUCEBOX_HOST_PROBED": "1", + "NO_COLOR": "1", +}) +pid, fd = pty.fork() +if pid == 0: + os.execve("/bin/bash", ["bash", script, "no-such-subcommand"], env) +buf = b"" +try: + while True: + chunk = os.read(fd, 4096) + if not chunk: + break + buf += chunk +except OSError: + pass +os.waitpid(pid, 0) +sys.stdout.write(buf.decode(errors="replace")) +PY +) + rm -rf "$sandbox" + if ! grep -qE '^DOCKER_INVOKED run .* -it( |$)' <<<"${out//$'\r'/}"; then + report fail "$label" "PTY route did not preserve docker -it: $(head -3 <<<"$out")" + return + fi + report ok "$label" +} +test_run_route_preserves_tty "docker-run route preserves caller TTY (-it)" + test_no_exec_flag_forces_run() { local label="$1" sandbox out sandbox=$(mktemp -d -t lucebox-route.XXXXXX) diff --git a/server/scripts/entrypoint.sh b/server/scripts/entrypoint.sh index 78f96c60f..95373aa1b 100755 --- a/server/scripts/entrypoint.sh +++ b/server/scripts/entrypoint.sh @@ -260,7 +260,7 @@ elif command -v amd-smi &>/dev/null; then amd_stats=$(amd-smi static --asic --vram --csv 2>/dev/null | awk -F',' ' NR == 1 { for (i = 1; i <= NF; i++) { - key = $i; gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", key); col[key] = i + key = tolower($i); gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", key); col[key] = i } next } @@ -282,12 +282,12 @@ elif command -v rocm-smi &>/dev/null; then amd_stats=$(rocm-smi --showproductname --showmeminfo vram --csv 2>/dev/null | awk -F',' ' NR == 1 { for (i = 1; i <= NF; i++) { - key = $i; gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", key); col[key] = i + key = tolower($i); gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", key); col[key] = i } next } { - bytes = $(col["VRAM Total Memory (B)"]); arch = $(col["GFX Version"]) + bytes = $(col["vram total memory (b)"]); arch = $(col["gfx version"]) gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", bytes) gsub(/^[[:space:]]+|[[:space:]\r]+$/, "", arch) if (bytes ~ /^[0-9]+$/) { From a65fabe000cdcab85d31f084e58367fb081120d1 Mon Sep 17 00:00:00 2001 From: mrciffa <49000955+davide221@users.noreply.github.com> Date: Mon, 27 Jul 2026 13:25:17 +0200 Subject: [PATCH 4/8] fix(lucebox): align fallback config paths --- lucebox.sh | 4 +++- lucebox/src/lucebox/docker_run.py | 6 ++++-- lucebox/tests/test_docker_run.py | 12 ++++++++++++ scripts/test_lucebox_sh.sh | 31 ++++++++++++++++++++++++++++--- 4 files changed, 47 insertions(+), 6 deletions(-) diff --git a/lucebox.sh b/lucebox.sh index d392064fd..e03445131 100755 --- a/lucebox.sh +++ b/lucebox.sh @@ -893,9 +893,11 @@ cmd_serve() { -p "$DEFAULT_PORT:8080" -v "$HOME:$HOME" -v "$CONFIG_HOME:$CONFIG_HOME" - -v "$fallback_models:/opt/lucebox-hub/server/models") + -v "$fallback_models:/opt/lucebox-hub/server/models" + -e "HOME=$HOME") _append_gpu_args fallback_argv "$variant" _append_host_env fallback_argv + _append_scalar_env fallback_argv "$variant" fallback_argv+=("${IMAGE_BASE}:${variant}") _serve_and_track "${fallback_argv[@]}" } diff --git a/lucebox/src/lucebox/docker_run.py b/lucebox/src/lucebox/docker_run.py index f8b688247..ab34c837a 100644 --- a/lucebox/src/lucebox/docker_run.py +++ b/lucebox/src/lucebox/docker_run.py @@ -82,7 +82,9 @@ def _runtime_volumes(cfg: Config) -> tuple[tuple[str, str], ...]: volumes = [(models, "/opt/lucebox-hub/server/models")] if home != models: volumes.append((home, home)) - config_home = Path(os.environ.get("LUCEBOX_HOME", home_path / ".lucebox")).absolute() + config_home = Path( + os.environ.get("LUCEBOX_HOME") or home_path / ".lucebox" + ).absolute() # The same-path home mount normally covers config. If models_dir == HOME, # that path is mounted at /opt/... instead, so add config explicitly too. if home == models or not config_home.is_relative_to(home_path): @@ -182,7 +184,7 @@ def server_run_spec(cfg: Config) -> DockerRunSpec: # making it obvious in `print-run` output what host facts get forwarded. env: list[tuple[str, str]] = list(_host_facts_env()) config_home = str( - Path(os.environ.get("LUCEBOX_HOME", Path.home() / ".lucebox")).absolute() + Path(os.environ.get("LUCEBOX_HOME") or Path.home() / ".lucebox").absolute() ) env += [ ("LUCEBOX_HOME", config_home), diff --git a/lucebox/tests/test_docker_run.py b/lucebox/tests/test_docker_run.py index d4652cc5c..f5a18ed59 100644 --- a/lucebox/tests/test_docker_run.py +++ b/lucebox/tests/test_docker_run.py @@ -132,6 +132,18 @@ def test_runtime_volumes_mounts_custom_config_home( assert (str(config_home), str(config_home)) in vols +def test_empty_lucebox_home_uses_default(monkeypatch, tmp_path: Path) -> None: + home = tmp_path / "home" + monkeypatch.setattr(Path, "home", staticmethod(lambda: home)) + monkeypatch.setenv("LUCEBOX_HOME", "") + + cfg = Config(models_dir=tmp_path / "models") + spec = docker_run.server_run_spec(cfg) + + assert _env(spec)["LUCEBOX_HOME"] == str(home / ".lucebox") + assert str(Path.cwd()) != _env(spec)["LUCEBOX_HOME"] + + # ── _resolve_model_files ───────────────────────────────────────────────────── diff --git a/scripts/test_lucebox_sh.sh b/scripts/test_lucebox_sh.sh index 80125e8f0..c9fc0577d 100755 --- a/scripts/test_lucebox_sh.sh +++ b/scripts/test_lucebox_sh.sh @@ -1061,11 +1061,29 @@ test_custom_lucebox_home_is_mounted_and_forwarded() { test_custom_lucebox_home_is_mounted_and_forwarded \ "custom LUCEBOX_HOME is mounted + forwarded to orchestrator" +test_serve_fallback_forwards_config_env() { + local label="$1" sandbox config_home out + sandbox=$(mktemp -d -t lucebox-serve-fallback.XXXXXX) + config_home=$(mktemp -d -t lucebox-config.XXXXXX) + _make_docker_shim "$sandbox" 0 + out=$(TEST_LUCEBOX_HOME="$config_home" \ + _run_wrapper_capture_docker "$sandbox" serve || true) + rm -rf "$sandbox" "$config_home" + if ! grep -qF "LUCEBOX_HOME=$config_home" <<<"$out" \ + || ! grep -qF "HOME=$sandbox" <<<"$out"; then + report fail "$label" "fallback server omitted HOME/config env: $(tail -3 <<<"$out")" + return + fi + report ok "$label" +} +test_serve_fallback_forwards_config_env \ + "serve fallback forwards HOME + LUCEBOX_HOME" + test_run_route_preserves_tty() { - local label="$1" sandbox out + local label="$1" sandbox out rc sandbox=$(mktemp -d -t lucebox-route-tty.XXXXXX) _make_docker_shim "$sandbox" 0 - out=$(python3 - "$SCRIPT" "$sandbox" <<'PY' 2>/dev/null + if out=$(timeout 15 python3 - "$SCRIPT" "$sandbox" <<'PY' 2>/dev/null import os import pty import sys @@ -1105,7 +1123,14 @@ except OSError: os.waitpid(pid, 0) sys.stdout.write(buf.decode(errors="replace")) PY -) + ); then + : + else + rc=$? + rm -rf "$sandbox" + report fail "$label" "PTY route timed out or failed (rc=$rc)" + return + fi rm -rf "$sandbox" if ! grep -qE '^DOCKER_INVOKED run .* -it( |$)' <<<"${out//$'\r'/}"; then report fail "$label" "PTY route did not preserve docker -it: $(head -3 <<<"$out")" From 5a2e48b2507c684d7591b056edd263b6749ca38f Mon Sep 17 00:00:00 2001 From: mrciffa <49000955+davide221@users.noreply.github.com> Date: Mon, 27 Jul 2026 15:35:13 +0200 Subject: [PATCH 5/8] fix(lucebox): harden the CLI package --- Dockerfile | 8 +- Dockerfile.rocm | 8 +- install.sh | 24 ++++ lucebox.sh | 155 ++++++++++++++++------- lucebox/README.md | 35 ++--- lucebox/pyproject.toml | 11 +- lucebox/src/lucebox/__main__.py | 6 +- lucebox/src/lucebox/cli.py | 99 ++++++++------- lucebox/src/lucebox/config.py | 117 +++++++++++++---- lucebox/src/lucebox/docker_run.py | 148 ++++++++++++++++++---- lucebox/src/lucebox/download.py | 44 +++++-- lucebox/src/lucebox/types.py | 33 ++++- lucebox/tests/test_cli.py | 10 +- lucebox/tests/test_config.py | 67 ++++++++++ lucebox/tests/test_config_cli.py | 13 ++ lucebox/tests/test_docker_run.py | 119 +++++++++++++++-- lucebox/tests/test_download.py | 18 +++ lucebox/tests/test_models_cli.py | 2 +- scripts/check_lucebox_wrapper_sandbox.sh | 13 +- scripts/test_lucebox_sh.sh | 148 ++++++++++++++++++++-- 20 files changed, 872 insertions(+), 206 deletions(-) diff --git a/Dockerfile b/Dockerfile index 26df32c78..0411927e7 100644 --- a/Dockerfile +++ b/Dockerfile @@ -125,7 +125,8 @@ COPY server/pyproject.toml server/README.md /src/server/ COPY server/scripts /src/server/scripts COPY optimizations/pflash /src/optimizations/pflash COPY optimizations/megakernel /src/optimizations/megakernel -COPY lucebox /src/lucebox +COPY lucebox/pyproject.toml lucebox/README.md /src/lucebox/ +COPY lucebox/src /src/lucebox/src # ─── Stage 2: runtime ─────────────────────────────────────────────────────── # Runtime image: ships nvidia driver libs but no nvcc / dev headers. Matches @@ -205,7 +206,12 @@ COPY --from=builder /src/server/build /opt/lucebox-hub/server/build # server/share/model_cards. The canonical copy also lives at # /opt/lucebox-hub/share/model_cards for any host-side tooling. COPY share/model_cards /opt/lucebox-hub/share/model_cards +# Dedicated targets for narrow, read-only binds when a selected model is a +# symlink to storage outside the main models directory. RUN mkdir -p /opt/lucebox-hub/server/share \ + /opt/lucebox-resolved/target \ + /opt/lucebox-resolved/draft \ + /opt/lucebox-resolved/draft-dir \ && ln -s /opt/lucebox-hub/share/model_cards \ /opt/lucebox-hub/server/share/model_cards diff --git a/Dockerfile.rocm b/Dockerfile.rocm index ca9fb1b65..79931c41d 100644 --- a/Dockerfile.rocm +++ b/Dockerfile.rocm @@ -122,7 +122,8 @@ COPY server/pyproject.toml server/README.md /src/server/ COPY server/scripts /src/server/scripts COPY optimizations/pflash /src/optimizations/pflash COPY optimizations/megakernel /src/optimizations/megakernel -COPY lucebox /src/lucebox +COPY lucebox/pyproject.toml lucebox/README.md /src/lucebox/ +COPY lucebox/src /src/lucebox/src # ─── Stage 2: runtime ─────────────────────────────────────────────────────── # Runtime reuses the ROCm base so the HIP runtime libs (libamdhip64, @@ -181,7 +182,12 @@ COPY --from=builder /src/server/pyproject.toml /src/server/README.md \ COPY --from=builder /src/server/build /opt/lucebox-hub/server/build COPY share/model_cards /opt/lucebox-hub/share/model_cards +# Dedicated targets for narrow, read-only binds when a selected model is a +# symlink to storage outside the main models directory. RUN mkdir -p /opt/lucebox-hub/server/share \ + /opt/lucebox-resolved/target \ + /opt/lucebox-resolved/draft \ + /opt/lucebox-resolved/draft-dir \ && ln -s /opt/lucebox-hub/share/model_cards \ /opt/lucebox-hub/server/share/model_cards diff --git a/install.sh b/install.sh index c6f959bdc..d67dbd9b4 100755 --- a/install.sh +++ b/install.sh @@ -36,6 +36,16 @@ die() { printf '%s[install] ✗%s %s\n' "$C_ERR" "$C_RST" "$*" >&2; exit 1; } command -v curl >/dev/null 2>&1 || die "curl is required (apt-get install curl)" +sha256_file() { + if command -v sha256sum >/dev/null 2>&1; then + sha256sum "$1" | awk '{print $1}' + elif command -v shasum >/dev/null 2>&1; then + shasum -a 256 "$1" | awk '{print $1}' + else + die "checksum requested, but neither sha256sum nor shasum is installed" + fi +} + # ── decide what gets baked in as the persisted channel ─────────────────── # Do this before fetching so a SHA-pinned URL is refused for the intended # reason even when that remote object is unavailable. @@ -68,6 +78,20 @@ info "fetching $LUCEBOX_INSTALL_URL" curl --connect-timeout 10 --max-time 120 -fsSL "$LUCEBOX_INSTALL_URL" -o "$tmp" \ || die "download failed from $LUCEBOX_INSTALL_URL" +# Release automation can pin the exact wrapper payload. This is optional for +# branch-channel installs, where the URL intentionally moves over time. +expected_sha="${LUCEBOX_WRAPPER_SHA256:-}" +if [ -n "$expected_sha" ]; then + [[ "$expected_sha" =~ ^[0-9a-fA-F]{64}$ ]] \ + || die "LUCEBOX_WRAPPER_SHA256 must be exactly 64 hexadecimal characters" + actual_sha=$(sha256_file "$tmp") + actual_sha=$(printf '%s' "$actual_sha" | tr '[:upper:]' '[:lower:]') + expected_sha=$(printf '%s' "$expected_sha" | tr '[:upper:]' '[:lower:]') + [ "$actual_sha" = "$expected_sha" ] \ + || die "wrapper checksum mismatch (expected $expected_sha, got $actual_sha)" + ok "wrapper sha256 verified" +fi + # ── sanity check ────────────────────────────────────────────────────────── # Refuse to install something that isn't recognizably lucebox.sh. Catches # 404 pages, redirects to HTML, and accidental URL typos. diff --git a/lucebox.sh b/lucebox.sh index e03445131..273f9afaf 100755 --- a/lucebox.sh +++ b/lucebox.sh @@ -250,6 +250,19 @@ err() { printf '%b[ERROR]%b %s\n' "$C_ERR" "$C_RST" "$*" >&2; } hint() { printf ' %b%s%b\n' "$C_DIM" "$*" "$C_RST"; } die() { err "$*"; exit 1; } +sha256_file() { + local sum + if command -v sha256sum >/dev/null 2>&1; then + sum=$(sha256sum "$1") + elif command -v shasum >/dev/null 2>&1; then + sum=$(shasum -a 256 "$1") + else + die "checksum requested, but neither sha256sum nor shasum is installed" + fi + sum="${sum%% *}" + printf '%s' "$sum" | tr '[:upper:]' '[:lower:]' +} + # ── host probing ────────────────────────────────────────────────────────── # Sets the LUCEBOX_HOST_* variables consumed by the in-container Python CLI # (passed through with -e). The Python side trusts these and doesn't reprobe @@ -676,9 +689,9 @@ require_systemd() { # ── docker run construction ─────────────────────────────────────────────── # All the Python-CLI subcommands share the same docker run incantation: # mount the host docker socket (so the in-container CLI can spawn server / -# bench containers on the host daemon), mount $HOME at the same path (so -# paths look identical in and out), and pass host facts via env. The selected -# image gets its native accelerator contract: --gpus all for +# bench containers on the host daemon), mount only Lucebox's config/models +# state, and pass host facts via env. The selected image gets its native +# accelerator contract: --gpus all for # CUDA; /dev/kfd + /dev/dri and the render/video groups for ROCm. DOCKER_SOCK_PATH="${DOCKER_HOST:-/var/run/docker.sock}" @@ -774,20 +787,21 @@ build_orchestrator_argv() { [ -n "$socket_gid" ] && argv+=(--group-add "$socket_gid") argv+=(-v "$DOCKER_SOCK_PATH:/var/run/docker.sock") fi - argv+=(-v "$HOME:$HOME") - # Bind-mount the XDG models dir explicitly (host = container path) so - # paths line up in/out. The $HOME mount above already covers it when - # XDG_DATA_HOME is unset, but an explicit -v is required when the user - # points XDG_DATA_HOME outside $HOME. + # Bind only Lucebox-owned state. The orchestrator used to receive all of + # $HOME read-write to support model symlinks; the Python launch builder + # now handles selected symlinks with narrow read-only mounts instead. + # Keeping credentials and unrelated user files outside the container is + # the safer default for both buyer appliances and contributor machines. mkdir -p "$DEFAULT_MODELS_DIR" argv+=(-v "$DEFAULT_MODELS_DIR:$DEFAULT_MODELS_DIR") - # A custom LUCEBOX_HOME may sit outside $HOME, so mount it explicitly. + # A custom LUCEBOX_HOME may sit anywhere, so mount it explicitly and use + # it as the ephemeral CLI's HOME (its caches then remain app-scoped too). # Use an image-owned working directory: callers may invoke lucebox from # /tmp or another path that is not bind-mounted into the container. mkdir -p "$CONFIG_HOME" argv+=(-v "$CONFIG_HOME:$CONFIG_HOME") argv+=(-w /opt/lucebox-hub) - argv+=(-e "HOME=$HOME") + argv+=(-e "HOME=$CONFIG_HOME") # Host facts — Python side reads these instead of reprobing. _append_host_env argv # User overrides for image/port/container/models scalars + HF_TOKEN. @@ -868,17 +882,38 @@ cmd_serve() { ;; esac - local orch_argv server_argv + local orch_argv server_argv server_output orch_error_file orch_rc=0 mapfile -t orch_argv < <(build_orchestrator_argv "$variant" 0 print-serve-argv) - if mapfile -t server_argv < <("${orch_argv[@]}" 2>/dev/null) \ - && [ "${#server_argv[@]}" -gt 0 ] \ - && [ "${server_argv[0]}" = "docker" ]; then - info "Starting lucebox server (variant=$variant, from config.toml)" - _serve_and_track "${server_argv[@]}" - return $? + orch_error_file=$(mktemp -t lucebox-orchestrator.XXXXXX) \ + || die "couldn't create temporary orchestrator log" + if server_output=$("${orch_argv[@]}" 2>"$orch_error_file"); then + if [ -n "$server_output" ]; then + mapfile -t server_argv <<<"$server_output" + if [ "${#server_argv[@]}" -gt 0 ] \ + && [ "${server_argv[0]}" = "docker" ]; then + rm -f "$orch_error_file" + info "Starting lucebox server (variant=$variant, from config.toml)" + _serve_and_track "${server_argv[@]}" + return $? + fi + fi + else + orch_rc=$? fi + # Configuration/path errors must never turn into a successful launch with + # defaults: that could start the wrong model or ignore explicit tuning. + # Docker infrastructure failures (for example an image not pulled yet) + # retain the conservative fallback below. + if [ "$orch_rc" -eq 2 ] \ + || grep -qE 'Invalid configuration:|Cannot build server command:' "$orch_error_file"; then + [ ! -s "$orch_error_file" ] || cat "$orch_error_file" >&2 + rm -f "$orch_error_file" + die "refusing to ignore invalid Lucebox configuration" + fi + rm -f "$orch_error_file" + warn "Couldn't fetch server argv from container (image not pulled?) — using fallback" info "Starting lucebox server (variant=$variant, port=$DEFAULT_PORT, defaults only)" local fallback_models="$DEFAULT_MODELS_DIR" @@ -891,10 +926,9 @@ cmd_serve() { local fallback_argv=(docker run --rm --name "$CONTAINER_NAME" -p "$DEFAULT_PORT:8080" - -v "$HOME:$HOME" -v "$CONFIG_HOME:$CONFIG_HOME" -v "$fallback_models:/opt/lucebox-hub/server/models" - -e "HOME=$HOME") + -e "HOME=$CONFIG_HOME") _append_gpu_args fallback_argv "$variant" _append_host_env fallback_argv _append_scalar_env fallback_argv "$variant" @@ -1129,38 +1163,68 @@ cmd_pull() { } cmd_update() { - # Re-run the bootstrap installer against the channel we were installed - # from. The installer is the source of truth for "how do you install - # lucebox correctly" — chmod, atomic mv, validation, baking the source - # URL back into the new copy so the channel is preserved across - # upgrades. Keeping the logic in install.sh means it can evolve - # independently (sha verify, signature check, etc.) and the installed - # `lucebox update` picks those changes up on the next run. - # - # The installer URL is derived from LUCEBOX_INSTALLED_FROM by swapping - # `lucebox.sh` → `install.sh` in the same directory, so forks don't - # need a separate registration. Override the source channel via - # $LUCEBOX_INSTALL_URL (e.g. to switch from canonical to a dev fork). - local source_url installer_url target + # Download the wrapper itself as data; never execute a second remote + # installer. Override the persisted channel with LUCEBOX_INSTALL_URL. + local source_url target wrapper_tmp expected_sha actual_sha escaped_url source_url="${LUCEBOX_INSTALL_URL:-$LUCEBOX_INSTALLED_FROM}" if [[ "$source_url" != */lucebox.sh ]]; then die "LUCEBOX_INSTALLED_FROM doesn't end in /lucebox.sh: $source_url" fi - installer_url="${source_url%/lucebox.sh}/install.sh" + case "$source_url" in + *['"$`\']*|*$'\n'*|*$'\r'*) + die "update URL contains unsafe characters: $source_url" ;; + esac target=$(realpath "$SCRIPT_PATH") - info "Updating lucebox via $installer_url" + info "Updating lucebox" info " source: $source_url" info " target: $target" - # Pass the URLs through to install.sh via env. The installer reads - # $LUCEBOX_INSTALL_URL (which we set to source_url) and - # $LUCEBOX_INSTALL_DEST (the realpath of *this* file, so a symlinked - # install replaces the actual file behind the link). - LUCEBOX_INSTALL_URL="$source_url" \ - LUCEBOX_INSTALL_DEST="$target" \ - bash -c "$(curl -fsSL "$installer_url")" \ - || die "update failed (installer exited non-zero)" + # Create the temporary file next to the destination so the final rename + # is atomic even when /tmp and the install directory are different mounts. + wrapper_tmp=$(mktemp "${target}.update.XXXXXX") \ + || die "couldn't create temporary update file next to $target" + trap 'if [ -n "${wrapper_tmp:-}" ]; then rm -f "$wrapper_tmp" "$wrapper_tmp.baked"; fi' EXIT + curl --connect-timeout 10 --max-time 120 --retry 2 --retry-delay 1 \ + -fsSL "$source_url" -o "$wrapper_tmp" \ + || die "failed to download wrapper from $source_url" + + [ "$(head -1 "$wrapper_tmp")" = '#!/usr/bin/env bash' ] \ + || die "downloaded wrapper has an unexpected shebang" + grep -Fqx 'set -euo pipefail' "$wrapper_tmp" \ + || die "downloaded wrapper is missing strict shell mode" + grep -q '^VERSION=' "$wrapper_tmp" \ + || die "downloaded file is missing the Lucebox version marker" + grep -q '^LUCEBOX_INSTALLED_FROM=' "$wrapper_tmp" \ + || die "downloaded file is missing the Lucebox update-channel marker" + bash -n "$wrapper_tmp" \ + || die "downloaded wrapper does not parse as valid Bash" + + expected_sha="${LUCEBOX_WRAPPER_SHA256:-}" + if [ -n "$expected_sha" ]; then + [[ "$expected_sha" =~ ^[0-9a-fA-F]{64}$ ]] \ + || die "LUCEBOX_WRAPPER_SHA256 must be exactly 64 hexadecimal characters" + actual_sha=$(sha256_file "$wrapper_tmp") + expected_sha=$(printf '%s' "$expected_sha" | tr '[:upper:]' '[:lower:]') + [ "$actual_sha" = "$expected_sha" ] \ + || die "wrapper checksum mismatch (expected $expected_sha, got $actual_sha)" + ok "wrapper sha256 verified" + fi + + # Preserve the chosen branch/fork in the new copy. The URL was validated + # above for safe embedding in a Bash double-quoted string. + escaped_url=$(printf '%s' "$source_url" | sed 's/[&|]/\\&/g') + sed "s|^LUCEBOX_INSTALLED_FROM=.*|LUCEBOX_INSTALLED_FROM=\"$escaped_url\"|" \ + "$wrapper_tmp" > "$wrapper_tmp.baked" + mv "$wrapper_tmp.baked" "$wrapper_tmp" + grep -Fqx "LUCEBOX_INSTALLED_FROM=\"$source_url\"" "$wrapper_tmp" \ + || die "failed to preserve update channel in downloaded wrapper" + bash -n "$wrapper_tmp" || die "updated wrapper failed validation after channel rewrite" + + chmod +x "$wrapper_tmp" + mv "$wrapper_tmp" "$target" + trap - EXIT + ok "updated lucebox → $target" } cmd_completion() { @@ -1445,7 +1509,7 @@ cmd_exec_in_container() { local argv=(docker exec "${tty[@]}") argv+=(--user "$(id -u):$(id -g)") argv+=(-w /opt/lucebox-hub) - argv+=(-e "HOME=$HOME") + argv+=(-e "HOME=$CONFIG_HOME") _append_host_env argv _append_scalar_env argv "$variant" # The image has no top-level `lucebox` binary on PATH — that name only @@ -1536,7 +1600,10 @@ Environment overrides: LUCEBOX_VARIANT image tag override (default: cuda12 on NVIDIA, rocm on AMD) LUCEBOX_PORT host port for the server (default: 8080) LUCEBOX_CONTAINER server container name (default: lucebox) - LUCEBOX_MODELS host model directory (default: \$XDG_DATA_HOME/lucebox/models + LUCEBOX_MODELS host model directory (default: \$XDG_DATA_HOME/lucebox/models) + LUCEBOX_HOME config/state directory (default: \$HOME/.lucebox) + LUCEBOX_WRAPPER_SHA256 + optional 64-hex checksum pin for install/update LUCEBOX_NO_EXEC=1 force docker-run for in-container subcommands even when the container is up (equivalent to --no-exec) HF_TOKEN propagated to \`models download\` for gated HF repos diff --git a/lucebox/README.md b/lucebox/README.md index 6af3d2a02..f5bb7d635 100644 --- a/lucebox/README.md +++ b/lucebox/README.md @@ -1,21 +1,22 @@ -# lucebox — host CLI for the lucebox-hub container +# lucebox — CLI for the Lucebox inference appliance -This package ships *inside* the `ghcr.io/luce-org/lucebox-hub` Docker image -and is invoked from the host via the [`lucebox.sh`](../lucebox.sh) wrapper: +This Python package ships inside the `ghcr.io/luce-org/lucebox-hub` image. Most +users do not install it directly: they install the small +[`lucebox` host wrapper](https://github.com/Luce-Org/lucebox/blob/main/lucebox.sh), +which invokes the package in the appropriate container: - lucebox.sh check # `docker run … lucebox check` - lucebox.sh config get - lucebox.sh print-run + lucebox check + lucebox models list + lucebox models download qwen3.6-27b --activate + lucebox start -The wrapper is the only thing that runs on the host. It selects CUDA for a -working NVIDIA GPU (including RTX 3090 + Strix builds) and ROCm for AMD builds -(including R9700 + Strix), then applies the matching Docker device flags. -Everything else (host -checks, TOML config, docker daemon calls, model download) is Python in the -container. Host facts (driver/runtime, GPU, RAM, VRAM, systemd availability) are -passed in via `LUCEBOX_HOST_*` environment variables so the Python side -doesn't reprobe. The autotune sweep, profiling, and agent-client launchers -land in follow-up PRs. +The wrapper detects the host and selects CUDA for NVIDIA builds (including RTX +3090 + Strix) or ROCm for AMD builds (including R9700 + Strix). The package then +handles readiness checks, TOML configuration, model selection and download, +optimization settings, and construction of the final server command. Host +facts are passed through `LUCEBOX_HOST_*` environment variables so the +container never has to guess the host configuration. -Subcommands are defined in [`lucebox/cli.py`](src/lucebox/cli.py). See the -top-level [README.md](../README.md) for the user-facing flow. +See the [project README](https://github.com/Luce-Org/lucebox#readme) for the +installation and user flow. Contributors can find the CLI implementation in +[`src/lucebox`](https://github.com/Luce-Org/lucebox/tree/main/lucebox/src/lucebox). diff --git a/lucebox/pyproject.toml b/lucebox/pyproject.toml index 5277b268d..bb12d90d7 100644 --- a/lucebox/pyproject.toml +++ b/lucebox/pyproject.toml @@ -7,7 +7,7 @@ name = "lucebox" dynamic = ["version"] description = "Host-side CLI for the lucebox-hub container: launch, config, model download" readme = "README.md" -requires-python = ">=3.11" +requires-python = ">=3.12,<3.13" authors = [{ name = "Lucebox" }] license = { text = "Apache-2.0" } @@ -30,8 +30,13 @@ dependencies = [ # `uv pip install luce-bench` on the host running the scorer. ] +[project.urls] +Homepage = "https://github.com/Luce-Org/lucebox" +Repository = "https://github.com/Luce-Org/lucebox" +Issues = "https://github.com/Luce-Org/lucebox/issues" + [project.scripts] -lucebox = "lucebox.cli:app" +lucebox = "lucebox.cli:main" [build-system] requires = ["hatchling", "hatch-vcs"] @@ -42,7 +47,7 @@ source = "vcs" # Untagged checkouts (e.g. fresh clone before tagging lucebox-v0.2.1) # resolve to this rather than 0.0.0.dev0. fallback-version = "0.2.1.dev0" -raw-options.tag_regex = '''^lucebox-v(?P\d+\.\d+\.\d+)$''' +tag-pattern = '''^lucebox-v(?P\d+\.\d+\.\d+)$''' [tool.hatch.build.hooks.vcs] # Build hook writes the resolved version into src/lucebox/_version.py diff --git a/lucebox/src/lucebox/__main__.py b/lucebox/src/lucebox/__main__.py index 128e2ca87..5c9d49ba1 100644 --- a/lucebox/src/lucebox/__main__.py +++ b/lucebox/src/lucebox/__main__.py @@ -1,6 +1,6 @@ -"""Entry point for `python -m lucebox`.""" +"""Entry point for ``python -m lucebox``.""" -from lucebox.cli import app +from lucebox.cli import main if __name__ == "__main__": - app() + main() diff --git a/lucebox/src/lucebox/cli.py b/lucebox/src/lucebox/cli.py index 9c0e1809c..622e6de5a 100644 --- a/lucebox/src/lucebox/cli.py +++ b/lucebox/src/lucebox/cli.py @@ -14,10 +14,8 @@ from __future__ import annotations -import os import sys from dataclasses import replace -from pathlib import Path from typing import Annotated import typer @@ -32,57 +30,64 @@ from lucebox import __version__ from lucebox.config import config_get, config_set, config_unset, live_config from lucebox.host_facts import from_env +from lucebox.types import Config app = typer.Typer( name="lucebox", help="Host CLI for the lucebox-hub container. Invoked by lucebox.sh.", no_args_is_help=True, + invoke_without_command=True, add_completion=False, ) console = Console() +error_console = Console(stderr=True) + + +@app.callback() +def root_options( + version_flag: Annotated[ + bool, + typer.Option("--version", help="Print lucebox version and exit.", is_eager=True), + ] = False, +) -> None: + """Apply options shared by the top-level CLI.""" + if version_flag: + print(__version__) + raise typer.Exit() # ── helpers ──────────────────────────────────────────────────────────────── -def _load_or_build() -> config_mod.Config: # type: ignore[name-defined] +def _load_or_build() -> Config: """env > config.toml > dataclass defaults — the canonical precedence. - Without the env-overlay step below, `config_mod.load()` returned the - persisted config verbatim and `LUCEBOX_IMAGE` / `LUCEBOX_VARIANT` / - `LUCEBOX_PORT` / `LUCEBOX_CONTAINER` / `LUCEBOX_MODELS` from the - systemd unit's `Environment=` (or any one-shot shell export) were - silently dropped. That contradicted the precedence lucebox.sh - documents and applies — and bit sindri when its config.toml had - `[image]` without `registry`, so the dataclass default - `ghcr.io/luce-org/lucebox-hub` won over the unit's - `LUCEBOX_IMAGE=ghcr.io/easel/lucebox-hub`. - - Fix: overlay env on top of the loaded config (or the live_config - fallback when config.toml is absent). Only the five top-level - scalars have env hooks — dflash/host/model don't, by design. + Only the five documented top-level scalars have environment overrides; + dflash, host, and model settings intentionally remain config-driven. """ - cfg = config_mod.load() - if cfg is None: - cfg = live_config() - # Overlay live host facts. When ``config.toml`` exists without a - # ``[host]`` block (the common case — operators don't hand-edit - # host facts), ``cfg.host`` defaults to a zero-filled ``HostFacts`` - # and the DFLASH_* serve heuristic silently falls through to the - # "no VRAM signal" path. Re-probe from env so the wrapper-exported - # LUCEBOX_HOST_* facts always win over the persisted (possibly - # absent) snapshot. - live_host = from_env() - host = live_host if live_host.vram_gb > 0 or live_host.nproc > 0 else cfg.host - return replace( - cfg, - variant=os.environ.get("LUCEBOX_VARIANT", cfg.variant), - image=os.environ.get("LUCEBOX_IMAGE", cfg.image), - container_name=os.environ.get("LUCEBOX_CONTAINER", cfg.container_name), - port=int(os.environ.get("LUCEBOX_PORT", str(cfg.port))), - models_dir=Path(os.environ.get("LUCEBOX_MODELS", str(cfg.models_dir))), - host=host, - ) + try: + cfg = config_mod.load() + if cfg is None: + return live_config() + # Host facts exported by the wrapper take precedence over an absent or + # stale persisted snapshot. A zero-filled environment means the CLI was + # invoked directly, so retain any snapshot already in the config. + live_host = from_env() + host = live_host if live_host.vram_gb > 0 or live_host.nproc > 0 else cfg.host + return config_mod.overlay_env(replace(cfg, host=host)) + except (OSError, ValueError) as exc: + error_console.print(f"[red]Invalid configuration:[/red] {escape(str(exc))}") + raise typer.Exit(code=2) from exc + + +def _server_spec() -> docker_run.DockerRunSpec: + """Build the server command and turn user-path errors into clean CLI output.""" + cfg = _load_or_build() + try: + return docker_run.server_run_spec(cfg) + except (OSError, RuntimeError, ValueError) as exc: + error_console.print(f"[red]Cannot build server command:[/red] {escape(str(exc))}") + raise typer.Exit(code=2) from exc # ── subcommands ──────────────────────────────────────────────────────────── @@ -112,9 +117,7 @@ def pull() -> None: @app.command("print-run") def print_run() -> None: """Print the docker-run command for the server (copy-pasteable).""" - cfg = _load_or_build() - spec = docker_run.server_run_spec(cfg) - print(spec.printable()) + print(_server_spec().printable()) @app.command("print-serve-argv") @@ -125,9 +128,7 @@ def print_serve_argv() -> None: a separate command from `print-run` so the bash side has a guaranteed machine-readable contract that's independent of the pretty formatter. """ - cfg = _load_or_build() - spec = docker_run.server_run_spec(cfg) - for tok in spec.argv(): + for tok in _server_spec().argv(): print(tok) @@ -145,8 +146,8 @@ def config_get_cmd( """Print a single key (or every reachable key) with its origin annotation.""" try: entries = config_get(key or None) - except KeyError as exc: - console.print(f"[red]{escape(str(exc))}[/red]") + except (KeyError, OSError, ValueError) as exc: + error_console.print(f"[red]{escape(str(exc))}[/red]") raise typer.Exit(code=2) from exc for k, (value, origin) in entries.items(): console.print(f"{k} = {escape(repr(value))} ([dim]from {origin}[/dim])") @@ -170,8 +171,8 @@ def config_set_cmd( value = value.strip() try: config_set(key, value) - except (KeyError, ValueError) as exc: - console.print(f"[red]{escape(str(exc))}[/red]") + except (KeyError, OSError, ValueError) as exc: + error_console.print(f"[red]{escape(str(exc))}[/red]") raise typer.Exit(code=2) from exc console.print(f"[green]Set[/green] {escape(key)} = {escape(value)}") @@ -183,8 +184,8 @@ def config_unset_cmd( """Remove a key from config.toml. Next read uses the live default.""" try: changed = config_unset(key) - except KeyError as exc: - console.print(f"[red]{escape(str(exc))}[/red]") + except (KeyError, OSError, ValueError) as exc: + error_console.print(f"[red]{escape(str(exc))}[/red]") raise typer.Exit(code=2) from exc if changed: console.print(f"[green]Unset[/green] {escape(key)}") diff --git a/lucebox/src/lucebox/config.py b/lucebox/src/lucebox/config.py index 7ca5d2445..75a601afb 100644 --- a/lucebox/src/lucebox/config.py +++ b/lucebox/src/lucebox/config.py @@ -9,7 +9,7 @@ The dotted-key surface area is small and flat: model.preset, model.target_file, model.draft_file port, models_dir, variant, image, container_name - dflash. for each of the 11 DflashRuntime knobs + think_max + dflash. for every registered DflashRuntime knob Load resolves the TOML file → ``Config`` object, with anything absent filled from ``Config()`` defaults. Save writes back only the keys that @@ -25,7 +25,7 @@ from collections.abc import Callable from dataclasses import asdict, replace from datetime import UTC -from pathlib import Path +from pathlib import Path, PurePosixPath from typing import Any, Literal, cast import tomli_w @@ -44,8 +44,8 @@ def default_config_path() -> Path: """Where .lucebox/config.toml lives. Convention: under $LUCEBOX_HOME if set, otherwise $HOME/.lucebox. Lives in - the bind-mounted host home dir so the config survives container teardown - and is editable from the host. + an explicitly bind-mounted application directory so the config survives + container teardown without exposing the rest of the host home directory. """ base = os.environ.get("LUCEBOX_HOME") if base: @@ -85,6 +85,50 @@ def _cast_bool(v: Any) -> bool: raise ValueError(f"cannot parse boolean: {v!r}") +def _cast_port(v: Any) -> int: + value = int(v) + if not 1 <= value <= 65535: + raise ValueError(f"port must be in the interval [1, 65535], got {value!r}") + return value + + +def _cast_models_dir(v: Any) -> str: + value = str(v) + if not Path(value).is_absolute(): + raise ValueError(f"models_dir must be an absolute path, got {value!r}") + return value + + +def _cast_model_relative_path(v: Any) -> str: + value = str(v) + if not value: + return value + path = PurePosixPath(value) + if path.is_absolute() or ".." in path.parts or value == ".": + raise ValueError(f"model file must be below models_dir, got {value!r}") + return value + + +def _cast_prefill_keep_ratio(v: Any) -> float: + value = float(v) + if not 0.0 < value <= 1.0: + raise ValueError( + "prefill_keep_ratio must be in the interval (0.0, 1.0], " + f"got {value!r}" + ) + return value + + +def _cast_think_soft_close_min_ratio(v: Any) -> float: + value = float(v) + if not 0.0 <= value <= 1.0: + raise ValueError( + "think_soft_close_min_ratio must be in the interval [0.0, 1.0], " + f"got {value!r}" + ) + return value + + # Each entry: dotted-key → (toml_path, type_caster, default_getter). # ``toml_path`` is the (section, field) pair on disk; ``"_root"`` means the # key lives at the top level (no [section]). ``default_getter`` returns the @@ -93,11 +137,11 @@ def _cast_bool(v: Any) -> bool: "variant": (("image", "variant"), str), "image": (("image", "registry"), str), "container_name": (("runtime", "container_name"), str), - "port": (("runtime", "port"), int), - "models_dir": (("paths", "models"), str), + "port": (("runtime", "port"), _cast_port), + "models_dir": (("paths", "models"), _cast_models_dir), "model.preset": (("model", "preset"), str), - "model.target_file": (("model", "target_file"), str), - "model.draft_file": (("model", "draft_file"), str), + "model.target_file": (("model", "target_file"), _cast_model_relative_path), + "model.draft_file": (("model", "draft_file"), _cast_model_relative_path), "dflash.budget": (("dflash", "budget"), int), "dflash.max_ctx": (("dflash", "max_ctx"), int), "dflash.lazy": (("dflash", "lazy"), _cast_bool), @@ -106,15 +150,22 @@ def _cast_bool(v: Any) -> bool: "dflash.cache_type_k": (("dflash", "cache_type_k"), str), "dflash.cache_type_v": (("dflash", "cache_type_v"), str), "dflash.prefill_mode": (("dflash", "prefill_mode"), _cast_prefill_mode), - "dflash.prefill_keep_ratio": (("dflash", "prefill_keep_ratio"), float), + "dflash.prefill_keep_ratio": ( + ("dflash", "prefill_keep_ratio"), + _cast_prefill_keep_ratio, + ), "dflash.prefill_threshold": (("dflash", "prefill_threshold"), int), "dflash.prefill_drafter": (("dflash", "prefill_drafter"), str), "dflash.think_max": (("dflash", "think_max"), int), "dflash.fa_window": (("dflash", "fa_window"), int), "dflash.think_soft_close_min_ratio": ( - ("dflash", "think_soft_close_min_ratio"), float), + ("dflash", "think_soft_close_min_ratio"), + _cast_think_soft_close_min_ratio, + ), "dflash.debug_thinking_logits": ( - ("dflash", "debug_thinking_logits"), _cast_bool), + ("dflash", "debug_thinking_logits"), + _cast_bool, + ), } @@ -226,11 +277,11 @@ def _from_dict(raw: dict[str, Any]) -> Config: registry = img.get("registry", "ghcr.io/luce-org/lucebox-hub") runtime = raw.get("runtime", {}) - port = int(runtime.get("port", 8080)) + port = _cast_port(runtime.get("port", 8080)) container_name = str(runtime.get("container_name", "lucebox")) paths = raw.get("paths", {}) - models_dir = Path(paths.get("models", str(default_models_dir()))) + models_dir = Path(_cast_models_dir(paths.get("models", str(default_models_dir())))) df = raw.get("dflash", {}) dflash = DflashRuntime( @@ -242,13 +293,16 @@ def _from_dict(raw: dict[str, Any]) -> Config: cache_type_k=str(df.get("cache_type_k", "")), cache_type_v=str(df.get("cache_type_v", "")), prefill_mode=_cast_prefill_mode(df.get("prefill_mode", "off")), - prefill_keep_ratio=float(df.get("prefill_keep_ratio", 0.05)), + prefill_keep_ratio=_cast_prefill_keep_ratio( + df.get("prefill_keep_ratio", 0.05) + ), prefill_threshold=int(df.get("prefill_threshold", 32000)), prefill_drafter=str(df.get("prefill_drafter", "")), think_max=int(df.get("think_max", 15488)), fa_window=int(df.get("fa_window", 0)), - think_soft_close_min_ratio=float( - df.get("think_soft_close_min_ratio", 0.0)), + think_soft_close_min_ratio=_cast_think_soft_close_min_ratio( + df.get("think_soft_close_min_ratio", 0.0) + ), debug_thinking_logits=_cast_bool(df.get("debug_thinking_logits", False)), ) @@ -281,8 +335,8 @@ def _from_dict(raw: dict[str, Any]) -> Config: # them from the registry so users only have to write one key. mdl = raw.get("model", {}) preset_name = str(mdl.get("preset", "")) - target_file = str(mdl.get("target_file", "")) - draft_file = str(mdl.get("draft_file", "")) + target_file = _cast_model_relative_path(mdl.get("target_file", "")) + draft_file = _cast_model_relative_path(mdl.get("draft_file", "")) if preset_name and (not target_file or not draft_file): from lucebox.download import PRESETS @@ -466,6 +520,24 @@ def config_get(key: str | None = None, *, path: Path | None = None) -> dict[str, return out +def overlay_env(cfg: Config) -> Config: + """Apply supported process overrides to an existing config. + + Keeping this in one helper makes the documented ``env > TOML > default`` + precedence identical for both persisted and first-run configurations. + """ + return replace( + cfg, + variant=os.environ.get("LUCEBOX_VARIANT", cfg.variant), + image=os.environ.get("LUCEBOX_IMAGE", cfg.image), + container_name=os.environ.get("LUCEBOX_CONTAINER", cfg.container_name), + port=_cast_port(os.environ.get("LUCEBOX_PORT", str(cfg.port))), + models_dir=Path( + _cast_models_dir(os.environ.get("LUCEBOX_MODELS", str(cfg.models_dir))) + ), + ) + + def live_config() -> Config: """Build a fresh Config from current host facts + the DFLASH_* heuristic. @@ -481,13 +553,10 @@ def live_config() -> Config: host = from_env() default = Config() default_variant = "rocm" if host.gpu_vendor == "amd" else "cuda12" - return replace( + cfg = replace( default, - variant=os.environ.get("LUCEBOX_VARIANT", default_variant), - image=os.environ.get("LUCEBOX_IMAGE", default.image), - container_name=os.environ.get("LUCEBOX_CONTAINER", default.container_name), - port=int(os.environ.get("LUCEBOX_PORT", str(default.port))), - models_dir=Path(os.environ.get("LUCEBOX_MODELS", str(default.models_dir))), + variant=default_variant, dflash=autotune_mod.runtime_from_host(host), host=host, ) + return overlay_env(cfg) diff --git a/lucebox/src/lucebox/docker_run.py b/lucebox/src/lucebox/docker_run.py index ab34c837a..f9b618007 100644 --- a/lucebox/src/lucebox/docker_run.py +++ b/lucebox/src/lucebox/docker_run.py @@ -13,10 +13,42 @@ import shlex import subprocess from dataclasses import dataclass -from pathlib import Path +from pathlib import Path, PurePosixPath from lucebox.types import Config, GpuVendor +_CONTAINER_MODELS = "/opt/lucebox-hub/server/models" +_CONTAINER_RESOLVED_MODELS = "/opt/lucebox-resolved" + + +@dataclass(frozen=True, slots=True) +class BindMount: + """One explicit Docker bind mount. + + ``read_only`` is part of the type so sensitive compatibility mounts cannot + accidentally become writable during argv rendering. + """ + + source: str + target: str + read_only: bool = False + + def __post_init__(self) -> None: + if not Path(self.source).is_absolute(): + raise ValueError(f"bind-mount source must be absolute, got {self.source!r}") + if not PurePosixPath(self.target).is_absolute(): + raise ValueError(f"bind-mount target must be absolute, got {self.target!r}") + for label, value in (("source", self.source), ("target", self.target)): + if "," in value or any(char in value for char in "\n\r\0"): + raise ValueError( + f"bind-mount {label} contains a character unsupported by " + f"Docker --mount: {value!r}" + ) + + def argument(self) -> str: + option = f"type=bind,source={self.source},target={self.target}" + return f"{option},readonly" if self.read_only else option + def _host_facts_env() -> list[tuple[str, str]]: """Forward LUCEBOX_HOST_* from the orchestrator's env into the server. @@ -69,27 +101,68 @@ def _resolve_model_files(cfg: Config) -> tuple[str, str, str]: draft = pres.draft_file if not draft and pres.speculator_dir: spec_path = cfg.models_dir / "draft" / pres.speculator_dir - if spec_path.is_dir(): + if spec_path.is_dir() or spec_path.is_symlink(): draft_dir = pres.speculator_dir return target, draft, draft_dir -def _runtime_volumes(cfg: Config) -> tuple[tuple[str, str], ...]: - """Mount models, $HOME, and any config dir outside the home mount.""" - home_path = Path.home() - home = str(home_path) - models = str(cfg.models_dir) - volumes = [(models, "/opt/lucebox-hub/server/models")] - if home != models: - volumes.append((home, home)) +def _runtime_volumes(cfg: Config) -> tuple[BindMount, ...]: + """Mount only the writable application data needed by the server.""" + models = str(cfg.models_dir.absolute()) config_home = Path( - os.environ.get("LUCEBOX_HOME") or home_path / ".lucebox" + os.environ.get("LUCEBOX_HOME") or Path.home() / ".lucebox" ).absolute() - # The same-path home mount normally covers config. If models_dir == HOME, - # that path is mounted at /opt/... instead, so add config explicitly too. - if home == models or not config_home.is_relative_to(home_path): - volumes.append((str(config_home), str(config_home))) - return tuple(volumes) + return ( + BindMount(models, _CONTAINER_MODELS), + BindMount(str(config_home), str(config_home)), + ) + + +def _validate_model_relative_path(value: str, field: str) -> PurePosixPath: + """Validate a config-provided path intended to live below models_dir.""" + relative = PurePosixPath(value) + if relative.is_absolute() or ".." in relative.parts or value in {"", "."}: + raise ValueError(f"{field} must be a path below models_dir, got {value!r}") + return relative + + +def _selected_model_path( + cfg: Config, + value: str, + *, + field: str, + role: str, + under_draft: bool, + directory: bool, + mounts: list[BindMount], +) -> str: + """Return the container path for one explicitly selected model artifact. + + Normal files are already covered by the models-directory bind mount. For + a symlink, bind only its resolved file (or the selected directory for a + speculator) into a dedicated read-only location. This preserves symlinked + model workflows without exposing the user's home or adjacent model files. + """ + relative = _validate_model_relative_path(value, field) + base = cfg.models_dir / "draft" if under_draft else cfg.models_dir + host_path = base.joinpath(*relative.parts) + container_base = PurePosixPath(_CONTAINER_MODELS) + if under_draft: + container_base /= "draft" + canonical = str(container_base.joinpath(*relative.parts)) + lexical = host_path.absolute() + resolved = host_path.resolve(strict=False) + if resolved == lexical: + return canonical + + mount_target = f"{_CONTAINER_RESOLVED_MODELS}/{role}" + if directory: + mounts.append(BindMount(str(resolved), mount_target, read_only=True)) + return mount_target + + container_path = f"{mount_target}/{resolved.name}" + mounts.append(BindMount(str(resolved), container_path, read_only=True)) + return container_path @dataclass(frozen=True, slots=True) @@ -103,7 +176,7 @@ class DockerRunSpec: detach: bool = False remove: bool = True port_publish: tuple[int, int] | None = None # (host, container) - volumes: tuple[tuple[str, str], ...] = () + volumes: tuple[BindMount, ...] = () env: tuple[tuple[str, str], ...] = () entrypoint_args: tuple[str, ...] = () extra: tuple[str, ...] = () @@ -133,8 +206,8 @@ def argv(self) -> list[str]: if self.port_publish is not None: host, container = self.port_publish out += ["-p", f"{host}:{container}"] - for host_path, container_path in self.volumes: - out += ["-v", f"{host_path}:{container_path}"] + for mount in self.volumes: + out += ["--mount", mount.argument()] for k, v in self.env: out += ["-e", f"{k}={v}"] out += list(self.extra) @@ -161,6 +234,7 @@ def printable(self) -> str: "--gpus", "--env", "--volume", + "--mount", "--publish", "--entrypoint", "--device", @@ -204,12 +278,40 @@ def server_run_spec(cfg: Config) -> DockerRunSpec: # draft_dir is a subdirectory of models/draft/ holding a safetensors speculator; # it takes effect only when draft_file is empty and the directory exists on disk. target_file, draft_file, draft_dir = _resolve_model_files(cfg) + volumes = list(_runtime_volumes(cfg)) if target_file: - env.append(("DFLASH_TARGET", f"/opt/lucebox-hub/server/models/{target_file}")) + target_path = _selected_model_path( + cfg, + target_file, + field="model.target_file", + role="target", + under_draft=False, + directory=False, + mounts=volumes, + ) + env.append(("DFLASH_TARGET", target_path)) if draft_file: - env.append(("DFLASH_DRAFT", f"/opt/lucebox-hub/server/models/draft/{draft_file}")) + draft_path = _selected_model_path( + cfg, + draft_file, + field="model.draft_file", + role="draft", + under_draft=True, + directory=False, + mounts=volumes, + ) + env.append(("DFLASH_DRAFT", draft_path)) elif draft_dir: - env.append(("DFLASH_DRAFT", f"/opt/lucebox-hub/server/models/draft/{draft_dir}")) + draft_path = _selected_model_path( + cfg, + draft_dir, + field="model.speculator_dir", + role="draft-dir", + under_draft=True, + directory=True, + mounts=volumes, + ) + env.append(("DFLASH_DRAFT", draft_path)) elif cfg.model.preset: # An active target-only preset is an explicit choice, not permission # for entrypoint.sh to scan models/draft and attach an unrelated stale @@ -267,7 +369,7 @@ def server_run_spec(cfg: Config) -> DockerRunSpec: remove=True, detach=False, port_publish=(cfg.port, 8080), - volumes=_runtime_volumes(cfg), + volumes=tuple(volumes), env=tuple(env), ) diff --git a/lucebox/src/lucebox/download.py b/lucebox/src/lucebox/download.py index df1d5c96f..43aeb83ca 100644 --- a/lucebox/src/lucebox/download.py +++ b/lucebox/src/lucebox/download.py @@ -28,17 +28,8 @@ import time from dataclasses import dataclass from pathlib import Path +from typing import TYPE_CHECKING -# hf-xet (huggingface_hub ≥ 1.16) streams the entire file in one final -# burst — the polling-based progress bar sits at 0% for ~14 minutes -# then snaps to 100% on a 17 GB GGUF. Force the chunked Python -# downloader instead so bytes grow continuously and the Rich bar tracks -# reality. Set before importing hf_hub_download so the import picks -# the env up. `setdefault` lets a user override on the command line. -os.environ.setdefault("HF_HUB_DISABLE_XET", "1") - -from huggingface_hub import HfApi, hf_hub_download # noqa: E402 -from huggingface_hub._local_folder import get_local_download_paths # noqa: E402 from rich.console import Console from rich.progress import ( BarColumn, @@ -51,6 +42,27 @@ from lucebox.types import Config, HostFacts +if TYPE_CHECKING: + from huggingface_hub import HfApi + + +def _configure_huggingface_download() -> None: + """Select the progress-friendly downloader before importing HF Hub. + + hf-xet streams a multi-GB file in one final burst, so the polling-based + progress bar remains at zero until completion. Keep this environment + change local to download/status operations rather than mutating every + ``lucebox`` process merely because the CLI module was imported. + """ + os.environ.setdefault("HF_HUB_DISABLE_XET", "1") + + +def _new_hf_api() -> HfApi: + _configure_huggingface_download() + from huggingface_hub import HfApi + + return HfApi() + @dataclass(frozen=True, slots=True) class ModelPreset: @@ -244,6 +256,9 @@ def _incomplete_path_candidates(local_dir: Path, filename: str, etag: str | None when ``local_dir`` is set hf-hub always uses the local staging dir, so the two candidates above cover every code path we hit. """ + _configure_huggingface_download() + from huggingface_hub._local_folder import get_local_download_paths + paths = get_local_download_paths(local_dir, filename) candidates: list[Path] = [] if etag: @@ -302,6 +317,9 @@ def _download_with_progress( actual hf-xet staging path (a hashed filename under ``.cache/huggingface/download/``), not a guess. """ + _configure_huggingface_download() + from huggingface_hub import hf_hub_download + local_dir.mkdir(parents=True, exist_ok=True) target = local_dir / filename candidates = _incomplete_path_candidates(local_dir, filename, etag) @@ -384,7 +402,7 @@ def download_preset(cfg: Config, preset: ModelPreset | None = None) -> int: """ preset = preset or DEFAULT_PRESET console = Console() - api = HfApi() + api = _new_hf_api() models = cfg.models_dir models.mkdir(parents=True, exist_ok=True) draft = models / "draft" @@ -437,7 +455,7 @@ def installed_status(cfg: Config, preset: ModelPreset) -> str: def installed_size_gb(cfg: Config, preset: ModelPreset) -> float: - """Sum of on-disk byte sizes for the preset's files, in GB (binary 1e9).""" + """Sum of on-disk byte sizes for the preset's files, in decimal GB (1e9).""" total = 0 target = _local_target_path(cfg, preset) if target.exists(): @@ -478,7 +496,7 @@ def status(cfg: Config, preset: ModelPreset | None = None) -> dict[str, bool]: not a draft exists. """ preset = preset or DEFAULT_PRESET - api = HfApi() + api = _new_hf_api() out: dict[str, bool] = {} try: size, _ = _file_meta(api, preset.target_repo, preset.target_file) diff --git a/lucebox/src/lucebox/types.py b/lucebox/src/lucebox/types.py index e9e2fa3ec..d755083c0 100644 --- a/lucebox/src/lucebox/types.py +++ b/lucebox/src/lucebox/types.py @@ -60,6 +60,24 @@ class HostFacts: docker_version: str = "" ctk: CtkStatus = "none" + def __post_init__(self) -> None: + """Reject malformed persisted host snapshots. + + ``Literal`` annotations help static type-checkers, but TOML is runtime + input and can still contain arbitrary strings. Failing here keeps an + invalid snapshot from silently selecting the wrong accelerator path. + """ + if self.gpu_vendor not in {"nvidia", "amd", "none"}: + raise ValueError( + "gpu_vendor must be nvidia, amd, or none; " + f"got {self.gpu_vendor!r}" + ) + if self.ctk not in {"runtime", "cdi", "installed-unwired", "none"}: + raise ValueError( + "ctk must be runtime, cdi, installed-unwired, or none; " + f"got {self.ctk!r}" + ) + @dataclass(frozen=True, slots=True) class DflashRuntime: @@ -92,7 +110,7 @@ class DflashRuntime: # attention (server default). On gemma4's hybrid iSWA the full-attn # layers grow KV linearly with max_ctx; a sparse fa_window keeps # decode compute bounded on long prompts without changing the KV - # footprint. Q: passed through to the server's `--fa-window ` + # footprint. Passed through to the server's `--fa-window ` # flag (see server/src/server/server_main.cpp). fa_window: int = 0 # Soft-close thinking termination dial (PR #326 in lucebox-hub). @@ -112,6 +130,19 @@ class DflashRuntime: # in-flight requests); leave off in production. debug_thinking_logits: bool = False + def __post_init__(self) -> None: + """Validate the bounded tuning knobs before they reach the server.""" + if not 0.0 < self.prefill_keep_ratio <= 1.0: + raise ValueError( + "prefill_keep_ratio must be in the interval (0.0, 1.0]; " + f"got {self.prefill_keep_ratio!r}" + ) + if not 0.0 <= self.think_soft_close_min_ratio <= 1.0: + raise ValueError( + "think_soft_close_min_ratio must be in the interval [0.0, 1.0]; " + f"got {self.think_soft_close_min_ratio!r}" + ) + @dataclass(frozen=True, slots=True) class ModelMeta: diff --git a/lucebox/tests/test_cli.py b/lucebox/tests/test_cli.py index f7628e8b7..ee254d9bd 100644 --- a/lucebox/tests/test_cli.py +++ b/lucebox/tests/test_cli.py @@ -5,7 +5,7 @@ import os import pytest -from lucebox.cli import app +from lucebox.cli import __version__, app from typer.testing import CliRunner @@ -57,6 +57,14 @@ def test_core_verbs_present_in_app() -> None: assert verb in registered +@pytest.mark.parametrize("args", [["version"], ["--version"]]) +def test_version_command_and_option_match(args: list[str]) -> None: + result = CliRunner().invoke(app, args) + + assert result.exit_code == 0 + assert result.stdout.strip() == __version__ + + def test_legacy_subcommands_are_removed() -> None: """`configure` and `download-models` were folded into config/models.""" cfg = CliRunner().invoke(app, ["configure", "--help"]) diff --git a/lucebox/tests/test_config.py b/lucebox/tests/test_config.py index afedc7d91..a61f411f6 100644 --- a/lucebox/tests/test_config.py +++ b/lucebox/tests/test_config.py @@ -57,6 +57,56 @@ def test_toml_load_rejects_invalid_prefill_mode(tmp_path: Path) -> None: config._load_toml(path) +@pytest.mark.parametrize( + ("key", "value"), + [ + ("dflash.prefill_keep_ratio", "0"), + ("dflash.prefill_keep_ratio", "1.01"), + ("dflash.think_soft_close_min_ratio", "-0.01"), + ("dflash.think_soft_close_min_ratio", "1.01"), + ("dflash.think_soft_close_min_ratio", "nan"), + ], +) +def test_config_set_rejects_out_of_range_ratios( + tmp_path: Path, key: str, value: str +) -> None: + path = tmp_path / "config.toml" + + with pytest.raises(ValueError, match="interval"): + config_set(key, value, path=path) + + assert not path.exists() + + +def test_toml_load_rejects_out_of_range_ratios(tmp_path: Path) -> None: + path = tmp_path / "config.toml" + path.write_text( + "[dflash]\n" + "prefill_keep_ratio = 0.05\n" + "think_soft_close_min_ratio = 2.0\n" + ) + + with pytest.raises(ValueError, match="think_soft_close_min_ratio"): + config._load_toml(path) + + +@pytest.mark.parametrize( + "host_body", + [ + 'gpu_vendor = "intel"\n', + 'ctk = "maybe"\n', + ], +) +def test_toml_load_rejects_invalid_host_literals( + tmp_path: Path, host_body: str +) -> None: + path = tmp_path / "config.toml" + path.write_text(f"[host]\n{host_body}") + + with pytest.raises(ValueError): + config._load_toml(path) + + def test_model_preset_round_trips_through_set_and_load(tmp_path: Path) -> None: """Setting model.preset writes a sparse TOML doc that loads back correctly.""" path = tmp_path / "config.toml" @@ -171,6 +221,23 @@ def test_config_set_rejects_unknown_key(tmp_path: Path) -> None: config_set("not.a.key", 1, path=path) +@pytest.mark.parametrize( + ("key", "value", "message"), + [ + ("port", "0", "port"), + ("port", "65536", "port"), + ("models_dir", "relative/models", "absolute"), + ("model.target_file", "../secret.gguf", "below models_dir"), + ("model.draft_file", "/tmp/draft.gguf", "below models_dir"), + ], +) +def test_config_set_rejects_unsafe_runtime_values( + tmp_path: Path, key: str, value: str, message: str +) -> None: + with pytest.raises(ValueError, match=message): + config_set(key, value, path=tmp_path / "config.toml") + + def test_config_set_auto_creates_file(tmp_path: Path) -> None: """`config set` creates a missing config.toml on first write.""" path = tmp_path / "config.toml" diff --git a/lucebox/tests/test_config_cli.py b/lucebox/tests/test_config_cli.py index 6525ea319..8d36da760 100644 --- a/lucebox/tests/test_config_cli.py +++ b/lucebox/tests/test_config_cli.py @@ -138,3 +138,16 @@ def test_load_or_build_no_toml_env_overrides_defaults( monkeypatch.setenv("LUCEBOX_IMAGE", "ghcr.io/myfork/lucebox-hub") cfg = _load_or_build() assert cfg.image == "ghcr.io/myfork/lucebox-hub" + + +def test_print_run_reports_invalid_configuration_without_traceback( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + _set_config_path(tmp_path, monkeypatch) + monkeypatch.setenv("LUCEBOX_PORT", "0") + + result = CliRunner().invoke(app, ["print-run"]) + + assert result.exit_code == 2 + assert "Invalid configuration" in result.output + assert "Traceback" not in result.output diff --git a/lucebox/tests/test_docker_run.py b/lucebox/tests/test_docker_run.py index f5a18ed59..9e693934f 100644 --- a/lucebox/tests/test_docker_run.py +++ b/lucebox/tests/test_docker_run.py @@ -10,6 +10,7 @@ from pathlib import Path +import pytest from lucebox.download import PRESETS from lucebox.types import Config, DflashRuntime, HostFacts, ModelMeta @@ -43,7 +44,9 @@ def test_argv_flags_and_ordering() -> None: detach=True, remove=False, port_publish=(8080, 8080), - volumes=(("/host/models", "/opt/lucebox-hub/server/models"),), + volumes=( + docker_run.BindMount("/host/models", "/opt/lucebox-hub/server/models"), + ), env=(("DFLASH_BUDGET", "22"),), entrypoint_args=("serve",), extra=("--shm-size", "1g"), @@ -53,8 +56,9 @@ def test_argv_flags_and_ordering() -> None: assert "-d" in argv # detach assert "--gpus" not in argv # gpus=False assert ["-p", "8080:8080"] == argv[argv.index("-p") : argv.index("-p") + 2] - assert ["-v", "/host/models:/opt/lucebox-hub/server/models"] == argv[ - argv.index("-v") : argv.index("-v") + 2 + mount_arg = "type=bind,source=/host/models,target=/opt/lucebox-hub/server/models" + assert ["--mount", mount_arg] == argv[ + argv.index("--mount") : argv.index("--mount") + 2 ] assert ["-e", "DFLASH_BUDGET=22"] == argv[argv.index("-e") : argv.index("-e") + 2] # extra flags precede the image; entrypoint_args follow it. @@ -63,6 +67,20 @@ def test_argv_flags_and_ordering() -> None: assert argv.index("--shm-size") < argv.index("img:tag") +@pytest.mark.parametrize( + ("source", "target"), + [ + ("relative", "/container"), + ("/host", "relative"), + ("/host,comma", "/container"), + ("/host", "/container\nnewline"), + ], +) +def test_bind_mount_rejects_ambiguous_paths(source: str, target: str) -> None: + with pytest.raises(ValueError, match="bind-mount"): + docker_run.BindMount(source, target) + + def test_argv_amd_uses_rocm_device_contract() -> None: spec = docker_run.DockerRunSpec(image="img:rocm", name="box", gpu_vendor="amd") argv = spec.argv() @@ -100,12 +118,21 @@ def test_printable_glues_value_taking_flags() -> None: # ── _runtime_volumes ───────────────────────────────────────────────────────── -def test_runtime_volumes_mounts_models_and_home(tmp_path: Path) -> None: +def test_runtime_volumes_mounts_only_models_and_config( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + home = tmp_path / "home" + monkeypatch.setattr(Path, "home", staticmethod(lambda: home)) + monkeypatch.delenv("LUCEBOX_HOME", raising=False) cfg = Config(models_dir=tmp_path / "models") vols = docker_run._runtime_volumes(cfg) - assert (str(tmp_path / "models"), "/opt/lucebox-hub/server/models") in vols - # $HOME is also mounted so absolute symlink targets resolve in-container. - assert any(host == str(Path.home()) for host, _ in vols) + assert docker_run.BindMount( + str(tmp_path / "models"), "/opt/lucebox-hub/server/models" + ) in vols + assert docker_run.BindMount( + str(home / ".lucebox"), str(home / ".lucebox") + ) in vols + assert all(mount.source != str(home) for mount in vols) def test_runtime_volumes_dedupes_when_models_is_home(monkeypatch, tmp_path: Path) -> None: @@ -113,10 +140,16 @@ def test_runtime_volumes_dedupes_when_models_is_home(monkeypatch, tmp_path: Path monkeypatch.delenv("LUCEBOX_HOME", raising=False) cfg = Config(models_dir=tmp_path) vols = docker_run._runtime_volumes(cfg) - # models_dir == home is mounted at /opt/..., not at its host path, so the - # nested config directory still needs an explicit same-path mount. + # models_dir is mounted at the image's canonical path, while config keeps + # its same-path mount. The parent home directory itself is never exposed. assert len(vols) == 2 - assert (str(tmp_path / ".lucebox"), str(tmp_path / ".lucebox")) in vols + assert docker_run.BindMount( + str(tmp_path / ".lucebox"), str(tmp_path / ".lucebox") + ) in vols + assert all( + mount.source != str(tmp_path) or mount.target != str(tmp_path) + for mount in vols + ) def test_runtime_volumes_mounts_custom_config_home( @@ -129,7 +162,7 @@ def test_runtime_volumes_mounts_custom_config_home( vols = docker_run._runtime_volumes(Config(models_dir=tmp_path / "models")) - assert (str(config_home), str(config_home)) in vols + assert docker_run.BindMount(str(config_home), str(config_home)) in vols def test_empty_lucebox_home_uses_default(monkeypatch, tmp_path: Path) -> None: @@ -190,7 +223,69 @@ def test_server_run_spec_top_level_shape(tmp_path: Path) -> None: assert spec.remove is True assert spec.detach is False assert spec.port_publish == (9000, 8080) - assert (str(tmp_path), "/opt/lucebox-hub/server/models") in spec.volumes + assert docker_run.BindMount( + str(tmp_path), "/opt/lucebox-hub/server/models" + ) in spec.volumes + + +def test_server_run_spec_mounts_selected_symlink_target_narrowly_read_only( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + home = tmp_path / "home" + models = tmp_path / "models" + external = home / "model-cache" + models.mkdir() + external.mkdir(parents=True) + target = external / "target.gguf" + target.write_bytes(b"model") + (models / "selected.gguf").symlink_to(target) + monkeypatch.setattr(Path, "home", staticmethod(lambda: home)) + + spec = docker_run.server_run_spec( + Config(models_dir=models, model=ModelMeta(target_file="selected.gguf")) + ) + + assert docker_run.BindMount( + str(target), + "/opt/lucebox-resolved/target/target.gguf", + read_only=True, + ) in spec.volumes + assert _env(spec)["DFLASH_TARGET"] == "/opt/lucebox-resolved/target/target.gguf" + assert all(mount.source != str(home) for mount in spec.volumes) + argv = spec.argv() + mount_args = [argv[i + 1] for i, token in enumerate(argv) if token == "--mount"] + assert any( + "target=/opt/lucebox-resolved/target/target.gguf,readonly" in argument + for argument in mount_args + ) + + +def test_server_run_spec_resolves_symlinked_model_parent(tmp_path: Path) -> None: + models = tmp_path / "models" + external = tmp_path / "external" + models.mkdir() + external.mkdir() + target = external / "nested.gguf" + target.write_bytes(b"model") + (models / "selected").symlink_to(external, target_is_directory=True) + + spec = docker_run.server_run_spec( + Config(models_dir=models, model=ModelMeta(target_file="selected/nested.gguf")) + ) + + assert docker_run.BindMount( + str(target), + "/opt/lucebox-resolved/target/nested.gguf", + read_only=True, + ) in spec.volumes + assert _env(spec)["DFLASH_TARGET"] == "/opt/lucebox-resolved/target/nested.gguf" + + +def test_server_run_spec_rejects_model_path_traversal(tmp_path: Path) -> None: + cfg = Config(models_dir=tmp_path, model=ModelMeta(target_file="../secret.gguf")) + + with pytest.raises(ValueError, match="below models_dir"): + docker_run.server_run_spec(cfg) def test_server_run_spec_rocm_uses_amd_devices_on_heterogeneous_host(tmp_path: Path) -> None: diff --git a/lucebox/tests/test_download.py b/lucebox/tests/test_download.py index 8b69e96b8..a4bf692e3 100644 --- a/lucebox/tests/test_download.py +++ b/lucebox/tests/test_download.py @@ -7,6 +7,8 @@ stays pinned without actually talking to the Hub. """ +import subprocess +import sys from pathlib import Path from types import SimpleNamespace @@ -23,6 +25,22 @@ from lucebox import download +def test_import_does_not_mutate_huggingface_environment() -> None: + code = ( + "import os; " + "os.environ.pop('HF_HUB_DISABLE_XET', None); " + "import lucebox.download; " + "assert 'HF_HUB_DISABLE_XET' not in os.environ" + ) + result = subprocess.run( + [sys.executable, "-c", code], + check=False, + capture_output=True, + text=True, + ) + assert result.returncode == 0, result.stderr + + def test_default_preset_uses_quantized_gguf_draft(): assert DEFAULT_PRESET.draft_repo == "spiritbuun/Qwen3.6-27B-DFlash-GGUF" assert DEFAULT_PRESET.draft_file == "dflash-draft-3.6-q4_k_m.gguf" diff --git a/lucebox/tests/test_models_cli.py b/lucebox/tests/test_models_cli.py index 744f40c29..dd4f244f5 100644 --- a/lucebox/tests/test_models_cli.py +++ b/lucebox/tests/test_models_cli.py @@ -44,7 +44,7 @@ def test_models_default_view_lists_only_installed( # No models on disk → default view says "no presets installed". result = CliRunner().invoke(app, ["models"]) assert result.exit_code == 0 - assert "No presets installed" in result.stdout or "Models dir" in result.stdout + assert "No presets installed" in result.stdout def test_models_download_recommends_when_empty( diff --git a/scripts/check_lucebox_wrapper_sandbox.sh b/scripts/check_lucebox_wrapper_sandbox.sh index df2b2b9bc..77f571c8b 100755 --- a/scripts/check_lucebox_wrapper_sandbox.sh +++ b/scripts/check_lucebox_wrapper_sandbox.sh @@ -95,8 +95,12 @@ run_logged_capture() { note "run: $* > $out" { printf '\n===== %s > %s =====\n' "$*" "$out" - "$@" - local rc=$? + local rc + if "$@"; then + rc=0 + else + rc=$? + fi printf '===== exit=%s =====\n' "$rc" return "$rc" } 2>&1 | tee "$out" | tee -a "$LOG" >/dev/null @@ -164,10 +168,9 @@ run_logged_capture "$ROOT/help.out" lucebox help assert_contains "$ROOT/help.out" "LUCEBOX_VARIANT" assert_contains "$ROOT/help.out" "LUCEBOX_IMAGE" -docker manifest inspect "${IMAGE}:${VARIANT}" >/dev/null -pass "image manifest exists: ${IMAGE}:${VARIANT}" - if [ "$RUN_PULL" = "1" ]; then + docker manifest inspect "${IMAGE}:${VARIANT}" >/dev/null + pass "image manifest exists: ${IMAGE}:${VARIANT}" run_logged_capture "$ROOT/pull.out" lucebox pull assert_contains "$ROOT/pull.out" "${IMAGE}:${VARIANT}" fi diff --git a/scripts/test_lucebox_sh.sh b/scripts/test_lucebox_sh.sh index c9fc0577d..d95513daf 100755 --- a/scripts/test_lucebox_sh.sh +++ b/scripts/test_lucebox_sh.sh @@ -26,6 +26,7 @@ ROOT="$(git rev-parse --show-toplevel 2>/dev/null || (cd "$(dirname "$0")/.." && SCRIPT="$ROOT/lucebox.sh" ENTRYPOINT="$ROOT/server/scripts/entrypoint.sh" INSTALLER="$ROOT/install.sh" +SELF_PATH="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)/$(basename "${BASH_SOURCE[0]}")" if [ ! -f "$SCRIPT" ]; then echo "FAIL: lucebox.sh not found at $SCRIPT" >&2 @@ -46,6 +47,23 @@ exit 1 STUB chmod +x "$SUITE_SHIMS/$binname" done +if ! command -v timeout >/dev/null 2>&1; then + # macOS has no GNU `timeout`; keep the contributor test suite portable + # with a tiny subprocess wrapper. CI still uses the native coreutils tool. + cat > "$SUITE_SHIMS/timeout" <<'PYTHON_TIMEOUT' +#!/usr/bin/env python3 +import subprocess +import sys + +duration = float(sys.argv[1].removesuffix("s")) +try: + result = subprocess.run(sys.argv[2:], timeout=duration, check=False) +except subprocess.TimeoutExpired: + raise SystemExit(124) +raise SystemExit(result.returncode) +PYTHON_TIMEOUT + chmod +x "$SUITE_SHIMS/timeout" +fi export HOME="$SUITE_SANDBOX/home" export XDG_CONFIG_HOME="$SUITE_SANDBOX/xdg" export XDG_DATA_HOME="$SUITE_SANDBOX/data" @@ -142,10 +160,10 @@ SHELLCHECK_TARGETS=( ) # Add every scripts/*.sh except this one (don't recurse into our own tests). while IFS= read -r -d '' f; do - [ "$f" = "${BASH_SOURCE[0]}" ] && continue + [ "$f" = "$SELF_PATH" ] && continue SHELLCHECK_TARGETS+=("$f") done < <(find "$ROOT/scripts" -maxdepth 1 -name '*.sh' -type f -print0 2>/dev/null) -SHELLCHECK_TARGETS+=("${BASH_SOURCE[0]}") +SHELLCHECK_TARGETS+=("$SELF_PATH") if command -v shellcheck >/dev/null 2>&1; then sc_out=$(shellcheck --severity=error "${SHELLCHECK_TARGETS[@]}" 2>&1) || sc_rc=$? @@ -636,15 +654,35 @@ test_entrypoint_host_info_json "entrypoint HOST_INFO JSON shape (populated + unk # preserve the user's channel across upgrades. test_install_sh_bakes_source_url() { local label="$1" - local tmp dest_dir dest_path src_url out rc + local tmp dest_dir dest_path bad_dest src_url src_sha out rc tmp=$(mktemp -d -t lucebox-install.XXXXXX) # Use the real lucebox.sh as the "remote" file — `file://` works with # curl out of the box and exercises the same install.sh code path as # an https fetch would. src_url="file://$SCRIPT" + if command -v sha256sum >/dev/null 2>&1; then + src_sha=$(sha256sum "$SCRIPT") + else + src_sha=$(shasum -a 256 "$SCRIPT") + fi + src_sha="${src_sha%% *}" dest_dir="$tmp/bin" dest_path="$dest_dir/lucebox" + bad_dest="$dest_dir/bad-checksum" + + out=$(LUCEBOX_INSTALL_URL="$src_url" LUCEBOX_INSTALL_DEST="$bad_dest" \ + LUCEBOX_WRAPPER_SHA256="$(printf '0%.0s' {1..64})" \ + NO_COLOR=1 bash "$INSTALLER" 2>&1) && rc=0 || rc=$? + if [ "$rc" -eq 0 ] || [ -e "$bad_dest" ] \ + || ! grep -qF "wrapper checksum mismatch" <<<"$out"; then + rm -rf "$tmp" + report fail "$label" "bad wrapper checksum was not rejected" + return + fi + + rc=0 out=$(LUCEBOX_INSTALL_URL="$src_url" LUCEBOX_INSTALL_DEST="$dest_path" \ + LUCEBOX_WRAPPER_SHA256="$src_sha" \ NO_COLOR=1 bash "$INSTALLER" 2>&1) || rc=$? rc="${rc:-0}" if [ "$rc" -ne 0 ]; then @@ -662,15 +700,20 @@ test_install_sh_bakes_source_url() { report fail "$label" "LUCEBOX_INSTALLED_FROM not rewritten in installed copy" return fi + if ! grep -qF "wrapper sha256 verified" <<<"$out"; then + rm -rf "$tmp" + report fail "$label" "installer did not report checksum verification" + return + fi rm -rf "$tmp" report ok "$label" } test_install_sh_bakes_source_url "install.sh bakes LUCEBOX_INSTALLED_FROM into installed copy" # ── update dispatch ─────────────────────────────────────────────────────── -# `lucebox update` must dispatch to cmd_update — verify it's wired in the -# main case statement and appears in --help. We can't actually run the -# update (it'd curl + replace this very script) so the test is parse-level. +# `lucebox update` must dispatch to cmd_update — first verify it's wired in +# the main case statement and appears in --help, then exercise an isolated +# wrapper copy against a file:// channel below. test_update_subcommand_wired() { local label="$1" local out @@ -687,6 +730,65 @@ test_update_subcommand_wired() { } test_update_subcommand_wired "lucebox update subcommand is wired" +test_update_downloads_verifies_and_replaces_atomically() { + local label="$1" tmp installed source_url checksum out rc + tmp=$(mktemp -d -t lucebox-update.XXXXXX) + installed="$tmp/lucebox" + mkdir -p "$tmp/channel" + + # Bake a file:// channel into an isolated wrapper copy. The update must + # replace only that copy, never the repository script. + source_url="file://$tmp/channel/lucebox.sh" + sed "s|^LUCEBOX_INSTALLED_FROM=.*|LUCEBOX_INSTALLED_FROM=\"$source_url\"|" \ + "$SCRIPT" > "$installed" + chmod +x "$installed" + cat > "$tmp/channel/lucebox.sh" <<'WRAPPER' +#!/usr/bin/env bash +set -euo pipefail +VERSION="9.9.9" +LUCEBOX_INSTALLED_FROM="${LUCEBOX_INSTALLED_FROM:-https://example.invalid/lucebox.sh}" +printf 'updated %s from %s\n' "$VERSION" "$LUCEBOX_INSTALLED_FROM" +WRAPPER + + # A wrong explicit pin must leave the installed wrapper untouched. + out=$(LUCEBOX_WRAPPER_SHA256="$(printf '0%.0s' {1..64})" \ + NO_COLOR=1 bash "$installed" update 2>&1) && rc=0 || rc=$? + if [ "$rc" -eq 0 ] || ! grep -qF "wrapper checksum mismatch" <<<"$out" \ + || grep -qF 'VERSION="9.9.9"' "$installed"; then + rm -rf "$tmp" + report fail "$label" "bad checksum was not rejected: $(head -3 <<<"$out")" + return + fi + + if command -v sha256sum >/dev/null 2>&1; then + checksum=$(sha256sum "$tmp/channel/lucebox.sh") + else + checksum=$(shasum -a 256 "$tmp/channel/lucebox.sh") + fi + checksum="${checksum%% *}" + rc=0 + out=$(LUCEBOX_WRAPPER_SHA256="$checksum" \ + NO_COLOR=1 bash "$installed" update 2>&1) || rc=$? + rc="${rc:-0}" + if [ "$rc" -ne 0 ]; then + rm -rf "$tmp" + report fail "$label" "verified update exited $rc: $(head -3 <<<"$out")" + return + fi + if ! grep -qF 'VERSION="9.9.9"' "$installed" \ + || ! grep -Fqx "LUCEBOX_INSTALLED_FROM=\"$source_url\"" "$installed" \ + || ! grep -qF "wrapper sha256 verified" <<<"$out"; then + rm -rf "$tmp" + report fail "$label" "validated wrapper did not replace the isolated copy" + return + fi + + rm -rf "$tmp" + report ok "$label" +} +test_update_downloads_verifies_and_replaces_atomically \ + "lucebox update verifies and atomically replaces the wrapper" + # ── IMAGE_BASE derived from install source ──────────────────────────────── # Source lucebox.sh in a subshell with LUCEBOX_INSTALLED_FROM pointing at # various URLs, then check that IMAGE_BASE comes out right. Uses @@ -929,6 +1031,11 @@ case "\$1" in exit 0 ;; *) + if [ "\${DOCKER_FAKE_CONFIG_ERROR:-0}" = "1" ] \ + && [[ "\$*" == *"print-serve-argv"* ]]; then + echo "Invalid configuration: test fixture" >&2 + exit 2 + fi printf 'DOCKER_INVOKED' for a in "\$@"; do printf ' %q' "\$a"; done printf '\n' @@ -1070,14 +1177,39 @@ test_serve_fallback_forwards_config_env() { _run_wrapper_capture_docker "$sandbox" serve || true) rm -rf "$sandbox" "$config_home" if ! grep -qF "LUCEBOX_HOME=$config_home" <<<"$out" \ - || ! grep -qF "HOME=$sandbox" <<<"$out"; then + || ! grep -qF "HOME=$config_home" <<<"$out"; then report fail "$label" "fallback server omitted HOME/config env: $(tail -3 <<<"$out")" return fi + if grep -qF -- "-v $sandbox:$sandbox" <<<"$out"; then + report fail "$label" "fallback server exposed the full host HOME" + return + fi report ok "$label" } test_serve_fallback_forwards_config_env \ - "serve fallback forwards HOME + LUCEBOX_HOME" + "serve fallback isolates HOME and forwards LUCEBOX_HOME" + +test_serve_refuses_invalid_config_fallback() { + local label="$1" sandbox out + sandbox=$(mktemp -d -t lucebox-serve-invalid.XXXXXX) + _make_docker_shim "$sandbox" 0 + out=$(DOCKER_FAKE_CONFIG_ERROR=1 \ + _run_wrapper_capture_docker "$sandbox" serve || true) + rm -rf "$sandbox" + if ! grep -qF "Invalid configuration: test fixture" <<<"$out" \ + || ! grep -qF "refusing to ignore invalid Lucebox configuration" <<<"$out"; then + report fail "$label" "configuration error was not surfaced: $(tail -4 <<<"$out")" + return + fi + if grep -qF "using fallback" <<<"$out"; then + report fail "$label" "invalid configuration silently launched fallback defaults" + return + fi + report ok "$label" +} +test_serve_refuses_invalid_config_fallback \ + "serve refuses to replace invalid config with fallback defaults" test_run_route_preserves_tty() { local label="$1" sandbox out rc From ed68176ea1f509670154dca67a3f202989245f56 Mon Sep 17 00:00:00 2001 From: mrciffa <49000955+davide221@users.noreply.github.com> Date: Mon, 27 Jul 2026 16:25:43 +0200 Subject: [PATCH 6/8] test(lucebox): validate speculator symlinks --- lucebox/src/lucebox/docker_run.py | 5 +++- lucebox/tests/test_docker_run.py | 44 +++++++++++++++++++++++++++++++ 2 files changed, 48 insertions(+), 1 deletion(-) diff --git a/lucebox/src/lucebox/docker_run.py b/lucebox/src/lucebox/docker_run.py index f9b618007..c8c827f35 100644 --- a/lucebox/src/lucebox/docker_run.py +++ b/lucebox/src/lucebox/docker_run.py @@ -101,7 +101,10 @@ def _resolve_model_files(cfg: Config) -> tuple[str, str, str]: draft = pres.draft_file if not draft and pres.speculator_dir: spec_path = cfg.models_dir / "draft" / pres.speculator_dir - if spec_path.is_dir() or spec_path.is_symlink(): + # is_dir() follows valid directory symlinks. It also rejects + # dangling links and links to files, which cannot satisfy the + # speculator-directory contract. + if spec_path.is_dir(): draft_dir = pres.speculator_dir return target, draft, draft_dir diff --git a/lucebox/tests/test_docker_run.py b/lucebox/tests/test_docker_run.py index 9e693934f..257b369f8 100644 --- a/lucebox/tests/test_docker_run.py +++ b/lucebox/tests/test_docker_run.py @@ -205,6 +205,25 @@ def test_resolve_model_files_no_preset_no_override(tmp_path: Path) -> None: assert docker_run._resolve_model_files(cfg) == ("", "", "") +@pytest.mark.parametrize("invalid_target", ["file", "missing"]) +def test_resolve_model_files_ignores_invalid_speculator_symlink( + invalid_target: str, tmp_path: Path +) -> None: + draft_root = tmp_path / "draft" + draft_root.mkdir() + target = tmp_path / "external-speculator" + if invalid_target == "file": + target.write_bytes(b"not a directory") + (draft_root / "laguna-xs2-speculator").symlink_to( + target, target_is_directory=invalid_target == "missing" + ) + cfg = Config(models_dir=tmp_path, model=ModelMeta(preset="laguna-xs.2")) + + _, _, draft_dir = docker_run._resolve_model_files(cfg) + + assert draft_dir == "" + + # ── server_run_spec ────────────────────────────────────────────────────────── @@ -281,6 +300,31 @@ def test_server_run_spec_resolves_symlinked_model_parent(tmp_path: Path) -> None assert _env(spec)["DFLASH_TARGET"] == "/opt/lucebox-resolved/target/nested.gguf" +def test_server_run_spec_mounts_symlinked_speculator_directory_read_only( + tmp_path: Path, +) -> None: + models = tmp_path / "models" + draft_root = models / "draft" + external = tmp_path / "external-speculator" + draft_root.mkdir(parents=True) + external.mkdir() + (external / "model.safetensors").write_bytes(b"speculator") + (draft_root / "laguna-xs2-speculator").symlink_to( + external, target_is_directory=True + ) + + spec = docker_run.server_run_spec( + Config(models_dir=models, model=ModelMeta(preset="laguna-xs.2")) + ) + + assert docker_run.BindMount( + str(external), + "/opt/lucebox-resolved/draft-dir", + read_only=True, + ) in spec.volumes + assert _env(spec)["DFLASH_DRAFT"] == "/opt/lucebox-resolved/draft-dir" + + def test_server_run_spec_rejects_model_path_traversal(tmp_path: Path) -> None: cfg = Config(models_dir=tmp_path, model=ModelMeta(target_file="../secret.gguf")) From a55a7ef268715bf0c1613b26f3f3df70cc672224 Mon Sep 17 00:00:00 2001 From: mrciffa <49000955+davide221@users.noreply.github.com> Date: Mon, 27 Jul 2026 17:04:38 +0200 Subject: [PATCH 7/8] refactor(lucebox): tighten package edge cases --- install.sh | 8 +++++--- lucebox/src/lucebox/docker_run.py | 12 +++++++++--- lucebox/tests/test_config_cli.py | 2 +- lucebox/tests/test_docker_run.py | 20 ++++++++++++++++++++ 4 files changed, 35 insertions(+), 7 deletions(-) diff --git a/install.sh b/install.sh index d67dbd9b4..545d64c8f 100755 --- a/install.sh +++ b/install.sh @@ -37,13 +37,16 @@ die() { printf '%s[install] ✗%s %s\n' "$C_ERR" "$C_RST" "$*" >&2; exit 1; } command -v curl >/dev/null 2>&1 || die "curl is required (apt-get install curl)" sha256_file() { + local sum if command -v sha256sum >/dev/null 2>&1; then - sha256sum "$1" | awk '{print $1}' + sum=$(sha256sum "$1") elif command -v shasum >/dev/null 2>&1; then - shasum -a 256 "$1" | awk '{print $1}' + sum=$(shasum -a 256 "$1") else die "checksum requested, but neither sha256sum nor shasum is installed" fi + sum="${sum%% *}" + printf '%s' "$sum" | tr '[:upper:]' '[:lower:]' } # ── decide what gets baked in as the persisted channel ─────────────────── @@ -85,7 +88,6 @@ if [ -n "$expected_sha" ]; then [[ "$expected_sha" =~ ^[0-9a-fA-F]{64}$ ]] \ || die "LUCEBOX_WRAPPER_SHA256 must be exactly 64 hexadecimal characters" actual_sha=$(sha256_file "$tmp") - actual_sha=$(printf '%s' "$actual_sha" | tr '[:upper:]' '[:lower:]') expected_sha=$(printf '%s' "$expected_sha" | tr '[:upper:]' '[:lower:]') [ "$actual_sha" = "$expected_sha" ] \ || die "wrapper checksum mismatch (expected $expected_sha, got $actual_sha)" diff --git a/lucebox/src/lucebox/docker_run.py b/lucebox/src/lucebox/docker_run.py index c8c827f35..1568fce02 100644 --- a/lucebox/src/lucebox/docker_run.py +++ b/lucebox/src/lucebox/docker_run.py @@ -153,9 +153,15 @@ def _selected_model_path( if under_draft: container_base /= "draft" canonical = str(container_base.joinpath(*relative.parts)) - lexical = host_path.absolute() resolved = host_path.resolve(strict=False) - if resolved == lexical: + # A symlink in models_dir or one of its ancestors is covered by the root bind. + # Resolve that root before comparing so only symlinks *within* the selected + # model path need a separate narrow mount. + expected = cfg.models_dir.resolve(strict=False) + if under_draft: + expected /= "draft" + expected = expected.joinpath(*relative.parts) + if resolved == expected: return canonical mount_target = f"{_CONTAINER_RESOLVED_MODELS}/{role}" @@ -308,7 +314,7 @@ def server_run_spec(cfg: Config) -> DockerRunSpec: draft_path = _selected_model_path( cfg, draft_dir, - field="model.speculator_dir", + field="preset.speculator_dir", role="draft-dir", under_draft=True, directory=True, diff --git a/lucebox/tests/test_config_cli.py b/lucebox/tests/test_config_cli.py index 8d36da760..a9a727723 100644 --- a/lucebox/tests/test_config_cli.py +++ b/lucebox/tests/test_config_cli.py @@ -149,5 +149,5 @@ def test_print_run_reports_invalid_configuration_without_traceback( result = CliRunner().invoke(app, ["print-run"]) assert result.exit_code == 2 - assert "Invalid configuration" in result.output + assert "Invalid configuration" in result.stderr assert "Traceback" not in result.output diff --git a/lucebox/tests/test_docker_run.py b/lucebox/tests/test_docker_run.py index 257b369f8..7753b8e5a 100644 --- a/lucebox/tests/test_docker_run.py +++ b/lucebox/tests/test_docker_run.py @@ -300,6 +300,26 @@ def test_server_run_spec_resolves_symlinked_model_parent(tmp_path: Path) -> None assert _env(spec)["DFLASH_TARGET"] == "/opt/lucebox-resolved/target/nested.gguf" +def test_server_run_spec_does_not_remount_file_for_symlinked_models_dir( + tmp_path: Path, +) -> None: + actual_models = tmp_path / "actual-models" + models = tmp_path / "models" + actual_models.mkdir() + (actual_models / "target.gguf").write_bytes(b"model") + models.symlink_to(actual_models, target_is_directory=True) + + spec = docker_run.server_run_spec( + Config(models_dir=models, model=ModelMeta(target_file="target.gguf")) + ) + + assert _env(spec)["DFLASH_TARGET"] == "/opt/lucebox-hub/server/models/target.gguf" + assert all( + mount.target != "/opt/lucebox-resolved/target/target.gguf" + for mount in spec.volumes + ) + + def test_server_run_spec_mounts_symlinked_speculator_directory_read_only( tmp_path: Path, ) -> None: From 875838c5532a2f0856ddb1a84967f61c48f38669 Mon Sep 17 00:00:00 2001 From: mrciffa <49000955+davide221@users.noreply.github.com> Date: Tue, 28 Jul 2026 16:14:14 +0200 Subject: [PATCH 8/8] feat(lucebox): add guided inference menu --- README.md | 29 ++ install.sh | 12 +- lucebox.sh | 654 +++++++++++++++++++++++++++++-- lucebox/README.md | 10 +- lucebox/src/lucebox/cli.py | 212 +++++++++- lucebox/src/lucebox/config.py | 22 +- lucebox/tests/test_cli.py | 11 +- lucebox/tests/test_models_cli.py | 63 +++ scripts/test_lucebox_sh.sh | 73 +++- server/scripts/entrypoint.sh | 6 +- 10 files changed, 1029 insertions(+), 63 deletions(-) diff --git a/README.md b/README.md index bc524efde..8cbab30cd 100644 --- a/README.md +++ b/README.md @@ -101,6 +101,35 @@ Reference target: **RTX 3090 (Ampere sm_86)** — all headline numbers. Other NV `server/` (DFlash) builds with CMake 3.18+ and vendors the required `ggml` sources directly; only `Block-Sparse-Attention` remains a git submodule. No PyTorch is needed for `server/`. `optimizations/megakernel/` is the only component requiring PyTorch 2.0+ (CUDAExtension links against torch C++ libs). Power-tune: `sudo nvidia-smi -pl 220` (3090 sweet spot, re-sweep for other cards). +## Lucebox CLI + +The easiest path for both Lucebox buyers and open-source users is one command: + +```bash +# Buyers receive this preinstalled. Other users install the small host wrapper: +curl -fsSL https://raw.githubusercontent.com/Luce-Org/lucebox/main/install.sh | bash + +lucebox +``` + +`lucebox` opens the branded menu. **Quick setup** detects NVIDIA CUDA or AMD +ROCm, lets you choose and download a model, applies safe hardware-aware +optimizations, and starts the inference service. Wi-Fi and device provisioning +are intentionally outside this inference-only CLI. + +Contributors can open the same menu from a repository checkout with +`./lucebox.sh`. Its **Developer tools** can build and run the native C++ engine +or launch Claude Code, Codex, OpenCode, Hermes, Pi, OpenClaw, and Open WebUI +harnesses. Every menu action also has a scriptable command, for example: + +```bash +lucebox models select +lucebox optimize --yes +lucebox start +./lucebox.sh build cuda +./lucebox.sh native cuda +``` + ## Quick Start On Harnesses [`harness/`](harness/) contains RTX 3090 client launchers and regression tests diff --git a/install.sh b/install.sh index 545d64c8f..1886bae18 100755 --- a/install.sh +++ b/install.sh @@ -3,7 +3,7 @@ # # Canonical install (Luce-Org main, stable channel): # -# curl -fsSL https://raw.githubusercontent.com/Luce-Org/lucebox-hub/main/install.sh | bash +# curl -fsSL https://raw.githubusercontent.com/Luce-Org/lucebox/main/install.sh | bash # # Install from a different fork / branch (dev channel). Note the env var # is on the `bash` side of the pipe — `VAR=val curl … | bash` would attach @@ -22,7 +22,7 @@ set -euo pipefail -LUCEBOX_INSTALL_URL="${LUCEBOX_INSTALL_URL:-https://raw.githubusercontent.com/Luce-Org/lucebox-hub/main/lucebox.sh}" +LUCEBOX_INSTALL_URL="${LUCEBOX_INSTALL_URL:-https://raw.githubusercontent.com/Luce-Org/lucebox/main/lucebox.sh}" DEST="${LUCEBOX_INSTALL_DEST:-$HOME/.local/bin/lucebox}" # ── helpers ─────────────────────────────────────────────────────────────── @@ -143,8 +143,10 @@ esac cat <&2; } hint() { printf ' %b%s%b\n' "$C_DIM" "$*" "$C_RST"; } die() { err "$*"; exit 1; } +print_logo() { + printf '%b' "$C_BRAND" + cat <<'EOF' + · ╱ + ·──✦──· █ █ █ ▄▀▀ █▀▀ █▀▀▄ ▄▀▀▄ █ █ + ╱ · █ █ █ █ █▀▀ █▀▀▄ █ █ █ + · ▀▀ ▀▀ ▀▀ ▀▀▀ ▀▀▀ ▀▀ ▀ ▀ +EOF + printf '%b local inference, made simple%b\n\n' "$C_DIM" "$C_RST" +} + +# Find a source checkout for contributor-only actions. An explicit path wins; +# otherwise inspect the current directory and the wrapper's own directory, +# walking upward until the repository markers are found. Buyer installs simply +# return no path and never see build/harness actions. +_find_repo_root() { + local candidate="${LUCEBOX_REPO:-}" dir + if [ -n "$candidate" ]; then + if [ -f "$candidate/server/CMakeLists.txt" ] && [ -d "$candidate/harness" ]; then + (cd "$candidate" && pwd) + return 0 + fi + return 1 + fi + + for candidate in "$PWD" "$(dirname "$SCRIPT_PATH")"; do + dir="$candidate" + while [ "$dir" != "/" ] && [ -n "$dir" ]; do + if [ -f "$dir/server/CMakeLists.txt" ] && [ -d "$dir/harness" ]; then + (cd "$dir" && pwd) + return 0 + fi + dir="$(dirname "$dir")" + done + done + return 1 +} + +_confirm() { + # usage: _confirm "question" [default_yes] + local question="$1" default_yes="${2:-1}" answer prompt + if [ "$default_yes" = "1" ]; then prompt="Y/n"; else prompt="y/N"; fi + printf '%s [%s] ' "$question" "$prompt" + IFS= read -r answer || return 1 + case "$answer" in + y|Y|yes|YES|Yes) return 0 ;; + n|N|no|NO|No) return 1 ;; + "") [ "$default_yes" = "1" ] ;; + *) return 1 ;; + esac +} + sha256_file() { local sum if command -v sha256sum >/dev/null 2>&1; then @@ -1141,13 +1200,46 @@ cmd_systemctl_passthrough() { } cmd_logs() { - require_systemd "logs" - # Pure passthrough: any flags the user wants (-f, -n, --since, ...) go - # straight to journalctl. Default is follow. - if [ $# -eq 0 ]; then - exec journalctl --user -u "$UNIT_NAME" -f + ensure_probed + if [ "$LUCEBOX_HOST_HAS_SYSTEMD" = "1" ] && [ -f "$UNIT_PATH" ]; then + # Pure passthrough: any flags the user wants (-f, -n, --since, ...) + # go straight to journalctl. Default is follow. + if [ $# -eq 0 ]; then + exec journalctl --user -u "$UNIT_NAME" -f + fi + exec journalctl --user -u "$UNIT_NAME" "$@" fi - exec journalctl --user -u "$UNIT_NAME" "$@" + if _lucebox_container_running; then + if [ $# -eq 0 ]; then + exec docker logs -f "$CONTAINER_NAME" + fi + exec docker logs "$@" "$CONTAINER_NAME" + fi + die "the inference engine is not running — use '$SCRIPT_NAME start' or '$SCRIPT_NAME serve'" +} + +cmd_status() { + ensure_probed + if [ "$LUCEBOX_HOST_HAS_SYSTEMD" = "1" ] && [ -f "$UNIT_PATH" ]; then + exec systemctl --user status "$UNIT_NAME" --no-pager + fi + if _lucebox_container_running; then + exec docker ps --filter "name=^${CONTAINER_NAME}\$" \ + --format 'table {{.Names}}\t{{.Status}}\t{{.Ports}}' + fi + info "Lucebox inference engine is stopped" + hint "Run '$SCRIPT_NAME setup' for first-time setup, or '$SCRIPT_NAME serve' in the foreground." +} + +cmd_stop() { + ensure_probed + if [ "$LUCEBOX_HOST_HAS_SYSTEMD" = "1" ] && [ -f "$UNIT_PATH" ]; then + exec systemctl --user stop "$UNIT_NAME" + fi + if _lucebox_container_running; then + exec docker stop "$CONTAINER_NAME" + fi + ok "Lucebox inference engine is already stopped" } cmd_pull() { @@ -1162,6 +1254,251 @@ cmd_pull() { exec docker pull "${IMAGE_BASE}:${variant}" } +# ── contributor-native workflow ─────────────────────────────────────────── + +_native_backend() { + local requested="${1:-}" variant + case "$requested" in + cuda|cuda12|nvidia) printf 'cuda'; return ;; + rocm|hip|amd) printf 'rocm'; return ;; + "") + variant=$(pick_variant) + if _variant_is_rocm "$variant"; then printf 'rocm'; else printf 'cuda'; fi + return + ;; + *) die "unknown backend '$requested' — choose cuda or rocm" ;; + esac +} + +_native_build_dir() { + local repo="$1" backend="$2" + if [ -n "${LUCEBOX_BUILD_DIR:-}" ]; then + printf '%s' "$LUCEBOX_BUILD_DIR" + elif [ "$backend" = "rocm" ]; then + printf '%s/server/build-hip' "$repo" + else + printf '%s/server/build-cuda' "$repo" + fi +} + +_safe_model_relative_path() { + local value="$1" part parts=() + [ -n "$value" ] || return 1 + [[ "$value" != /* ]] || return 1 + IFS='/' read -r -a parts <<<"$value" + for part in "${parts[@]}"; do + [ "$part" != ".." ] || return 1 + done + return 0 +} + +_selected_model_paths() { + # Emit target path, draft path (or "none"), and model id on separate lines. + # `models select` persists the filenames, so preset mapping is only a + # compatibility fallback for hand-written config files. + local preset target_file draft_file target_path draft_path="none" + preset=$(_lucebox_config_get model.preset) + target_file=$(_lucebox_config_get model.target_file) + draft_file=$(_lucebox_config_get model.draft_file) + if [ -z "$target_file" ]; then + case "$preset" in + qwen3.6-27b) target_file="Qwen3.6-27B-Q4_K_M.gguf" ;; + gemma-4-26b) target_file="google_gemma-4-26B-A4B-it-Q4_K_M.gguf" ;; + gemma-4-31b) target_file="google_gemma-4-31B-it-Q4_K_M.gguf" ;; + laguna-xs.2) target_file="laguna-xs2-Q4_K_M.gguf" ;; + qwen3.6-moe) target_file="Qwen3.6-35B-A3B-UD-Q4_K_M.gguf" ;; + esac + fi + if [ -z "$draft_file" ]; then + case "$preset" in + qwen3.6-27b) draft_file="dflash-draft-3.6-q4_k_m.gguf" ;; + gemma-4-26b) draft_file="gemma-4-26B-A4B-it-DFlash-q8_0.gguf" ;; + gemma-4-31b) draft_file="gemma-4-31B-it-DFlash-q8_0.gguf" ;; + esac + fi + _safe_model_relative_path "$target_file" \ + || die "no valid model is selected — run '$SCRIPT_NAME models select' first" + target_path="$DEFAULT_MODELS_DIR/$target_file" + if [ -n "$draft_file" ]; then + _safe_model_relative_path "$draft_file" \ + || die "invalid model.draft_file in $(_lucebox_config_path)" + draft_path="$DEFAULT_MODELS_DIR/draft/$draft_file" + elif [ "$preset" = "laguna-xs.2" ] \ + && [ -d "$DEFAULT_MODELS_DIR/draft/laguna-xs2-speculator" ]; then + draft_path="$DEFAULT_MODELS_DIR/draft/laguna-xs2-speculator" + fi + printf '%s\n%s\n%s\n' "$target_path" "$draft_path" "${preset:-lucebox}" +} + +_export_native_config() { + local key env_name value + while IFS='|' read -r key env_name; do + [ -n "$key" ] || continue + value=$(_lucebox_config_get "$key") + [ -n "$value" ] || continue + case "$key" in + dflash.lazy|dflash.debug_thinking_logits) + case "$value" in + true|1|yes|on) value=1 ;; + *) value=0 ;; + esac + ;; + esac + export "$env_name=$value" + done <<'EOF' +dflash.budget|DFLASH_BUDGET +dflash.max_ctx|DFLASH_MAX_CTX +dflash.lazy|DFLASH_LAZY +dflash.prefix_cache_slots|DFLASH_PREFIX_CACHE_SLOTS +dflash.prefill_cache_slots|DFLASH_PREFILL_CACHE_SLOTS +dflash.cache_type_k|DFLASH_CACHE_TYPE_K +dflash.cache_type_v|DFLASH_CACHE_TYPE_V +dflash.prefill_mode|DFLASH_PREFILL_MODE +dflash.prefill_keep_ratio|DFLASH_PREFILL_KEEP +dflash.prefill_threshold|DFLASH_PREFILL_THRESHOLD +dflash.prefill_drafter|DFLASH_PREFILL_DRAFTER +dflash.think_max|DFLASH_THINK_MAX +dflash.fa_window|DFLASH_FA_WINDOW +dflash.think_soft_close_min_ratio|DFLASH_THINK_SOFT_CLOSE_MIN_RATIO +dflash.debug_thinking_logits|DFLASH_DEBUG_THINKING_LOGITS +EOF +} + +cmd_native_build() { + local repo backend build_dir jobs hip_wmma=OFF + repo=$(_find_repo_root) \ + || die "native build requires a lucebox repository checkout (cd into it or set LUCEBOX_REPO)" + ensure_probed + backend=$(_native_backend "${1:-}") + build_dir=$(_native_build_dir "$repo" "$backend") + command -v cmake >/dev/null 2>&1 || die "cmake is required to build the inference engine" + [ -f "$repo/server/deps/llama.cpp/ggml/CMakeLists.txt" ] \ + || die "git submodules are missing — run: git -C '$repo' submodule update --init --recursive" + + local configure=(cmake -S "$repo/server" -B "$build_dir" -DCMAKE_BUILD_TYPE=Release) + if [ "$backend" = "rocm" ]; then + [ "$LUCEBOX_HOST_HAS_AMD_GPU" = "1" ] \ + || die "ROCm native build selected but no AMD GPU was detected" + if [ -f /opt/rocm/include/rocwmma/rocwmma.hpp ] \ + || [ -f /usr/include/rocwmma/rocwmma.hpp ]; then + hip_wmma=ON + fi + configure+=( + -DDFLASH27B_GPU_BACKEND=hip + "-DDFLASH27B_HIP_ARCHITECTURES=${LUCEBOX_HOST_AMD_GPU_ARCH:-gfx1151}" + "-DDFLASH27B_HIP_SM80_EQUIV=$hip_wmma" + ) + else + [ "$LUCEBOX_HOST_HAS_NVIDIA_GPU" = "1" ] \ + || die "CUDA native build selected but no NVIDIA GPU was detected" + configure+=(-DDFLASH27B_GPU_BACKEND=cuda) + if [[ "$LUCEBOX_HOST_GPU_SM" =~ ^[0-9]+$ ]]; then + configure+=("-DCMAKE_CUDA_ARCHITECTURES=$LUCEBOX_HOST_GPU_SM") + fi + fi + info "Configuring native $backend build in $build_dir" + "${configure[@]}" + jobs="${LUCEBOX_HOST_NPROC:-1}" + [ "$jobs" -gt 0 ] 2>/dev/null || jobs=1 + info "Building dflash_server ($jobs jobs)" + cmake --build "$build_dir" --target dflash_server -j "$jobs" + ok "Native engine ready: $build_dir/dflash_server" +} + +cmd_native_serve() { + local repo backend build_dir binary selected=() target draft model_id + repo=$(_find_repo_root) \ + || die "native run requires a lucebox repository checkout (cd into it or set LUCEBOX_REPO)" + ensure_probed + backend=$(_native_backend "${1:-}") + build_dir=$(_native_build_dir "$repo" "$backend") + binary="$build_dir/dflash_server" + [ -x "$binary" ] \ + || die "native engine is not built — run '$SCRIPT_NAME build $backend' first" + mapfile -t selected < <(_selected_model_paths) + target="${selected[0]:-}" + draft="${selected[1]:-none}" + model_id="${selected[2]:-lucebox}" + [ -f "$target" ] \ + || die "selected target is not installed: $target — run '$SCRIPT_NAME models select'" + if [ "$draft" != "none" ] && [ ! -e "$draft" ]; then + die "selected draft is not installed: $draft — run '$SCRIPT_NAME models select'" + fi + + _export_native_config + export DFLASH_DIR="$repo/server" + export DFLASH_SERVER_BIN="$binary" + export DFLASH_TARGET="$target" + if [ "$draft" = "none" ]; then + export DFLASH_DRAFT="$DEFAULT_MODELS_DIR/.lucebox-no-draft" + else + export DFLASH_DRAFT="$draft" + fi + export DFLASH_HOST="${LUCEBOX_NATIVE_HOST:-127.0.0.1}" + export DFLASH_PORT="$DEFAULT_PORT" + export DFLASH_MODEL_NAME="$model_id" + export LUCEBOX_NATIVE=1 + info "Starting native $backend engine at http://$DFLASH_HOST:$DFLASH_PORT" + exec "$repo/server/scripts/entrypoint.sh" serve +} + +cmd_harness() { + local repo name="${1:-}" backend build_dir binary script selected=() target draft + repo=$(_find_repo_root) \ + || die "harness launchers are contributor tools; run this command inside the lucebox repository" + if [ -z "$name" ]; then + cat <<'EOF' +Choose a harness: + 1 Claude Code + 2 Codex + 3 OpenCode + 4 Hermes + 5 Pi + 6 OpenClaw + 7 Open WebUI +EOF + printf 'Harness number: ' + IFS= read -r name || return 1 + fi + case "$name" in + 1|claude|claude-code) name=claude_code ;; + 2|codex) name=codex ;; + 3|opencode) name=opencode ;; + 4|hermes) name=hermes ;; + 5|pi) name=pi ;; + 6|openclaw) name=openclaw ;; + 7|openwebui|webui) name=openwebui ;; + *) die "unknown harness '$name'" ;; + esac + script="$repo/harness/clients/run_${name}.sh" + [ -x "$script" ] || die "harness launcher is missing: $script" + ensure_probed + backend=$(_native_backend "${LUCEBOX_HARNESS_BACKEND:-}") + build_dir=$(_native_build_dir "$repo" "$backend") + binary="$build_dir/dflash_server" + [ -x "$binary" ] \ + || die "native engine is not built — run '$SCRIPT_NAME build $backend' first" + mapfile -t selected < <(_selected_model_paths) + target="${selected[0]:-}" + draft="${selected[1]:-none}" + [ -f "$target" ] \ + || die "selected target is not installed: $target — run '$SCRIPT_NAME models select'" + + export REPO_DIR="$repo" + export DFLASH_SERVER_BIN="$binary" + export DFLASH_TARGET="$target" + export TARGET="$target" + export DFLASH_DRAFT="$draft" + export DRAFT="$draft" + local max_ctx budget + max_ctx=$(_lucebox_config_get dflash.max_ctx) + budget=$(_lucebox_config_get dflash.budget) + [ -z "$max_ctx" ] || export MAX_CTX="$max_ctx" + [ -z "$budget" ] || export BUDGET="$budget" + info "Launching $name with the selected Lucebox model" + exec "$script" +} + cmd_update() { # Download the wrapper itself as data; never execute a second remote # installer. Override the persisted channel with LUCEBOX_INSTALL_URL. @@ -1240,7 +1577,7 @@ cmd_completion() { # lucebox completion fish | source # # Keep this in sync with the dispatch table in main() and the sub-app - # verbs (config get/set/unset, models list/download). Adding a new + # verbs (config get/set/unset, models list/download/select). Adding a new # top-level command means adding it here too. local shell="${1:-}" case "$shell" in @@ -1253,11 +1590,11 @@ _lucebox_complete() { COMPREPLY=() cur="${COMP_WORDS[COMP_CWORD]}" prev="${COMP_WORDS[COMP_CWORD-1]}" - cmds="install uninstall start stop restart enable disable status logs \ - serve pull update check completion config models \ - print-run help version" + cmds="menu setup install uninstall start stop restart enable disable status logs \ + serve build native harness pull update check completion config models \ + optimize print-run help version" config_verbs="get set unset" - models_verbs="list download" + models_verbs="list download select" completion_shells="bash zsh fish" # Sub-app verbs / shell args. @@ -1295,14 +1632,14 @@ ZSH # lucebox fish completion. Source from ~/.config/fish/config.fish: # lucebox completion fish | source complete -c lucebox -f -set -l __lucebox_cmds install uninstall start stop restart enable disable \ - status logs serve pull update check completion config models \ - print-run help version +set -l __lucebox_cmds menu setup install uninstall start stop restart enable disable \ + status logs serve build native harness pull update check completion config models \ + optimize print-run help version for cmd in $__lucebox_cmds complete -c lucebox -n "not __fish_seen_subcommand_from $__lucebox_cmds" -a $cmd end complete -c lucebox -n "__fish_seen_subcommand_from config" -a "get set unset" -complete -c lucebox -n "__fish_seen_subcommand_from models" -a "list download" +complete -c lucebox -n "__fish_seen_subcommand_from models" -a "list download select" complete -c lucebox -n "__fish_seen_subcommand_from completion" -a "bash zsh fish" FISH ;; @@ -1534,7 +1871,7 @@ cmd_exec_in_container() { _lucebox_prefer_exec() { local cmd="$1"; shift case "$cmd" in - config|models|check|print-run|print-serve-argv) + config|models|optimize|check|print-run|print-serve-argv) return 0 ;; *) @@ -1567,9 +1904,250 @@ cmd_route_to_container() { cmd_in_container "$cmd" "$@" } +# ── interactive product surface ─────────────────────────────────────────── + +_engine_state() { + if command -v systemctl >/dev/null 2>&1 \ + && systemctl --user is-active --quiet "$UNIT_NAME" 2>/dev/null; then + printf 'running' + elif _lucebox_container_running; then + printf 'running' + else + printf 'stopped' + fi +} + +_menu_clear() { + if [ -t 1 ] && [ "${TERM:-dumb}" != "dumb" ] \ + && [ "${LUCEBOX_NO_CLEAR:-0}" != "1" ]; then + printf '\033[2J\033[H' + fi +} + +_menu_pause() { + [ -t 0 ] || return 0 + printf '\nPress Enter to return to the menu…' + IFS= read -r _ || true +} + +_menu_run() { + # Always invoke through bash: repository copies are not necessarily marked + # executable, while installed buyer copies are. This behaves identically + # in both places and lets exec-heavy subcommands return to the menu. + local rc + if bash "$SCRIPT_PATH" "$@"; then + return 0 + else + rc=$? + warn "Command failed (exit $rc)." + return "$rc" + fi +} + +_menu_start() { + ensure_probed + if [ "$LUCEBOX_HOST_HAS_SYSTEMD" = "1" ] && [ -f "$UNIT_PATH" ]; then + _menu_run start + return + fi + warn "The background service is not installed." + if _confirm "Run the Docker engine in this terminal now?" 1; then + _menu_run serve + else + hint "Run '$SCRIPT_NAME setup' to install the background service." + fi +} + +_menu_restart_if_running() { + [ "$(_engine_state)" = "running" ] || return 0 + if ! _confirm "Restart the running engine to apply this change?" 1; then + warn "The change is saved and will apply at the next restart." + return 0 + fi + if [ "$LUCEBOX_HOST_HAS_SYSTEMD" = "1" ] && [ -f "$UNIT_PATH" ]; then + _menu_run restart + else + _menu_run stop + warn "The foreground engine was stopped. Start it again from menu option 4." + fi +} + +cmd_setup() { + [ -t 0 ] && [ -t 1 ] \ + || die "guided setup needs a terminal — use '$SCRIPT_NAME --help' for non-interactive commands" + ensure_probed + _menu_clear + print_logo + printf '%bQuick setup%b\n\n' "$C_INFO" "$C_RST" + cmd_check + printf '\n' + + local variant + variant=$(pick_variant) + if [ "$LUCEBOX_HOST_HAS_NVIDIA_GPU" = "1" ] \ + && [ "$LUCEBOX_HOST_HAS_AMD_GPU" = "1" ]; then + printf 'This build has NVIDIA and AMD graphics. Which accelerator should run Lucebox?\n' + printf ' 1 NVIDIA / CUDA %b(recommended for RTX + Strix builds)%b\n' "$C_DIM" "$C_RST" + printf ' 2 AMD / ROCm\n' + printf 'Choice [1]: ' + local backend_choice + IFS= read -r backend_choice || return 1 + case "$backend_choice" in + 2) variant=rocm ;; + *) variant=cuda12 ;; + esac + fi + info "Selected backend: $variant" + + if docker image inspect "${IMAGE_BASE}:${variant}" >/dev/null 2>&1; then + ok "Inference image is already installed (${IMAGE_BASE}:${variant})" + else + _confirm "Download the ${variant} inference image now?" 1 \ + || { warn "Setup stopped before the image download."; return 1; } + LUCEBOX_VARIANT="$variant" bash "$SCRIPT_PATH" pull || return $? + fi + + # Persist the accelerator choice only after its image is available; the + # config writer lives in that image on buyer installations. + LUCEBOX_VARIANT="$variant" bash "$SCRIPT_PATH" config set "variant=$variant" \ + || return $? + + printf '\n' + bash "$SCRIPT_PATH" models select || return $? + if [ -z "$(_lucebox_config_get model.preset)" ]; then + warn "No model was selected; setup stopped without starting the engine." + return 1 + fi + + printf '\n' + if _confirm "Use automatic GPU optimization?" 1; then + bash "$SCRIPT_PATH" optimize --yes || return $? + fi + + if [ "$LUCEBOX_HOST_HAS_SYSTEMD" = "1" ]; then + if [ ! -f "$UNIT_PATH" ]; then + printf '\n' + if _confirm "Install Lucebox as a background service?" 1; then + bash "$SCRIPT_PATH" install || return $? + fi + fi + if [ -f "$UNIT_PATH" ] && _confirm "Start the inference engine now?" 1; then + if [ "$(_engine_state)" = "running" ]; then + bash "$SCRIPT_PATH" restart || return $? + else + bash "$SCRIPT_PATH" start || return $? + fi + fi + else + warn "Background services are unavailable on this host." + hint "Use '$SCRIPT_NAME serve' to run the engine in the foreground." + fi + + printf '\n' + ok "Lucebox is configured" + hint "API: http://127.0.0.1:$DEFAULT_PORT/v1" + hint "Menu: $SCRIPT_NAME" + hint "Status: $SCRIPT_NAME status" +} + +cmd_developer_menu() { + local repo choice + repo=$(_find_repo_root) \ + || { warn "Developer tools require a lucebox repository checkout."; return 1; } + while true; do + _menu_clear + print_logo + printf '%bDeveloper tools%b\n' "$C_INFO" "$C_RST" + printf 'Repository: %s\n\n' "$repo" + printf ' 1 Build the native inference engine\n' + printf ' 2 Run the native inference engine\n' + printf ' 3 Choose and run a client harness\n' + printf ' b Back\n\n' + printf 'Choose: ' + IFS= read -r choice || return 0 + case "$choice" in + 1) _menu_run build; _menu_pause ;; + 2) _menu_run native; _menu_pause ;; + 3) _menu_run harness; _menu_pause ;; + b|B|q|Q) return 0 ;; + *) warn "Choose 1–3 or b"; _menu_pause ;; + esac + done +} + +cmd_menu() { + local choice model variant state repo_hint + while true; do + ensure_probed + model=$(_lucebox_config_get model.preset) + model="${model:-not selected}" + variant=$(pick_variant) + state=$(_engine_state) + if _find_repo_root >/dev/null 2>&1; then + repo_hint="available" + else + repo_hint="not a source checkout" + fi + + _menu_clear + print_logo + printf ' GPU: %s\n' "${LUCEBOX_HOST_GPU_NAME:-${LUCEBOX_HOST_AMD_GPU_NAME:-not detected}}" + if [ "$LUCEBOX_HOST_HAS_NVIDIA_GPU" = "1" ] \ + && [ "$LUCEBOX_HOST_HAS_AMD_GPU" = "1" ]; then + printf ' Other GPU: %s\n' "${LUCEBOX_HOST_AMD_GPU_NAME:-AMD GPU}" + fi + printf ' Backend: %s\n' "$variant" + printf ' Model: %s\n' "$model" + printf ' Optimization: automatic\n' + printf ' Engine: %s\n\n' "$state" + + printf ' 1 Quick setup\n' + printf ' 2 Choose or download a model\n' + printf ' 3 Apply automatic optimization\n' + printf ' 4 Start the inference engine\n' + printf ' 5 Stop the inference engine\n' + printf ' 6 Status\n' + printf ' 7 Recent logs\n' + printf ' d Developer tools %b(%s)%b\n' "$C_DIM" "$repo_hint" "$C_RST" + printf ' q Quit\n\n' + printf 'Choose: ' + IFS= read -r choice || return 0 + case "$choice" in + 1) cmd_setup; _menu_pause ;; + 2) + if _menu_run models select; then _menu_restart_if_running; fi + _menu_pause + ;; + 3) + if _menu_run optimize; then _menu_restart_if_running; fi + _menu_pause + ;; + 4) _menu_start; _menu_pause ;; + 5) _menu_run stop; _menu_pause ;; + 6) _menu_run status; _menu_pause ;; + 7) + if [ "$LUCEBOX_HOST_HAS_SYSTEMD" = "1" ] && [ -f "$UNIT_PATH" ]; then + _menu_run logs -n 80 --no-pager + else + _menu_run logs --tail 80 + fi + _menu_pause + ;; + d|D) cmd_developer_menu ;; + q|Q|quit|exit) return 0 ;; + *) warn "Choose 1–7, d, or q"; _menu_pause ;; + esac + done +} + usage() { cat < print shell completion script (bash / zsh / fish) + models select numbered model picker; download + activate in one step models list / download / activate model presets + optimize apply safe hardware-aware inference defaults config read / write keys in .lucebox/config.toml print-run print the docker-run command for the server @@ -1633,22 +2219,42 @@ main() { *) args+=("$1"); shift ;; esac done - set -- "${args[@]}" + if [ "${#args[@]}" -gt 0 ]; then + set -- "${args[@]}" + else + set -- + fi - local cmd="${1:-help}" - [ $# -gt 0 ] && shift + local cmd + if [ $# -eq 0 ]; then + if [ -t 0 ] && [ -t 1 ]; then cmd=menu; else cmd=help; fi + else + cmd="$1" + shift + fi case "$cmd" in + # Branded interactive surface / guided first run. + menu) cmd_menu "$@" ;; + setup) cmd_setup "$@" ;; + # Systemd surface install) cmd_systemd_install "$@" ;; uninstall) cmd_systemd_uninstall "$@" ;; - start|stop|restart|enable|disable|status) + start|restart|enable|disable) cmd_systemctl_passthrough "$cmd" "$@" ;; + stop) cmd_stop "$@" ;; + status) cmd_status "$@" ;; logs) cmd_logs "$@" ;; # Direct server serve) cmd_serve "$@" ;; pull) cmd_pull "$@" ;; + # Native source-repository workflow. + build) cmd_native_build "$@" ;; + native) cmd_native_serve "$@" ;; + harness) cmd_harness "$@" ;; + # Self-update — re-runs the bootstrap installer against the channel # this script was installed from (LUCEBOX_INSTALLED_FROM). update) cmd_update "$@" ;; diff --git a/lucebox/README.md b/lucebox/README.md index f5bb7d635..6e5533fc6 100644 --- a/lucebox/README.md +++ b/lucebox/README.md @@ -5,9 +5,11 @@ users do not install it directly: they install the small [`lucebox` host wrapper](https://github.com/Luce-Org/lucebox/blob/main/lucebox.sh), which invokes the package in the appropriate container: + lucebox # branded interactive menu + lucebox setup # guided first run lucebox check - lucebox models list - lucebox models download qwen3.6-27b --activate + lucebox models select # numbered picker + download + activate + lucebox optimize # safe hardware-aware defaults lucebox start The wrapper detects the host and selects CUDA for NVIDIA builds (including RTX @@ -17,6 +19,10 @@ optimization settings, and construction of the final server command. Host facts are passed through `LUCEBOX_HOST_*` environment variables so the container never has to guess the host configuration. +Inside a source checkout, the same wrapper also exposes `lucebox build`, +`lucebox native`, and `lucebox harness` for contributors. Buyer installations +keep using the prebuilt container, so no compiler or source checkout is needed. + See the [project README](https://github.com/Luce-Org/lucebox#readme) for the installation and user flow. Contributors can find the CLI implementation in [`src/lucebox`](https://github.com/Luce-Org/lucebox/tree/main/lucebox/src/lucebox). diff --git a/lucebox/src/lucebox/cli.py b/lucebox/src/lucebox/cli.py index 622e6de5a..ff11ebf8c 100644 --- a/lucebox/src/lucebox/cli.py +++ b/lucebox/src/lucebox/cli.py @@ -4,8 +4,10 @@ doesn't intercept (everything outside the systemd surface) ends up here. Subcommand inventory: + (no command) — branded interactive menu check — readiness report config get/set/unset — read / write a single key in config.toml + optimize — apply the recommended hardware profile pull — docker pull the selected CUDA or ROCm image print-run — emit the docker-run command for the server print-serve-argv — same, raw argv lines (consumed by `lucebox serve`) @@ -23,6 +25,7 @@ from rich.markup import escape from rich.table import Table +import lucebox.autotune as autotune_mod import lucebox.config as config_mod import lucebox.docker_run as docker_run import lucebox.download as download_mod @@ -35,7 +38,7 @@ app = typer.Typer( name="lucebox", help="Host CLI for the lucebox-hub container. Invoked by lucebox.sh.", - no_args_is_help=True, + no_args_is_help=False, invoke_without_command=True, add_completion=False, ) @@ -45,6 +48,7 @@ @app.callback() def root_options( + ctx: typer.Context, version_flag: Annotated[ bool, typer.Option("--version", help="Print lucebox version and exit.", is_eager=True), @@ -54,11 +58,32 @@ def root_options( if version_flag: print(__version__) raise typer.Exit() + if ctx.invoked_subcommand is None: + if sys.stdin.isatty() and sys.stdout.isatty(): + _package_menu() + else: + _print_logo() + console.print(ctx.get_help()) + raise typer.Exit() # ── helpers ──────────────────────────────────────────────────────────────── +_LOGO = r""" + · ╱ + ·──✦──· █ █ █ ▄▀▀ █▀▀ █▀▀▄ ▄▀▀▄ █ █ + ╱ · █ █ █ █ █▀▀ █▀▀▄ █ █ █ + · ▀▀ ▀▀ ▀▀ ▀▀▀ ▀▀▀ ▀▀ ▀ ▀ +""".strip("\n") + + +def _print_logo() -> None: + """Render the compact Lucebox mark used by both interactive surfaces.""" + console.print(f"[bold gold1]{_LOGO}[/bold gold1]") + console.print("[dim] local inference, made simple[/dim]\n") + + def _load_or_build() -> Config: """env > config.toml > dataclass defaults — the canonical precedence. @@ -90,6 +115,43 @@ def _server_spec() -> docker_run.DockerRunSpec: raise typer.Exit(code=2) from exc +def _package_menu() -> None: + """Small in-package menu for direct installs and contributor workflows. + + Service lifecycle remains host-owned by ``lucebox.sh``. This menu covers + the package's own responsibilities and makes a direct ``python -m lucebox`` + invocation useful instead of dropping users into a wall of help text. + """ + while True: + _print_logo() + cfg = _load_or_build() + active = cfg.model.preset or "not selected" + console.print(f"Model: [bold]{escape(active)}[/bold]") + console.print("Optimization: [bold]Automatic[/bold] (safe defaults for this GPU)\n") + console.print(" [bold cyan]1[/bold cyan] Choose or download a model") + console.print(" [bold cyan]2[/bold cyan] Apply automatic optimization") + console.print(" [bold cyan]3[/bold cyan] Show configuration") + console.print(" [bold cyan]4[/bold cyan] Show Docker launch command") + console.print(" [bold cyan]q[/bold cyan] Quit") + try: + choice = typer.prompt("\nChoose", default="1").strip().lower() + except (EOFError, typer.Abort): + return + if choice in {"q", "quit", "exit"}: + return + if choice == "1": + models_select() + elif choice == "2": + optimize() + elif choice == "3": + config_get_cmd() + elif choice == "4": + print_run() + else: + console.print("[yellow]Choose 1–4 or q.[/yellow]") + console.print() + + # ── subcommands ──────────────────────────────────────────────────────────── @@ -223,6 +285,27 @@ def _print_installed_presets() -> None: console.print(f"[dim]Total disk usage: {total:.1f} GB[/dim]") +def _activate_preset(cfg: Config, preset: download_mod.ModelPreset) -> None: + """Persist one selected preset and seed first-run tuning.""" + config_set("model.preset", preset.name) + config_set("model.target_file", preset.target_file) + if preset.has_draft and preset.draft_file: + config_set("model.draft_file", preset.draft_file) + else: + # Drop any stale draft_file from a previous activation; the selected + # preset is explicitly target-only. + config_unset("model.draft_file") + console.print(f"[green]Activated:[/green] model.preset = {preset.name}") + # Never clobber a user-edited [dflash] section. Explicit profile resets + # go through `lucebox optimize` instead. + if config_mod.seed_dflash_from_host(cfg.host): + max_ctx = config_get("dflash.max_ctx")["dflash.max_ctx"][0] + console.print( + f"[green]Auto-tuned:[/green] VRAM-tier DFLASH_* defaults " + f"(max_ctx={max_ctx}) written to config.toml" + ) + + @models_app.callback(invoke_without_command=True) def models_default(ctx: typer.Context) -> None: """Default action: list installed presets, mark active with `*`.""" @@ -250,6 +333,82 @@ def models_list() -> None: console.print(table) +@models_app.command("select") +def models_select( + preset: Annotated[ + str, + typer.Argument(help="Preset name (omit for a numbered menu)."), + ] = "", + yes: Annotated[ + bool, + typer.Option("--yes", "-y", help="Download and activate without confirmation."), + ] = False, +) -> None: + """Choose, download, and activate a model in one guided step.""" + cfg = _load_or_build() + if not preset: + names = sorted(download_mod.PRESETS) + recommended = download_mod.recommend_preset(cfg.host) + default_name = cfg.model.preset or recommended or names[0] + default_index = names.index(default_name) + 1 if default_name in names else 1 + + table = Table(title="Choose a model", show_lines=False) + table.add_column("#", justify="right", style="cyan") + table.add_column("model") + table.add_column("download") + table.add_column("status") + table.add_column("notes") + for index, name in enumerate(names, start=1): + candidate = download_mod.PRESETS[name] + labels: list[str] = [] + if name == cfg.model.preset: + labels.append("active") + if name == recommended: + labels.append("recommended") + table.add_row( + str(index), + name, + f"~{candidate.approx_total_gb} GB", + download_mod.installed_status(cfg, candidate), + ", ".join(labels) or candidate.description, + ) + console.print(table) + try: + answer = typer.prompt( + "Model number or name", + default=str(default_index), + ).strip() + except (EOFError, typer.Abort): + console.print("[dim]No model changed.[/dim]") + return + if answer.isdigit() and 1 <= int(answer) <= len(names): + preset = names[int(answer) - 1] + else: + preset = answer + + try: + selected = download_mod.resolve_preset(preset) + except KeyError as exc: + error_console.print(f"[red]{escape(str(exc))}[/red]") + raise typer.Exit(code=2) from exc + + state = download_mod.installed_status(cfg, selected) + if state == "installed": + # Selection must work on a preloaded buyer appliance even when it is + # offline. Do not query Hugging Face merely to activate files that are + # already present locally. + _activate_preset(cfg, selected) + return + if state != "installed" and not yes: + if not typer.confirm( + f"Download about {selected.approx_total_gb} GB and activate {selected.name}?", + default=True, + ): + console.print("[dim]No model changed.[/dim]") + return + models_download(selected.name, activate=True) + + @models_app.command("download") def models_download( preset: Annotated[str, typer.Argument(help="Preset name (empty = recommend)")] = "", @@ -317,26 +476,37 @@ def models_download( console.print("[green]Done.[/green]") if activate: - config_set("model.preset", preset) - if pres.target_file: - config_set("model.target_file", pres.target_file) - if pres.has_draft and pres.draft_file: - config_set("model.draft_file", pres.draft_file) - else: - # Drop any stale draft_file from a previous activation; the - # active preset has no draft. - config_unset("model.draft_file") - console.print(f"[green]Activated:[/green] model.preset = {preset}") - # First-time setup: bake the VRAM-tier DFLASH_* heuristic into - # config.toml so `lucebox serve` is auto-tuned to this host instead - # of falling back to the conservative class defaults. Never clobbers - # an existing [dflash] section. - if config_mod.seed_dflash_from_host(cfg.host): - max_ctx = config_get("dflash.max_ctx")["dflash.max_ctx"][0] - console.print( - f"[green]Auto-tuned:[/green] VRAM-tier DFLASH_* defaults " - f"(max_ctx={max_ctx}) written to config.toml" - ) + _activate_preset(cfg, pres) + + +@app.command() +def optimize( + yes: Annotated[ + bool, + typer.Option("--yes", "-y", help="Apply without confirmation."), + ] = False, +) -> None: + """Apply the recommended hardware-aware inference profile. + + Stable CUDA/ROCm fast paths remain enabled by the engine itself. This + profile selects a safe context size, cache format, and DFlash budget from + detected VRAM while leaving experimental features off. + """ + cfg = _load_or_build() + recommended = autotune_mod.runtime_from_host(cfg.host) + console.print("[bold]Automatic optimization (recommended)[/bold]") + console.print(" DFlash decode automatic when the model has a matching draft") + console.print(" GPU fast paths enabled by the engine on CUDA and ROCm") + console.print(f" Maximum context {recommended.max_ctx:,} tokens") + cache = recommended.cache_type_k or "model default" + console.print(f" KV cache {cache}") + console.print(" Experimental paths off") + if not yes and not typer.confirm("Apply this profile?", default=True): + console.print("[dim]Optimization unchanged.[/dim]") + return + config_mod.seed_dflash_from_host(cfg.host, force=True) + console.print("[green]Automatic optimization applied.[/green]") + console.print("[dim]Advanced users can still use `lucebox config set dflash.KEY=VALUE`.[/dim]") @app.command() diff --git a/lucebox/src/lucebox/config.py b/lucebox/src/lucebox/config.py index 75a601afb..778d81385 100644 --- a/lucebox/src/lucebox/config.py +++ b/lucebox/src/lucebox/config.py @@ -394,12 +394,19 @@ def save(cfg: Config, path: Path | None = None, *, doc: dict[str, Any] | None = return path -def seed_dflash_from_host(host: HostFacts, *, path: Path | None = None) -> bool: +def seed_dflash_from_host( + host: HostFacts, + *, + path: Path | None = None, + force: bool = False, +) -> bool: """Persist the VRAM-tier DFLASH_* heuristic to config.toml on first setup. - Returns True when it wrote. No-op (returns False) when a ``[dflash]`` - section already exists, so it never clobbers values a prior tune or the - user set. Called when a preset is first activated: without it a fresh + Returns True when it wrote. By default this is a no-op when a ``[dflash]`` + section already exists, so first-time setup never clobbers values a prior + tune or the user set. ``force=True`` intentionally replaces that section; + the interactive CLI uses it when the user explicitly selects the automatic + profile. Called when a preset is first activated: without it a fresh install serves at the conservative ``DflashRuntime`` class defaults (``load()`` returns those for a config.toml that has no ``[dflash]``), ignoring the host's VRAM tier. The ``live_config`` heuristic only fires @@ -417,8 +424,13 @@ def seed_dflash_from_host(host: HostFacts, *, path: Path | None = None) -> bool: if not path.exists() and path.with_suffix(".env").exists(): load(path) doc = load_doc(path) - if "dflash" in doc: + if "dflash" in doc and not force: return False + if force: + # Replace the whole section rather than updating known keys in place. + # That removes stale experimental fields and makes "Automatic" a real + # reset to the current hardware-derived defaults. + doc.pop("dflash", None) runtime = autotune_mod.runtime_from_host(host) for field, value in asdict(runtime).items(): _doc_set(doc, "dflash", field, _value_to_toml(value)) diff --git a/lucebox/tests/test_cli.py b/lucebox/tests/test_cli.py index ee254d9bd..7fb146ffd 100644 --- a/lucebox/tests/test_cli.py +++ b/lucebox/tests/test_cli.py @@ -53,10 +53,19 @@ def test_core_verbs_present_in_app() -> None: c.name or (c.callback.__name__ if c.callback else "") for c in app.registered_commands } - for verb in ("check", "pull", "print-run", "print-serve-argv", "version"): + for verb in ("check", "pull", "optimize", "print-run", "print-serve-argv", "version"): assert verb in registered +def test_no_args_has_branded_noninteractive_help() -> None: + """Pipes and CI get useful output instead of an input prompt.""" + result = CliRunner().invoke(app, []) + assert result.exit_code == 0 + assert "local inference, made simple" in result.output + assert "models" in result.output + assert "optimize" in result.output + + @pytest.mark.parametrize("args", [["version"], ["--version"]]) def test_version_command_and_option_match(args: list[str]) -> None: result = CliRunner().invoke(app, args) diff --git a/lucebox/tests/test_models_cli.py b/lucebox/tests/test_models_cli.py index dd4f244f5..97eaba8c7 100644 --- a/lucebox/tests/test_models_cli.py +++ b/lucebox/tests/test_models_cli.py @@ -122,6 +122,69 @@ def test_models_download_explicit_preset_with_activate( assert entries["model.preset"] == ("gemma-4-26b", "file") +def test_models_select_activates_preloaded_model_offline( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """A factory-preloaded buyer can switch models without a network call.""" + cfg_path = _set_config_path(tmp_path, monkeypatch) + _stub_host(monkeypatch, vram_gb=24) + preset = PRESETS["qwen3.6-27b"] + models = tmp_path / "models" + (models / preset.target_file).parent.mkdir(parents=True, exist_ok=True) + (models / preset.target_file).touch() + assert preset.draft_file is not None + (models / "draft").mkdir() + (models / "draft" / preset.draft_file).touch() + + def fail_network(*args: object, **kwargs: object) -> object: + raise AssertionError("preloaded selection must not contact Hugging Face") + + monkeypatch.setattr(download_mod, "status", fail_network) + monkeypatch.setattr(download_mod, "download_preset", fail_network) + + result = CliRunner().invoke(app, ["models", "select", preset.name, "--yes"]) + assert result.exit_code == 0 + assert "Activated" in result.output + entries = config_mod.config_get(path=cfg_path) + assert entries["model.preset"] == (preset.name, "file") + assert entries["model.target_file"] == (preset.target_file, "file") + assert entries["model.draft_file"] == (preset.draft_file, "file") + + +def test_models_select_numbered_picker_downloads_and_activates( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + cfg_path = _set_config_path(tmp_path, monkeypatch) + _stub_host(monkeypatch, vram_gb=24) + monkeypatch.setattr(download_mod, "download_preset", lambda cfg, pres: 0) + monkeypatch.setattr( + download_mod, + "status", + lambda cfg, pres: {"target_present": False, "draft_present": False}, + ) + # qwen3.6-27b is the fourth entry in the stable alphabetical menu. + result = CliRunner().invoke(app, ["models", "select"], input="4\ny\n") + assert result.exit_code == 0 + assert "Choose a model" in result.output + entries = config_mod.config_get(path=cfg_path) + assert entries["model.preset"] == ("qwen3.6-27b", "file") + + +def test_optimize_resets_to_hardware_profile( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + cfg_path = _set_config_path(tmp_path, monkeypatch) + _stub_host(monkeypatch, vram_gb=24) + config_mod.config_set("dflash.max_ctx", 4096, path=cfg_path) + + result = CliRunner().invoke(app, ["optimize", "--yes"]) + assert result.exit_code == 0 + assert "Automatic optimization applied" in result.output + entries = config_mod.config_get(path=cfg_path) + assert entries["dflash.max_ctx"] == (98304, "file") + assert entries["dflash.cache_type_k"] == ("tq3_0", "file") + + def test_installed_helpers_track_presence( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ) -> None: diff --git a/scripts/test_lucebox_sh.sh b/scripts/test_lucebox_sh.sh index d95513daf..46134abb3 100755 --- a/scripts/test_lucebox_sh.sh +++ b/scripts/test_lucebox_sh.sh @@ -184,9 +184,14 @@ if bash -n "$ENTRYPOINT"; then report ok "bash -n entrypoint.sh parses cleanly" else report fail "bash -n entrypoint.sh"; fi # ── 3. Trivial subcommands (zero-exit expected) ─────────────────────────── -assert_runs "help" "bash '$SCRIPT' help" "host-side wrapper" -assert_runs "--help" "bash '$SCRIPT' --help" "host-side wrapper" -assert_runs "-h" "bash '$SCRIPT' -h" "host-side wrapper" +assert_runs "help" "bash '$SCRIPT' help" "simple CLI for the Lucebox inference engine" +assert_runs "--help" "bash '$SCRIPT' --help" "simple CLI for the Lucebox inference engine" +assert_runs "-h" "bash '$SCRIPT' -h" "simple CLI for the Lucebox inference engine" +assert_runs "no args (non-interactive help)" \ + "bash '$SCRIPT' "$native_tmp/bin/cmake" <<'STUB' +#!/usr/bin/env bash +printf '%s\n' "$*" >> "${LUCEBOX_TEST_CMAKE_LOG:?}" +STUB +chmod +x "$native_tmp/bin/cmake" +native_log="$native_tmp/cmake.log" +native_out="" +native_rc=0 +native_out=$( + PATH="$native_tmp/bin:$PATH" \ + LUCEBOX_TEST_CMAKE_LOG="$native_log" \ + LUCEBOX_REPO="$ROOT" \ + LUCEBOX_BUILD_DIR="$native_tmp/build" \ + LUCEBOX_VARIANT=cuda12 \ + _LUCEBOX_HOST_PROBED=1 \ + LUCEBOX_HOST_HAS_NVIDIA_GPU=1 \ + LUCEBOX_HOST_GPU_SM=86 \ + LUCEBOX_HOST_NPROC=2 \ + bash "$SCRIPT" build cuda 2>&1 +) || native_rc=$? +if [ "$native_rc" -ne 0 ]; then + report fail "native CUDA build dispatch" "exit=$native_rc output=$(head -3 <<<"$native_out")" +elif ! grep -qF -- "-DDFLASH27B_GPU_BACKEND=cuda" "$native_log" \ + || ! grep -qF -- "-DCMAKE_CUDA_ARCHITECTURES=86" "$native_log" \ + || ! grep -qF -- "--target dflash_server -j 2" "$native_log"; then + report fail "native CUDA build dispatch" "unexpected cmake argv: $(tr '\n' ' ' < "$native_log")" +else + report ok "native CUDA build dispatch" +fi + +: > "$native_log" +native_rc=0 +native_out=$( + PATH="$native_tmp/bin:$PATH" \ + LUCEBOX_TEST_CMAKE_LOG="$native_log" \ + LUCEBOX_REPO="$ROOT" \ + LUCEBOX_BUILD_DIR="$native_tmp/build-hip" \ + LUCEBOX_VARIANT=rocm \ + _LUCEBOX_HOST_PROBED=1 \ + LUCEBOX_HOST_HAS_NVIDIA_GPU=0 \ + LUCEBOX_HOST_HAS_AMD_GPU=1 \ + LUCEBOX_HOST_AMD_GPU_ARCH=gfx1201 \ + LUCEBOX_HOST_NPROC=2 \ + bash "$SCRIPT" build rocm 2>&1 +) || native_rc=$? +if [ "$native_rc" -ne 0 ]; then + report fail "native ROCm build dispatch" "exit=$native_rc output=$(head -3 <<<"$native_out")" +elif ! grep -qF -- "-DDFLASH27B_GPU_BACKEND=hip" "$native_log" \ + || ! grep -qF -- "-DDFLASH27B_HIP_ARCHITECTURES=gfx1201" "$native_log"; then + report fail "native ROCm build dispatch" "unexpected cmake argv: $(tr '\n' ' ' < "$native_log")" +else + report ok "native ROCm build dispatch" +fi +rm -rf "$native_tmp" + # ── 7. Unknown subcommand → cmd_in_container fallback path. Same rule: # clean error, no raw bash leak. assert_no_set_u_leak "unknown subcommand dispatch" "$SCRIPT" no-such-subcommand @@ -797,6 +861,7 @@ test_image_base_derives_from_install_url() { local label="$1" url expected got for case in \ "https://raw.githubusercontent.com/easel/lucebox-hub/feat/lucebox-docker/lucebox.sh|ghcr.io/easel/lucebox-hub" \ + "https://raw.githubusercontent.com/Luce-Org/lucebox/main/lucebox.sh|ghcr.io/luce-org/lucebox-hub" \ "https://raw.githubusercontent.com/Luce-Org/lucebox-hub/main/lucebox.sh|ghcr.io/luce-org/lucebox-hub" \ "https://raw.githubusercontent.com/easel/lucebox-hub/601ab52/lucebox.sh|ghcr.io/easel/lucebox-hub" \ "https://example.com/bogus|ghcr.io/luce-org/lucebox-hub" @@ -817,7 +882,7 @@ test_image_base_derives_from_install_url() { done report ok "$label" } -test_image_base_derives_from_install_url "IMAGE_BASE derived from LUCEBOX_INSTALLED_FROM (4 URL shapes)" +test_image_base_derives_from_install_url "IMAGE_BASE derived from LUCEBOX_INSTALLED_FROM (5 URL shapes)" # ── config.toml reader + resolver ───────────────────────────────────────── # Drive _lucebox_config_get + _lucebox_resolve against a fixture diff --git a/server/scripts/entrypoint.sh b/server/scripts/entrypoint.sh index 95373aa1b..6187c8af8 100755 --- a/server/scripts/entrypoint.sh +++ b/server/scripts/entrypoint.sh @@ -610,7 +610,11 @@ if [ "$DFLASH_PREFILL_MODE" != "off" ]; then --prefill-drafter "$DFLASH_PREFILL_DRAFTER") fi -info "lucebox-hub container starting (target=$(basename "$DFLASH_TARGET"), max_ctx=$DFLASH_MAX_CTX, budget=$DFLASH_BUDGET, lazy=$DFLASH_LAZY)" +if [ "${LUCEBOX_NATIVE:-0}" = "1" ]; then + info "lucebox native server starting (target=$(basename "$DFLASH_TARGET"), max_ctx=$DFLASH_MAX_CTX, budget=$DFLASH_BUDGET, lazy=$DFLASH_LAZY)" +else + info "lucebox-hub container starting (target=$(basename "$DFLASH_TARGET"), max_ctx=$DFLASH_MAX_CTX, budget=$DFLASH_BUDGET, lazy=$DFLASH_LAZY)" +fi cd "$DFLASH_DIR" exec "${CMD[@]}"