diff --git a/public/lite/clients-en.webp b/public/lite/clients-en.webp index c81ec90..82f5fa7 100644 Binary files a/public/lite/clients-en.webp and b/public/lite/clients-en.webp differ diff --git a/public/lite/clients.webp b/public/lite/clients.webp index e3b3b24..1764886 100644 Binary files a/public/lite/clients.webp and b/public/lite/clients.webp differ diff --git a/public/lite/dry-run-en.webp b/public/lite/dry-run-en.webp index be2efb9..99de9c0 100644 Binary files a/public/lite/dry-run-en.webp and b/public/lite/dry-run-en.webp differ diff --git a/public/lite/dry-run.webp b/public/lite/dry-run.webp index 755b4cd..e6fa70c 100644 Binary files a/public/lite/dry-run.webp and b/public/lite/dry-run.webp differ diff --git a/public/lite/keys-en.webp b/public/lite/keys-en.webp index 4cab5c5..520fa6c 100644 Binary files a/public/lite/keys-en.webp and b/public/lite/keys-en.webp differ diff --git a/public/lite/keys.webp b/public/lite/keys.webp index 03d318a..fa83ea6 100644 Binary files a/public/lite/keys.webp and b/public/lite/keys.webp differ diff --git a/public/lite/menubar-menu-en.webp b/public/lite/menubar-menu-en.webp index 77a4180..876d12c 100644 Binary files a/public/lite/menubar-menu-en.webp and b/public/lite/menubar-menu-en.webp differ diff --git a/public/lite/menubar-menu.webp b/public/lite/menubar-menu.webp index d710d9b..323e5c5 100644 Binary files a/public/lite/menubar-menu.webp and b/public/lite/menubar-menu.webp differ diff --git a/public/lite/menubar.png b/public/lite/menubar.png index bc2c71a..093f4eb 100644 Binary files a/public/lite/menubar.png and b/public/lite/menubar.png differ diff --git a/public/lite/overview-en.webp b/public/lite/overview-en.webp index 0f6467b..c2c7251 100644 Binary files a/public/lite/overview-en.webp and b/public/lite/overview-en.webp differ diff --git a/public/lite/overview.webp b/public/lite/overview.webp index d82184a..711398e 100644 Binary files a/public/lite/overview.webp and b/public/lite/overview.webp differ diff --git a/public/lite/remote-add-en.webp b/public/lite/remote-add-en.webp index c963379..3cb29eb 100644 Binary files a/public/lite/remote-add-en.webp and b/public/lite/remote-add-en.webp differ diff --git a/public/lite/remote-add.webp b/public/lite/remote-add.webp index 6e4b377..d037aff 100644 Binary files a/public/lite/remote-add.webp and b/public/lite/remote-add.webp differ diff --git a/public/lite/remote-clients-en.webp b/public/lite/remote-clients-en.webp index 9f08516..776cf65 100644 Binary files a/public/lite/remote-clients-en.webp and b/public/lite/remote-clients-en.webp differ diff --git a/public/lite/remote-clients.webp b/public/lite/remote-clients.webp index 878107b..592f0b8 100644 Binary files a/public/lite/remote-clients.webp and b/public/lite/remote-clients.webp differ diff --git a/public/lite/remote-switcher-en.webp b/public/lite/remote-switcher-en.webp index 83a3001..5d1a9c4 100644 Binary files a/public/lite/remote-switcher-en.webp and b/public/lite/remote-switcher-en.webp differ diff --git a/public/lite/remote-switcher.webp b/public/lite/remote-switcher.webp index 276973b..11ce0fa 100644 Binary files a/public/lite/remote-switcher.webp and b/public/lite/remote-switcher.webp differ diff --git a/public/lite/routing-en.webp b/public/lite/routing-en.webp index 5a05efa..bc78805 100644 Binary files a/public/lite/routing-en.webp and b/public/lite/routing-en.webp differ diff --git a/public/lite/routing.webp b/public/lite/routing.webp index 4318d8b..2b602c4 100644 Binary files a/public/lite/routing.webp and b/public/lite/routing.webp differ diff --git a/public/lite/security-en.webp b/public/lite/security-en.webp index 6829747..0c23e8b 100644 Binary files a/public/lite/security-en.webp and b/public/lite/security-en.webp differ diff --git a/public/lite/security.webp b/public/lite/security.webp index 0ceb8f3..b29e79a 100644 Binary files a/public/lite/security.webp and b/public/lite/security.webp differ diff --git a/public/lite/settings-en.webp b/public/lite/settings-en.webp index 3875009..81a67d0 100644 Binary files a/public/lite/settings-en.webp and b/public/lite/settings-en.webp differ diff --git a/public/lite/settings.webp b/public/lite/settings.webp index 5de9b3f..1090447 100644 Binary files a/public/lite/settings.webp and b/public/lite/settings.webp differ diff --git a/public/lite/traffic-en.webp b/public/lite/traffic-en.webp index 9a8ea47..1d8ebbd 100644 Binary files a/public/lite/traffic-en.webp and b/public/lite/traffic-en.webp differ diff --git a/public/lite/traffic.webp b/public/lite/traffic.webp index 2b5369d..7c00192 100644 Binary files a/public/lite/traffic.webp and b/public/lite/traffic.webp differ diff --git a/public/lite/upstreams-en.webp b/public/lite/upstreams-en.webp index 982c15b..d3a1724 100644 Binary files a/public/lite/upstreams-en.webp and b/public/lite/upstreams-en.webp differ diff --git a/public/lite/upstreams.webp b/public/lite/upstreams.webp index de2d730..0692c91 100644 Binary files a/public/lite/upstreams.webp and b/public/lite/upstreams.webp differ diff --git a/src/components/pages/LitePage.astro b/src/components/pages/LitePage.astro index b08f2e1..f272a49 100644 --- a/src/components/pages/LitePage.astro +++ b/src/components/pages/LitePage.astro @@ -130,7 +130,11 @@ const archLabel = { x64: "x64", arm64: "ARM64" } as const; {c.status.badge} -

{c.status.body}

+ diff --git a/src/content/docs-lite/en/features.md b/src/content/docs-lite/en/features.md new file mode 100644 index 0000000..e9ded41 --- /dev/null +++ b/src/content/docs-lite/en/features.md @@ -0,0 +1,86 @@ +# Features + +A page-by-page reference to what ThinkWatch Lite shows and does. The [Lite page](/lite) gives the short version. + +The main window has nine pages: Overview, Traffic, Clients, Keys, Upstreams, Routing, Security, MCP and Settings. + +## Usage and cost + +The Overview page reports tokens, cost and requests for the last 24 hours, 7 days, 30 days or a custom range, each against the period before; a live view follows the last ten minutes. A trend chart stacks tokens or cost by model, and the model ranking beneath it opens the matching requests. Further sections cover the prompt cache (hit rate, the net savings it brought and the hit rate per model), latency (median and 95th-percentile time to first token, per model and per upstream), generation speed (median tokens per second, per model and per upstream) and what each protection found. + +The cost figure states how much of it is estimated, for instance for a response that was cut off before it finished. Requests whose model has no price, and requests whose upstream reported no usage, are counted separately and never added in as zero. Prices come from price sheets: the default one follows LiteLLM's public prices and is refreshed once a day, and a custom one applies a multiplier and its own prices for particular models, as needed for a relay whose prices differ from the official ones. A subscription account such as a ChatGPT sign-in is priced from the price sheet like any other upstream, and an upstream such as a local model can be set to free. Each request's cost is fixed when the request finishes, and the request records the price sheet and the date of the prices it was costed with. + +## Traffic and sessions + +The Traffic page lists requests as they arrive: status, key, model, upstream, time to first token and total time (with the generation speed on hover), tokens and cost, with marks for a converted API format, redacted keys and a blocked or suspicious tool call. The list can be filtered by key, upstream and model, or narrowed to failed or unpriced requests. The Sessions view groups the requests of one conversation into turns, with the input tokens and the cost of each turn. + +A request opens into its timeline, its routing (the rule it matched, the group it went through and each attempt with its status and duration), the request and response bodies, and its usage and cost. A request from DeepSeek Harness also shows the size of the session log it carried, the whole conversation the client attaches to every request; the gateway removes it before a request goes to an upstream other than DeepSeek. A finished request can be sent again, unchanged, to another upstream after an estimate of its cost, and the two responses are shown side by side. + +## Client setup + +The Clients page points Claude Code, Claude Desktop, Codex, opencode, Zed, Aider and DeepSeek Harness at the gateway. Before anything is written, it lists the fields that change and what else the change affects (the ChatGPT desktop app, for instance, reads the same configuration file as Codex), shows the full diff and backs up the original file. Only the settings that point the client at the gateway change, and each client receives a key of its own. Claude Desktop is connected through its official third-party inference mode, and the page lists each of the files that change for it; a Claude Desktop managed by an organization is left as it is. A connected client can be restored at any time, on its own or together with all the others; a restored Codex keeps a plain OpenAI entry in place of the gateway's, so sessions started while it was connected can still be opened. opencode (v1 and v2) also gets the list of models its key can use on the gateway; when that list changes, the page offers to update it, through the same diff. Cursor, Continue and Antigravity CLI come with step-by-step instructions and a key created for them. For every client the page shows whether it is in use, waiting for its first request or not in effect, and its requests over the last 24 hours. + +On Windows, Claude Code and Codex installed inside WSL appear in a group of their own for each distribution, next to the clients on the computer itself. They are pointed at the gateway on Windows, restored and diagnosed the same way, each with a key separate from the Windows copy, and their files are edited through `\\wsl.localhost`. They are given `127.0.0.1`, the same address as the clients on Windows, which WSL reaches in two setups: + +- **WSL 1**, which shares the network with Windows. +- **WSL 2 with mirrored networking**: `networkingMode=mirrored` under `[wsl2]` (or the older `[experimental]`) in `%USERPROFILE%\.wslconfig`. It needs Windows 11 22H2 or later and WSL 2.0.5 or later. + +WSL 2 uses NAT networking by default, and the gateway on Windows cannot be reached from inside WSL that way; the gateway does not listen on the WSL virtual adapter for it. The WSL group then explains this instead of offering to connect, and offers to switch to mirrored networking: `networkingMode` in `.wslconfig` is added or changed, and nothing else in the file is touched, through the same diff, confirmation and full backup as a client. The switch takes effect once WSL restarts, which the page also offers (`wsl --shutdown`, which stops every running distribution). A full uninstall leaves `.wslconfig` as it is, and its backup is kept. On Windows 10 and Windows 11 21H2, which have no mirrored networking, and with a WSL older than 2.0.5, the group says so. When connected to a remote core, clients in WSL are pointed at the server like those on Windows, whatever the networking. + +## Keys + +Clients reach the gateway with a key, on the local machine as well. The Keys page lists the default key, used by clients that were not given one of their own, and a key for each connected client, labelled with the client it belongs to so that its requests can be told apart in Traffic. Each key has a route, the models it may use (all, none, or chosen models and patterns such as `gpt-5*`), an optional limit on concurrent requests, and its requests and cost over the last 24 hours. A key can be disabled, which rejects every request made with it, or rotated; rotating writes the new key into the configuration of the client that uses it. + +## Upstreams + +Upstreams are the services requests are forwarded to: API keys for Anthropic, OpenAI, Google Gemini, DeepSeek or any compatible endpoint, Amazon Bedrock (with an API key, access keys or an AWS profile), a ChatGPT account or a Z.ai / BigModel account signed in from the app, relays such as OpenRouter, and local models such as Ollama. A ChatGPT account shows its usage limits and reset times. So does an upstream on a GLM Coding Plan, that is, one whose address is on `api.z.ai` or `open.bigmodel.cn`, whether it was signed in from the app or added with a key: its 5-hour and weekly limits and, on a plan billed in credits, the credits left (“1,976 / 2,000 credits left”). When a client and an upstream use different API formats, requests are converted between Anthropic Messages, OpenAI Chat Completions, OpenAI Responses and Gemini, and the fields that cannot be carried over are listed on the request. Upstreams can be reached through an outbound proxy and priced with a price sheet of their own; proxies and price sheets have tabs on the same page. A connection test times the DNS lookup and the TCP, TLS and proxy handshakes without incurring any cost; an inference test measures the time to first token and estimates its cost before it runs. + +API keys and header values can be written as `${NAME}` to read a system environment variable. On macOS these come from the login shell, so variables exported in `~/.zshrc` and similar files apply, and the same holds on Linux (`~/.bashrc`, `~/.profile` and so on); on Windows they are the environment variables configured in system settings. After a variable changes, reopening the app picks it up. Proxy variables such as `HTTPS_PROXY`, and `PATH`, are not read. + +A relay or vendor can hand out an import link, `thinkwatch://import?…` or its web form `https://thinkwat.ch/import#…`, that pre-fills a new upstream with a name, base URL, protocol, API key and model list. The app shows the settings and the host that will receive requests and the key in a confirmation dialog, and writes nothing and contacts nothing before Create is chosen. A link only ever adds one upstream: it cannot change existing ones, headers, proxies, pricing or routing, and a key that refers to an environment variable is rejected. The parameters and a link builder are in [Import links](/docs/lite/import-links). + +## Routing and failover + +Each key follows a route, and keys without one follow the default route. A route is a list of rules evaluated in order. A rule matches on the model, the key, the client's API format, input tokens, `max_tokens`, the number of tools, images, extended thinking, streaming, prompt caching or the kind of auxiliary request; it then forwards the request to an upstream or a group, or refuses it, and can rewrite the model, `max_tokens` or extended thinking. A group puts several upstreams behind one name and decides the order in which they are tried: as listed, manually selected, in turn, lowest latency first or lowest cost first. When an attempt fails, the request moves on to the next upstream, and by default a session stays on one upstream so that its prompt cache keeps hitting. A map at the top of the page traces every key through its route and groups to the upstreams. + +Auxiliary requests that clients send on their own (health checks, warm-ups, titles, topic detection and input suggestions) can be answered locally at no cost, passed through, or routed by the rules. + +Every request records the rule it matched, the group it went through and each attempt with its status and duration. A dry run evaluates the rules for a given request and shows where it would go and why, without sending anything and without incurring any cost. + +## Security + +The Security page holds five protections. They apply to every upstream and every key alike, and each runs in one of three modes: Off, Observe (detect and record, change nothing) or Enforce. The output limit starts Off and the other four start in Observe, so out of the box no request is changed or blocked. + +- **Outbound redaction** looks for credentials in a request before it leaves: API keys and tokens for Anthropic, OpenAI, GitHub, Slack, AWS, Google, GitLab, Stripe, npm, DigitalOcean and SendGrid, private keys, JWTs and passwords in connection strings. In Enforce mode they are replaced with placeholders and restored where the response repeats them. Rules for internal IP addresses and internal domains are included and start off. +- **Tool-call inspection** checks the tool calls a model returns for commands that download or decode code and run it, send out environment variables or credential files, read private keys or cloud credentials, or install startup items and scheduled jobs. In Enforce mode such a call cuts the response off, so the client never receives a complete call to run. Deleting the home or root directory and making files world-writable are only recorded by default. +- **Hidden characters** looks for Unicode tag characters and bidirectional control characters in what the client sends, tool results included, and in Enforce mode refuses the request. +- **Content filter** matches keywords or regular expressions against the messages the client sends, tool results included, and in Enforce mode refuses a request that matches a blocking rule. Of the built-in rules, the three against explicit "ignore previous instructions" phrasing are on by default; rules for jailbreaks, persona manipulation, prompt extraction and their Chinese counterparts can be switched on. +- **Output limit** stops an answer that grows past a set number of characters, 100,000 by default: a streamed answer is cut off at that point and a non-streamed one is replaced with an error. Reasoning and tool-call arguments do not count towards the limit. + +The page lists every rule. Built-in rules can be switched on or off one at a time, and those for tool calls and content can be set to act or only record in Enforce mode. Custom rules are regular expressions, or keywords for the content filter, and any rule can be tried on a sample text first. Everything the protections find is kept in the log on the first tab, together with the request it came from. + +## MCP + +The MCP page covers what clients load from their own configuration files, which does not pass through the gateway. + +- **Servers:** the MCP servers configured in Claude Code, Claude Desktop, Cursor, Codex, opencode, Antigravity CLI, Zed and DeepSeek Harness, side by side. A server can be copied from one client to another or removed from a client; the change is shown before anything is written, and the original file is backed up. Copying and removing work for Claude Code, Claude Desktop, Cursor and Codex; opencode, Antigravity CLI, Zed and DeepSeek Harness are listed but not written to. A remote server on another host is marked as third party, since using it sends the surrounding context to that host, and a server configured differently in different clients is marked as well and can be compared side by side. +- **Skills and hooks:** the installed skills and configured hooks, with the client each belongs to. +- **Findings:** client configuration, skills, hooks, slash commands, subagents and project instruction files are scanned for hidden characters, prompt injection, dangerous commands and overly broad permissions, and each finding is graded high, medium or low. The scan only reports; it never changes a file. + +The app watches these files while it runs, and a new finding raises a system notification. + +## Settings + +Settings has six sections. Connection lists the local core and the saved remote cores, described in [Connecting to a remote core](/docs/lite/remote-core). General sets the language, the appearance, what the menu bar item shows on macOS, launch at login, and whether notices arrive as system notifications, in the app only or not at all. Listening sets who can reach the gateway (this machine only, the local network of a chosen interface, or every interface), its port and the allowed address ranges. Log retention sets how long request payloads and request records are kept, and a size cap for payloads. About shows the version, checks for updates and produces a diagnostics bundle with keys and addresses masked. Uninstall restores every connected client and removes the autostart entry, and is meant to be run before the app is deleted. + +## Menu bar, system tray and notifications + +On macOS the menu bar shows today's tokens above today's cost; the numbers turn orange when a subscription quota is nearly used up and red when it is, and Settings can reduce the item to the icon or to the numbers. + +Clicking it opens a native menu with the gateway's address, its generation speed over the last minute and its state, unread notices, today's requests, tokens and cost, the quotas and reset times of each subscription account and GLM Coding Plan upstream (with the credits left under the bar on a plan billed in credits), and the requests in progress, followed by actions: choosing the upstream of a manually selected group, copying the gateway address or the default key, switching connections and checking for updates, all without opening the main window. + +On Windows the icon sits in the notification area. Hovering over it shows the gateway's state and today's tokens and cost; a left click opens the main window, and a right click opens the same menu, with quota bars written out as text. + +On Linux the icon sits in the system tray. Clicking it opens the same menu, with Open ThinkWatch Lite as its first item and quota bars written out as text. + +System notifications, native on macOS and Windows and sent through the desktop's notification service on Linux, report when the gateway stops forwarding or keeps restarting, the connection to a remote core drops, a subscription quota runs out, a sign-in expires or an upstream rejects its credential, a proxy cannot be reached, the configuration file fails validation, a tool call matches a rule that cuts the response off, or suspicious content appears in a client's configuration. A new version found by the automatic check is announced the same way (see [Updates](/docs/lite/install#updates)). An unreachable upstream, which a fallback usually covers, is only listed in the app. Notices as a whole can be set to system notifications, in-app only, or off. Marking a notice as read stops the bell from counting it; the notice stays in the list until the problem behind it clears or the list is cleared. diff --git a/src/content/docs-lite/en/overview.md b/src/content/docs-lite/en/overview.md index a1dd987..b0da8ab 100644 --- a/src/content/docs-lite/en/overview.md +++ b/src/content/docs-lite/en/overview.md @@ -1,16 +1,24 @@ # ThinkWatch Lite -ThinkWatch Lite is a desktop app that runs a local AI API gateway on macOS, Windows and Linux, from the macOS menu bar, the Windows notification area or the Linux system tray. Claude Code, Codex and other clients of the Anthropic, OpenAI and Gemini APIs send their requests through the gateway, and the app records what each request cost, which upstream served it and why, and which credentials were redacted before it was sent. - -The gateway is [ThinkWatch Core](/docs/core). It runs beside the app on the same computer, or on a Linux server that the app connects to; see [Connecting to a remote core](/docs/lite/remote-core). +ThinkWatch Lite is a local gateway for Claude Code, Codex and other AI clients, on macOS, Windows and Linux. Each client is connected once; after that, upstreams and models change in the gateway without touching the client. Every request is recorded with its cost and route, and the API keys in it can be replaced before it leaves the machine. > It runs on macOS 12 or later on Apple silicon, on Windows 10 21H2 or later on x64 or ARM64 and on Linux on x86_64 or aarch64, is [installed](/docs/lite/install) with Homebrew, a disk image, the Windows installer or an AppImage, and updates itself. +## Highlights + +- **Connect once, switch freely.** Seven clients are pointed at the gateway in one step, with the change previewed and the original backed up; Cursor, Continue and Antigravity CLI come with instructions. +- **Keys replaced before sending.** Outbound redaction, tool-call inspection, hidden-character detection, a content filter and an output limit apply to every request, each in Off, Observe or Enforce. +- **MCP servers, skills and hooks, scanned.** The MCP servers of eight clients side by side, and a scan of client configuration for hidden characters, prompt injection, dangerous commands and overly broad permissions. +- **Every request traceable.** The matched rule, each attempt, any format conversion and the cost, with replay against another upstream. +- **Routing and failover.** Rules by model, tools, images and more; groups that fail over before the answer begins and keep each session on one upstream. +- **Any upstream.** API keys, Amazon Bedrock, ChatGPT and Z.ai accounts, relays and local models, with conversion between the Anthropic, OpenAI and Gemini APIs. +- **Costs stated as they are.** Estimates marked, unpriced requests counted separately rather than as zero. + ## Pages | Page | Contents | |---|---| -| Overview | Tokens, cost and requests over a period, trends by model, cache, latency and security results | +| Overview | Tokens, cost and requests over a period, trends by model, cache, latency, generation speed and security results | | Traffic | Every request, or requests grouped into sessions, with the details of each | | Clients | Pointing clients at the gateway, and restoring them | | Keys | The gateway keys clients connect with, each with its route and limits | @@ -20,97 +28,13 @@ The gateway is [ThinkWatch Core](/docs/core). It runs beside the app on the same | MCP | MCP servers, skills and hooks in each client, and the configuration scan | | Settings | Connection, language, appearance, menu bar, notifications, listening, retention, updates and uninstall | -## Usage and cost - -Tokens, cost and requests live or over the last 24 hours, 7 days, 30 days or a custom range, each compared with the period before, with a trend by model and the models ranked by usage. The Overview page also reports the cache hit rate and the net savings from caching, time-to-first-byte percentiles by model and by upstream, and what each security protection found in the period. - -Measured costs, estimated costs and unpriced requests are kept apart: an estimate, such as the cost of a request whose client disconnected before the response finished, is marked as one, and requests whose model has no price are counted separately instead of being added as zero. Costs follow LiteLLM's public price data, which the gateway refreshes daily, or a custom price sheet. - -## Requests and sessions - -The Traffic page lists each request with its key, model, upstream, time to first byte, total time, tokens and cost as it arrives. The list can be filtered by key, upstream, failures, unpriced requests or text, and grouped into sessions, so that the requests of one task can be read together. - -A request's details show the rule it matched, the group it went through, each attempt with its status and duration, and any failover to the next upstream. When the client and the upstream use different API formats, the request is converted between Anthropic Messages, OpenAI Chat Completions, OpenAI Responses and Gemini, and the fields that could not be carried over are listed. The details also include the request and response bodies and the usage the cost was calculated from. A finished request can be replayed against another upstream and the two results compared side by side. - -Request and response bodies are masked before they are displayed; a secret is never shown in the interface. - -## Routing and failover - -Each key uses a route. A route's rules are checked in order against the model, the key, the client's API format, input tokens, `max_tokens`, the number of tools, images, extended thinking, streaming and the prompt cache, and the first rule that matches decides: it forwards the request to an upstream or a group, or refuses it with a message. - -A group picks its upstreams in order, by manual choice, in rotation, by lowest latency or by lowest cost, and moves on to the next when one is unavailable. In rotation, sticky sessions keep each session on one upstream so that its prompt cache stays valid; they are on by default and can be turned off for a group. Auxiliary requests that clients send on their own, such as health checks, warm-ups and title generation, can be answered locally at no cost, passed through, or handled by the routing rules, which can send them to a lower-cost upstream. - -A dry run takes a key, a model, a client format and the properties of a request, and shows where the request would go: the rule that matched, why each rule before it did not, the upstreams that would be tried and any format conversion. It sends nothing and costs nothing. - -## Upstreams - -API-key upstreams such as Anthropic, OpenAI, Gemini, DeepSeek or any compatible endpoint; ChatGPT and Z.ai accounts signed in from the app; relays such as OpenRouter; and local models served by Ollama or another OpenAI-compatible server. The usage limits of ChatGPT accounts and of GLM Coding Plan keys on Z.ai and BigModel are shown with their reset times. - -Upstreams can connect through an outbound HTTP or SOCKS proxy, whose connection and authentication can be checked from the Upstreams page. Each upstream is billed per token or free. Prices come from the default price sheet, LiteLLM's public price data refreshed daily, or from a custom price sheet that applies a multiplier and prices for individual models on top of it. - -## Security - -Five protections apply to every request that passes through the gateway. They are global: the same modes and rules apply to every upstream and every key. - -| Protection | What it checks | In Enforce | -|---|---|---| -| Outbound redaction | API keys, private keys, JWTs and connection-string passwords in a request before it is sent; internal addresses and domains once their rules are turned on | Replaces them with placeholders and restores them in the response | -| Tool-call inspection | Tool calls returned by the upstream, against rules for dangerous commands such as downloading and running a script | Cuts off the response, for rules set to cut off | -| Hidden characters | Unicode tag characters and bidirectional controls in what the client sends, tool results included | Refuses the request | -| Content filter | Phrases and patterns in what the client sends, tool results included, such as instructions to ignore previous instructions | Refuses the request, for rules set to refuse | -| Output limit | The length of an answer, in characters | Cuts off the answer at the limit | - -Each protection is Off, Observe or Enforce. Observe detects and records matches without changing the request; it is the initial mode of every protection except the output limit, which starts Off and uses a limit of 100,000 characters once turned on. Built-in rules can be turned off one by one and custom rules added. Every match is listed in the security log with the request it came from. - -## MCP servers, skills and hooks - -The MCP page lists the MCP servers configured in Claude Code, Claude Desktop, Cursor, Codex, opencode, Zed, Antigravity CLI and DeepSeek Harness side by side. Remote and third-party servers are marked, and a server configured differently in two clients can be compared field by field. A server can be copied to another client or removed; the change is shown before it is written, and the original file is backed up. - -The page also lists hooks and skills, and scans client configuration, skills, hooks, slash commands, subagents and project instructions for hidden characters, prompt injection, dangerous commands and overly broad permissions. The scan reports what it finds and changes no file. The files are scanned again when they change, and a new finding raises a notification. - -## Client setup - -Claude Code, Codex, opencode, Zed, Aider, Claude Desktop and DeepSeek Harness can be pointed at the gateway from the app. The change is shown as a diff before anything is written, the original file is backed up, only the settings that point the client at the gateway change, and the change can be restored at any time. Cursor, Continue and Antigravity CLI come with step-by-step instructions. - -Claude Desktop is pointed at the gateway through its third-party inference mode. It has to be quit completely and reopened afterwards, and conversations in that mode are kept apart from the others; a Claude Desktop managed by an organization is not changed. For opencode, the models the client's key can use are written into its configuration, and the Clients page says when that list needs updating after upstreams or routes change. After Codex is restored, the sessions started while it pointed at the gateway can still be opened. - -On Windows, the Clients page also lists each WSL distribution. Claude Code and Codex inside WSL can be pointed at the gateway under WSL 1, or under WSL 2 with mirrored networking; they reach it at 127.0.0.1, and the gateway keeps listening on this computer only. When WSL 2 uses NAT networking, the page can switch it to mirrored networking, which needs Windows 11 22H2 or later, and restart WSL. - -## Keys - -Clients connect to the gateway with gateway keys. Connecting a client creates a key for it, so that traffic, cost and limits are attributed to that client. Each key has a route, the set of models its client sees, and an optional concurrency limit. A key can be disabled, except the default key, or rotated; when a key is rotated, the new key is written into the configuration of the client that uses it. - -## Menu bar and tray - -On macOS, the menu bar shows today's tokens above today's cost. The numbers turn orange when a subscription quota is nearly used up and red once it has run out; the item can also show only the icon or only the numbers. The item is drawn as a bitmap because the menu bar cannot display two lines of text. - -Its menu shows the gateway's state and output rate, today's requests, tokens and cost, each subscription quota with its reset time, and the requests in progress, with items to open the main window, copy the gateway address or the default key, switch connection and check for updates. - -On Windows the icon sits in the notification area. Hovering over it shows the gateway's state and today's tokens and cost; a left click opens the main window, and a right click opens the same menu, with quota bars written out as text. On Linux the icon sits in the system tray. Clicking it opens the same menu, with Open ThinkWatch Lite as its first item. - -## Notifications - -The following raise a system notification: - -- the gateway stops forwarding, cannot start or keeps exiting; -- the connection to a remote core is lost; -- a subscription quota runs out; -- an upstream's sign-in expires, an upstream rejects its credential, or a renewed credential cannot be saved to the configuration; -- an outbound proxy cannot be reached; -- an edited configuration file fails validation, or a change to the listening address does not take effect; -- a tool call matches a rule set to cut off; -- the configuration scan finds new suspicious content. - -An upstream that became unreachable is listed in the app without a system notification. A notice is withdrawn once its problem is resolved. One setting decides how all of them are delivered: as system notifications, in the app only, or not at all. System notifications are native on each platform: the macOS notification center, Windows notifications and the Linux desktop's notification service. - -## Interface language - -The interface is available in English and Simplified Chinese. It follows the system language by default; another language can be chosen in Settings. - -## Relationship to ThinkWatch Core +Each page is described in [Features](/docs/lite/features). -The gateway is implemented in ThinkWatch Core. Lite contains no routing, forwarding or accounting logic. It controls Core over a unix socket on macOS and Linux and a loopback port on Windows, or over a TCP port when it is connected to a core on a server. Every control connection is encrypted and authenticated by a handshake keyed by `listen.control.key` in `config.yaml`. See [Architecture](/docs/lite/architecture) and [Connecting to a remote core](/docs/lite/remote-core). +## Further reading -## License +- [Install and update](/docs/lite/install) +- [Connecting to a remote core](/docs/lite/remote-core): the gateway is [ThinkWatch Core](/docs/core), which runs beside the app or on a Linux server. +- [Architecture](/docs/lite/architecture): Lite holds no routing, forwarding or accounting logic; it controls Core over an encrypted control channel. +- [Build from source](/docs/lite/run-from-source) ThinkWatch Lite is licensed under the MIT License. diff --git a/src/content/docs-lite/zh-CN/features.md b/src/content/docs-lite/zh-CN/features.md new file mode 100644 index 0000000..8425376 --- /dev/null +++ b/src/content/docs-lite/zh-CN/features.md @@ -0,0 +1,86 @@ +# 功能详解 + +逐页说明 ThinkWatch Lite 展示的内容与提供的功能。简要介绍见 [Lite 产品页](/zh-CN/lite)。 + +主窗口共有九个页面:概览、流量、客户端、密钥、上游、路由、安全、MCP 和设置。 + +## 用量与费用 + +概览页按最近 24 小时、7 天、30 天或自定义区间统计 token、费用与请求数,并与上一个同等长度的区间对比;实时档显示最近十分钟。趋势图按模型分层显示 token 或费用,其下的模型排行可以直接打开对应的请求。页面下方依次是缓存(命中率、缓存带来的净节省、各模型的命中率)、延迟(首 token 时间的中位数与 P95,按模型和按上游)、生成速度(每秒 token 数的中位数,按模型和按上游)以及各项防护的检出情况。 + +费用会注明其中估算的部分,例如响应结束前被中断的请求。模型未定价的请求和上游未报告用量的请求单独计数,从不按零计入。价格来自价目表:默认价目表采用 LiteLLM 的公开价格,每天更新一次;自定义价目表在其基础上设置倍率,并可单独为个别模型定价,适用于价格与官方不同的中转服务。ChatGPT 这类订阅账号同样按价目表计价,本地模型等上游可设为不计费。每个请求的费用在请求结束时确定,并注明计价所用的价目表及其数据日期。 + +## 流量与会话 + +流量页实时列出请求:状态、密钥、模型、上游、首 token 时间与总耗时(悬停时显示生成速度)、token 和费用,并标出格式转换、被脱敏的密钥,以及被拦截或可疑的工具调用。列表可以按密钥、上游和模型筛选,或只看失败、无法计价的请求。「会话」视图把同一段对话的请求归为若干轮次,给出每一轮的输入 token 与费用。 + +打开一个请求可以查看时间线、路由(命中的规则、经过的策略组,以及每一次尝试的状态与耗时)、请求与响应正文、用量与费用。DeepSeek Harness 发出的请求还会显示所带会话日志的大小:这是客户端随每个请求附带的整段对话记录,发往 DeepSeek 以外的上游之前由网关去除。已结束的请求可以在预估费用后原样发送到另一个上游,两次的响应并排对照。 + +## 客户端接管 + +客户端页可以把 Claude Code、Claude Desktop、Codex、opencode、Zed、Aider 与 DeepSeek Harness 指向网关。写入之前,页面列出将要修改的字段和这次接管的其他影响(例如 ChatGPT 桌面版与 Codex 读取同一份配置文件),给出完整的改动差异,并完整备份原文件。只修改指向网关所需的配置,每个客户端使用各自的密钥。Claude Desktop 通过官方的第三方推理模式接入,页面逐一列出要修改的各个文件;由组织统一管理的 Claude Desktop 不做修改。已接管的客户端可以随时单独还原或全部还原;Codex 还原后保留一项直连 OpenAI 的配置,接管期间的会话仍可打开。opencode(v1 与 v2)的配置中同时写入其密钥在网关上可用的模型列表;网关上可用的模型变化后,页面提示更新,更新同样先给出改动差异。Cursor、Continue 与 Antigravity CLI 提供逐步的配置方法,并为其创建密钥。页面列出每个客户端处于使用中、等待首个请求还是未生效,以及最近 24 小时的请求。 + +在 Windows 上,安装在 WSL 中的 Claude Code 与 Codex 按发行版单独成组,列在这台电脑的客户端之后。它们同样可以指向 Windows 上的网关、还原和检查,使用与 Windows 上那一份分开的密钥,配置文件经由 `\\wsl.localhost` 修改。写入的地址与 Windows 上的客户端相同,是 `127.0.0.1`,WSL 在以下两种情况下可以访问: + +- **WSL 1**:与 Windows 共用网络。 +- **使用 mirrored 网络模式的 WSL 2**:在 `%USERPROFILE%\.wslconfig` 的 `[wsl2]` 段(或旧的 `[experimental]` 段)中设置 `networkingMode=mirrored`,需要 Windows 11 22H2 及以上版本、WSL 2.0.5 及以上版本。 + +WSL 2 默认使用 NAT 网络,此时 Windows 上的网关无法从 WSL 内访问,网关也不会为此另外监听 WSL 的虚拟网卡。WSL 分组因此不提供接管,改为说明原因,并提供「改为 mirrored 模式」:在 `.wslconfig` 中新增或修改 `networkingMode`,文件的其他内容保持不变,与接管客户端一样先给出完整差异,确认后全文备份再写入。修改在 WSL 重启后生效,页面同时提供「重启 WSL」(执行 `wsl --shutdown`,会停止所有正在运行的发行版)。完全卸载时 `.wslconfig` 不会改回,备份保留。Windows 10 与 Windows 11 21H2 没有 mirrored 网络模式,WSL 版本低于 2.0.5 时也无法使用,页面会分别说明。连接远程 core 时,WSL 中的客户端与 Windows 上的一样指向服务器,不受网络模式限制。 + +## 密钥 + +客户端连接网关必须携带密钥,本机也不例外。密钥页列出默认密钥(供未单独分配密钥的客户端使用)和每个已接管客户端的专用密钥,并注明所属客户端,便于在流量页中区分各客户端的请求。每把密钥有各自的路由、可用模型(全部、无,或指定的模型与 `gpt-5*` 这类通配模式)、可选的并发上限,以及最近 24 小时的请求数与费用。密钥可以停用,停用后使用它的请求一律被拒绝;也可以更换,新密钥会写入使用它的客户端的配置。 + +## 上游 + +上游是网关转发请求的目标:Anthropic、OpenAI、Google Gemini、DeepSeek 或任何兼容接口的 API 密钥,Amazon Bedrock(API 密钥、访问密钥或 AWS 配置文件),在应用内登录的 ChatGPT 账号或 Z.ai / BigModel 账号,OpenRouter 等中转服务,以及 Ollama 等本机模型。ChatGPT 账号显示订阅额度与重置时间;GLM Coding Plan 的上游(地址在 `api.z.ai` 或 `open.bigmodel.cn` 上,在应用内登录或手动填写密钥均可)同样显示:5 小时与每周额度,积分制套餐另外显示剩余积分(「剩余 1,976 / 2,000 积分」)。客户端与上游的 API 格式不同时,请求在 Anthropic Messages、OpenAI Chat Completions、OpenAI Responses 与 Gemini 之间自动转换,无法转换的字段会在请求上逐一列出。上游可以经出站代理访问,也可以使用单独的价目表计价,代理与价目表在同一页的标签中管理。链路测速测量 DNS 解析以及 TCP、TLS、代理握手的耗时,不产生费用;推理测速测量首个 token 的时间,运行前先给出费用预估。 + +API 密钥和请求头的值可以写成 `${变量名}`,读取系统环境变量。macOS 上读的是登录 shell 里的环境变量,`~/.zshrc` 等文件中 `export` 的变量都会生效,Linux 同理(`~/.bashrc`、`~/.profile` 等);Windows 上读的是系统设置里配置的环境变量。修改变量后,重新打开应用即可生效。代理相关的变量(`HTTPS_PROXY` 等)和 `PATH` 不会被读取。 + +中转站或服务商可以提供导入链接(`thinkwatch://import?…`,或网页形式 `https://thinkwat.ch/import#…`),预填新上游的名称、接口地址、接口协议、API 密钥与模型清单。应用在确认对话框中列出这些设置,并写明请求与密钥将发往的主机;选择「创建」之前不写入配置,也不连接该地址。一条链接只能新增一个上游,不能修改已有的上游、请求头、代理、价目表或路由,引用环境变量的密钥一律拒绝。参数说明与链接生成器见[导入链接](/zh-CN/docs/lite/import-links)。 + +## 路由与故障转移 + +每把密钥使用一条路由,未指定的使用默认路由。路由由按顺序匹配的规则组成。规则的条件包括模型、密钥、客户端的 API 格式、输入 token 数、`max_tokens`、工具数量、图片、扩展思考、流式、提示缓存以及辅助请求的类型;命中后把请求交给某个上游或策略组,或拒绝请求,也可以改写模型、`max_tokens` 或扩展思考。策略组把多个上游放在同一个名字下,并决定尝试的先后:按顺序、手动选择、轮询、延迟最低优先或费用最低优先。一次尝试失败时,请求转到下一个上游;同一会话默认保持在同一个上游上,以便提示缓存持续命中。页面顶部的路由图显示每把密钥经过的路由、策略组和上游。 + +客户端自行发出的辅助请求(连通性检查、预热、生成标题、话题识别、输入建议)可以由网关在本地应答而不产生费用,也可以直接转发,或交给路由规则处理。 + +每个请求都记录命中的规则、经过的策略组,以及每一次尝试的状态与耗时。试算按给定的请求条件逐条匹配规则,说明请求会交给哪个上游及其原因,不发出请求,也不产生费用。 + +## 安全 + +安全页有五项防护,对所有上游和所有密钥统一生效,各有「关闭」「观察」「拦截」三档,其中「观察」只检测和记录,不做任何改动。输出长度出厂为「关闭」,其余四项出厂为「观察」,因此默认不会改动或拦截任何请求。 + +- **出站脱敏**:请求发出之前查找其中的凭据,包括 Anthropic、OpenAI、GitHub、Slack、AWS、Google、GitLab、Stripe、npm、DigitalOcean、SendGrid 的 API 密钥与令牌,以及私钥、JWT 和连接串中的口令。「拦截」档下把它们替换为占位符,响应中回显时再还原。内网 IP 地址和内网域名两条规则出厂为停用,可以按需启用。 +- **工具调用审查**:检查模型返回的工具调用中是否含有下载或解码后执行代码、外发环境变量或凭据文件、读取私钥或云服务凭据、写入启动项或定时任务等命令。「拦截」档下命中即切断响应,客户端收不到一个完整、可执行的调用。删除主目录或根目录、设置全员可写权限两条规则出厂只记录。 +- **隐藏字符**:检查客户端发送的内容(含工具结果)中的 Unicode 标签字符和双向控制符,「拦截」档下拒绝发出请求。 +- **内容过滤**:用关键词或正则表达式匹配客户端发送的消息(含工具结果),「拦截」档下拒绝命中拒绝类规则的请求。内置规则中,出厂只启用三条明确要求「忽略先前指令」的规则;越狱、身份操纵、套取提示词等规则及其中文版本可以按需启用。 +- **输出长度**:回答超过设定的字符数(默认 100,000)时,流式回答在超出处切断,非流式回答整份替换为错误。思考内容和工具调用的参数不计入。 + +安全页列出全部规则:内置规则可以逐条启用或停用,工具调用审查和内容过滤的内置规则还可以设定在「拦截」档下执行处置还是仅记录;自定义规则为正则表达式,内容过滤也可以使用关键词。任何规则都可以先用一段文本测试。各项防护检出的内容都记入第一个标签页的日志,并注明所属的请求。 + +## MCP + +MCP 页管理客户端从自己的配置文件中加载的内容,这些内容不经过网关。 + +- **服务器**:并排列出 Claude Code、Claude Desktop、Cursor、Codex、opencode、Antigravity CLI、Zed 与 DeepSeek Harness 配置的 MCP 服务器。可以把一个服务器从一个客户端复制到另一个客户端,或从某个客户端移除;写入前先显示改动,并备份原文件。复制与移除支持 Claude Code、Claude Desktop、Cursor 与 Codex,opencode、Antigravity CLI、Zed 与 DeepSeek Harness 只列出、不写入。位于其他主机的远程服务器标为「第三方」,使用它会把相关上下文发送到该地址;同名服务器在各客户端中配置不同时标为「配置不一致」,可以并排比较。 +- **技能与钩子**:列出已安装的技能和配置的钩子,以及各自所属的客户端。 +- **发现**:扫描客户端配置、技能、钩子、斜杠命令、subagent 与项目指令文件,检查隐藏字符、提示注入、危险命令与过宽权限四类问题,每项发现按高、中、低分级。扫描只报告,不修改任何文件。 + +应用运行期间会监视这些文件,出现新的发现时发送系统通知。 + +## 设置 + +设置页分为六节。「连接」列出本机 core 和已保存的远程 core,详见[连接远程 core](/zh-CN/docs/lite/remote-core)。「通用」设置语言、外观、菜单栏显示的内容(仅 macOS)、开机启动,以及提醒以系统通知发送、仅在应用内显示还是关闭。「网关监听」设置网关的访问范围(仅本机、所选网卡所在的局域网或所有网卡)、端口和放行网段。「日志保留」分别设置请求报文与请求记录的保留天数,以及报文的空间上限。「关于」显示版本、检查更新,并可生成诊断包,其中的密钥与地址均已脱敏。「卸载」还原所有已接管的客户端并取消开机启动,应在删除应用之前执行。 + +## 菜单栏、系统托盘与通知 + +macOS 菜单栏显示今日 token 与今日费用,订阅额度紧张时数字变橙、用完变红;设置里可以改为仅标识或仅数值。 + +点开是原生菜单:网关地址、最近一分钟的生成速度与状态、未读的提醒、今日的请求数、token 与费用、各订阅账号与 GLM Coding Plan 上游的额度与重置时间(积分制套餐在额度条下方显示剩余积分)、进行中的请求,以及切换手动选择策略组中的上游、复制网关地址和默认密钥、切换连接、检查更新等常用操作,不必先打开主界面。 + +Windows 上图标位于通知区域:悬停显示网关状态与今日 token、费用;左键打开主界面,右键打开同一份菜单,其中的额度条改为文字。 + +Linux 上图标位于系统托盘:点击打开同一份菜单,第一项为「打开主界面」,额度条同样改为文字。 + +以下情况会发送系统通知(macOS 与 Windows 使用原生通知,Linux 通过桌面环境的通知服务):网关停止转发或反复重启、与远程 core 的连接断开、订阅额度用完、账号登录失效或上游拒绝当前凭据、代理不通、配置文件未通过校验、工具调用命中切断类规则、客户端配置中出现可疑内容。自动检查到新版本时也以系统通知告知(见[更新](/zh-CN/docs/lite/install#更新))。上游无法连接时通常由回退上游承接,因此只记录在应用内。提醒可以整体设为系统通知、仅在应用内显示或关闭。标为已读的提醒不再计入铃铛上的数字,但在问题解决或清空列表之前仍留在列表中。 diff --git a/src/content/docs-lite/zh-CN/install.md b/src/content/docs-lite/zh-CN/install.md index 75b26c0..009fc86 100644 --- a/src/content/docs-lite/zh-CN/install.md +++ b/src/content/docs-lite/zh-CN/install.md @@ -1,6 +1,6 @@ # 安装与更新 -ThinkWatch Lite 支持 macOS 12 及以上版本的 Apple silicon 机型,Windows 10 21H2 及以上版本的 x64 与 ARM64 机型,以及 x86_64 与 aarch64 机型的 Linux。网关 ThinkWatch Core 随应用一同安装,无需另行安装其他组件。 +ThinkWatch Lite 支持 macOS 12 及以上(Apple silicon)、Windows 10 21H2 及以上(x64、ARM64)和 Linux(x86_64、aarch64)。网关 ThinkWatch Core 随应用一同安装,无需另行安装其他组件。 ## macOS:Homebrew diff --git a/src/content/docs-lite/zh-CN/overview.md b/src/content/docs-lite/zh-CN/overview.md index 8eb592a..60a606e 100644 --- a/src/content/docs-lite/zh-CN/overview.md +++ b/src/content/docs-lite/zh-CN/overview.md @@ -1,16 +1,24 @@ # ThinkWatch Lite -ThinkWatch Lite 是一款在本机运行 AI API 网关的桌面应用,支持 macOS、Windows 和 Linux,常驻 macOS 菜单栏、Windows 通知区域或 Linux 系统托盘。Claude Code、Codex 以及其他使用 Anthropic、OpenAI、Gemini 接口的客户端经由网关发出请求,应用记录每个请求的费用、由哪个上游处理及其原因,以及发出前被脱敏的密钥。 +ThinkWatch Lite 是 Claude Code、Codex 等 AI 客户端的本地网关,支持 macOS、Windows 与 Linux。客户端只需接入一次,此后在网关中更换上游与模型,无需改动客户端。每个请求的费用与去向都有记录,发出前可替换其中的 API 密钥。 -网关本体是 [ThinkWatch Core](/zh-CN/docs/core)。它随应用在本机运行,也可以部署在 Linux 服务器上,由应用远程连接,参见[连接远程 core](/zh-CN/docs/lite/remote-core)。 +> 支持 macOS 12 及以上(Apple silicon)、Windows 10 21H2 及以上(x64、ARM64)和 Linux(x86_64、aarch64),可用 Homebrew、磁盘映像、Windows 安装程序或 AppImage [安装](/zh-CN/docs/lite/install),安装后自动更新。 -> 支持 macOS 12 及以上版本的 Apple silicon 机型、Windows 10 21H2 及以上版本的 x64 与 ARM64 机型,以及 x86_64 与 aarch64 机型的 Linux,可通过 Homebrew、磁盘映像、Windows 安装程序或 AppImage [安装](/zh-CN/docs/lite/install),并由应用自动更新。 +## 要点 + +- **一次接入,随时切换。** 七款客户端可一键指向网关,写入前预览改动并备份原文件;Cursor、Continue 与 Antigravity CLI 提供配置说明。 +- **发出前替换密钥。** 出站脱敏、工具调用审查、隐藏字符检测、内容过滤与输出长度限制作用于每个请求,每项可设为关闭、观察或拦截。 +- **扫描 MCP、技能与钩子。** 八款客户端的 MCP 服务器并列显示,并扫描客户端配置中的隐藏字符、提示注入、危险命令与过宽权限。 +- **每个请求都可追溯。** 命中的规则、每一次尝试、格式转换与费用都有记录,也可以重放到另一个上游对比。 +- **路由与故障转移。** 按模型、工具、图片等条件分流;回答开始前上游出错时换用下一个,同一会话固定使用同一上游。 +- **多种上游。** API 密钥、Amazon Bedrock、ChatGPT 与 Z.ai 账号、中转服务与本机模型,Anthropic、OpenAI、Gemini 接口之间自动转换。 +- **费用如实计算。** 估算的金额单独标注,无法计价的请求单独计数,不按零计入。 ## 页面 | 页面 | 内容 | |---|---| -| 概览 | 一段时间内的 token、费用与请求数,按模型的趋势,缓存、延迟与安全检查结果 | +| 概览 | 一段时间内的 token、费用与请求数,按模型的趋势,缓存、延迟、生成速度与安全检查结果 | | 流量 | 逐条请求或按会话归组的请求,以及每个请求的详情 | | 客户端 | 把客户端指向网关,以及还原 | | 密钥 | 客户端连接网关所用的密钥,及其路由与限制 | @@ -20,97 +28,13 @@ ThinkWatch Lite 是一款在本机运行 AI API 网关的桌面应用,支持 m | MCP | 各客户端的 MCP 服务器、技能与钩子,以及配置扫描 | | 设置 | 连接、语言、外观、菜单栏、提醒、网关监听、日志保留、更新与卸载 | -## 用量与费用 - -按实时、24 小时、7 天、30 天或自定义区间统计 token、费用与请求数,并与上一个同等区间对比;按模型分层显示趋势,并按用量列出模型排行。概览页还给出缓存命中率与缓存带来的净节省、按模型和按上游统计的首字节延迟分位,以及各项安全防护在该区间内的检查结果。 - -实测费用、估算费用与无法计价的请求分别列出:估算的金额(例如客户端在响应结束前断开的请求)会标明是估算,模型没有价格的请求单独计数,不按零计入。费用按网关每日更新的 LiteLLM 公开价格数据计算,也可以使用自定义价目表。 - -## 请求与会话 - -流量页逐条列出请求的密钥、模型、上游、首字节延迟、总耗时、token 与费用,新请求实时加入。列表可按密钥、上游、失败、无法计价或关键词筛选,也可以按会话归组,把同一项任务的请求放在一起查看。 - -请求详情给出命中的规则、经过的策略组、每一次尝试的状态与耗时,以及向下一个上游的故障转移。客户端与上游的 API 格式不同时,请求在 Anthropic Messages、OpenAI Chat Completions、OpenAI Responses 与 Gemini 之间自动转换,无法转换的字段会逐一列出。详情中还有请求体、响应体,以及计算费用所依据的用量。已结束的请求可以原样重放到另一个上游,并排对比两次的结果。 - -请求体与响应体在显示之前即已遮蔽,界面上不会出现密钥原文。 - -## 路由与故障转移 - -每把密钥对应一条路由。路由中的规则按顺序与请求的模型、密钥、客户端 API 格式、输入 token、`max_tokens`、工具数量、图片、扩展思考、流式与提示缓存等条件匹配,由第一条命中的规则决定去向:交给某个上游或策略组,或者附带说明直接拒绝。 - -策略组按顺序、手动选择、轮询、延迟最低或费用最低选用成员,某个上游不可用时换用下一个。轮询时,会话粘滞使同一会话固定使用同一上游,提示缓存因此持续有效;该选项默认开启,可以按策略组关闭。客户端自行发出的辅助请求(如健康检查、预热、生成标题)可以由网关在本地应答且不产生费用,也可以直接转发,或交给路由规则处理,由规则把它们转给费用更低的上游。 - -试算按给定的密钥、模型、客户端格式与请求条件,说明请求会被转发到哪里:命中的规则、前面每条规则未命中的原因、将依次尝试的上游,以及途中的格式转换。试算不发出请求,也不产生费用。 - -## 上游 - -支持 Anthropic、OpenAI、Gemini、DeepSeek 等 API 密钥上游与任意兼容端点,在应用内登录的 ChatGPT 与 Z.ai 账号,OpenRouter 等中转服务,以及 Ollama 或其他兼容 OpenAI 接口的本机模型服务。ChatGPT 账号与 Z.ai、BigModel 的 GLM Coding Plan 显示额度及重置时间。 - -上游可以经 HTTP 或 SOCKS 出站代理连接,代理的连通性与认证可以在上游页检查。每个上游按 token 计费或不计费。价格来自默认价目表(每日更新的 LiteLLM 公开价格数据),也可以使用在其基础上设置倍率与单个模型价格的自定义价目表。 - -## 安全 - -五项防护作用于经过网关的每个请求。防护是全局的:档位与规则对所有上游、所有密钥一致。 - -| 防护 | 检查什么 | 拦截时 | -|---|---|---| -| 出站脱敏 | 请求发出前其中的 API 密钥、私钥、JWT 与连接串口令;开启相应规则后也包括内网地址与内部域名 | 替换为占位符,并在响应中还原 | -| 工具调用审查 | 上游返回的工具调用,按危险命令规则检查,例如下载并执行脚本 | 切断响应(仅限设为「切断」的规则) | -| 隐藏字符 | 客户端发来的内容(含工具结果)中的 Unicode 标签字符与双向控制符 | 拒绝请求 | -| 内容过滤 | 客户端发来的内容(含工具结果)中的词语与模式,例如要求忽略先前指令的语句 | 拒绝请求(仅限设为「拒绝」的规则) | -| 输出长度 | 单次回答的字符数 | 在上限处切断回答 | - -每项防护有关闭、观察、拦截三档。观察只检测并记录命中,不改变请求;除输出长度外,各项防护出厂均为观察。输出长度出厂为关闭,开启后默认上限为 100,000 个字符。内置规则可以逐条停用,也可以添加自定义规则。每一次命中都连同对应的请求记录在安全日志中。 - -## MCP 服务器、技能与钩子 - -MCP 页并列显示 Claude Code、Claude Desktop、Cursor、Codex、opencode、Zed、Antigravity CLI 与 DeepSeek Harness 中配置的 MCP 服务器,标出远程与第三方服务器;同名服务器在两个客户端中配置不同时,可以逐字段对比。服务器可以复制到其他客户端或移除,写入前显示改动,并完整备份原文件。 - -该页还列出钩子与技能,并扫描客户端配置、技能、钩子、斜杠命令、subagent 与项目指令,检查隐藏字符、提示注入、危险命令与过宽权限四类问题。扫描只报告,不修改任何文件。这些文件发生变化时会重新扫描,出现新发现时发送提醒。 - -## 客户端接管 - -Claude Code、Codex、opencode、Zed、Aider、Claude Desktop 与 DeepSeek Harness 可以在应用内一键指向网关。写入前先显示改动差异并完整备份原文件,只修改指向网关所需的设置,随时可以还原。Cursor、Continue 与 Antigravity CLI 提供逐步的手动配置说明。 - -Claude Desktop 通过其第三方推理模式指向网关,完成后需完全退出再重新打开,此模式下的对话与原有对话分开保存;由组织统一管理的 Claude Desktop 不做修改。opencode 接管时写入该密钥可用的模型,上游或路由变化后,客户端页提示更新模型列表。Codex 还原后,接管期间开启的会话仍可打开。 - -在 Windows 上,客户端页还列出各个 WSL 发行版。WSL 1 与使用 mirrored 网络模式的 WSL 2 中的 Claude Code 和 Codex 可以指向网关,经 127.0.0.1 访问,网关仍只监听本机。WSL 2 使用 NAT 网络时,客户端页可将其改为 mirrored 网络模式并重启 WSL;mirrored 网络模式需要 Windows 11 22H2 及以上。 - -## 密钥 - -客户端凭网关密钥连接网关。接管客户端时为它单独生成一把密钥,流量、费用与限制因此按客户端区分。每把密钥有各自的路由、客户端可见的模型范围与可选的并发上限。除默认密钥外,密钥可以停用;密钥也可以更换,更换后的新密钥会写入使用它的客户端的配置。 - -## 菜单栏与托盘 - -在 macOS 上,菜单栏上行显示今日 token,下行显示今日费用。订阅额度即将用完时数字变为橙色,用完后变为红色;也可以只显示标识或只显示数字。由于菜单栏放不下两行文字,这一块以位图形式绘制。 - -点开的菜单列出网关状态与输出速率、今日请求数、token 与费用、各订阅额度及其重置时间、进行中的请求,并提供打开主界面、复制网关地址、复制默认密钥、切换连接与检查更新等菜单项。 - -Windows 上图标位于通知区域:悬停显示网关状态与今日 token、费用;左键打开主界面,右键打开同一份菜单,其中的额度条改为文字。Linux 上图标位于系统托盘:点击打开同一份菜单,第一项为「打开主界面」。 - -## 系统通知 - -以下情况会发送系统通知: - -- 网关停止转发、无法启动或反复退出; -- 与远程 core 的连接断开; -- 订阅额度用完; -- 上游的登录过期、上游拒绝当前凭据,或续期得到的新凭据未能写回配置; -- 出站代理无法连接; -- 手动修改的配置文件未通过校验,或监听地址的修改未能生效; -- 工具调用命中设为「切断」的规则; -- 配置扫描发现新的可疑内容。 - -上游无法连接只在应用内列出,不发送系统通知。问题解除后,对应的提醒随之撤回。提醒方式由一项设置统一决定:系统通知、仅在应用内显示或关闭。系统通知在各平台上均为原生通知:macOS 通知中心、Windows 通知与 Linux 桌面环境的通知服务。 - -## 界面语言 - -界面提供英文与简体中文,默认跟随系统语言,也可以在设置中另选。 - -## 与 ThinkWatch Core 的关系 +各页的详细说明见[功能详解](/zh-CN/docs/lite/features)。 -网关本体在 ThinkWatch Core 中实现。Lite 不包含路由、转发或计费逻辑,在 macOS 与 Linux 上通过 unix socket、在 Windows 上通过回环端口控制 Core;连接服务器上的 core 时则通过 TCP 端口。每条控制连接都经过以 `config.yaml` 中 `listen.control.key` 为密钥的握手加密与认证。参见[架构](/zh-CN/docs/lite/architecture)与[连接远程 core](/zh-CN/docs/lite/remote-core)。 +## 延伸阅读 -## 许可证 +- [安装与更新](/zh-CN/docs/lite/install) +- [连接远程 core](/zh-CN/docs/lite/remote-core):网关本体是 [ThinkWatch Core](/zh-CN/docs/core),随应用在本机运行,也可以部署在 Linux 服务器上。 +- [架构](/zh-CN/docs/lite/architecture):Lite 不含路由、转发与计费逻辑,通过加密的控制通道控制 Core。 +- [从源码构建](/zh-CN/docs/lite/run-from-source) ThinkWatch Lite 采用 MIT 许可证。 diff --git a/src/content/docs/_meta.ts b/src/content/docs/_meta.ts index 641383d..696eb70 100644 --- a/src/content/docs/_meta.ts +++ b/src/content/docs/_meta.ts @@ -135,11 +135,21 @@ export const products: Product[] = [ base: "/docs/lite", editUrl: "https://github.com/ThinkWatchProject/thinkwatch.github.io/tree/main/src/content/docs-lite", tagline: { - en: "The desktop app for a local AI API gateway on macOS, Windows and Linux: what it shows, how to install it, how to connect it to a core on a server, and how it is built.", - "zh-CN": "在本机运行 AI API 网关的 macOS、Windows 与 Linux 桌面应用:展示的内容、安装方法、连接服务器上的 core 的方法,以及构建方式。", + en: "The local gateway for Claude Code, Codex and other AI clients on macOS, Windows and Linux: what it does, how to install it, how to connect it to a core on a server, and how it is built.", + "zh-CN": "Claude Code、Codex 等 AI 客户端的本地网关,支持 macOS、Windows 与 Linux:功能、安装方法、连接服务器上的 core 的方法,以及构建方式。", }, docs: [ overview(), + { + slug: "features", + label: { en: "Features", "zh-CN": "功能详解" }, + locales: both, + group: "getStarted", + summary: { + en: "Page by page: usage and cost, traffic, client setup, keys, upstreams, routing, the five protections, MCP, settings, the menu bar and notifications.", + "zh-CN": "逐页说明:用量与费用、流量、客户端接管、密钥、上游、路由、五项防护、MCP、设置、菜单栏与通知。", + }, + }, { slug: "install", label: { en: "Install and update", "zh-CN": "安装与更新" }, diff --git a/src/i18n/pages/home.ts b/src/i18n/pages/home.ts index 509b81a..dcbe033 100644 --- a/src/i18n/pages/home.ts +++ b/src/i18n/pages/home.ts @@ -14,7 +14,7 @@ export const homeCopy = { sub: "ThinkWatch routes, inspects, and meters model requests and MCP tool calls. It is available as a self-hosted server for organizations and as a desktop application for individual developers.", doors: { teams: { label: "For organizations", name: "ThinkWatch Enterprise", text: "Self-hosted AI API and MCP gateway.", cta: "Deploy ThinkWatch" }, - machine: { label: "For individual developers", name: "ThinkWatch Lite", text: "Desktop application backed by a local gateway, for macOS on Apple silicon, Windows on x64 or ARM64, and Linux on x86_64 or aarch64.", cta: "Install Lite" }, + machine: { label: "For individual developers", name: "ThinkWatch Lite", text: "A local gateway for Claude Code, Codex and other AI clients, on macOS, Windows and Linux.", cta: "Install Lite" }, }, trace: { tag: "Sample trace", @@ -57,7 +57,7 @@ export const homeCopy = { { t: "AI API gateway", b: "OpenAI, Anthropic, Gemini, Azure OpenAI, and Bedrock behind one endpoint, with scoped virtual keys." }, { t: "MCP gateway with per-user identity", b: "Per-user OAuth and tokens, tool-level RBAC, and an audit log for every call." }, { t: "SSO and RBAC", b: "Five roles and support for any OIDC provider." }, - { t: "Audit logs, rate limits, and budgets", b: "Sliding rate limits and token budgets per user, key, or provider." }, + { t: "Audit logs, rate limits, and budgets", b: "Sliding-window request and token limits and spending budgets per user, API key or role." }, ], stepsLabel: "Inside a ThinkWatch request", steps: [ @@ -78,12 +78,12 @@ export const homeCopy = { }, lite: { eyebrow: "ThinkWatch Lite · For individual developers", - title: "A local AI gateway for macOS, Windows and Linux", + title: "A local gateway for Claude Code, Codex and other AI clients", points: [ - { t: "Cost reporting", b: "Every request is priced from the price table; estimated amounts are marked as such, and requests that cannot be priced are counted as unpriced rather than as zero." }, - { t: "Request routing", b: "The matched rule, the group, and every upstream attempt, with a dry run for rules before any traffic." }, - { t: "Security checks", b: "Five guards on requests and responses, from secret redaction and tool-call inspection to an output limit, and a scan of client configurations, skills and hooks for hidden characters, prompt injection and dangerous commands." }, - { t: "Remote core", b: "The app can also connect to ThinkWatch Core running on a server, over an encrypted control channel." }, + { t: "Connect once, switch freely", b: "Each client is pointed at the gateway once; upstreams and models then change in the gateway, with no client to reconfigure or restart." }, + { t: "Keys replaced before sending", b: "Outbound redaction can replace credentials before a request leaves, tool-call inspection can cut off dangerous commands, and MCP servers, skills and hooks are scanned." }, + { t: "Every request traceable", b: "The matched rule, each upstream attempt and the cost of every request, with replay against another upstream." }, + { t: "Costs stated as they are", b: "Estimated amounts are marked, and requests without a price are counted separately rather than as zero." }, ], pills: ["Available", "macOS · Apple silicon", "Windows · x64 · ARM64", "Linux · x86_64 · aarch64", "MIT"], shotAlt: @@ -138,7 +138,7 @@ export const homeCopy = { sub: "ThinkWatch 对模型请求与 MCP 工具调用进行路由、检查与计量,提供面向组织的自托管服务端,以及面向个人开发者的桌面应用。", doors: { teams: { label: "面向组织", name: "ThinkWatch 企业版", text: "自托管的 AI API 与 MCP 网关。", cta: "部署 ThinkWatch" }, - machine: { label: "面向个人开发者", name: "ThinkWatch Lite", text: "基于本地网关的桌面应用,支持 Apple silicon 机型的 macOS、x64 与 ARM64 机型的 Windows,以及 x86_64 与 aarch64 机型的 Linux。", cta: "安装 Lite" }, + machine: { label: "面向个人开发者", name: "ThinkWatch Lite", text: "Claude Code、Codex 等 AI 客户端的本地网关,支持 macOS、Windows 与 Linux。", cta: "安装 Lite" }, }, trace: { tag: "示例追踪", @@ -181,7 +181,7 @@ export const homeCopy = { { t: "AI API 网关", b: "OpenAI、Anthropic、Gemini、Azure OpenAI 和 Bedrock 通过统一入口接入,并支持限定范围的虚拟密钥。" }, { t: "基于用户身份的 MCP 网关", b: "每位用户使用本人的 OAuth 凭据,支持工具级 RBAC,每次调用均记录审计日志。" }, { t: "SSO 与 RBAC", b: "五级角色,支持任意 OIDC 身份提供方。" }, - { t: "审计、限流与预算", b: "按用户、密钥或上游服务商设置滑动窗口限流和 token 预算。" }, + { t: "审计、限流与预算", b: "按用户、API 密钥或角色设置滑动窗口限流与费用预算。" }, ], stepsLabel: "一次 ThinkWatch 请求的内部", steps: [ @@ -202,12 +202,12 @@ export const homeCopy = { }, lite: { eyebrow: "ThinkWatch Lite · 面向个人开发者", - title: "适用于 macOS、Windows 与 Linux 的本地 AI 网关", + title: "Claude Code、Codex 等 AI 客户端的本地网关", points: [ - { t: "费用报告", b: "每个请求按价目表计算费用;估算的金额另行标注,无法计价的请求单独计数,不按零计入。" }, - { t: "请求路由", b: "展示每个请求命中的规则、策略组与每一次尝试,改规则前可以先试算。" }, - { t: "安全检查", b: "五项防护作用于请求与响应,包括出站脱敏、工具调用审查与输出长度限制等;另可扫描客户端配置、技能与钩子,检查隐藏字符、提示注入与危险命令。" }, - { t: "连接远程 core", b: "应用也可以通过加密的控制通道,连接运行在服务器上的 ThinkWatch Core。" }, + { t: "一次接入,随时切换", b: "客户端只需接入一次,此后在网关中更换上游与模型,客户端无需改配置或重启。" }, + { t: "发出前替换密钥", b: "出站脱敏可在请求发出前替换其中的凭据,工具调用审查可切断危险命令,MCP 服务器、技能与钩子也会被扫描。" }, + { t: "每个请求都可追溯", b: "每个请求命中的规则、尝试过的上游与费用都有记录,也可以重放到另一个上游对比。" }, + { t: "费用如实计算", b: "估算的金额单独标注,无法计价的请求单独计数,不按零计入。" }, ], pills: ["已发布", "macOS · Apple silicon", "Windows · x64 · ARM64", "Linux · x86_64 · aarch64", "MIT"], shotAlt: diff --git a/src/i18n/pages/lite.ts b/src/i18n/pages/lite.ts index c2fe4a7..f52c65d 100644 --- a/src/i18n/pages/lite.ts +++ b/src/i18n/pages/lite.ts @@ -20,84 +20,78 @@ const overviewAlt = { export const liteCopy = { en: { meta: { - title: "ThinkWatch Lite — Local AI API gateway for macOS, Windows and Linux", + title: "ThinkWatch Lite — Local gateway for Claude Code, Codex and other AI clients", description: - "Desktop app for a local AI API gateway on macOS, Windows and Linux. Shows what Claude Code, Codex and other OpenAI and Anthropic clients cost, where each request was routed and what was redacted, and can connect to ThinkWatch Core on a server. MIT License.", + "A local gateway for Claude Code, Codex and other AI clients on macOS, Windows and Linux. Connect each client once and switch upstreams freely, replace API keys before a request leaves, stop dangerous tool calls, and see the cost and route of every request. MIT License.", }, hero: { eyebrow: "ThinkWatch Lite · For individual developers", - titleA: "A local AI API gateway ", - titleHighlight: "for macOS, Windows and Linux", - sub: "Claude Code, Codex and other clients of the Anthropic, OpenAI and Gemini APIs send their requests through a local gateway, and the app records what each request cost, which upstream served it and why, and which credentials were redacted before it was sent. The gateway can also run on a server; the app then connects to it over an encrypted control channel.", + titleA: "A local gateway for ", + titleHighlight: "Claude Code, Codex and other AI clients", + sub: "Each client is connected once; after that, upstreams and models change without touching its configuration. Every request is recorded with its cost and route, and the API keys in it can be replaced before it leaves the machine. For macOS, Windows and Linux, under the MIT License.", ctaSecondary: "Other platforms and installation methods", shotAlt: overviewAlt.en, }, status: { badge: "Available", - body: "macOS 12 or later on Apple silicon, installed with Homebrew or a disk image. Windows 10 21H2 or later on x64 or ARM64, installed with an installer. Ubuntu 22.04, Debian 12, Fedora 36 or later on x86_64 or aarch64, as an AppImage. The interface is in English and Simplified Chinese, and the app updates itself; a Homebrew installation is updated through Homebrew.", + items: ["macOS 12+ · Apple silicon", "Windows 10 21H2+ · x64 · ARM64", "Linux · x86_64 · aarch64", "English · 简体中文", "Updates itself", "MIT"], }, features: { - eyebrow: "Features by page", + eyebrow: "Features", items: [ { - id: "overview", - title: "Usage and cost", - body: "Tokens, cost and requests live or for the last 24 hours, 7 days, 30 days or a custom range, each compared with the period before, with a trend by model and the models ranked by usage. The page also reports the cache hit rate and the net savings from caching, time-to-first-byte percentiles by model and by upstream, and what each security protection found. Measured and estimated costs are marked as such, and requests without a price are counted separately rather than as zero.", - alt: overviewAlt.en, - }, - { - id: "traffic", - title: "Requests and sessions", - body: "Every request with its key, model, upstream, time to first byte, total time, tokens and cost, listed as it arrives and filtered by key, upstream, failures, unpriced requests or text; requests can also be grouped into sessions. A request's details show the rule it matched, every attempt and any failover, the conversion between API formats with the fields that could not be carried over, the request and response bodies with credentials masked, and the usage its cost was calculated from. A finished request can be replayed against another upstream and the results compared side by side.", - alt: "The Traffic page: the latest requests with their key, model, upstream, latency, tokens and cost, among them one in progress, requests converted between API formats, one with two credentials redacted, one blocked, one answered locally, and failed and canceled requests", + id: "clients", + title: "Connect once, switch freely", + body: "Claude Code, Codex, opencode and four other clients are pointed at the gateway in one step, with the change previewed, the original file backed up and a restore always available; on Windows, Claude Code and Codex inside WSL as well. From then on, switching upstreams happens in the gateway, with no client to reconfigure or restart.", + alt: "The Clients page: Claude Code and Codex connected, each with its own key and its requests over the last 24 hours; opencode not connected; Cursor set up by hand and in use; Continue and Antigravity CLI not yet set up; Zed and Aider not detected", }, { - id: "clients", - title: "Client setup", - body: "Claude Code, Codex, opencode, Zed, Aider, Claude Desktop and DeepSeek Harness are pointed at the gateway from the app. Each change is shown as a diff before it is written, the original file is backed up, only the settings that point the client at the gateway are changed, and any client can be restored. On Windows, Claude Code and Codex inside WSL are pointed at the gateway as well, under WSL 1 or under WSL 2 with mirrored networking. Cursor, Continue and Antigravity CLI come with step-by-step instructions. Each client's requests over the last 24 hours are listed beside it.", - alt: "The Clients page: Claude Code and Codex in use, each with its own key and its requests over the last 24 hours; opencode not connected; Cursor set up by hand and in use; Continue and Gemini CLI not set up; Zed and Aider not detected", + id: "security", + title: "Keys replaced before sending, dangerous commands stopped", + body: "Outbound redaction swaps API keys, private keys, JWTs and connection-string passwords for placeholders before a request leaves and restores them in the response, so a relay never sees the real values. Tool-call inspection cuts off download-and-run commands and similar calls before the client can run them, and hidden characters and prompt injection can be refused. The five protections start in Observe, recording without changing anything, and each switches to Enforce on its own.", + alt: "The Security page log: credentials replaced before a request left, one of them matched by a custom rule; a download-and-run tool call cut off; and hidden characters, a delete command and an injected instruction recorded, each with the key, client, model and upstream of its request", }, { - id: "keys", - title: "A key for each client", - body: "Clients connect with gateway keys, and connecting a client creates a key for it, so that traffic and cost are attributed to that client. Each key has its own route, the set of models its client sees, an optional concurrency limit, and its requests and cost over the last 24 hours. A key can be disabled or rotated; a rotated key is written into the configuration of the client that uses it.", - alt: "The Keys page: four keys, default, claude-code, codex and cursor, with the client each belongs to, its route, the models its client sees, and its requests and cost over the last 24 hours", + id: "mcp", + title: "MCP servers, skills and hooks, scanned", + body: "The MCP servers of eight clients appear side by side, with third-party remote servers and inconsistent configurations marked, and can be copied or removed between clients. Client configuration, skills, hooks and project instructions are scanned for hidden characters, prompt injection, dangerous commands and overly broad permissions, and a new finding raises a notification.", + alt: "The MCP page: the MCP servers configured in Claude Code, Claude Desktop, Cursor, Codex, opencode, Antigravity CLI and Zed side by side, with remote third-party servers and a server configured differently in two clients marked; one high, one medium and one low finding in 11 scanned files", }, { - id: "upstreams", - title: "Upstreams and prices", - body: "API-key upstreams such as Anthropic, OpenAI, Gemini, DeepSeek or any compatible endpoint; ChatGPT and Z.ai accounts signed in from the app; relays such as OpenRouter; and local models served by Ollama. The usage limits of ChatGPT accounts and of GLM Coding Plan keys on Z.ai and BigModel are shown with their reset times. When a client and an upstream use different API formats, requests are converted between Anthropic Messages, OpenAI Chat Completions, OpenAI Responses and Gemini. Upstreams can connect through an outbound proxy. Costs follow LiteLLM's public prices, refreshed daily, or a custom price sheet with a multiplier and prices for individual models.", - alt: "The Upstreams page: seven upstreams, including Anthropic, a relay priced with a discounted price sheet, a ChatGPT Plus account with 58% of its 5-hour limit used, OpenRouter through a proxy, DeepSeek, Gemini and a local Ollama, each with its billing, its requests and cost over 24 hours and its median time to first byte; tabs for proxies and price sheets", + id: "traffic", + title: "Every request, traceable", + body: "A request shows the rule it matched, each upstream it tried, any conversion between API formats and how its cost was calculated. A finished request can be replayed against another upstream and the two answers compared side by side.", + alt: "The Traffic page: each request with its key, model, upstream, time to first token, total time, tokens and cost, with marks for converted formats, redacted keys and a blocked request, and one request answered locally by the gateway", }, { id: "routing", - title: "Routing and failover", - body: "Each key uses a route, whose rules are checked in order against the model, the client's API format, input tokens, max_tokens, tools, images, extended thinking and other properties of a request. A rule forwards the request to an upstream or a group, or refuses it. A group picks its upstreams in order, by manual choice, in rotation, by lowest latency or by lowest cost, and moves on to the next when one is unavailable; in rotation, sticky sessions, on by default, keep each session on one upstream so that its prompt cache stays valid. Auxiliary requests that clients send on their own, such as title generation or warm-up, can be answered locally.", - alt: "The Routing page: a map of how four keys lead through three routes and the groups main and budget to seven upstreams, with one rule refusing requests; below it, each route with its keys and its rules in order", + title: "Routing by rule, with failover", + body: "Rules send requests to different upstreams by model, tools, images, extended thinking and more. When an upstream fails before the answer begins, the next one takes over, and each session stays on one upstream so its prompt cache keeps hitting. Auxiliary requests such as title generation and warm-ups can be answered locally without using any quota.", + alt: "The Routing page: a map from keys through routes and groups to upstreams, and the routes with the rules each applies in order", }, { - id: "dry-run", - title: "Dry run", - body: "A dry run takes a key, a model, a client format and the properties of a request, and shows where the request would go: the rule that matched and why the rules before it did not, each upstream that would be tried, and any format conversion on the way. Nothing is sent and no cost is incurred.", - alt: "A routing dry run for the cursor key requesting claude-sonnet-5 in the OpenAI Chat Completions format: the gemini rule did not match because the model must be gemini-*, the catch-all rule matched, and the request goes to the budget group, ordered by lowest cost, which tries relay and then anthropic, each with a conversion from OpenAI Chat Completions to Anthropic Messages", + id: "upstreams", + title: "Any upstream, any API format", + body: "API keys, Amazon Bedrock, ChatGPT and Z.ai accounts, relays such as OpenRouter and local Ollama models all serve as upstreams, with subscription quotas and reset times shown. Requests are converted between the Anthropic, OpenAI and Gemini APIs, so Codex can also use models that only speak Chat Completions.", + alt: "The Upstreams page: API-key upstreams for Anthropic, DeepSeek and Gemini, a relay priced with its own price sheet, a ChatGPT Plus account with 58% of its 5-hour limit used, OpenRouter through a proxy and a local Ollama set to free, each with its requests, cost and latency over 24 hours", }, { - id: "security", - title: "Five security protections", - body: "Five protections apply to every request through the gateway, the same for every upstream and key. Outbound redaction finds API keys, private keys, JWTs and connection-string passwords in a request; tool-call inspection checks the tool calls an upstream returns for dangerous commands; hidden characters and the content filter check what a client sends, tool results included, for invisible Unicode and prompt-injection phrases; the output limit measures the length of an answer. Each protection is set to Off, Observe or Enforce. Observe records matches without changing anything and is the initial setting for all but the output limit, which starts off. Enforce acts on a match: redaction replaces the credentials before the request is sent and restores them in the response, a dangerous tool call or an answer over the limit is cut off, and a request carrying hidden characters or a phrase set to be refused is refused. Built-in rules can be turned off one by one, custom rules can be added, and every match appears in the security log.", - alt: "The Security page log: redaction and tool-call inspection enforcing, hidden characters and the content filter observing, the output limit off, and eight matches in the last 24 hours, including an AWS access key ID replaced, a download-and-run command cut off, a custom customer-id rule, Unicode tag characters hidden in a tool result, and a prompt-injection phrase recorded", + id: "overview", + title: "Costs stated as they are", + body: "Tokens, cost, cache savings, time to first token and generation speed, by model and by upstream. Estimated amounts are marked, and requests without a price are counted separately instead of as zero; prices follow LiteLLM's public list, refreshed daily, or a custom price sheet.", + alt: overviewAlt.en, }, { - id: "mcp", - title: "MCP servers, skills and hooks", - body: "The MCP servers configured in Claude Code, Claude Desktop, Cursor, Codex, opencode, Zed, Antigravity CLI and DeepSeek Harness appear side by side, with remote and third-party servers marked and differences between clients highlighted; a server can be copied to another client or removed, with the change shown before it is written. Hooks and skills are listed as well. Client configuration, skills, hooks, slash commands, subagents and project instructions are scanned for hidden characters, prompt injection, dangerous commands and overly broad permissions. Findings are reported without changing any file, and a new finding raises a system notification.", - alt: "The MCP page: five MCP servers across Claude Code, Claude Desktop, Cursor, Codex, opencode and Zed, with context7 and linear marked as third-party remote servers and github marked as configured differently between clients; one high, one medium and one low finding from 11 scanned files", + id: "dry-run", + title: "Dry run before changing rules", + body: "A dry run shows which rule a request would match, why the rules before it did not, and which upstreams would be tried in turn. Nothing is sent and nothing is charged.", + alt: "A routing dry run: a request from the cursor key for claude-sonnet-5 in the OpenAI Chat Completions format does not match the gemini rule, which says why, matches the catch-all rule and goes to the lowest-cost group, which tries relay and then anthropic, converting the request to Anthropic Messages", }, { - id: "settings", - title: "Settings and languages", - body: "The interface is available in English and Simplified Chinese and follows the system language unless another is chosen. Settings also cover the connection to a local or remote core, the appearance, what the menu bar item shows, launch at login, notifications, the gateway's listening address and the networks allowed to reach it, how long request records and bodies are kept, updates, and a full uninstall that restores every connected client and turns off launch at login.", - alt: "The Settings page: connections to this Mac, which is current, and to two remote cores, homelab and build-server; the connection used at startup; the interface language set to follow the system; the appearance; and what the menu bar item shows", + id: "keys", + title: "A key for each client", + body: "Connecting a client gives it a key of its own, so traffic and cost are counted per client. Each key has its own route, visible models and concurrency limit, and a rotated key is written into its client's configuration.", + alt: "The Keys page: the default key and one key each for Claude Code, Codex and Cursor, with the route each key uses, the models it may use, and its requests and cost over the last 24 hours", }, ], }, @@ -105,14 +99,13 @@ export const liteCopy = { eyebrow: "Remote core", title: "ThinkWatch Core on a server", body: [ - "The gateway can also run on a Linux server, where twcore runs as a systemd service and serves clients across the network. The app connects to it through the server's remote control port and shows that server's traffic, cost and configuration in the same pages.", - "A connection is added in Settings › Connection with the server's address, its control port and the key that twcore control-key prints. Before switching to a connection, the app tests it: it completes the handshake and compares versions, since the server has to run the core version the app requires; when they differ, sudo twcore upgrade --version --restart on the server installs that version, newer or older. The app connects to one core at a time; while it is connected to a server, the core on the local computer stops and its data is kept. The Clients and MCP pages still act on the computer the app runs on, and the Clients page can point that computer's clients at the server's gateway.", - "Every control connection, local or remote, is encrypted and authenticated by a Noise handshake keyed by listen.control.key in config.yaml; no certificates are involved. The app keeps connection keys in a file in its data directory that only the current user can read.", + "The gateway also runs on a Linux server as a systemd service, serving clients across the network. The app connects to it and shows the server's traffic, cost and configuration in the same pages, and the clients on the computer can be pointed at the server's gateway in one step.", + "The control connection is encrypted and authenticated by a Noise handshake, with no certificates involved. The app connects to one core at a time, and the local data stays as it was while it is connected elsewhere.", ], serverDocs: "Server deployment guide", docs: "Connecting to a remote core", addAlt: - "Adding a remote connection: the name homelab, the address 192.168.1.40, control port 24817 and the key, with a successful test that reports core 0.49.0 and the gateway at 192.168.1.40:8788", + "Adding a remote connection: the name homelab, the address 192.168.1.40, the control port and the key, with a successful test that reports the core version and the gateway at 192.168.1.40:8788", figures: [ { id: "remote-switcher", @@ -129,12 +122,12 @@ export const liteCopy = { menubar: { eyebrow: "Outside the main window", title: "Menu bar, tray and notifications", - body: "On macOS, the menu bar item shows today's tokens and cost, which turn orange when a subscription quota is nearly used up and red once it has run out; it can also show only the icon or only the numbers. Its menu lists the gateway's state and output rate, today's requests, tokens and cost, each subscription quota with its reset time, and the requests in progress, with actions to open the main window, copy the gateway address or the default key and switch connection. On Windows the icon sits in the notification area and on Linux in the system tray, with the same menu in text form.", + body: "The macOS menu bar shows today's tokens and cost, in orange when a subscription quota is nearly used up and red once it has run out. Its menu gives the gateway's state, each quota with its reset time and the requests in progress, and copies the gateway address or default key without opening the window. Windows and Linux have the same menu in the tray.", notices: - "System notifications report a gateway that stopped forwarding, a lost connection to a remote core, a subscription quota that ran out, a credential that expired, was rejected or could not be saved, an unreachable proxy, configuration that did not take effect, a tool call that matched a rule set to cut off, and suspicious content newly found in client configuration. An unreachable upstream is listed in the app without a system notification. One setting sends these notices as system notifications, keeps them in the app, or turns them off.", + "A system notification reports a gateway that stopped forwarding, a lost remote connection, a quota that ran out, a credential that stopped working, an unreachable proxy, configuration that did not take effect, a dangerous tool call that was cut off, and suspicious content in client configuration.", chipAlt: "The menu bar item: the ThinkWatch mark with today's 13.3M tokens above today's cost, $9.34", menuAlt: - "The menu bar menu: the gateway running at 127.0.0.1:8788 at 64 tokens per second; the chatgpt quota at 58% of its 5-hour window, resetting in 2 hours, and 31% of its weekly window, resetting in 3 days; today's 204 requests with 4 failed, 13.3M tokens and $9.34 in cost; one Claude Code request in progress; and items to copy the gateway address or the default key, undo the last configuration change, open the app, switch connection, open settings and check for updates", + "The menu bar menu: the gateway's address, output speed and state; the ChatGPT account's 5-hour and weekly quotas with their reset times; today's requests, tokens and cost; the request in progress; and items to open the app, copy the gateway address or the default key, switch connection, open settings and check for updates", }, built: { eyebrow: "Architecture", @@ -197,84 +190,78 @@ export const liteCopy = { }, "zh-CN": { meta: { - title: "ThinkWatch Lite — 适用于 macOS、Windows 与 Linux 的本地 AI API 网关", + title: "ThinkWatch Lite — Claude Code、Codex 等 AI 客户端的本地网关", description: - "在本机运行 AI API 网关的桌面应用,支持 macOS、Windows 与 Linux。记录 Claude Code、Codex 以及其他使用 OpenAI、Anthropic 接口的客户端每个请求的费用、所用的上游与发出前被脱敏的密钥,也可以连接部署在服务器上的 ThinkWatch Core。采用 MIT 许可证。", + "Claude Code、Codex 等 AI 客户端的本地网关,支持 macOS、Windows 与 Linux。客户端接入一次即可随时切换上游,请求发出前可替换其中的 API 密钥、切断危险的工具调用,每个请求的费用与去向都有记录。MIT 开源。", }, hero: { eyebrow: "ThinkWatch Lite · 面向个人开发者", - titleA: "适用于 macOS、Windows 与 Linux 的", - titleHighlight: "本地 AI API 网关", - sub: "Claude Code、Codex 以及其他使用 Anthropic、OpenAI、Gemini 接口的客户端经由本机的网关发出请求,应用记录每个请求的费用、由哪个上游处理及其原因,以及发出前被脱敏的密钥。网关也可以部署在服务器上,此时应用通过加密的控制通道连接服务器上的 ThinkWatch Core。", + titleA: "Claude Code、Codex 等 AI 客户端的", + titleHighlight: "本地网关", + sub: "客户端只需接入一次,此后更换上游或模型无需改动客户端配置。每个请求的费用与去向都有记录,发出前可替换其中的 API 密钥。支持 macOS、Windows 与 Linux,MIT 开源。", ctaSecondary: "其他平台与安装方式", shotAlt: overviewAlt["zh-CN"], }, status: { badge: "已发布", - body: "支持 macOS 12 及以上版本的 Apple silicon 机型,可通过 Homebrew 或磁盘映像安装;支持 Windows 10 21H2 及以上版本的 x64 与 ARM64 机型,通过安装程序安装;支持 Ubuntu 22.04、Debian 12、Fedora 36 及以上版本的 x86_64 与 aarch64 机型,以 AppImage 发布。界面提供英文与简体中文;应用自动更新,通过 Homebrew 安装的由 Homebrew 更新。", + items: ["macOS 12+ · Apple silicon", "Windows 10 21H2+ · x64 · ARM64", "Linux · x86_64 · aarch64", "English · 简体中文", "自动更新", "MIT"], }, features: { - eyebrow: "各页功能", + eyebrow: "功能", items: [ { - id: "overview", - title: "用量与费用", - body: "按实时、24 小时、7 天、30 天或自定义区间统计 token、费用与请求数,并与上一个同等区间对比;按模型分层显示趋势,并按用量列出模型排行。页面还给出缓存命中率与缓存带来的净节省、按模型和按上游统计的首字节延迟分位,以及各项安全防护的检查结果。实测与估算的费用分别标明,无法计价的请求单独计数,不按零计入。", - alt: overviewAlt["zh-CN"], - }, - { - id: "traffic", - title: "请求与会话", - body: "逐条列出请求的密钥、模型、上游、首字节延迟、总耗时、token 与费用,新请求实时加入,可按密钥、上游、失败、无法计价或关键词筛选,也可以按会话归组。请求详情给出命中的规则、每一次尝试与故障转移、API 格式转换及无法转换的字段、遮蔽凭据后的请求体与响应体,以及计算费用所依据的用量。已结束的请求可以原样重放到另一个上游,并排对比结果。", - alt: "流量页:最近的请求及其密钥、模型、上游、延迟、token 与费用,其中有一条进行中的请求、经格式转换的请求、脱敏 2 处凭据的请求、被拦截的请求、本地应答的请求,以及失败和已取消的请求", + id: "clients", + title: "一次接入,随时切换", + body: "一键接入 Claude Code、Codex、opencode 等七款客户端,写入前预览改动、备份原文件,随时可以还原;Windows 上 WSL 中的 Claude Code 与 Codex 同样支持。此后切换上游只在网关中完成,客户端无需改配置或重启。", + alt: "客户端页:Claude Code 与 Codex 已接管,各用一把密钥,并列出最近 24 小时的请求;opencode 尚未接管;Cursor 已手动配置并在使用;Continue 与 Antigravity CLI 尚未配置;Zed 与 Aider 未检测到", }, { - id: "clients", - title: "客户端接管", - body: "Claude Code、Codex、opencode、Zed、Aider、Claude Desktop 与 DeepSeek Harness 可以在应用内一键指向网关。写入前先显示改动差异,原文件完整备份,只修改指向网关所需的设置,随时可以还原。在 Windows 上,WSL 中的 Claude Code 与 Codex 同样可以指向网关,适用于 WSL 1 和使用 mirrored 网络模式的 WSL 2。Cursor、Continue 与 Antigravity CLI 提供逐步的手动配置说明。每个客户端旁列出其最近 24 小时的请求。", - alt: "客户端页:Claude Code 与 Codex 已接管,各用一把密钥,并列出最近 24 小时的请求;opencode 尚未接管;Cursor 已手动配置并在使用;Continue 与 Gemini CLI 尚未配置;Zed 与 Aider 未检测到", + id: "security", + title: "发出前替换密钥,拦下危险命令", + body: "出站脱敏在请求发出前把 API 密钥、私钥、JWT 与连接串口令换成占位符,并在响应中还原,中转服务看不到原值。工具调用审查在客户端执行之前切断下载即执行等危险命令,隐藏字符与提示注入也可以直接拒绝。五项防护出厂只记录、不改动请求,逐项切换到拦截即可生效。", + alt: "安全页日志:请求发出前替换的凭据(其中一条由自定义规则命中)、被切断的下载即执行工具调用,以及记录在案的隐藏字符、删除命令与注入指令,每条都注明所属请求的密钥、客户端、模型与上游", }, { - id: "keys", - title: "每个客户端一把密钥", - body: "客户端凭网关密钥连接网关。接管客户端时为它单独生成一把密钥,流量与费用因此按客户端区分。每把密钥有各自的路由、客户端可见的模型范围、可选的并发上限,以及最近 24 小时的请求数与费用。密钥可以停用或更换,更换后的新密钥会写入使用它的客户端的配置。", - alt: "密钥页:default、claude-code、codex、cursor 四把密钥,列出各自所属的客户端、路由、可见模型,以及最近 24 小时的请求数与费用", + id: "mcp", + title: "扫描 MCP、技能与钩子", + body: "八款客户端的 MCP 服务器集中显示,标出第三方远程服务器与各客户端之间不一致的配置,可以在客户端之间复制或移除。客户端配置、技能、钩子与项目指令中的隐藏字符、提示注入、危险命令与过宽权限会被找出,出现新发现时发送通知。", + alt: "MCP 页:Claude Code、Claude Desktop、Cursor、Codex、opencode、Antigravity CLI 与 Zed 中配置的 MCP 服务器并列显示,标出第三方远程服务器与两个客户端间配置不一致的服务器;共扫描 11 个文件,发现高、中、低风险各一项", }, { - id: "upstreams", - title: "上游与价目表", - body: "支持 Anthropic、OpenAI、Gemini、DeepSeek 等 API 密钥上游与任意兼容端点,在应用内登录的 ChatGPT 与 Z.ai 账号,OpenRouter 等中转服务,以及 Ollama 提供的本机模型。ChatGPT 账号与 Z.ai、BigModel 的 GLM Coding Plan 显示额度及重置时间。客户端与上游的 API 格式不同时,请求在 Anthropic Messages、OpenAI Chat Completions、OpenAI Responses 与 Gemini 之间自动转换。上游可以经出站代理连接。费用按每日更新的 LiteLLM 公开价格计算,也可以使用设有倍率与单个模型价格的自定义价目表。", - alt: "上游页:七个上游,包括 Anthropic、按折扣价目表计价的中转、5 小时额度已用 58% 的 ChatGPT Plus 账号、经代理访问的 OpenRouter、DeepSeek、Gemini 与本机 Ollama,以及各自的计费方式、24 小时请求数与费用和首字节延迟中位数;另有代理与价目表两个标签", + id: "traffic", + title: "每个请求都可追溯", + body: "请求详情给出命中的规则、尝试过的每个上游、API 格式转换,以及费用的计算依据。已结束的请求可以重放到另一个上游,并排对比两次回答。", + alt: "流量页:逐条列出请求的密钥、模型、上游、首 token 时间、总耗时、token 与费用,标出格式转换、密钥脱敏与被拦截的请求,其中一条由网关在本地应答", }, { id: "routing", - title: "路由与故障转移", - body: "每把密钥对应一条路由,路由中的规则按顺序与请求的模型、客户端 API 格式、输入 token、max_tokens、工具、图片、扩展思考等条件匹配。规则把请求交给某个上游或策略组,或者直接拒绝。策略组按顺序、手动选择、轮询、延迟最低或费用最低选用成员,某个上游不可用时换用下一个;轮询时默认开启会话粘滞,同一会话固定使用同一上游,提示缓存因此持续有效。客户端自行发出的辅助请求(如生成标题、预热)可以由网关在本地应答。", - alt: "路由页:从四把密钥经三条路由与 main、budget 两个策略组到七个上游的链路图,其中一条规则直接拒绝请求;下方逐条列出每条路由的密钥与规则", + title: "按规则分流,失败自动换", + body: "按模型、工具、图片、扩展思考等条件把请求分给不同上游。回答开始前上游出错时自动换用下一个,同一会话固定使用同一上游,提示缓存保持有效。标题生成、预热等辅助请求可在本地应答,不占用额度。", + alt: "路由页:从密钥经路由与策略组到上游的链路图,下方逐条列出每条路由的规则", }, { - id: "dry-run", - title: "路由试算", - body: "试算按给定的密钥、模型、客户端格式与请求条件,说明请求会被转发到哪里:命中的规则、前面各条规则未命中的原因、将依次尝试的上游,以及途中的格式转换。不发出请求,也不产生费用。", - alt: "路由试算:cursor 密钥以 OpenAI Chat Completions 格式请求 claude-sonnet-5;gemini 规则因模型须为 gemini-* 而未命中,catch-all 规则命中,请求交给按费用最低排序的 budget 策略组,依次尝试 relay 与 anthropic,两者都将 OpenAI Chat Completions 转换为 Anthropic Messages", + id: "upstreams", + title: "多种上游,接口互转", + body: "API 密钥、Amazon Bedrock、ChatGPT 与 Z.ai 账号、OpenRouter 等中转服务以及本机 Ollama 均可作为上游,订阅额度与重置时间一并显示。Anthropic、OpenAI、Gemini 接口之间自动转换,Codex 也能使用只支持 Chat Completions 的模型。", + alt: "上游页:Anthropic、DeepSeek、Gemini 等 API 密钥上游,按自有价目表计价的中转,5 小时额度已用 58% 的 ChatGPT Plus 账号,经代理访问的 OpenRouter,以及设为免费的本机 Ollama,并列出各自 24 小时的请求数、费用与延迟", }, { - id: "security", - title: "五项安全防护", - body: "五项防护作用于经过网关的每个请求,对所有上游与密钥一致:出站脱敏查找请求中的 API 密钥、私钥、JWT 与连接串口令;工具调用审查检查上游返回的工具调用中的危险命令;隐藏字符与内容过滤检查客户端发来的内容(含工具结果)中的不可见 Unicode 字符与提示注入语句;输出长度衡量单次回答的长度。每项防护可设为关闭、观察或拦截。观察只记录命中,不改变请求;除输出长度出厂为关闭外,其余各项出厂均为观察。拦截时,出站脱敏在请求发出前替换凭据并在响应中还原,危险的工具调用与超出上限的回答被切断,含隐藏字符或命中「拒绝」规则的请求被拒绝。内置规则可以逐条停用,也可以添加自定义规则,所有命中都记录在安全日志中。", - alt: "安全页日志:出站脱敏与工具调用审查处于拦截,隐藏字符与内容过滤处于观察,输出长度关闭;最近 24 小时 8 次命中,包括已替换的 AWS 访问密钥 ID、已切断的下载即执行命令、自定义规则 customer-id 的命中、工具结果中隐藏的 Unicode 标签字符,以及记录在案的提示注入语句", + id: "overview", + title: "费用如实计算", + body: "按模型与上游统计 token、费用、缓存节省、首 token 时间与生成速度。估算的金额单独标注,无法计价的请求单独计数,不按零计入;价格每日按 LiteLLM 公开价更新,也可以使用自定义价目表。", + alt: overviewAlt["zh-CN"], }, { - id: "mcp", - title: "MCP 服务器、技能与钩子", - body: "Claude Code、Claude Desktop、Cursor、Codex、opencode、Zed、Antigravity CLI 与 DeepSeek Harness 中配置的 MCP 服务器并列显示,标出远程与第三方服务器以及各客户端之间不一致的配置;服务器可以复制到其他客户端或移除,写入前显示改动。钩子与技能同样逐一列出。客户端配置、技能、钩子、斜杠命令、subagent 与项目指令会被扫描,检查隐藏字符、提示注入、危险命令与过宽权限四类问题。扫描只报告、不修改任何文件,出现新发现时发送系统通知。", - alt: "MCP 页:五个 MCP 服务器在 Claude Code、Claude Desktop、Cursor、Codex、opencode 与 Zed 中的配置情况,context7 与 linear 标为第三方远程服务器,github 标为各客户端配置不一致;共扫描 11 个文件,发现高、中、低风险各一项", + id: "dry-run", + title: "改规则前先试算", + body: "试算给出请求会命中哪条规则、前面的规则为何未命中,以及将依次尝试哪些上游。不发出请求,也不产生费用。", + alt: "路由试算:cursor 密钥以 OpenAI Chat Completions 格式请求 claude-sonnet-5;gemini 规则未命中并说明原因,catch-all 规则命中,请求交给费用最低优先的策略组,依次尝试 relay 与 anthropic,并转换为 Anthropic Messages", }, { - id: "settings", - title: "设置与界面语言", - body: "界面提供英文与简体中文,默认跟随系统语言,也可以另选。设置页还包括连接本机或远程 core、外观、菜单栏显示内容、开机启动、提醒方式、网关的监听地址与允许访问的网段、请求记录与正文的保留时长、自动更新,以及完全卸载(还原所有已接管的客户端并关闭开机启动)。", - alt: "设置页:连接列表中的本机(当前)与 homelab、build-server 两个远程 core,启动时使用的连接,跟随系统的界面语言,外观,以及菜单栏的显示内容", + id: "keys", + title: "每个客户端一把密钥", + body: "接入客户端时为它生成专用密钥,流量与费用按客户端分开统计。每把密钥可单独设置路由、可见模型与并发上限,更换后的新密钥自动写入客户端配置。", + alt: "密钥页:默认密钥与 Claude Code、Codex、Cursor 各自的密钥,列出各自使用的路由、可用模型,以及最近 24 小时的请求数与费用", }, ], }, @@ -282,14 +269,13 @@ export const liteCopy = { eyebrow: "连接远程 core", title: "服务器上的 ThinkWatch Core", body: [ - "网关也可以部署在 Linux 服务器上:twcore 作为 systemd 服务运行,为网络中的客户端提供网关。应用通过服务器的远程控制端口连接它,在同样的页面中显示该服务器的流量、费用与配置。", - "在「设置 › 连接」中添加连接,填写服务器地址、控制端口,以及在服务器上执行 twcore control-key 得到的密钥。切换到某个连接之前,应用先测试该连接:完成握手并核对版本,服务器上的 core 须为应用要求的版本;版本不一致时,在服务器上执行 sudo twcore upgrade --version <版本号> --restart 即可换成该版本,升级或降级均可。应用同一时间只连接一个 core;连接服务器期间,本机的 core 停止运行,本机数据保留。客户端页与 MCP 页始终作用于运行应用的这台电脑,客户端页可以把这台电脑上的客户端改为指向服务器的网关。", - "无论本机还是远程,每条控制连接都经过以 config.yaml 中 listen.control.key 为密钥的 Noise 握手加密与认证,不涉及证书。连接密钥保存在应用数据目录中仅当前用户可读的文件里。", + "网关也可以作为 systemd 服务部署在 Linux 服务器上,为网络中的客户端提供服务。应用连接后,在同样的页面中查看与管理服务器的流量、费用和配置,本机的客户端也可以一步改为指向服务器的网关。", + "控制连接经 Noise 握手加密与认证,不涉及证书。应用同一时间只连接一个 core,连接服务器期间本机数据原样保留。", ], serverDocs: "服务器部署指南", docs: "连接远程 core", addAlt: - "添加远程连接:名称 homelab、地址 192.168.1.40、控制端口 24817 与密钥,测试连接成功,显示 core 0.49.0 与网关地址 192.168.1.40:8788", + "添加远程连接:名称 homelab、地址 192.168.1.40、控制端口与密钥,测试连接成功,显示 core 版本与网关地址 192.168.1.40:8788", figures: [ { id: "remote-switcher", @@ -306,12 +292,12 @@ export const liteCopy = { menubar: { eyebrow: "主窗口之外", title: "菜单栏、托盘与系统通知", - body: "在 macOS 上,菜单栏显示今日 token 与费用;订阅额度即将用完时数字变为橙色,用完后变为红色;也可以只显示标识或只显示数字。点开的菜单列出网关状态与输出速率、今日请求数、token 与费用、各订阅额度及其重置时间、进行中的请求,并提供打开主界面、复制网关地址、复制默认密钥、切换连接等操作。Windows 上图标位于通知区域,Linux 上位于系统托盘,菜单内容相同,以文字呈现。", + body: "macOS 菜单栏显示今日 token 与费用,订阅额度将尽时变为橙色,用完后变为红色。点开的菜单给出网关状态、各项额度及重置时间与进行中的请求,不打开主窗口即可复制网关地址或默认密钥。Windows 与 Linux 的托盘提供相同的菜单。", notices: - "网关停止转发、与远程 core 的连接断开、订阅额度用完、凭据过期、被拒绝或未能保存、代理无法连接、配置未能生效、工具调用命中「切断」规则,以及客户端配置中出现新的可疑内容时,应用会发送系统通知。上游无法连接只在应用内列出,不发送系统通知。提醒方式由一项设置统一决定:系统通知、仅在应用内显示或关闭。", + "网关停止转发、远程连接断开、额度用完、凭据失效、代理无法连接、配置未能生效、危险工具调用被切断,以及客户端配置中出现可疑内容时,应用发送系统通知。", chipAlt: "菜单栏:ThinkWatch 标识,右侧上行为今日 token 13.3M,下行为今日费用 $9.34", menuAlt: - "菜单栏菜单:网关运行中,地址 127.0.0.1:8788,输出 64 token/秒;chatgpt 的 5 小时额度已用 58%、2 小时后重置,每周额度已用 31%、3 天后重置;今日请求 204 次(失败 4 次)、token 13.3M、费用 $9.34;一条 Claude Code 请求进行中;以及复制网关地址、复制默认密钥、撤销上一次配置修改、打开主界面、连接、设置、检查更新等菜单项", + "菜单栏菜单:网关地址、输出速率与状态;ChatGPT 账号的 5 小时与每周额度及重置时间;今日请求数、token 与费用;进行中的请求;以及打开主界面、复制网关地址、复制默认密钥、切换连接、设置与检查更新等菜单项", }, built: { eyebrow: "架构",