diff --git a/CHANGELOG.md b/CHANGELOG.md index 79d5dac..7060801 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -65,6 +65,19 @@ lives under a budget, and what it is doing is visible (plan 2026-10-02, `C-7…C `package.nls*`, the runtime words through `l10n/` bundles, with key-parity tests holding the two together; the server's logs, CLI and prompts stay English. A `bundlesDirectory` field joins `cxxModules/cache`'s `paths`, so a client never guesses where a diagnostic bundle lands. +- **Hardened by the pre-release review.** A 0.0.9 instance working beside a 0.0.10 one keeps its + cache: the lease tick now gives an undescribed instance directory the same 24-hour grace the + sweep does, and a reset no longer deletes the resetting instance's own `instance.json` (the + heartbeat that says it is alive). The sweep command and a stale `cxxModules/cache` walk the + tree off the event loop — the reply arrives when the work is done, and no request or lease + renewal waits behind minutes of filesystem work; every sweep, background or interactive, is one + at a time with later ones coalesced. An unreadable process identity (another user's process on + Windows, a failed `ps` on macOS) no longer reads as a dead lease owner — only a definite answer + releases a lease early. A crashed clangd no longer pins its generation's start forever, so + `staleCommands` sweeps are possible again after a crash. Clock steps back no longer reap every + live instance. Budget settings that do not parse are rejected whole (`1.5G` is not `1G`) and + logged; deletions count what they could not take instead of what they attempted; a dry run + changes nothing the server remembers; a multi-root sweep answers with the sum over its roots. ## [0.0.9] — 2026-10-02 diff --git a/conformance/traceability.json b/conformance/traceability.json index 23499fa..eb76935 100644 --- a/conformance/traceability.json +++ b/conformance/traceability.json @@ -2083,7 +2083,7 @@ ], "S3-5.8-2": [ { - "manual": "the sweep answers from the session loop after the removal, the way mcppls.resetCache already did; it cancels nothing and fails no request an engine owed (review of src/server/session.cpp and src/orchestrator/workspace.cpp)" + "manual": "the sweep's removals run on a thread of their own and the reply returns to the session loop as an event (review of src/server/session.cpp, mcppls.sweepCache branch); it cancels nothing and fails no request an engine owed" } ], "S3-5.8-3": [ @@ -2110,7 +2110,11 @@ "S3-5.8-7": [ { "script": "src/orchestrator/workspace.cpp", - "contains": "impl.sweepRunning_.exchange(true)" + "contains": "sweepPending_" + }, + { + "script": "src/orchestrator/workspace.cpp", + "contains": "impl.sweepRunning_.store(true)" } ], "S3-5.8-8": [ @@ -2120,5 +2124,43 @@ { "check": "cache-budget/sweep-dry-run" } + ], + "S3-5.7-8": [ + { + "script": "src/orchestrator/workspace.cpp", + "contains": "start_cache_task_(\"report\"" + } + ], + "S3-5.8-9": [ + { + "script": "src/orchestrator/workspace.cpp", + "contains": "failed += removed.failed" + }, + { + "script": "src/cli/cache.cpp", + "contains": "failed += removed.failed" + } + ], + "S3-5.8-10": [ + { + "script": "src/server/session.cpp", + "contains": "run_prepared_sweep" + }, + { + "script": "src/orchestrator/workspace.cppm", + "contains": "deferred_answer" + } + ], + "S3-5.8-11": [ + { + "script": "src/server/session.cpp", + "contains": "freed += outcome.value(\"bytes\"" + } + ], + "S3-5.8-12": [ + { + "script": "src/orchestrator/workspace.cpp", + "contains": "no numbers are remembered" + } ] } diff --git a/docs/30-settings.md b/docs/30-settings.md index f3ad54e..d1944bb 100644 --- a/docs/30-settings.md +++ b/docs/30-settings.md @@ -92,7 +92,7 @@ either wrapped in a top-level `mcppls` object or not. | `mcppls.mcpp` | a path | *(empty)* | `--mcpp` | reload | The `mcpp` executable for mcpp projects; empty means found on `PATH`. | | `mcppls.payload` | a path | *(empty)* | `--payload` | restart | Payload directory with clangd and the semantic kit; overridden per-file by `clangd` and `kit` below. | | `mcppls.clangd` | a path | *(empty)* | `--clangd` | restart | clangd executable, overriding the one the payload carries. | -| `mcppls.cache.maxBytes` | ? | ? | `--cache-max-bytes` | restart | How large one workspace's module cache may get. Copies and dead instance directories are removed to stay under it; the published BMIs never are, so a cache that cannot get under the limit without them is reported instead (the status bar and the cache menu say so). `unlimited` turns the budget off. | +| `mcppls.cache.maxBytes` | ? | ? | `--cache-max-bytes` | restart | How large one workspace's module cache may get. Reaching it is reported -- the status bar and the cache menu say near or over; keeping under it is the sweeps' own work (copies and dead instance directories go, published BMIs never), and all workspaces together answer to cache.totalBytes. A cache that cannot get under the limit without a published BMI is only reported. `unlimited` turns the budget off, `0` keeps none of what a sweep may remove. | | `mcppls.cache.totalBytes` | ? | ? | `--cache-total-bytes` | restart | How large all workspaces' module caches may get together. Only workspaces no instance has open give anything up, oldest-used first; published BMIs are never removed. | | `mcppls.cache.instanceGrace` | a non-negative number of seconds | `86400` | `--cache-instance-grace` | restart | How long an instance directory that says nothing about itself (a leftover of mcppls 0.0.9 or older) is kept before it is removed: 86400, the default, is 24 hours. Directories that do describe themselves are judged by their own heartbeat instead. | | `mcppls.cache.showInStatusBar` | `auto`, `always`, `never` | `auto` | — | immediately | Whether the status bar shows the cache size. `auto` shows it only when the cache is near or over its budget; `always` and `never` do what they say. The hover card and the menu answer for the rest either way. | diff --git a/docs/specs/s3-lsp-extensions.md b/docs/specs/s3-lsp-extensions.md index 3669a1e..87475f3 100644 --- a/docs/specs/s3-lsp-extensions.md +++ b/docs/specs/s3-lsp-extensions.md @@ -314,7 +314,7 @@ interface CacheInstanceInfo { } ``` -A server that declared `cxxModules` **MUST** answer `cxxModules/cache` for every root it serves with the numbers of the cache it actually holds. S3-5.7-1 A server **MUST NOT** remove, move or rewrite anything as a result of the request: it is a read. S3-5.7-2 A server **MAY** answer from a report it cached for at most 30 seconds, and **MUST** recompute that report before answering when a sweep of the same root finished after the cached one was made, so what a client shows after a sweep is what the sweep left. S3-5.7-3 The `prompts` the report carries are rendered by the server itself, for a person to hand to a local agent; they name the read-only commands to look at and the paths on this machine, and the server **MUST NOT** send them, or any other part of the report, anywhere. S3-5.7-4 A client **MUST** treat every path and name in the report as text: it renders them escaped, and never turns a server-sent string into a command, a URL or markup of its own. S3-5.7-5 A server that does not know the request answers `MethodNotFound`, and a client that receives it falls back to the status's `cache` field or to the CLI. S3-5.7-6 `paths` names the three places a person investigating the cache is sent to: the root's own module cache (`cacheRoot`), the server's logs (`logDirectory`), and where diagnostic bundles are written (`bundlesDirectory`, the default the bundle writer uses); a client that reveals a directory reveals one of these, and nothing it guesses itself. S3-5.7-7 The agent prompt is a task book, not a transcript: the verified facts, the read-only checks each with what healthy looks like, the output contract (a verdict, the evidence, what could be done without deleting), and a bug branch that asks the developer first and only then -- with their agreement -- drafts the issue, shows the draft for approval, and names the bundle paths for the person to attach; the prompt **MUST** state that the agent never uploads logs or bundles itself. +A server that declared `cxxModules` **MUST** answer `cxxModules/cache` for every root it serves with the numbers of the cache it actually holds. S3-5.7-1 A server **MUST NOT** remove, move or rewrite anything as a result of the request: it is a read. S3-5.7-2 A server **MAY** answer from a report it cached for at most 30 seconds, and **MUST** recompute that report before answering when a sweep of the same root finished after the cached one was made, so what a client shows after a sweep is what the sweep left. S3-5.7-3 The `prompts` the report carries are rendered by the server itself, for a person to hand to a local agent; they name the read-only commands to look at and the paths on this machine, and the server **MUST NOT** send them, or any other part of the report, anywhere. S3-5.7-4 A client **MUST** treat every path and name in the report as text: it renders them escaped, and never turns a server-sent string into a command, a URL or markup of its own. S3-5.7-5 A server that does not know the request answers `MethodNotFound`, and a client that receives it falls back to the status's `cache` field or to the CLI. S3-5.7-6 `paths` names the three places a person investigating the cache is sent to: the root's own module cache (`cacheRoot`), the server's logs (`logDirectory`), and where diagnostic bundles are written (`bundlesDirectory`, the default the bundle writer uses); a client that reveals a directory reveals one of these, and nothing it guesses itself. S3-5.7-7 The agent prompt is a task book, not a transcript: the verified facts, the read-only checks each with what healthy looks like, the output contract (a verdict, the evidence, what could be done without deleting), and a bug branch that asks the developer first and only then -- with their agreement -- drafts the issue, shows the draft for approval, and names the bundle paths for the person to attach; the prompt **MUST** state that the agent never uploads logs or bundles itself. S3-5.7-8 A server **MUST NOT** hold the loop that serves requests to recompute a stale report: walking a grown cache is filesystem work, and it runs off that loop -- the request is answered from the numbers already computed, and the refreshed numbers arrive with the next answer. ### 5.8 `mcppls.sweepCache` @@ -331,15 +331,16 @@ interface SweepCacheParams { interface SweepCacheResult { ok: true; freedBytes: number; // what the sweep freed, or would have under `dryRun` - files: number; // the copies and command directories counted in `freedBytes` + files: number; // the copies counted in `freedBytes` instances: number; // the instance directories removed + failed: number; // entries a removal could not take (a lock, a scanner holding the file) roots: number; // how many roots were swept dryRun: boolean; - alreadyRunning?: boolean; // a sweep was in flight; nothing was done by this one + alreadyRunning?: boolean; // every swept root had a sweep in flight; nothing was done by this one } ``` -A server that advertises the command **MUST NOT** stop, restart or interrupt any engine for a sweep's sake. S3-5.8-1 A sweep **MUST NOT** let a request any engine owed fail. S3-5.8-2 A server **MUST NOT** remove a published BMI -- a `.pcm` whose name is not the versioned copy shape. S3-5.8-3 A server **MUST NOT** remove a file a live engine generation could hold mapped: a copy is swept only when its mtime is older than the start of the oldest live generation of the engines using that cache, or when no engine uses the cache at all. S3-5.8-4 A server **MUST NOT** sweep a workspace's cache that another instance has open: an instance directory whose own heartbeat is fresh belongs to a live instance, whatever the workspace's lease says. S3-5.8-5 The `staleCommands` category -- the command directories older units' BMIs occupy -- is the engine start path's own work (C-2), so a server **MUST** skip it unless the client asked for it by name and no engine is live in that root. S3-5.8-6 A server **MUST** run one sweep at a time, and answer a sweep that arrives while one runs with `alreadyRunning: true` and nothing removed by it. S3-5.8-7 With `dryRun: true` a server **MUST** compute the answer over exactly the set it would have removed, so what a client reports as "would free" is what a sweep would free. S3-5.8-8 +A server that advertises the command **MUST NOT** stop, restart or interrupt any engine for a sweep's sake. S3-5.8-1 A sweep **MUST NOT** let a request any engine owed fail. S3-5.8-2 A server **MUST NOT** remove a published BMI -- a `.pcm` whose name is not the versioned copy shape. S3-5.8-3 A server **MUST NOT** remove a file a live engine generation could hold mapped: a copy is swept only when its mtime is older than the start of the oldest live generation of the engines using that cache, or when no engine uses the cache at all. S3-5.8-4 A server **MUST NOT** sweep a workspace's cache that another instance has open: an instance directory whose own heartbeat is fresh belongs to a live instance, whatever the workspace's lease says. S3-5.8-5 The `staleCommands` category -- the command directories older units' BMIs occupy -- is the engine start path's own work (C-2), so a server **MUST** skip it unless the client asked for it by name and no engine is live in that root. S3-5.8-6 A server **MUST** run one sweep at a time, and answer a sweep that arrives while one runs with `alreadyRunning: true` and nothing removed by it. S3-5.8-7 With `dryRun: true` a server **MUST** compute the answer over exactly the set it would have removed, so what a client reports as "would free" is what a sweep would free. S3-5.8-8 What a sweep counts freed is what left the disk: an entry a removal could not take is counted in `failed`, never in `freedBytes`. S3-5.8-9 A server **MUST NOT** run the removals on the loop that serves requests and renews leases: the sweep of a grown cache is minutes of filesystem work, the reply arrives when the work is done, and nothing a client asked before it -- the lease's own renewal included -- waits behind it. S3-5.8-10 With more than one root swept, the numbers are the sum over those roots and `roots` says how many there were; `alreadyRunning` appears only when every swept root had a sweep in flight. S3-5.8-11 A `dryRun` sweep also changes nothing the server remembers: no `lastSweep`, no recomputed report -- a preview a client showed must not later read as a sweep that happened. S3-5.8-12 ## 6. Module features through standard LSP diff --git a/docs/zh-CN/30-settings.md b/docs/zh-CN/30-settings.md index 3863285..170182d 100644 --- a/docs/zh-CN/30-settings.md +++ b/docs/zh-CN/30-settings.md @@ -87,7 +87,7 @@ VS Code 扩展已经会这样做);`重新加载模型` 只重新加载项目 | `mcppls.mcpp` | 路径 | (空) | `--mcpp` | 重新加载模型 | mcpp 项目所用的 `mcpp` 可执行文件;空表示在 `PATH` 上查找。 | | `mcppls.payload` | 路径 | (空) | `--payload` | 重启 | 包含 clangd 和语义工具包的 payload 目录;下面的 `clangd` 和 `kit` 可以分别覆盖其中一项。 | | `mcppls.clangd` | 路径 | (空) | `--clangd` | 重启 | clangd 可执行文件,覆盖 payload 自带的那一份。 | -| `mcppls.cache.maxBytes` | ? | ? | `--cache-max-bytes` | 重启 | 单个工作区的模块缓存上限。超出时先清理副本与死实例目录回到预算内;已发布的模块本体(BMI)永远不会被删——删净副本仍超限时只报告(状态栏与缓存菜单可见)。`unlimited` 关闭预算。 | +| `mcppls.cache.maxBytes` | ? | ? | `--cache-max-bytes` | 重启 | 单个工作区的模块缓存上限。接近或超出时会报告——状态栏与缓存菜单显示 near/over;回到预算内是自动清扫的事(删副本与死实例目录,已发布的 BMI 永不删除),所有工作区加在一起的上限由 cache.totalBytes 负责。删净可删仍超限时只报告不硬删。`unlimited` 关闭预算,`0` 表示可清理的一律不留。 | | `mcppls.cache.totalBytes` | ? | ? | `--cache-total-bytes` | 重启 | 所有工作区模块缓存的总上限。只有没有实例打开的工作区按最久未用的先后让出副本;已发布的模块本体不会被删。 | | `mcppls.cache.instanceGrace` | 非负整数(秒) | `86400` | `--cache-instance-grace` | 重启 | 一个不自述的实例目录(0.0.9 及更早版本的遗留)在删除前保留多久:默认 86400 秒,即 24 小时。会自述的目录按它自己的心跳判断。 | | `mcppls.cache.showInStatusBar` | `auto`, `always`, `never` | `auto` | — | 立即生效 | 状态栏是否显示缓存大小。`auto` 只在缓存接近或超过预算时显示;`always` 与 `never` 如字面。无论如何,其余数字看悬停卡片与菜单。 | diff --git a/editors/vscode/l10n/bundle.l10n.json b/editors/vscode/l10n/bundle.l10n.json index 19a9a0f..8cfdc3d 100644 --- a/editors/vscode/l10n/bundle.l10n.json +++ b/editors/vscode/l10n/bundle.l10n.json @@ -7,7 +7,7 @@ "{0} modules · {1} units": "{0} modules · {1} units", "{0} s ago": "{0} s ago", "A sweep is already running.": "A sweep is already running.", - "A sweep would free {0} bytes ({1} files). Nothing was removed.": "A sweep would free {0} bytes ({1} files). Nothing was removed.", + "A sweep would free {0} ({1} files). Nothing was removed.": "A sweep would free {0} ({1} files). Nothing was removed.", "A sweep would free nothing: there is nothing to remove.": "A sweep would free nothing: there is nothing to remove.", "Back": "Back", "Cache in use": "Cache in use", @@ -33,7 +33,7 @@ "Error": "Error", "Feedback": "Feedback", "freed {0} ({1} files) · {2} ago": "freed {0} ({1} files) · {2} ago", - "Freed {0} bytes ({1} files). No restart, no rebuild.": "Freed {0} bytes ({1} files). No restart, no rebuild.", + "Freed {0} ({1} files). No restart, no rebuild.": "Freed {0} ({1} files). No restart, no rebuild.", "Instances": "Instances", "Largest modules": "Largest modules", "Last sweep": "Last sweep", diff --git a/editors/vscode/l10n/bundle.l10n.zh-cn.json b/editors/vscode/l10n/bundle.l10n.zh-cn.json index 7aaa7b7..d4de684 100644 --- a/editors/vscode/l10n/bundle.l10n.zh-cn.json +++ b/editors/vscode/l10n/bundle.l10n.zh-cn.json @@ -7,7 +7,7 @@ "{0} modules · {1} units": "{0} 个模块 · {1} 个单元", "{0} s ago": "{0} 秒前", "A sweep is already running.": "已有一次清理正在进行。", - "A sweep would free {0} bytes ({1} files). Nothing was removed.": "预演:可释放 {0} 字节({1} 个文件)。未删除任何东西。", + "A sweep would free {0} ({1} files). Nothing was removed.": "预演:可释放 {0}({1} 个文件)。未删除任何东西。", "A sweep would free nothing: there is nothing to remove.": "预演:无可释放的内容,没有要删除的。", "Back": "返回", "Cache in use": "缓存占用", @@ -33,7 +33,7 @@ "Error": "错误", "Feedback": "反馈", "freed {0} ({1} files) · {2} ago": "释放 {0}({1} 个文件)· {2}", - "Freed {0} bytes ({1} files). No restart, no rebuild.": "已释放 {0} 字节({1} 个文件)。不重启、不重编。", + "Freed {0} ({1} files). No restart, no rebuild.": "已释放 {0}({1} 个文件)。不重启、不重编。", "Instances": "实例目录", "Largest modules": "最大模块", "Last sweep": "上次清理", diff --git a/editors/vscode/src/cacheSweep.ts b/editors/vscode/src/cacheSweep.ts index c29ddd4..f127d0c 100644 --- a/editors/vscode/src/cacheSweep.ts +++ b/editors/vscode/src/cacheSweep.ts @@ -2,7 +2,7 @@ // extension's command id and the server's are different on purpose -- `vscode-languageclient` // registers every command the server advertises, and a clash fails the client at startup. Pure: no // `vscode`, so the parsing and the ids are unit-testable. -import { CacheDetail } from './cacheSegment'; +import { CacheDetail, sizeText } from './cacheSegment'; import { t } from './strings'; export const SWEEP_WORKSPACE_CACHE_COMMAND = 'mcppls.sweepWorkspaceCache'; @@ -51,11 +51,11 @@ export function sweepResultText(result: SweepResult): string { if (result.alreadyRunning === true) return t('A sweep is already running.'); if (result.dryRun) { return result.freedBytes > 0 - ? t('A sweep would free {0} bytes ({1} files). Nothing was removed.', result.freedBytes, result.files) + ? t('A sweep would free {0} ({1} files). Nothing was removed.', sizeText(result.freedBytes), result.files) : t('A sweep would free nothing: there is nothing to remove.'); } if (result.freedBytes > 0) { - return t('Freed {0} bytes ({1} files). No restart, no rebuild.', result.freedBytes, result.files); + return t('Freed {0} ({1} files). No restart, no rebuild.', sizeText(result.freedBytes), result.files); } return t('Nothing to remove: the cache is already swept.'); } diff --git a/editors/vscode/src/tooltipCard.ts b/editors/vscode/src/tooltipCard.ts index 47fc72c..56691f8 100644 --- a/editors/vscode/src/tooltipCard.ts +++ b/editors/vscode/src/tooltipCard.ts @@ -23,7 +23,11 @@ export function escapeCell(text: string): string { return text.replace(/([\\`|[\]])/g, '\\$1').replace(/\r?\n/g, ' '); } -const BAR_CELLS = 12; +// 13 cells is the plain default (review 2026-10-04): the en table's text columns leave exactly +// this much of the width band. The card itself computes its own count per language -- see +// `barCells` below -- so the zh table, with narrower words, runs its bars a few cells longer and +// closes on the same right edge. +const BAR_CELLS = 13; /** The dot-matrix bar, monochrome: `█` for the filled share, `░` for the scale behind it. */ export function bar(share: number, cells = BAR_CELLS): string { @@ -114,10 +118,22 @@ function ageText(seconds: number): string { } /** One row of the composition table: the class, its size right-aligned, its share, its bar. */ -function compositionRow(label: string, bytes: number, total: number): string { +function compositionRow(label: string, bytes: number, total: number, cells: number): string { const share = total > 0 ? bytes / total : 0; const percent = total > 0 ? Math.round(share * 100) : 0; - return `| ${label} | ${sizeText(bytes)} | ${percent}% | ${bar(share)} |`; + return `| ${label} | ${sizeText(bytes)} | ${percent}% | ${bar(share, cells)} |`; +} + +// The bar column fills what this card's own text columns leave of the width band (review +// 2026-10-04): the cells come from the labels and numbers the table actually shows, so the en +// table lands on 13 cells and the zh one -- narrower words around the same Latin sizes -- runs a +// few longer, both tables closing on the same right edge as the footer. One count a card: the +// rows, the total row and the preparation line all share the scale, so nothing jitters. +const BAR_BAND_WIDTH = 44; // the widest row's plain text columns, measured with en at 13 cells +const widest = (texts: string[]): number => Math.max(...texts.map(visibleWidth)); +function barCells(columnWidths: number[]): number { + const nonBar = columnWidths.reduce((sum, width) => sum + width, 0); + return Math.max(12, Math.min(20, BAR_BAND_WIDTH - nonBar)); } // Markdown folds a single newline into a space; a row only gets its own line from a HARD break @@ -131,6 +147,18 @@ export function cacheCardZones(input: CardInput): string[] { const detail = input.detail; const coarse = input.coarse; const zones: string[] = []; + const bytes = detail?.bytes ?? coarse?.bytes ?? 0; + const limit = detail?.limits.perWorkspace ?? coarse?.limitBytes ?? 0; + const fill = limit > 0 ? Math.min(1, bytes / limit) : 0; + const percent = limit > 0 ? Math.round(fill * 100) : 0; + // The card's one bar scale, from the columns the table actually shows (the total row's + // "bytes / limit" is the widest cell the Used column ever holds). + const usedText = limit > 0 ? `${sizeText(bytes)} / ${sizeText(limit)}` : sizeText(bytes); + const cells = barCells([ + widest([t('Class'), t('Published'), t('Copies'), t('Instances'), t('Trash'), t('Total')]), + widest([t('Used'), usedText]), + widest([t('Share'), `${percent}%`]), + ]); if (input.status) { // One line for what the project IS: dot, name, state, counts, source -- each part drops @@ -149,14 +177,10 @@ export function cacheCardZones(input: CardInput): string[] { const progress = input.status.progress ?? detail?.progress; if (progress && progress.total > 0) { const share = progress.done / progress.total; - zones.push(`${t('Preparing index {0}/{1}', progress.done, progress.total)} ${bar(share)} ${Math.round(share * 100)}%`); + zones.push(`${t('Preparing index {0}/{1}', progress.done, progress.total)} ${bar(share, cells)} ${Math.round(share * 100)}%`); } } - const bytes = detail?.bytes ?? coarse?.bytes ?? 0; - const limit = detail?.limits.perWorkspace ?? coarse?.limitBytes ?? 0; - const fill = limit > 0 ? Math.min(1, bytes / limit) : 0; - const percent = limit > 0 ? Math.round(fill * 100) : 0; if (detail) { // One table for the whole cache zone: the classes, then the bold total row against the // budget -- the grid keeps every column aligned. All four class rows are always there. @@ -164,15 +188,15 @@ export function cacheCardZones(input: CardInput): string[] { const table = [ `| ${t('Class')} | ${t('Used')} | ${t('Share')} | |`, '|---|---:|---:|:--|', - compositionRow(t('Published'), detail.canonical?.bytes ?? 0, total), - compositionRow(t('Copies'), detail.copies.bytes, total), - compositionRow(t('Instances'), detail.instances.bytes, total), - compositionRow(t('Trash'), detail.trash?.bytes ?? 0, total), + compositionRow(t('Published'), detail.canonical?.bytes ?? 0, total, cells), + compositionRow(t('Copies'), detail.copies.bytes, total, cells), + compositionRow(t('Instances'), detail.instances.bytes, total, cells), + compositionRow(t('Trash'), detail.trash?.bytes ?? 0, total, cells), ]; if (limit > 0) { - table.push(`| **${t('Total')}** | **${sizeText(bytes)} / ${sizeText(limit)}** | **${percent}%** | ${bar(fill)} |`); + table.push(`| **${t('Total')}** | **${usedText}** | **${percent}%** | ${bar(fill, cells)} |`); } else { - table.push(`| **${t('Total')}** | **${sizeText(bytes)}** | | ${bar(fill)} |`); + table.push(`| **${t('Total')}** | **${usedText}** | | ${bar(fill, cells)} |`); } zones.push(zone(table)); if (detail.lastSweep && detail.lastSweep.at > 0) { @@ -183,10 +207,14 @@ export function cacheCardZones(input: CardInput): string[] { } else if (coarse) { // No detail yet (or an old server): the total row against the budget in the same table // shape the full card will show, so the card never changes form while the detail loads. + const coarseCells = barCells([ + widest([t('Used / budget'), `${sizeText(coarse.bytes)} / ${sizeText(coarse.limitBytes)}`]), + widest([t('Share'), `${percent}%`]), + ]); zones.push(zone([ `| ${t('Used / budget')} | ${t('Share')} | |`, '|---:|---:|:--|', - `| **${sizeText(coarse.bytes)} / ${sizeText(coarse.limitBytes)}** | ${percent}% | ${bar(fill)} |`, + `| **${sizeText(coarse.bytes)} / ${sizeText(coarse.limitBytes)}** | ${percent}% | ${bar(fill, coarseCells)} |`, ])); zones.push(t('Copies {0} ({1} files) · instances {2} ({3})', sizeText(coarse.copies.bytes), coarse.copies.files, sizeText(coarse.instances.bytes), coarse.instances.count)); @@ -229,7 +257,7 @@ export function cardMarkdown(input: CardInput): string { // triple lands 3 columns wide of it, the zh 1 narrow, and the no-break-space pass // below closes whatever remains (at most a column or two). const actions = `[$(clear-all) ${t('Sweep cache')}](command:${input.sweepCommand})` - + ` · [$(folder-opened) ${t('Open logs')}](command:${input.revealCommand}?%5B%22root%22%5D)` + + ` · [$(folder-opened) ${t('Open logs')}](command:${input.revealCommand}?%5B%22logs%22%5D)` + ` · [$(copy) ${t('Copy agent prompt')}](command:${input.copyPromptCommand})`; const actionsPlain = `$(clear-all) ${t('Sweep cache')} · $(folder-opened) ${t('Open logs')} · $(copy) ${t('Copy agent prompt')}`; const repo = `[$(github) ${escapeCell(repoLabel(REPOSITORY))}](${REPOSITORY}) · [$(copy)](command:${input.copyRepositoryCommand})`; diff --git a/editors/vscode/test/unit/cacheHub.test.ts b/editors/vscode/test/unit/cacheHub.test.ts index 96b53f7..9e91db3 100644 --- a/editors/vscode/test/unit/cacheHub.test.ts +++ b/editors/vscode/test/unit/cacheHub.test.ts @@ -87,6 +87,8 @@ suite('cache hub v2', () => { test('sweep results say what happened, including that nothing restarts', () => { assert.ok(sweepResultText(parseSweepResult({ ok: true, freedBytes: 1_288_490_188_288, files: 6837, roots: 1, dryRun: false })).includes('No restart')); + assert.ok(sweepResultText(parseSweepResult({ ok: true, freedBytes: 1_288_490_188_288, files: 6837, roots: 1, dryRun: false })).includes('1.29 TB'), + 'the receipt reads like a person reads sizes, not raw bytes'); assert.ok(sweepResultText(parseSweepResult({ ok: true, freedBytes: 0, files: 0, roots: 1, dryRun: false })).includes('already')); assert.ok(sweepResultText(parseSweepResult({ ok: true, freedBytes: 100, files: 1, roots: 1, dryRun: true })).includes('would')); assert.ok(sweepResultText(parseSweepResult({ ok: true, alreadyRunning: true })).includes('already running')); diff --git a/editors/vscode/test/unit/tooltipCard.test.ts b/editors/vscode/test/unit/tooltipCard.test.ts index 8fd4613..b28abce 100644 --- a/editors/vscode/test/unit/tooltipCard.test.ts +++ b/editors/vscode/test/unit/tooltipCard.test.ts @@ -52,11 +52,11 @@ suite('tooltip card v3.1 (markdown, aligned footer)', () => { }); test('the bar is one monochrome dot-matrix language, fixed width', () => { - assert.strictEqual(bar(0.5), '`██████░░░░░░`'); - assert.strictEqual(bar(0), '`░░░░░░░░░░░░`'); - assert.strictEqual(bar(1), '`████████████`'); - assert.strictEqual(bar(2), '`████████████`', 'out-of-range shares clamp, never overflow'); - assert.strictEqual(bar(Number.NaN), '`░░░░░░░░░░░░`'); + assert.strictEqual(bar(0.5), '`███████░░░░░░`'); + assert.strictEqual(bar(0), '`░░░░░░░░░░░░░`'); + assert.strictEqual(bar(1), '`█████████████`'); + assert.strictEqual(bar(2), '`█████████████`', 'out-of-range shares clamp, never overflow'); + assert.strictEqual(bar(Number.NaN), '`░░░░░░░░░░░░░`'); }); test('widths the way the hover lays them out: icons two, CJK two, latin one', () => { @@ -85,11 +85,11 @@ suite('tooltip card v3.1 (markdown, aligned footer)', () => { test('the one cache table: all four classes, code-span bars, the bold total row', () => { const markdown = cardMarkdown(input()); assert.ok(markdown.includes('| Class | Used | Share | |')); - assert.ok(markdown.includes('| Published | 1.90 GB | 50% | `██████░░░░░░` |')); - assert.ok(markdown.includes('| Copies | 1.70 GB | 45% | `█████░░░░░░░` |')); + assert.ok(markdown.includes('| Published | 1.90 GB | 50% | `███████░░░░░░` |')); + assert.ok(markdown.includes('| Copies | 1.70 GB | 45% | `██████░░░░░░░` |')); assert.ok(markdown.includes('| Instances | 100 MB | 3% |'), 'a class below a cell keeps its row'); assert.ok(markdown.includes('| Trash | 1.00 KB | 0% |')); - assert.ok(markdown.includes('| **Total** | **3.80 GB / 4.00 GB** | **95%** | `███████████░` |')); + assert.ok(markdown.includes('| **Total** | **3.80 GB / 4.00 GB** | **95%** | `████████████░` |')); assert.ok(markdown.includes('failed to delete'), 'failures are visible, never silent'); const zones = markdown.split('\n\n'); assert.ok(zones[0].includes('GalTranslPP') && !zones[0].includes(' \n'), 'the title is ONE line'); @@ -106,7 +106,7 @@ suite('tooltip card v3.1 (markdown, aligned footer)', () => { test('without a detail the card keeps its shape: the total row, coarse counts, no half-empty grid', () => { const markdown = cardMarkdown(input({ detail: undefined, coarse })); - assert.ok(markdown.includes('| **3.80 GB / 4.00 GB** | 95% | `███████████░` |')); + assert.ok(markdown.includes('| **3.80 GB / 4.00 GB** | 95% | `███████████████████░` |')); assert.ok(markdown.includes('Copies 300 B (3 files) · instances 100 B (1)')); assert.ok(!markdown.includes('| Class |'), 'no half-empty table'); }); @@ -118,7 +118,8 @@ suite('tooltip card v3.1 (markdown, aligned footer)', () => { const lines = footer.split(' \n'); assert.strictEqual(lines.length, 2, 'the actions and the repository, two lines'); assert.ok(lines[0].includes('[$(clear-all) Sweep cache](command:mcppls.sweepWorkspaceCache)')); - assert.ok(lines[0].includes('[$(folder-opened) Open logs](command:mcppls.revealCacheDirectory?%5B%22root%22%5D)')); + assert.ok(lines[0].includes('[$(folder-opened) Open logs](command:mcppls.revealCacheDirectory?%5B%22logs%22%5D)'), + 'the label says logs, so the link opens the logs directory itself'); assert.ok(lines[0].includes('[$(copy) Copy agent prompt](command:mcppls.copyAgentPrompt)')); assert.ok(lines[1].includes('](https://github.com/Sunrisepeak/mcpp-language-server)'), 'the repository link is a real link'); assert.ok(lines[1].includes('[$(copy)](command:mcppls.copyRepositoryUrl)'), 'the copy next to it is a command link'); @@ -146,11 +147,34 @@ suite('tooltip card v3.1 (markdown, aligned footer)', () => { try { const markdown = cardMarkdown(input()); assert.ok(markdown.startsWith('● **GalTranslPP — 就绪** · 48 个模块 · 176 个单元 · mcpp'), markdown.split('\n')[0]); - assert.ok(markdown.includes('| 已发布 | 1.90 GB | 50% | `██████░░░░░░` |')); + assert.ok(markdown.includes('| 已发布 | 1.90 GB | 50% | `████████░░░░░░░` |')); assert.ok(markdown.includes('| **合计** | **3.80 GB / 4.00 GB** |')); assert.ok(markdown.includes('清理缓存') && markdown.includes('复制 Agent 提示词')); } finally { setLocalizer((message, ...args) => args.length > 0 ? message.replace(/\{(\d+)\}/g, (_, index) => String(args[Number(index)])) : message); } }); + + test('the bar column adapts per language: en 13, zh longer, both landing the same band', () => { + // The review's ask (2026-10-04): the bars carry their share of the table's width, en at 13 + // cells, and the narrower zh words leave more of the band -- so the two languages' tables + // close on the same right edge instead of the zh one reading narrower. + const en = cardMarkdown(input()); + const cellsOf = (markdown: string, row: RegExp): number => markdown.match(row)![1].length; + assert.strictEqual(cellsOf(en, /\| Published[^\n]*`([█░]+)`/), 13, 'the en table, the width the person picked'); + const widest = (markdown: string): number => Math.max(...markdown.split('\n\n').find((block) => block.startsWith('|'))! + .split('\n').map((row) => visibleWidth(row.replace(/\*\*|`/g, '')))); + const zh = JSON.parse(fs.readFileSync(path.resolve(__dirname, '..', '..', '..', 'l10n', 'bundle.l10n.zh-cn.json'), 'utf8')) as Record; + setLocalizer((message, ...args) => { + const translated = zh[message] ?? message; + return args.length > 0 ? translated.replace(/\{(\d+)\}/g, (_, index) => String(args[Number(index)])) : translated; + }); + try { + const zhMarkdown = cardMarkdown(input()); + assert.ok(cellsOf(zhMarkdown, /\| 已发布[^\n]*`([█░]+)`/) > 13, 'the zh bars run longer for the same band'); + assert.ok(Math.abs(widest(zhMarkdown) - widest(en)) <= 1, `both tables land on one width band: zh ${widest(zhMarkdown)} en ${widest(en)}`); + } finally { + setLocalizer((message, ...args) => args.length > 0 ? message.replace(/\{(\d+)\}/g, (_, index) => String(args[Number(index)])) : message); + } + }); }); diff --git a/modules/platform/src/process.cpp b/modules/platform/src/process.cpp index eef7f4d..0301932 100644 --- a/modules/platform/src/process.cpp +++ b/modules/platform/src/process.cpp @@ -641,6 +641,9 @@ std::optional process_alive(std::int64_t pid) { options.program = "/bin/ps"; options.arguments = { "-o", "state=", "-p", std::to_string(pid) }; options.pipeInput = false; + // A fixed locale: `ps` spells its columns by the process locale, and a lease's reader must + // see what its writer saw even when the two run under different settings (review 2026-10-03). + options.environment = std::vector { "LC_ALL=C" }; auto ran = run(std::move(options), std::chrono::seconds { 2 }); if (!ran || ran->timedOut) return std::nullopt; // ps exits 1 when no process has the pid. @@ -692,6 +695,7 @@ std::optional cpu_seconds(std::int64_t pid) { options.program = "/bin/ps"; options.arguments = { "-o", "time=", "-p", std::to_string(pid) }; options.pipeInput = false; + options.environment = std::vector { "LC_ALL=C" }; // one spelling of the columns, every reader auto ran = run(std::move(options), std::chrono::seconds { 2 }); if (!ran || ran->timedOut || ran->exitCode != 0) return std::nullopt; return parse_cpu_time(ran->output); @@ -729,6 +733,9 @@ std::optional process_identity(std::int64_t pid) { options.program = "/bin/ps"; options.arguments = { "-o", "lstart=", "-p", std::to_string(pid) }; options.pipeInput = false; + // `lstart` is a formatted date, so its spelling follows the process locale: pin it, or a + // lease read under one locale would judge one written under another as a reused pid. + options.environment = std::vector { "LC_ALL=C" }; auto ran = run(std::move(options), std::chrono::seconds { 2 }); if (!ran || ran->timedOut || ran->exitCode != 0) return std::nullopt; const auto started { base::trim(ran->output) }; diff --git a/src/cli/cache.cpp b/src/cli/cache.cpp index 0142a0a..6f95876 100644 --- a/src/cli/cache.cpp +++ b/src/cli/cache.cpp @@ -107,9 +107,15 @@ int prune(const std::string& workspaces, const PruneOptions& options, cache::Bud freed += copies.bytes; failed += copies.failed; for (const auto& build : engine::clangd::stale_module_builds(base::join_path(context, "cdb"), 2)) { - const std::uint64_t bytes { cache::tree_bytes(build) }; - if (!options.dryRun) fs::remove_all(build); - freed += bytes; + // Counted is what left the disk, no more (review 2026-10-03): a removal that + // fails says so in `failed` instead of freeing numbers that did not happen. + if (options.dryRun) { + freed += cache::tree_bytes(build); + continue; + } + const fs::Removal removed { fs::tree_remove(build) }; + freed += removed.bytes; + failed += removed.failed; } const auto trash { cache::sweep_trash(context, options.dryRun) }; freed += trash.bytes; diff --git a/src/config/settings.cpp b/src/config/settings.cpp index 6504c03..12af3d0 100644 --- a/src/config/settings.cpp +++ b/src/config/settings.cpp @@ -304,8 +304,8 @@ const std::vector& shipped_registry() { Setting { .key = "cache.maxBytes", .kind = Kind::bytes, .defaultValue = "4G", .commandLine = "--cache-max-bytes", .surface = Surface::server, .applies = Applies::restart, .category = "paths", .since = "0.0.10", - .summary = "How large one workspace's module cache may get. Copies and dead instance directories are removed to stay under it; the published BMIs never are, so a cache that cannot get under the limit without them is reported instead (the status bar and the cache menu say so). `unlimited` turns the budget off.", - .summaryZh = "单个工作区的模块缓存上限。超出时先清理副本与死实例目录回到预算内;已发布的模块本体(BMI)永远不会被删——删净副本仍超限时只报告(状态栏与缓存菜单可见)。`unlimited` 关闭预算。", + .summary = "How large one workspace's module cache may get. Reaching it is reported -- the status bar and the cache menu say near or over; keeping under it is the sweeps' own work (copies and dead instance directories go, published BMIs never), and all workspaces together answer to cache.totalBytes. A cache that cannot get under the limit without a published BMI is only reported. `unlimited` turns the budget off, `0` keeps none of what a sweep may remove.", + .summaryZh = "单个工作区的模块缓存上限。接近或超出时会报告——状态栏与缓存菜单显示 near/over;回到预算内是自动清扫的事(删副本与死实例目录,已发布的 BMI 永不删除),所有工作区加在一起的上限由 cache.totalBytes 负责。删净可删仍超限时只报告不硬删。`unlimited` 关闭预算,`0` 表示可清理的一律不留。", }, Setting { .key = "cache.totalBytes", .kind = Kind::bytes, .defaultValue = "16G", .commandLine = "--cache-total-bytes", diff --git a/src/engine/clangd.cpp b/src/engine/clangd.cpp index 2708d4b..1be13b1 100644 --- a/src/engine/clangd.cpp +++ b/src/engine/clangd.cpp @@ -2363,6 +2363,11 @@ class ClangdEngine final : public Engine { const bool early { !handshakeDone_ }; upSince_.reset(); closedGeneration_ = generation_; + // C-7: the generation that protected its own writes from sweeps is over -- nothing of it + // may be running to hold a file mapped, so the bound falls back to "no live generation" + // (an interactive sweep may take the stale command directories of S3 5.8, and no bound + // protects a generation that no longer exists). The next start stamps a new one. + generationStartedAt_ = 0; handshakeDone_ = false; accepting_ = false; forget_primes_(); diff --git a/src/orchestrator/cache.cpp b/src/orchestrator/cache.cpp index 7d2eb13..0be1424 100644 --- a/src/orchestrator/cache.cpp +++ b/src/orchestrator/cache.cpp @@ -80,11 +80,14 @@ std::optional instance_file(const std::string& directory) { } // Whether an instance (its own directory's heartbeat) says it is alive: within twice the lease -// expiry, the same window the lease takeover already uses, and wider than a renewal interval. +// expiry either way, the same window the lease takeover already uses, and wider than a renewal +// interval. A heartbeat a little in the future is a clock stepped back (NTP, waking from sleep), +// not a dead instance -- reaping every live one for that would be worse than trusting it. bool instance_alive(const std::optional& described, std::chrono::system_clock::time_point now) { if (!described || described->at <= 0) return false; + const auto window { std::chrono::duration_cast(LEASE_EXPIRY * 2).count() }; const auto age { std::chrono::duration_cast(now.time_since_epoch()).count() - described->at }; - return age >= 0 && age < std::chrono::duration_cast(LEASE_EXPIRY * 2).count(); + return age < window && age > -window; } // Whether anything is working in a workspace directory this process does not own: a fresh lease, a @@ -96,7 +99,9 @@ bool workspace_live(const std::string& directory, std::chrono::system_clock::tim if (!lease.is_discarded() && lease.is_object()) { const auto heartbeat = lease.value("heartbeat", std::int64_t { 0 }); const auto age { std::chrono::duration_cast(now.time_since_epoch()).count() - heartbeat }; - if (heartbeat > 0 && age >= 0 && age < std::chrono::duration_cast(LEASE_EXPIRY).count()) return true; + // The same window either way as instance_alive: a lease a moment ahead of this clock + // is a stepped clock, not a dead workspace. + if (heartbeat > 0 && std::abs(age) < std::chrono::duration_cast(LEASE_EXPIRY).count()) return true; } } if (instance_alive(instance_file(directory), now)) return true; @@ -137,12 +142,11 @@ std::optional parse_bytes(std::string_view text) { scale = std::uint64_t { 1 } << 10; text.remove_suffix(1); } - const std::uint64_t count { [&] { - std::uint64_t value { 0 }; - const auto [_, error] = std::from_chars(text.data(), text.data() + text.size(), value); - return error == std::errc {} ? value : std::uint64_t { 0 }; - }() }; - if (count == 0 && text != "0") return std::nullopt; + // The number must be consumed whole: "1.5G" is not 1G and "12abc" is not 12 -- a setting that + // does not parse says so (nullopt) instead of silently meaning something smaller. + std::uint64_t count { 0 }; + const auto [end, error] { std::from_chars(text.data(), text.data() + text.size(), count) }; + if (error != std::errc {} || end != text.data() + text.size()) return std::nullopt; if (count > std::numeric_limits::max() / scale) return std::numeric_limits::max(); return count * scale; } @@ -153,16 +157,26 @@ Sweep sweep_copies(std::string_view modulesRoot, std::int64_t before, bool dryRu return sweep; } -std::size_t rename_dead_instances(std::string_view workspaceDirectory, std::chrono::milliseconds now, std::string_view ownToken) { +std::size_t rename_dead_instances(std::string_view workspaceDirectory, std::chrono::milliseconds now, std::string_view ownToken, + std::chrono::seconds grace) { const std::string instances { base::join_path(workspaceDirectory, "instances") }; if (!fs::is_directory(instances)) return 0; std::size_t renamed { 0 }; const auto nowPoint { std::chrono::system_clock::time_point { std::chrono::duration_cast(now) } }; + // The file clock is linear: its reading of "now minus the grace" is its reading of now, less + // the grace in nanoseconds -- the same stale-before sweep_instances computes. + const std::int64_t staleBefore { fs::modified_now() - std::chrono::duration_cast(grace).count() }; for (const auto& entry : fs::list_directory(instances)) { const std::string name { base::file_name(entry) }; if (!fs::is_directory(entry) || name.find(".trash-") != std::string::npos) continue; const auto described = instance_file(entry); if (instance_alive(described, nowPoint)) continue; + if (!described) { + // No self-description (a version before 0.0.10): the same grace sweep_instances gives + // it -- the newest write in the tree stands in for a heartbeat, so a 0.0.9 guest that + // is working right now keeps its directory, whoever owns the tick (review 2026-10-03). + if (newest_modified(entry) >= staleBefore) continue; + } // Renamed aside in one cheap metadata operation; the removal (seconds to minutes on a // full directory) happens in the background, and a failed rename waits for the next tick. if (fs::rename(entry, std::format("{}.trash-{}", entry, ownToken))) ++renamed; @@ -243,6 +257,27 @@ Sweep enforce_budget(std::string_view workspacesRoot, const Budget& budget, std: std::chrono::system_clock::time_point now) { Sweep sweep; if (!fs::is_directory(workspacesRoot) || budget.total == std::numeric_limits::max()) return sweep; + // One walk of a workspace answers both questions the budget asks of it -- how big it is, and + // when it was last used -- where two walks doubled the cost on exactly the machines with the + // most to walk (review 2026-10-03). + struct Measured { + std::uint64_t bytes { 0 }; + std::int64_t newest { std::numeric_limits::min() }; + }; + constexpr auto measured = [](this auto&& self, const std::string& directory) -> Measured { + Measured whole; + for (const auto& entry : fs::list_directory(directory)) { + if (fs::is_directory(entry)) { + const Measured nested { self(entry) }; + whole.bytes += nested.bytes; + whole.newest = std::max(whole.newest, nested.newest); + } else if (const auto stamp = fs::stamp(entry)) { + whole.bytes += stamp->size; + whole.newest = std::max(whole.newest, stamp->modified); + } + } + return whole; + }; struct Candidate { std::string path; std::int64_t lastUse; @@ -252,12 +287,12 @@ Sweep enforce_budget(std::string_view workspacesRoot, const Budget& budget, std: for (const auto& entry : fs::list_directory(workspacesRoot)) { if (!fs::is_directory(entry)) continue; const std::string name { base::file_name(entry) }; - const std::uint64_t bytes { directory_bytes(entry) }; - total += bytes; + const Measured whole { measured(entry) }; + total += whole.bytes; if (name == ownKey || workspace_live(entry, now)) continue; // Only dead workspaces are ranked, so the wall-clock heartbeat of the instance file is // stale by definition; the tree's newest file mtime is the ranking that is left. - candidates.push_back({ entry, newest_modified(entry) }); + candidates.push_back({ entry, whole.newest }); } if (total <= budget.total) return sweep; // Oldest-used first: the workspace nobody has touched for the longest gives up its copies. diff --git a/src/orchestrator/cache.cppm b/src/orchestrator/cache.cppm index bf65536..44e7224 100644 --- a/src/orchestrator/cache.cppm +++ b/src/orchestrator/cache.cppm @@ -66,8 +66,12 @@ Sweep sweep_copies(std::string_view modulesRoot, std::int64_t before, bool dryRu // The cheap part of the tick (plan C-9): stat a few `instance.json` files and rename the dead // directories aside (`.trash-`); the removal itself happens elsewhere, so the // tick stays at milliseconds however large the dead directory is. Never touches a live one -- -// a live guest is protected by its own heartbeat, not by the owner's lease. -std::size_t rename_dead_instances(std::string_view workspaceDirectory, std::chrono::milliseconds now, std::string_view ownToken); +// a live guest is protected by its own heartbeat, not by the owner's lease, and a directory with +// no self-description at all (a 0.0.8/0.0.9 instance) gets the same `grace` `sweep_instances` +// gives it: its newest write stands in for a heartbeat, so a 0.0.9 guest working right now is +// not renamed out from under a 0.0.10 owner's tick (review 2026-10-03). +std::size_t rename_dead_instances(std::string_view workspaceDirectory, std::chrono::milliseconds now, std::string_view ownToken, + std::chrono::seconds grace); // Removes what `rename_dead_instances` renamed aside and, in a full sweep (startup task, CLI // `--prune`, the sweep command), judges every instance directory directly: heartbeat fresh -> diff --git a/src/orchestrator/instance.cpp b/src/orchestrator/instance.cpp index 18b7d9f..ae733a3 100644 --- a/src/orchestrator/instance.cpp +++ b/src/orchestrator/instance.cpp @@ -50,11 +50,7 @@ struct Lease { // Whether the process a lease names is gone: false when it cannot be told (no pid recorded, or the // platform cannot answer -- the heartbeat alone decides there, as before). X-6 makes the check real // on Windows, where before it could never say anything. -bool owner_gone(const Lease& lease) { - if (lease.pid <= 0 || lease.started.empty()) return false; - const auto owner = platform::process_identity(lease.pid); - return !owner || owner->started != lease.started; -} +bool owner_gone(const Lease& lease) { return lease_owner_gone(lease.pid, lease.started); } std::optional read_lease(std::string_view workspaceDirectory) { auto text = platform::fs::read_file(lease_path(workspaceDirectory)); @@ -90,6 +86,17 @@ void write_instance(bool shared, std::string_view workspaceDirectory, std::strin } // namespace +bool lease_owner_gone(std::int64_t pid, std::string_view started) { + if (pid <= 0 || started.empty()) return false; + const auto owner = platform::process_identity(pid); + if (owner) return owner->started != started; + // Who it is cannot be read; whether it is ANY process still can. Only a definite "no such + // process" is a gone owner -- anything else (another user's process on Windows, a ps(1) that + // hiccuped) leaves the heartbeat to decide, exactly as before X-6. + const auto alive = platform::process_alive(pid); + return alive.has_value() && !*alive; +} + WorkspaceLease WorkspaceLease::acquire(std::string_view workspaceDirectory, std::chrono::system_clock::time_point now, std::string_view root) { WorkspaceLease lease; diff --git a/src/orchestrator/instance.cppm b/src/orchestrator/instance.cppm index 0bebdb3..7345f0b 100644 --- a/src/orchestrator/instance.cppm +++ b/src/orchestrator/instance.cppm @@ -12,6 +12,13 @@ export namespace mcppls::orchestrator { inline constexpr std::chrono::seconds LEASE_RENEWAL { 10 }; inline constexpr std::chrono::seconds LEASE_EXPIRY { 30 }; +// Whether the process a lease names (its pid and the incarnation stamp X-6 recorded) is gone. +// A lease is taken over early only on a definite answer: a different incarnation, or a process +// the platform itself says is gone. An identity the platform cannot read (another user's process +// on Windows, a ps(1) that failed on macOS) is NOT death -- treating it as such let a live +// owner's lease be taken over (review 2026-10-03); there the heartbeat alone decides, as before. +bool lease_owner_gone(std::int64_t pid, std::string_view started); + class WorkspaceLease { public: // `workspaceDirectory` is /workspaces/; `now` is wall-clock time (a lease is read by diff --git a/src/orchestrator/workspace.cpp b/src/orchestrator/workspace.cpp index b34d6b2..68dc052 100644 --- a/src/orchestrator/workspace.cpp +++ b/src/orchestrator/workspace.cpp @@ -193,7 +193,12 @@ struct Workspace::Impl final : engine::Host { std::uint64_t lastSweepFreed_ { 0 }; std::size_t lastSweepFiles_ { 0 }; std::size_t lastSweepFailed_ { 0 }; - std::atomic sweepRunning_ { false }; // one sweep at a time, in this process + // One cache pass at a time in this process, background or interactive alike (0.0.10 review: + // startup, an engine start and a command used to race on the same tree). A pass asked for + // while one runs is merged into `sweepPending_` and retriggered when the running one reports. + std::atomic sweepRunning_ { false }; + bool sweepPending_ { false }; // under cacheMutex_ + bool sweepPendingReportOnly_ { true }; // the coalesced pass's mode: report-only unless a removal asked too std::optional leaseRenewAt; engine::PayloadPaths payload; bool payloadCorrupt { false }; @@ -564,8 +569,16 @@ struct Workspace::Impl final : engine::Host { cache::Budget cache_budget() const { cache::Budget budget; - if (const auto bytes = cache::parse_bytes(options.settings.string_value("cache.maxBytes"))) budget.perWorkspace = *bytes; - if (const auto bytes = cache::parse_bytes(options.settings.string_value("cache.totalBytes"))) budget.total = *bytes; + // A setting that does not parse says so (review 2026-10-03): the default quietly holding is + // how a typo'd budget used to look exactly like no budget at all. + if (const std::string given { options.settings.string_value("cache.maxBytes") }; !given.empty()) { + if (const auto bytes = cache::parse_bytes(given)) budget.perWorkspace = *bytes; + else log::warning("setting cache.maxBytes \"{}\" does not parse (4G, 512M, 100K, bytes or unlimited); keeping the default", given); + } + if (const std::string given { options.settings.string_value("cache.totalBytes") }; !given.empty()) { + if (const auto bytes = cache::parse_bytes(given)) budget.total = *bytes; + else log::warning("setting cache.totalBytes \"{}\" does not parse (4G, 512M, 100K, bytes or unlimited); keeping the default", given); + } return budget; } @@ -574,8 +587,21 @@ struct Workspace::Impl final : engine::Host { // One background pass: dead instances first (they can free the most), then the copies, then the // budget across the workspaces nothing has open. `before` is the sweep's bound on the file // clock; the result arrives as a `cache_swept` event, so the journal, the status and the report - // cache are all touched on the session loop, as everything else is. - void start_cache_task_(std::string_view origin, std::int64_t before = platform::fs::modified_now()) { + // cache are all touched on the session loop, as everything else is. `reportOnly` skips every + // removal and only refreshes the numbers (a stale `cxxModules/cache` answer). Single-flight: + // while one pass runs, another asked-for pass is coalesced and retriggered by + // `handle_cache_swept`, never queued behind it (review 2026-10-03: a crash-looping engine + // starting one full pass per restart is many threads on one tree otherwise). + void start_cache_task_(std::string_view origin, std::int64_t before = platform::fs::modified_now(), bool reportOnly = false) { + { + std::lock_guard lock(cacheMutex_); + if (sweepRunning_.load()) { + sweepPending_ = true; + sweepPendingReportOnly_ = sweepPendingReportOnly_ && reportOnly; // a removal pass asked for later wins + return; + } + sweepRunning_.store(true); + } const cache::Budget budget { cache_budget() }; const auto grace { cache_grace() }; const std::string workspaceDirectory { workspaceDirectory_ }; @@ -584,25 +610,27 @@ struct Workspace::Impl final : engine::Host { const std::string ownToken { ownToken_ }; auto queue = events; const std::string rootKey { key }; - std::thread { [budget, grace, workspaceDirectory, ownCacheDirectory, ownKey, ownToken, origin = std::string { origin }, before, queue, rootKey]() mutable { + std::thread { [budget, grace, workspaceDirectory, ownCacheDirectory, ownKey, ownToken, origin = std::string { origin }, before, queue, rootKey, reportOnly]() mutable { const auto now { std::chrono::system_clock::now() }; std::uint64_t bytes { 0 }; std::size_t files { 0 }, instances { 0 }, failed { 0 }; - const cache::Sweep dead { cache::sweep_instances(workspaceDirectory, now, grace) }; - bytes += dead.bytes; - files += dead.files; - instances += dead.instances; - failed += dead.failed; - for (const auto& context : cache::contexts_of(ownCacheDirectory)) { - const cache::Sweep one { cache::sweep_copies(cache::modules_root(context), before) }; - bytes += one.bytes; - files += one.files; - failed += one.failed; + if (!reportOnly) { + const cache::Sweep dead { cache::sweep_instances(workspaceDirectory, now, grace) }; + bytes += dead.bytes; + files += dead.files; + instances += dead.instances; + failed += dead.failed; + for (const auto& context : cache::contexts_of(ownCacheDirectory)) { + const cache::Sweep one { cache::sweep_copies(cache::modules_root(context), before) }; + bytes += one.bytes; + files += one.files; + failed += one.failed; + } + const cache::Sweep over { cache::enforce_budget(base::join_path(platform::dirs::cache_directory(), "workspaces"), budget, ownKey, now) }; + bytes += over.bytes; + files += over.files; + failed += over.failed; } - const cache::Sweep over { cache::enforce_budget(base::join_path(platform::dirs::cache_directory(), "workspaces"), budget, ownKey, now) }; - bytes += over.bytes; - files += over.files; - failed += over.failed; Json report; try { report = cache::report(ownCacheDirectory, budget, std::chrono::system_clock::now(), ownToken); @@ -610,8 +638,8 @@ struct Workspace::Impl final : engine::Host { report = Json::object(); // a report is never worth a crashed sweeper thread } queue->push(Event { EventKind::cache_swept, - Json { { "origin", origin }, { "bytes", bytes }, { "files", files }, { "instances", instances }, - { "failed", failed }, { "report", std::move(report) } }, + Json { { "origin", origin }, { "reportOnly", reportOnly }, { "bytes", bytes }, { "files", files }, + { "instances", instances }, { "failed", failed }, { "report", std::move(report) } }, 0, {}, rootKey, {} }); } }.detach(); } @@ -2300,9 +2328,10 @@ struct Workspace::Impl final : engine::Host { lease->renew(std::chrono::system_clock::now()); // C-9: the tick's cheap part -- stat the few `instance.json` files and rename the dead // directories aside; their removal (however large) runs in the background, so the tick - // stays at milliseconds. A live guest's own heartbeat protects it here, owner or not. + // stays at milliseconds. A live guest's own heartbeat protects it here, owner or not, + // and a directory that says nothing at all gets the same grace the sweep gives it. const auto heartbeat { std::chrono::duration_cast(std::chrono::system_clock::now().time_since_epoch()) }; - if (cache::rename_dead_instances(workspaceDirectory_, heartbeat, ownToken_) > 0) start_cache_task_("tick", platform::fs::modified_now()); + if (cache::rename_dead_instances(workspaceDirectory_, heartbeat, ownToken_, cache_grace()) > 0) start_cache_task_("tick", platform::fs::modified_now()); leaseRenewAt = now + LEASE_RENEWAL; } if (reloadAt && *reloadAt <= now) { @@ -2760,9 +2789,11 @@ std::uint64_t Workspace::reset_cache() { for (const auto& engine : impl.engines) freed += engine->clear_cache_on_request(); for (const auto& entry : platform::fs::list_directory(impl.cacheDirectory)) { const std::string_view name { base::file_name(entry) }; - // C-9: the instance's self-description is reset with everything else -- the next heartbeat - // tick writes it anew; leaving it would make this instance look gone to a reaper. - if (name != "instance.json" && (!name.starts_with("model.") || !name.ends_with(".json"))) continue; + // The cached models go; the instance's own `instance.json` stays (review 2026-10-03). It is + // not cache -- it is the heartbeat that says this directory is alive. Removing it here made + // a guest look dead to the reapers for up to a renewal interval while it was rebuilding + // (the old comment claimed the opposite); like `owner.lease`, it survives a reset. + if (!name.starts_with("model.") || !name.ends_with(".json")) continue; if (const auto stamp = platform::fs::stamp(entry)) freed += stamp->size; platform::fs::remove_all(entry); } @@ -2789,36 +2820,63 @@ void Workspace::handle_cache_swept(const Json& outcome) { const std::size_t files { outcome.value("files", std::size_t { 0 }) }; const std::size_t instances { outcome.value("instances", std::size_t { 0 }) }; const std::size_t failed { outcome.value("failed", std::size_t { 0 }) }; + const bool dryRun { outcome.value("dryRun", false) }; + bool retrigger { false }; + bool pendingReportOnly { false }; { std::lock_guard lock(impl.cacheMutex_); - impl.cacheSnapshot_ = outcome.value("report", Json::object()); - impl.cacheReport_ = impl.cacheSnapshot_; - impl.cacheReportAt_ = Clock::now(); - if (bytes > 0 || files > 0 || instances > 0 || failed > 0) { - impl.lastSweepAt_ = std::chrono::duration_cast(std::chrono::system_clock::now().time_since_epoch()).count(); - impl.lastSweepFreed_ = bytes; - impl.lastSweepFiles_ = files; - impl.lastSweepFailed_ = failed; - } - } - impl.journal.add("cache-swept", Json { { "bytes", bytes }, { "files", files }, { "instances", instances }, - { "failed", failed }, { "origin", outcome.value("origin", std::string {}) } }); - impl.update_status(); + // The single-flight token is released here for every origin (a dry-run command posts its + // event too, with nothing to remember); a pass coalesced while this one ran starts now. + impl.sweepRunning_.store(false); + if (impl.sweepPending_) { + impl.sweepPending_ = false; + retrigger = true; + pendingReportOnly = impl.sweepPendingReportOnly_; + impl.sweepPendingReportOnly_ = true; // the next batch starts from the neutral mode again + } + // A preview changed nothing: no numbers are remembered and nothing is journalled as swept. + if (!dryRun) { + impl.cacheSnapshot_ = outcome.value("report", Json::object()); + impl.cacheReport_ = impl.cacheSnapshot_; + impl.cacheReportAt_ = Clock::now(); + if (bytes > 0 || files > 0 || instances > 0 || failed > 0) { + impl.lastSweepAt_ = std::chrono::duration_cast(std::chrono::system_clock::now().time_since_epoch()).count(); + impl.lastSweepFreed_ = bytes; + impl.lastSweepFiles_ = files; + impl.lastSweepFailed_ = failed; + } + } + } + if (!dryRun) { + impl.journal.add("cache-swept", Json { { "bytes", bytes }, { "files", files }, { "instances", instances }, + { "failed", failed }, { "origin", outcome.value("origin", std::string {}) } }); + impl.update_status(); + } + if (retrigger) impl.start_cache_task_("coalesced", platform::fs::modified_now(), pendingReportOnly); } Json Workspace::cache_report() const { auto& impl = *impl_; Json numbers; + bool refresh { false }; { std::lock_guard lock(impl.cacheMutex_); - // At most 30 s old: a hub opening twice in a minute does not walk the same tree twice - // (C-13.1: the server is the only source, and it answers from its cache). + // At most 30 s old (C-13.1: the server is the only source, and it answers from its cache). + // A stale snapshot is answered at once and refreshed off the loop (review 2026-10-03): + // walking a full cache tree on the event loop held every request behind it. The one + // synchronous walk left is a server's very first answer, before any background pass has + // reported -- small by definition, and never repeated. if (!impl.cacheReportAt_ || Clock::now() - *impl.cacheReportAt_ > std::chrono::seconds { 30 }) { - impl.cacheReport_ = cache::report(impl.cacheDirectory, impl.cache_budget(), std::chrono::system_clock::now(), impl.ownToken_); - impl.cacheReportAt_ = Clock::now(); + if (impl.cacheReport_.is_null()) { + impl.cacheReport_ = cache::report(impl.cacheDirectory, impl.cache_budget(), std::chrono::system_clock::now(), impl.ownToken_); + impl.cacheReportAt_ = Clock::now(); + } else { + refresh = true; + } } numbers = impl.cacheReport_; } + if (refresh) impl.start_cache_task_("report", platform::fs::modified_now(), true); Json project { { "name", std::string { base::file_name(root_) } }, { "source", impl.model ? std::string { project::to_string(impl.model->source) } : std::string { "inferred" } } }; if (impl.model) { @@ -2832,6 +2890,10 @@ Json Workspace::cache_report() const { } const std::optional core { impl.coreEngine != nullptr ? std::optional { impl.coreEngine->status() } : std::nullopt }; const normalize::EnginePlan& plan { impl.plan }; + // The two prompts share one facts object: it was rendered twice per report (review 2026-10-03). + // Copy-initialisation on purpose: `Json facts { json }` is nlohmann's initializer-list trap and + // wraps the object in a one-element array, which the prompt renderers then reject. + const Json promptFacts = impl.cache_prompt_facts_(); Json envelope { { "state", std::string { to_string(impl.compute_state()) } }, { "project", std::move(project) }, { "plan", Json { { "units", plan.entries.size() }, { "modules", plan.modules.size() } } }, @@ -2848,8 +2910,8 @@ Json Workspace::cache_report() const { { "paths", Json { { "cacheRoot", impl.workspaceDirectory_ }, { "logDirectory", base::parent_path(log::file_path()) }, { "bundlesDirectory", bundle::default_directory() } } }, { "cli", Json { { "cacheQuery", "mcppls cache --format json" }, { "sweep", "mcppls cache --prune --dry-run" } } }, - { "prompts", Json { { "agent", cache::agent_prompt(impl.cache_prompt_facts_()) }, - { "issue", cache::issue_prompt(impl.cache_prompt_facts_()) } } } }; + { "prompts", Json { { "agent", cache::agent_prompt(promptFacts) }, + { "issue", cache::issue_prompt(promptFacts) } } } }; if (core && core->toPrepare > 0) envelope["progress"] = Json { { "done", core->prepared }, { "total", core->toPrepare } }; std::lock_guard lock(impl.cacheMutex_); if (impl.lastSweepAt_) { @@ -2859,86 +2921,103 @@ Json Workspace::cache_report() const { return envelope; } -Json Workspace::sweep_cache(const Json& params) { +SweepStart Workspace::prepare_sweep(const Json& params) { auto& impl = *impl_; + SweepStart start; const Json given { params.value("categories", Json::array()) }; const auto wants = [&](std::string_view category, bool byDefault) { if (!given.is_array() || given.empty()) return byDefault; return std::ranges::find(given, Json(std::string { category })) != given.end(); }; - const bool dryRun { params.value("dryRun", false) }; - if (impl.sweepRunning_.exchange(true)) { + start.dryRun = params.value("dryRun", false); + if (impl.sweepRunning_.load()) { // C-13.1: one sweep at a time; a caller while one runs learns it and does the math itself. - return Json { { "ok", true }, { "alreadyRunning", true }, { "dryRun", dryRun }, { "freedBytes", std::uint64_t { 0 } }, - { "files", std::size_t { 0 } }, { "instances", std::size_t { 0 } }, { "roots", 1 } }; - } - struct Running { - std::atomic& flag; - ~Running() { flag = false; } - } running { impl.sweepRunning_ }; - + start.immediate = Json { { "ok", true }, { "alreadyRunning", true }, { "dryRun", start.dryRun }, + { "freedBytes", std::uint64_t { 0 } }, { "files", std::size_t { 0 } }, + { "instances", std::size_t { 0 } }, { "roots", 1 } }; + return start; + } + impl.sweepRunning_.store(true); + start.started = true; + start.copies = wants("copies", true); + start.instances = wants("instances", true); + start.trash = wants("trash", true); + start.staleCommands = wants("staleCommands", false); + start.budget = wants("budget", true); // The safety rules of S3 5.8: nothing an engine holds, no canonical BMI, no engine stopped. std::int64_t bound { platform::fs::modified_now() }; - bool engineLive { false }; for (const auto& engine : impl.engines) { - if (const std::int64_t started { engine->generation_started_at() }; started > 0) { - engineLive = true; - bound = std::min(bound, started); - } - } - const auto now { std::chrono::system_clock::now() }; + if (const std::int64_t began { engine->generation_started_at() }; began > 0) { + start.engineLive = true; + bound = std::min(bound, began); + } + } + start.bound = bound; + start.cacheDirectory = impl.cacheDirectory; + start.workspaceDirectory = impl.workspaceDirectory_; + start.workspacesRoot = base::join_path(platform::dirs::cache_directory(), "workspaces"); + start.ownToken = impl.ownToken_; + start.limits = impl.cache_budget(); + start.grace = impl.cache_grace(); + start.now = std::chrono::system_clock::now(); + return start; +} + +Json run_prepared_sweep(const SweepStart& start) { std::uint64_t freed { 0 }; - std::size_t files { 0 }, removed { 0 }, failed { 0 }; - if (wants("copies", true)) { - for (const auto& context : cache::contexts_of(impl.cacheDirectory)) { - const cache::Sweep one { cache::sweep_copies(cache::modules_root(context), bound, dryRun) }; + std::size_t files { 0 }, instances { 0 }, failed { 0 }; + if (start.copies) { + for (const auto& context : cache::contexts_of(start.cacheDirectory)) { + const cache::Sweep one { cache::sweep_copies(cache::modules_root(context), start.bound, start.dryRun) }; freed += one.bytes; files += one.files; failed += one.failed; } } - if (wants("instances", true)) { - const cache::Sweep one { cache::sweep_instances(impl.workspaceDirectory_, now, impl.cache_grace(), dryRun) }; + if (start.instances) { + const cache::Sweep one { cache::sweep_instances(start.workspaceDirectory, start.now, start.grace, start.dryRun) }; freed += one.bytes; - removed += one.instances; + instances += one.instances; failed += one.failed; } - if (wants("trash", true)) { - const cache::Sweep one { cache::sweep_trash(impl.cacheDirectory, dryRun) }; + if (start.trash) { + const cache::Sweep one { cache::sweep_trash(start.cacheDirectory, start.dryRun) }; freed += one.bytes; failed += one.failed; } // C-2's job on the engine path, and only ever by explicit request, and only where no engine - // lives: a command that stops for nothing must not take the command directories either. - if (wants("staleCommands", false) && !engineLive) { - for (const auto& context : cache::contexts_of(impl.cacheDirectory)) { + // lives: a command that stops for nothing must not take the command directories either. What + // it counts is what it removed (review 2026-10-03): a removal that fails says so, and nothing + // is counted freed that is still on disk. + if (start.staleCommands && !start.engineLive) { + for (const auto& context : cache::contexts_of(start.cacheDirectory)) { for (const auto& build : engine::clangd::stale_module_builds(base::join_path(context, "cdb"), 2)) { - const std::uint64_t bytes { cache::tree_bytes(build) }; - if (!dryRun) platform::fs::remove_all(build); - freed += bytes; - ++files; + if (start.dryRun) { + freed += cache::tree_bytes(build); + continue; + } + const platform::fs::Removal removed { platform::fs::tree_remove(build) }; + freed += removed.bytes; + failed += removed.failed; } } } - if (wants("budget", true) && !dryRun) { - const cache::Sweep one { cache::enforce_budget(base::join_path(platform::dirs::cache_directory(), "workspaces"), impl.cache_budget(), - base::file_name(impl.workspaceDirectory_), now) }; + if (start.budget && !start.dryRun) { + const cache::Sweep one { cache::enforce_budget(start.workspacesRoot, start.limits, base::file_name(start.workspaceDirectory), start.now) }; freed += one.bytes; files += one.files; failed += one.failed; } - { - std::lock_guard lock(impl.cacheMutex_); - impl.lastSweepAt_ = std::chrono::duration_cast(std::chrono::system_clock::now().time_since_epoch()).count(); - impl.lastSweepFreed_ = freed; - impl.lastSweepFiles_ = files; - impl.lastSweepFailed_ = failed; - impl.cacheReportAt_.reset(); // the next report is computed at once, from what is left - } - impl.journal.add("cache-swept", Json { { "bytes", freed }, { "files", files }, { "instances", removed }, - { "failed", failed }, { "origin", "command" } }); - impl.update_status(); - return Json { { "ok", true }, { "freedBytes", freed }, { "files", files }, { "instances", removed }, { "roots", 1 }, { "dryRun", dryRun } }; + Json outcome { { "origin", "command" }, { "dryRun", start.dryRun }, { "bytes", freed }, { "files", files }, + { "instances", instances }, { "failed", failed } }; + if (!start.dryRun) { + try { + outcome["report"] = cache::report(start.cacheDirectory, start.limits, std::chrono::system_clock::now(), start.ownToken); + } catch (...) { + outcome["report"] = Json::object(); // the reply stands even when a fresh report does not + } + } + return outcome; } bool Workspace::restart_core_engine() { diff --git a/src/orchestrator/workspace.cppm b/src/orchestrator/workspace.cppm index e0316f3..1b15576 100644 --- a/src/orchestrator/workspace.cppm +++ b/src/orchestrator/workspace.cppm @@ -15,6 +15,7 @@ import mcppls.engine; import mcppls.engine.payload; import mcppls.engine.native.index; import mcppls.orchestrator.client; +import mcppls.orchestrator.cache; import mcppls.config.settings; export namespace mcppls::orchestrator { @@ -92,7 +93,10 @@ struct SessionOptions { // `tool_run` carries what the external-program runner wrote down about one run, so the // workspace it belongs to can journal it (design 4.6). // `bundle_written` carries the outcome of a diagnostic bundle an editor asked for (issue #23 fix plan F18). -enum class EventKind { client_message, client_closed, engine_event, model_loaded, external, review_finished, tool_run, bundle_written, cache_swept }; +// `deferred_answer` carries the reply of a request whose heavy half ran off the loop (the sweep +// command, 0.0.10 review): a thread posts it, the loop answers the client, so the loop itself is +// never held by the work -- the same shape as `bundle_written`, for a request reply. +enum class EventKind { client_message, client_closed, engine_event, model_loaded, external, review_finished, tool_run, bundle_written, cache_swept, deferred_answer }; struct Event { EventKind kind { EventKind::client_message }; @@ -116,6 +120,34 @@ bool is_build_file(std::string_view name); // The files a watch (dynamic or the W9.3 polling fallback) cares about under `root`. std::map watched_files_snapshot(std::string_view root); +// A sweep the session asked for, prepared on its loop (0.0.10 review): the loop-side half +// (`Workspace::prepare_sweep`) decides the categories and bounds and takes the single-flight +// token, and this -- plain data only -- is what the heavy half runs with, on any thread. When +// `started` is false nothing runs and `immediate` is that root's whole answer (a sweep was in +// flight already). +struct SweepStart { + bool started { false }; + bool dryRun { false }; + Json immediate { Json::object() }; + // Valid when started; every path and limit the walks need, with no Workspace state in it. + bool copies { true }, instances { true }, trash { true }, staleCommands { false }, budget { true }; + bool engineLive { false }; // S3 5.8: staleCommands only with no live engine + std::int64_t bound { 0 }; // the file-clock bound a live generation's writes stay above + std::string cacheDirectory; // this instance's own (its contexts) + std::string workspaceDirectory; // the shared one (its `instances/`) + std::string workspacesRoot; // /workspaces, the budget's view + std::string ownToken; // the report's "own" marker + cache::Budget limits {}; + std::chrono::seconds grace { 0 }; + std::chrono::system_clock::time_point now {}; +}; + +// The heavy half of a prepared sweep (0.0.10 review): the walks and removals, the filesystem and +// pure helpers only -- no Workspace state, so any thread may run it. Returns the `cache_swept` +// event's payload (`origin`, `dryRun`, the numbers, and a fresh `report` for a real sweep); the +// caller delivers it as an event and answers the request from the same numbers. +Json run_prepared_sweep(const SweepStart& start); + class Workspace { public: Workspace(std::string root, std::string key, SessionOptions options, engine::PayloadPaths payload, bool payloadCorrupt, bool kitEnabled, @@ -175,15 +207,19 @@ public: // ---- the cache mcppls owns (0.0.10 plan C-7, C-8, C-9, C-13.1) --------------------- // The classified cache report this root serves (`cxxModules/cache`): numbers cached for at most // 30 s and recomputed at once after a sweep, plus the state, engines and paths a hub needs. - // Read-only: it never cleans. + // Read-only: it never cleans. A stale snapshot is answered at once and refreshed off the loop + // (0.0.10 review): walking a full cache tree is not what the event loop is for. Json cache_report() const; - // `mcppls.sweepCache` (S3 5.8): removes what `categories` name under the sweep's safety rules -- - // no engine stopped, no canonical BMI touched, nothing a live generation holds mapped. Answers - // `{ok, freedBytes, files, instances, roots, dryRun}` (+ `alreadyRunning` when a sweep is in - // flight and nothing was done). - Json sweep_cache(const Json& params); - // A background sweep (start path, tick or budget) finished; its numbers go to the journal, the - // status' `cache` fragment and the report cache. + // `mcppls.sweepCache` (S3 5.8), split in two (0.0.10 review) so the session's event loop is + // never held by the removals: this, the loop-side half, decides the categories under the + // sweep's safety rules -- no engine stopped, no canonical BMI touched, nothing a live + // generation holds mapped -- takes the single-flight token, and packs everything the heavy + // half needs into plain data. `run_prepared_sweep` does the walking, anywhere. + SweepStart prepare_sweep(const Json& params); + // A background sweep (start path, tick, budget or report refresh) finished; its numbers go to + // the journal, the status' `cache` fragment and the report cache -- and the single-flight + // token is released here, on the loop, for every origin (a pass asked for while one ran is + // coalesced and retriggered from here, never queued behind it). void handle_cache_swept(const Json& outcome); // ---- the review an editor asks for (overall design 7.7) ---------------------------- diff --git a/src/server/session.cpp b/src/server/session.cpp index c51ca45..9efbedb 100644 --- a/src/server/session.cpp +++ b/src/server/session.cpp @@ -216,6 +216,11 @@ class Session { case EventKind::cache_swept: if (auto* root = root_by_key_(event.rootKey)) root->handle_cache_swept(event.message); break; + case EventKind::deferred_answer: + // A request whose heavy half ran off the loop (the sweep command): the loop answers, + // exactly as if the work had finished here -- see the comment at EventKind. + if (event.message.is_object()) reply_(event.message.value("id", Json {}), event.message.value("result", Json(nullptr))); + break; } } @@ -310,7 +315,10 @@ class Session { } // C-13.1 (plan 2026-10-03): the sweep command -- what no engine holds is removed, nothing is // stopped and no canonical BMI is touched (S3 5.8). `arguments: [{ root, categories, - // dryRun, maxBytes }]`; no arguments names every root. + // dryRun, maxBytes }]`; no arguments names every root. The removals run off the loop + // (review 2026-10-03: a full walk of a grown cache held every request behind it, and the + // lease's own renewal with them): this decides per root, a thread does the walking, and + // the answer arrives as a deferred_answer event -- every root's numbers added together. if (method == lsp::method::WORKSPACE_EXECUTE_COMMAND && params.value("command", std::string {}) == "mcppls.sweepCache") { const Json arguments = params.value("arguments", Json::array()); Json options; @@ -320,18 +328,51 @@ class Session { if (auto path = base::uri_to_path(wanted)) wantedPath = platform::fs::canonical_path(*path); else wantedPath = platform::fs::canonical_path(wanted); } - std::size_t swept { 0 }; - Json answer; + struct Target { + std::string key; + orchestrator::SweepStart start; + }; + std::vector targets; for (auto& root : roots_) { if (wantedPath && !base::same_path(*wantedPath, root->root())) continue; - answer = root->sweep_cache(options); - ++swept; + targets.push_back({ root->key(), root->prepare_sweep(options) }); } - if (swept == 0) { + if (targets.empty()) { reply_error_(id, lsp::INVALID_PARAMS, "no root serves the cache a sweep was asked for"); return; } - reply_(id, std::move(answer)); + std::thread { [events = events_, targets = std::move(targets), id]() mutable { + std::uint64_t freed { 0 }; + std::size_t files { 0 }, instances { 0 }, failed { 0 }, running { 0 }; + bool dryRun { false }; + bool allBusy { true }; + for (auto& target : targets) { + if (!target.start.started) continue; // that root had a sweep in flight; it freed nothing here + allBusy = false; + dryRun = target.start.dryRun; + ++running; + Json outcome; + try { + outcome = orchestrator::run_prepared_sweep(target.start); + } catch (...) { + // The walk itself failed: the event still releases the single-flight token. + outcome = Json { { "origin", "command" }, { "dryRun", target.start.dryRun }, { "bytes", std::uint64_t { 0 } }, + { "files", std::size_t { 0 } }, { "instances", std::size_t { 0 } }, { "failed", std::size_t { 1 } } }; + } + events->push(Event { EventKind::cache_swept, outcome, 0, {}, target.key, {} }); + freed += outcome.value("bytes", std::uint64_t { 0 }); + files += outcome.value("files", std::size_t { 0 }); + instances += outcome.value("instances", std::size_t { 0 }); + failed += outcome.value("failed", std::size_t { 0 }); + } + events->push(Event { EventKind::deferred_answer, + Json { { "id", id }, + { "result", Json { { "ok", true }, { "freedBytes", freed }, { "files", files }, + { "instances", instances }, { "failed", failed }, + { "roots", targets.size() }, { "dryRun", dryRun }, + { "alreadyRunning", allBusy } } } }, + 0, {}, {}, {} }); + } }.detach(); return; } // overall design 7.7: the review of the workspace's changes, run in the background, its findings published as diagnostics. @@ -572,11 +613,16 @@ class Session { // Indexed by EventKind's own value: every enum member has its row, in the enum's // order. A new EventKind without its row here is an out-of-bounds read of a // string_view -- the 0.0.10 cache_swept crash was exactly that, found by the E2E. - static constexpr std::array KINDS { "a client message", "the client closing", "an engine event", "a loaded model", - "an external event", "a finished review", "a tool run", "a written bundle", - "a swept cache" }; + // The assert makes the compiler say it first (review 2026-10-03), the guard below + // makes a future miss degrade to a name instead of undefined behaviour. + static constexpr std::array KINDS { "a client message", "the client closing", "an engine event", "a loaded model", + "an external event", "a finished review", "a tool run", "a written bundle", + "a swept cache", "a deferred answer" }; + static_assert(KINDS.size() == static_cast(EventKind::deferred_answer) + 1, + "every EventKind has its row in KINDS (the last member is the count)"); + const auto index { static_cast(event->kind) }; what = event->kind == EventKind::client_message ? event->message.value("method", std::string { "a response" }) - : std::string { KINDS[static_cast(event->kind)] }; + : index < KINDS.size() ? std::string { KINDS[index] } : std::format("event {}", index); if (event->kind == EventKind::engine_event && event->message.is_object()) { std::string detail { event->message.value("kind", std::string {}) }; if (const Json* inner = lsp::find(event->message, "message"); inner != nullptr && inner->is_object()) { diff --git a/tests/test_cache.cpp b/tests/test_cache.cpp index 486a020..ed5a0fd 100644 --- a/tests/test_cache.cpp +++ b/tests/test_cache.cpp @@ -6,6 +6,7 @@ import mcppls.os; import mcppls.base.path; import mcppls.platform.dirs; import mcppls.platform.fs; +import mcppls.platform.process; import mcppls.engine.clangd.bmi; import mcppls.orchestrator.cache; import mcppls.orchestrator.instance; @@ -149,6 +150,7 @@ int main() { const auto now { std::chrono::system_clock::now() }; const std::int64_t nowMs { std::chrono::duration_cast(now.time_since_epoch()).count() }; const std::string instances { mcppls::base::join_path(workspace, "instances") }; + const std::chrono::hours grace { 24 }; write_instance(mcppls::base::join_path(instances, "aaaaaaaaaaaaaaaa"), nowMs, "aaaaaaaaaaaaaaaa"); write_instance(mcppls::base::join_path(instances, "bbbbbbbbbbbbbbbb"), nowMs - 120'000, "bbbbbbbbbbbbbbbb"); // 2 min stale @@ -157,21 +159,51 @@ int main() { // `cccc` says nothing (a 0.0.9 leftover); its tree is fresh, so the grace keeps it. (void)fs::write_file(mcppls::base::join_path(instances, "cccccccccccccccc/model.a.json"), "{}"); - const cache::Sweep sweep { cache::sweep_instances(workspace, now, std::chrono::hours { 24 }) }; + const cache::Sweep sweep { cache::sweep_instances(workspace, now, grace) }; expect(sweep.instances == 1) << sweep.instances; expect(fs::is_directory(mcppls::base::join_path(instances, "aaaaaaaaaaaaaaaa"))) << "the live one stays"; expect(!fs::exists(mcppls::base::join_path(instances, "bbbbbbbbbbbbbbbb"))); expect(fs::is_directory(mcppls::base::join_path(instances, "cccccccccccccccc"))) << "fresh without a self-description: the grace holds"; - // The tick's cheap half: rename only, never remove. - const std::size_t renamed { cache::rename_dead_instances(workspace, std::chrono::duration_cast(now.time_since_epoch()), "dddddddddddddddd") }; - expect(renamed == 1) << "the stale one is renamed aside"; - expect(fs::is_directory(mcppls::base::join_path(instances, "cccccccccccccccc.trash-dddddddddddddddd"))); + // The tick's cheap half: rename only, never remove -- and the SAME grace the sweep gives a + // directory that says nothing (review 2026-10-03): a 0.0.9 guest working right now must not + // be renamed out from under a 0.0.10 owner's tick, whatever its mtime says elsewhere. + const std::size_t renamedFresh { cache::rename_dead_instances(workspace, std::chrono::duration_cast(now.time_since_epoch()), "dddddddddddddddd", grace) }; + expect(renamedFresh == 0) << std::format("the fresh undescribed directory is not renamed: {}", renamedFresh); + expect(fs::is_directory(mcppls::base::join_path(instances, "cccccccccccccccc"))) << "still under its own name"; + + // A described one whose heartbeat is stale, and an undescribed one past its grace: both go. + write_instance(mcppls::base::join_path(instances, "ffffffffffffffff"), nowMs - 120'000, "ffffffffffffffff"); + const std::string oldLeftover { mcppls::base::join_path(instances, "gggggggggggggggg") }; + const std::string oldModel { mcppls::base::join_path(oldLeftover, "model.old.json") }; + (void)fs::create_directories(oldLeftover); + (void)fs::write_file(oldModel, "{}"); + std::filesystem::last_write_time(oldModel, std::chrono::file_clock::now() - std::chrono::hours { 48 }); + const std::size_t renamed { cache::rename_dead_instances(workspace, std::chrono::duration_cast(now.time_since_epoch()), "dddddddddddddddd", grace) }; + expect(renamed == 2) << std::format("the stale heartbeat and the past-grace leftover: {}", renamed); + expect(fs::is_directory(mcppls::base::join_path(instances, "ffffffffffffffff.trash-dddddddddddddddd"))); + expect(fs::is_directory(oldLeftover + ".trash-dddddddddddddddd")); // And a sweep removes what the tick renamed aside. - const cache::Sweep taken { cache::sweep_instances(workspace, now, std::chrono::hours { 24 }) }; - expect(taken.instances == 1) << "the renamed-aside directory is taken"; + const cache::Sweep taken { cache::sweep_instances(workspace, now, grace) }; + expect(taken.instances == 2) << "the renamed-aside directories are taken"; expect(fs::is_directory(mcppls::base::join_path(instances, "aaaaaaaaaaaaaaaa"))) << "the live one survived the second sweep"; - expect(!fs::exists(mcppls::base::join_path(instances, "cccccccccccccccc.trash-dddddddddddddddd"))); + expect(fs::is_directory(mcppls::base::join_path(instances, "cccccccccccccccc"))) << "the fresh leftover survived it too"; + fs::remove_all(workspace); + }; + + "a heartbeat slightly ahead of this clock is alive; one far ahead is not trusted"_test = [] { + // A clock stepped back (NTP, waking from sleep) puts fresh heartbeats in the future: reaping + // every live instance for that would be worse than believing them (review 2026-10-03). + const std::string workspace { scratch("clock") }; + const auto now { std::chrono::system_clock::now() }; + const std::int64_t nowMs { std::chrono::duration_cast(now.time_since_epoch()).count() }; + const std::string instances { mcppls::base::join_path(workspace, "instances") }; + write_instance(mcppls::base::join_path(instances, "stepedback"), nowMs + 10'000, "stepedback"); + write_instance(mcppls::base::join_path(instances, "farfuture"), nowMs + 7 * 24 * 3600'000, "farfuture"); + const cache::Sweep sweep { cache::sweep_instances(workspace, now, std::chrono::hours { 24 }) }; + expect(sweep.instances == 1) << "only the far-future one is judged dead"; + expect(fs::is_directory(mcppls::base::join_path(instances, "stepedback"))); + expect(!fs::exists(mcppls::base::join_path(instances, "farfuture"))); fs::remove_all(workspace); }; @@ -209,13 +241,36 @@ int main() { expect(cache::parse_bytes("512M") == std::uint64_t { 512 } << 20); expect(cache::parse_bytes("100K") == std::uint64_t { 100 } << 10); expect(cache::parse_bytes("4096") == std::uint64_t { 4'096 }); + expect(cache::parse_bytes("0") == std::uint64_t { 0 }) << "an explicit zero asks for none of what can be removed"; expect(cache::parse_bytes("unlimited") == std::numeric_limits::max()); expect(!cache::parse_bytes("four").has_value()); + // Consumed whole, or not at all (review 2026-10-03): "12abcG" used to mean 12G and "1.5G" used to mean 1G. + expect(!cache::parse_bytes("12abcG").has_value()); + expect(!cache::parse_bytes("1.5G").has_value()); + expect(!cache::parse_bytes("4 G").has_value()); + expect(!cache::parse_bytes("G").has_value()); expect(cache::level_of(1'000, 4'000) == "ok"); expect(cache::level_of(3'000, 4'000) == "near"); expect(cache::level_of(4'001, 4'000) == "over"); }; + "a lease is taken over early only on a process the platform can name as gone"_test = [] { + using mcppls::orchestrator::lease_owner_gone; + // No pid or no stamp recorded: the heartbeat alone decides, the process tables are not asked. + expect(!lease_owner_gone(0, "stamp")); + expect(!lease_owner_gone(42, "")); + // This process, by its own identity: alive, and a different stamp means a reused pid. + const auto self { mcppls::platform::process_self() }; + if (self) { + expect(!lease_owner_gone(self->pid, self->started)) << "the process writing this lease is not gone"; + expect(lease_owner_gone(self->pid, self->started + "!")) << "the same pid with another incarnation is"; + } + // A pid nothing answers for is gone on every platform; an unreadable identity is NOT death + // (another user's process on Windows), but that case needs one to exist -- the definite + // branches are what a test can promise everywhere. + expect(lease_owner_gone(2'000'000'000, "stamp")) << "a pid no process has"; + }; + "the agent prompt is the task book the plan wrote down (D19; UI-12/UI-13 of 2026-10-03)"_test = [] { const Json facts { { "version", "0.0.10" }, { "editor", "VS Code" }, { "editorVersion", "1.95" }, { "os", "linux" }, { "arch", "x64" }, { "root", "/project" }, { "cacheRoot", "/cache" },