From 80e18ce4848ba3dec6485b50e3048ba4258cf862 Mon Sep 17 00:00:00 2001 From: Karn Date: Tue, 29 Sep 2026 01:16:45 +0530 Subject: [PATCH] docs(web): trim site copy, add benchmark charts The benchmarks page was 1,300 lines of tables and tuning history that the full report on GitHub already carries. It is now a short summary with two charts (runs passed, median time per task) and a link to the report. Every other page loses repeated facts and wordy sentences. Co-Authored-By: Claude Opus 5.5 --- packages/web/src/components/bench-charts.tsx | 217 +++ packages/web/src/routes/docs/architecture.tsx | 37 +- packages/web/src/routes/docs/benchmarks.tsx | 1295 +---------------- packages/web/src/routes/docs/commands.tsx | 19 +- packages/web/src/routes/docs/comparison.tsx | 94 +- packages/web/src/routes/docs/faq.tsx | 24 +- packages/web/src/routes/docs/index.tsx | 36 +- packages/web/src/routes/docs/permissions.tsx | 30 +- packages/web/src/routes/docs/security.tsx | 176 +-- packages/web/src/routes/index.tsx | 101 +- 10 files changed, 443 insertions(+), 1586 deletions(-) create mode 100644 packages/web/src/components/bench-charts.tsx diff --git a/packages/web/src/components/bench-charts.tsx b/packages/web/src/components/bench-charts.tsx new file mode 100644 index 0000000..aae23bb --- /dev/null +++ b/packages/web/src/components/bench-charts.tsx @@ -0,0 +1,217 @@ +import { useState } from "react"; +import { TEXT } from "@/components/md"; +import { cn } from "@/lib/utils"; + +/** + * The two benchmark charts, drawn in HTML so they reflow like the rest of the + * page. reins do wears the site's violet and Claude an orange; the pair passes + * the colorblind and contrast checks on both themes. Figures come from + * docs/benchmarks/2026-09-reins-do.md. + */ +const DO = "bg-primary"; +const CLAUDE = "bg-[#eb6834] dark:bg-[#d95926]"; + +function Legend() { + return ( +

+ + + + +

+ ); +} + +const PASS: Array<{ set: string; bars: Array<[string, number, string]> }> = [ + { + set: "8 unseen tasks", + bars: [ + [DO, 87.5, "35/40"], + [CLAUDE, 87.5, "21/24"], + ], + }, + { + set: "30 tuned tasks", + bars: [ + [DO, 82.6, "124/150"], + [CLAUDE, 90, "81/90"], + ], + }, +]; + +/** Share of runs passed, per set. */ +export function PassChart() { + return ( +
+ +
+ {PASS.map((group) => ( +
+

{group.set}

+
+ {group.bars.map(([color, pct, runs]) => ( +
+
+
+
+ + {pct}% {runs} + +
+ ))} +
+
+ ))} +
+
+ Runs passed. 8 unseen tasks: reins do 35 of 40, Claude 21 of 24, both 87.5%. 30 tuned tasks: + reins do 124 of 150 (82.6%), Claude 81 of 90 (90%). +
+
+ ); +} + +/** + * Median seconds of the passing runs, on the 30 tasks both passed at least + * once, and the speed-up the report computes from the unrounded medians. + */ +const TIMES: Array<[string, number, number, string]> = [ + ["fx-form", 8.1, 141.5, "17.4x"], + ["datecalc", 10.1, 115.7, "11.4x"], + ["mdn", 4.8, 108.1, "22.6x"], + ["flights", 15.9, 64.4, "4.0x"], + ["rustdocs", 5.9, 46.8, "7.9x"], + ["fx-settings", 3.4, 45.8, "13.5x"], + ["lit", 7.7, 42.3, "5.4x"], + ["fx-autocomplete", 3.9, 39.5, "10.2x"], + ["github", 11.0, 34.6, "3.1x"], + ["npm", 3.5, 31.7, "8.9x"], + ["fx-consent", 5.3, 31.3, "5.9x"], + ["pypi", 7.4, 30.1, "4.0x"], + ["hnsearch", 8.9, 29.1, "3.2x"], + ["cambridge", 6.5, 27.8, "4.2x"], + ["crates", 6.7, 25.1, "3.7x"], + ["debian", 4.5, 24.5, "5.4x"], + ["pydocs", 5.9, 23.5, "3.9x"], + ["amazon", 9.3, 23.2, "2.5x"], + ["cookies", 8.5, 22.6, "2.6x"], + ["gopkg", 9.1, 22.6, "2.4x"], + ["osm", 6.4, 21.9, "3.4x"], + ["fx-faq", 2.5, 21.4, "8.5x"], + ["huggingface", 5.1, 21.3, "4.1x"], + ["hackernews", 3.8, 20.0, "5.2x"], + ["wikipedia", 4.1, 19.3, "4.6x"], + ["gutenberg", 2.8, 19.0, "6.7x"], + ["wolframalpha", 6.8, 18.9, "2.7x"], + ["musicbrainz", 4.9, 17.4, "3.5x"], + ["iana", 1.8, 13.5, "7.7x"], + ["fx-risky", 0.7, 8.0, "12.2x"], +]; +const MAX = 150; +const TICKS = [0, 50, 100, 150]; +const pos = (s: number) => `${(s / MAX) * 100}%`; + +/** One row per task: reins do's median run against Claude's. */ +export function TimeChart() { + const [active, setActive] = useState(null); + const row = active === null ? null : TIMES[active]; + + return ( +
+ + +
    setActive(null)}> + {TIMES.map(([task, fast, slow], i) => ( +
  • setActive(i)} + className={cn( + "grid grid-cols-[7.5rem_1fr] items-center gap-3 rounded-sm", + active === i && "bg-muted", + )} + > + + {task}: reins do {fast} s, Claude {slow} s. + + +
  • + ))} +
+ +
+ Median seconds per task for the 30 tasks both passed. reins do was faster on every one. +
+
+ ); +} diff --git a/packages/web/src/routes/docs/architecture.tsx b/packages/web/src/routes/docs/architecture.tsx index 842e749..5f5bf84 100644 --- a/packages/web/src/routes/docs/architecture.tsx +++ b/packages/web/src/routes/docs/architecture.tsx @@ -43,47 +43,38 @@ function ArchitecturePage() {

The CLI

- The CLI is the entire interface: reins tabs, reins click,{" "} - reins screenshot and the rest of the command set. Agents use it because they - already have a shell: no MCP server to register, no per-agent setup. A skill ( - npx skills add karnstack/reins) teaches agents the loop. + The whole interface. Agents already have a shell, so there is no MCP server to register. A + skill (npx skills add karnstack/reins) teaches them the commands.

The daemon

- You never run the daemon yourself. Any CLI command spawns it on demand. It exposes an HTTP{" "} - /rpc endpoint for the CLI and holds the WebSocket that extensions dial into. - One daemon serves any number of browsers. reins kill stops it, and logs live in{" "} - ~/.reins/logs/. + Started by any CLI command. It takes HTTP /rpc calls from the CLI and holds the + WebSocket each browser's extension connects to. Logs are in ~/.reins/logs/.

The extension

- A Manifest V3 extension. Its service worker executes commands against tabs through{" "} - chrome.debugger (the Chrome DevTools Protocol), and an offscreen document holds - the persistent WebSocket to the daemon, because MV3 service workers are suspended when idle - and cannot keep long-lived sockets. + Its service worker runs commands through chrome.debugger. An offscreen document + holds the WebSocket, because Chrome suspends idle MV3 service workers.

- The extension discovers the daemon by probing a small set of candidate localhost ports, and - authenticates itself by its chrome-extension://<id> origin, a header the - browser stamps itself, which web pages and other extensions cannot forge. + It finds the daemon by trying a few localhost ports, and proves who it is by its{" "} + chrome-extension://<id> origin, which pages cannot fake.

Multiple browsers

- Install the extension in several Chromium browsers (Chrome, Brave, Edge, Arc, Dia) and each - connects to the same daemon. reins tabs lists every tab with a browser id. Pass{" "} - --browser <id> only when more than one browser is connected. reins never - guesses which browser you meant: it errors and names the ones it can see. + Each browser with the extension connects to the same daemon. With more than one connected, + pass --browser <id>. reins never guesses; it errors and lists the ones it + sees.

Element refs

- reins snapshot assigns stable refs (e5: button "Submit") to - interactive elements. Commands act by ref, which survives page repaints better than a - hand-written selector, and a CSS --selector fallback exists for everything - else. + reins snapshot gives each control a ref (e5: button "Submit"). + Refs survive repaints better than hand-written selectors. --selector takes CSS + when you need it.

Full command reference diff --git a/packages/web/src/routes/docs/benchmarks.tsx b/packages/web/src/routes/docs/benchmarks.tsx index 5ce9842..a466cbc 100644 --- a/packages/web/src/routes/docs/benchmarks.tsx +++ b/packages/web/src/routes/docs/benchmarks.tsx @@ -1,15 +1,14 @@ import { createFileRoute } from "@tanstack/react-router"; -import type { ReactNode } from "react"; -import { A, Code, H1, H2, H3, Ol, P, Shell, Table, TEXT, Ul } from "@/components/md"; +import { PassChart, TimeChart } from "@/components/bench-charts"; +import { A, Code, H1, H2, H3, P, Shell, Table, Ul } from "@/components/md"; import { seo } from "@/lib/seo"; -import { cn } from "@/lib/utils"; export const Route = createFileRoute("/docs/benchmarks")({ head: () => ({ ...seo({ title: "Benchmarks · reins", description: - "reins do (Jev) against Claude driving reins step by step on 38 browsing tasks: pass rates on a tuned dev set and an unseen holdout, speed, cost, the self-check, and every limitation we know of.", + "reins do (Jev) against Claude driving reins step by step on 38 browsing tasks: about as accurate, 4.6x faster and about 100x cheaper at the median.", path: "/docs/benchmarks", }), }), @@ -19,1278 +18,108 @@ export const Route = createFileRoute("/docs/benchmarks")({ const GH = "https://github.com/karnstack/reins/blob/main"; const REPORT = `${GH}/docs/benchmarks/2026-09-reins-do.md`; const SCRIPT = `${GH}/packages/cli/scripts/bench-do.mjs`; -const TASKS_FILE = `${GH}/packages/cli/scripts/bench/tasks.mjs`; const JEV_PRICING = "https://typesafe.ai/blog/introducing-system-one-models-and-jev"; -/** - * One row per task, computed from the benchmark runs (dev on e9331fa, holdout3 - * on f1af855, Claude on all 38 tasks, with the three Claude fx-risky runs - * scored by their page check); the raw run files aren't kept in the repo. Medians are over - * passing runs, lower-middle for even counts. Jev cost is rounded up to the - * next $0.00001 and Claude cost down to the $0.001; speed-up and cost ratio - * come from unrounded medians and are truncated. `null` is "no passing run". - * `note` points at the numbered notes under the tables. - */ -type Row = { - id: string; - tier: "fixture" | "live"; - set: "dev" | "holdout3"; - note?: string; - doPassed: string; - doS: string | null; - jevTokens: string | null; - jevCost: string | null; - manualPassed: string; - manualS: string | null; - claudeCost: string | null; - claudeTurns: number | null; - speedup: string | null; - costRatio: string | null; -}; - -const ROWS: Row[] = [ - { - id: "fx-autocomplete", - tier: "fixture", - set: "dev", - doPassed: "5/5", - doS: "3.9", - jevTokens: "8,339", - jevCost: "$0.00036", - manualPassed: "3/3", - manualS: "39.5", - claudeCost: "$0.144", - claudeTurns: 15, - speedup: "10.2x", - costRatio: "413x", - }, - { - id: "fx-datepicker", - tier: "fixture", - set: "dev", - doPassed: "0/5", - doS: null, - jevTokens: null, - jevCost: null, - manualPassed: "3/3", - manualS: "25.1", - claudeCost: "$0.087", - claudeTurns: 12, - speedup: null, - costRatio: null, - }, - { - id: "fx-consent", - tier: "fixture", - set: "dev", - doPassed: "5/5", - doS: "5.3", - jevTokens: "9,623", - jevCost: "$0.00041", - manualPassed: "3/3", - manualS: "31.3", - claudeCost: "$0.069", - claudeTurns: 9, - speedup: "5.9x", - costRatio: "172x", - }, - { - id: "fx-filters", - tier: "fixture", - set: "dev", - doPassed: "0/5", - doS: null, - jevTokens: null, - jevCost: null, - manualPassed: "3/3", - manualS: "55.7", - claudeCost: "$0.134", - claudeTurns: 12, - speedup: null, - costRatio: null, - }, - { - id: "fx-newtab", - tier: "fixture", - set: "dev", - note: "1", - doPassed: "5/5", - doS: "6.1", - jevTokens: "12,993", - jevCost: "$0.00055", - manualPassed: "0/3", - manualS: null, - claudeCost: null, - claudeTurns: null, - speedup: null, - costRatio: null, - }, - { - id: "fx-form", - tier: "fixture", - set: "dev", - doPassed: "5/5", - doS: "8.1", - jevTokens: "22,227", - jevCost: "$0.00094", - manualPassed: "3/3", - manualS: "141.5", - claudeCost: "$0.324", - claudeTurns: 40, - speedup: "17.4x", - costRatio: "347x", - }, - { - id: "fx-risky", - tier: "fixture", - set: "dev", - note: "2", - doPassed: "5/5", - doS: "0.7", - jevTokens: "1,361", - jevCost: "$0.00006", - manualPassed: "3/3", - manualS: "8.0", - claudeCost: "$0.030", - claudeTurns: 2, - speedup: "12.2x", - costRatio: "535x", - }, - { - id: "flights", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "15.9", - jevTokens: "105,994", - jevCost: "$0.00446", - manualPassed: "3/3", - manualS: "64.4", - claudeCost: "$0.559", - claudeTurns: 28, - speedup: "4.0x", - costRatio: "125x", - }, - { - id: "wikipedia", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "4.1", - jevTokens: "34,846", - jevCost: "$0.00147", - manualPassed: "3/3", - manualS: "19.3", - claudeCost: "$0.101", - claudeTurns: 7, - speedup: "4.6x", - costRatio: "69x", - }, - { - id: "github", - tier: "live", - set: "dev", - doPassed: "4/5", - doS: "11.0", - jevTokens: "51,100", - jevCost: "$0.00215", - manualPassed: "3/3", - manualS: "34.6", - claudeCost: "$0.170", - claudeTurns: 17, - speedup: "3.1x", - costRatio: "79x", - }, - { - id: "cookies", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "8.5", - jevTokens: "32,297", - jevCost: "$0.00136", - manualPassed: "3/3", - manualS: "22.6", - claudeCost: "$0.151", - claudeTurns: 8, - speedup: "2.6x", - costRatio: "111x", - }, - { - id: "arxiv", - tier: "live", - set: "dev", - doPassed: "0/5", - doS: null, - jevTokens: null, - jevCost: null, - manualPassed: "3/3", - manualS: "53.0", - claudeCost: "$0.291", - claudeTurns: 25, - speedup: null, - costRatio: null, - }, - { - id: "npm", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "3.5", - jevTokens: "16,441", - jevCost: "$0.00070", - manualPassed: "3/3", - manualS: "31.7", - claudeCost: "$0.150", - claudeTurns: 9, - speedup: "8.9x", - costRatio: "217x", - }, - { - id: "mdn", - tier: "live", - set: "dev", - note: "3", - doPassed: "5/5", - doS: "4.8", - jevTokens: "32,335", - jevCost: "$0.00136", - manualPassed: "1/3", - manualS: "108.1", - claudeCost: "$0.492", - claudeTurns: 34, - speedup: "22.6x", - costRatio: "362x", - }, - { - id: "cambridge", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "6.5", - jevTokens: "25,691", - jevCost: "$0.00108", - manualPassed: "3/3", - manualS: "27.8", - claudeCost: "$0.107", - claudeTurns: 9, - speedup: "4.2x", - costRatio: "100x", - }, - { - id: "huggingface", - tier: "live", - set: "dev", - note: "3", - doPassed: "1/5", - doS: "5.1", - jevTokens: "76,747", - jevCost: "$0.00323", - manualPassed: "3/3", - manualS: "21.3", - claudeCost: "$0.171", - claudeTurns: 8, - speedup: "4.1x", - costRatio: "53x", - }, - { - id: "amazon", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "9.3", - jevTokens: "38,873", - jevCost: "$0.00164", - manualPassed: "3/3", - manualS: "23.2", - claudeCost: "$0.165", - claudeTurns: 9, - speedup: "2.5x", - costRatio: "101x", - }, - { - id: "hackernews", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "3.8", - jevTokens: "48,945", - jevCost: "$0.00206", - manualPassed: "3/3", - manualS: "20.0", - claudeCost: "$0.133", - claudeTurns: 8, - speedup: "5.2x", - costRatio: "65x", - }, - { - id: "wolframalpha", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "6.8", - jevTokens: "36,208", - jevCost: "$0.00153", - manualPassed: "3/3", - manualS: "18.9", - claudeCost: "$0.068", - claudeTurns: 7, - speedup: "2.7x", - costRatio: "44x", - }, - { - id: "pydocs", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "5.9", - jevTokens: "46,391", - jevCost: "$0.00195", - manualPassed: "3/3", - manualS: "23.5", - claudeCost: "$0.114", - claudeTurns: 10, - speedup: "3.9x", - costRatio: "58x", - }, - { - id: "pypi", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "7.4", - jevTokens: "24,222", - jevCost: "$0.00102", - manualPassed: "3/3", - manualS: "30.1", - claudeCost: "$0.142", - claudeTurns: 12, - speedup: "4.0x", - costRatio: "139x", - }, - { - id: "datecalc", - tier: "live", - set: "dev", - doPassed: "4/5", - doS: "10.1", - jevTokens: "143,793", - jevCost: "$0.00604", - manualPassed: "3/3", - manualS: "115.7", - claudeCost: "$0.572", - claudeTurns: 47, - speedup: "11.4x", - costRatio: "94x", - }, - { - id: "fx-settings", - tier: "fixture", - set: "dev", - doPassed: "5/5", - doS: "3.4", - jevTokens: "8,795", - jevCost: "$0.00037", - manualPassed: "2/3", - manualS: "45.8", - claudeCost: "$0.136", - claudeTurns: 17, - speedup: "13.5x", - costRatio: "369x", - }, - { - id: "fx-orders", - tier: "fixture", - set: "dev", - doPassed: "0/5", - doS: null, - jevTokens: null, - jevCost: null, - manualPassed: "3/3", - manualS: "23.7", - claudeCost: "$0.084", - claudeTurns: 10, - speedup: null, - costRatio: null, - }, - { - id: "lit", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "7.7", - jevTokens: "38,667", - jevCost: "$0.00163", - manualPassed: "3/3", - manualS: "42.3", - claudeCost: "$0.203", - claudeTurns: 18, - speedup: "5.4x", - costRatio: "125x", - }, - { - id: "musicbrainz", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "4.9", - jevTokens: "60,964", - jevCost: "$0.00257", - manualPassed: "3/3", - manualS: "17.4", - claudeCost: "$0.127", - claudeTurns: 8, - speedup: "3.5x", - costRatio: "49x", - }, - { - id: "openlibrary", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "14.6", - jevTokens: "72,927", - jevCost: "$0.00307", - manualPassed: "0/3", - manualS: null, - claudeCost: null, - claudeTurns: null, - speedup: null, - costRatio: null, - }, - { - id: "crates", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "6.7", - jevTokens: "28,637", - jevCost: "$0.00121", - manualPassed: "3/3", - manualS: "25.1", - claudeCost: "$0.149", - claudeTurns: 10, - speedup: "3.7x", - costRatio: "124x", - }, - { - id: "iana", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "1.8", - jevTokens: "9,814", - jevCost: "$0.00042", - manualPassed: "3/3", - manualS: "13.5", - claudeCost: "$0.208", - claudeTurns: 5, - speedup: "7.7x", - costRatio: "505x", - }, - { - id: "osm", - tier: "live", - set: "dev", - doPassed: "5/5", - doS: "6.4", - jevTokens: "26,382", - jevCost: "$0.00111", - manualPassed: "3/3", - manualS: "21.9", - claudeCost: "$0.088", - claudeTurns: 11, - speedup: "3.4x", - costRatio: "80x", - }, - { - id: "fx-faq", - tier: "fixture", - set: "holdout3", - doPassed: "5/5", - doS: "2.5", - jevTokens: "8,796", - jevCost: "$0.00037", - manualPassed: "3/3", - manualS: "21.4", - claudeCost: "$0.114", - claudeTurns: 8, - speedup: "8.5x", - costRatio: "310x", - }, - { - id: "fx-wizard", - tier: "fixture", - set: "holdout3", - doPassed: "5/5", - doS: "4.6", - jevTokens: "13,056", - jevCost: "$0.00055", - manualPassed: "0/3", - manualS: null, - claudeCost: null, - claudeTurns: null, - speedup: null, - costRatio: null, - }, - { - id: "gutenberg", - tier: "live", - set: "holdout3", - doPassed: "5/5", - doS: "2.8", - jevTokens: "18,850", - jevCost: "$0.00080", - manualPassed: "3/3", - manualS: "19.0", - claudeCost: "$0.096", - claudeTurns: 8, - speedup: "6.7x", - costRatio: "121x", - }, - { - id: "hnsearch", - tier: "live", - set: "holdout3", - doPassed: "5/5", - doS: "8.9", - jevTokens: "72,835", - jevCost: "$0.00306", - manualPassed: "3/3", - manualS: "29.1", - claudeCost: "$0.184", - claudeTurns: 13, - speedup: "3.2x", - costRatio: "60x", - }, - { - id: "debian", - tier: "live", - set: "holdout3", - note: "4", - doPassed: "5/5", - doS: "4.5", - jevTokens: "33,937", - jevCost: "$0.00143", - manualPassed: "3/3", - manualS: "24.5", - claudeCost: "$0.150", - claudeTurns: 13, - speedup: "5.4x", - costRatio: "105x", - }, - { - id: "rustdocs", - tier: "live", - set: "holdout3", - doPassed: "5/5", - doS: "5.9", - jevTokens: "45,163", - jevCost: "$0.00190", - manualPassed: "3/3", - manualS: "46.8", - claudeCost: "$0.288", - claudeTurns: 15, - speedup: "7.9x", - costRatio: "151x", - }, - { - id: "gopkg", - tier: "live", - set: "holdout3", - doPassed: "5/5", - doS: "9.1", - jevTokens: "30,018", - jevCost: "$0.00127", - manualPassed: "3/3", - manualS: "22.6", - claudeCost: "$0.099", - claudeTurns: 10, - speedup: "2.4x", - costRatio: "79x", - }, - { - id: "xe", - tier: "live", - set: "holdout3", - doPassed: "0/5", - doS: null, - jevTokens: null, - jevCost: null, - manualPassed: "3/3", - manualS: "37.4", - claudeCost: "$0.289", - claudeTurns: 19, - speedup: null, - costRatio: null, - }, -]; - -/** - * A table with a header row, for data with more than two columns. It scrolls - * sideways inside its own box on a narrow screen so the page never does. - */ -function Grid({ - head, - rows, - wrap, -}: { - head: string[]; - rows: Array<{ key: string; cells: ReactNode[] }>; - wrap?: boolean; -}) { - return ( -
- - - - {head.map((h) => ( - - ))} - - - - {rows.map((row) => ( - - {row.cells.map((cell, i) => ( - - ))} - - ))} - -
- {h} -
- {cell} -
-
- ); -} - -const dash = (v: string | number | null, unit = "") => (v === null ? "-" : `${v}${unit}`); - -function TaskTable({ set }: { set: Row["set"] }) { - const rows = ROWS.filter((r) => r.set === set).sort((a, b) => - a.tier === b.tier ? 0 : a.tier === "fixture" ? -1 : 1, - ); - return ( - ({ - key: r.id, - cells: [ - - {r.id} - {r.note ? ({r.note}) : null} - , - r.tier, - r.doPassed, - dash(r.doS, " s"), - dash(r.jevTokens), - dash(r.jevCost), - r.manualPassed, - dash(r.manualS, " s"), - dash(r.claudeCost), - dash(r.claudeTurns), - dash(r.speedup), - dash(r.costRatio), - ], - }))} - /> - ); -} - -const TOTALS: Array<[string, string, string, string, string]> = [ - ["dev", "fixture", "30/45", "23/27", "23/24"], - ["dev", "live", "94/105", "58/63", "58/63"], - ["dev", "all", "124/150 (82.6%)", "81/90 (90%)", "81/87 (93.1%)"], - ["holdout3", "fixture", "10/10", "3/6", "3/6"], - ["holdout3", "live", "25/30", "18/18", "18/18"], - ["holdout3", "all", "35/40 (87.5%)", "21/24 (87.5%)", "21/24"], - ["both", "all", "159/190", "102/114", "102/111"], -]; - -/** Dev pass rate by round. The dev set grew as holdouts were folded in. */ -const ROUNDS: Array<[string, string, string, string, string]> = [ - ["baseline", "7b9d105", "15 tasks × 3", "26/45 (57.7%)", ""], - ["after round 1", "031e35c", "15 × 3", "33/45 (73.3%)", ""], - ["after round 2", "307e204", "15 × 5", "61/75 (81.3%)", ""], - ["holdout v1 run", "48c758d", "15 × 5", "58/75 (77.3%)", "holdout v1: 25/35 (71.4%)"], - ["after round 3", "3c8a364", "22 × 5", "87/110 (79.0%)", "holdout2: 16/40 (40%)"], - ["round 4 start", "3c8a364", "30 × 5", "101/150 (67.3%)", ""], - ["after round 4", "f1af855", "30 × 5", "122/150 (81.3%)", "holdout3: 35/40 (87.5%)"], - ["after round 5", "e9331fa", "30 × 5", "124/150 (82.6%)", ""], -]; - -/** Every kept change, with what it moved in its own A/B on the dev set. */ -const KEPT: Array<[string, string, string, string]> = [ - [ - '"Accept" inside a cookie or consent banner is not a risky click', - "ffdb1b9", - "1", - "26 to 27 of 45; consent walls like Cambridge's can be dismissed without the goal naming the button", - ], - [ - "The observation waits for a loading or navigating tab, and re-attaches after Chrome drops the debugger session", - "384ce2b", - "1", - "amazon 1/3 to 3/3; datecalc's page became readable at all", - ], - [ - "A stop verdict right after typing a search query submits it first", - "cc8217a", - "1", - 'flights 1/3 to 2/3; github\'s "blocked with the query typed" runs became passes', - ], - [ - "After a click or a typed value, wait for the page to settle", - "031e35c", - "1", - "flights 2/3 to 3/3, github 0/3 to 2/3, huggingface 1/3 to 2/3 (npm 3/3 to 2/3)", - ], - [ - "Type key by key, and resume after a dropped debugger session", - "d3401e3", - "2", - "fields that rewrite their value on keyup keep it; 52 to 55 of 75, the gain on the noisy tasks", - ], - [ - "Name an unlabelled control by its table row or group, never by its options", - "547849b", - "2", - "datecalc 0/20 to 8/10 over two runs (kept below the bar)", - ], - [ - "DONE self-check, then report-only", - "307e204, 48c758d", - "2", - "no pass moved; refusing a DONE never changed an outcome, so it now only reports doneConfidence", - ], - [ - "A rounded-probability tie is not an unusable answer", - "21a1fcf", - "2 to 3", - "removes an error stop seen in about 1 run in 6 in round 2", - ], - [ - "The observation reads open shadow roots", - "08b668d", - "3", - "mdn 0/5 to 5/5 (kept below the bar)", - ], - [ - "A stuck run is self-checked and ends done when its page is the goal", - "3c8a364", - "3", - "changed no outcome in its A/B; adds a self-check to stuck results", - ], - [ - "A control is named by what its shadow root or slot renders", - "3c4e846", - "4", - "lit 0/5 to 4/5 (kept below the bar)", - ], - [ - "A link to a new tab is opened by the extension, not by Chrome", - "a77ab0d", - "4", - "no pass change; Chrome no longer raises itself over the user's app when a run follows a target=_blank link", - ], - [ - "Off-screen pagers and controls named by a goal word are in the observation", - "e9d0438", - "4", - "iana 0/5 to 5/5; +8% tokens per call", - ], - [ - "The observation waits for in-flight XHR and fetch requests", - "47840f2", - "4", - "osm 2/5 to 5/5; median run +20%", - ], - [ - "No refocus click on the field just typed into, and clicks reach slotted web-component controls (a pair)", - "f1af855", - "4", - "openlibrary 0/5 to 4/5; each half alone moved nothing", - ], - [ - "A leading www. never makes a different site", - "8d69688", - "5", - "cannot move dev; debian left_site to done in a confirmation run (not counted)", - ], - [ - "A field whose fill head says NONE retargets to a field whose head names a fill", - "e9331fa", - "5", - "fired on no dev run; did not fix xe", - ], -]; - -function Label({ children }: { children: ReactNode }) { - return {children}; -} - function BenchmarksPage() { return ( <>

Benchmarks

- reins do hands a whole browsing goal to TypeSafe's Jev model. The other way to - get the same thing done is the agent itself driving reins one command at a time: snapshot, - click, type, read, repeat. This page measures the two against each other on 38 tasks in the - maintainer's real Chrome, on 2026-09-28. Both are scored by the same independent check on - the page the run ended on. Every number is recomputed from the raw JSON, and nothing is - rounded in a flattering direction. The full report is on GitHub:{" "} - 2026-09-reins-do.md. -

-

- {" "} - The dev figure is from a set the fixes were tuned on; the unseen figure is 8 tasks; and each - arm fails tasks the other passes. + reins do hands a whole browsing goal to TypeSafe's Jev model. The alternative + is your agent driving reins itself: snapshot, click, type, repeat. We ran both on 38 tasks + in a real Chrome on 2026-09-28, and scored both with the same check on the page each run + ended on.

-

Headline

-
    -
  • - reins do passed - 124 of 150 runs (82.6%). Claude step by step passed 81 of 90 (90%); 81 of 87 (93.1%) - without fx-newtab, a task its rules made impossible. -
  • -
  • - {" "} - reins do passed 35 of 40 runs (87.5%) on its first and only run. Claude step - by step passed 21 of 24 (87.5%). -
  • -
  • - the median - task was 4.6x faster with reins do (range 2.4x to 22.6x) and 111x cheaper in - model spend (range 44x to 535x). Pooled over those tasks' passing runs, the median{" "} - reins do run took 5.7 s and the median Claude run 24.5 s. The cost counts - only Jev; the calling agent's own turn is not in it. -
  • -
  • - of the 166 reins do runs that ended{" "} - done, 17 were wrong, and all 17 carried a self-check under 0.5 ("unsure"); 20 - of the 149 right ones (13%) were marked unsure too. -
  • -
- -

Results

- reins do columns: runs passed, then the median time, Jev input tokens and Jev - cost of the passing runs. Claude columns: runs passed, then the median time, cost and turns - of the passing runs. Speed-up and cost ratio are Claude's median over reins do - 's, shown only where both arms passed at least once. Times are wall-clock from spawn to - exit, with the tab's initial 2 s load outside the clock. Numbers in brackets point at the - notes below the tables. + About as accurate on tasks it had never seen, 4.6x faster and about 100x cheaper at the + median. Neither side wins everywhere: each one fails tasks the other passes.

-

Dev set, 30 tasks

- -

Holdout3, 8 unseen tasks

- -
    -
  1. - fx-newtab is not a fair comparison. Its link opens the docs in a new tab, and the - step-by-step prompt forbids touching any tab but the one it was given, so Claude stopped - every time and said so. It is in the Claude totals, which are also given without it. -
  2. -
  3. - fx-risky's correct outcome is a stop. reins do ended{" "} - risky_action 5 of 5 times without clicking; Claude stopped before the delete - 3 of 3 times. The runner first scored those Claude runs as failures by a rule meant for{" "} - reins do's status; they are counted here by the page check, as the runner now - does. -
  4. -
  5. - One passing run only, so that median is a single sample: mdn's Claude figures, - huggingface's reins do figures. -
  6. -
  7. - Every debian run ended left_site, not done: the search form on - www.debian.org submits to packages.debian.org, and on the holdout's code{" "} - reins do stopped there as a site change. The page it stopped on was the goal, - so the check passes. Round 5 treats a leading www. as the same site; a later confirmation - run ended done 5 of 5 but is not counted, because a holdout is measured once. -
  8. -
- ({ key: `${t[0]}-${t[1]}`, cells: t }))} - /> -

- Per set, the median task speed-up is 4.2x on dev (24 tasks) and 5.4x on holdout3 (6 tasks); - the median cost ratio is 111x and 105x. On the same code as the holdout3 run (f1af855), the - dev set scored 122/150 (81.3%). + +

Runs passed

+ +

+ The 30 tuned tasks are the ones reins do was fixed against over five rounds, so + its score there flatters it. The 8 unseen tasks were run once, before anyone looked at them.

-
    -
  • - reins do, dev: 901 Jev calls, 6,256,057 input tokens, $0.263. -
  • -
  • - reins do, holdout3: 205 Jev calls, 1,177,962 input tokens, $0.050. -
  • -
  • - Claude, all 38 tasks: 114 runs, 1,728 turns, $21.83. The dearest run was openlibrary #1 - (52 turns, $0.91, failed); the slowest was openlibrary #3 (318.9 s, failed). -
  • -
-

The self-check

+

Time per task

- Every done is put back to Jev once as a yes/no question (is every requirement - in the goal visibly satisfied on this page?), and the answer's probability comes back as{" "} - doneConfidence. Below 0.5 the output reads{" "} - done (unsure: self-check 0.43) and the next: line says to verify. - Recomputed from every done run of the dev and holdout3 runs: + Median run on each of the 30 tasks both passed. reins do was faster on all 30, + from 2.4x to 22.6x.

- + +

Cost

+
-
    -
  • - The 17 wrong ones: fx-filters ×5 (0.32 to 0.49), fx-datepicker ×5 (0.12 to 0.14), arxiv ×5 - (0.38 to 0.45), github ×1 (0.44), huggingface ×1 (0.08). The highest, 0.49, sits one - hundredth under the line. -
  • -
  • - The 20 right-but-unsure ones are four whole tasks: cookies, fx-newtab, wolframalpha and - gutenberg, 5 runs each (0.11 to 0.49). Unsure means "check the page", not "failed". -
  • -
- -

Where step by step failed and reins do passed

-

- Read from each failed run's final reply. Several point at reins' own step commands rather - than at Claude's reasoning: reins snapshot, which the step-by-step arm reads - pages with, did not look inside shadow roots at the time, while reins do's - observation did, and stale refs from an earlier snapshot could point at hidden elements - ("zero size"). Both are fixed in 0.6.1. The fx-wizard page also opened with its modal - already showing, for both arms, because of a CSS bug in the fixture (fixed since). +

+ Jev's spend covers Jev only. Your agent still spends a turn sending the command and checking + the result.

-
    -
  • - The link opens a new tab and the prompt forbids other tabs. - Claude stopped and asked, in 3 to 4 turns. -
  • -
  • - Run 1 (52 turns, $0.91) searched but never opened the - "Sort by" menu: its options are inside a web component and never appeared in a snapshot. - Run 2 went through Advanced Search, landed on a "Verify you are human" page and stopped - rather than click it. Run 3 never found the search field, which sits in a web component's - modal. -
  • -
  • - Runs 1 and 2: clicking the "Team" card came back "element - has zero size", Next still advanced with the default Private, and Back and Close could not - be used, so Claude stopped rather than create a Private project. Run 3 pressed Enter and - then Space on "Create project", both worked, and it created the project twice; Claude - noticed and said so. -
  • -
  • - Runs 2 and 3 never found MDN's search box (a web component): the - search button reported zero size and reins press / was rejected as an unknown - key. Run 1 got there in 34 turns. -
  • -
  • - Run 2 saw the switches without names, a click on the - right one reported zero size, and a selector probe toggled a different switch; Claude - stopped without saving and said which switch might have changed. -
  • -
-

- No step-by-step run hit its $2 cap or the 600 s limit; each failure is a stop Claude chose, - with a reply saying why. -

- -

Method

-
    -
  • - local pages served by the runner, each built - around one widget that trips browser agents: an autocomplete that only commits on a picked - suggestion, a calendar popover with a Done button, a consent modal over a search, a search - with a filter and a sort menu, a link to a new tab, a three-step form, a "Delete draft" - button (the right outcome is a stop), a settings panel with switches and Save, a paginated - table, an FAQ accordion with a vote, and a modal wizard with a custom radio group. -
  • -
  • - public sites, no logins, no purchases: Google - Flights, Wikipedia, GitHub, BBC Weather, arXiv, npm, MDN, Cambridge Dictionary, Hugging - Face, Amazon, Hacker News, Wolfram|Alpha, the Python docs, PyPI, a date calculator, - lit.dev, MusicBrainz, Open Library, crates.io, IANA, OpenStreetMap, Project Gutenberg, HN - Search, Debian packages, the Rust std docs, pkg.go.dev and xe.com. -
  • -
  • - a JavaScript check per task, run in the tab the run ended on. Each - was validated without reins do: false on the start page and on near-miss - states (searched but not sorted, the wrong edition, a date picked but not confirmed), true - on a goal state reached another way. They are in tasks.mjs. -
  • -
  • - the check returns true and the arm finished on its own inside its - limit. A timeout never counts (none happened). -
  • -
  • - {" "} - reins do '<goal>' --fill … --timeout 60 --tab <id> --json, 90 s - for four slow sites, at most 30 steps. 5 runs per task. -
  • -
  • - claude -p with the same goal and fill values, - allowed only the reins step commands (snapshot, click, type, fill, select, press, hover, - scroll, wait, text, screenshot), one per Bash call, --restricted with an - allowlist, no reins do, no navigation by URL, no other tabs, stop before - anything irreversible, $2 cap. 3 runs per task, because each costs about a hundred times - more. Its cost is Claude Code's reported total_cost_usd. -
  • -
  • - a holdout set is used once. After its first run it has - been seen and its failures discussed, so it folds into dev: holdout v1 (7 tasks, 25/35) - and holdout2 (8 tasks, 16/40) did. Holdout3 was frozen at f1af855 before any{" "} - reins do run on it; its one run is the unseen number here, measured before - round 5. -
  • -
  • - the maintainer's Mac, reins 0.5.0, Node 24, on a - residential network in India. Jev runs on the US West Coast, so every call crossed the - Pacific, and the reins do times include that. -
  • -
  • - Jev input tokens at $0.042 per million, output free ( - TypeSafe's announcement, read 2026-09-27; TypeSafe says it - cannot prove the price is not subsidised). Tokens are the API's own count. -
  • -
  • - medians over passing runs, lower-middle for even counts. Jev - cost rounded up, Claude cost rounded down, ratios truncated, so no rounding favours{" "} - reins do. -
  • -
-

How it got here

+

Check every done

- reins do was tuned in five rounds on the dev set. Each change was tried alone - and A/B'd on the whole dev set, and kept when the total rose by at least 3 runs, no task at - 4/5 or better fell below 3/5, and fx-risky still stopped every time. Three were kept below - that bar on judgment, because their target moved from 0/5 and the drops elsewhere traced to - unrelated coin flips. Tasks were only replaced or corrected when the task itself was broken, - never because reins do failed them. The dev set grew as holdouts were folded - in, so these rates are not one series on one set. + reins do ends done when Jev thinks the goal is met, then asks Jev + once more to confirm. 17 of 166 done runs were wrong, and all 17 came back + marked unsure. So an unsure done is the one to look at before you + report success. It is not proof of failure: 20 correct runs were marked unsure too.

- ({ key: r[0], cells: r }))} - /> - ({ key: k[1], cells: k }))} - /> -

- Round 5's total rose from 122 to 124, which the traces put down to noise: neither round-5 - mechanism fired on a dev task. -

-

Tried and reverted, each on its own A/B:

-
    -
  • - Observe settle without the re-attach: fx-form 3/3 to 1/3 (a password-manager frame drops - the debugger session). -
  • -
  • - Offer the "Open field" click only on fields that open something: flights 2/3 to 0/3. -
  • -
  • Stale-element guard scoped to row containers: no change.
  • -
  • - Page text joined per block, tried twice: arxiv's failure changed shape but did not pass. -
  • -
  • Key-by-key typing without the resume: fx-form 5/5 to 1/5.
  • -
  • A 300 ms grace for a click's navigation to start: no gain.
  • -
  • - Stable select option ids, and refusing to re-select the current option: datecalc 4/5 to - 2/5 and 1/5. -
  • -
  • A stuck self-check that credits the run's recorded actions: no outcome changed.
  • -
  • - The refocus rule alone, and the slotted-click fix alone: no gain each; kept together. -
  • -
-

Limitations

+

Where each one fails

- + reins do never passed these:

    +
  • A date picker. It picks the date and never presses the calendar's Done button.
  • +
  • A search with filters. It sets the language filter before submitting, and loses it.
  • +
  • arXiv. It opens a different paper whose title starts with the same words.
  • +
  • A paginated table. The order is on page 2, and it gives up on page 1.
  • - Jev types the query, clicks the TypeScript filter before - submitting it (the page drops unsubmitted text, as GitHub does), sorts, retypes and - submits. The final URL has the query and the sort but no language. Every run ends{" "} - done, self-check 0.32 to 0.49. -
  • -
  • - Jev opens the calendar, goes forward two months, picks - November 18, then answers DONE without pressing the calendar's Done button, which is in - its observation. Every run ends done, self-check 0.12 to 0.14. -
  • -
  • - Jev searches, then opens a different paper whose title starts - with the same words. The wanted paper is not on the newest-first results page, and the - sort control is offered and never chosen. Every run ends done, self-check - 0.38 to 0.45. -
  • -
  • - The order is on page 2. The pager is in the viewport and - among Jev's candidates, and Jev answers BLOCKED (0.94 to 0.95) at step 0 every run. No - self-check runs on a blocked stop. -
  • -
  • - The page has two search boxes. In 4 runs Jev typed into - the site-wide one, whose Enter opens the top model, then was stuck (×3, - self-check 0.04) or said done there (×1, 0.08). The pass used the list's own - "Filter by name" field. -
  • -
  • - The miss followed the "advanced search" link, which drops the - typed query. After typing, Jev's choice between that link and giving up (which reins turns - into a submit) is a near tie, so it flips from run to run. -
  • -
  • - The miss clicked Calculate before setting the end day, then - flipped the end-day select between 1 and 2 until the 30-step budget ran out. -
  • -
  • - Stops needs_text at step 0: xe labels its - amount input "Receiving amount", the same as the converted output, so no field matches the - amount fill. A re-run with the currency fills the task first lacked was still 0/5. - Unsolved. + xe.com. The site labels its input and its output the same, so the amount never lands.
+

Claude driving reins step by step never passed these:

    +
  • Open Library. The sort menu and the search field sit inside web components.
  • - On this suite every wrong done was - marked unsure, but that is 166 runs on 38 tasks, and the 17 wrong ones come from 5 tasks. - Verify the page before reporting success. -
  • -
  • - reins do never makes up - text; a field the fills do not cover stops the run with needs_text, and a - site that mislabels its fields (xe) stops it even when the value was given. -
  • -
  • - It reads open shadow roots, never closed ones. A - page that opens a tab from script (window.open) can still bring Chrome to the - front. The risky-click stop is a heuristic on English words (buy, pay, send, delete…) and - unlabelled buttons; other languages and odd labels can slip past it. -
  • -
  • - 38 tasks, one machine, one network, one day; n=5 for{" "} - reins do, n=3 for Claude. A live site can change tomorrow. The fixture pages, - the checks and every fix were written by Claude agents working for the maintainer. -
  • -
  • - Five rounds of fixes were tuned on the dev set, so 82.6% is - optimistic. The unseen 35/40 is the number that says how it generalises, and it was - measured before round 5. It is 8 tasks: xe alone is 5 of its 40 runs, and its 5 debian - passes ended left_site (note 4). -
  • -
  • - The Jev figure is only Jev: the calling agent - still spends its own turn to issue the command, read the result and verify the page. The - Claude figure is the whole session, including Claude Code's own system prompt and tool - definitions (17 of 114 replies mention the maintainer's claude.ai connectors, so those - were in context too). -
  • -
  • - Claude plus reins' step commands, and - some of its failures are the step commands' (reins snapshot does not read - shadow roots). A better step toolkit, URL navigation or other tabs would score higher and - change the speed and cost. -
  • -
  • - reins do sends the goal, your fill values, the URL - and title, the visible text and the page's control labels to TypeSafe. The full list is - under reins do and TypeSafe on the security page. - Step by step sends page content to Anthropic. Either way it is opt-in. + A modal wizard. Clicks on a custom radio card reported "zero size". This and Open Library + were mostly gaps in reins snapshot, fixed in 0.6.1.
  • +
  • A link that opens a new tab, because its rules forbade other tabs.
+

+ It is a small benchmark: 38 tasks, one machine, one day, from India to Jev's US servers. + Live sites change. Jev input costs $0.042 per million tokens,{" "} + TypeSafe's published price. reins do sends page text + to TypeSafe; here is exactly what. +

-

Reproduce

+

Run it yourself

- Other flags: --tier fixture|live, --tasks id1,id2,{" "} - --claude-budget-usd, --out-dir and --dry (a self-test - with no browser and no spend). It needs a TypeSafe key (reins key set typesafe) - and claude on PATH. It costs money: about $0.32 of Jev for the 190 runs here - and $21.83 of Claude for 114. It opens tabs in your real browser, in the foreground, for the - whole run: about 25 minutes for dev, 5 for holdout3 and 90 for the Claude arm. Holdout3 is - no longer unseen. The script is bench-do.mjs. Raw run files aren't kept - in the repo; these commands regenerate them. + Needs a TypeSafe key (reins key set typesafe) and claude on your + PATH. It opens tabs in your real browser for about two hours and spends about $22, nearly + all of it on Claude. --dry checks the setup with no browser and no spend. The + script is bench-do.mjs.

-

Earlier version of this benchmark

+

Full report

- The first version (2026-09-27) ran 4 live tasks (flights, wikipedia, github, cookies) 5 - times per arm. On the build that shipped then, reins do passed 14/20 and Claude - step by step 19/20; github was 1/5 for reins do, and on the tasks it passed{" "} - reins do was 4.6x to 7.9x faster. It had no holdout and no fixture tier, which - is why this version exists. + Per-task numbers, how each task is checked, every fix and what it moved, the ones we + reverted, and the traces behind each failure: 2026-09-reins-do.md on + GitHub.

); diff --git a/packages/web/src/routes/docs/commands.tsx b/packages/web/src/routes/docs/commands.tsx index 83ad5d7..aac7df2 100644 --- a/packages/web/src/routes/docs/commands.tsx +++ b/packages/web/src/routes/docs/commands.tsx @@ -23,11 +23,9 @@ const GROUPS: Array<{ id: string; title: string; intro?: ReactNode; rows: [strin title: "Delegate", intro: ( <> - reins do hands a small task to TypeSafe's Jev model, which picks each click and field in - about 0.2 s. The agent supplies every typed value and verifies the result. Page labels are - single-quoted so a $ or a backtick in them never expands. Needs a TypeSafe key (reins key - set typesafe). Measured against an agent driving the step commands itself:{" "} - Benchmarks. + reins do hands a small task to TypeSafe's Jev model, which picks each click. Your agent + supplies every typed value and checks the result. Needs a TypeSafe key (reins key set + typesafe). How it does: Benchmarks. ), rows: [ @@ -152,7 +150,7 @@ const GROUPS: Array<{ id: string; title: string; intro?: ReactNode; rows: [strin id: "policy", title: "Site permissions", intro: - "The shell can inspect and tighten the per-site policy, never loosen it. Grants happen in the extension popup. See the Site permissions page for the full model.", + "The shell can tighten the per-site policy, never loosen it. Grants happen in the extension popup.", rows: [ [ "reins policy [--browser ]", @@ -220,12 +218,9 @@ function CommandsPage() { <>

Commands

- The CLI is the whole interface: agents shell out to it, and so can you. The commands that - act on a page or a tab share three flags: --tab <id> (the active tab by - default), --browser <id> (only needed when several browsers are - connected, and the ids come from reins tabs) and --json for raw - results. The management commands (status, doctor,{" "} - kill, help) take none of them. + Page and tab commands share three flags: --tab <id> (default: the active + tab), --browser <id> (only when several browsers are connected) and{" "} + --json. Ids come from reins tabs.

{GROUPS.map((group) => ( diff --git a/packages/web/src/routes/docs/comparison.tsx b/packages/web/src/routes/docs/comparison.tsx index 18d4d5c..f2b97cb 100644 --- a/packages/web/src/routes/docs/comparison.tsx +++ b/packages/web/src/routes/docs/comparison.tsx @@ -1,5 +1,5 @@ import { createFileRoute } from "@tanstack/react-router"; -import { A, H1, H2, P, Table } from "@/components/md"; +import { A, H1, H2, P } from "@/components/md"; import { seo } from "@/lib/seo"; export const Route = createFileRoute("/docs/comparison")({ @@ -19,91 +19,49 @@ function ComparisonPage() { <>

How it compares

- agent-browser and{" "} - dev3000 (Vercel Labs) and{" "} - playwright-mcp (Microsoft) are all - browser tooling for coding agents, and they all start from the same place: by default each - one launches and manages a browser for the agent. reins starts from the other end. It hands - the agent the browser you already have open. + The other browser tools for coding agents launch a browser of their own by default. reins + uses the one you already have open, so you are already logged in everywhere.

-

reins

-
-

agent-browser

-

- agent-browser is a fast, general automation CLI that owns its browser. reins puts an - extension inside the browsers you already run, so every session is authenticated by - definition, nothing new launches, and no debug port is ever exposed. The daemon only accepts - the extension's unforgeable origin on 127.0.0.1. + agent-browser (Vercel Labs) is a + fast automation CLI that launches its own Chrome for Testing. It can reuse a profile's + logins or attach to a running Chrome if you set that up. It adds HAR recording, request + mocking and web vitals.

- If you need headless fleets, request mocking or CI runs, agent-browser is the better fit. If - the task is "act as me, in my browser", that is reins. + Pick it for headless runs, CI and request mocking. Pick reins to act as you, in your + browser.

dev3000

-

- dev3000 solves a different problem: it wraps your dev server, launches a monitored browser, - and merges server logs, console, network and screenshots into one timeline an AI can debug - from. That is dev-loop observability, not general browser control. + dev3000 (Vercel Labs) wraps your dev + server, launches a monitored browser, and merges server logs, console, network and + screenshots into one timeline for an AI to debug from.

- They compose. dev3000 watches the app you are building, and reins drives the rest of your - browser: dashboards, docs, the third-party service you are integrating. + Different job, and they combine well: dev3000 watches the app you are building, reins drives + everything else in your browser.

playwright-mcp

-

- The closest comparison: its extension mode can also drive existing tabs in your real browser - (Chrome and Edge only). The defaults differ. playwright-mcp launches a Playwright-managed - browser with its own persistent profile, and everything flows through an MCP server you - register in each client. reins is a plain CLI, so any agent with a shell drives your - everyday browsers with no per-agent setup, and one daemon serves them all at once. + playwright-mcp (Microsoft) is an + MCP server that launches a Playwright browser with its own profile. It has an opt-in + extension mode that drives your real Chrome or Edge tabs, the closest thing to reins. You + register it in each agent. +

+

+ Pick it for Firefox and WebKit, device emulation, or clean isolated sessions. Pick reins for + a plain CLI any agent can use with no setup, across all your Chromium browsers at once.

+ +

What reins adds

- Pick playwright-mcp for cross-engine coverage (Firefox, WebKit), device emulation, or - clean-room isolated sessions. Pick reins when the point is acting as you, in the browser you - already work in. + An extension inside your browser instead of a debug port, per-site permission tiers, a + redacted audit trail, and raw CDP when the built-in commands are not enough.

); diff --git a/packages/web/src/routes/docs/faq.tsx b/packages/web/src/routes/docs/faq.tsx index 99d9ca3..d0e1d98 100644 --- a/packages/web/src/routes/docs/faq.tsx +++ b/packages/web/src/routes/docs/faq.tsx @@ -7,61 +7,60 @@ const FAQS = [ id: "remote", question: "Is anything ever sent to a remote server?", answer: - "Not by default. The extension talks to exactly one thing: the reins daemon on 127.0.0.1 on your machine. There is no analytics, no telemetry, and no remote code. The one exception is opt-in: once you save a TypeSafe API key, reins do sends page state (the goal, the tab's URL and title, visible text, interactive element labels and values, recent actions) from the daemon to TypeSafe. The security page lists exactly what is sent. Your browser still reaches the internet the way it always did, because reins drives the browser you already use rather than replacing it.", + "Not by default. The extension talks only to the reins daemon on 127.0.0.1. No analytics, no telemetry, no remote code. The one opt-in exception is reins do: once you save a TypeSafe key, it sends page state to TypeSafe while a run is working. The security page lists exactly what.", }, { id: "browsers", question: "Which browsers work?", answer: - "Any Chromium browser that supports Manifest V3 extensions. Chrome, Brave, Edge, Arc and Dia are all known to work. Install the extension in each browser you want agents to reach; one daemon serves them all.", + "Any Chromium browser with Manifest V3 extensions: Chrome, Brave, Edge, Arc, Dia. Install the extension in each one; one daemon serves them all.", }, { id: "agents", question: "Which agents work?", answer: - "Anything with a shell: Claude Code, Cursor, Codex, GitHub Copilot, Gemini CLI, and plain scripts. Agents with skill support learn the commands via npx skills add karnstack/reins; everything else can read reins help.", + "Anything with a shell: Claude Code, Cursor, Codex, Copilot, Gemini CLI, plain scripts. Teach it with npx skills add karnstack/reins, or point it at reins help.", }, { id: "banner", question: 'Why does Chrome show an "is being debugged" banner?', answer: - "reins executes commands through chrome.debugger, the same Chrome DevTools Protocol that powers DevTools. Chrome shows its native banner whenever a debugger is attached. That is deliberate transparency: you always know when an agent is acting on a tab.", + "reins drives tabs through chrome.debugger, the same protocol DevTools uses, and Chrome shows that banner whenever a debugger is attached. It is how you know an agent is acting.", }, { id: "daemon", question: "Do I need to run or configure the daemon?", answer: - "No. Any reins command starts the daemon on demand, and the extension finds it on its own through localhost port discovery. reins kill stops it; reins status shows what is connected.", + "No. Any reins command starts it, and the extension finds it on its own. reins status shows what is connected; reins kill stops it.", }, { id: "update", question: "How do I update reins?", answer: - "Run npm i -g @karnstack/reins@latest. The next tool command (reins tabs, say) notices the running daemon is older than the CLI and restarts it on the new version; reins restart does it right away. reins status and reins doctor only report the mismatch. The Chrome Web Store extension updates itself.", + "Run npm i -g @karnstack/reins@latest. The next command restarts the daemon on the new version, or run reins restart. The Web Store extension updates itself.", }, { id: "mcp", question: "How is this different from an MCP browser server?", answer: - "There is nothing to register per agent. reins is a plain CLI, so any tool that can run shell commands can drive the browser. And it drives your real, logged-in profile rather than a separate automation browser.", + "Nothing to register per agent. reins is a plain CLI, so anything that runs shell commands can use it, and it drives your real logged-in browser rather than a separate one.", }, { id: "stop", question: "How do I stop an agent while it is running?", answer: - "Click the reins toolbar icon and press Disconnect. The connection is cut at once. reins kill stops the daemon entirely.", + "Click the reins toolbar icon and press Disconnect. reins kill stops the daemon entirely.", }, { id: "store", question: "Can I install the extension without the Chrome Web Store?", answer: - "Yes. reins extension stages the bundled extension for Chrome's Load unpacked, with no reins allow step. The npm package carries a full copy, so it works with no store access at all. The docs page Install without the store has the walkthrough.", + "Yes. reins extension stages the copy bundled in the npm package for Chrome's Load unpacked. See Install without the store.", }, { id: "dev-builds", question: "Does reins work with unpacked dev builds of the extension?", - answer: - "Yes. Load the unpacked extension, then allow its ID once with reins allow . Store-installed extensions are allowlisted automatically.", + answer: "Yes. Load the unpacked build, then run reins allow once.", }, ]; @@ -97,8 +96,7 @@ function FaqPage() { <>

FAQ

- Quick answers about how reins works. Anything missing? Ask on{" "} - GitHub. + Missing something? Ask on GitHub.

{FAQS.map((faq) => (
diff --git a/packages/web/src/routes/docs/index.tsx b/packages/web/src/routes/docs/index.tsx index fb5896e..915cc59 100644 --- a/packages/web/src/routes/docs/index.tsx +++ b/packages/web/src/routes/docs/index.tsx @@ -19,18 +19,13 @@ function GettingStarted() { return ( <>

Getting started

-

- reins gives coding agents full control of your actual, logged-in Chromium browser, through a - CLI and a Manifest V3 extension. Claude Code, Cursor, Codex, Copilot, anything with a shell. - This page takes you from nothing to an agent driving a tab. -

+

Four steps from nothing to your agent driving your logged-in browser.

1. Install the CLI

- This installs the reins command and the daemon it manages. You never run the - daemon yourself: any command starts it on demand, it binds 127.0.0.1, and{" "} - reins kill stops it. There is nothing to configure and nothing to keep running. + This installs reins. Its daemon starts on its own when you run any command, and{" "} + reins kill stops it. Nothing to configure.

{/* Before the extension, not after it. The extension step sends the @@ -40,24 +35,16 @@ function GettingStarted() {

2. Teach your agent

- The skill teaches agents the command set and the loop below. This is the step people skip, - and skipping it is why an agent with reins installed still says it cannot open a browser. - Agents without skill support can run reins help; the CLI is self-describing. + This teaches your agent the commands. Skip it and your agent will still say it cannot open a + browser. No skill support? Point it at reins help.

3. Add the extension

- Install the reins extension from the Chrome Web Store in - every Chromium browser you want agents to reach. Chrome, Brave, Edge, Arc and Dia all work. - The extension finds the daemon on its own through localhost port discovery, and the toolbar - popover turns green when it is connected. -

-

- Prefer to skip the store? reins extension stages the bundled copy for Chrome's - Load unpacked. The walkthrough is on Install without the store. + Add the reins extension to each Chromium browser you want + agents to reach: Chrome, Brave, Edge, Arc, Dia. Its icon turns green once it finds the + daemon. No store access? See Install without the store.

-

Working from a dev build instead? Load the unpacked extension and allow its ID once:

- "]} />

4. Check

The loop agents use

-

Every page interaction is the same three beats: look, act, check.

+

Look, act, check.

- The commands that act on a page or a tab share three flags: --tab <id>{" "} - (the active tab by default), --browser <id> (only needed when several - browsers are connected) and --json for raw output. + Page commands act on the active tab unless you pass --tab <id>. Add{" "} + --json for raw output.

Full command reference How the pieces fit together diff --git a/packages/web/src/routes/docs/permissions.tsx b/packages/web/src/routes/docs/permissions.tsx index b5d55c1..86fb0be 100644 --- a/packages/web/src/routes/docs/permissions.tsx +++ b/packages/web/src/routes/docs/permissions.tsx @@ -19,15 +19,12 @@ function PermissionsPage() { <>

Site permissions

- Every site resolves to one of three tiers, in order of power: deny <{" "} - read < full. The extension checks the tier before it runs any - command against a tab. The check lives in the extension itself, the one place a process on - your machine cannot reach around, so even a misbehaving agent (or a compromised daemon) - cannot skip it. + Every site gets one of three tiers: deny, read or{" "} + full. The extension checks it before any command touches a tab, so neither the + agent nor the daemon can skip it.

- The shipped default is full everywhere, so a fresh install behaves exactly as - before. The policy model is opt-in hardening: tighten the sites you care about, or flip the + The default is full everywhere. Tighten the sites you care about, or flip the default and grant sites back one by one.

@@ -46,9 +43,8 @@ function PermissionsPage() { ]} />

- Navigation is checked on both ends: reins nav and reins open need{" "} - full on the destination host as well as the current one, so a read-only page - cannot be steered somewhere permissive. + reins nav and reins open need full on both the + current and the destination site.

Rules and matching

@@ -77,10 +73,8 @@ function PermissionsPage() {

Granting and tightening

- Grants happen only in the extension popup: click the reins icon, and the Site permissions - section offers a tier control for the current tab, the rules list, and the default. That is - deliberate. The popup is a user gesture, and an agent in your shell cannot perform one. From - the CLI you can inspect the policy and tighten it, never loosen it: + Grant access in the extension popup, under Site permissions. An agent in your shell cannot + click it, which is the point. The CLI can only look and tighten:

What a blocked agent sees

- A blocked command fails with a policy_denied error. It names the host, its - current tier, and what to do about it: + A blocked command fails with policy_denied and says what to do:

-

- The CLI prints the message and exits nonzero, so agents relay the instruction instead of - retrying. -

+

It exits nonzero, so agents pass the message on instead of retrying.

How this fits the broader trust model ); diff --git a/packages/web/src/routes/docs/security.tsx b/packages/web/src/routes/docs/security.tsx index 94c91cb..d556bc7 100644 --- a/packages/web/src/routes/docs/security.tsx +++ b/packages/web/src/routes/docs/security.tsx @@ -7,7 +7,7 @@ export const Route = createFileRoute("/docs/security")({ ...seo({ title: "Security · reins", description: - "The reins security model: localhost-only daemon, DNS-rebinding protection, extension-origin allowlisting, Chrome's native debug banner, and instant disconnect.", + "The reins security model: localhost-only daemon, extension-origin allowlisting, per-site permissions, a redacted audit trail, and exactly what reins do sends to TypeSafe.", path: "/docs/security", }), }), @@ -19,171 +19,107 @@ function SecurityPage() { <>

Security

- A tool that drives your logged-in browser has to be careful with it. reins keeps the attack - surface small by having no cloud half at all: the pieces only ever talk to each other, on + reins has no cloud half. The CLI, the daemon and the extension only talk to each other, on your machine.

-

Network surface

+

Network

  • - Everything binds 127.0.0.1. Neither the daemon nor the extension is reachable - from the network. + Everything binds 127.0.0.1. Nothing is reachable from the network.
  • - /rpc and the other daemon endpoints validate the Host header, so - web pages cannot reach the daemon even through rebound DNS. + The daemon checks the Host header, so a web page cannot reach it through DNS + rebinding.
  • - The daemon accepts extension WebSocket connections only from exact allowlisted{" "} - chrome-extension://<id> origins. The browser stamps that header itself, - so pages and other extensions cannot forge it. Dev builds are added explicitly with{" "} - reins allow <id>. + It accepts only allowlisted chrome-extension://<id> origins. Chrome + sets that header itself, so pages and other extensions cannot fake it.
-

Visibility and control

+

Seeing and stopping it

    -
  • - Chrome shows its native "is being debugged" banner whenever the extension is attached to a - tab, so you always know when an agent is acting. -
  • -
  • The toolbar popup's Disconnect toggle cuts the daemon connection at once.
  • -
  • - Nothing happens in the background: the extension only acts on explicit commands sent - through the CLI on your machine. -
  • +
  • Chrome shows its "is being debugged" banner on any tab reins is attached to.
  • +
  • Disconnect in the toolbar popup cuts the connection at once.
  • +
  • The extension acts only on commands from the CLI. Nothing runs in the background.
-

Per-site permissions

-
    -
  • - Every host resolves to a tier: deny, read or full. - The extension enforces it before any command touches a tab. The check runs inside the - extension, so nothing that speaks the protocol can skip or loosen it. That includes the - CLI, the daemon, and any other local client. -
  • -
  • - Grants happen only in the extension popup. That is a user gesture, and an agent cannot - perform it from the shell. The CLI (reins policy) can view and tighten the - policy, never loosen it. -
  • -
  • - The shipped default is full everywhere, which is today's behavior, so - tightening is opt-in. deny also redacts the site's tabs from{" "} - reins tabs. -
  • -
- Tiers, wildcard rules and matching precedence +

Site permissions

+

+ Every site is deny, read or full, checked inside the + extension before any command runs. Only a click in the popup can grant more; the CLI can + only tighten. The default is full everywhere, so this is opt-in. +

+ How site permissions work -

Trust boundary

+

What it does not protect against

- The tiers contain the agent you invited in. They are not a defense against other software on - your machine. Anything already running as your OS user sits inside the trust boundary: it - could talk to the daemon or rewrite the policy store directly, and no browser automation - tool's permission model survives local malware. The honest write-up covers what the tiers - protect against, what they do not, prompt injection, and a hardening checklist. It is the{" "} - - threat model (SECURITY.md) - - . + The tiers contain the agent you invited in. They do not stop other software running as you, + which could talk to the daemon or edit the policy directly. The{" "} + threat model{" "} + covers this, prompt injection, and a hardening checklist.

Audit trail

-
    -
  • - Every command the daemon executes, and every one the policy blocks, appends one structured - line (timestamp, command, browser, tab, host, tier, outcome, duration) to{" "} - ~/.reins/logs/audit-YYYY-MM-DD.jsonl. reins audit renders the - trail, and --denied shows only what policy blocked. -
  • -
  • - Value-bearing params are redacted before the line is written: typed text, fill values,{" "} - eval code and CDP payloads. The trail never stores what the agent typed, only - that it typed. -
  • -
  • - Audit files are pruned after 30 days. Writes are best-effort: a full disk never blocks a - command. -
  • -
+

+ Every command, and every one policy blocks, adds a line to{" "} + ~/.reins/logs/audit-YYYY-MM-DD.jsonl. Typed text, fill values, eval code and + CDP payloads are redacted, so the log shows that the agent typed, never what. Files are + deleted after 30 days. reins audit shows the trail; --denied shows + only blocks. +

reins do and TypeSafe

- reins do is off until you save a TypeSafe API key ( - reins key set typesafe, or the Jev section of the extension popup). Once a key - is saved, and only while a reins do run is working, the daemon (not the - extension) sends this to api.typesafe.ai, under your own TypeSafe account: + reins do is off until you save a TypeSafe key ( + reins key set typesafe). While a run is working, the daemon sends this to{" "} + api.typesafe.ai, under your account:

  • - the goal you gave, and your --fill names and values + your goal and your --fill values
  • the tab's URL and title
  • -
  • visible text in the viewport (up to about 6,000 characters)
  • -
  • labels, roles and current values of the page's interactive elements
  • +
  • visible text, up to about 6,000 characters
  • +
  • labels and values of the page's controls
  • the run's last 10 actions

- Never sent: password, file and hidden inputs. The extension itself makes no remote requests. -

-

- Before every click or keystroke, reins re-checks the chosen element (still present, visible, - not covered, not moving) and refuses to act when the check fails. The page still controls - its own DOM, so what sits under a chosen element can change between the check and the - action; keep --confirm for anything you would not click blind. + Password, file and hidden inputs are never sent. The key lives in{" "} + ~/.reins/credentials.json (mode 0600) and is never printed, logged or sent to a + page.

- reins do hands page state to TypeSafe's Jev model, which answers typed - multiple-choice questions: which operation, and which observed element. Jev can only choose - among elements reins actually read from the page; its output never becomes a selector, - coordinate or code. Page text can still try to steer it (prompt injection), so: + Jev can only pick from elements reins read off the page. To limit what a page can talk it + into:

  • - A click whose label contains a money, messaging or deletion word (buy, pay, send, delete, - …) stops the run unless the agent passed --confirm for that label or the goal - names it word for word. Unlabeled buttons stop too. This is a heuristic, not a guarantee: - other languages and odd labels can slip past. + Clicks labelled with words like buy, pay, send or delete stop the run unless you pass{" "} + --confirm or the goal names that label. Unlabelled buttons stop too. It + matches English words, so it is not a guarantee.
  • - A run that moves to another site stops (left_site), and site permissions - still apply on every step (full required). + Leaving the site stops the run (left_site). Site permissions apply on every + step.
  • - Ctrl-C, a dead agent, --timeout or a daemon restart stop the run before its - next action. + Ctrl-C, --timeout or a daemon restart stops it before the next action.
  • - The key file is ~/.reins/credentials.json (0600). The key is never returned - by any command, never logged, and never sent to a page. + reins re-checks each element right before acting, but a page can still swap what sits + under it. Use --confirm for anything you would not click blind.
-

- What a run costs in time and tokens, measured against an agent driving the step commands - itself, is on the Benchmarks page. -

-

Data handling

-
    -
  • - Page content and tab metadata are read through the Chrome DevTools Protocol only when your - local daemon asks, and are sent only to that daemon over localhost. -
  • -
  • - No analytics, no telemetry, no tracking, no remote code. No remote servers unless you opt - in to reins do with a TypeSafe key (see above). -
  • -
  • - The only stored state is the extension's own settings (auto-connect, cached daemon port, - connection status) and your site-permission policy, kept in chrome.storage on - your device. -
  • -
+

Data

- The full policy is at reins.tech/privacy. The code is MIT-licensed - and auditable at github.com/karnstack/reins - . + No analytics, no telemetry, no remote code. Page content goes only to your local daemon, + unless you opt in to reins do. The extension stores its settings and your site + policy in chrome.storage. Full policy:{" "} + reins.tech/privacy. Source:{" "} + github.com/karnstack/reins.

); diff --git a/packages/web/src/routes/index.tsx b/packages/web/src/routes/index.tsx index f604f8f..8d8c659 100644 --- a/packages/web/src/routes/index.tsx +++ b/packages/web/src/routes/index.tsx @@ -131,36 +131,21 @@ function WhatItDoes() { <>

What it does

    +
  • Tabs in every connected browser: list, open, focus, close.
  • +
  • Click, type, fill, select, hover, scroll, press keys, upload files, answer dialogs.
  • - Every tab, every browser. List, open, focus and close tabs across Chrome, Brave, Edge, Arc - and Dia. One daemon serves every browser that connects to it. + reins snapshot lists what you can click, with short refs to act on. CSS + selectors work too.
  • +
  • Read the page as text or a screenshot.
  • +
  • Console messages and network requests, without opening DevTools.
  • - Act on the page. Click, type, fill, select, hover, scroll, press keys, upload files, - answer dialogs, resize the window. + reins eval for JavaScript, reins cdp for any raw DevTools + Protocol call.
  • - Refs, not selectors. reins snapshot lists the interactive elements with - stable refs, and commands act by ref. A CSS --selector is there when you need - it. -
  • -
  • Read the page. Visible text, and screenshots your agent can open and reason about.
  • -
  • - Console and network, without opening DevTools. Recent messages and requests, filtered by - level, age or URL. -
  • -
  • - An escape hatch. reins eval runs JavaScript in the page.{" "} - reins cdp sends a raw Chrome DevTools Protocol command when the curated set - is not enough. -
  • -
  • - Site permissions. Every host resolves to deny, read or full, and the extension enforces it - before a command touches a tab. -
  • -
  • - An audit trail. Every command the daemon runs, and every one the policy blocks, appends - one line to ~/.reins/logs, with the values redacted. + reins do hands a whole task to Jev, TypeSafe's action model, instead of going + click by click.
@@ -183,12 +168,7 @@ function Loop() { return ( <>

The loop

-

- Every page interaction is the same three beats: look, act, check. The commands that act on a - page or a tab share three flags: --tab <id> (the active tab by default),{" "} - --browser <id> (only when more than one browser is connected) and{" "} - --json for raw output. -

+

Look, act, check.

Full command reference @@ -213,17 +193,10 @@ function HowItWorks() { <>

How it works

- Three pieces with one narrow contract between them, and all three run on your machine. The - daemon ships inside the CLI and starts on demand, so there is nothing to keep running and - nothing to register per agent. + Three pieces, all on your machine. The daemon ships inside the CLI and starts on its own. + Chrome shows its "is being debugged" banner while reins is attached to a tab.

{PATH}
-

- The extension finds the daemon by probing a small set of localhost ports, and authenticates - by its chrome-extension://<id> origin, a header the browser stamps itself - and a page cannot forge. Chrome shows its native debugging banner the whole time it is - attached. -

How the three pieces fit together ); @@ -236,9 +209,8 @@ function Permissions() { <>

Site permissions

- Every site your agent touches resolves to one of three tiers. The check lives in the - extension, the one place no process on your machine can reach around, so a misbehaving agent - cannot skip it. + Every site gets one of three tiers, checked inside the extension where an agent cannot skip + it.

- Granting more access takes a click in the extension popup. That is a user gesture, and an - agent in your shell cannot perform one. From the CLI, reins policy can inspect - the policy and tighten it. It can never loosen it. + Only a click in the extension popup grants more access. From the shell,{" "} + reins policy can tighten, never loosen.

How site permissions work @@ -270,15 +241,14 @@ function Facts() {
{INSTALL_COMMAND}], - ["Skill", {SKILL_COMMAND}], ["Browsers", "Chrome, Brave, Edge, Arc, Dia. Any Chromium that takes MV3 extensions."], ["Agents", "Claude Code, Cursor, Codex, Copilot, Gemini CLI. Anything with a shell."], - ["Binds", 127.0.0.1], - ["Hosted service", "None"], - ["Account", "None"], - ["Telemetry", "None. No analytics, no tracking, no remote code."], - ["Version", "0.x. Commands and output can still change."], + [ + "Network", + <> + Binds 127.0.0.1. No account, no hosted service, no telemetry. + , + ], ]} /> @@ -295,17 +265,7 @@ function Limits() { rows={[ ["Browsers", "Chromium only. No Firefox, no WebKit."], ["Headless", "Not supported. reins drives a browser you already have open."], - [ - "CI", - "Not the target. Use Playwright or agent-browser for a machine with nobody at it.", - ], - [ - "Two browsers", - <> - Supported, but commands then need --browser <id>. reins never - guesses which one you meant. - , - ], + ["CI", "Not the target. Use Playwright or agent-browser there."], ["Releases", "0.x. Commands, flags and output can still change."], ]} /> @@ -334,23 +294,20 @@ function Install() { return ( <>

Install

-

- Three pieces, and the second is the one people skip. Without the skill, your agent has the - CLI installed and no idea the commands exist. -

+

Three steps. Do not skip the second: without it your agent never learns the commands.

  1. The CLI. The daemon rides inside it and starts on demand.
  2. - The skill, so your agent knows the command set. Agents without skill support can read{" "} - reins help instead, but do not skip this if yours supports it. + The skill, so your agent knows the commands. No skill support? It can read{" "} + reins help.
  3. - The extension, in every browser you want agents to reach. It finds the daemon on its own, - and the toolbar icon turns green once it connects. + The extension, in each browser you want agents to reach. Its icon turns green once + connected.

    Add reins from the Chrome Web Store