From 50ba34c7235f9f97cf67b2e315d80d13a16ca7e7 Mon Sep 17 00:00:00 2001 From: Janghoon Lee <44862514+savagemanage@users.noreply.github.com> Date: Fri, 2 Oct 2026 07:23:58 +0000 Subject: [PATCH 1/3] feat(codemode): K-1 typed domain objects as code-mode globals A skill document is cheap only because the objects it drives already exist. Code Mode could previously only be written as `tools..(...)`, so a skill had to describe a tool call instead of naming a method. Mechanism, in @redrob-code/codemode: `globals` on ExecuteOptions names top-level tool namespaces that are ALSO bound as bare identifiers in the interpreter's global scope. A global is an alias for the same tool path, so there is one implementation and one authorization point, not two. Code Mode stays host-neutral: it never learns what a given name means. The builtin global seeding moves into seedBuiltinGlobals(), and BUILTIN_GLOBAL_NAMES is derived from it, so a builtin added later cannot silently become available as a host global name. assertValidGlobals refuses a name that is not a namespace of the tool tree, one that shadows a builtin, a duplicate, a non-identifier, and a tool rather than a namespace. The generated instructions gain a "Domain globals" section stating the bare and tools forms are the same call. Objects, in packages/redrob/src/tool/domain.ts: Page and Channel as real TypeScript interfaces with a single implementation each, injected under the names `page` and `channel`. - Channel is implemented: `channel.send` appends a text part to the assistant message that owns the execution, so every surface the session is attached to renders it without the engine knowing which surface that is. `channel.id` returns the session id. - Page is NOT implemented, and says so. Nothing in the engine process controls a browser page today, so its single implementation is unavailablePage, whose every method fails with the capability named: "the engine process has no browser-page control surface". It returns no plausible value, so a skill written against the interface fails loudly rather than reading an empty string. The domain namespaces are spread after the MCP catalog deliberately: a connected MCP server named `page` must not shadow the object a skill is written against. --- packages/codemode/src/codemode.ts | 15 +- packages/codemode/src/interpreter/runtime.ts | 131 ++++++++--- packages/codemode/src/tool-runtime.ts | 27 ++- packages/codemode/test/globals.test.ts | 83 +++++++ packages/redrob/src/tool/code-mode.ts | 26 ++- packages/redrob/src/tool/domain.ts | 232 +++++++++++++++++++ packages/redrob/test/tool/code-mode.test.ts | 9 +- packages/redrob/test/tool/domain.test.ts | 185 +++++++++++++++ 8 files changed, 663 insertions(+), 45 deletions(-) create mode 100644 packages/codemode/test/globals.test.ts create mode 100644 packages/redrob/src/tool/domain.ts create mode 100644 packages/redrob/test/tool/domain.test.ts diff --git a/packages/codemode/src/codemode.ts b/packages/codemode/src/codemode.ts index 14782d7c8a..1cd36e56c6 100644 --- a/packages/codemode/src/codemode.ts +++ b/packages/codemode/src/codemode.ts @@ -1,5 +1,5 @@ import { Effect, Schema } from "effect" -import { executeWithLimits } from "./interpreter/runtime.js" +import { assertValidGlobals, executeWithLimits } from "./interpreter/runtime.js" import { type HostTools, type Services, type ToolDescription, ToolRuntime } from "./tool-runtime.js" import type { Definition } from "./tool.js" @@ -38,6 +38,14 @@ export type ExecuteOptions = {}> = { code: string /** Explicit tool tree exposed to the program as `tools`. */ tools?: Tools & ToolTree> + /** + * Top-level tool namespaces ALSO bound as bare globals, so a skill document can be + * written as `await page.text()` rather than `await tools.page.text()`. Each name must + * be a namespace at the top level of `tools` and must not shadow a builtin global. + * A global is an alias for the same tool path: one implementation, one authorization + * point. Code Mode stays host-neutral - it never knows what `page` means. + */ + globals?: ReadonlyArray /** Per-execution overrides for the default resource limits. */ limits?: ExecutionLimits /** Observes decoded tool input immediately before tool execution. */ @@ -139,6 +147,7 @@ export const execute = >( ): Effect.Effect> => { const tools = (options.tools ?? {}) as HostTools> ToolRuntime.assertValidTools(tools) + assertValidGlobals(tools, options.globals ?? []) return executeWithLimits(options, resolveExecutionLimits(options.limits), ToolRuntime.searchIndex(tools)) } @@ -148,8 +157,10 @@ export const make = = {}>( ): Runtime> => { const tools = (options.tools ?? {}) as HostTools> ToolRuntime.assertValidTools(tools) + const globals = options.globals ?? [] + assertValidGlobals(tools, globals) const limits = resolveExecutionLimits(options.limits) - const prepared = ToolRuntime.prepare(tools, options.discovery?.catalogBudget) + const prepared = ToolRuntime.prepare(tools, options.discovery?.catalogBudget, globals) return { catalog: () => prepared.catalog, diff --git a/packages/codemode/src/interpreter/runtime.ts b/packages/codemode/src/interpreter/runtime.ts index 093f577765..573cdc493b 100644 --- a/packages/codemode/src/interpreter/runtime.ts +++ b/packages/codemode/src/interpreter/runtime.ts @@ -13,6 +13,8 @@ import { type Services, } from "../tool-runtime.js" import { ToolError } from "../tool-error.js" +import { isDefinition as isToolDefinition } from "../tool.js" +import { identifierSegment } from "../tool-schema.js" import type { DataValue, Diagnostic, @@ -599,6 +601,85 @@ const collectPatternNames = (pattern: AstNode, out: Array = []): Array => { + const globalScope = new Map() + globalScope.set("tools", { mutable: false, value: new ToolReference([]) }) + globalScope.set("Promise", { mutable: false, value: new PromiseNamespace() }) + globalScope.set("undefined", { mutable: false, value: undefined }) + globalScope.set("Object", { mutable: false, value: new GlobalNamespace("Object") }) + globalScope.set("Math", { mutable: false, value: new GlobalNamespace("Math") }) + globalScope.set("JSON", { mutable: false, value: new GlobalNamespace("JSON") }) + globalScope.set("Number", { mutable: false, value: new CoercionFunction("Number") }) + globalScope.set("String", { mutable: false, value: new CoercionFunction("String") }) + globalScope.set("Boolean", { mutable: false, value: new CoercionFunction("Boolean") }) + globalScope.set("Array", { mutable: false, value: new GlobalNamespace("Array") }) + globalScope.set("console", { mutable: false, value: new GlobalNamespace("console") }) + globalScope.set("parseInt", { mutable: false, value: new CoercionFunction("parseInt") }) + globalScope.set("parseFloat", { mutable: false, value: new CoercionFunction("parseFloat") }) + globalScope.set("Date", { mutable: false, value: new GlobalNamespace("Date") }) + globalScope.set("RegExp", { mutable: false, value: new GlobalNamespace("RegExp") }) + globalScope.set("Map", { mutable: false, value: new GlobalNamespace("Map") }) + globalScope.set("Set", { mutable: false, value: new GlobalNamespace("Set") }) + globalScope.set("URL", { mutable: false, value: new GlobalNamespace("URL") }) + globalScope.set("URLSearchParams", { mutable: false, value: new GlobalNamespace("URLSearchParams") }) + globalScope.set("encodeURI", { mutable: false, value: new UriFunction("encodeURI") }) + globalScope.set("encodeURIComponent", { mutable: false, value: new UriFunction("encodeURIComponent") }) + globalScope.set("decodeURI", { mutable: false, value: new UriFunction("decodeURI") }) + globalScope.set("decodeURIComponent", { mutable: false, value: new UriFunction("decodeURIComponent") }) + // Error constructors are real values, so `x instanceof Error` works and `Error("msg")` + // (with or without `new`) constructs a branded { name, message } error object. + for (const name of errorConstructors) { + globalScope.set(name, { mutable: false, value: new ErrorConstructorReference(name) }) + } + // NaN/Infinity flow as ordinary in-sandbox values (normalized to null only at the data + // boundary - see copyOut), so their global bindings must exist too, e.g. `reduce(max, -Infinity)`. + globalScope.set("NaN", { mutable: false, value: NaN }) + globalScope.set("Infinity", { mutable: false, value: Infinity }) + return globalScope +} + +/** + * Every name the interpreter seeds into the builtin global scope. A host global + * (see `ExecuteOptions.globals`) may not shadow one of these. + */ +export const BUILTIN_GLOBAL_NAMES: ReadonlySet = new Set(seedBuiltinGlobals().keys()) + +/** + * Validates the names a host asks to bind as bare globals. A global is an alias for a + * top-level namespace of the tool tree, so it must name one, must not shadow a builtin + * global, and must be a plain identifier (a bare binding cannot be written in bracket + * notation). A duplicate is refused rather than deduplicated: a repeated name in a host's + * list is a configuration mistake worth surfacing. + * + * Lives here, beside `seedBuiltinGlobals`, so the reserved-name check reads the same set + * the interpreter actually seeds. + */ +export const assertValidGlobals = (tools: HostTools, globals: ReadonlyArray): void => { + const seen = new Set() + for (const name of globals) { + if (!identifierSegment.test(name)) { + throw new Error(`Global '${name}' is not a plain identifier and cannot be bound as a global.`) + } + if (BUILTIN_GLOBAL_NAMES.has(name)) { + throw new Error(`Global '${name}' collides with a builtin global.`) + } + if (seen.has(name)) throw new Error(`Global '${name}' is listed more than once.`) + seen.add(name) + if (!Object.hasOwn(tools, name)) { + throw new Error(`Global '${name}' is not a top-level namespace of the tool tree.`) + } + const namespace = tools[name] + if (typeof namespace === "function" || isToolDefinition(namespace)) { + throw new Error(`Global '${name}' must be a namespace of tools, not a tool itself.`) + } + } +} + class Interpreter { private scopes: Array> private readonly invokeTool: (path: ReadonlyArray, args: Array) => Effect.Effect @@ -618,46 +699,28 @@ class Interpreter { invokeTool: (path: ReadonlyArray, args: Array) => Effect.Effect, toolKeys: (path: ReadonlyArray) => ReadonlyArray, logs: Array = [], + /** + * Host tool namespaces additionally bound at the top level, so a program writes + * `page.text()` instead of `tools.page.text()`. Each name resolves to the same tool + * path, so there is one implementation and one authorization point, not two. + */ + hostGlobals: ReadonlyArray = [], ) { - const globalScope = new Map() + const globalScope = seedBuiltinGlobals() this.scopes = [globalScope] this.invokeTool = invokeTool this.toolKeys = toolKeys this.logs = logs this.lastValue = undefined this.callPermits = Semaphore.makeUnsafe(TOOL_CALL_CONCURRENCY) - globalScope.set("tools", { mutable: false, value: new ToolReference([]) }) - globalScope.set("Promise", { mutable: false, value: new PromiseNamespace() }) - globalScope.set("undefined", { mutable: false, value: undefined }) - globalScope.set("Object", { mutable: false, value: new GlobalNamespace("Object") }) - globalScope.set("Math", { mutable: false, value: new GlobalNamespace("Math") }) - globalScope.set("JSON", { mutable: false, value: new GlobalNamespace("JSON") }) - globalScope.set("Number", { mutable: false, value: new CoercionFunction("Number") }) - globalScope.set("String", { mutable: false, value: new CoercionFunction("String") }) - globalScope.set("Boolean", { mutable: false, value: new CoercionFunction("Boolean") }) - globalScope.set("Array", { mutable: false, value: new GlobalNamespace("Array") }) - globalScope.set("console", { mutable: false, value: new GlobalNamespace("console") }) - globalScope.set("parseInt", { mutable: false, value: new CoercionFunction("parseInt") }) - globalScope.set("parseFloat", { mutable: false, value: new CoercionFunction("parseFloat") }) - globalScope.set("Date", { mutable: false, value: new GlobalNamespace("Date") }) - globalScope.set("RegExp", { mutable: false, value: new GlobalNamespace("RegExp") }) - globalScope.set("Map", { mutable: false, value: new GlobalNamespace("Map") }) - globalScope.set("Set", { mutable: false, value: new GlobalNamespace("Set") }) - globalScope.set("URL", { mutable: false, value: new GlobalNamespace("URL") }) - globalScope.set("URLSearchParams", { mutable: false, value: new GlobalNamespace("URLSearchParams") }) - globalScope.set("encodeURI", { mutable: false, value: new UriFunction("encodeURI") }) - globalScope.set("encodeURIComponent", { mutable: false, value: new UriFunction("encodeURIComponent") }) - globalScope.set("decodeURI", { mutable: false, value: new UriFunction("decodeURI") }) - globalScope.set("decodeURIComponent", { mutable: false, value: new UriFunction("decodeURIComponent") }) - // Error constructors are real values, so `x instanceof Error` works and `Error("msg")` - // (with or without `new`) constructs a branded { name, message } error object. - for (const name of errorConstructors) { - globalScope.set(name, { mutable: false, value: new ErrorConstructorReference(name) }) - } - // NaN/Infinity flow as ordinary in-sandbox values (normalized to null only at the data - // boundary - see copyOut), so their global bindings must exist too, e.g. `reduce(max, -Infinity)`. - globalScope.set("NaN", { mutable: false, value: NaN }) - globalScope.set("Infinity", { mutable: false, value: Infinity }) + for (const name of hostGlobals) { + // Validated by ToolRuntime.assertValidGlobals before construction; a collision + // here would silently replace a builtin, so refuse rather than overwrite. + if (BUILTIN_GLOBAL_NAMES.has(name)) { + throw new Error(`Host global '${name}' collides with a builtin global.`) + } + globalScope.set(name, { mutable: false, value: new ToolReference([name]) }) + } } run(program: ProgramNode): Effect.Effect { @@ -3365,7 +3428,7 @@ export const executeWithLimits = >( const operation = Effect.gen(function* () { const program = parseProgram(options.code) - const interpreter = new Interpreter>(tools.invoke, tools.keys, logs) + const interpreter = new Interpreter>(tools.invoke, tools.keys, logs, options.globals ?? []) const value = yield* interpreter.run(program) const result = copyOut(copyIn(value, "Execution result"), true) as DataValue return { diff --git a/packages/codemode/src/tool-runtime.ts b/packages/codemode/src/tool-runtime.ts index f4ccc61d4c..a5f70823a0 100644 --- a/packages/codemode/src/tool-runtime.ts +++ b/packages/codemode/src/tool-runtime.ts @@ -481,6 +481,24 @@ export const assertValidTools = (tools: HostTools): void => { } } +/** + * Renders the host-global aliases. Each listed namespace is reachable both as a bare + * identifier and under `tools`, so a skill document can open with `await page.text()` + * without the model having to be told that `page` is "really" a tool namespace. + */ +const globalsSection = (globals: ReadonlyArray): Array => { + if (globals.length === 0) return [] + const ordered = [...globals].sort((left, right) => left.localeCompare(right)) + return [ + "", + "## Domain globals", + "", + `These namespaces are also bound as bare globals: ${ordered.map((name) => `\`${name}\``).join(", ")}.`, + "`.(input)` and `tools..(input)` are the same call; prefer the bare form.", + "A global's tools are listed under its namespace in the catalog below, like any other tool.", + ] +} + /** * Budgeted catalog: every namespace is always listed with its tool count; full call * signatures are inlined against the `catalogBudget` (estimated tokens, @@ -492,7 +510,12 @@ export const assertValidTools = (tools: HostTools): void => { * namespace. Namespace stub lines are never budgeted: every namespace appears with its * tool count even at budget 0. */ -export const prepare = (tools: HostTools, catalogBudget = defaultCatalogBudget): DiscoveryPlan => { +export const prepare = ( + tools: HostTools, + catalogBudget = defaultCatalogBudget, + /** Namespaces also bound as bare globals; rendered so the model knows it may write `page.text()`. */ + globals: ReadonlyArray = [], +): DiscoveryPlan => { if (!Number.isSafeInteger(catalogBudget) || catalogBudget < 0) { throw new RangeError("discovery.catalogBudget must be a non-negative safe integer") } @@ -639,7 +662,7 @@ export const prepare = (tools: HostTools, catalogBudget = defaultCatalogBu } } - const lines = [...intro, ...workflow, ...rules, ...language, ...toolSection] + const lines = [...intro, ...workflow, ...rules, ...globalsSection(globals), ...language, ...toolSection] return { catalog: described, instructions: lines.join("\n"), diff --git a/packages/codemode/test/globals.test.ts b/packages/codemode/test/globals.test.ts new file mode 100644 index 0000000000..c27d9bc1a6 --- /dev/null +++ b/packages/codemode/test/globals.test.ts @@ -0,0 +1,83 @@ +import { describe, expect, test } from "bun:test" +import { Effect, Schema } from "effect" +import { CodeMode, Tool } from "../src/index.js" + +const text = Tool.make({ + description: "Visible text of the page", + input: Schema.Struct({}), + output: Schema.String, + run: () => Effect.succeed("hello from the page"), +}) + +const click = Tool.make({ + description: "Click the first node matching a selector", + input: Schema.Struct({ selector: Schema.String }), + output: Schema.Boolean, + run: (input) => Effect.succeed(input.selector === "button"), +}) + +const page = { text, click } + +const run = (code: string, globals: ReadonlyArray = ["page"]) => + Effect.runPromise(CodeMode.make({ tools: { page }, globals }).execute(code)) + +describe("CodeMode host globals", () => { + test("a bare global resolves to its tool namespace", async () => { + const result = await run("return await page.text({})") + expect(result).toMatchObject({ ok: true, value: "hello from the page" }) + }) + + test("the bare form and the tools form are the same call", async () => { + const result = await run( + 'return [await page.click({ selector: "button" }), await tools.page.click({ selector: "a" })]', + ) + expect(result).toMatchObject({ ok: true, value: [true, false] }) + expect(result.toolCalls.map((call) => call.name)).toStrictEqual(["page.click", "page.click"]) + }) + + test("a global is enumerable like any namespace", async () => { + const result = await run("return Object.keys(page).toSorted()") + expect(result).toMatchObject({ ok: true, value: ["click", "text"] }) + }) + + test("a top-level declaration still shadows a global, as in a JS module", async () => { + const result = await run('const page = "shadowed"; return page') + expect(result).toMatchObject({ ok: true, value: "shadowed" }) + }) + + test("an unlisted namespace is not bound as a global", async () => { + const result = await Effect.runPromise(CodeMode.make({ tools: { page } }).execute("return await page.text({})")) + expect(result.ok).toBe(false) + expect(result.ok ? "" : result.error.message).toContain("page") + }) + + test("the instructions name the globals and the equivalence", async () => { + const instructions = CodeMode.make({ tools: { page }, globals: ["page"] }).instructions() + expect(instructions).toContain("## Domain globals") + expect(instructions).toContain("`page`") + expect(instructions).toContain("`tools..(input)` are the same call") + }) + + test("no globals section is rendered when no globals are declared", () => { + expect(CodeMode.make({ tools: { page } }).instructions()).not.toContain("## Domain globals") + }) + + describe("refuses a global a host cannot honestly bind", () => { + const cases: ReadonlyArray, string]> = [ + ["a name that is not in the tool tree", ["screen"], "not a top-level namespace"], + ["a builtin global", ["Math"], "collides with a builtin global"], + ["the tools binding itself", ["tools"], "collides with a builtin global"], + ["a duplicate", ["page", "page"], "listed more than once"], + ["a non-identifier", ["page-object"], "not a plain identifier"], + ] + for (const [name, globals, message] of cases) { + test(name, () => { + expect(() => CodeMode.make({ tools: { page }, globals })).toThrow(message) + }) + } + + test("a tool rather than a namespace", () => { + expect(() => CodeMode.make({ tools: { text }, globals: ["text"] })).toThrow("not a tool itself") + }) + }) +}) diff --git a/packages/redrob/src/tool/code-mode.ts b/packages/redrob/src/tool/code-mode.ts index 98df376495..162a45b9c9 100644 --- a/packages/redrob/src/tool/code-mode.ts +++ b/packages/redrob/src/tool/code-mode.ts @@ -8,6 +8,7 @@ import { Agent } from "@/agent/agent" import { Session } from "@/session/session" import { Permission } from "@/permission" import { Plugin } from "@/plugin" +import { DOMAIN_GLOBALS, domainTools, sessionChannel, unavailableChannel, unavailablePage } from "./domain" export const CODE_MODE_TOOL = "execute" @@ -57,10 +58,16 @@ function groupByServer(mcpTools: Record, servers: readonly export function describeCatalog(mcpTools: Record, servers: readonly string[]): string { return CodeMode.make({ - tools: toolTree( - [...groupByServer(mcpTools, servers).values()].flat(), - () => () => Effect.fail(toolError("Tool preview is not executable.")), - ), + tools: { + ...toolTree( + [...groupByServer(mcpTools, servers).values()].flat(), + () => () => Effect.fail(toolError("Tool preview is not executable.")), + ), + // The preview must list the same domain globals a live execution binds, so the + // description a model reads matches the scope it will actually run in. + ...domainTools({ page: unavailablePage(), channel: unavailableChannel() }), + }, + globals: DOMAIN_GLOBALS, }).instructions() } @@ -213,6 +220,11 @@ export const CodeModeTool = Tool.define( const calls: CallEntry[] = [] const attachments: Attachment[] = [] + // K-1 domain objects for this execution. `channel` posts into the session that owns + // the run; `page` has no working implementation in the engine process yet and + // refuses by name rather than returning a plausible value. + const page = unavailablePage() + const channel = sessionChannel({ sessions, sessionID: ctx.sessionID, messageID: ctx.messageID }) const publish = () => ctx.metadata({ title: CODE_MODE_TOOL, metadata: { toolCalls: calls.map((c) => ({ ...c })) } }) @@ -237,7 +249,11 @@ export const CodeModeTool = Tool.define( ) const runtime = CodeMode.make({ - tools: toolTree(catalog, callTool), + // Domain namespaces are spread last deliberately: a connected MCP server named + // `page` must not shadow the typed domain object a skill document is written + // against. The domain objects are part of the language; the MCP catalog is not. + tools: { ...toolTree(catalog, callTool), ...domainTools({ page, channel }) }, + globals: DOMAIN_GLOBALS, onToolCallStart: ({ index, name, input }) => Effect.suspend(() => { const shown = (() => { diff --git a/packages/redrob/src/tool/domain.ts b/packages/redrob/src/tool/domain.ts new file mode 100644 index 0000000000..165a9aa9a8 --- /dev/null +++ b/packages/redrob/src/tool/domain.ts @@ -0,0 +1,232 @@ +/** + * K-1: the typed domain objects a skill document is written against. + * + * A skill is cheap only because these objects exist. Each one is a real TypeScript + * interface with a single implementation, exported from this module and injected into the + * Code Mode interpreter scope under its own name, so a skill body reads + * + * const body = await page.text() + * await channel.send({ text: body.slice(0, 200) }) + * + * rather than describing a tool call. Code Mode itself stays host-neutral: it is handed a + * namespace of tools plus the list of names to bind as globals, and never learns what + * `page` or `channel` mean (see packages/codemode/AGENTS.md). + * + * `page` has no implementation that can act today: nothing in the engine process controls + * a browser page. Its single implementation is therefore `unavailablePage`, which THROWS a + * message naming what is missing rather than returning a plausible value, so a skill + * written against the interface fails loudly instead of silently reading an empty string. + */ +import { SessionV1 } from "@redrob-code/core/v1/session" +import { Effect, Schema } from "effect" +import { Tool as SandboxTool, toolError } from "@redrob-code/codemode" +import { PartID, type MessageID, type SessionID } from "../session/schema" +import type { Session } from "@/session/session" + +/** One node returned by a page query. Kept to what a skill can act on without a handle. */ +export interface PageNode { + readonly selector: string + readonly text: string + readonly attributes: Readonly> +} + +/** + * A domain object the running engine cannot reach. Carries the missing capability by name + * so the model, the logs, and a person reading a transcript all see the same reason. + */ +export class DomainUnavailableError extends Schema.TaggedErrorClass()( + "CodeModeDomainUnavailable", + { object: Schema.String, missing: Schema.String }, +) { + override get message() { + return `The \`${this.object}\` domain object is not available in this session: ${this.missing}` + } +} + +/** + * The browser page the agent is acting in. + * + * Every method can fail with `DomainUnavailableError`, because whether a page exists is a + * property of the session, not of the call. + */ +export interface Page { + /** The page's current URL. */ + readonly url: () => Effect.Effect + /** The page's visible text. */ + readonly text: () => Effect.Effect + /** Nodes matching a CSS selector, in document order. */ + readonly query: (input: { + readonly selector: string + readonly limit?: number + }) => Effect.Effect, DomainUnavailableError> + /** Clicks the first node matching a selector. */ + readonly click: (input: { readonly selector: string }) => Effect.Effect + /** Types text into the first node matching a selector, optionally submitting afterwards. */ + readonly type: (input: { + readonly selector: string + readonly text: string + readonly submit?: boolean + }) => Effect.Effect + /** Navigates the page and returns the URL actually landed on. */ + readonly navigate: (input: { readonly url: string }) => Effect.Effect +} + +/** The channel this session belongs to: where a message or a result is delivered. */ +export interface Channel { + /** Identifier of the conversation this session posts into. */ + readonly id: () => Effect.Effect + /** Posts a message visible to the person in the conversation. Returns the part id written. */ + readonly send: (input: { readonly text: string }) => Effect.Effect +} + +/** + * The single `Page` implementation available in the engine process: none of it works. + * + * Kept deliberately rather than omitted, so the interface, the tool schemas, the generated + * instructions, and the skills written against them all exist and are exercised before the + * browser side lands. `missing` names the capability, not the symptom. + */ +export const unavailablePage = ( + missing = "the engine process has no browser-page control surface; the browser must expose page control to the engine first", +): Page => { + const refuse = () => + Effect.fail(new DomainUnavailableError({ object: "page", missing })) as Effect.Effect + return { + url: () => refuse(), + text: () => refuse(), + query: () => refuse>(), + click: () => refuse(), + type: () => refuse(), + navigate: () => refuse(), + } +} + +/** + * The single `Channel` implementation: posts into the session the program is running in, + * by appending a text part to the assistant message that owns the execution. Every surface + * the session is attached to renders that part, so this is the channel in the sense the + * skill means it, without the engine having to know which surface is attached. + */ +export const sessionChannel = (input: { + readonly sessions: Session.Interface + readonly sessionID: SessionID + readonly messageID: MessageID +}): Channel => ({ + id: () => Effect.succeed(input.sessionID), + send: ({ text }) => + Effect.gen(function* () { + const part = yield* input.sessions.updatePart({ + id: PartID.ascending(), + messageID: input.messageID, + sessionID: input.sessionID, + type: "text", + text, + } satisfies SessionV1.TextPart) + return part.id + }), +}) + +/** + * A `Channel` that refuses. Used where the catalog is described rather than executed, so + * the preview instructions list the same globals a real execution binds without a preview + * being able to post anything into a conversation. + */ +export const unavailableChannel = (missing = "this is a catalog preview, not a live execution"): Channel => { + const refuse = () => + Effect.fail(new DomainUnavailableError({ object: "channel", missing })) as Effect.Effect + return { id: () => refuse(), send: () => refuse() } +} + +const Empty = Schema.Struct({}) + +/** Narrow JSON projection of a page node, so results cross the sandbox data boundary. */ +const PageNodeSchema = Schema.Struct({ + selector: Schema.String, + text: Schema.String, + attributes: Schema.Record(Schema.String, Schema.String), +}) + +/** + * Lifts a domain call into a Code Mode tool. A `DomainUnavailableError` becomes a + * model-safe tool failure carrying the same sentence, so an unavailable object reads as a + * precise refusal in the program rather than as an opaque execution failure. + */ +const lift = + (call: (input: I) => Effect.Effect) => + (input: I) => + call(input).pipe(Effect.mapError((error) => toolError(error.message, error))) + +/** `page` as a Code Mode namespace. The tool names are the interface's method names. */ +export const pageTools = (page: Page) => ({ + url: SandboxTool.make({ + description: "Current URL of the page the agent is acting in.", + input: Empty, + output: Schema.String, + run: lift(() => page.url()), + }), + text: SandboxTool.make({ + description: "Visible text of the current page.", + input: Empty, + output: Schema.String, + run: lift(() => page.text()), + }), + query: SandboxTool.make({ + description: "Nodes matching a CSS selector, in document order.", + input: Schema.Struct({ + selector: Schema.String.annotate({ description: "CSS selector." }), + limit: Schema.optionalKey(Schema.Number.annotate({ description: "Maximum nodes to return." })), + }), + output: Schema.Array(PageNodeSchema), + run: lift((input: { selector: string; limit?: number }) => page.query(input)), + }), + click: SandboxTool.make({ + description: "Click the first node matching a CSS selector.", + input: Schema.Struct({ selector: Schema.String.annotate({ description: "CSS selector." }) }), + output: Schema.Null, + run: lift((input: { selector: string }) => page.click(input).pipe(Effect.as(null))), + }), + type: SandboxTool.make({ + description: "Type text into the first node matching a CSS selector.", + input: Schema.Struct({ + selector: Schema.String.annotate({ description: "CSS selector." }), + text: Schema.String.annotate({ description: "Text to type." }), + submit: Schema.optionalKey(Schema.Boolean.annotate({ description: "Submit the field afterwards." })), + }), + output: Schema.Null, + run: lift((input: { selector: string; text: string; submit?: boolean }) => page.type(input).pipe(Effect.as(null))), + }), + navigate: SandboxTool.make({ + description: "Navigate the page and return the URL landed on.", + input: Schema.Struct({ url: Schema.String.annotate({ description: "Absolute URL to open." }) }), + output: Schema.String, + run: lift((input: { url: string }) => page.navigate(input)), + }), +}) + +/** `channel` as a Code Mode namespace. */ +export const channelTools = (channel: Channel) => ({ + id: SandboxTool.make({ + description: "Identifier of the conversation this session posts into.", + input: Empty, + output: Schema.String, + run: lift(() => channel.id()), + }), + send: SandboxTool.make({ + description: "Post a message into the conversation this session belongs to.", + input: Schema.Struct({ text: Schema.String.annotate({ description: "Message text." }) }), + output: Schema.String, + run: lift((input: { text: string }) => channel.send(input)), + }), +}) + +/** The names bound as bare globals in the interpreter scope, in the order they are declared. */ +export const DOMAIN_GLOBALS = ["channel", "page"] as const + +/** + * The domain namespaces for one execution, keyed by the global name each is bound to. + * Returned as one object so the tool tree and `DOMAIN_GLOBALS` cannot drift apart. + */ +export const domainTools = (input: { readonly page: Page; readonly channel: Channel }) => ({ + channel: channelTools(input.channel), + page: pageTools(input.page), +}) diff --git a/packages/redrob/test/tool/code-mode.test.ts b/packages/redrob/test/tool/code-mode.test.ts index 16cc08bbec..053cad8cc3 100644 --- a/packages/redrob/test/tool/code-mode.test.ts +++ b/packages/redrob/test/tool/code-mode.test.ts @@ -256,7 +256,7 @@ describe("code mode execute", () => { expect(output.metadata.toolCalls).toEqual([]) }) - test("Object.keys(tools) enumerates the MCP server and CodeMode namespaces", async () => { + test("Object.keys(tools) enumerates the MCP server, domain, and CodeMode namespaces", async () => { const tool = await build({ github_list_issues: mcpTool("list_issues", () => ""), linear_search: mcpTool("search", () => ""), @@ -267,7 +267,12 @@ describe("code mode execute", () => { ctx, ), ) - expect(JSON.parse(output.output)).toEqual({ namespaces: ["github", "linear", "$codemode"], count: 3 }) + // The K-1 domain namespaces (`channel`, `page`) are part of the language, so they are + // always present alongside whatever MCP servers are connected. + expect(JSON.parse(output.output)).toEqual({ + namespaces: ["github", "linear", "channel", "page", "$codemode"], + count: 5, + }) }) test("calls a namespaced MCP tool and flows its text result back into the program", async () => { diff --git a/packages/redrob/test/tool/domain.test.ts b/packages/redrob/test/tool/domain.test.ts new file mode 100644 index 0000000000..9315ac2864 --- /dev/null +++ b/packages/redrob/test/tool/domain.test.ts @@ -0,0 +1,185 @@ +import { describe, expect, test } from "bun:test" +import { Agent } from "@/agent/agent" +import { MCP } from "@/mcp" +import { Plugin } from "@/plugin" +import { Session } from "@/session/session" +import { Tool } from "@/tool/tool" +import * as Truncate from "@/tool/truncate" +import { CODE_MODE_TOOL, CodeModeTool, describeCatalog } from "@/tool/code-mode" +import { + DOMAIN_GLOBALS, + DomainUnavailableError, + channelTools, + domainTools, + pageTools, + sessionChannel, + unavailableChannel, + unavailablePage, +} from "@/tool/domain" +import { MessageID, PartID, SessionID } from "@/session/schema" +import type { SessionV1 } from "@redrob-code/core/v1/session" +import { Cause, Effect, Exit, Layer } from "effect" + +const sessionID = SessionID.make("ses_domain") +const messageID = MessageID.make("msg_domain") + +const ctx: Tool.Context = { + sessionID, + messageID, + agent: "build", + abort: new AbortController().signal, + callID: "call_domain", + messages: [], + metadata: () => Effect.void, + ask: () => Effect.void, +} + +/** Collects the parts a program writes, so `channel.send` is asserted on its effect. */ +function recordingSessions() { + const parts: SessionV1.Part[] = [] + const sessions = { + get: () => Effect.succeed({ permission: [] } as any), + updatePart: (part: SessionV1.Part) => + Effect.sync(() => { + parts.push(part) + return part + }), + } + return { parts, sessions } +} + +function harness(sessions: Record) { + return Layer.mergeAll( + Layer.mock(Plugin.Service, { + trigger: ((_name, _input, output) => Effect.succeed(output)) as Plugin.Interface["trigger"], + }), + Layer.mock(Truncate.Service, { + output: (text: string) => Effect.succeed({ content: text, truncated: false as const }), + }), + Layer.mock(Agent.Service, { get: () => Effect.succeed({ name: "build", permission: [] } as any) }), + Layer.mock(Session.Service, sessions as any), + Layer.mock(MCP.Service, { tools: () => Effect.succeed({}), clients: () => Effect.succeed({}) }), + ) +} + +/** Runs one Code Mode program through the real `execute` tool. */ +function run(code: string, sessions: Record) { + return Effect.runPromise( + CodeModeTool.pipe( + Effect.flatMap(Tool.init), + Effect.flatMap((def) => def.execute({ code }, ctx)), + Effect.provide(harness(sessions)), + ), + ) +} + +/** Program failures die at the tool boundary; recover the defect for message assertions. */ +async function failureOf(code: string, sessions: Record) { + const exit = await Effect.runPromise( + CodeModeTool.pipe( + Effect.flatMap(Tool.init), + Effect.flatMap((def) => def.execute({ code }, ctx)), + Effect.provide(harness(sessions)), + Effect.exit, + ), + ) + if (Exit.isSuccess(exit)) throw new Error("expected the program to fail") + return (Cause.squash(exit.cause) as Error).message +} + +describe("K-1 typed domain objects", () => { + test("every global names a namespace of the domain tool tree", () => { + const tools = domainTools({ page: unavailablePage(), channel: unavailableChannel() }) + expect([...(DOMAIN_GLOBALS as ReadonlyArray)].toSorted()).toStrictEqual(Object.keys(tools).toSorted()) + }) + + test("the tool names are the interface method names", () => { + expect(Object.keys(pageTools(unavailablePage())).toSorted()).toStrictEqual([ + "click", + "navigate", + "query", + "text", + "type", + "url", + ]) + expect(Object.keys(channelTools(unavailableChannel())).toSorted()).toStrictEqual(["id", "send"]) + }) + + test("the globals are advertised in the catalog description", () => { + const instructions = describeCatalog({}, []) + expect(instructions).toContain("## Domain globals") + expect(instructions).toContain("`page`") + expect(instructions).toContain("`channel`") + }) + + describe("channel", () => { + test("send posts a text part into the session that owns the run", async () => { + const { parts, sessions } = recordingSessions() + const result = await run('return await channel.send({ text: "posted by a skill" })', sessions) + expect(result.output).toStartWith("prt_") + expect(parts).toHaveLength(1) + expect(parts[0]).toMatchObject({ + type: "text", + text: "posted by a skill", + sessionID, + messageID, + }) + }) + + test("id returns the session the program is bound to", async () => { + const { sessions } = recordingSessions() + expect((await run("return await channel.id({})", sessions)).output).toBe(sessionID) + }) + + test("the bare global and the tools path are the same call", async () => { + const { parts, sessions } = recordingSessions() + await run('await channel.send({ text: "a" }); return await tools.channel.send({ text: "b" })', sessions) + expect(parts.map((part) => (part.type === "text" ? part.text : undefined))).toStrictEqual(["a", "b"]) + }) + }) + + describe("page is unavailable rather than faked", () => { + test("every method refuses, naming the missing capability", async () => { + const page = unavailablePage() + const calls = [ + page.url(), + page.text(), + page.query({ selector: "h1" }), + page.click({ selector: "h1" }), + page.type({ selector: "input", text: "x" }), + page.navigate({ url: "https://example.invalid" }), + ] + for (const call of calls) { + const exit = await Effect.runPromise(call.pipe(Effect.exit)) + expect(Exit.isFailure(exit)).toBe(true) + const error = Exit.isFailure(exit) ? Cause.squash(exit.cause) : undefined + expect(error).toBeInstanceOf(DomainUnavailableError) + expect((error as DomainUnavailableError).message).toContain("no browser-page control surface") + } + }) + + test("a program calling page.text fails with that reason, not an empty string", async () => { + const { sessions } = recordingSessions() + const message = await failureOf("return await page.text({})", sessions) + expect(message).toContain("`page` domain object is not available") + expect(message).toContain("no browser-page control surface") + }) + + test("the refusal carries the named object", () => { + const error = new DomainUnavailableError({ object: "page", missing: "nothing drives a page here" }) + expect(error.message).toBe( + "The `page` domain object is not available in this session: nothing drives a page here", + ) + }) + }) + + test("the execute tool still reports its own name", async () => { + const { sessions } = recordingSessions() + const result = await run('return await channel.send({ text: "t" })', sessions) + expect(result.title).toBe(CODE_MODE_TOOL) + }) + + test("PartID is the id shape channel.send returns", () => { + expect(PartID.ascending()).toStartWith("prt_") + }) +}) From 0dfe0fbe0bbc7da53885a8624d902dfa71b8436e Mon Sep 17 00:00:00 2001 From: Janghoon Lee <44862514+savagemanage@users.noreply.github.com> Date: Fri, 2 Oct 2026 07:38:57 +0000 Subject: [PATCH 2/3] feat(skill): K-2 auto-arming from frontmatter keywords and URL globs Two optional frontmatter keys beside name/description/slash: icon: string # a URL autoInject: keywords: [string] # matched against the prompt url: [string] # globs, e.g. docs.google.com/document/** Both lists are optional and an empty list matches nothing, so `autoInject: {}` is a skill that still never arms rather than one that arms on everything. The arming decision is a pure function in packages/core/src/skill/arming.ts: SkillArming.arm(input: Input): ReadonlyArray SkillArming.armedNames(input: Input): ReadonlyArray It takes the loaded skills, the current prompt text and the current tab URL, and returns which skills arm, with the patterns that armed each one. It reads nothing and caches nothing, so the decision is identical in the prompt builder, in a UI preview and in a test. Keyword and URL matching are both case-insensitive; keywords match on word boundaries so `ai` does not match inside `said`; `*` and `?` stop at a `/` while `**` crosses segments; a `.` in a glob is a literal dot and not a wildcard. The scheme is ignored on both sides, so globs are written without one. A skill with NO autoInject block never auto-arms. It stays explicitly loadable, which is the point: a skill that arms on everything is always in the prompt. Loading decodes one document at a time, so a malformed frontmatter costs that one document and not the set. The decoders are the Result-returning form rather than the Option form, because the Option form answers only "no" and makes the cause unfindable; every rejection is logged at warning level with the file and the decoder's own reason. A malformed autoInject is narrower still: it is ignored with its own warning and the skill loads without it, since that leaves the skill exactly where a skill with no block already sits. The reasons are in the log MESSAGE rather than in annotations, because the default logger renders an annotation object as [object Object] and loses precisely that detail. Tests in packages/core/test/skill/arming.test.ts: 13 pass, 0 fail. They cover a keyword hit, a URL-glob hit, three glob near-misses that must not arm, a skill with no block, an empty block, case-insensitivity in both directions, and a malformed autoInject that is ignored rather than fatal while a sibling bad document is dropped with a logged reason. Each of five restored defects (explicit-only guard removed, `*` crossing `/`, glob compiled as a regex, malformed autoInject made fatal, reason text dropped from the log) was confirmed to fail the suite. Note: packages/schema has 2 failing tests in test/event-manifest.test.ts. They fail identically on a clean tree (git stash push -u, 13 pass / 2 fail both ways) and are unrelated to this change. --- packages/core/src/skill.ts | 83 +++++++++++- packages/core/src/skill/arming.ts | 131 ++++++++++++++++++ packages/core/test/skill/arming.test.ts | 170 ++++++++++++++++++++++++ packages/schema/src/skill.ts | 18 +++ 4 files changed, 395 insertions(+), 7 deletions(-) create mode 100644 packages/core/src/skill/arming.ts create mode 100644 packages/core/test/skill/arming.test.ts diff --git a/packages/core/src/skill.ts b/packages/core/src/skill.ts index a0ffeae896..8c82d56a46 100644 --- a/packages/core/src/skill.ts +++ b/packages/core/src/skill.ts @@ -2,7 +2,7 @@ export * as SkillV2 from "./skill" import { makeLocationNode } from "./effect/app-node" import path from "path" -import { Context, Effect, Layer, Schema, Types } from "effect" +import { Context, Effect, Layer, Result, Schema, Types } from "effect" import { Skill } from "@redrob-code/schema/skill" import { AgentV2 } from "./agent" import { ConfigMarkdown } from "./config/markdown" @@ -27,6 +27,9 @@ export type Source = typeof Source.Type export const Info = Skill.Info export type Info = Skill.Info +export const AutoInject = Skill.AutoInject +export type AutoInject = Skill.AutoInject + export const available = (skills: ReadonlyArray, agent: AgentV2.Info) => skills.filter((skill) => PermissionV2.evaluate("skill", skill.name, agent.permissions).effect !== "deny") @@ -34,8 +37,25 @@ const Frontmatter = Schema.Struct({ name: Schema.String.pipe(Schema.optional), description: Schema.String.pipe(Schema.optional), slash: Schema.Boolean.pipe(Schema.optional), + icon: Schema.String.pipe(Schema.optional), }) -const decodeFrontmatter = Schema.decodeUnknownOption(Frontmatter) +/** + * `decodeUnknownResult`, not `decodeUnknownOption`: the Option form answers only "no", which + * makes a malformed document unfindable among the ones that loaded. The Result form carries + * the issue, so the log below can name the file AND why it was rejected. + */ +const decodeFrontmatter = Schema.decodeUnknownResult(Frontmatter) + +/** + * K-2's auto-arm block is decoded SEPARATELY from the rest of the frontmatter, so a + * malformed `autoInject` costs only the arming behaviour: the skill still loads and stays + * explicitly loadable, instead of the whole document being dropped for a typo in an + * optional block. + */ +const decodeAutoInject = Schema.decodeUnknownResult(Skill.AutoInject) + +/** The reason one document did not become a skill, kept so the log can say why. */ +type Drop = { readonly file: string; readonly reason: string } export type Data = { sources: Types.DeepMutable[] @@ -74,33 +94,82 @@ const layer = Layer.effect( const skills: Info[] = [] if (source.type === "embedded") return [source.skill] const directories = source.type === "directory" ? [source.path] : yield* discovery.pull(source.url) + // Decoded ONE DOCUMENT AT A TIME: a malformed frontmatter costs that document and not + // the set, and every rejection is recorded with its reason so the cause is findable. + const drops: Drop[] = [] + let ignoredAutoInject = 0 for (const directory of directories) { const files = yield* fs .glob("{*.md,**/SKILL.md}", { cwd: directory, absolute: true, include: "file", symlink: true, dot: true }) .pipe(Effect.catch(() => Effect.succeed([] as string[]))) for (const filepath of files.toSorted()) { const content = yield* fs.readFileStringSafe(filepath).pipe(Effect.catch(() => Effect.succeed(undefined))) - if (!content) continue + if (!content) { + drops.push({ file: filepath, reason: "unreadable: the file could not be read" }) + continue + } const markdown = ConfigMarkdown.parseOption(content) - if (!markdown) continue - const frontmatter = decodeFrontmatter(markdown.data).valueOrUndefined - if (!frontmatter) continue + if (!markdown) { + drops.push({ file: filepath, reason: "unparsable: no valid markdown frontmatter block" }) + continue + } + const decoded = decodeFrontmatter(markdown.data) + if (Result.isFailure(decoded)) { + drops.push({ file: filepath, reason: `frontmatter rejected: ${decoded.failure.message}` }) + continue + } + const frontmatter = decoded.success const name = frontmatter.name !== undefined ? frontmatter.name : path.dirname(filepath) === directory ? path.basename(filepath, ".md") : undefined - if (!name) continue + if (!name) { + drops.push({ + file: filepath, + reason: "unnamed: no `name` in frontmatter and the path gives no fallback name", + }) + continue + } + // A malformed `autoInject` is IGNORED, not fatal: the skill loads and stays + // explicitly loadable, which is what a skill without the block does anyway. + let autoInject: AutoInject | undefined + const declared = (markdown.data as Record | undefined)?.["autoInject"] + if (declared !== undefined) { + const block = decodeAutoInject(declared) + if (Result.isFailure(block)) { + ignoredAutoInject += 1 + // The reason goes in the MESSAGE, not only in an annotation: the default logger + // renders an annotation object as `[object Object]`, which loses exactly the + // detail that makes this findable. + yield* Effect.logWarning( + `SkillV2.load ignored a malformed autoInject block in ${filepath} (skill "${name}"): ${block.failure.message}`, + ) + } else autoInject = block.success + } skills.push({ name, description: frontmatter.description, slash: frontmatter.slash, + icon: frontmatter.icon, + autoInject, location: AbsolutePath.make(filepath), content: markdown.content, }) } } + if (drops.length > 0) { + // Warning, not debug: a dropped skill is silently missing behaviour, so the count and + // every individual reason have to reach a default log level to be findable at all. + // Both are in the message text for the same reason the per-block warning above is. + const total = skills.length + drops.length + yield* Effect.logWarning( + `SkillV2.load dropped ${drops.length} of ${total} skill documents from ${Source.key(source)} ` + + `(loaded ${skills.length}, ignoredAutoInject ${ignoredAutoInject}): ` + + drops.map((drop) => `${drop.file}: ${drop.reason}`).join("; "), + ) + } return skills }) diff --git a/packages/core/src/skill/arming.ts b/packages/core/src/skill/arming.ts new file mode 100644 index 0000000000..ca5e7a59e9 --- /dev/null +++ b/packages/core/src/skill/arming.ts @@ -0,0 +1,131 @@ +export * as SkillArming from "./arming" + +import type { Skill } from "@redrob-code/schema/skill" + +/** + * K-2: the arming decision, as a pure function. + * + * Given the loaded skills, the current prompt text and the current tab URL, `arm` returns + * which skills auto-arm. It reads nothing, caches nothing and logs nothing, so the decision + * is testable on its own and identical wherever it is asked: the session prompt builder, a + * UI preview, or a test. + * + * Two rules, both case-insensitive: + * + * - keyword: a keyword matches when it appears in the prompt on word boundaries, so `ai` + * does not match `said` but `docs.google.com` still matches when followed by a slash. + * - url: a glob matches the current tab URL in full. `*` and `?` stop at `/`; `**` crosses + * path segments. The scheme is ignored on both sides, so a glob is written + * `docs.google.com/document/**` rather than `https://docs.google.com/document/**`. + * + * A skill with no `autoInject` block NEVER auto-arms. It stays explicitly loadable, which + * is the point: a skill that arms on everything is a skill that is always in the prompt. + */ + +/** The shape `arm` needs from a skill. Narrower than `Skill.Info` so callers can test it directly. */ +export type Armable = { + readonly name: string + readonly autoInject?: Skill.AutoInject | undefined +} + +/** One skill that armed, with the patterns that armed it. */ +export type Armed = { + readonly name: string + /** Keywords from the skill's block that matched the prompt. */ + readonly keywords: ReadonlyArray + /** URL globs from the skill's block that matched the current tab URL. */ + readonly urls: ReadonlyArray +} + +export type Input = { + readonly skills: ReadonlyArray + /** The current prompt text. An empty prompt matches no keyword. */ + readonly prompt: string + /** The current tab URL, when the session is attached to one. Absent matches no glob. */ + readonly url?: string | undefined +} + +const isWordCharacter = (character: string | undefined): boolean => + character !== undefined && /[\p{L}\p{N}_]/u.test(character) + +/** + * Case-insensitive word-boundary containment. A boundary is required only where the + * keyword's own edge is a word character, so a keyword like `.pdf` or `docs.google.com` + * matches in running text while `ai` does not match inside `said`. + */ +const matchesKeyword = (prompt: string, keyword: string): boolean => { + const needle = keyword.trim().toLowerCase() + if (needle.length === 0) return false + const haystack = prompt.toLowerCase() + const checkStart = isWordCharacter(needle[0]) + const checkEnd = isWordCharacter(needle[needle.length - 1]) + let from = 0 + for (;;) { + const at = haystack.indexOf(needle, from) + if (at === -1) return false + const before = at === 0 ? undefined : haystack[at - 1] + const after = haystack[at + needle.length] + if ((!checkStart || !isWordCharacter(before)) && (!checkEnd || !isWordCharacter(after))) return true + from = at + 1 + } +} + +/** Drops a `scheme://` prefix and a trailing slash, so globs are written without a scheme. */ +const normalizeUrl = (value: string): string => + value + .trim() + .toLowerCase() + .replace(/^[a-z][a-z0-9+.-]*:\/\//, "") + .replace(/\/+$/, "") + +/** + * Compiles a glob to an anchored regular expression. `**` crosses `/`; `*` and `?` do not. + * Every other character is matched literally, so a `.` in a hostname is a dot and not a + * wildcard - the common authoring mistake if globs were treated as regular expressions. + */ +const globToRegExp = (glob: string): RegExp => { + let pattern = "" + for (let index = 0; index < glob.length; index += 1) { + const character = glob[index]! + if (character === "*") { + if (glob[index + 1] === "*") { + pattern += ".*" + index += 1 + continue + } + pattern += "[^/]*" + continue + } + if (character === "?") { + pattern += "[^/]" + continue + } + pattern += character.replace(/[.*+?^${}()|[\]\\]/g, "\\$&") + } + return new RegExp(`^${pattern}$`) +} + +const matchesUrlGlob = (url: string, glob: string): boolean => { + const normalized = normalizeUrl(glob) + if (normalized.length === 0) return false + return globToRegExp(normalized).test(normalizeUrl(url)) +} + +/** Returns the skills that auto-arm, in the order they were given. */ +export const arm = (input: Input): ReadonlyArray => { + const armed: Array = [] + for (const skill of input.skills) { + const block = skill.autoInject + // No block means never auto-arm. This is the explicit-only case, not a default-on one. + if (block === undefined) continue + const keywords = (block.keywords ?? []).filter((keyword) => matchesKeyword(input.prompt, keyword)) + const urls = + input.url === undefined ? [] : (block.url ?? []).filter((glob) => matchesUrlGlob(input.url as string, glob)) + if (keywords.length === 0 && urls.length === 0) continue + armed.push({ name: skill.name, keywords, urls }) + } + return armed +} + +/** Just the names that armed, for callers that only need the set. */ +export const armedNames = (input: Input): ReadonlyArray => arm(input).map((entry) => entry.name) diff --git a/packages/core/test/skill/arming.test.ts b/packages/core/test/skill/arming.test.ts new file mode 100644 index 0000000000..00dc901fd1 --- /dev/null +++ b/packages/core/test/skill/arming.test.ts @@ -0,0 +1,170 @@ +import fs from "fs/promises" +import path from "path" +import { describe, expect, test } from "bun:test" +import { Effect, Layer } from "effect" +import * as TestConsole from "effect/testing/TestConsole" +import { AgentV2 } from "@redrob-code/core/agent" +import { AppNodeBuilder } from "@redrob-code/core/effect/app-node-builder" +import { LayerNode } from "@redrob-code/core/effect/layer-node" +import { AbsolutePath } from "@redrob-code/core/schema" +import { SkillV2 } from "@redrob-code/core/skill" +import { SkillArming } from "@redrob-code/core/skill/arming" +import { SkillDiscovery } from "@redrob-code/core/skill/discovery" +import { tmpdir } from "../fixture/tmpdir" +import { testEffect } from "../lib/effect" + +/** `arm` is pure, so most of K-2 needs no layer, no filesystem and no clock. */ +const skill = (name: string, autoInject?: SkillV2.AutoInject): SkillArming.Armable => ({ name, autoInject }) + +const names = (input: SkillArming.Input) => SkillArming.armedNames(input) + +describe("SkillArming.arm", () => { + const docs = skill("google-docs", { url: ["docs.google.com/document/**"] }) + const deploy = skill("deploy", { keywords: ["deploy", "rollback"] }) + /** No `autoInject` block at all: explicitly loadable, never auto-armed. */ + const manual = skill("manual") + + test("arms on a keyword hit", () => { + expect(names({ skills: [deploy, manual], prompt: "can you deploy this branch" })).toEqual(["deploy"]) + }) + + test("reports which keyword armed the skill", () => { + expect(SkillArming.arm({ skills: [deploy], prompt: "time to ROLLBACK" })).toEqual([ + { name: "deploy", keywords: ["rollback"], urls: [] }, + ]) + }) + + test("keyword matching is case-insensitive in both directions", () => { + expect(names({ skills: [skill("x", { keywords: ["RollBack"] })], prompt: "rollback now" })).toEqual(["x"]) + }) + + test("a keyword does not match inside a longer word", () => { + expect(names({ skills: [skill("ai", { keywords: ["ai"] })], prompt: "she said nothing" })).toEqual([]) + }) + + test("arms on a URL-glob hit", () => { + expect( + names({ + skills: [docs, manual], + prompt: "summarise this", + url: "https://docs.google.com/document/d/abc123/edit", + }), + ).toEqual(["google-docs"]) + }) + + test("URL matching ignores the scheme and is case-insensitive", () => { + expect(names({ skills: [docs], prompt: "", url: "HTTPS://Docs.Google.com/Document/d/ABC" })).toEqual([ + "google-docs", + ]) + }) + + test("a glob NEAR-MISS does not arm", () => { + // `documents` is not `document`, and a single `*` does not cross a `/`. + expect(names({ skills: [docs], prompt: "", url: "https://docs.google.com/documents/d/abc" })).toEqual([]) + expect( + names({ + skills: [skill("one-segment", { url: ["docs.google.com/document/*"] })], + prompt: "", + url: "https://docs.google.com/document/d/abc", + }), + ).toEqual([]) + // A `.` in the glob is a literal dot, not a regex wildcard. + expect(names({ skills: [docs], prompt: "", url: "https://docsXgoogle.com/document/d/abc" })).toEqual([]) + }) + + test("a skill with no autoInject block NEVER arms", () => { + expect( + names({ skills: [manual], prompt: "manual deploy rollback everything", url: "docs.google.com/document/d/abc" }), + ).toEqual([]) + }) + + test("an empty autoInject block matches nothing", () => { + expect(names({ skills: [skill("empty", {})], prompt: "deploy", url: "docs.google.com/document/d/a" })).toEqual([]) + }) + + test("no current URL means no glob can arm", () => { + expect(names({ skills: [docs], prompt: "summarise this" })).toEqual([]) + }) + + test("keyword and URL hits are both reported on one skill", () => { + expect( + SkillArming.arm({ + skills: [skill("both", { keywords: ["summarise"], url: ["docs.google.com/**"] })], + prompt: "Summarise it", + url: "https://docs.google.com/document/d/a", + }), + ).toEqual([{ name: "both", keywords: ["summarise"], urls: ["docs.google.com/**"] }]) + }) + + test("skills are returned in the order given", () => { + expect(names({ skills: [deploy, manual, skill("also", { keywords: ["deploy"] })], prompt: "deploy" })).toEqual([ + "deploy", + "also", + ]) + }) +}) + +const urls = new Map() +const discovery = Layer.succeed( + SkillDiscovery.Service, + SkillDiscovery.Service.of({ pull: (url) => Effect.succeed(urls.get(url) ?? []) }), +) +const it = testEffect( + AppNodeBuilder.build(LayerNode.group([SkillV2.node, AgentV2.node]), [[SkillDiscovery.node, discovery]]), +) + +describe("SkillV2.load frontmatter", () => { + it.live("ignores a malformed autoInject instead of failing the load, and drops only the bad document", () => + Effect.acquireRelease( + Effect.promise(() => tmpdir()), + (tmp) => Effect.promise(() => tmp[Symbol.asyncDispose]()), + ).pipe( + Effect.flatMap((tmp) => + Effect.gen(function* () { + yield* Effect.promise(async () => { + // `autoInject` is a string where the schema wants an object: ignored, not fatal. + await fs.writeFile( + path.join(tmp.path, "broken-block.md"), + "---\nname: broken-block\nautoInject: nonsense\n---\n# broken-block", + ) + // A good neighbour, to prove one bad document does not cost the set. + await fs.writeFile( + path.join(tmp.path, "good.md"), + "---\nname: good\nicon: https://example.test/i.png\nautoInject:\n keywords: [deploy]\n url: [docs.google.com/document/**]\n---\n# good", + ) + // `slash` must be a boolean: this document IS dropped, with a logged reason. + await fs.writeFile(path.join(tmp.path, "bad-type.md"), "---\nname: bad-type\nslash: yes please\n---\n# bad") + }) + + const skills = yield* SkillV2.Service + yield* skills.transform((editor) => editor.source({ type: "directory", path: AbsolutePath.make(tmp.path) })) + const loaded = yield* skills.list() + + // The undecodable document is gone; the two others survived. + expect(loaded.map((item) => item.name).toSorted()).toEqual(["broken-block", "good"]) + + const broken = loaded.find((item) => item.name === "broken-block")! + expect(broken.autoInject).toBeUndefined() + // Still loadable, just never auto-armed. + expect(names({ skills: [broken], prompt: "nonsense" })).toEqual([]) + + const good = loaded.find((item) => item.name === "good")! + expect(good.icon).toBe("https://example.test/i.png") + expect(good.autoInject).toEqual({ keywords: ["deploy"], url: ["docs.google.com/document/**"] }) + expect(names({ skills: [good], prompt: "please deploy" })).toEqual(["good"]) + expect(names({ skills: [good], prompt: "", url: "https://docs.google.com/document/d/x" })).toEqual(["good"]) + + // The drop must be FINDABLE: the count, the offending file and the reason all logged. + const logged = (yield* TestConsole.logLines).map((line) => String(line)).join("\n") + expect(logged).toContain("SkillV2.load dropped 1 of 3 skill documents") + expect(logged).toContain("bad-type.md") + expect(logged).toContain("frontmatter rejected") + expect(logged).toContain("ignored a malformed autoInject block") + expect(logged).toContain("broken-block.md") + // The documents that loaded are not reported as dropped. + expect(logged).not.toContain("good.md:") + }), + ), + ), + ) +}) diff --git a/packages/schema/src/skill.ts b/packages/schema/src/skill.ts index ec299180ed..4dfe2afdba 100644 --- a/packages/schema/src/skill.ts +++ b/packages/schema/src/skill.ts @@ -16,11 +16,29 @@ export const UrlSource = Schema.Struct({ url: Schema.String, }).annotate({ identifier: "SkillV2.UrlSource" }) +/** + * K-2 auto-arming block. A skill carrying one arms itself when the current prompt or the + * current tab URL matches; a skill WITHOUT one never auto-arms and stays explicitly + * loadable. Both lists are optional and an empty list matches nothing, so + * `autoInject: {}` is a skill that still never arms rather than one that arms on + * everything - a skill that arms on everything is a skill that is always in the prompt. + */ +export interface AutoInject extends Schema.Schema.Type {} +export const AutoInject = Schema.Struct({ + /** Case-insensitive keywords matched against the prompt text on word boundaries. */ + keywords: Schema.Array(Schema.String).pipe(optional), + /** Globs matched against the current tab URL, e.g. `docs.google.com/document/**`. */ + url: Schema.Array(Schema.String).pipe(optional), +}).annotate({ identifier: "SkillV2.AutoInject" }) + export interface Info extends Schema.Schema.Type {} export const Info = Schema.Struct({ name: Schema.String, description: Schema.String.pipe(optional), slash: Schema.Boolean.pipe(optional), + /** Icon URL declared in frontmatter, for surfaces that list skills. */ + icon: Schema.String.pipe(optional), + autoInject: AutoInject.pipe(optional), location: AbsolutePath, content: Schema.String, }).annotate({ identifier: "SkillV2.Info" }) From 9278641eb783540cca80aa6a3a4aab5cd6d68abb Mon Sep 17 00:00:00 2001 From: Janghoon Lee <44862514+savagemanage@users.noreply.github.com> Date: Fri, 2 Oct 2026 07:50:32 +0000 Subject: [PATCH 3/3] feat(client): K-2 regenerate the client types for the new skill keys `packages/client` commits its generated SDK and gates it with `check:generated` (bun run generate && git diff --exit-code), so adding `icon` and `autoInject` to `SkillV2.Info` in packages/schema left the committed artifact behind and that gate red on CI - the core (linux) job passed its tests and then failed on this step, and unit (linux) is only the aggregation job reporting it. Regenerated with the repo's own generator (`bun run generate` in packages/client), not hand-edited, because the gate rejects a hand edit by construction. The whole diff is the two new optional fields on SkillsListOutput, which is exactly the propagation of the schema change: readonly icon?: string readonly autoInject?: { readonly keywords?: ReadonlyArray readonly url?: ReadonlyArray } Measured after the change: packages/client typecheck passes; its suite is 15 pass / 1 fail, and that one failure ("exposes every standard HTTP API group") also fails on a clean origin/develop worktree, so it is not from this change. --- packages/client/src/generated/types.ts | 2 ++ 1 file changed, 2 insertions(+) diff --git a/packages/client/src/generated/types.ts b/packages/client/src/generated/types.ts index 5db37e3579..7cc6071d18 100644 --- a/packages/client/src/generated/types.ts +++ b/packages/client/src/generated/types.ts @@ -2609,6 +2609,8 @@ export type SkillsListOutput = { readonly name: string readonly description?: string readonly slash?: boolean + readonly icon?: string + readonly autoInject?: { readonly keywords?: ReadonlyArray; readonly url?: ReadonlyArray } readonly location: string readonly content: string }>