Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 13 additions & 4 deletions .github/workflows/sandbox-isolation.yml
Original file line number Diff line number Diff line change
Expand Up @@ -234,8 +234,8 @@ jobs:
executed=$(grep -oE 'executed="[0-9]+"' "$trx" | head -1 | grep -oE '[0-9]+')
passed=$(grep -oE 'passed="[0-9]+"' "$trx" | head -1 | grep -oE '[0-9]+')
echo "executed=${executed:-0} passed=${passed:-0}"
if [ "${executed:-0}" -lt 92 ]; then
echo "::error::Expected >=92 sandbox isolation tests to run (bwrap/prlimit confinement + cap-drop + cgroup-namespace re-root + egress-allowlist filter + cgroup resource cap + durable-launch cgroup wiring + argv/envp per-string kernel ceiling + a prompt past it riding stdin + a read-only workspace mount + a network-off run reaching its broker through the relay and nothing else + an allowlist run relayed to its broker + a read-only reviewer reading its diff with the real CLIs + a bwrap probe that runs the launch argv + the MCP helper bound file by file behind a read-only socket dir + a CLI reaching its broker socket through the relay + a severed child reaching its broker over the lease socket across a worker restart + a pre-relay namespaced run re-bound at its gateway and torn down with its seal + an allowlist run's veth guarded both ways, a flow the worker opened before the run included + forwarding a root worker may not write named before an allowlist is planned + an allowlist run's port 53 open only at the resolvers of the resolv.conf its namespace reads, and still to a resolver address the worker's own NAT rewrites before its forward and its input hooks + a target repository's own CLI config kept out of the run with the real CLIs, every repository's memory read in a multi-repo workspace included + a multi-repo Codex run starting at a workspace root that is no git repository and loading no config from it + a repo-less Codex run starting in a scratch directory that is no git repository), but only ${executed:-0} did — the Category=Sandbox filter matched too few (trait regression?). If a case was deliberately removed, lower this number in the same PR."
if [ "${executed:-0}" -lt 96 ]; then
echo "::error::Expected >=96 sandbox isolation tests to run (bwrap/prlimit confinement + cap-drop + cgroup-namespace re-root + egress-allowlist filter + cgroup resource cap + durable-launch cgroup wiring + argv/envp per-string kernel ceiling + a prompt past it riding stdin + a read-only workspace mount + a network-off run reaching its broker through the relay and nothing else + an allowlist run relayed to its broker + a read-only reviewer reading its diff with the real CLIs + a bwrap probe that runs the launch argv + the MCP helper bound file by file behind a read-only socket dir + a CLI reaching its broker socket through the relay + a severed child reaching its broker over the lease socket across a worker restart + a pre-relay namespaced run re-bound at its gateway and torn down with its seal + an allowlist run's veth guarded both ways, a flow the worker opened before the run included + forwarding a root worker may not write named before an allowlist is planned + an allowlist run's port 53 open only at the resolvers of the resolv.conf its namespace reads, and still to a resolver address the worker's own NAT rewrites before its forward and its input hooks + a target repository's own CLI config kept out of the run with the real CLIs, every repository's memory read in a multi-repo workspace included + a multi-repo Codex run starting at a workspace root that is no git repository and loading no config from it + a repo-less Codex run starting in a scratch directory that is no git repository + a goal handed to the real Claude CLI as text it cannot act on, fresh and resumed, against the text channel as its control), but only ${executed:-0} did — the Category=Sandbox filter matched too few (trait regression?). If a case was deliberately removed, lower this number in the same PR."
exit 1
fi

Expand Down Expand Up @@ -275,6 +275,15 @@ jobs:
assert len(cases) == 1 and cases[0].get('outcome') == 'Passed', f'{method}: must pass'
print('All 5 repository-config E2E arms ran and passed.')

# The goal-channel E2E is armed by the same CLI pins and returns early the same way; require each arm's marker,
# confined, so the positive control it checks against is this lane's posture and not an unconfined one.
for arm in ('mentions', 'resume', 'slash /fix', 'slash /security-review'):
assert f'[goal-channel-e2e] ran {arm} claude-code Confined uid=0 confined=True' in text, f'goal-channel E2E arm "{arm}" did not run confined — check CODESPACE_REQUIRE_REVIEW_CLIS and the CLI install step'
for method, rows in (('A_claude_goal_naming_secrets_reaches_the_model_verbatim_and_reads_none_of_them', 1), ('A_continued_claude_session_takes_its_prompt_the_same_way', 1), ('A_claude_goal_that_opens_with_a_slash_word_reaches_the_model_as_text', 2)):
cases = [r for r in results if 'GoalChannelE2ETests.' + method in r.get('testName', '')]
assert len(cases) == rows and all(r.get('outcome') == 'Passed' for r in cases), f'{method}: all {rows} case(s) must pass'
print('All 4 goal-channel E2E arms ran and passed.')

# The sealed-egress E2E returns early on a host that cannot confine, which reads as Passed; require each arm's marker.
for arm in ('durable', 'non-durable', 'relay-ipv6', 'restart', 'relay-refused', 'relay-policy-route'):
assert f'[sealed-egress-e2e] ran {arm}' in text, f'sealed-egress E2E arm "{arm}" did not run — this lane is root with bwrap and the relay helper, so it must relay'
Expand Down Expand Up @@ -387,9 +396,9 @@ jobs:
root = ET.parse(path).getroot()
counters = root.find('.//{*}Counters')
executed, passed = int(counters.get('executed')), int(counters.get('passed'))
assert executed >= 11 and passed == executed, f'expected all 11 non-root arms to run and pass, got executed={executed} passed={passed}'
assert executed >= 15 and passed == executed, f'expected all 15 non-root arms to run and pass, got executed={executed} passed={passed}'
text = open(path, encoding='utf-8').read()
for marker in ('[non-root-e2e] ran admission uid=1654', '[sealed-egress-e2e] ran non-root durable uid=1654', '[sealed-egress-e2e] ran non-root non-durable uid=1654', '[sealed-egress-e2e] ran non-root restart uid=1654', '[sealed-egress-e2e] ran non-root relay-refused uid=1654', '[sealed-egress-e2e] ran non-root relay-ipv6', '[broker-socket-e2e] ran non-root socket-channel uid=1654', '[review-diff-e2e] ran non-root network-off-relayed claude-code uid=1654', '[review-diff-e2e] ran non-root network-off-relayed codex-cli uid=1654', '[durable-egress-e2e] ran non-root allowlist-severed uid=1654', '[repo-config-e2e] ran non-root claude-code single-repo Standard uid=1654'):
for marker in ('[non-root-e2e] ran admission uid=1654', '[sealed-egress-e2e] ran non-root durable uid=1654', '[sealed-egress-e2e] ran non-root non-durable uid=1654', '[sealed-egress-e2e] ran non-root restart uid=1654', '[sealed-egress-e2e] ran non-root relay-refused uid=1654', '[sealed-egress-e2e] ran non-root relay-ipv6', '[broker-socket-e2e] ran non-root socket-channel uid=1654', '[review-diff-e2e] ran non-root network-off-relayed claude-code uid=1654', '[review-diff-e2e] ran non-root network-off-relayed codex-cli uid=1654', '[durable-egress-e2e] ran non-root allowlist-severed uid=1654', '[repo-config-e2e] ran non-root claude-code single-repo Standard uid=1654', '[goal-channel-e2e] ran non-root mentions claude-code Standard uid=1654 confined=True', '[goal-channel-e2e] ran non-root resume claude-code Standard uid=1654 confined=True', '[goal-channel-e2e] ran non-root slash /fix claude-code Standard uid=1654 confined=True', '[goal-channel-e2e] ran non-root slash /security-review claude-code Standard uid=1654 confined=True'):
assert marker in text, f'non-root arm marker "{marker}" is missing — the arm returned early or did not run as the worker uid'
print(f'All {executed} non-root arms ran as uid 1654 and passed.')
PY
Expand Down
Original file line number Diff line number Diff line change
@@ -1,4 +1,6 @@
using System.Text.Encodings.Web;
using System.Text.Json;
using System.Text.Json.Nodes;
using CodeSpace.Core.DependencyInjection;
using CodeSpace.Core.Services.Agents.Mcp;
using CodeSpace.Core.Services.Agents.Skills;
Expand Down Expand Up @@ -117,6 +119,26 @@ public sealed class ClaudeCodeHarness : IAgentHarness, IAgentHarnessBinary, IAge

private const string DefaultCommand = "claude";

/// <summary>
/// The text block that follows the goal in the one user message a run's stdin carries (<see cref="PromptMessage"/>).
/// The pinned CLI reads @-mentions and a leading /command out of a message's LAST text block only, so this constant
/// is the block it parses and the goal never is. It must give the CLI nothing to act on: no '@', no leading '/',
/// '!' or '#', no keyword it reacts to. Pinned by a unit test and by GoalChannelE2ETests against the real binary.
/// </summary>
internal const string GoalTrailer = "Begin with the task above.";

/// <summary>
/// How the message is serialized. Relaxed, because the launch pipe measures and carries stdin through its own JSON
/// encoder, which already escapes every non-ASCII character (and HTML-sensitive ASCII such as '+', '&lt;', '&gt;'): the
/// default encoder here would escape them first and the pipe would then escape the escapes. So a goal of characters
/// this encoder writes as they are (ASCII and most of the BMP) costs the pipe what the bare goal would, plus a constant
/// envelope. What it does escape — a quote, a backslash, every control character (a newline included, which keeps
/// the message exactly one line), a few format and separator characters such as U+FEFF and U+2028, and any character
/// outside the BMP, which every System.Text.Json encoder writes as an escaped surrogate pair — has its backslash
/// escaped again on the pipe, so it costs up to twice the goal's own pipe size. Nothing here is ever rendered as HTML.
/// </summary>
private static readonly JsonSerializerOptions PromptMessageJson = new() { Encoder = JavaScriptEncoder.UnsafeRelaxedJsonEscaping };

public string Kind => HarnessKind;

public string Version => System.Environment.GetEnvironmentVariable(VersionEnvVar) is { Length: > 0 } v ? v : DefaultVersion;
Expand Down Expand Up @@ -173,12 +195,14 @@ public sealed class ClaudeCodeHarness : IAgentHarness, IAgentHarnessBinary, IAge

public SandboxSpec BuildInvocation(AgentTask task)
{
// --output-format stream-json REQUIRES --verbose in --print mode (the CLI rejects it otherwise).
var args = new List<string> { "--print", "--output-format", "stream-json", "--verbose" };
// --output-format stream-json REQUIRES --verbose in --print mode (the CLI rejects it otherwise). The prompt is
// read as stream-json too (see PromptMessage): on a text stdin the CLI acts on the goal's own text before the
// model sees it.
var args = new List<string> { "--print", "--output-format", "stream-json", "--verbose", "--input-format", "stream-json" };

// P3.2: a CONTINUE re-stage threads the prior session id as `--resume <id>` to pick up the conversation.
// Placed right after the seed — before the variadic --allowed-tools / --permission-mode — so the variadic can
// never swallow it. The continuation prompt rides stdin like any other. Null (a fresh run) → omitted.
// never swallow it. The continuation prompt rides stdin like any other, in the same message. Null (a fresh run) → omitted.
if (task.ResumeFromSessionId is { Length: > 0 } resumeSessionId)
{
args.Add("--resume");
Expand Down Expand Up @@ -227,7 +251,7 @@ public SandboxSpec BuildInvocation(AgentTask task)
{
Command = ResolveCommand(),
Args = args,
StandardInput = task.Goal,
StandardInput = PromptMessage(task.Goal),
WorkingDirectory = task.WorkspaceDirectory,
Environment = BuildEnvironment(task),
TimeoutSeconds = task.TimeoutSeconds,
Expand All @@ -245,6 +269,60 @@ public SandboxSpec BuildInvocation(AgentTask task)
};
}

/// <summary>
/// The goal as the ONE stream-json user message the CLI reads from stdin under <c>--input-format stream-json</c>: the
/// goal as its first text block, <see cref="GoalTrailer"/> as its last, on one line. Every Claude prompt is built
/// here — a fresh run, a CONTINUE on <c>--resume</c>, a revise round, a reviewer — because every one is a
/// <see cref="BuildInvocation"/>.
///
/// <para>Why not the goal as text: on a text stdin the pinned 2.1.263 CLI, before the model acts and in plan mode
/// too, reads every <c>@path</c> the goal names after start of text, whitespace (JavaScript's <c>\s</c>, so a BOM,
/// NBSP, U+3000 and U+2028 count) or 。、?! into the request and the session transcript — an absolute path, a
/// <c>~</c> path (HOME is the run's config home under bubblewrap, beside its MCP declaration and transcripts), a
/// directory listing, a symlink out of the workspace. A goal that starts with <c>/word</c> runs as a command:
/// <c>/security-review</c> runs git, <c>/config</c> rewrites settings, and an unknown word ends the run as a success
/// with no turn. Text reaches a goal from pull requests, repositories and other models, so none of it may act before
/// the model reads it.</para>
///
/// <para>The CLI parses mentions and commands out of a message's LAST text block only, so the goal sits first and
/// reaches the model byte for byte — never escaped or rewritten, the same text Codex gets. Escaping the sigils instead
/// would change what the model reads and would have to copy the CLI's JavaScript <c>\s</c> exactly: a .NET <c>\s</c>
/// misses the BOM, and the CLI then reads the file. The block order is undocumented CLI behaviour, which is why
/// GoalChannelE2ETests pins it against the real binary, with the text channel as its positive control.</para>
///
/// <para>A blank goal is refused: the CLI drops a block its JavaScript <c>trim()</c> empties and runs the model on the
/// trailer alone, reporting success for a run that had no task. On a text stdin it refused a blank prompt itself.</para>
/// </summary>
internal static string PromptMessage(string goal)
{
EnsureGoal(goal);

var message = new JsonObject
{
["type"] = "user",
["message"] = new JsonObject { ["role"] = "user", ["content"] = new JsonArray(TextBlock(goal), TextBlock(GoalTrailer)) },
["parent_tool_use_id"] = null,
["session_id"] = "",
};

return message.ToJsonString(PromptMessageJson) + "\n";
}

private static JsonObject TextBlock(string text) => new() { ["type"] = "text", ["text"] = text };

private static void EnsureGoal(string goal)
{
if (IsBlankToTheCli(goal))
throw new ArgumentException("A Claude Code run cannot take a blank goal: the CLI drops a blank prompt block and would run the model with no task.", nameof(goal));
}

/// <summary>
/// Whether the CLI would drop <paramref name="goal"/> as a blank block. It tests a block with JavaScript's
/// <c>trim()</c>, whose whitespace is .NET's plus U+FEFF: <c>string.IsNullOrWhiteSpace</c> alone passes a BOM-only
/// goal the CLI then drops. (.NET also counts U+0085, which only refuses a goal the CLI would have kept.)
/// </summary>
private static bool IsBlankToTheCli(string goal) => goal.All(c => char.IsWhiteSpace(c) || c == '\uFEFF');

/// <summary>
/// The config-home files the runner materializes: the persona's projected skills, PLUS — on a CONTINUE — the prior
/// session's restored transcript at <c>projects/&lt;sanitized-cwd&gt;/&lt;sessionId&gt;.jsonl</c> where
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -211,7 +211,7 @@ public async Task A_continuation_whose_transcript_overflows_the_launch_frame_run
warm.Args.ShouldContain("--resume", customMessage: "fixture check: and would have resumed it");
launched.ConfigHomeFiles.ShouldNotContain(file => file.Content.Length == transcript.Length, "the launched spec restores nothing");
launched.Args.ShouldNotContain("--resume");
launched.StandardInput.ShouldNotBeNull().ShouldEndWith(AgentRetryContinuity.OversizedTranscriptHint, customMessage: "the goal said the conversation was restored; it must be told that it is not");
FakeAgentCliDialect.ClaudeGoal(launched.StandardInput).ShouldEndWith(AgentRetryContinuity.OversizedTranscriptHint, customMessage: "the goal said the conversation was restored; it must be told that it is not");

var persisted = JsonSerializer.Deserialize<AgentTask>(run.TaskJson, AgentJson.Options).ShouldNotBeNull();
persisted.ResumeFromSessionId.ShouldBeNull("the Room's 'resumed' mark reads the persisted envelope, and this attempt resumed nothing");
Expand Down Expand Up @@ -260,7 +260,7 @@ public async Task A_locally_graded_continuation_that_overflows_the_frame_runs_co

run.Status.ShouldBe(AgentRunStatus.Succeeded, $"a cold continuation must launch and be graded on its contract — it ended {result.ExitReason}: {result.AcceptanceDetail ?? run.Error}");
result.AcceptancePassed.ShouldBe(true);
harness.Specs[^1].StandardInput.ShouldNotBeNull().ShouldEndWith(AgentRetryContinuity.OversizedTranscriptHint, customMessage: "the agent is still told");
FakeAgentCliDialect.ClaudeGoal(harness.Specs[^1].StandardInput).ShouldEndWith(AgentRetryContinuity.OversizedTranscriptHint, customMessage: "the agent is still told");
}
finally
{
Expand Down
Loading
Loading