diff --git a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs
index 0f58bbb43..83394808e 100644
--- a/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs
+++ b/backend/src/CodeSpace.Core/Services/Agents/AgentRunExecutor.cs
@@ -3672,7 +3672,7 @@ private static (string SocketPath, string Token) MintMcpConnect(Guid runId) =>
try
{
- return new AgentMcpEndpoint(runId, registry, autonomy, teamId, redactor, socketPath, token, connects, scope, ct, _logger, fenceEpoch, governanceEnabled, approvalConversationId, catalogMode);
+ return new AgentMcpEndpoint(runId, registry, autonomy, teamId, redactor, socketPath, token, connects, scope, ct, _logger, fenceEpoch, governanceEnabled, approvalConversationId, catalogMode, task.Permissions);
}
// An over-length socket path throws ArgumentOutOfRangeException (UDS endpoint ctor); CreateDirectory can throw
// IOException / UnauthorizedAccessException. The endpoint is optional infra, not the run, so any of these is a
diff --git a/backend/src/CodeSpace.Core/Services/Agents/Commands/CallerCommandLanes.cs b/backend/src/CodeSpace.Core/Services/Agents/Commands/CallerCommandLanes.cs
new file mode 100644
index 000000000..ea7edd2d3
--- /dev/null
+++ b/backend/src/CodeSpace.Core/Services/Agents/Commands/CallerCommandLanes.cs
@@ -0,0 +1,86 @@
+using CodeSpace.Core.DependencyInjection;
+
+namespace CodeSpace.Core.Services.Agents.Commands;
+
+///
+/// One agent.run_command at a time per calling agent run. Each command gets a cgroup leaf of its own, beside the
+/// agent's leaf rather than inside it, carrying the run's whole tier row (RunCommandService.BuildSpec). The run
+/// token in the agent's config lets it open as many endpoint connections as it likes, so commands it started at once
+/// would each hold a full row. Queued here, the commands one run has running never hold more than one row between them.
+/// The agent's own leaf is separate, so an agent and its one running command can together hold up to two rows.
+///
+/// Process-local by design: a run's MCP endpoint, and so every command its agent asks for, lives in the worker
+/// that launched it. A lane exists only while a command of its run holds or awaits it.
+///
+public sealed class CallerCommandLanes : ISingletonDependency
+{
+ private readonly Dictionary _lanes = new();
+ private readonly object _gate = new();
+
+ /// How many runs currently hold or await a lane — what a test reads to prove a finished run leaves nothing behind.
+ internal int Count
+ {
+ get { lock (_gate) return _lanes.Count; }
+ }
+
+ /// Wait for 's lane and hold it until the returned handle is disposed. A cancelled wait gives its place back.
+ public async Task EnterAsync(Guid runId, CancellationToken cancellationToken)
+ {
+ var lane = Join(runId);
+
+ try
+ {
+ await lane.Semaphore.WaitAsync(cancellationToken).ConfigureAwait(false);
+ }
+ catch
+ {
+ Leave(runId, lane, held: false);
+ throw;
+ }
+
+ return new Held(this, runId, lane);
+ }
+
+ private Lane Join(Guid runId)
+ {
+ lock (_gate)
+ {
+ if (!_lanes.TryGetValue(runId, out var lane)) _lanes[runId] = lane = new Lane();
+
+ lane.Users++;
+
+ return lane;
+ }
+ }
+
+ private void Leave(Guid runId, Lane lane, bool held)
+ {
+ if (held) lane.Semaphore.Release();
+
+ lock (_gate)
+ {
+ if (--lane.Users > 0) return;
+
+ _lanes.Remove(runId);
+ lane.Semaphore.Dispose();
+ }
+ }
+
+ private sealed class Lane
+ {
+ public SemaphoreSlim Semaphore { get; } = new(1, 1);
+ public int Users { get; set; }
+ }
+
+ private sealed class Held(CallerCommandLanes lanes, Guid runId, Lane lane) : IAsyncDisposable
+ {
+ private int _released;
+
+ public ValueTask DisposeAsync()
+ {
+ if (Interlocked.Exchange(ref _released, 1) == 0) lanes.Leave(runId, lane, held: true);
+
+ return ValueTask.CompletedTask;
+ }
+ }
+}
diff --git a/backend/src/CodeSpace.Core/Services/Agents/Commands/RunCommandService.cs b/backend/src/CodeSpace.Core/Services/Agents/Commands/RunCommandService.cs
index 682fd0aa5..a2c5de193 100644
--- a/backend/src/CodeSpace.Core/Services/Agents/Commands/RunCommandService.cs
+++ b/backend/src/CodeSpace.Core/Services/Agents/Commands/RunCommandService.cs
@@ -2,6 +2,7 @@
using CodeSpace.Core.Persistence.Db;
using CodeSpace.Core.Persistence.Entities;
using CodeSpace.Core.Services.Agents.Sandbox;
+using CodeSpace.Core.Services.Agents.Sandbox.Isolation;
using CodeSpace.Core.Services.Agents.Workspace;
using CodeSpace.Core.Services.Providers;
using CodeSpace.Core.Services.Providers.Auth;
@@ -18,14 +19,16 @@ public sealed class RunCommandService : IRunCommandService, IScopedDependency
private readonly ISandboxRunnerRegistry _runners;
private readonly IWorkspaceProviderRegistry _workspaces;
private readonly AgentDefaultRunnerSetting _defaultRunner;
+ private readonly CallerCommandLanes _lanes;
- public RunCommandService(CodeSpaceDbContext db, IProviderAuthResolver auth, ISandboxRunnerRegistry runners, IWorkspaceProviderRegistry workspaces, AgentDefaultRunnerSetting defaultRunner)
+ public RunCommandService(CodeSpaceDbContext db, IProviderAuthResolver auth, ISandboxRunnerRegistry runners, IWorkspaceProviderRegistry workspaces, AgentDefaultRunnerSetting defaultRunner, CallerCommandLanes lanes)
{
_db = db;
_auth = auth;
_runners = runners;
_workspaces = workspaces;
_defaultRunner = defaultRunner;
+ _lanes = lanes;
}
public async Task RunAsync(RunCommandRequest request, CancellationToken cancellationToken)
@@ -33,6 +36,8 @@ public async Task RunAsync(RunCommandRequest request, Cancellatio
if (string.IsNullOrWhiteSpace(request.Command))
throw new InvalidOperationException("A command is required.");
+ await using var lane = await EnterCallerLaneAsync(request.CallerPosture, cancellationToken).ConfigureAwait(false);
+
var runnerKind = string.IsNullOrWhiteSpace(request.RunnerKind) ? _defaultRunner.Value : request.RunnerKind;
var runner = _runners.Resolve(runnerKind);
@@ -52,6 +57,38 @@ public async Task RunAsync(RunCommandRequest request, Cancellatio
}
}
+ ///
+ /// A command an agent asked for waits its turn among its run's commands (), so the
+ /// commands one run has running never hold more than one tier row of cgroup ceilings between them. A workflow
+ /// node's command has no calling run and never waits.
+ ///
+ private async Task EnterCallerLaneAsync(AgentRunPosture? caller, CancellationToken cancellationToken) =>
+ caller is null ? null : await _lanes.EnterAsync(caller.RunId, cancellationToken).ConfigureAwait(false);
+
+ ///
+ /// What a command an agent asked network for lost to its calling run's posture, as one line the agent and whoever
+ /// approved the call can read on the tool result — or null when nothing was lost: no calling run (a workflow node),
+ /// no network asked for (or none the deployment ceiling allows anyway), or granted as asked. Derived from the same
+ /// projection runs, so it can never disagree with the sandbox the command got. "Off" carries
+ /// the confinement caveat: it is severed only where the sandbox confines.
+ ///
+ public static string? CallerNetworkNarrowing(RunCommandRequest request)
+ {
+ if (request.CallerPosture is not { } caller) return null;
+
+ var authored = AuthoredSpec(request, workingDirectory: null);
+
+ if (!authored.AllowNetwork) return null;
+
+ return WithinCallerPosture(authored, caller) switch
+ {
+ { AllowNetwork: false } when caller.Permissions.Network != AgentNetworkAccess.On => $"off: the calling run ({caller.Autonomy}) has no network{AgentAutonomyPolicy.ConfinementCaveat}",
+ { AllowNetwork: false } => $"off: the calling run's egress allowlist names no host a command may reach{AgentAutonomyPolicy.ConfinementCaveat}",
+ { EgressAllowlist: { Count: > 0 } hosts } => $"narrowed to the calling run's egress allowlist ({string.Join(", ", hosts)})",
+ _ => null,
+ };
+ }
+
///
/// The request → projection, with the deployment autonomy ceiling
/// (Sandbox:MaxAutonomy) narrowing the requested egress. This lane has NO autonomy tier anywhere in its
@@ -61,10 +98,16 @@ public async Task RunAsync(RunCommandRequest request, Cancellatio
/// sandbox enforces) has the last word instead. NARROW-ONLY: a ceiling that grants network leaves the request
/// exactly as asked, so the committed default clamps nothing.
///
+ /// A command an AGENT asked for ( set) is then narrowed to that
+ /// agent's own run by ; a workflow node's command is exactly as above.
+ ///
/// Internal (not private) so the narrowing is unit-pinned directly (InternalsVisibleTo) rather than only
/// through a runner that would have to be confining to show it.
///
- internal static SandboxSpec BuildSpec(RunCommandRequest request, string? workingDirectory) => new()
+ internal static SandboxSpec BuildSpec(RunCommandRequest request, string? workingDirectory) => WithinCallerPosture(AuthoredSpec(request, workingDirectory), request.CallerPosture);
+
+ /// The command as authored, under the deployment ceiling alone — what a workflow node's command runs as, and what an agent's is narrowed from.
+ private static SandboxSpec AuthoredSpec(RunCommandRequest request, string? workingDirectory) => new()
{
Command = request.Command,
Args = request.Args,
@@ -77,6 +120,42 @@ public async Task RunAsync(RunCommandRequest request, Cancellatio
MaxFileSizeMb = request.MaxFileSizeMb,
};
+ ///
+ /// Narrow a command an agent asked for through its tool fabric to the posture of the agent's OWN run. The command's
+ /// sandbox is a sandbox of its own, so without this a network-off agent could hand itself the internet by asking
+ /// for "network": true, and every command it ran was uncapped. NARROW-ONLY: the network stays only when the
+ /// command asked for it, the deployment ceiling allows it (above) AND the run has it; the ceilings are those of the
+ /// run's tier clamped by the deployment ceiling, narrowed by the operator's host memory budget — the same table and
+ /// budget AgentRunExecutor.ApplyResourceCeilings holds the run itself to. Those ceilings land on a cgroup leaf
+ /// of the command's own, beside the agent's: they bound the command, not the run as a whole, which is why a run's
+ /// commands also queue (). No caller (a workflow node) ⇒ the spec is returned untouched.
+ ///
+ private static SandboxSpec WithinCallerPosture(SandboxSpec spec, AgentRunPosture? caller)
+ {
+ if (caller is null) return spec;
+
+ var ceilings = AgentAutonomyPolicy.Ceilings(AgentAutonomyPolicy.Clamp(caller.Autonomy, AgentAutonomyPolicy.DeploymentCeiling), RuntimeSettings.Current.AgentMemoryCeilingMb);
+ var narrowed = spec with { AllowNetwork = spec.AllowNetwork && caller.Permissions.Network == AgentNetworkAccess.On, MaxMemoryMb = ceilings.MemoryMb, MaxCpuPercent = ceilings.CpuPercent };
+
+ return WithinCallerEgress(narrowed, caller.Permissions);
+ }
+
+ ///
+ /// An allowlisted caller's command reaches ONLY the operator's extra hosts ().
+ /// The run's own allowlist adds its model host and its repositories' git hosts; a command needs neither, and a
+ /// repository the command names may sit on a host the run never had, so the extra hosts are the one part that is a
+ /// strict subset of the run's reach. None ⇒ severed, never full egress — the same fail-closed rule
+ /// AgentRunExecutor.ApplyEgressPolicy applies to the run itself.
+ ///
+ private static SandboxSpec WithinCallerEgress(SandboxSpec spec, AgentPermissions permissions)
+ {
+ if (!spec.AllowNetwork || permissions.Egress != AgentEgressPolicy.Allowlist) return spec;
+
+ var hosts = EgressAllowlistBuilder.Build(modelBaseUrl: null, modelProvider: null, Array.Empty(), permissions.EgressAllowHosts);
+
+ return hosts.Count == 0 ? spec with { AllowNetwork = false } : spec with { EgressAllowlist = hosts };
+ }
+
///
/// Repo → clone request: load the repository (by id, like the git.* node services), resolve a short-lived
/// token through the same provider auth layer the resolver uses, and reuse its provider→username table so
diff --git a/backend/src/CodeSpace.Core/Services/Agents/Mcp/AgentMcpEndpoint.cs b/backend/src/CodeSpace.Core/Services/Agents/Mcp/AgentMcpEndpoint.cs
index b13740dd8..a4ea6f3f3 100644
--- a/backend/src/CodeSpace.Core/Services/Agents/Mcp/AgentMcpEndpoint.cs
+++ b/backend/src/CodeSpace.Core/Services/Agents/Mcp/AgentMcpEndpoint.cs
@@ -17,7 +17,7 @@ namespace CodeSpace.Core.Services.Agents.Mcp;
/// One run's live MCP endpoint over a PER-RUN Unix-domain socket: it binds + listens on the run's socket path, accepts
/// connections in a loop, and for each connection validates the per-run CODESPACE_RUN_TOKEN on the FIRST line
/// before serving — then pumps one (a fresh bound to the
-/// run's tool registry + autonomy + team + secret redactor) over the socket's . Every
+/// run's tool registry + autonomy + permissions + team + secret redactor) over the socket's . Every
/// tool-result text the handler returns is run through the run's , so an echoed model key
/// never reaches the model. The connect descriptor
/// (socket path + token) is registered with the under the run id so a consumer
@@ -48,6 +48,7 @@ public sealed class AgentMcpEndpoint : IAsyncDisposable
private readonly bool _governanceEnabled;
private readonly Guid? _approvalConversationId;
private readonly McpCatalogMode _catalogMode;
+ private readonly AgentPermissions? _permissions;
private readonly ILogger _logger;
private readonly CancellationTokenSource _cts;
private readonly Socket _listener;
@@ -56,7 +57,7 @@ public sealed class AgentMcpEndpoint : IAsyncDisposable
private bool _disposed;
- public AgentMcpEndpoint(Guid runId, IAgentToolRegistry registry, AgentAutonomyLevel autonomy, Guid teamId, SecretRedactor redactor, string socketPath, string token, IAgentMcpConnectRegistry connects, IServiceScope scope, CancellationToken ct, ILogger logger, long fenceEpoch = 0, bool governanceEnabled = false, Guid? approvalConversationId = null, McpCatalogMode catalogMode = McpCatalogMode.Full)
+ public AgentMcpEndpoint(Guid runId, IAgentToolRegistry registry, AgentAutonomyLevel autonomy, Guid teamId, SecretRedactor redactor, string socketPath, string token, IAgentMcpConnectRegistry connects, IServiceScope scope, CancellationToken ct, ILogger logger, long fenceEpoch = 0, bool governanceEnabled = false, Guid? approvalConversationId = null, McpCatalogMode catalogMode = McpCatalogMode.Full, AgentPermissions? permissions = null)
{
_runId = runId;
_registry = registry;
@@ -71,6 +72,7 @@ public AgentMcpEndpoint(Guid runId, IAgentToolRegistry registry, AgentAutonomyLe
_governanceEnabled = governanceEnabled;
_approvalConversationId = approvalConversationId;
_catalogMode = catalogMode;
+ _permissions = permissions;
_logger = logger;
_cts = CancellationTokenSource.CreateLinkedTokenSource(ct);
_counters = new McpFabricCounters();
@@ -187,7 +189,7 @@ private async Task ServeConnectionAsync(Socket conn, CancellationToken ct)
var authorityContext = new McpAuthorityContext(_runId, _teamId, connectionScope.ServiceProvider.GetRequiredService(), _counters);
var authorizedRegistry = new AuthorityCheckedToolRegistry(_registry, authorityContext);
- var protocol = new McpRequestHandler(authorizedRegistry, _autonomy, _teamId, _redactor, _runId, ledger, _fenceEpoch, _governanceEnabled, _approvalConversationId, bot, waiters, components, _catalogMode, _counters, _logger);
+ var protocol = new McpRequestHandler(authorizedRegistry, _autonomy, _teamId, _redactor, _runId, ledger, _fenceEpoch, _governanceEnabled, _approvalConversationId, bot, waiters, components, _catalogMode, _counters, _logger, _permissions);
var handler = new AuthorizedMcpRequestHandler(protocol, authorityContext);
diff --git a/backend/src/CodeSpace.Core/Services/Agents/Mcp/McpRequestHandler.cs b/backend/src/CodeSpace.Core/Services/Agents/Mcp/McpRequestHandler.cs
index 672945fbe..c5abaa722 100644
--- a/backend/src/CodeSpace.Core/Services/Agents/Mcp/McpRequestHandler.cs
+++ b/backend/src/CodeSpace.Core/Services/Agents/Mcp/McpRequestHandler.cs
@@ -30,7 +30,9 @@ namespace CodeSpace.Core.Services.Agents.Mcp;
/// tool the tier does not permit comes back as a tool result with isError:true + a reason — never silently
/// run. Per-call TENANCY is enforced: the handler stamps the run's teamId onto every AgentToolCall,
/// which NodeAgentTool writes to the synthetic scope's sys.team_id so a repo-touching tool resolves
-/// the run's tenant (a foreign repository id still fail-closes; a null team → no team → fail-closed). EVERY
+/// the run's tenant (a foreign repository id still fail-closes; a null team → no team → fail-closed). The run's sandbox
+/// posture rides every call the same way (), so a tool that starts a sandbox of
+/// its own (agent.run_command) runs it no wider than the run. EVERY
/// tool-result text the model receives — success output, tool error, AND the caught-exception message — is run
/// through the run's at the single choke point, so an echoed
/// model key can never reach the model through a tool call.
@@ -105,12 +107,17 @@ public sealed class McpRequestHandler : IMcpRequestHandler
// run that did NOT opt into the side-effecting fabric) serves only read-only tools — they are the only ones listed,
// allow-listed, and callable. Full (the existing opt-in) serves the whole registry, byte-identical to before.
private readonly McpCatalogMode _catalogMode;
+ // The posture of the run this connection serves, stamped onto every tool call so a tool that starts a sandbox of
+ // its own runs it no wider than the run. The run's own permissions when the endpoint passed them; a handler built
+ // without them (tests) serves its tier's derived permissions.
+ private readonly AgentRunPosture _posture;
private readonly ILogger _logger;
- public McpRequestHandler(IAgentToolRegistry registry, AgentAutonomyLevel autonomy, Guid? teamId = null, SecretRedactor? redactor = null, Guid runId = default, IToolCallLedgerService? ledger = null, long fenceEpoch = 0, bool governanceEnabled = false, Guid? approvalConversationId = null, IChatBotService? bot = null, IToolApprovalWaiterRegistry? waiters = null, IInteractionComponentRegistry? components = null, McpCatalogMode catalogMode = McpCatalogMode.Full, McpFabricCounters? counters = null, ILogger? logger = null)
+ public McpRequestHandler(IAgentToolRegistry registry, AgentAutonomyLevel autonomy, Guid? teamId = null, SecretRedactor? redactor = null, Guid runId = default, IToolCallLedgerService? ledger = null, long fenceEpoch = 0, bool governanceEnabled = false, Guid? approvalConversationId = null, IChatBotService? bot = null, IToolApprovalWaiterRegistry? waiters = null, IInteractionComponentRegistry? components = null, McpCatalogMode catalogMode = McpCatalogMode.Full, McpFabricCounters? counters = null, ILogger? logger = null, AgentPermissions? permissions = null)
{
_registry = registry;
_autonomy = autonomy;
+ _posture = new AgentRunPosture { RunId = runId, Autonomy = autonomy, Permissions = permissions ?? AgentAutonomyPolicy.Derive(autonomy) };
_counters = counters;
_teamId = teamId;
_redactor = redactor ?? SecretRedactor.None;
@@ -877,7 +884,7 @@ private async Task ExecuteAndRecordAsync(IAgentTool tool, JsonEleme
{
try
{
- var result = await tool.CallAsync(new AgentToolCall { Input = arguments, TeamId = _teamId, RunId = _runId }, cancellationToken).ConfigureAwait(false);
+ var result = await tool.CallAsync(CallFor(arguments), cancellationToken).ConfigureAwait(false);
if (result.IsError)
{
@@ -981,7 +988,7 @@ private async Task InvokeToolAsync(IAgentTool tool, JsonElement arg
{
try
{
- var result = await tool.CallAsync(new AgentToolCall { Input = arguments, TeamId = _teamId, RunId = _runId }, cancellationToken).ConfigureAwait(false);
+ var result = await tool.CallAsync(CallFor(arguments), cancellationToken).ConfigureAwait(false);
if (result.IsError) return ToolResult(isError: true, result.Error ?? "Tool failed.");
@@ -1000,6 +1007,9 @@ private async Task InvokeToolAsync(IAgentTool tool, JsonElement arg
}
}
+ /// The call a tool receives on either path: the model's arguments, plus the run's team, id and sandbox posture — server-stamped, never read from the arguments.
+ private AgentToolCall CallFor(JsonElement arguments) => new() { Input = arguments, TeamId = _teamId, RunId = _runId, CallerPosture = _posture };
+
private static bool IsSupportedVersion(JsonElement request) =>
!request.TryGetProperty("jsonrpc", out var v) || (v.ValueKind == JsonValueKind.String && v.GetString() == "2.0");
diff --git a/backend/src/CodeSpace.Core/Services/Agents/Tools/NodeAgentTool.cs b/backend/src/CodeSpace.Core/Services/Agents/Tools/NodeAgentTool.cs
index 1be24deaa..500466be1 100644
--- a/backend/src/CodeSpace.Core/Services/Agents/Tools/NodeAgentTool.cs
+++ b/backend/src/CodeSpace.Core/Services/Agents/Tools/NodeAgentTool.cs
@@ -17,7 +17,8 @@ namespace CodeSpace.Core.Services.Agents.Tools;
/// Only synchronous nodes are tool-callable: a node that SUSPENDS for an async wait (e.g. agent.run)
/// returns a typed error rather than silently parking — a tool call must produce a concrete result. The node
/// runs against a minimal synthetic context (the tool input as its inputs, no upstream scope, no-op
-/// observability); the agent loop / MCP layer owns its own auditing around the call.
+/// observability, the calling run's ); the agent loop / MCP layer owns its own
+/// auditing around the call.
///
public sealed class NodeAgentTool : IAgentTool
{
@@ -85,6 +86,7 @@ public async Task CallAsync(AgentToolCall call, CancellationTok
Scope = new NodeRunScope { Trigger = new Dictionary(), Sys = sys },
Logger = _logger,
Observability = NodeObservability.NoOp,
+ CallerPosture = call.CallerPosture,
};
var result = await _invocations.ExecuteAsync(new NodeInvocation(_node.TypeKey, context), cancellationToken).ConfigureAwait(false);
diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentRunCommandNode.cs b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentRunCommandNode.cs
index cb2ab6fa2..c241964d8 100644
--- a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentRunCommandNode.cs
+++ b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/Builtin/AgentRunCommandNode.cs
@@ -15,7 +15,9 @@ namespace CodeSpace.Core.Services.Workflows.Nodes.Builtin;
/// need a full AI agent. With a repositoryId the command runs inside a freshly-cloned, per-run
/// workspace (so npm test / make lint see real code); without one it runs ephemerally. The
/// command itself never touches the network unless network is set — secure by default — and runs
-/// under the runner's process / file-size rlimits (a fork-bomb + runaway-write cap).
+/// under the runner's process / file-size rlimits (a fork-bomb + runaway-write cap). Called as an agent tool it also
+/// runs no wider than the calling run (, applied by
+/// ): no network the run lacks, and the run's tier ceilings.
///
/// A non-zero exit is a NORMAL outcome: the node SUCCEEDS with status=Failed/TimedOut + the exit code,
/// so a workflow branches on the result (e.g. tests-passed? → open a PR). The node only FAILS on an
@@ -76,7 +78,7 @@ public AgentRunCommandNode(IRunCommandService runCommand, IArtifactStore artifac
"command": { "type": "string", "minLength": 1, "description": "Executable to run (resolved on PATH, e.g. \"npm\", \"make\", \"pytest\"). Not shell-interpreted — put each argument in Args.", "x-spotlight": 1 },
"args": { "type": "array", "items": { "type": "string" }, "description": "Arguments, one per entry (e.g. [\"test\", \"--silent\"]). No shell splitting or globbing." },
"branch": { "type": "string", "description": "Branch / tag / sha to check out (repo runs only). Empty → the repository's default branch." },
- "network": { "type": "boolean", "description": "Allow the command to reach the network. Off by default — the sandbox severs egress so the command can't call out or exfiltrate." },
+ "network": { "type": "boolean", "description": "Allow the command to reach the network. Off by default — the sandbox severs egress so the command can't call out or exfiltrate. Called as an agent tool, it is granted only when the calling run has network itself, and an allowlisted run's command reaches only the run's operator-named hosts." },
"timeoutSeconds": { "type": "integer", "minimum": 1, "description": "Wall-clock cap. On expiry the command (and its children) are killed and status is TimedOut. Default 600.", "x-spotlight": 3 },
"runnerKind": { "type": "string", "description": "Sandbox backend to run on (e.g. \"local\"). Empty → the deployment default, set by the Agents:DefaultRunnerKind configuration key (Agents__DefaultRunnerKind in the environment); \"local\" when that is unset." },
"maxOutputChars": { "type": "integer", "minimum": 1, "description": "Cap the captured stdout/stderr to this many characters (a head+tail preview is kept). Leave empty to keep the returned capture. Source byte counts and lower-bound flags report whether the runner reached EOF; capture completeness states whether output was lost before this inline cap." }
@@ -103,7 +105,8 @@ public AgentRunCommandNode(IRunCommandService runCommand, IArtifactStore artifac
"stdoutCapturedArtifactId": { "type": "string", "format": "uuid", "description": "Artifact holding only the captured stdout excerpt; missing source content is not recoverable from this artifact." },
"stderrCapturedArtifactId": { "type": "string", "format": "uuid", "description": "Artifact holding only the captured stderr excerpt; missing source content is not recoverable from this artifact." },
"stdoutArtifactId": { "type": "string", "format": "uuid", "description": "Set only when stdout was capped — the artifact id holding the FULL stdout (fetch via /api/artifacts/{id}). Absent when nothing was dropped." },
- "stderrArtifactId": { "type": "string", "format": "uuid", "description": "Set only when stderr was capped — the artifact id holding the FULL stderr. Absent when nothing was dropped." }
+ "stderrArtifactId": { "type": "string", "format": "uuid", "description": "Set only when stderr was capped — the artifact id holding the FULL stderr. Absent when nothing was dropped." },
+ "networkNarrowed": { "type": "string", "description": "Set only when the command was called as an agent tool, asked for the network, and the calling run's posture took it away or narrowed it — says which, so a connection error is not mistaken for a network fault. Absent otherwise, and always absent on a workflow node." }
}
}
""")
@@ -124,6 +127,7 @@ public async Task RunAsync(NodeRunContext context, CancellationToken
Ref = TryReadNonEmpty(context, "branch", out var branch) ? branch : null,
AllowNetwork = TryReadBool(context, "network"),
RunnerKind = TryReadNonEmpty(context, "runnerKind", out var rk) ? rk : null,
+ CallerPosture = context.CallerPosture,
};
if (TryReadPositiveInt(context, "timeoutSeconds", out var timeout)) request = request with { TimeoutSeconds = timeout };
@@ -161,6 +165,7 @@ public async Task RunAsync(NodeRunContext context, CancellationToken
["exitCode"] = JsonSerializer.SerializeToElement(result.ExitCode),
["status"] = JsonSerializer.SerializeToElement(result.Status.ToString()),
};
+ if (RunCommandService.CallerNetworkNarrowing(request) is { } narrowing) outputs["networkNarrowed"] = JsonSerializer.SerializeToElement(narrowing);
foreach (var capture in captures)
{
outputs[capture.Name] = JsonSerializer.SerializeToElement(capture.Inline.Text);
diff --git a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/NodeRunContext.cs b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/NodeRunContext.cs
index f4446bd52..4392f262b 100644
--- a/backend/src/CodeSpace.Core/Services/Workflows/Nodes/NodeRunContext.cs
+++ b/backend/src/CodeSpace.Core/Services/Workflows/Nodes/NodeRunContext.cs
@@ -1,5 +1,6 @@
using System.Text.Json;
using CodeSpace.Core.Services.Workflows.Runtime;
+using CodeSpace.Messages.Agents;
using Microsoft.Extensions.Logging;
namespace CodeSpace.Core.Services.Workflows.Nodes;
@@ -126,4 +127,11 @@ public sealed record NodeRunContext
/// to hand a restored conversation to, so checkpointing one is pure waste.
///
public bool RetriesOnFailure { get; init; }
+
+ ///
+ /// The sandbox posture of the agent run this node is serving as a tool — set only by NodeAgentTool, from the
+ /// run's MCP endpoint. A node that starts a sandbox of its own (agent.run_command) runs it no wider than
+ /// this. Null on the workflow engine path, where the node keeps its own authored posture.
+ ///
+ public AgentRunPosture? CallerPosture { get; init; }
}
diff --git a/backend/src/CodeSpace.Messages/Agents/AgentRunPosture.cs b/backend/src/CodeSpace.Messages/Agents/AgentRunPosture.cs
new file mode 100644
index 000000000..69a4b560b
--- /dev/null
+++ b/backend/src/CodeSpace.Messages/Agents/AgentRunPosture.cs
@@ -0,0 +1,20 @@
+namespace CodeSpace.Messages.Agents;
+
+///
+/// The sandbox posture of the agent run a tool call serves: its autonomy tier and the permissions its own sandbox was
+/// launched with. A tool that starts a sandbox of its own on the run's behalf (agent.run_command) runs it no
+/// wider than this, so a tool call can never reach a network, a host or a resource ceiling the calling agent was
+/// denied. Stamped by the run's MCP endpoint from the run's task; absent off the agent-tool path, where a workflow node
+/// keeps its own authored posture.
+///
+public sealed record AgentRunPosture
+{
+ /// The run this posture is of — the unit its commands queue by, so one run never has two running at once.
+ public Guid RunId { get; init; }
+
+ /// The run's autonomy tier — what its resource ceilings derive from.
+ public required AgentAutonomyLevel Autonomy { get; init; }
+
+ /// The run's effective permissions — its network and egress allowlist.
+ public required AgentPermissions Permissions { get; init; }
+}
diff --git a/backend/src/CodeSpace.Messages/Agents/AgentToolContracts.cs b/backend/src/CodeSpace.Messages/Agents/AgentToolContracts.cs
index 97d817fb6..bc18ece3e 100644
--- a/backend/src/CodeSpace.Messages/Agents/AgentToolContracts.cs
+++ b/backend/src/CodeSpace.Messages/Agents/AgentToolContracts.cs
@@ -24,6 +24,14 @@ public sealed record AgentToolCall
/// may read. Null → no run context (a retrieval tool then has no session to read → fail-closed / empty).
///
public Guid? RunId { get; init; }
+
+ ///
+ /// The sandbox posture of the agent run this call is serving — stamped from the per-run MCP endpoint (same
+ /// provenance as ), never from the model's input. A tool that starts a sandbox of its own runs
+ /// it no wider than this. Null → no calling run (a test, a future non-agent caller): such a tool keeps its own
+ /// posture.
+ ///
+ public AgentRunPosture? CallerPosture { get; init; }
}
/// Result of the pure, I/O-free input-validation stage — the first gate before any permission check or side effect.
diff --git a/backend/src/CodeSpace.Messages/Agents/RunCommandRequest.cs b/backend/src/CodeSpace.Messages/Agents/RunCommandRequest.cs
index 681e28752..15f5d1194 100644
--- a/backend/src/CodeSpace.Messages/Agents/RunCommandRequest.cs
+++ b/backend/src/CodeSpace.Messages/Agents/RunCommandRequest.cs
@@ -49,4 +49,12 @@ public sealed record RunCommandRequest
/// Sandbox runner + workspace backend to use — "local" (v0), later "docker" / "k8s". null → the deployment default (the Agents:DefaultRunnerKind configuration key, itself defaulting to "local").
public string? RunnerKind { get; init; }
+
+ ///
+ /// The posture of the agent run that asked for this command through its tool fabric. Set → the command runs no
+ /// wider than that run: no network unless the run has network, only the run's operator-named hosts when the run is
+ /// allowlisted, and under the resource ceilings of the run's tier — all narrowed further by the deployment ceiling.
+ /// null → a workflow node's own command: its authored posture, narrowed by the deployment ceiling alone.
+ ///
+ public AgentRunPosture? CallerPosture { get; init; }
}
diff --git a/backend/src/CodeSpace.Messages/Agents/SandboxSpec.cs b/backend/src/CodeSpace.Messages/Agents/SandboxSpec.cs
index 622abd934..1bf2c484b 100644
--- a/backend/src/CodeSpace.Messages/Agents/SandboxSpec.cs
+++ b/backend/src/CodeSpace.Messages/Agents/SandboxSpec.cs
@@ -156,8 +156,11 @@ public sealed record SandboxSpec
/// tier via AgentAutonomyPolicy.Ceilings — a committed per-tier value,
/// narrowable per deployment by the operator's host budget (RuntimeSettings.AgentMemoryCeilingMb) and by
/// nothing else. It is applied at the executor's one spec choke point, so every harness's invocation and every
- /// revise round carries it. 0 = unlimited, which is what a spec built OUTSIDE that path still means —
- /// RunCommandService's repo-scoped command runs on the NON-durable runner, which has no cgroup path at all.
+ /// revise round carries it. 0 = unlimited, which is what a spec built OUTSIDE that path still means — a
+ /// workflow node's agent.run_command. The same command asked for by an AGENT carries its calling run's tier
+ /// row instead (RunCommandService), on a cgroup leaf of its own beside the agent's — it bounds the command, it
+ /// does not share the agent's. A run's commands queue, one at a time (CallerCommandLanes), so they never hold
+ /// more than one row between them; the agent and its one running command can together hold up to two.
///
/// ENFORCED only by a runner with cgroup-v2 delegation (the durable local runner on Linux under an
/// operator-delegated Sandbox:CgroupRoot); carried and ignored otherwise, including on macOS development.
diff --git a/backend/tests/CodeSpace.IntegrationTests/Agents/AgentMcpEndpointFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Agents/AgentMcpEndpointFlowTests.cs
index c393efe32..4aff302fd 100644
--- a/backend/tests/CodeSpace.IntegrationTests/Agents/AgentMcpEndpointFlowTests.cs
+++ b/backend/tests/CodeSpace.IntegrationTests/Agents/AgentMcpEndpointFlowTests.cs
@@ -428,6 +428,36 @@ public async Task Endpoint_at_Confined_denies_a_destructive_tool_before_executio
await run;
}
+ [Fact]
+ public async Task A_run_whose_own_network_is_off_cannot_open_one_through_run_command_over_its_endpoint()
+ {
+ if (OperatingSystem.IsWindows()) return;
+ if (!Socket.OSSupportsUnixDomainSockets) return;
+
+ var teamId = await SeedTeamAsync();
+
+ // Unleashed, so the destructive tool is gate-Allowed; the run's OWN permissions pin the network off. The posture
+ // must travel executor → endpoint → handler → tool → node → service from the PERSISTED task, not the tier.
+ var runId = await CreateRunAsync(teamId, AgentAutonomyLevel.Unleashed, permissions: new AgentPermissions { Network = AgentNetworkAccess.Off });
+ var commands = new RecordingCommandRunner();
+
+ using var connects = ConnectRegistryFromFixture();
+ var run = Task.Run(() => ExecuteAsync(runId, new ScriptedHarness("sleep 6"), commandRunner: commands));
+
+ var connect = await WaitForConnectAsync(connects, runId, run);
+ await using var client = await McpClient.ConnectAsync(connect);
+
+ var call = await client.CallToolAsync(1, "agent.run_command", new { command = "true", network = true });
+ call.GetProperty("isError").GetBoolean().ShouldBeFalse(customMessage: $"the command must still run — narrowed, not refused: {call.GetRawText()}");
+
+ var spec = commands.Specs.ShouldHaveSingleItem("the tool call must reach the command runner exactly once");
+ spec.AllowNetwork.ShouldBeFalse("a run whose own network is off cannot open one by asking agent.run_command for it");
+ spec.MaxMemoryMb.ShouldBe(6144, "the command is held to the run's Unleashed memory row");
+ spec.MaxCpuPercent.ShouldBe(400);
+
+ await run;
+ }
+
[Fact]
public async Task A_team_A_endpoint_naming_team_Bs_repo_fails_closed_without_leaking_existence()
{
@@ -1119,7 +1149,7 @@ private async Task ReadAgentRunStatusAsync(Guid runId)
/// at a real existing stand-in (the test only File.Exists-checks it; the scripted harness runs /bin/sh, not the proxy);
/// when false we point it at a missing path to exercise the fail-closed "no declaration" branch.
///
- private async Task ExecuteAsync(Guid runId, IAgentHarness harness, bool proxyPresent = true, bool useGovernanceContainer = false, string? proxyPath = null, CancellationToken cancellationToken = default)
+ private async Task ExecuteAsync(Guid runId, IAgentHarness harness, bool proxyPresent = true, bool useGovernanceContainer = false, string? proxyPath = null, CancellationToken cancellationToken = default, ISandboxRunner? commandRunner = null)
{
// The catalog choice rides the RUN now (CreateRunAsync's enableMcp) and governance is a committed constant, so
// the only environment this still drives is the proxy path — a genuine filesystem seam, not a feature flag.
@@ -1132,7 +1162,11 @@ private async Task ExecuteAsync(Guid runId, IAgentHarness harness, bool proxyPre
// useGovernanceContainer routes the run through the SECOND, governance-on container so the endpoint's DI
// IAgentToolRegistry actually contains decision.request (registry composition is fixed at container build).
using var scope = useGovernanceContainer ? _fixture.BeginGovernanceOnScope() : _fixture.BeginScope();
- await NewExecutor(scope, harness).ExecuteAsync(runId, cancellationToken);
+
+ // A command runner, when given, is what the scopes the executor makes for itself — the MCP endpoint's among
+ // them, and so every tool call's — run commands on. The harness keeps the real runner the executor is handed.
+ using var toolScope = commandRunner is null ? null : scope.BeginLifetimeScope(b => b.RegisterInstance(new SandboxRunnerRegistry([commandRunner])).As());
+ await NewExecutor(scope, harness, toolScope).ExecuteAsync(runId, cancellationToken);
}
finally
{
@@ -1157,7 +1191,7 @@ private async Task ReattachAsync(AgentRunReattachReservation reservation, IAgent
}
}
- private static AgentRunExecutor NewExecutor(ILifetimeScope scope, IAgentHarness harness) => new(
+ private static AgentRunExecutor NewExecutor(ILifetimeScope scope, IAgentHarness harness, ILifetimeScope? toolScope = null) => new(
scope.Resolve(),
new AgentHarnessRegistry(new[] { harness }),
new HarnessModelReconciler(new AgentHarnessRegistry(new[] { harness }), scope.Resolve(), scope.Resolve()),
@@ -1166,7 +1200,7 @@ private async Task ReattachAsync(AgentRunReattachReservation reservation, IAgent
scope.Resolve(),
scope.Resolve(),
scope.Resolve(),
- scope.Resolve(),
+ toolScope is null ? scope.Resolve() : new ScopedServiceScopeFactory(toolScope),
scope.Resolve(),
scope.Resolve(),
scope.Resolve(),
@@ -1471,11 +1505,11 @@ private static string[] ToolNames(JsonElement listResponse) =>
// ── Seeding (mirrors McpToolTeamScopeFlowTests + AgentRunExecutorTests) ──
/// is the per-run catalog choice — null takes the committed default (full), false narrows the run to the read-only slice. It replaced the ambient env flag the helpers used to set.
- private async Task CreateRunAsync(Guid teamId, AgentAutonomyLevel autonomy, IReadOnlyList? tools = null, bool? enableMcp = null, string harnessKind = "scripted", string? model = "test-model")
+ private async Task CreateRunAsync(Guid teamId, AgentAutonomyLevel autonomy, IReadOnlyList? tools = null, bool? enableMcp = null, string harnessKind = "scripted", string? model = "test-model", AgentPermissions? permissions = null)
{
using var scope = _fixture.BeginScopeAs(_operators[teamId], teamId);
var run = await scope.Resolve().CreateAsync(
- new AgentTask { Goal = "scripted", Harness = harnessKind, Model = model, TimeoutSeconds = 1800, Autonomy = autonomy, Tools = tools, EnableMcpEndpoint = enableMcp },
+ new AgentTask { Goal = "scripted", Harness = harnessKind, Model = model, TimeoutSeconds = 1800, Autonomy = autonomy, Permissions = permissions ?? new(), Tools = tools, EnableMcpEndpoint = enableMcp },
teamId, null, null, iterationKey: "", cancellationToken: CancellationToken.None);
return run.Id;
}
@@ -1549,6 +1583,40 @@ private sealed class TempDir : IDisposable
}
/// Holds the fixture scope open while a test reads the connect-registry singleton it resolved from it.
+ ///
+ /// An rooted at THIS scope, so a per-test registration override is visible to the
+ /// scopes the executor creates for itself. The container's own factory is a singleton holding the ROOT lifetime
+ /// scope, so every scope it makes would see nothing a test registered.
+ ///
+ private sealed class ScopedServiceScopeFactory(ILifetimeScope scope) : IServiceScopeFactory
+ {
+ public IServiceScope CreateScope() => new Scope(scope.BeginLifetimeScope());
+
+ private sealed class Scope(ILifetimeScope child) : IServiceScope
+ {
+ public IServiceProvider ServiceProvider { get; } = new Autofac.Extensions.DependencyInjection.AutofacServiceProvider(child);
+
+ public void Dispose() => child.Dispose();
+ }
+ }
+
+ /// Records every command spec it is handed, then runs it on the real local runner.
+ private sealed class RecordingCommandRunner : ISandboxRunner
+ {
+ private readonly LocalProcessRunner _real = new();
+
+ public System.Collections.Concurrent.ConcurrentQueue Specs { get; } = new();
+
+ public string Kind => LocalProcessRunner.LocalKind;
+
+ public Task RunAsync(SandboxSpec spec, CancellationToken cancellationToken)
+ {
+ Specs.Enqueue(spec);
+
+ return _real.RunAsync(spec, cancellationToken);
+ }
+ }
+
private sealed class FlagScope : IDisposable
{
private readonly IDisposable _scope;
diff --git a/backend/tests/CodeSpace.IntegrationTests/Agents/RunCommandCallerPostureFlowTests.cs b/backend/tests/CodeSpace.IntegrationTests/Agents/RunCommandCallerPostureFlowTests.cs
new file mode 100644
index 000000000..3eb319ca3
--- /dev/null
+++ b/backend/tests/CodeSpace.IntegrationTests/Agents/RunCommandCallerPostureFlowTests.cs
@@ -0,0 +1,204 @@
+using System.Collections.Concurrent;
+using System.Text.Json;
+using Autofac;
+using CodeSpace.Core.Persistence.Db;
+using CodeSpace.Core.Services.Agents.Mcp;
+using CodeSpace.Core.Services.Agents.Sandbox;
+using CodeSpace.Core.Services.Agents.Sandbox.Runners;
+using CodeSpace.Core.Services.Agents.Tools;
+using CodeSpace.Core.Services.Workflows.Engine;
+using CodeSpace.IntegrationTests.Infrastructure;
+using CodeSpace.IntegrationTests.Workflows.Infrastructure;
+using CodeSpace.Messages.Agents;
+using CodeSpace.Messages.Commands.Workflows;
+using CodeSpace.Messages.Constants;
+using CodeSpace.Messages.Dtos.Workflows;
+using CodeSpace.Messages.Enums;
+using MediatR;
+using Microsoft.EntityFrameworkCore;
+using Shouldly;
+
+namespace CodeSpace.IntegrationTests.Agents;
+
+///
+/// The two lanes into agent.run_command, side by side over the production registrations: an AGENT's call
+/// through the real MCP tool dispatch (McpRequestHandler → NodeAgentTool → the node →
+/// RunCommandService), and a WORKFLOW NODE run by the real engine. The agent's command must run no wider than the
+/// agent's own run — a network-off caller that asks for "network": true gets a severed sandbox and its tier's
+/// ceilings — while the workflow node's authored network is exactly what it was.
+///
+/// Tier 🟡 medium-mock: every production class runs for real, including the real
+/// that executes the command; the only stand-in is a recording decorator on that runner, because the posture is a
+/// property of the spec it receives and a macOS / unconfinable host cannot show a severed namespace. The spec it
+/// recorded is then run through the production child chain () with a
+/// stand-in bwrap path, so the assertion is the argv a confining Linux worker would exec. The real-kernel proof of
+/// --unshare-net itself belongs to the Linux sandbox lane.
+///
+[Collection(PostgresCollection.Name)]
+[Trait("Category", "Integration")]
+public sealed class RunCommandCallerPostureFlowTests(PostgresFixture fixture)
+{
+ private const string FakeBwrap = "/usr/bin/bwrap";
+
+ [Fact]
+ public async Task A_network_off_agent_asking_run_command_for_the_network_gets_a_severed_ceilinged_sandbox_through_the_real_tool_dispatch()
+ {
+ if (OperatingSystem.IsWindows()) return;
+
+ var (teamId, _) = await WorkflowsTestSeed.SeedTeamAsync(fixture);
+ var runner = new RecordingRunner();
+ using var scope = ScopeRunningCommandsOn(runner);
+
+ // Unleashed so the destructive tool is gate-Allowed and needs no approval; the run's OWN permissions keep the
+ // network off — the posture an author pins with an agent.run node's network override.
+ var handler = new McpRequestHandler(scope.Resolve(), AgentAutonomyLevel.Unleashed, teamId, permissions: new AgentPermissions { Network = AgentNetworkAccess.Off });
+
+ var result = await CallToolAsync(handler, "agent.run_command", new { command = "true", network = true });
+
+ result.GetProperty("isError").GetBoolean().ShouldBeFalse(customMessage: $"the command must still run — narrowed, not refused: {result.GetRawText()}");
+ var spec = runner.Specs.ShouldHaveSingleItem("the tool call must reach the runner exactly once");
+ spec.AllowNetwork.ShouldBeFalse("a network-off run cannot open a network by asking a tool for one");
+ spec.MaxMemoryMb.ShouldBe(6144, "the command is held to the calling run's Unleashed memory row");
+ spec.MaxCpuPercent.ShouldBe(400);
+ Chain(spec).ShouldContain("--unshare-net", customMessage: $"a confining worker must sever the command's network: [{string.Join(' ', Chain(spec))}]");
+ result.GetRawText().ShouldContain("networkNarrowed", customMessage: "the tool result tells the agent its run's posture took the network it asked for");
+ }
+
+ [Theory]
+ [InlineData(true, 1)] // two connections of ONE run: the run's commands queue
+ [InlineData(false, 2)] // two runs: never queued behind each other
+ public async Task Overlapping_commands_from_one_run_never_run_at_once_through_the_real_tool_dispatch(bool sameRun, int expectedMaxConcurrent)
+ {
+ // Each command gets a cgroup leaf of its own carrying the run's whole tier row. The run token in the agent's
+ // config lets it open as many endpoint connections as it likes, each with its own handler — so a run that
+ // starts two commands at once would hold two rows. Queued per run, its commands never hold more than one.
+ if (OperatingSystem.IsWindows()) return;
+
+ var (teamId, _) = await WorkflowsTestSeed.SeedTeamAsync(fixture);
+ var runner = new OverlapRunner(expectedConcurrency: 2);
+ using var scope = fixture.BeginScope(builder => builder.RegisterInstance(new SandboxRunnerRegistry([runner])).As());
+ var firstRun = Guid.NewGuid();
+ var connections = new[] { firstRun, sameRun ? firstRun : Guid.NewGuid() }
+ .Select(runId => new McpRequestHandler(scope.Resolve(), AgentAutonomyLevel.Unleashed, teamId, runId: runId))
+ .ToList();
+
+ var results = await Task.WhenAll(connections.Select(handler => CallToolAsync(handler, "agent.run_command", new { command = "true" })));
+
+ results.ShouldAllBe(result => !result.GetProperty("isError").GetBoolean(), customMessage: "fixture check: both commands ran");
+ runner.Started.ShouldBe(2);
+ runner.MaxConcurrent.ShouldBe(expectedMaxConcurrent, sameRun ? "one run's commands must never run at the same time" : "another run's command is not queued behind this one");
+ }
+
+ [Fact]
+ public async Task A_workflow_node_asking_for_the_network_keeps_it_through_the_real_engine()
+ {
+ if (OperatingSystem.IsWindows()) return;
+
+ var (teamId, userId) = await WorkflowsTestSeed.SeedTeamAsync(fixture);
+ var workflowId = await CreateWorkflowAsync(teamId, userId);
+ var runId = await WorkflowsTestSeed.SeedManualRunAsync(fixture, workflowId, teamId);
+ var runner = new RecordingRunner();
+
+ using (var scope = ScopeRunningCommandsOn(runner))
+ await scope.Resolve().ExecuteRunAsync(runId, CancellationToken.None);
+
+ using (var verify = fixture.BeginScope())
+ {
+ var node = await verify.Resolve().WorkflowRunNode.AsNoTracking().SingleAsync(candidate => candidate.RunId == runId && candidate.NodeId == "command");
+ node.Status.ShouldBe(NodeStatus.Success, node.Error);
+ }
+
+ var spec = runner.Specs.ShouldHaveSingleItem("the engine must run the node's command exactly once");
+ spec.AllowNetwork.ShouldBeTrue("a workflow node has no calling run: its authored network stands, as it always did");
+ spec.MaxMemoryMb.ShouldBe(0, "no caller ceilings on the workflow lane — unchanged");
+ spec.MaxCpuPercent.ShouldBe(0);
+ spec.EgressAllowlist.ShouldBeNull();
+ Chain(spec).ShouldNotContain("--unshare-net", customMessage: $"the workflow node's command keeps the host network: [{string.Join(' ', Chain(spec))}]");
+ }
+
+ // ── Helpers ───────────────────────────────────────────────────────────────
+
+ /// A scope whose "local" sandbox runner is ; everything else is the production registration, and child scopes (the tool's node invocation, the engine's) inherit it.
+ private ILifetimeScope ScopeRunningCommandsOn(RecordingRunner runner) =>
+ fixture.BeginScope(builder => builder.RegisterInstance(new SandboxRunnerRegistry([runner])).As());
+
+ private async Task CreateWorkflowAsync(Guid teamId, Guid userId)
+ {
+ using var author = fixture.BeginScopeAs(userId, teamId, Roles.Admin);
+
+ return await author.Resolve().Send(new CreateWorkflowCommand
+ {
+ Name = "networked-command", Enabled = true, Activations = new List(),
+ Definition = new WorkflowDefinition
+ {
+ SchemaVersion = 1,
+ Nodes = new List
+ {
+ new() { Id = "start", TypeKey = "trigger.manual", Config = WorkflowsTestSeed.EmptyJson(), Inputs = WorkflowsTestSeed.EmptyJson() },
+ new() { Id = "command", TypeKey = "agent.run_command", Config = WorkflowsTestSeed.EmptyJson(), Inputs = JsonSerializer.SerializeToElement(new { command = "true", network = true }) },
+ new() { Id = "end", TypeKey = "builtin.terminal", Config = WorkflowsTestSeed.EmptyJson(), Inputs = WorkflowsTestSeed.EmptyJson() },
+ },
+ Edges = new List { new() { From = "start", To = "command" }, new() { From = "command", To = "end" } },
+ },
+ });
+ }
+
+ private static async Task CallToolAsync(McpRequestHandler handler, string name, object arguments)
+ {
+ var request = JsonSerializer.SerializeToElement(new { jsonrpc = "2.0", id = 1, method = "tools/call", @params = new { name, arguments } });
+
+ var response = await handler.HandleAsync(request, CancellationToken.None);
+
+ return response!.Value.GetProperty("result");
+ }
+
+ /// The production child chain for on a host whose bwrap is at — what a confining worker execs.
+ private static IReadOnlyList Chain(SandboxSpec spec) =>
+ LocalProcessRunner.ChildCommand(new LocalProcessRunner.CommandIsolationContext(spec, null, null, Array.Empty(), Array.Empty()), FakeBwrap, prlimit: null);
+
+ /// Records how many commands run at once; each waits until run together or a short grace elapses, so commands that can overlap always do.
+ private sealed class OverlapRunner(int expectedConcurrency) : ISandboxRunner
+ {
+ private readonly object _gate = new();
+ private readonly TaskCompletionSource _all = new(TaskCreationOptions.RunContinuationsAsynchronously);
+ private int _running;
+
+ public int Started { get; private set; }
+ public int MaxConcurrent { get; private set; }
+ public string Kind => LocalProcessRunner.LocalKind;
+
+ public async Task RunAsync(SandboxSpec spec, CancellationToken cancellationToken)
+ {
+ lock (_gate)
+ {
+ Started++;
+ _running++;
+ MaxConcurrent = Math.Max(MaxConcurrent, _running);
+ if (_running >= expectedConcurrency) _all.TrySetResult();
+ }
+
+ await Task.WhenAny(_all.Task, Task.Delay(TimeSpan.FromSeconds(2), cancellationToken));
+
+ lock (_gate) _running--;
+
+ return new SandboxResult { Status = SandboxStatus.Success, ExitCode = 0, Stdout = "", Stderr = "" };
+ }
+ }
+
+ /// Records every spec it is handed, then runs it on the real local runner.
+ private sealed class RecordingRunner : ISandboxRunner
+ {
+ private readonly LocalProcessRunner _real = new();
+
+ public ConcurrentQueue Specs { get; } = new();
+
+ public string Kind => LocalProcessRunner.LocalKind;
+
+ public Task RunAsync(SandboxSpec spec, CancellationToken cancellationToken)
+ {
+ Specs.Enqueue(spec);
+
+ return _real.RunAsync(spec, cancellationToken);
+ }
+ }
+}
diff --git a/backend/tests/CodeSpace.UnitTests/Agents/CallerCommandLanesTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/CallerCommandLanesTests.cs
new file mode 100644
index 000000000..5c90c4be4
--- /dev/null
+++ b/backend/tests/CodeSpace.UnitTests/Agents/CallerCommandLanesTests.cs
@@ -0,0 +1,60 @@
+using CodeSpace.Core.Services.Agents.Commands;
+using Shouldly;
+
+namespace CodeSpace.UnitTests.Agents;
+
+///
+/// 🟢 Unit: the per-run queue agent.run_command enters when an agent calls it. One entry per run at a time,
+/// none across runs, a cancelled waiter gives its place back, and a run whose commands are done holds no lane.
+///
+[Trait("Category", "Unit")]
+public sealed class CallerCommandLanesTests
+{
+ [Fact]
+ public async Task A_second_entry_for_the_same_run_waits_until_the_first_leaves()
+ {
+ var lanes = new CallerCommandLanes();
+ var runId = Guid.NewGuid();
+
+ var first = await lanes.EnterAsync(runId, CancellationToken.None);
+ var second = lanes.EnterAsync(runId, CancellationToken.None);
+
+ second.IsCompleted.ShouldBeFalse("fixture check: the second entry starts out waiting");
+
+ await first.DisposeAsync();
+ await using var entered = await second.WaitAsync(TimeSpan.FromSeconds(5));
+
+ lanes.Count.ShouldBe(1, "the run still has a command in its lane");
+ }
+
+ [Fact]
+ public async Task Entries_for_different_runs_never_wait_on_each_other()
+ {
+ var lanes = new CallerCommandLanes();
+
+ await using var first = await lanes.EnterAsync(Guid.NewGuid(), CancellationToken.None);
+ var second = lanes.EnterAsync(Guid.NewGuid(), CancellationToken.None);
+
+ second.IsCompletedSuccessfully.ShouldBeTrue("another run's command is not queued behind this one");
+ await (await second).DisposeAsync();
+ }
+
+ [Fact]
+ public async Task A_cancelled_waiter_gives_its_place_back_and_a_finished_run_holds_no_lane()
+ {
+ var lanes = new CallerCommandLanes();
+ var runId = Guid.NewGuid();
+ using var cancel = new CancellationTokenSource();
+
+ var first = await lanes.EnterAsync(runId, CancellationToken.None);
+ var waiter = lanes.EnterAsync(runId, cancel.Token);
+
+ await cancel.CancelAsync();
+ await Should.ThrowAsync(() => waiter);
+
+ await first.DisposeAsync();
+
+ lanes.Count.ShouldBe(0, "a run whose commands are all done (or gave up) leaves nothing behind");
+ await (await lanes.EnterAsync(runId, CancellationToken.None).WaitAsync(TimeSpan.FromSeconds(5))).DisposeAsync();
+ }
+}
diff --git a/backend/tests/CodeSpace.UnitTests/Agents/DefaultRunnerKindTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/DefaultRunnerKindTests.cs
index 1b0eff84c..a48615d44 100644
--- a/backend/tests/CodeSpace.UnitTests/Agents/DefaultRunnerKindTests.cs
+++ b/backend/tests/CodeSpace.UnitTests/Agents/DefaultRunnerKindTests.cs
@@ -73,7 +73,7 @@ public void The_configured_kind_arrives_through_the_environment_form_of_the_key(
public async Task RunCommandService_resolves_the_deployment_default_only_when_the_request_pins_none(string? requestKind, string expectedKind)
{
var runners = new RecordingRunnerRegistry();
- var service = new RunCommandService(null!, null!, runners, null!, Setting("cfg-runner"));
+ var service = new RunCommandService(null!, null!, runners, null!, Setting("cfg-runner"), new CallerCommandLanes());
// Ephemeral (no repositoryId) so the DbContext / auth / workspace collaborators are never touched.
await service.RunAsync(new RunCommandRequest { Command = "true", RunnerKind = requestKind }, CancellationToken.None);
diff --git a/backend/tests/CodeSpace.UnitTests/Agents/McpRequestHandlerTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/McpRequestHandlerTests.cs
index bfd61e699..101b5d569 100644
--- a/backend/tests/CodeSpace.UnitTests/Agents/McpRequestHandlerTests.cs
+++ b/backend/tests/CodeSpace.UnitTests/Agents/McpRequestHandlerTests.cs
@@ -521,6 +521,41 @@ public async Task ToolsCall_stamps_the_handlers_run_id_onto_the_tool_call()
seen.ShouldBe(runId, "the run id must travel handler → AgentToolCall so a retrieval tool can scope to the run's session");
}
+ [Theory]
+ [InlineData(false)]
+ [InlineData(true)]
+ public async Task ToolsCall_stamps_the_runs_posture_onto_the_tool_call_on_the_ungoverned_and_the_governed_path(bool governed)
+ {
+ // The posture travels handler → AgentToolCall like TeamId and RunId, from the run's endpoint and never from the
+ // model's arguments, so agent.run_command's sandbox can be no wider than the run's. Both paths that invoke a
+ // tool (the ledger-governed write and the ungoverned call) must carry it.
+ var permissions = new AgentPermissions { Network = AgentNetworkAccess.On, Egress = AgentEgressPolicy.Allowlist, EgressAllowHosts = ["registry.npmjs.org"] };
+ AgentRunPosture? seen = null;
+ var tool = new FakeTool { Kind = "agent.run_command", IsDestructiveOverride = true, OnCall = (c, _) => { seen = c.CallerPosture; return Task.FromResult(AgentToolResult.Ok(Parse("{}"), 2)); } };
+ var runId = Guid.NewGuid();
+ var handler = new McpRequestHandler(new FakeRegistry(tool), AgentAutonomyLevel.Unleashed, Guid.NewGuid(), null, runId, new SpyLedger(), fenceEpoch: 1, governanceEnabled: governed, permissions: permissions);
+
+ await Respond(handler, Call("agent.run_command", """{"network":true}"""));
+
+ var posture = seen.ShouldNotBeNull("the tool must see the run's posture");
+ posture.RunId.ShouldBe(runId, "the run the posture belongs to — the unit a run's commands queue by");
+ posture.Autonomy.ShouldBe(AgentAutonomyLevel.Unleashed);
+ posture.Permissions.ShouldBeSameAs(permissions, "the run's own permissions, not a re-derivation from its tier");
+ }
+
+ [Fact]
+ public async Task ToolsCall_on_a_handler_built_without_permissions_stamps_its_tiers_derived_posture()
+ {
+ AgentRunPosture? seen = null;
+ var tool = new FakeTool { Kind = "echo", OnCall = (c, _) => { seen = c.CallerPosture; return Task.FromResult(AgentToolResult.Ok(Parse("{}"), 2)); } };
+
+ await Respond(Handler(AgentAutonomyLevel.Standard, tool), Call("echo", "{}"));
+
+ var posture = seen.ShouldNotBeNull();
+ posture.Autonomy.ShouldBe(AgentAutonomyLevel.Standard);
+ posture.Permissions.ShouldBe(AgentAutonomyPolicy.Derive(AgentAutonomyLevel.Standard), "with no run permissions the tier's own derivation stands in — never an open network");
+ }
+
[Fact]
public async Task ToolsCall_with_no_run_on_the_handler_stamps_the_empty_run_id()
{
diff --git a/backend/tests/CodeSpace.UnitTests/Agents/NodeAgentToolTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/NodeAgentToolTests.cs
index af78f9a9c..b505c485a 100644
--- a/backend/tests/CodeSpace.UnitTests/Agents/NodeAgentToolTests.cs
+++ b/backend/tests/CodeSpace.UnitTests/Agents/NodeAgentToolTests.cs
@@ -187,6 +187,22 @@ public async Task A_call_with_no_team_leaves_sys_empty_so_the_scope_reader_fails
NodeScopeReader.TryReadTeamId(captured, out _).ShouldBeFalse();
}
+ [Fact]
+ public async Task A_call_carries_its_runs_posture_onto_the_synthetic_context_and_a_call_without_one_carries_none()
+ {
+ // The posture rides the call (stamped by the run's endpoint), never the model's input, and lands on the typed
+ // context field a sandbox-starting node reads — so agent.run_command runs no wider than the calling run.
+ var posture = new AgentRunPosture { Autonomy = AgentAutonomyLevel.Unleashed, Permissions = new AgentPermissions { Network = AgentNetworkAccess.Off } };
+ var withPosture = new CapturingNode();
+ var without = new CapturingNode();
+
+ await Tool(withPosture).CallAsync(new AgentToolCall { Input = EmptyObject, CallerPosture = posture }, CancellationToken.None);
+ await Tool(without).CallAsync(new AgentToolCall { Input = EmptyObject }, CancellationToken.None);
+
+ withPosture.Captured.ShouldNotBeNull().CallerPosture.ShouldBeSameAs(posture);
+ without.Captured.ShouldNotBeNull().CallerPosture.ShouldBeNull("no calling run → the node keeps its own posture");
+ }
+
[Fact]
public async Task TeamId_does_not_alter_inputs_rawinputs_config_or_observability()
{
diff --git a/backend/tests/CodeSpace.UnitTests/Agents/RunCommandCallerPostureTests.cs b/backend/tests/CodeSpace.UnitTests/Agents/RunCommandCallerPostureTests.cs
new file mode 100644
index 000000000..6bbe25fd7
--- /dev/null
+++ b/backend/tests/CodeSpace.UnitTests/Agents/RunCommandCallerPostureTests.cs
@@ -0,0 +1,276 @@
+using CodeSpace.Core.Services.Agents;
+using CodeSpace.Core.Services.Agents.Commands;
+using CodeSpace.Core.Services.Agents.Sandbox;
+using CodeSpace.Core.Services.Agents.Sandbox.Isolation;
+using CodeSpace.Core.Services.Agents.Sandbox.Runners;
+using CodeSpace.Core.Settings;
+using CodeSpace.Messages.Agents;
+using Microsoft.Extensions.Configuration;
+using Shouldly;
+
+namespace CodeSpace.UnitTests.Agents;
+
+///
+/// agent.run_command called by an AGENT runs no wider than that agent's own run. The command's sandbox is a
+/// sandbox of its own — the tier clamps that bound the agent's spec never reach it — so without the caller's posture
+/// a network-off agent could hand itself the internet by asking a tool for "network": true, and every command it
+/// ran escaped the cgroup ceilings its own run is held to. Pinned at (the one
+/// request → spec projection) and then through the production child chain
+/// () with a stand-in bwrap, so the severed network is proven as the argv
+/// a confining host would exec, on any host. A workflow node — no caller — keeps exactly the posture it had.
+///
+[Trait("Category", "Unit")]
+public class RunCommandCallerPostureTests
+{
+ private const string FakeBwrap = "/usr/bin/bwrap";
+
+ // ── The matrix: caller tier × deployment ceiling ──────────────────────────
+
+ [Theory]
+ // caller tier deployment network memory cpu
+ [InlineData(AgentAutonomyLevel.Confined, null, false, 1024, 100)]
+ [InlineData(AgentAutonomyLevel.Confined, "Trusted", false, 1024, 100)]
+ [InlineData(AgentAutonomyLevel.Confined, "Standard", false, 1024, 100)]
+ [InlineData(AgentAutonomyLevel.Confined, "Confined", false, 1024, 100)]
+ [InlineData(AgentAutonomyLevel.Standard, null, false, 4096, 400)]
+ [InlineData(AgentAutonomyLevel.Standard, "Trusted", false, 4096, 400)]
+ [InlineData(AgentAutonomyLevel.Standard, "Standard", false, 4096, 400)]
+ [InlineData(AgentAutonomyLevel.Standard, "Confined", false, 1024, 100)]
+ [InlineData(AgentAutonomyLevel.Trusted, null, true, 6144, 400)]
+ [InlineData(AgentAutonomyLevel.Trusted, "Trusted", true, 6144, 400)]
+ [InlineData(AgentAutonomyLevel.Trusted, "Standard", false, 4096, 400)]
+ [InlineData(AgentAutonomyLevel.Trusted, "Confined", false, 1024, 100)]
+ [InlineData(AgentAutonomyLevel.Unleashed, null, true, 6144, 400)]
+ [InlineData(AgentAutonomyLevel.Unleashed, "Trusted", true, 6144, 400)]
+ [InlineData(AgentAutonomyLevel.Unleashed, "Standard", false, 4096, 400)]
+ [InlineData(AgentAutonomyLevel.Unleashed, "Confined", false, 1024, 100)]
+ public void An_agent_called_command_runs_no_wider_than_its_callers_tier_under_the_deployment_ceiling(AgentAutonomyLevel caller, string? deployment, bool expectedNetwork, int expectedMemoryMb, int expectedCpuPercent)
+ {
+ var spec = WithSettings(deployment, hostMemoryBudgetMb: null, () => RunCommandService.BuildSpec(AgentAsks(PostureOf(caller)), workingDirectory: null));
+
+ spec.AllowNetwork.ShouldBe(expectedNetwork, $"a {caller} caller under deployment ceiling '{deployment ?? "(unset)"}' asking for the network must reach the runner with AllowNetwork={expectedNetwork}");
+ spec.MaxMemoryMb.ShouldBe(expectedMemoryMb, $"a {caller} caller's command is held to its tier's memory row, clamped by the deployment ceiling '{deployment ?? "(unset)"}'");
+ spec.MaxCpuPercent.ShouldBe(expectedCpuPercent);
+ spec.EgressAllowlist.ShouldBeNull("a caller with full egress narrows nothing to an allowlist");
+ }
+
+ [Theory]
+ [InlineData(null, true)]
+ [InlineData("Trusted", true)]
+ [InlineData("Standard", false)]
+ public void A_workflow_node_command_keeps_its_authored_posture_under_the_deployment_ceiling_alone(string? deployment, bool expectedNetwork)
+ {
+ var spec = WithSettings(deployment, hostMemoryBudgetMb: null, () => RunCommandService.BuildSpec(new RunCommandRequest { Command = "curl", AllowNetwork = true }, workingDirectory: null));
+
+ spec.AllowNetwork.ShouldBe(expectedNetwork, "no calling run: the authored network flag under the deployment ceiling, exactly as before");
+ spec.MaxMemoryMb.ShouldBe(0, "a workflow node's command carries no caller ceiling — unchanged");
+ spec.MaxCpuPercent.ShouldBe(0);
+ spec.EgressAllowlist.ShouldBeNull();
+ }
+
+ // ── The run's own permissions, not just its tier ──────────────────────────
+
+ [Fact]
+ public void A_caller_whose_own_network_is_off_cannot_reach_the_network_through_a_command_whatever_its_tier()
+ {
+ // An Unleashed run whose author pinned network off: the agent's own sandbox is severed, so the command it asks
+ // for must be too — the tier alone would have granted it.
+ var caller = new AgentRunPosture { Autonomy = AgentAutonomyLevel.Unleashed, Permissions = new AgentPermissions { Network = AgentNetworkAccess.Off } };
+
+ var spec = WithSettings(null, hostMemoryBudgetMb: null, () => RunCommandService.BuildSpec(AgentAsks(caller), workingDirectory: null));
+
+ spec.AllowNetwork.ShouldBeFalse("a command an agent asks for never has a network its caller does not");
+ spec.MaxMemoryMb.ShouldBe(6144, "the ceilings still follow the caller's tier");
+ }
+
+ [Fact]
+ public void A_command_that_did_not_ask_for_the_network_gets_none_from_a_networked_caller()
+ {
+ var spec = WithSettings(null, hostMemoryBudgetMb: null, () => RunCommandService.BuildSpec(new RunCommandRequest { Command = "make", AllowNetwork = false, CallerPosture = PostureOf(AgentAutonomyLevel.Unleashed) }, workingDirectory: null));
+
+ spec.AllowNetwork.ShouldBeFalse("the caller's posture only ever narrows: a command that asked for no network gets none");
+ }
+
+ [Theory]
+ [InlineData(new[] { " Registry.NPMjs.org " }, true, new[] { "registry.npmjs.org" })]
+ [InlineData(new string[0], false, null)]
+ public void An_allowlisted_caller_hands_its_command_only_the_operator_named_hosts_and_severs_it_when_there_are_none(string[] extraHosts, bool expectedNetwork, string[]? expectedAllowlist)
+ {
+ // The run's own allowlist adds its model host and its repositories' git hosts; a command needs neither, and a
+ // repository the command names may sit on a host the run never had. Only the operator's extra hosts are a
+ // strict subset of what the run may reach, so they are all the command gets — and none means severed, never
+ // full egress.
+ var caller = new AgentRunPosture { Autonomy = AgentAutonomyLevel.Trusted, Permissions = new AgentPermissions { Network = AgentNetworkAccess.On, Egress = AgentEgressPolicy.Allowlist, EgressAllowHosts = extraHosts } };
+
+ var spec = WithSettings(null, hostMemoryBudgetMb: null, () => RunCommandService.BuildSpec(AgentAsks(caller), workingDirectory: null));
+
+ spec.AllowNetwork.ShouldBe(expectedNetwork);
+ spec.EgressAllowlist.ShouldBe(expectedAllowlist);
+ }
+
+ [Fact]
+ public void The_operators_host_memory_budget_narrows_an_agent_called_commands_ceiling()
+ {
+ var spec = WithSettings(null, hostMemoryBudgetMb: 512, () => RunCommandService.BuildSpec(AgentAsks(PostureOf(AgentAutonomyLevel.Trusted)), workingDirectory: null));
+
+ spec.MaxMemoryMb.ShouldBe(512, "the same host budget that narrows the agent's own run narrows the command it asks for");
+ spec.MaxCpuPercent.ShouldBe(400);
+ }
+
+ // ── The argv a confining host execs ───────────────────────────────────────
+
+ [Theory]
+ [InlineData("workflow-node", false)] // the authored network, unchanged: the host network is shared
+ [InlineData("networked-agent", false)] // a caller with network passes its network through
+ [InlineData("network-off-agent", true)] // the bypass this closes: the caller's own severed network holds
+ [InlineData("allowlisted-agent", true)] // an allowlist bwrap cannot enforce outside a filtered namespace severs
+ public void Under_bubblewrap_the_command_chain_severs_exactly_the_commands_whose_caller_has_no_open_network(string lane, bool expectedSevered)
+ {
+ var spec = WithSettings(null, hostMemoryBudgetMb: null, () => RunCommandService.BuildSpec(RequestFor(lane), workingDirectory: null));
+
+ var argv = LocalProcessRunner.ChildCommand(new LocalProcessRunner.CommandIsolationContext(spec, null, null, Array.Empty(), Array.Empty()), FakeBwrap, prlimit: null);
+
+ argv[0].ShouldBe(FakeBwrap, "a confining host runs the command inside bubblewrap");
+ argv.Contains("--unshare-net").ShouldBe(expectedSevered, $"the {lane} command's chain must {(expectedSevered ? "" : "not ")}carry --unshare-net: [{string.Join(' ', argv)}]");
+ }
+
+ [Fact]
+ public void A_cgroup_host_caps_an_agent_called_command_at_its_callers_ceilings()
+ {
+ var spec = WithSettings(null, hostMemoryBudgetMb: null, () => RunCommandService.BuildSpec(RequestFor("network-off-agent"), workingDirectory: null));
+
+ // The plan the command launch builds from the spec on a host with a delegated cgroup-v2 root.
+ var plan = CgroupResourcePlan.Build("/sys/fs/cgroup/codespace", "command", spec.MaxMemoryMb, spec.MaxCpuPercent, maxPids: 0).ShouldNotBeNull("an agent's command is capped");
+
+ plan.Limits.Single(limit => limit.FileName == "memory.max").Value.ShouldBe("6442450944", "the calling Unleashed run's memory row");
+ plan.Limits.Single(limit => limit.FileName == "cpu.max").Value.ShouldBe("400000 100000", "and its cpu row");
+ }
+
+ [Fact]
+ public void A_cgroup_host_leaves_a_workflow_nodes_command_uncapped()
+ {
+ var spec = WithSettings(null, hostMemoryBudgetMb: null, () => RunCommandService.BuildSpec(RequestFor("workflow-node"), workingDirectory: null));
+
+ CgroupResourcePlan.Build("/sys/fs/cgroup/codespace", "command", spec.MaxMemoryMb, spec.MaxCpuPercent, maxPids: 0).ShouldBeNull("a workflow node's command asks for no cgroup at all — unchanged");
+ }
+
+ // ── What the agent is told when its command loses the network it asked for ─
+
+ [Theory]
+ [InlineData("workflow-node", null)] // no calling run: nothing narrowed it
+ [InlineData("networked-agent", null)] // granted as asked
+ [InlineData("network-off-agent", "off: the calling run (Unleashed) has no network — severed only where the sandbox confines")]
+ [InlineData("allowlisted-agent", "narrowed to the calling run's egress allowlist (registry.npmjs.org)")]
+ [InlineData("empty-allowlist-agent", "off: the calling run's egress allowlist names no host a command may reach — severed only where the sandbox confines")]
+ [InlineData("quiet-agent", null)] // a command that asked for no network lost nothing
+ public void The_command_says_when_its_callers_posture_took_the_network_it_asked_for(string lane, string? expected)
+ {
+ var notice = WithSettings(null, hostMemoryBudgetMb: null, () => RunCommandService.CallerNetworkNarrowing(RequestFor(lane)));
+
+ notice.ShouldBe(expected);
+ }
+
+ // ── Commands one run has running never hold more than one tier row between them ─
+
+ [Fact]
+ public async Task Two_overlapping_commands_from_one_run_run_one_after_the_other()
+ {
+ // Each command gets a cgroup leaf of its own, beside the agent's, carrying the run's whole tier row. Two
+ // commands a run starts at once would each get a full row; queued, they never hold more than one between them.
+ var runner = new OverlapRunner(expectedConcurrency: 2);
+ var service = new RunCommandService(null!, null!, new SingleRunnerRegistry(runner), null!, LocalDefault(), new CallerCommandLanes());
+ var runId = Guid.NewGuid();
+
+ await Task.WhenAll(service.RunAsync(AgentAsks(PostureOf(AgentAutonomyLevel.Trusted) with { RunId = runId }), CancellationToken.None), service.RunAsync(AgentAsks(PostureOf(AgentAutonomyLevel.Trusted) with { RunId = runId }), CancellationToken.None));
+
+ runner.Started.ShouldBe(2, "fixture check: both commands ran");
+ runner.MaxConcurrent.ShouldBe(1, "a run's two commands must never run at the same time");
+ }
+
+ [Fact]
+ public async Task Commands_from_two_runs_and_workflow_node_commands_are_not_queued_behind_each_other()
+ {
+ var runner = new OverlapRunner(expectedConcurrency: 3);
+ var service = new RunCommandService(null!, null!, new SingleRunnerRegistry(runner), null!, LocalDefault(), new CallerCommandLanes());
+
+ await Task.WhenAll(
+ service.RunAsync(AgentAsks(PostureOf(AgentAutonomyLevel.Trusted) with { RunId = Guid.NewGuid() }), CancellationToken.None),
+ service.RunAsync(AgentAsks(PostureOf(AgentAutonomyLevel.Trusted) with { RunId = Guid.NewGuid() }), CancellationToken.None),
+ service.RunAsync(new RunCommandRequest { Command = "true" }, CancellationToken.None));
+
+ runner.MaxConcurrent.ShouldBe(3, "the queue is per calling run: other runs' commands and a workflow node's are untouched");
+ }
+
+ // ── Helpers ───────────────────────────────────────────────────────────────
+
+ private static RunCommandRequest RequestFor(string lane) => lane switch
+ {
+ "workflow-node" => new RunCommandRequest { Command = "curl", AllowNetwork = true },
+ "empty-allowlist-agent" => AgentAsks(new AgentRunPosture { Autonomy = AgentAutonomyLevel.Trusted, Permissions = new AgentPermissions { Network = AgentNetworkAccess.On, Egress = AgentEgressPolicy.Allowlist, EgressAllowHosts = [] } }),
+ "quiet-agent" => AgentAsks(new AgentRunPosture { Autonomy = AgentAutonomyLevel.Unleashed, Permissions = new AgentPermissions { Network = AgentNetworkAccess.Off } }) with { AllowNetwork = false },
+ "networked-agent" => AgentAsks(PostureOf(AgentAutonomyLevel.Trusted)),
+ "network-off-agent" => AgentAsks(new AgentRunPosture { Autonomy = AgentAutonomyLevel.Unleashed, Permissions = new AgentPermissions { Network = AgentNetworkAccess.Off } }),
+ "allowlisted-agent" => AgentAsks(new AgentRunPosture { Autonomy = AgentAutonomyLevel.Trusted, Permissions = new AgentPermissions { Network = AgentNetworkAccess.On, Egress = AgentEgressPolicy.Allowlist, EgressAllowHosts = ["registry.npmjs.org"] } }),
+ _ => throw new ArgumentOutOfRangeException(nameof(lane), lane, null),
+ };
+
+ /// A command an agent asked for through its tool fabric, wanting the network.
+ private static RunCommandRequest AgentAsks(AgentRunPosture caller) => new() { Command = "curl", AllowNetwork = true, CallerPosture = caller };
+
+ /// A run launched at with the permissions that tier derives — what every producer stamps when no per-field override applies.
+ private static AgentRunPosture PostureOf(AgentAutonomyLevel tier) => new() { Autonomy = tier, Permissions = AgentAutonomyPolicy.Derive(tier) };
+
+ private static AgentDefaultRunnerSetting LocalDefault() =>
+ new(new ConfigurationBuilder().AddInMemoryCollection(new Dictionary { [AgentDefaultRunnerSetting.ConfigurationKey] = "local" }).Build());
+
+ private sealed class SingleRunnerRegistry(ISandboxRunner runner) : ISandboxRunnerRegistry
+ {
+ public IReadOnlyList All => [runner];
+ public ISandboxRunner Resolve(string kind) => runner;
+ }
+
+ ///
+ /// Records how many commands run at once. Each one waits until are running
+ /// together or a short grace elapses, so commands that CAN overlap always do, and the recorded maximum is what the
+ /// service allowed rather than an accident of scheduling.
+ ///
+ private sealed class OverlapRunner(int expectedConcurrency) : ISandboxRunner
+ {
+ private readonly object _gate = new();
+ private readonly TaskCompletionSource _all = new(TaskCreationOptions.RunContinuationsAsynchronously);
+ private int _running;
+
+ public int Started { get; private set; }
+ public int MaxConcurrent { get; private set; }
+ public string Kind => "local";
+
+ public async Task RunAsync(SandboxSpec spec, CancellationToken cancellationToken)
+ {
+ lock (_gate)
+ {
+ Started++;
+ _running++;
+ MaxConcurrent = Math.Max(MaxConcurrent, _running);
+ if (_running >= expectedConcurrency) _all.TrySetResult();
+ }
+
+ await Task.WhenAny(_all.Task, Task.Delay(TimeSpan.FromSeconds(1), cancellationToken));
+
+ lock (_gate) _running--;
+
+ return new SandboxResult { Status = SandboxStatus.Success, ExitCode = 0, Stdout = "", Stderr = "" };
+ }
+ }
+
+ /// Bind the deployment ceiling and host memory budget through the REAL configuration read for one projection; null is an unconfigured deployment.
+ private static T WithSettings(string? deployment, int? hostMemoryBudgetMb, Func act)
+ {
+ var configuration = new ConfigurationBuilder().AddInMemoryCollection(new Dictionary
+ {
+ [RuntimeSettings.MaxAutonomyKey] = deployment,
+ [RuntimeSettings.AgentMemoryCeilingMbKey] = hostMemoryBudgetMb?.ToString(),
+ }).Build();
+
+ using (RuntimeSettings.Override(RuntimeSettings.Read(configuration))) return act();
+ }
+}
diff --git a/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunCommandNodeTests.cs b/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunCommandNodeTests.cs
index 42bad643b..5cc6dedb9 100644
--- a/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunCommandNodeTests.cs
+++ b/backend/tests/CodeSpace.UnitTests/Workflows/AgentRunCommandNodeTests.cs
@@ -87,6 +87,36 @@ public async Task A_timeout_surfaces_as_status_TimedOut()
result.Outputs["exitCode"].GetInt32().ShouldBe(-1);
}
+ [Fact]
+ public async Task Carries_the_calling_runs_posture_into_the_request_and_a_workflow_context_carries_none()
+ {
+ var posture = new AgentRunPosture { Autonomy = AgentAutonomyLevel.Standard, Permissions = new AgentPermissions { Network = AgentNetworkAccess.Off } };
+ var asTool = new StubRunCommandService();
+ var asNode = new StubRunCommandService();
+
+ await new AgentRunCommandNode(asTool, new FakeArtifactStore()).RunAsync(Context() with { CallerPosture = posture }, CancellationToken.None);
+ await new AgentRunCommandNode(asNode, new FakeArtifactStore()).RunAsync(Context(), CancellationToken.None);
+
+ asTool.Request!.CallerPosture.ShouldBeSameAs(posture, "a command an agent asks for carries that agent's run posture to the service that builds its sandbox");
+ asNode.Request!.CallerPosture.ShouldBeNull("a workflow node's command has no calling run — its authored posture stands");
+ }
+
+ [Fact]
+ public async Task Tells_the_agent_when_its_runs_posture_took_the_network_its_command_asked_for()
+ {
+ // The agent (and whoever approved the call) asked for "network": true; a network-off run's command runs
+ // severed. Without saying so, the only trace is the command's own connection error.
+ var posture = new AgentRunPosture { Autonomy = AgentAutonomyLevel.Standard, Permissions = new AgentPermissions { Network = AgentNetworkAccess.Off } };
+ var inputs = new Dictionary { ["command"] = JsonSerializer.SerializeToElement("npm"), ["network"] = JsonSerializer.SerializeToElement(true) };
+
+ var asTool = await new AgentRunCommandNode(new StubRunCommandService(), new FakeArtifactStore()).RunAsync(ContextFrom(inputs) with { CallerPosture = posture }, CancellationToken.None);
+ var asNode = await new AgentRunCommandNode(new StubRunCommandService(), new FakeArtifactStore()).RunAsync(ContextFrom(inputs), CancellationToken.None);
+
+ asTool.Outputs["networkNarrowed"].GetString().ShouldBe(RunCommandService.CallerNetworkNarrowing(new RunCommandRequest { Command = "npm", AllowNetwork = true, CallerPosture = posture }));
+ asTool.Outputs["networkNarrowed"].GetString().ShouldNotBeNull().ShouldStartWith("off");
+ asNode.Outputs.ContainsKey("networkNarrowed").ShouldBeFalse("a workflow node's command is never narrowed by a caller — its outputs are unchanged");
+ }
+
[Fact]
public async Task Runs_ephemerally_when_no_repository_is_given()
{