Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 3 additions & 1 deletion crates/astra-cli/src/cli/stream/mcp_result_tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -94,7 +94,9 @@ async fn callback_through_pipeline(
80,
false,
);
host.on_server_tool_surface_admission(&tool).unwrap();
host.executor
.accept_server_tool_surface_admission(&tool)
.unwrap();
let results = host
.execute_tools_batch(vec![ToolBatchRequest {
session_id: "mcp-session".into(),
Expand Down
65 changes: 58 additions & 7 deletions crates/astra-cli/src/cli/stream/stream_render.rs
Original file line number Diff line number Diff line change
Expand Up @@ -4730,8 +4730,16 @@ impl SseStreamHost for CliSseStreamHost<'_> {
.await;
}

fn on_server_tool_surface_admission(&mut self, tool: &str) -> Result<(), String> {
self.executor.accept_server_tool_surface_admission(tool)
fn on_server_tool_surface_admission(
&mut self,
request: &ToolBatchRequest,
) -> Result<(), String> {
self.tool_result_identities.insert(
request.request_id.clone(),
ToolResultIdentity::from_batch_request(request),
);
self.executor
.accept_server_tool_surface_admission(&request.tool)
}

async fn on_render_effects(&mut self, effects: Vec<SseRenderEffect>) {
Expand Down Expand Up @@ -4837,7 +4845,12 @@ impl SseStreamHost for CliSseStreamHost<'_> {
}

fn on_tool_result(&mut self, result: &EdgeToolExecResult) {
self.sync_incremental_tool_result(result);
// The foreground snapshot belongs to one run, not every callback
// transported through its stream. Child results retain their own
// durable journal and AgentLive publication.
if self.callback_tool_belongs_to_foreground(&result.request_id) {
self.sync_incremental_tool_result(result);
}
}

async fn execute_tool(&mut self, request: &ToolBatchRequest) -> EdgeToolExecResult {
Expand Down Expand Up @@ -11995,7 +12008,7 @@ mod tests {
cache: &'a mut EdgeToolCache,
cancel: Option<&'a tokio_util::sync::CancellationToken>,
) -> CliSseStreamHost<'a> {
let mut host = CliSseStreamHost::from_edge_ctx(
let host = CliSseStreamHost::from_edge_ctx(
EdgeSseContext {
api: &self.api,
token: "tok",
Expand All @@ -12021,7 +12034,8 @@ mod tests {
80,
false,
);
host.on_server_tool_surface_admission("memory")
host.executor
.accept_server_tool_surface_admission("memory")
.expect("existing cloud memory binding accepts server admission");
host
}
Expand Down Expand Up @@ -12107,7 +12121,6 @@ mod tests {
let workspace = tempdir().unwrap();
let mut cache = EdgeToolCache::new(8);
let mut host = fixture.host(workspace.path(), &mut cache, None);
host.on_server_tool_surface_admission("bash").unwrap();
let mut requests = Vec::new();
let mut settled = Vec::new();
for (id, args) in [
Expand All @@ -12119,6 +12132,7 @@ mod tests {
] {
let mut request = parallel_batch_request("budget", id, "bash", args.clone());
request.command_timeout_cap_ms = Some(200);
host.on_server_tool_surface_admission(&request).unwrap();
let results = tokio::time::timeout(
std::time::Duration::from_secs(5),
host.execute_tools_batch(vec![request.clone()]),
Expand Down Expand Up @@ -12437,6 +12451,8 @@ mod tests {
let executor = std::sync::Arc::new(crate::edge_tools::ToolExecutor::new(&project));

let (tx, mut rx) = tokio::sync::mpsc::channel(32);
let incremental =
std::sync::Arc::new(astra_turn_core::turn_event_sink::IncrementalTurnState::default());
let mut tool_cache = EdgeToolCache::new(8);
let mut pm =
crate::cli::permission_manager::PermissionManager::with_project(false, &project);
Expand All @@ -12458,7 +12474,7 @@ mod tests {
skill_continuation: false,
turn_rollback_on_failure: false,
tool_cache: &mut tool_cache,
incremental_state: None,
incremental_state: Some(incremental.clone()),
request_session_execution_lease: None,
},
80,
Expand Down Expand Up @@ -12500,6 +12516,16 @@ mod tests {
.await;

assert_eq!(results.len(), 2);
for result in &results {
host.on_tool_result(result);
}
assert_eq!(incremental.snapshot().tool_call_records.len(), 1);
assert_eq!(
incremental.snapshot().tool_call_records[0]
.tool_call_id
.as_deref(),
Some("pf-1")
);
assert!(results.iter().all(|result| result.status == "completed"));
assert!(results[0].output.contains("one"), "{}", results[0].output);
assert!(results[1].output.contains("two"), "{}", results[1].output);
Expand Down Expand Up @@ -12542,6 +12568,7 @@ mod tests {
}])
.await;
assert_eq!(results[0].status, "completed");
host.on_tool_result(&results[0]);
if request_id == "serial-child" {
let fields = results[0]
.tool_result_fields
Expand Down Expand Up @@ -12584,6 +12611,8 @@ mod tests {
}])
.await;
assert_eq!(rejected[0].status, "failed");
host.on_tool_result(&rejected[0]);
assert_eq!(incremental.snapshot().tool_call_records.len(), 2);
assert!(
rx.try_recv().is_err(),
"child synthetic rejection stays out of root UI"
Expand Down Expand Up @@ -12968,6 +12997,28 @@ mod tests {
}),
..Default::default()
});
// A rejected server request still has an exact owner before any
// execution starts; unknown callbacks must not enter this snapshot.
for (id, run) in [("child-rejected", "child-live"), ("tool-1", "run-live")] {
let mut request =
parallel_batch_request(run, id, "unavailable_tool", serde_json::json!({}));
request.run_id = run.into();
request.session_id = "sess-live".into();
assert!(host.on_server_tool_surface_admission(&request).is_err());
host.on_tool_result(&EdgeToolExecResult {
execution_completion: None,
request_id: id.into(),
tool: request.tool,
args: request.args,
output: "surface rejected".into(),
tool_result_fields: None,
status: "failed".into(),
duration_ms: 0,
});
}
assert_eq!(incremental_state.snapshot().tool_call_records.len(), 1);
incremental_state.replace_tool_records(Vec::new());
incremental_state.replace_tools_used(Vec::new());
host.on_tool_result(&EdgeToolExecResult {
execution_completion: None,
request_id: "tool-1".to_string(),
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -106,6 +106,15 @@ pass. Both final answers must be exactly `42`. Report all physical calls,
auxiliary judgments, token/cache coverage, tool attempts, and elapsed phases,
not just whether the answer is right. Use repeated runs for latency claims.

`flash_child_compact_json` adds a non-arithmetic, compact JSON contract for
both the actual GLM child result and the parent's answer. Together with the
integer-only cases, it checks whether requested formats survive shared persona
and summary guidance without runtime output rewriting. Keep failed samples;
one later pass does not demonstrate reliable format compliance. The fanout
case also reports a soft primary-round bound of four: automatic delivery should
avoid re-fetching sufficient observed results, while inspection, missing or
truncated output, pagination and recovery remain valid reasons to read results.

`flash_semantic_model_reference_glm` isolates candidate-aware semantic
selection from live websites. The user says `5.2glm` without tool syntax;
the case requires the authorized `glm-5.2` child to make a real provider call
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,86 @@
name: flash_child_compact_json
description: |
Natural named-model delegation with a compact JSON answer contract. Verify
real child inference, exact result adoption, and the parent's final answer.
This complements arithmetic cases without a task-specific runtime formatter.
prompt: |
用 glm5.2 子代理判断“所有鸟类都会飞”是否成立。让它只回复紧凑 JSON,
唯一字段 valid,值为布尔值,不要解释或 Markdown。你收到并采用它的结果后,
也只回复该 JSON。不要访问文件或网络。
debug_log: true
capability: delegation
timeout_seconds: 120
criteria:
- type: exit_code
code: 0
- type: session_event_count
event_type: agent_spawned
min: 1
max: 1
- type: session_event_count
event_type: agent_spawned
min: 1
max: 1
json_match:
path: /metadata/model_configuration/prepared_selection/model_name
equals: glm-5.2
- type: session_event_count
event_type: LlmRoundCompleted
min: 1
json_match:
path: /payload/model
equals: glm-5.2
same_run_as:
event_type: agent_spawned
run_id_path: /metadata/run_id
- type: session_child_result_adopted
expected_result: '{"valid":false}'
spawn_match:
path: /metadata/model_configuration/prepared_selection/model_name
equals: glm-5.2
allow_get_result: true
- type: text_equals
expected: '{"valid":false}'
- type: turn_rounds_between
min: 2
max: 3
- type: journal_tool_call_count
name: model_catalog
min: 0
max: 0
- type: journal_tool_call_count
name: tool_search
min: 0
max: 0
- type: journal_tool_call_count
name: bash
min: 0
max: 0
- type: journal_tool_call_count
name: read_file
min: 0
max: 0
- type: journal_tool_call_count
name: list_dir
min: 0
max: 0
- type: journal_tool_call_count
name: grep
min: 0
max: 0
- type: journal_tool_call_count
name: web_fetch
min: 0
max: 0
- type: journal_tool_call_count
name: web_search
min: 0
max: 0
- type: journal_tool_call_count
name: write_file
min: 0
max: 0
- type: journal_tool_call_count
name: str_replace
min: 0
max: 0
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,9 @@ timeout_seconds: 240
criteria:
- type: exit_code
code: 0
- type: turn_rounds_between
min: 2
max: 4
- type: journal_tool_call_count
name: agent_fanout
min: 1
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,87 @@
name: flash_scoped_child_and_parent
description: |
Natural named-model delegation with independent primary work. Verify actual
child inference and result adoption, not model self-reporting. Unnecessary
discovery, filesystem/network access and excessive primary rounds fail.
prompt: |
请让 glm5.2 子代理独立计算 17×19,最终只回复整数。它运行时你独立计算
9×11,最后汇总两份结果。不要访问文件或网络。
debug_log: true
capability: delegation
timeout_seconds: 120
criteria:
- type: exit_code
code: 0
- type: session_event_count
event_type: agent_spawned
min: 1
max: 1
- type: session_event_count
event_type: agent_spawned
min: 1
max: 1
json_match:
path: /metadata/model_configuration/prepared_selection/model_name
equals: glm-5.2
- type: session_event_count
event_type: LlmRoundCompleted
min: 1
json_match:
path: /payload/model
equals: glm-5.2
same_run_as:
event_type: agent_spawned
run_id_path: /metadata/run_id
- type: session_child_result_adopted
expected_result: "323"
spawn_match:
path: /metadata/model_configuration/prepared_selection/model_name
equals: glm-5.2
allow_get_result: true
- type: text_contains
needle: "323"
- type: text_contains
needle: "99"
- type: turn_rounds_between
min: 2
max: 3
- type: journal_tool_call_count
name: model_catalog
min: 0
max: 0
- type: journal_tool_call_count
name: tool_search
min: 0
max: 0
- type: journal_tool_call_count
name: bash
min: 0
max: 0
- type: journal_tool_call_count
name: read_file
min: 0
max: 0
- type: journal_tool_call_count
name: list_dir
min: 0
max: 0
- type: journal_tool_call_count
name: grep
min: 0
max: 0
- type: journal_tool_call_count
name: web_fetch
min: 0
max: 0
- type: journal_tool_call_count
name: web_search
min: 0
max: 0
- type: journal_tool_call_count
name: write_file
min: 0
max: 0
- type: journal_tool_call_count
name: str_replace
min: 0
max: 0
20 changes: 15 additions & 5 deletions crates/astra-test-harness/src/criteria.rs
Original file line number Diff line number Diff line change
Expand Up @@ -6700,12 +6700,22 @@ mod tests {
));
let mut outcome = outcome_with_tools(&[]);
outcome.text = "FLASH-SLOT-ALPHA FLASH-SLOT-BETA 2/2".into();
let results = evaluate_deterministic_with_session(
&case.criteria,
&outcome,
Some(&mk_session(&adopted)),
);
outcome.turn_rounds = 4;
let session = mk_session(&adopted);
let results = evaluate_deterministic_with_session(&case.criteria, &outcome, Some(&session));
assert!(results.iter().all(|result| result.passed), "{results:?}");
for rounds in [0, 5] {
outcome.turn_rounds = rounds;
let results =
evaluate_deterministic_with_session(&case.criteria, &outcome, Some(&session));
assert!(
results.iter().any(|result| {
matches!(result.criterion, Criterion::TurnRoundsBetween { .. })
&& !result.passed
}),
"invalid or excessive rounds must fail: {results:?}"
);
}
}

#[test]
Expand Down
Loading
Loading