@setoelkahfi / sigit / commits / 96c0879

Add task tool: delegate research to a read-only subagent

Every file the model reads stays in the main conversation forever, which fills small context windows fast. The new task tool hands a research prompt to a fresh conversation that only gets the read-only tools (read_file, list_directory, search_files, glob, read_website) and returns its final answer, capped at 8000 chars. The main history gains one tool result instead of the whole transcript. tools.rs cannot construct a backend, so each surface registers a subagent factory at startup: with an OpenAI-compatible provider it builds a fresh backend with the same endpoint and model; on-device it returns None and the tool tells the model to research directly (onde has a single shared history, so a second context needs onde support first). The tool is offered only when a factory is available, the same conditional pattern the skill tool uses. task is classified read-only for permissions: the nested loop hard-gates execution to the read-only set, so a hallucinated call to anything else returns an error instead of running. That also keeps delegated research usable in plan mode.

paydii committed Jul 4, 2026 at 22:26 UTC 96c0879965f27d99c05e88f4504402438e9698a8
4 files changed +512 -1
src/chat.rs
+6
@@ -1600,6 +1600,12 @@ mod tui {
1600 });
1601 }
1602
1603 + // Delegated research (`task`) is offered only when a subagent backend
1604 + // can actually be built — same conditional pattern as `skill` above.
1605 + if crate::tools::subagent_available() {
1606 + specs.push(crate::tools::task_tool_spec());
1607 + }
1608 +
1609 // Tools discovered from configured MCP servers (incl. the official one).
1610 specs.extend(crate::mcp::tool_specs());
1611
src/main.rs
+35
@@ -305,12 +305,37 @@ fn agent_tools_as_specs() -> Vec<ToolSpec> {
305 });
306 }
307
308 + // Delegated research (`task`) is offered only when a subagent backend can
309 + // actually be built — same conditional pattern as the `skill` tool above.
310 + if tools::subagent_available() {
311 + specs.push(tools::task_tool_spec());
312 + }
313 +
314 // Tools discovered from configured MCP servers (incl. the official one).
315 specs.extend(mcp::tool_specs());
316
317 specs
318 }
319
320 +/// Register the `task` tool's subagent factory for an OpenAI-compatible
321 +/// provider: each subagent run gets a FRESH `OpenAiBackend` against the same
322 +/// endpoint (its own conversation history), seeded with the subagent system
323 +/// prompt. The provider config is captured by clone. Called once at startup by
324 +/// whichever surface resolved the provider; on-device paths register a factory
325 +/// that returns `None` instead (onde has a single shared history, so a second
326 +/// concurrent context is not possible yet).
327 +fn register_subagent_factory_for(cfg: &provider::ProviderConfig) {
328 + let cfg = cfg.clone();
329 + tools::set_subagent_factory(Box::new(move || {
330 + Some(Arc::new(OpenAiBackend::new(
331 + cfg.base_url.clone(),
332 + cfg.api_key.clone(),
333 + cfg.model.clone(),
334 + Some(tools::SUBAGENT_SYSTEM_PROMPT.to_string()),
335 + )) as Arc<dyn InferenceBackend>)
336 + }));
337 +}
338 +
339 fn initialize_meta() -> Meta {
340 let startup_selection = setup::startup_model_selection();
341
@@ -2700,6 +2725,7 @@ async fn run_interactive(tty: std::fs::File, mut cleanup_tty: std::fs::File) ->
2725 // No local model to load; the endpoint is ready immediately.
2726 let _ = load_tx.send(Ok(()));
2727 let label = provider.display_name.clone();
2728 + register_subagent_factory_for(&provider);
2729 let backend = Arc::new(OpenAiBackend::new(
2730 provider.base_url,
2731 provider.api_key,
@@ -2729,6 +2755,7 @@ async fn run_interactive(tty: std::fs::File, mut cleanup_tty: std::fs::File) ->
2755 );
2756 let _ = load_tx.send(Ok(()));
2757 let label = provider.display_name.clone();
2758 + register_subagent_factory_for(&provider);
2759 let backend = Arc::new(OpenAiBackend::new(
2760 provider.base_url,
2761 provider.api_key,
@@ -2745,6 +2772,9 @@ async fn run_interactive(tty: std::fs::File, mut cleanup_tty: std::fs::File) ->
2772 );
2773 }
2774 let _ = load_tx.send(Ok(()));
2775 + // On-device inference has a single shared history; no
2776 + // subagent context is possible yet.
2777 + tools::set_subagent_factory(Box::new(|| None));
2778 let backend = Arc::new(LocalBackend::new(Arc::clone(&engine)))
2779 as Arc<dyn InferenceBackend>;
2780 (backend, startup_model_name)
@@ -2867,6 +2897,7 @@ async fn run_acp_server() -> anyhow::Result<()> {
2897 cfg.model,
2898 cfg.base_url
2899 );
2900 + register_subagent_factory_for(&cfg);
2901 let override_backend: Arc<dyn InferenceBackend> = Arc::new(OpenAiBackend::new(
2902 cfg.base_url,
2903 cfg.api_key,
@@ -2874,6 +2905,10 @@ async fn run_acp_server() -> anyhow::Result<()> {
2905 Some(system_prompt_for_model(true).to_string()),
2906 ));
2907 *state.backend.lock().await = override_backend;
2908 + } else {
2909 + // On-device inference has a single shared conversation history, so a
2910 + // second concurrent subagent context is not possible yet.
2911 + tools::set_subagent_factory(Box::new(|| None));
2912 }
2913
2914 let stdin = tokio::io::stdin().compat();
src/permissions.rs
+5 -1
@@ -64,10 +64,13 @@ pub enum Decision {
64
65 /// Classify a tool by name. Unknown names and MCP tools are mutating: the
66 /// conservative default for anything whose side effects we can't see.
67 +/// `task` is read-only because the subagent it launches is restricted to the
68 +/// read-only toolset (see `SUBAGENT_TOOL_NAMES` in `tools.rs`), so delegated
69 +/// research stays available in plan mode.
70 pub fn classify(tool_name: &str) -> ToolRisk {
71 match tool_name {
72 "read_file" | "list_directory" | "search_files" | "glob" | "read_website"
70 - | "write_todos" | "skill" => ToolRisk::ReadOnly,
73 + | "write_todos" | "skill" | "task" => ToolRisk::ReadOnly,
74 _ => ToolRisk::Mutating,
75 }
76 }
@@ -247,6 +250,7 @@ mod tests {
250 "read_website",
251 "write_todos",
252 "skill",
253 + "task",
254 ] {
255 assert_eq!(classify(tool), ToolRisk::ReadOnly, "{tool}");
256 assert_eq!(decision_for("t-ro", tool), Decision::Allow, "{tool}");
src/tools.rs
+466
@@ -5,6 +5,9 @@ use serde_json::{Value, json};
5 use std::fs;
6 use std::path::{Path, PathBuf};
7 use std::process::Command;
8 +use std::sync::{Arc, OnceLock};
9 +
10 +use crate::backend::{InferenceBackend, ToolResult, ToolSpec};
11
12 const WEBSITE_READ_CHAR_LIMIT: usize = 20_000;
13 const WEBSITE_READ_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(20);
@@ -400,6 +403,7 @@ pub async fn execute_tool(name: &str, arguments: &str) -> String {
403 "delete_file" => exec_delete_file(arguments),
404 "run_command" => exec_run_command(arguments),
405 "skill" => crate::skills::activate_skill(arguments),
406 + TASK_TOOL_NAME => exec_task(arguments).await,
407 // Tools discovered from MCP servers are namespaced `mcp__<server>__<tool>`
408 // and forwarded to the owning server.
409 _ if crate::mcp::is_mcp_tool(name) => crate::mcp::call_tool(name, arguments).await,
@@ -407,6 +411,224 @@ pub async fn execute_tool(name: &str, arguments: &str) -> String {
411 }
412 }
413
414 +// ── task (subagent) ──────────────────────────────────────────────────────────
415 +//
416 +// The `task` tool delegates a self-contained research question to a *nested*
417 +// agent loop running in a fresh conversation, so the main thread receives only
418 +// the final answer instead of every intermediate file read. The subagent's
419 +// toolset is strictly read-only and never includes `task` itself (no
420 +// recursion), so a delegated agent can research but not mutate state.
421 +//
422 +// This module is backend-agnostic and cannot construct a backend, so the
423 +// surface that knows the active provider (`run_acp_server` / `run_interactive`
424 +// in `main.rs`) registers a factory at startup. The factory returns `None`
425 +// when inference runs on-device: onde has a single shared conversation
426 +// history, so a second concurrent context is not possible yet.
427 +
428 +pub const TASK_TOOL_NAME: &str = "task";
429 +
430 +/// The tool names a subagent may call, filtered from [`all_tools`].
431 +const SUBAGENT_TOOL_NAMES: &[&str] = &[
432 + "read_file",
433 + "list_directory",
434 + "search_files",
435 + "glob",
436 + "read_website",
437 +];
438 +
439 +/// System prompt seeding every subagent conversation.
440 +pub const SUBAGENT_SYSTEM_PROMPT: &str = "You are a focused research subagent. \
441 +You are given one self-contained task by a calling agent. Investigate it using \
442 +the read-only tools available to you (read_file, list_directory, search_files, \
443 +glob, read_website) and answer it thoroughly but concisely. You cannot modify \
444 +files or run commands. Your final message is returned verbatim to the caller, \
445 +so make it a complete, self-contained answer — include the concrete facts, \
446 +paths, and code excerpts the caller needs, and nothing else.";
447 +
448 +/// Rounds of tool calls a subagent may use before it is forced to answer.
449 +const SUBAGENT_MAX_TOOL_ROUNDS: usize = 8;
450 +/// Cap on the answer text returned to the caller.
451 +const SUBAGENT_RESULT_CHAR_LIMIT: usize = 8_000;
452 +
453 +/// Returned when `task` is called but no subagent backend can be built.
454 +const SUBAGENT_UNAVAILABLE: &str = "The task tool is not available on-device \
455 +yet: on-device inference has a single conversation context. Do the research \
456 +yourself with the read-only tools.";
457 +
458 +/// Builds a fresh backend for one subagent run, or `None` when the active
459 +/// inference cannot host a second conversation (on-device).
460 +pub type SubagentFactory = Box<dyn Fn() -> Option<Arc<dyn InferenceBackend>> + Send + Sync>;
461 +
462 +static SUBAGENT_FACTORY: OnceLock<SubagentFactory> = OnceLock::new();
463 +
464 +/// Register the process-wide subagent factory. Called once at startup by the
465 +/// surface that resolved the inference provider; later calls are ignored.
466 +pub fn set_subagent_factory(factory: SubagentFactory) {
467 + let _ = SUBAGENT_FACTORY.set(factory);
468 +}
469 +
470 +/// Whether a `task` call could run right now: a factory is registered and it
471 +/// can build a backend. The spec builders (`agent_tools_as_specs` /
472 +/// `build_tool_specs`) use this to offer the tool only when it works, the same
473 +/// conditional pattern as the `skill` tool.
474 +pub fn subagent_available() -> bool {
475 + SUBAGENT_FACTORY
476 + .get()
477 + .is_some_and(|factory| factory().is_some())
478 +}
479 +
480 +/// The read-only toolset offered to a subagent, filtered from [`all_tools`] by
481 +/// name. Never contains `task` (recursion) or any mutating tool.
482 +pub fn subagent_tool_specs() -> Vec<ToolSpec> {
483 + all_tools()
484 + .into_iter()
485 + .filter(|tool| SUBAGENT_TOOL_NAMES.contains(&tool.name))
486 + .map(|tool| ToolSpec {
487 + name: tool.name.to_string(),
488 + description: tool.description.to_string(),
489 + parameters_schema: tool.parameters_schema.to_string(),
490 + })
491 + .collect()
492 +}
493 +
494 +/// Spec for the `task` tool. Lives in the `*_as_specs`/`build_tool_specs`
495 +/// layer (like `skill` and MCP tools), not in [`all_tools`], because it is
496 +/// only offered when [`subagent_available`] is true.
497 +pub fn task_tool_spec() -> ToolSpec {
498 + ToolSpec {
499 + name: TASK_TOOL_NAME.to_string(),
500 + description: "Delegate a self-contained research task to a subagent that \
501 + runs in a fresh conversation and returns only its final answer. The \
502 + subagent can read files, list directories, search, glob, and read \
503 + websites, but cannot modify anything. Use this for exploratory \
504 + questions whose intermediate file reads would otherwise clutter this \
505 + conversation (e.g. \"where is X implemented and how does it work?\"). \
506 + The subagent cannot see this conversation, so the prompt must be \
507 + fully self-contained: include absolute paths, symbol names, and \
508 + exactly what the answer should contain."
509 + .to_string(),
510 + parameters_schema: json!({
511 + "type": "object",
512 + "properties": {
513 + "description": {
514 + "type": "string",
515 + "description": "A short (3-5 word) summary of the task, for progress display."
516 + },
517 + "prompt": {
518 + "type": "string",
519 + "description": "The full task for the subagent. Must be self-contained: the subagent sees nothing of this conversation."
520 + }
521 + },
522 + "required": ["description", "prompt"],
523 + "additionalProperties": false
524 + })
525 + .to_string(),
526 + }
527 +}
528 +
529 +/// The `task` tool entry point used by [`execute_tool`].
530 +async fn exec_task(arguments: &str) -> String {
531 + exec_task_with(arguments, SUBAGENT_FACTORY.get()).await
532 +}
533 +
534 +/// Core of the `task` tool, parameterized on the factory so tests can exercise
535 +/// the unavailable path without touching the process-global `OnceLock`.
536 +async fn exec_task_with(arguments: &str, factory: Option<&SubagentFactory>) -> String {
537 + let args: Value = match serde_json::from_str(arguments) {
538 + Ok(v) => v,
539 + Err(err) => return format!("Error: failed to parse arguments: {err}"),
540 + };
541 +
542 + let prompt = match args.get("prompt").and_then(Value::as_str) {
543 + Some(p) if !p.trim().is_empty() => p,
544 + _ => return "Error: missing required parameter \"prompt\"".to_string(),
545 + };
546 + let description = args
547 + .get("description")
548 + .and_then(Value::as_str)
549 + .unwrap_or("(no description)");
550 +
551 + let Some(backend) = factory.and_then(|build| build()) else {
552 + return SUBAGENT_UNAVAILABLE.to_string();
553 + };
554 +
555 + log::info!("task: running subagent — {description}");
556 + run_subagent(backend.as_ref(), prompt).await
557 +}
558 +
559 +/// The nested agent loop: a fresh conversation, read-only tools, a round cap,
560 +/// and only the final text returned. Mirrors the main loops in `main.rs` /
561 +/// `chat.rs`: offer tools each round, and pass `tools = None` on the last
562 +/// round to force a text answer.
563 +async fn run_subagent(backend: &dyn InferenceBackend, prompt: &str) -> String {
564 + let specs = subagent_tool_specs();
565 +
566 + let mut result = match backend.send_message_with_tools(prompt, &specs, None).await {
567 + Ok(r) => r,
568 + Err(err) => return format!("Error: subagent inference failed: {err}"),
569 + };
570 +
571 + let mut round = 0;
572 + while !result.tool_calls.is_empty() && round < SUBAGENT_MAX_TOOL_ROUNDS {
573 + round += 1;
574 + log::info!(
575 + "task: subagent tool round {round} — {} call(s)",
576 + result.tool_calls.len()
577 + );
578 +
579 + let mut tool_results = Vec::with_capacity(result.tool_calls.len());
580 + for call in &result.tool_calls {
581 + // Hard gate, not just advertisement: even if the model asks for a
582 + // tool outside the offered set, only read-only tools execute here.
583 + let content = if SUBAGENT_TOOL_NAMES.contains(&call.name.as_str()) {
584 + // Boxed to break the async cycle: execute_tool → task →
585 + // run_subagent → execute_tool.
586 + Box::pin(execute_tool(&call.name, &call.arguments)).await
587 + } else {
588 + format!(
589 + "Error: `{}` is not available to a subagent. Only these \
590 + read-only tools are: {}.",
591 + call.name,
592 + SUBAGENT_TOOL_NAMES.join(", ")
593 + )
594 + };
595 + tool_results.push(ToolResult {
596 + tool_call_id: call.id.clone(),
597 + content,
598 + });
599 + }
600 +
601 + // On the last round, offer no tools so the model must produce text.
602 + let next_tools = if round < SUBAGENT_MAX_TOOL_ROUNDS {
603 + Some(specs.as_slice())
604 + } else {
605 + None
606 + };
607 + result = match backend
608 + .send_tool_results(tool_results, next_tools, None)
609 + .await
610 + {
611 + Ok(r) => r,
612 + Err(err) => return format!("Error: subagent inference failed: {err}"),
613 + };
614 + }
615 +
616 + let text = result.text.trim();
617 + if text.is_empty() {
618 + return "The subagent finished without a text answer.".to_string();
619 + }
620 +
621 + let total = text.chars().count();
622 + if total > SUBAGENT_RESULT_CHAR_LIMIT {
623 + let truncated: String = text.chars().take(SUBAGENT_RESULT_CHAR_LIMIT).collect();
624 + return format!(
625 + "{truncated}\n\n--- truncated (showing {SUBAGENT_RESULT_CHAR_LIMIT} of {total} \
626 + characters of the subagent's answer) ---"
627 + );
628 + }
629 + text.to_string()
630 +}
631 +
632 fn absolute_path(path: &Path) -> PathBuf {
633 if path.is_absolute() {
634 path.to_path_buf()
@@ -2201,4 +2423,248 @@ mod tests {
2423 let result = exec_run_command(&args);
2424 assert!(result.contains("err"), "got: {result}");
2425 }
2426 +
2427 + // ── task (subagent) tests ────────────────────────────────────────────
2428 +
2429 + #[test]
2430 + fn test_subagent_toolset_is_read_only() {
2431 + let specs = subagent_tool_specs();
2432 + let mut names: Vec<&str> = specs.iter().map(|spec| spec.name.as_str()).collect();
2433 + names.sort_unstable();
2434 + assert_eq!(
2435 + names,
2436 + [
2437 + "glob",
2438 + "list_directory",
2439 + "read_file",
2440 + "read_website",
2441 + "search_files"
2442 + ]
2443 + );
2444 +
2445 + // No recursion and no mutating tools, ever.
2446 + assert!(!names.contains(&TASK_TOOL_NAME));
2447 + for banned in [
2448 + "create_file",
2449 + "create_directory",
2450 + "edit_file",
2451 + "multi_edit",
2452 + "delete_file",
2453 + "run_command",
2454 + "write_todos",
2455 + "remember",
2456 + ] {
2457 + assert!(!names.contains(&banned), "{banned} leaked into subagent");
2458 + }
2459 +
2460 + // Every spec came from `all_tools()` (schemas intact).
2461 + for spec in &specs {
2462 + assert!(
2463 + serde_json::from_str::<Value>(&spec.parameters_schema)
2464 + .unwrap()
2465 + .is_object()
2466 + );
2467 + }
2468 + }
2469 +
2470 + #[tokio::test]
2471 + async fn test_task_reports_unavailable_without_factory() {
2472 + let args = serde_json::json!({
2473 + "description": "look around",
2474 + "prompt": "What is in the current directory?"
2475 + })
2476 + .to_string();
2477 + // `None` is exactly what `exec_task` passes before any surface has
2478 + // registered the process-global factory.
2479 + let result = exec_task_with(&args, None).await;
2480 + assert!(
2481 + result.contains("not available on-device yet"),
2482 + "got: {result}"
2483 + );
2484 + }
2485 +
2486 + #[tokio::test]
2487 + async fn test_task_missing_prompt() {
2488 + let result = exec_task_with(r#"{"description": "x"}"#, None).await;
2489 + assert!(
2490 + result.contains("missing required parameter \"prompt\""),
2491 + "got: {result}"
2492 + );
2493 + }
2494 +
2495 + // ── task (subagent) end-to-end against a scripted endpoint ──────────
2496 +
2497 + /// A completion whose assistant message requests one tool call.
2498 + fn completion_tool_call(id: &str, name: &str, arguments: &str) -> String {
2499 + serde_json::json!({
2500 + "choices": [{"message": {
2501 + "role": "assistant",
2502 + "content": serde_json::Value::Null,
2503 + "tool_calls": [{
2504 + "id": id,
2505 + "type": "function",
2506 + "function": {"name": name, "arguments": arguments},
2507 + }],
2508 + }}]
2509 + })
2510 + .to_string()
2511 + }
2512 +
2513 + /// A completion whose assistant message is a plain text answer.
2514 + fn completion_text(text: &str) -> String {
2515 + serde_json::json!({
2516 + "choices": [{"message": {"role": "assistant", "content": text}}]
2517 + })
2518 + .to_string()
2519 + }
2520 +
2521 + /// Minimal scripted OpenAI-compatible endpoint (same pattern as
2522 + /// `tests/acp_permissions.rs`): serves one canned chat-completion JSON body
2523 + /// per request and records each request body. The subagent loop passes
2524 + /// `sink: None`, so the backend takes the non-streaming path and expects
2525 + /// plain JSON rather than SSE.
2526 + fn start_scripted_endpoint(responses: Vec<String>) -> (u16, Arc<std::sync::Mutex<Vec<Value>>>) {
2527 + use std::io::{BufRead, BufReader, Read, Write};
2528 +
2529 + let listener = std::net::TcpListener::bind("127.0.0.1:0").expect("bind endpoint");
2530 + let port = listener.local_addr().unwrap().port();
2531 + let requests: Arc<std::sync::Mutex<Vec<Value>>> = Arc::default();
2532 + let recorded = Arc::clone(&requests);
2533 + let queue = std::sync::Mutex::new(std::collections::VecDeque::from(responses));
2534 +
2535 + std::thread::spawn(move || {
2536 + // `connection: close` means one request per connection, matching
2537 + // the backend's serial completion requests.
2538 + for stream in listener.incoming() {
2539 + let Ok(mut stream) = stream else { continue };
2540 + let mut reader = BufReader::new(match stream.try_clone() {
2541 + Ok(clone) => clone,
2542 + Err(_) => continue,
2543 + });
2544 + let mut content_length = 0usize;
2545 + loop {
2546 + let mut line = String::new();
2547 + if reader.read_line(&mut line).unwrap_or(0) == 0 {
2548 + break;
2549 + }
2550 + let line = line.trim();
2551 + if line.is_empty() {
2552 + break;
2553 + }
2554 + if let Some(length) = line.to_ascii_lowercase().strip_prefix("content-length:")
2555 + {
2556 + content_length = length.trim().parse().unwrap_or(0);
2557 + }
2558 + }
2559 + let mut body = vec![0u8; content_length];
2560 + if reader.read_exact(&mut body).is_err() {
2561 + continue;
2562 + }
2563 + if let Ok(request) = serde_json::from_slice::<Value>(&body) {
2564 + recorded.lock().unwrap().push(request);
2565 + }
2566 + let payload = queue
2567 + .lock()
2568 + .unwrap()
2569 + .pop_front()
2570 + .unwrap_or_else(|| completion_text("out of scripted responses"));
2571 + let response = format!(
2572 + "HTTP/1.1 200 OK\r\ncontent-type: application/json\r\n\
2573 + content-length: {}\r\nconnection: close\r\n\r\n{}",
2574 + payload.len(),
2575 + payload
2576 + );
2577 + let _ = stream.write_all(response.as_bytes());
2578 + }
2579 + });
2580 +
2581 + (port, requests)
2582 + }
2583 +
2584 + #[tokio::test]
2585 + async fn test_task_runs_subagent_end_to_end() {
2586 + // A file only the subagent's read_file call can surface.
2587 + let dir = std::env::temp_dir().join("sigit_test_subagent_e2e");
2588 + let _ = fs::remove_dir_all(&dir);
2589 + fs::create_dir_all(&dir).unwrap();
2590 + let file = dir.join("notes.txt");
2591 + fs::write(&file, "subagent secret: 4217").unwrap();
2592 +
2593 + // Script: one read_file tool call, then a final text answer.
2594 + let (port, requests) = start_scripted_endpoint(vec![
2595 + completion_tool_call(
2596 + "call_1",
2597 + "read_file",
2598 + &serde_json::json!({ "path": file }).to_string(),
2599 + ),
2600 + completion_text("The file contains the number 4217."),
2601 + ]);
2602 +
2603 + // Register the real process-global factory, pointing a fresh
2604 + // OpenAiBackend at the scripted endpoint — exactly what the surfaces
2605 + // do at startup. This is the only test that touches the OnceLock.
2606 + set_subagent_factory(Box::new(move || {
2607 + Some(Arc::new(crate::backend::OpenAiBackend::new(
2608 + format!("http://127.0.0.1:{port}"),
2609 + "test-key",
2610 + "scripted-model",
2611 + Some(SUBAGENT_SYSTEM_PROMPT.to_string()),
2612 + )) as Arc<dyn InferenceBackend>)
2613 + }));
2614 + assert!(subagent_available());
2615 +
2616 + let args = serde_json::json!({
2617 + "description": "read the notes file",
2618 + "prompt": format!("What number is recorded in {}?", file.display()),
2619 + })
2620 + .to_string();
2621 + let result = execute_tool(TASK_TOOL_NAME, &args).await;
2622 +
2623 + // Only the subagent's final text comes back.
2624 + assert_eq!(result, "The file contains the number 4217.");
2625 +
2626 + let recorded = requests.lock().unwrap();
2627 + assert_eq!(recorded.len(), 2, "expected exactly two completions");
2628 +
2629 + // The subagent conversation is fresh (subagent system prompt) and was
2630 + // offered only the read-only toolset.
2631 + let first = &recorded[0];
2632 + assert_eq!(first["messages"][0]["role"], "system");
2633 + assert!(
2634 + first["messages"][0]["content"]
2635 + .as_str()
2636 + .unwrap()
2637 + .contains("research subagent")
2638 + );
2639 + let offered: Vec<&str> = first["tools"]
2640 + .as_array()
2641 + .unwrap()
2642 + .iter()
2643 + .map(|tool| tool["function"]["name"].as_str().unwrap())
2644 + .collect();
2645 + for name in &offered {
2646 + assert!(
2647 + SUBAGENT_TOOL_NAMES.contains(name),
2648 + "non-read-only tool offered: {name}"
2649 + );
2650 + }
2651 + assert!(!offered.contains(&TASK_TOOL_NAME));
2652 +
2653 + // The read-only tool actually executed: its output travelled back to
2654 + // the endpoint as a tool result on the second request.
2655 + let messages = recorded[1]["messages"].as_array().unwrap();
2656 + let tool_message = messages
2657 + .iter()
2658 + .find(|message| message["role"] == "tool")
2659 + .expect("no tool result in second request");
2660 + assert_eq!(tool_message["tool_call_id"], "call_1");
2661 + assert!(
2662 + tool_message["content"]
2663 + .as_str()
2664 + .unwrap()
2665 + .contains("subagent secret: 4217")
2666 + );
2667 +
2668 + let _ = fs::remove_dir_all(&dir);
2669 + }
2670 }