5
use std::fs;
6
use std::path::{Path, PathBuf};
7
use std::process::Command;
8
+use std::sync::{Arc, OnceLock};
9
+
10
+use crate::backend::{InferenceBackend, ToolResult, ToolSpec};
11
12
const WEBSITE_READ_CHAR_LIMIT: usize = 20_000;
13
const WEBSITE_READ_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(20);
403
"delete_file" => exec_delete_file(arguments),
404
"run_command" => exec_run_command(arguments),
405
"skill" => crate::skills::activate_skill(arguments),
406
+ TASK_TOOL_NAME => exec_task(arguments).await,
407
// Tools discovered from MCP servers are namespaced `mcp__<server>__<tool>`
408
// and forwarded to the owning server.
409
_ if crate::mcp::is_mcp_tool(name) => crate::mcp::call_tool(name, arguments).await,
411
}
412
}
413
414
+// ── task (subagent) ──────────────────────────────────────────────────────────
415
+//
416
+// The `task` tool delegates a self-contained research question to a *nested*
417
+// agent loop running in a fresh conversation, so the main thread receives only
418
+// the final answer instead of every intermediate file read. The subagent's
419
+// toolset is strictly read-only and never includes `task` itself (no
420
+// recursion), so a delegated agent can research but not mutate state.
421
+//
422
+// This module is backend-agnostic and cannot construct a backend, so the
423
+// surface that knows the active provider (`run_acp_server` / `run_interactive`
424
+// in `main.rs`) registers a factory at startup. The factory returns `None`
425
+// when inference runs on-device: onde has a single shared conversation
426
+// history, so a second concurrent context is not possible yet.
427
+
428
+pub const TASK_TOOL_NAME: &str = "task";
429
+
430
+/// The tool names a subagent may call, filtered from [`all_tools`].
431
+const SUBAGENT_TOOL_NAMES: &[&str] = &[
432
+ "read_file",
433
+ "list_directory",
434
+ "search_files",
435
+ "glob",
436
+ "read_website",
437
+];
438
+
439
+/// System prompt seeding every subagent conversation.
440
+pub const SUBAGENT_SYSTEM_PROMPT: &str = "You are a focused research subagent. \
441
+You are given one self-contained task by a calling agent. Investigate it using \
442
+the read-only tools available to you (read_file, list_directory, search_files, \
443
+glob, read_website) and answer it thoroughly but concisely. You cannot modify \
444
+files or run commands. Your final message is returned verbatim to the caller, \
445
+so make it a complete, self-contained answer — include the concrete facts, \
446
+paths, and code excerpts the caller needs, and nothing else.";
447
+
448
+/// Rounds of tool calls a subagent may use before it is forced to answer.
449
+const SUBAGENT_MAX_TOOL_ROUNDS: usize = 8;
450
+/// Cap on the answer text returned to the caller.
451
+const SUBAGENT_RESULT_CHAR_LIMIT: usize = 8_000;
452
+
453
+/// Returned when `task` is called but no subagent backend can be built.
454
+const SUBAGENT_UNAVAILABLE: &str = "The task tool is not available on-device \
455
+yet: on-device inference has a single conversation context. Do the research \
456
+yourself with the read-only tools.";
457
+
458
+/// Builds a fresh backend for one subagent run, or `None` when the active
459
+/// inference cannot host a second conversation (on-device).
460
+pub type SubagentFactory = Box<dyn Fn() -> Option<Arc<dyn InferenceBackend>> + Send + Sync>;
461
+
462
+static SUBAGENT_FACTORY: OnceLock<SubagentFactory> = OnceLock::new();
463
+
464
+/// Register the process-wide subagent factory. Called once at startup by the
465
+/// surface that resolved the inference provider; later calls are ignored.
466
+pub fn set_subagent_factory(factory: SubagentFactory) {
467
+ let _ = SUBAGENT_FACTORY.set(factory);
468
+}
469
+
470
+/// Whether a `task` call could run right now: a factory is registered and it
471
+/// can build a backend. The spec builders (`agent_tools_as_specs` /
472
+/// `build_tool_specs`) use this to offer the tool only when it works, the same
473
+/// conditional pattern as the `skill` tool.
474
+pub fn subagent_available() -> bool {
475
+ SUBAGENT_FACTORY
476
+ .get()
477
+ .is_some_and(|factory| factory().is_some())
478
+}
479
+
480
+/// The read-only toolset offered to a subagent, filtered from [`all_tools`] by
481
+/// name. Never contains `task` (recursion) or any mutating tool.
482
+pub fn subagent_tool_specs() -> Vec<ToolSpec> {
483
+ all_tools()
484
+ .into_iter()
485
+ .filter(|tool| SUBAGENT_TOOL_NAMES.contains(&tool.name))
486
+ .map(|tool| ToolSpec {
487
+ name: tool.name.to_string(),
488
+ description: tool.description.to_string(),
489
+ parameters_schema: tool.parameters_schema.to_string(),
490
+ })
491
+ .collect()
492
+}
493
+
494
+/// Spec for the `task` tool. Lives in the `*_as_specs`/`build_tool_specs`
495
+/// layer (like `skill` and MCP tools), not in [`all_tools`], because it is
496
+/// only offered when [`subagent_available`] is true.
497
+pub fn task_tool_spec() -> ToolSpec {
498
+ ToolSpec {
499
+ name: TASK_TOOL_NAME.to_string(),
500
+ description: "Delegate a self-contained research task to a subagent that \
501
+ runs in a fresh conversation and returns only its final answer. The \
502
+ subagent can read files, list directories, search, glob, and read \
503
+ websites, but cannot modify anything. Use this for exploratory \
504
+ questions whose intermediate file reads would otherwise clutter this \
505
+ conversation (e.g. \"where is X implemented and how does it work?\"). \
506
+ The subagent cannot see this conversation, so the prompt must be \
507
+ fully self-contained: include absolute paths, symbol names, and \
508
+ exactly what the answer should contain."
509
+ .to_string(),
510
+ parameters_schema: json!({
511
+ "type": "object",
512
+ "properties": {
513
+ "description": {
514
+ "type": "string",
515
+ "description": "A short (3-5 word) summary of the task, for progress display."
516
+ },
517
+ "prompt": {
518
+ "type": "string",
519
+ "description": "The full task for the subagent. Must be self-contained: the subagent sees nothing of this conversation."
520
+ }
521
+ },
522
+ "required": ["description", "prompt"],
523
+ "additionalProperties": false
524
+ })
525
+ .to_string(),
526
+ }
527
+}
528
+
529
+/// The `task` tool entry point used by [`execute_tool`].
530
+async fn exec_task(arguments: &str) -> String {
531
+ exec_task_with(arguments, SUBAGENT_FACTORY.get()).await
532
+}
533
+
534
+/// Core of the `task` tool, parameterized on the factory so tests can exercise
535
+/// the unavailable path without touching the process-global `OnceLock`.
536
+async fn exec_task_with(arguments: &str, factory: Option<&SubagentFactory>) -> String {
537
+ let args: Value = match serde_json::from_str(arguments) {
538
+ Ok(v) => v,
539
+ Err(err) => return format!("Error: failed to parse arguments: {err}"),
540
+ };
541
+
542
+ let prompt = match args.get("prompt").and_then(Value::as_str) {
543
+ Some(p) if !p.trim().is_empty() => p,
544
+ _ => return "Error: missing required parameter \"prompt\"".to_string(),
545
+ };
546
+ let description = args
547
+ .get("description")
548
+ .and_then(Value::as_str)
549
+ .unwrap_or("(no description)");
550
+
551
+ let Some(backend) = factory.and_then(|build| build()) else {
552
+ return SUBAGENT_UNAVAILABLE.to_string();
553
+ };
554
+
555
+ log::info!("task: running subagent — {description}");
556
+ run_subagent(backend.as_ref(), prompt).await
557
+}
558
+
559
+/// The nested agent loop: a fresh conversation, read-only tools, a round cap,
560
+/// and only the final text returned. Mirrors the main loops in `main.rs` /
561
+/// `chat.rs`: offer tools each round, and pass `tools = None` on the last
562
+/// round to force a text answer.
563
+async fn run_subagent(backend: &dyn InferenceBackend, prompt: &str) -> String {
564
+ let specs = subagent_tool_specs();
565
+
566
+ let mut result = match backend.send_message_with_tools(prompt, &specs, None).await {
567
+ Ok(r) => r,
568
+ Err(err) => return format!("Error: subagent inference failed: {err}"),
569
+ };
570
+
571
+ let mut round = 0;
572
+ while !result.tool_calls.is_empty() && round < SUBAGENT_MAX_TOOL_ROUNDS {
573
+ round += 1;
574
+ log::info!(
575
+ "task: subagent tool round {round} — {} call(s)",
576
+ result.tool_calls.len()
577
+ );
578
+
579
+ let mut tool_results = Vec::with_capacity(result.tool_calls.len());
580
+ for call in &result.tool_calls {
581
+ // Hard gate, not just advertisement: even if the model asks for a
582
+ // tool outside the offered set, only read-only tools execute here.
583
+ let content = if SUBAGENT_TOOL_NAMES.contains(&call.name.as_str()) {
584
+ // Boxed to break the async cycle: execute_tool → task →
585
+ // run_subagent → execute_tool.
586
+ Box::pin(execute_tool(&call.name, &call.arguments)).await
587
+ } else {
588
+ format!(
589
+ "Error: `{}` is not available to a subagent. Only these \
590
+ read-only tools are: {}.",
591
+ call.name,
592
+ SUBAGENT_TOOL_NAMES.join(", ")
593
+ )
594
+ };
595
+ tool_results.push(ToolResult {
596
+ tool_call_id: call.id.clone(),
597
+ content,
598
+ });
599
+ }
600
+
601
+ // On the last round, offer no tools so the model must produce text.
602
+ let next_tools = if round < SUBAGENT_MAX_TOOL_ROUNDS {
603
+ Some(specs.as_slice())
604
+ } else {
605
+ None
606
+ };
607
+ result = match backend
608
+ .send_tool_results(tool_results, next_tools, None)
609
+ .await
610
+ {
611
+ Ok(r) => r,
612
+ Err(err) => return format!("Error: subagent inference failed: {err}"),
613
+ };
614
+ }
615
+
616
+ let text = result.text.trim();
617
+ if text.is_empty() {
618
+ return "The subagent finished without a text answer.".to_string();
619
+ }
620
+
621
+ let total = text.chars().count();
622
+ if total > SUBAGENT_RESULT_CHAR_LIMIT {
623
+ let truncated: String = text.chars().take(SUBAGENT_RESULT_CHAR_LIMIT).collect();
624
+ return format!(
625
+ "{truncated}\n\n--- truncated (showing {SUBAGENT_RESULT_CHAR_LIMIT} of {total} \
626
+ characters of the subagent's answer) ---"
627
+ );
628
+ }
629
+ text.to_string()
630
+}
631
+
632
fn absolute_path(path: &Path) -> PathBuf {
633
if path.is_absolute() {
634
path.to_path_buf()
2423
let result = exec_run_command(&args);
2424
assert!(result.contains("err"), "got: {result}");
2425
}
2426
+
2427
+ // ── task (subagent) tests ────────────────────────────────────────────
2428
+
2429
+ #[test]
2430
+ fn test_subagent_toolset_is_read_only() {
2431
+ let specs = subagent_tool_specs();
2432
+ let mut names: Vec<&str> = specs.iter().map(|spec| spec.name.as_str()).collect();
2433
+ names.sort_unstable();
2434
+ assert_eq!(
2435
+ names,
2436
+ [
2437
+ "glob",
2438
+ "list_directory",
2439
+ "read_file",
2440
+ "read_website",
2441
+ "search_files"
2442
+ ]
2443
+ );
2444
+
2445
+ // No recursion and no mutating tools, ever.
2446
+ assert!(!names.contains(&TASK_TOOL_NAME));
2447
+ for banned in [
2448
+ "create_file",
2449
+ "create_directory",
2450
+ "edit_file",
2451
+ "multi_edit",
2452
+ "delete_file",
2453
+ "run_command",
2454
+ "write_todos",
2455
+ "remember",
2456
+ ] {
2457
+ assert!(!names.contains(&banned), "{banned} leaked into subagent");
2458
+ }
2459
+
2460
+ // Every spec came from `all_tools()` (schemas intact).
2461
+ for spec in &specs {
2462
+ assert!(
2463
+ serde_json::from_str::<Value>(&spec.parameters_schema)
2464
+ .unwrap()
2465
+ .is_object()
2466
+ );
2467
+ }
2468
+ }
2469
+
2470
+ #[tokio::test]
2471
+ async fn test_task_reports_unavailable_without_factory() {
2472
+ let args = serde_json::json!({
2473
+ "description": "look around",
2474
+ "prompt": "What is in the current directory?"
2475
+ })
2476
+ .to_string();
2477
+ // `None` is exactly what `exec_task` passes before any surface has
2478
+ // registered the process-global factory.
2479
+ let result = exec_task_with(&args, None).await;
2480
+ assert!(
2481
+ result.contains("not available on-device yet"),
2482
+ "got: {result}"
2483
+ );
2484
+ }
2485
+
2486
+ #[tokio::test]
2487
+ async fn test_task_missing_prompt() {
2488
+ let result = exec_task_with(r#"{"description": "x"}"#, None).await;
2489
+ assert!(
2490
+ result.contains("missing required parameter \"prompt\""),
2491
+ "got: {result}"
2492
+ );
2493
+ }
2494
+
2495
+ // ── task (subagent) end-to-end against a scripted endpoint ──────────
2496
+
2497
+ /// A completion whose assistant message requests one tool call.
2498
+ fn completion_tool_call(id: &str, name: &str, arguments: &str) -> String {
2499
+ serde_json::json!({
2500
+ "choices": [{"message": {
2501
+ "role": "assistant",
2502
+ "content": serde_json::Value::Null,
2503
+ "tool_calls": [{
2504
+ "id": id,
2505
+ "type": "function",
2506
+ "function": {"name": name, "arguments": arguments},
2507
+ }],
2508
+ }}]
2509
+ })
2510
+ .to_string()
2511
+ }
2512
+
2513
+ /// A completion whose assistant message is a plain text answer.
2514
+ fn completion_text(text: &str) -> String {
2515
+ serde_json::json!({
2516
+ "choices": [{"message": {"role": "assistant", "content": text}}]
2517
+ })
2518
+ .to_string()
2519
+ }
2520
+
2521
+ /// Minimal scripted OpenAI-compatible endpoint (same pattern as
2522
+ /// `tests/acp_permissions.rs`): serves one canned chat-completion JSON body
2523
+ /// per request and records each request body. The subagent loop passes
2524
+ /// `sink: None`, so the backend takes the non-streaming path and expects
2525
+ /// plain JSON rather than SSE.
2526
+ fn start_scripted_endpoint(responses: Vec<String>) -> (u16, Arc<std::sync::Mutex<Vec<Value>>>) {
2527
+ use std::io::{BufRead, BufReader, Read, Write};
2528
+
2529
+ let listener = std::net::TcpListener::bind("127.0.0.1:0").expect("bind endpoint");
2530
+ let port = listener.local_addr().unwrap().port();
2531
+ let requests: Arc<std::sync::Mutex<Vec<Value>>> = Arc::default();
2532
+ let recorded = Arc::clone(&requests);
2533
+ let queue = std::sync::Mutex::new(std::collections::VecDeque::from(responses));
2534
+
2535
+ std::thread::spawn(move || {
2536
+ // `connection: close` means one request per connection, matching
2537
+ // the backend's serial completion requests.
2538
+ for stream in listener.incoming() {
2539
+ let Ok(mut stream) = stream else { continue };
2540
+ let mut reader = BufReader::new(match stream.try_clone() {
2541
+ Ok(clone) => clone,
2542
+ Err(_) => continue,
2543
+ });
2544
+ let mut content_length = 0usize;
2545
+ loop {
2546
+ let mut line = String::new();
2547
+ if reader.read_line(&mut line).unwrap_or(0) == 0 {
2548
+ break;
2549
+ }
2550
+ let line = line.trim();
2551
+ if line.is_empty() {
2552
+ break;
2553
+ }
2554
+ if let Some(length) = line.to_ascii_lowercase().strip_prefix("content-length:")
2555
+ {
2556
+ content_length = length.trim().parse().unwrap_or(0);
2557
+ }
2558
+ }
2559
+ let mut body = vec![0u8; content_length];
2560
+ if reader.read_exact(&mut body).is_err() {
2561
+ continue;
2562
+ }
2563
+ if let Ok(request) = serde_json::from_slice::<Value>(&body) {
2564
+ recorded.lock().unwrap().push(request);
2565
+ }
2566
+ let payload = queue
2567
+ .lock()
2568
+ .unwrap()
2569
+ .pop_front()
2570
+ .unwrap_or_else(|| completion_text("out of scripted responses"));
2571
+ let response = format!(
2572
+ "HTTP/1.1 200 OK\r\ncontent-type: application/json\r\n\
2573
+ content-length: {}\r\nconnection: close\r\n\r\n{}",
2574
+ payload.len(),
2575
+ payload
2576
+ );
2577
+ let _ = stream.write_all(response.as_bytes());
2578
+ }
2579
+ });
2580
+
2581
+ (port, requests)
2582
+ }
2583
+
2584
+ #[tokio::test]
2585
+ async fn test_task_runs_subagent_end_to_end() {
2586
+ // A file only the subagent's read_file call can surface.
2587
+ let dir = std::env::temp_dir().join("sigit_test_subagent_e2e");
2588
+ let _ = fs::remove_dir_all(&dir);
2589
+ fs::create_dir_all(&dir).unwrap();
2590
+ let file = dir.join("notes.txt");
2591
+ fs::write(&file, "subagent secret: 4217").unwrap();
2592
+
2593
+ // Script: one read_file tool call, then a final text answer.
2594
+ let (port, requests) = start_scripted_endpoint(vec![
2595
+ completion_tool_call(
2596
+ "call_1",
2597
+ "read_file",
2598
+ &serde_json::json!({ "path": file }).to_string(),
2599
+ ),
2600
+ completion_text("The file contains the number 4217."),
2601
+ ]);
2602
+
2603
+ // Register the real process-global factory, pointing a fresh
2604
+ // OpenAiBackend at the scripted endpoint — exactly what the surfaces
2605
+ // do at startup. This is the only test that touches the OnceLock.
2606
+ set_subagent_factory(Box::new(move || {
2607
+ Some(Arc::new(crate::backend::OpenAiBackend::new(
2608
+ format!("http://127.0.0.1:{port}"),
2609
+ "test-key",
2610
+ "scripted-model",
2611
+ Some(SUBAGENT_SYSTEM_PROMPT.to_string()),
2612
+ )) as Arc<dyn InferenceBackend>)
2613
+ }));
2614
+ assert!(subagent_available());
2615
+
2616
+ let args = serde_json::json!({
2617
+ "description": "read the notes file",
2618
+ "prompt": format!("What number is recorded in {}?", file.display()),
2619
+ })
2620
+ .to_string();
2621
+ let result = execute_tool(TASK_TOOL_NAME, &args).await;
2622
+
2623
+ // Only the subagent's final text comes back.
2624
+ assert_eq!(result, "The file contains the number 4217.");
2625
+
2626
+ let recorded = requests.lock().unwrap();
2627
+ assert_eq!(recorded.len(), 2, "expected exactly two completions");
2628
+
2629
+ // The subagent conversation is fresh (subagent system prompt) and was
2630
+ // offered only the read-only toolset.
2631
+ let first = &recorded[0];
2632
+ assert_eq!(first["messages"][0]["role"], "system");
2633
+ assert!(
2634
+ first["messages"][0]["content"]
2635
+ .as_str()
2636
+ .unwrap()
2637
+ .contains("research subagent")
2638
+ );
2639
+ let offered: Vec<&str> = first["tools"]
2640
+ .as_array()
2641
+ .unwrap()
2642
+ .iter()
2643
+ .map(|tool| tool["function"]["name"].as_str().unwrap())
2644
+ .collect();
2645
+ for name in &offered {
2646
+ assert!(
2647
+ SUBAGENT_TOOL_NAMES.contains(name),
2648
+ "non-read-only tool offered: {name}"
2649
+ );
2650
+ }
2651
+ assert!(!offered.contains(&TASK_TOOL_NAME));
2652
+
2653
+ // The read-only tool actually executed: its output travelled back to
2654
+ // the endpoint as a tool result on the second request.
2655
+ let messages = recorded[1]["messages"].as_array().unwrap();
2656
+ let tool_message = messages
2657
+ .iter()
2658
+ .find(|message| message["role"] == "tool")
2659
+ .expect("no tool result in second request");
2660
+ assert_eq!(tool_message["tool_call_id"], "call_1");
2661
+ assert!(
2662
+ tool_message["content"]
2663
+ .as_str()
2664
+ .unwrap()
2665
+ .contains("subagent secret: 4217")
2666
+ );
2667
+
2668
+ let _ = fs::remove_dir_all(&dir);
2669
+ }
2670
}