use std::sync::Arc; use dashmap::DashMap; use tokio_util::sync::CancellationToken; use uuid::Uuid; use crate::agent::SubAgentManager; use crate::agent::inbox::AgentInboxNotifier; use crate::agent::projection::{AgentProjection, AgentProjectionHub}; use crate::agent::run::AgentExecutionContext; use crate::agent::sub_agent::{ ExecutionMode, ResolvedAgentRun, SubAgentConfig, SubAgentError, SubAgentResult, TaskStatus, }; use crate::storage::Storage; use crate::storage::agent_inbox::AgentEventType; use crate::storage::agent_run::{ AcceptAgentRequest, AcceptedAgentRuns, AgentRunMode, AgentRunRecord, AgentRunStatus, AgentTerminalOutcome, NewAgentRun, }; use crate::tools::ToolExecutionContext; use crate::tools::emit_signal::{SignalAccepted, SignalAcceptedStatus, SignalInput}; /// Durable Agent orchestration. Every named run is persisted in /// `agent_runs` before execution, transitions are execution-ID conditional, /// and the terminal commit is the single writer of final state. Background /// runs reserve an inbox completion slot at admission; their completion event /// is materialized by the terminal commit and delivered through the Session /// continuation lane instead of a direct channel notification. /// Admission result for a background batch: the run ids actually spawned. #[derive(Debug, Clone)] pub struct BackgroundAdmission { pub run_ids: Vec, } #[derive(Debug, Clone)] pub struct ScheduledAgentExecution { pub agent_run_id: String, pub status: crate::storage::ScheduledRunStatus, pub outcome: Option, pub error: Option, pub agent_terminal: AgentTerminalOutcome, pub runtime_generation: i64, } pub struct AgentCoordinator { storage: Arc, manager: Arc, work_manager: Option>, notifier: Arc, projection: Arc, execution_gate: Arc, admission: crate::gateway::reload::RuntimeAdmission, task_supervisor: crate::task_supervisor::TaskSupervisor, runtime_generation: i64, max_pending_inbox_events_per_session: i64, max_inbox_delivery_attempts: i64, active_tokens: DashMap, } #[derive(Debug, thiserror::Error)] pub enum CoordinatorError { #[error("agent orchestration rejected the request: {0}")] Rejected(String), #[error("agent run storage error: {0}")] Storage(#[from] crate::storage::StorageError), #[error(transparent)] SubAgent(#[from] SubAgentError), } impl AgentCoordinator { #[allow(clippy::too_many_arguments)] pub(crate) fn new( storage: Arc, manager: Arc, work_manager: Arc, notifier: Arc, projection: Arc, execution_gate: Arc, admission: crate::gateway::reload::RuntimeAdmission, task_supervisor: crate::task_supervisor::TaskSupervisor, runtime_generation: u64, orchestration: &crate::config::AgentOrchestrationConfig, ) -> Arc { Arc::new(Self { storage, manager, work_manager: Some(work_manager), notifier, projection, execution_gate, admission, task_supervisor, runtime_generation: runtime_generation as i64, max_pending_inbox_events_per_session: orchestration.max_pending_inbox_events_per_session as i64, max_inbox_delivery_attempts: i64::from(orchestration.max_inbox_delivery_attempts), active_tokens: DashMap::new(), }) } #[allow(clippy::too_many_arguments)] pub async fn execute_scheduled( self: &Arc, job_run_id: i64, lease_owner: &str, job_id: &str, job_name: &str, agent_id: Option<&str>, prompt: &str, timeout_secs: u64, ) -> Result { let run_id = Uuid::new_v4().to_string(); let root_session_id = format!("scheduled-run:{job_run_id}"); let sink = Arc::new(crate::tools::ScheduledCompletionSink::default()); let caller = ToolExecutionContext::for_session(root_session_id.clone()) .with_turn_id(format!("scheduled:{job_run_id}")) .with_execution_origin(crate::tools::ExecutionOrigin::Scheduled { job_run_id }); let contract = format!( "## Unattended Scheduled Run\n\nYou are executing scheduled task “{job_name}” ({job_id}). The user will not see ordinary final text. After completing all necessary work, you must call complete_scheduled_run exactly once. Use ok only when the task completed and found nothing requiring attention; use alert for actionable findings; use failed when the task did not complete reliably; use refused for a permission or safety refusal. Legacy textual suppression and direct-messaging instructions are obsolete." ); let config = SubAgentConfig { target: agent_id.map(str::to_string), prompt: prompt.to_string(), context: Some(contract.clone()), mode: ExecutionMode::Foreground, allowed_tools: None, max_iterations: None, timeout_secs: Some(timeout_secs), plan_item_id: None, session_id: Some(root_session_id.clone()), }; let mut resolution = if agent_id.is_some() { let mut resolution = self.manager.resolve_agent(&config, &caller, &run_id)?; resolution .tools .register(crate::tools::CompleteScheduledRunTool::new()); resolution.tool_context.session_id = Some(root_session_id.clone()); resolution.tool_context.scheduled_completion = Some(sink.clone()); resolution.timeout_secs = resolution.timeout_secs.min(timeout_secs); resolution.signal_contract = None; resolution } else { self.manager .resolve_scheduled_root(&caller, &run_id, timeout_secs, sink.clone())? }; resolution.tool_context.execution_origin = crate::tools::ExecutionOrigin::Scheduled { job_run_id }; let now = chrono::Utc::now().timestamp_millis(); let deadline_at = now.saturating_add((resolution.timeout_secs * 1000) as i64); let new_run = NewAgentRun { id: run_id.clone(), root_session_id, root_turn_id: None, parent_run_id: None, caller_agent_id: "SCHEDULER".to_string(), caller_scope_id: format!("scheduled:{job_id}"), idempotency_key: Some(format!("scheduled:{job_run_id}")), agent_id: agent_id.unwrap_or("ROOT").to_string(), definition_hash: resolution.definition_hash.clone().unwrap_or_default(), provider_profile: resolution.llm_profile.clone().unwrap_or_default(), provider_name: resolution.provider_config.name.clone(), model_id: resolution.provider_config.model_id.clone(), mode: AgentRunMode::Foreground, depth: 1, plan_item_id: None, execution_id: run_id.clone(), task: prompt.to_string(), context_json: None, budget_json: serde_json::json!({ "remaining_runs": self.manager.catalog().max_runs_per_tree().saturating_sub(1), "remaining_depth": self.manager.catalog().max_tree_depth().saturating_sub(1), }) .to_string(), signal_contract_json: None, signal_delivery: None, deadline_at, runtime_generation: self.runtime_generation, completion_slot_reserved: false, }; match self .storage .accept_agent_runs(AcceptAgentRequest { runs: vec![new_run], now, }) .await? { AcceptedAgentRuns::Accepted { .. } => {} AcceptedAgentRuns::Existing { .. } => { return Err(CoordinatorError::Rejected(format!( "scheduled occurrence {job_run_id} already has an Agent run" ))); } } let mut mark_attempt = 0_u64; let marked = match loop { match self .storage .mark_scheduled_run_running(job_run_id, lease_owner, Some(&run_id), now) .await { Err(error) if error.is_transient() && mark_attempt < 2 => { mark_attempt += 1; tokio::time::sleep(std::time::Duration::from_millis(50 * mark_attempt)).await; } result => break result, } } { Ok(marked) => marked, Err(error) => { let _ = self .storage .cancel_agent_run_with_completion( &run_id, "scheduled occurrence could not enter running state", true, now, ) .await; return Err(error.into()); } }; if !marked { let _ = self .storage .cancel_agent_run_with_completion( &run_id, "scheduled occurrence was no longer active", true, now, ) .await; return Err(CoordinatorError::Rejected(format!( "scheduled occurrence {job_run_id} lost its lease" ))); } let (execution, agent_terminal) = match self .execute_scheduled_agent_run(&run_id, &config, resolution) .await { Ok(execution) => execution, Err(error) => ( Err(error), AgentTerminalOutcome::Failed { error: "scheduled Agent could not enter its execution lifecycle".to_string(), prompt_tokens: None, completion_tokens: None, cost: None, signal_ids: Vec::new(), }, ), }; let status = match &execution { Ok(result) => match &result.status { TaskStatus::Completed => crate::storage::ScheduledRunStatus::Completed, TaskStatus::Failed(_) => crate::storage::ScheduledRunStatus::Failed, TaskStatus::TimedOut => crate::storage::ScheduledRunStatus::TimedOut, TaskStatus::Cancelled => crate::storage::ScheduledRunStatus::Interrupted, }, Err(_) => crate::storage::ScheduledRunStatus::Failed, }; let error = match &execution { Ok(result) => match &result.status { TaskStatus::Completed => None, TaskStatus::Failed(_) => Some( "scheduled Agent execution failed; inspect Gateway logs for details" .to_string(), ), TaskStatus::TimedOut => Some("scheduled Agent timed out".to_string()), TaskStatus::Cancelled => Some("scheduled Agent was cancelled".to_string()), }, Err(_) => Some( "scheduled Agent execution failed; inspect Gateway logs for details".to_string(), ), }; Ok(ScheduledAgentExecution { agent_run_id: run_id, status, outcome: sink.outcome(), error, agent_terminal, runtime_generation: self.runtime_generation, }) } async fn execute_scheduled_agent_run( self: &Arc, run_id: &str, config: &SubAgentConfig, resolution: ResolvedAgentRun, ) -> Result< ( Result, AgentTerminalOutcome, ), CoordinatorError, > { let execution_id = run_id.to_string(); let token = resolution.tool_context.cancellation.clone(); self.active_tokens.insert(run_id.to_string(), token); let started = self .storage .mark_agent_run_running(run_id, &execution_id, chrono::Utc::now().timestamp_millis()) .await?; if !started { self.active_tokens.remove(run_id); return Err(CoordinatorError::Rejected(format!( "scheduled Agent run {run_id} was closed before execution started" ))); } let result = self .manager .execute_resolved(config, resolution, run_id) .await .map_err(CoordinatorError::SubAgent); self.active_tokens.remove(run_id); let terminal = match &result { Ok(result) => match &result.status { TaskStatus::Completed => AgentTerminalOutcome::Completed { result: result.full_content.clone(), prompt_tokens: None, completion_tokens: None, cost: None, tool_calls: result.tool_calls_count as i64, iterations: result.iterations as i64, signal_ids: Vec::new(), }, TaskStatus::Failed(error) => AgentTerminalOutcome::Failed { error: error.clone(), prompt_tokens: None, completion_tokens: None, cost: None, signal_ids: Vec::new(), }, TaskStatus::TimedOut => AgentTerminalOutcome::TimedOut { deadline_at: chrono::Utc::now().timestamp_millis(), signal_ids: Vec::new(), }, TaskStatus::Cancelled => AgentTerminalOutcome::Interrupted { reason: "scheduled Agent interrupted by shutdown".to_string(), signal_ids: Vec::new(), }, }, Err(error) => AgentTerminalOutcome::Failed { error: error.to_string(), prompt_tokens: None, completion_tokens: None, cost: None, signal_ids: Vec::new(), }, }; Ok((result, terminal)) } /// Admit a named background run for the root caller and spawn its runner. /// Completion is guaranteed by the reserved inbox slot; the returned ID /// is only valid when every durable step succeeded. /// Admit one or more named background runs and spawn their runners. /// Returns immediately: run quota is acquired inside each runner (queuing /// time counts toward the run timeout), and completion capacity is /// reserved up front so no completion can ever be lost. pub async fn delegate_background( self: &Arc, caller: &ToolExecutionContext, configs: Vec, ) -> Result { if caller.execution_origin.is_scheduled() { return Err(CoordinatorError::Rejected( "scheduled Agents cannot create background runs".to_string(), )); } if caller.agent.is_some() { return Err(CoordinatorError::Rejected( "nested background runs are not available yet; only the root Agent may delegate background work".to_string(), )); } if configs.is_empty() { return Err(CoordinatorError::Rejected( "background delegation requires at least one task".to_string(), )); } // Hard cap: a batch larger than the run quota would never execute // concurrently, so reject it up front. if configs.len() > self.execution_gate.max_concurrent_runs() { return Err(CoordinatorError::Rejected(format!( "background batch of {} runs exceeds max_concurrent_runs ({})", configs.len(), self.execution_gate.max_concurrent_runs() ))); } let root_session_id = caller .session_id .clone() .or_else(|| { caller .agent .as_ref() .map(|agent| agent.root_session_id.clone()) }) .ok_or_else(|| { CoordinatorError::Rejected("delegate requires a session-bound context".to_string()) })?; let now = chrono::Utc::now().timestamp_millis(); // Resolve every target before any durable write so a bad request // fails closed without leaving orphan rows. let mut run_ids = Vec::with_capacity(configs.len()); let mut resolved = Vec::with_capacity(configs.len()); for config in &configs { if config.target.is_none() { return Err(CoordinatorError::Rejected( "named background targets only".to_string(), )); } let run_id = Uuid::new_v4().to_string(); resolved.push(self.manager.resolve_agent(config, caller, &run_id)?); run_ids.push(run_id); } // 1. Reserve one completion slot per run; failure rejects the whole // batch so nothing is admitted under capacity. if self .storage .reserve_completion_slots( &root_session_id, configs.len() as i64, self.max_pending_inbox_events_per_session, now, ) .await? .is_none() { return Err(CoordinatorError::Rejected( "inbox capacity exceeded; cannot accept the background batch".to_string(), )); } // 2. Persist the queued runs atomically with the reservation. let mut runs = Vec::with_capacity(configs.len()); for (index, config) in configs.iter().enumerate() { let resolution = &resolved[index]; runs.push(NewAgentRun { id: run_ids[index].clone(), root_session_id: root_session_id.clone(), root_turn_id: caller.turn_id.clone(), parent_run_id: None, caller_agent_id: "ROOT".to_string(), caller_scope_id: "ROOT".to_string(), idempotency_key: None, agent_id: resolution.agent_id.clone().unwrap_or_default(), definition_hash: resolution.definition_hash.clone().unwrap_or_default(), provider_profile: resolution.llm_profile.clone().unwrap_or_default(), provider_name: resolution.provider_config.name.clone(), model_id: resolution.provider_config.model_id.clone(), mode: AgentRunMode::Background, depth: 1, plan_item_id: config.plan_item_id.clone(), execution_id: run_ids[index].clone(), task: config.prompt.clone(), context_json: config.context.clone(), budget_json: serde_json::json!({ "remaining_runs": self.manager.catalog().max_runs_per_tree(), "remaining_depth": self.manager.catalog().max_tree_depth(), }) .to_string(), signal_contract_json: resolution .signal_contract .as_ref() .map(|contract| serde_json::to_string(contract).unwrap_or_default()), signal_delivery: resolution .signal_contract .as_ref() .map(|contract| contract.delivery.as_str().to_string()), deadline_at: now + (resolution.timeout_secs * 1000) as i64, runtime_generation: self.runtime_generation, completion_slot_reserved: true, }); } match self .storage .accept_agent_runs(AcceptAgentRequest { runs, now }) .await? { AcceptedAgentRuns::Accepted { .. } => {} AcceptedAgentRuns::Existing { .. } => { self.storage .release_completion_slots(&root_session_id, configs.len() as i64, now) .await?; return Err(CoordinatorError::Rejected( "background admission conflicted with an existing run id".to_string(), )); } } // 3. Spawn one runner per run. Each runner acquires its own run // quota permit and admission guard; delegate returns immediately. let coordinator = self.clone(); let mut spawned_ids = Vec::with_capacity(configs.len()); for (index, config) in configs.iter().enumerate() { let run_id = run_ids[index].clone(); let token = CancellationToken::new(); self.active_tokens.insert(run_id.clone(), token.clone()); let config = config.clone(); let resolution = resolved[index].clone(); let spawned = self .task_supervisor .spawn_graceful(format!("agent-run:{run_id}"), { let coordinator = coordinator.clone(); let run_id = run_id.clone(); async move { coordinator .run_background_runner(&run_id, &config, resolution, token) .await; } }); if !spawned { // Compensation: the rejected closure was dropped by the // supervisor. Cancel the run and release its slot. self.active_tokens.remove(&run_id); let _ = self .storage .cancel_agent_run_with_completion(&run_id, "gateway shutdown", true, now) .await; continue; } spawned_ids.push(run_id); } if spawned_ids.is_empty() { return Err(CoordinatorError::Rejected( "gateway is shutting down and cannot accept background tasks".to_string(), )); } Ok(BackgroundAdmission { run_ids: spawned_ids, }) } async fn run_background_runner( self: &Arc, run_id: &str, config: &SubAgentConfig, resolved: ResolvedAgentRun, token: CancellationToken, ) { let now = chrono::Utc::now().timestamp_millis(); let execution_id = run_id.to_string(); let root_session_id = resolved .tool_context .agent .as_ref() .map(|agent| agent.root_session_id.clone()) .unwrap_or_default(); // Run quota + admission guard, acquired inside the runner so // `delegate_background` returns immediately. Queuing time counts // toward the run timeout and cancellation aborts the wait. let run_permit = match self .execution_gate .acquire_run(&root_session_id, &token) .await { Ok(permit) => permit, Err(_) => { let _ = self .storage .cancel_agent_run_with_completion(run_id, "cancelled before start", true, now) .await; self.active_tokens.remove(run_id); return; } }; let Some(activity) = self.admission.try_enter() else { drop(run_permit); let _ = self .storage .cancel_agent_run_with_completion(run_id, "gateway shutdown", true, now) .await; self.active_tokens.remove(run_id); return; }; let _run_permit = run_permit; let _activity = activity; if !self .storage .mark_agent_run_running(run_id, &execution_id, now) .await .unwrap_or(false) { // Cancelled before start; the canceller already resolved the // reservation and completion. self.active_tokens.remove(run_id); return; } let emitted_signals: Vec = resolved .tool_context .agent .as_ref() .and_then(|agent| agent.emitted_signals.lock().ok()) .map(|signals| { signals .iter() .map(|signal| signal.signal_id.clone()) .collect() }) .unwrap_or_default(); let result = self .manager .execute_resolved(config, resolved, run_id) .await; let outcome = match &result { Ok(result) => match &result.status { TaskStatus::Completed => AgentTerminalOutcome::Completed { result: result.full_content.clone(), prompt_tokens: None, completion_tokens: None, cost: None, tool_calls: result.tool_calls_count as i64, iterations: result.iterations as i64, signal_ids: emitted_signals.clone(), }, TaskStatus::Failed(error) => AgentTerminalOutcome::Failed { error: error.clone(), prompt_tokens: None, completion_tokens: None, cost: None, signal_ids: emitted_signals.clone(), }, TaskStatus::TimedOut => AgentTerminalOutcome::TimedOut { deadline_at: chrono::Utc::now().timestamp_millis(), signal_ids: emitted_signals.clone(), }, TaskStatus::Cancelled => AgentTerminalOutcome::Cancelled { reason: "cancelled by user, parent or shutdown".to_string(), signal_ids: emitted_signals.clone(), }, }, Err(error) => AgentTerminalOutcome::Failed { error: error.to_string(), prompt_tokens: None, completion_tokens: None, cost: None, signal_ids: emitted_signals, }, }; let summary = result .as_ref() .ok() .map(|result| truncate_summary(&result.full_content)); match self .storage .commit_agent_terminal( run_id, &execution_id, self.runtime_generation, &outcome, summary.as_deref(), chrono::Utc::now().timestamp_millis(), ) .await { Ok(Some(commit)) => { if let Some(plan_item_id) = commit.run.plan_item_id.clone() { self.refresh_work_plan(&commit.run.root_session_id, Some(plan_item_id)); } // The completion event (if any) was committed; wake the // session so it claims the inbox soon. A lost wake is not // fatal — the next claim pass finds the event anyway. self.projection.publish(AgentProjection { session_id: commit.run.root_session_id.clone(), revision: commit.run.revision, run: Some(crate::protocol::AgentRunView::from_record( &commit.run, 2000, )), event: None, }); self.notifier .notify(&commit.run.root_session_id, commit.run.revision) .await; } Ok(None) => { tracing::warn!( run_id, "late background result discarded by terminal commit" ); } Err(error) => { tracing::error!(run_id, error = %error, "background terminal commit failed"); } } self.active_tokens.remove(run_id); drop(token); } /// Cancel every nonterminal run of a session (archive/delete or `/stop`). /// Completion events are written consumed so no continuation starts after /// the session was closed. pub async fn cancel_session( &self, session_id: &str, reason: &str, ) -> Result { let runs = self.storage.list_agent_runs(session_id, None, 200).await?; let mut count = 0; for run in runs { if run.status.is_terminal() { continue; } if self .storage .cancel_agent_run_with_completion( &run.id, reason, true, chrono::Utc::now().timestamp_millis(), ) .await? { if let Some((_, token)) = self.active_tokens.remove(&run.id) { token.cancel(); } count += 1; } } Ok(count) } /// Startup/activation recovery: interrupt runs of older generations, /// expire stale leases and reconcile the per-session capacity rows. /// Safe to call once per activation. pub async fn recover_on_activation( &self, ) -> Result { let now = chrono::Utc::now().timestamp_millis(); let report = self .storage .recover_agent_state( self.runtime_generation, now, self.max_inbox_delivery_attempts, 60_000, ) .await?; // One merged wake per session with due events; the notifier is the // accelerator and a dead target is not an error (the periodic timer // inside each live worker re-claims anyway). if report.interrupted_runs > 0 || report.leases_expired > 0 { for session_id in self.storage.sessions_with_due_events(now).await? { self.notifier.notify(&session_id, now).await; } } Ok(report) } /// Execute a foreground delegation batch with durable run persistence. /// Results keep request order even though runs execute concurrently. pub async fn delegate_foreground( self: &Arc, caller: &ToolExecutionContext, configs: Vec, ) -> Result, CoordinatorError> { if configs.is_empty() { return Err(CoordinatorError::Rejected( "foreground delegation requires at least one task".to_string(), )); } if configs .iter() .any(|config| config.mode != ExecutionMode::Foreground) { return Err(CoordinatorError::Rejected( "coordinator foreground path received a non-foreground request".to_string(), )); } // Resolve every target before persisting anything so a bad request // fails closed without leaving orphan rows. let mut run_ids = Vec::with_capacity(configs.len()); let mut resolved: Vec = Vec::with_capacity(configs.len()); for config in &configs { if config.target.is_none() { return Err(CoordinatorError::Rejected( "legacy general Agent is not persisted; named targets only".to_string(), )); } let run_id = Uuid::new_v4().to_string(); resolved.push(self.manager.resolve_agent(config, caller, &run_id)?); run_ids.push(run_id); } let root_session_id = caller .agent .as_ref() .map(|agent| agent.root_session_id.clone()) .or_else(|| caller.session_id.clone()) .ok_or_else(|| { CoordinatorError::Rejected("delegate requires a session-bound context".to_string()) })?; let now = chrono::Utc::now().timestamp_millis(); let caller_scope_id = caller .turn_id .clone() .or_else(|| caller.agent.as_ref().map(|agent| agent.run_id.clone())) .unwrap_or_else(|| "root".to_string()); let mut runs = Vec::with_capacity(configs.len()); for (index, config) in configs.iter().enumerate() { let resolution = &resolved[index]; let deadline_at = now + (resolution.timeout_secs * 1000) as i64; runs.push(NewAgentRun { id: run_ids[index].clone(), root_session_id: root_session_id.clone(), root_turn_id: caller.turn_id.clone(), parent_run_id: caller.agent.as_ref().map(|agent| agent.run_id.clone()), caller_agent_id: caller .agent .as_ref() .map(|agent| agent.current_agent_id.clone()) .unwrap_or_else(|| "ROOT".to_string()), caller_scope_id: caller_scope_id.clone(), idempotency_key: None, agent_id: resolution.agent_id.clone().unwrap_or_default(), definition_hash: resolution.definition_hash.clone().unwrap_or_default(), provider_profile: resolution.llm_profile.clone().unwrap_or_default(), provider_name: resolution.provider_config.name.clone(), model_id: resolution.provider_config.model_id.clone(), mode: AgentRunMode::Foreground, depth: caller .agent .as_ref() .map_or(1, |agent| agent.depth.saturating_add(1) as i64), plan_item_id: config.plan_item_id.clone(), execution_id: run_ids[index].clone(), task: config.prompt.clone(), context_json: config.context.clone(), budget_json: serde_json::to_string(&serde_json::json!({ "remaining_runs": caller.agent.as_ref().map(|agent| agent.budget.remaining_runs), "remaining_depth": caller.agent.as_ref().map(|agent| agent.budget.remaining_depth), })) .unwrap_or_default(), signal_contract_json: resolution .signal_contract .as_ref() .map(|contract| serde_json::to_string(contract).unwrap_or_default()), signal_delivery: resolution .signal_contract .as_ref() .map(|contract| contract.delivery.as_str().to_string()), deadline_at, runtime_generation: self.runtime_generation, completion_slot_reserved: false, }); } match self .storage .accept_agent_runs(AcceptAgentRequest { runs, now }) .await? { AcceptedAgentRuns::Accepted { .. } => {} AcceptedAgentRuns::Existing { .. } => { return Err(CoordinatorError::Rejected( "foreground delegation conflicted with an existing run id".to_string(), )); } } // A named parent waits structurally for its children: it moves to // waiting_children and holds no step permits while waiting. let parent = caller.agent.clone(); if let Some(parent) = parent.as_ref() { let _ = self .storage .mark_agent_run_waiting_children( &parent.run_id, &parent.execution_id, AgentRunStatus::Running, chrono::Utc::now().timestamp_millis(), ) .await; } let futures: Vec<_> = configs .iter() .enumerate() .map(|(index, config)| { let coordinator = self.clone(); let run_id = run_ids[index].clone(); let resolution = resolved[index].clone(); let config = config.clone(); async move { coordinator.execute_run(&run_id, &config, resolution).await } }) .collect(); let results = futures_util::future::join_all(futures) .await .into_iter() .enumerate() .map(|(index, result)| { result.unwrap_or_else(|error| SubAgentResult { task_id: run_ids[index].clone(), content: String::new(), content_truncated: false, full_content: String::new(), status: TaskStatus::Failed(error.to_string()), tool_calls_count: 0, iterations: 0, duration_ms: 0, }) }) .collect(); if let Some(parent) = parent.as_ref() { let _ = self .storage .restore_agent_run_running( &parent.run_id, &parent.execution_id, chrono::Utc::now().timestamp_millis(), ) .await; } Ok(results) } async fn execute_run( self: &Arc, run_id: &str, config: &SubAgentConfig, resolution: ResolvedAgentRun, ) -> Result { let execution_id = run_id.to_string(); let token = resolution.tool_context.cancellation.clone(); self.active_tokens.insert(run_id.to_string(), token); let started = self .storage .mark_agent_run_running(run_id, &execution_id, chrono::Utc::now().timestamp_millis()) .await?; if !started { self.active_tokens.remove(run_id); let status = self .storage .get_agent_run(run_id) .await? .map(|run| run.status) .unwrap_or(AgentRunStatus::Cancelled); return Ok(SubAgentResult { task_id: run_id.to_string(), content: String::new(), content_truncated: false, full_content: String::new(), status: match status { AgentRunStatus::Cancelled => TaskStatus::Cancelled, AgentRunStatus::TimedOut => TaskStatus::TimedOut, AgentRunStatus::Interrupted => { TaskStatus::Failed("run interrupted before start".to_string()) } _ => TaskStatus::Failed("run was closed before execution started".to_string()), }, tool_calls_count: 0, iterations: 0, duration_ms: 0, }); } let emitted_signals: Vec = resolution .tool_context .agent .as_ref() .and_then(|agent| agent.emitted_signals.lock().ok()) .map(|signals| { signals .iter() .map(|signal| signal.signal_id.clone()) .collect() }) .unwrap_or_default(); let result = self .manager .execute_resolved(config, resolution, run_id) .await; let outcome_result = match &result { Ok(result) => match &result.status { TaskStatus::Completed => AgentTerminalOutcome::Completed { result: result.full_content.clone(), prompt_tokens: None, completion_tokens: None, cost: None, tool_calls: result.tool_calls_count as i64, iterations: result.iterations as i64, signal_ids: emitted_signals.clone(), }, TaskStatus::Failed(error) => AgentTerminalOutcome::Failed { error: error.clone(), prompt_tokens: None, completion_tokens: None, cost: None, signal_ids: emitted_signals.clone(), }, TaskStatus::TimedOut => AgentTerminalOutcome::TimedOut { deadline_at: chrono::Utc::now().timestamp_millis(), signal_ids: emitted_signals.clone(), }, TaskStatus::Cancelled => AgentTerminalOutcome::Cancelled { reason: "cancelled by user, parent or shutdown".to_string(), signal_ids: emitted_signals.clone(), }, }, Err(error) => AgentTerminalOutcome::Failed { error: error.to_string(), prompt_tokens: None, completion_tokens: None, cost: None, signal_ids: emitted_signals, }, }; let summary = result .as_ref() .ok() .map(|result| truncate_summary(&result.full_content)); let commit = self .storage .commit_agent_terminal( run_id, &execution_id, self.runtime_generation, &outcome_result, summary.as_deref(), chrono::Utc::now().timestamp_millis(), ) .await?; self.active_tokens.remove(run_id); match commit { Some(commit) => { if commit.run.plan_item_id.is_some() { self.refresh_work_plan( &commit.run.root_session_id, commit.run.plan_item_id.clone(), ); } } None => { tracing::warn!(run_id, "late agent run result discarded by terminal commit"); } } result.map_err(CoordinatorError::SubAgent) } /// The plan item was already mutated inside the Storage transaction; this /// only re-reads, refreshes the WorkManager cache and broadcasts. fn refresh_work_plan(&self, session_id: &str, item_id: Option) { let Some(work_manager) = self.work_manager.as_ref() else { return; }; let Some(item_id) = item_id else { return; }; let work_manager = work_manager.clone(); let session_id = session_id.to_string(); tokio::spawn(async move { if let Err(error) = work_manager .refresh_after_external_commit(&session_id, "agent_run", vec![item_id]) .await { tracing::warn!(error = %error, "failed to refresh plan after agent run commit"); } }); } /// Cancel a nonterminal run owned by the caller's session/tree. pub async fn cancel_run( &self, caller: &ToolExecutionContext, run_id: &str, reason: &str, ) -> Result { self.cancel_run_inner(caller, run_id, reason, false).await } /// `suppress_continuation` writes the completion event consumed so no /// continuation Turn restarts after lifecycle cancellation (`/stop`). pub(crate) async fn cancel_run_inner( &self, caller: &ToolExecutionContext, run_id: &str, reason: &str, suppress_continuation: bool, ) -> Result { let Some(run) = self.storage.get_agent_run(run_id).await? else { return Ok(false); }; self.authorize_access(caller, &run).await?; if run.status.is_terminal() { return Ok(false); } let cancelled = self .storage .cancel_agent_run_with_completion( run_id, reason, suppress_continuation, chrono::Utc::now().timestamp_millis(), ) .await?; if cancelled { if let Some((_, token)) = self.active_tokens.remove(run_id) { token.cancel(); } // An explicit cancel supersedes the run's unconsumed signals: // they remain as audit facts but no continuation will report // them. Completions are never superseded. if let Err(error) = self .storage .supersede_agent_events( run_id, AgentEventType::Signal, chrono::Utc::now().timestamp_millis(), ) .await { tracing::warn!(run_id, error = %error, "failed to supersede signals on cancel"); } } Ok(cancelled) } /// Persist one signal from a running Agent into its root session inbox. /// The tool only exists for runs with a signal contract; this method is /// the single writer that enforces capacity and generates the wake. pub async fn emit_signal( &self, context: &AgentExecutionContext, input: SignalInput, ) -> Result { let Some(run) = self.storage.get_agent_run(&context.run_id).await? else { return Err(CoordinatorError::Rejected( "signal rejected: run no longer exists".to_string(), )); }; if run.execution_id != context.execution_id { return Err(CoordinatorError::Rejected( "signal rejected: stale execution".to_string(), )); } if run.status.is_terminal() { return Err(CoordinatorError::Rejected( "signal rejected: run is no longer active".to_string(), )); } let now = chrono::Utc::now().timestamp_millis(); let event = crate::tools::emit_signal::build_signal_event( context, &input, Uuid::new_v4().to_string(), run.signal_delivery .as_deref() .map(crate::storage::agent_inbox::AgentEventDelivery::parse) .transpose() .map_err(CoordinatorError::from)? .unwrap_or(crate::storage::agent_inbox::AgentEventDelivery::Queue), ); match self .storage .insert_agent_signal(&event, self.max_pending_inbox_events_per_session, now) .await? { Some((record, deduplicated)) => { if deduplicated { return Ok(SignalAccepted { signal_id: record.id, status: SignalAcceptedStatus::Deduplicated, delivery: event.delivery, }); } self.projection.publish(AgentProjection { session_id: run.root_session_id.clone(), revision: record.revision, run: None, event: Some(crate::protocol::AgentEventView::from_record(&record)), }); self.notifier .notify(&run.root_session_id, record.revision) .await; Ok(SignalAccepted { signal_id: record.id, status: SignalAcceptedStatus::Accepted, delivery: event.delivery, }) } None => Err(CoordinatorError::Rejected( "inbox capacity exceeded; signal rejected".to_string(), )), } } pub async fn get_run( &self, caller: &ToolExecutionContext, run_id: &str, ) -> Result, CoordinatorError> { let Some(run) = self.storage.get_agent_run(run_id).await? else { return Ok(None); }; self.authorize_access(caller, &run).await?; Ok(Some(run)) } pub async fn list_runs( &self, caller: &ToolExecutionContext, cursor: Option<(i64, String)>, limit: i64, ) -> Result, CoordinatorError> { let session_id = caller_session(caller)?; self.storage .list_agent_runs(&session_id, cursor, limit) .await .map_err(CoordinatorError::Storage) } /// Session-scoped run projection for management/WebSocket clients. pub async fn list_runs_for_session( &self, session_id: &str, cursor: Option<(i64, String)>, limit: i64, ) -> Result<(i64, Vec, Option), CoordinatorError> { let runs = self .storage .list_agent_runs(session_id, cursor, limit) .await?; let next_cursor = runs .last() .map(|run| format!("{}:{}", run.created_at, run.id)); let views = runs .iter() .map(|run| crate::protocol::AgentRunView::from_record(run, 2_000)) .collect(); let revision = self.storage.get_session_agent_revision(session_id).await?; Ok((revision, views, next_cursor)) } /// Session-scoped single-run projection. pub async fn get_run_for_session( &self, session_id: &str, run_id: &str, ) -> Result<(i64, Option), CoordinatorError> { let revision = self.storage.get_session_agent_revision(session_id).await?; let run = self.storage.get_agent_run(run_id).await?; let run = run.filter(|run| run.root_session_id == session_id); Ok(( revision, run.map(|run| crate::protocol::AgentRunView::from_record(&run, 2_000)), )) } /// Inbox events of one run (audit/projection). pub async fn list_run_events( &self, run_id: &str, limit: i64, ) -> Result, CoordinatorError> { self.storage .list_agent_inbox_events_for_run(run_id, limit) .await .map_err(CoordinatorError::Storage) } /// Management-API cancel: session-scoped, suppresses continuation so a /// cancelled run never restarts a background Turn by itself. pub async fn cancel_run_for_session( &self, session_id: &str, run_id: &str, reason: &str, ) -> Result { let Some(run) = self.storage.get_agent_run(run_id).await? else { return Ok(false); }; if run.root_session_id != session_id { return Err(CoordinatorError::Rejected( "run does not belong to this session".to_string(), )); } if run.status.is_terminal() { return Ok(false); } let caller = ToolExecutionContext::for_session(session_id); self.cancel_run_inner(&caller, run_id, reason, true).await } /// Full durable result for a finished run; `None` while nonterminal. pub async fn get_result( &self, caller: &ToolExecutionContext, run_id: &str, ) -> Result, CoordinatorError> { let Some(run) = self.get_run(caller, run_id).await? else { return Ok(None); }; if run.status.is_terminal() { Ok(Some(run)) } else { Ok(None) } } /// Root may access every run of its session; a named Agent may only /// access its own run and descendants. Run IDs are never credentials. async fn authorize_access( &self, caller: &ToolExecutionContext, run: &AgentRunRecord, ) -> Result<(), CoordinatorError> { let Some(agent) = caller.agent.as_ref() else { let session_id = caller_session(caller)?; if run.root_session_id != session_id { return Err(CoordinatorError::Rejected( "run belongs to a different session".to_string(), )); } return Ok(()); }; if run.root_session_id != agent.root_session_id { return Err(CoordinatorError::Rejected( "run belongs to a different session".to_string(), )); } let mut current = run.clone(); loop { if current.id == agent.run_id { return Ok(()); } let Some(parent_id) = current.parent_run_id.clone() else { return Err(CoordinatorError::Rejected( "run is not part of the caller's delegation tree".to_string(), )); }; let Some(parent) = self.storage.get_agent_run(&parent_id).await? else { return Err(CoordinatorError::Rejected( "run ancestry is corrupt".to_string(), )); }; current = parent; } } } fn caller_session(caller: &ToolExecutionContext) -> Result { caller .agent .as_ref() .map(|agent| agent.root_session_id.clone()) .or_else(|| caller.session_id.clone()) .ok_or_else(|| { CoordinatorError::Rejected("caller context is not session-bound".to_string()) }) } fn truncate_summary(content: &str) -> String { const MAX: usize = 500; if content.len() <= MAX { content.to_string() } else { let cut = content.floor_char_boundary(MAX); format!("{}...", &content[..cut]) } } #[cfg(test)] mod tests { use super::*; use crate::agent::AgentCatalog; use crate::config::LLMProviderConfig; use crate::tools::ToolRegistry; use std::collections::HashMap; fn provider_config() -> LLMProviderConfig { LLMProviderConfig { provider_type: "openai".into(), name: "test".into(), base_url: "http://localhost".into(), api_key: "test".into(), extra_headers: HashMap::new(), model_id: "test".into(), temperature: None, max_tokens: None, model_extra: HashMap::new(), max_tool_iterations: 1, token_limit: 4096, workspace_dir: std::env::temp_dir(), input_types: vec!["text".into()], price_input_per_million: None, price_output_per_million: None, } } fn write_catalog(root: &std::path::Path) -> AgentCatalog { std::fs::create_dir_all(root.join("agents")).unwrap(); std::fs::write( root.join("agents/researcher.md"), "---\nid: researcher\ndescription: research role\nllm_profile: research\n---\n# Role\n\nDo the assigned work.\n", ) .unwrap(); let tools = ToolRegistry::new(); let loader = crate::skills::SkillsLoader::new_for_testing( root.join("skills"), root.join("external-skills"), ); let profiles = HashMap::from([("research".to_string(), provider_config())]); let config = crate::config::AgentOrchestrationConfig { definitions_dir: "agents".to_string(), ..Default::default() }; AgentCatalog::load( &config, root, &profiles, &HashMap::new(), &HashMap::new(), root, &tools, &loader, 1, ) .unwrap() } async fn coordinator() -> (Arc, tempfile::TempDir) { coordinator_with_inbox_limit(1).await } async fn coordinator_with_inbox_limit( max_pending: usize, ) -> (Arc, tempfile::TempDir) { coordinator_with_inbox_and_run_limit(max_pending, None).await } async fn coordinator_with_inbox_and_run_limit( max_pending: usize, max_concurrent_runs: Option, ) -> (Arc, tempfile::TempDir) { let dir = tempfile::tempdir().unwrap(); let storage = Arc::new(Storage::new(&dir.path().join("coord.db")).await.unwrap()); let catalog = write_catalog(dir.path()); let manager = Arc::new( SubAgentManager::new( provider_config(), Arc::new(ToolRegistry::new()), Some(storage.clone()), None, ) .with_catalog(Arc::new(catalog)), ); let work_manager = Arc::new(crate::work::WorkManager::new(storage.clone())); let notifier = crate::agent::AgentInboxNotifier::new(); let supervisor = crate::task_supervisor::TaskSupervisor::new(); let orchestration = crate::config::AgentOrchestrationConfig { max_pending_inbox_events_per_session: max_pending, ..Default::default() }; let gate = match max_concurrent_runs { Some(limit) => { let config = crate::config::AgentOrchestrationConfig { max_concurrent_runs: limit, ..Default::default() }; crate::agent::gate::ExecutionGate::new(&config) } None => crate::agent::gate::ExecutionGate::unbounded(), }; ( AgentCoordinator::new( storage, manager, work_manager, notifier, Arc::new(crate::agent::AgentProjectionHub::new()), gate, crate::gateway::reload::RuntimeAdmission::open(), supervisor, 1, &orchestration, ), dir, ) } fn foreground_config(target: &str) -> SubAgentConfig { SubAgentConfig { target: Some(target.to_string()), prompt: "work".to_string(), context: None, mode: ExecutionMode::Foreground, allowed_tools: None, max_iterations: Some(1), timeout_secs: Some(5), plan_item_id: None, session_id: Some("cli:test:dialog".to_string()), } } #[tokio::test] async fn foreground_run_is_persisted_with_terminal_state() { let (coordinator, _dir) = coordinator().await; let caller = ToolExecutionContext::for_session("cli:test:dialog"); let results = coordinator .delegate_foreground(&caller, vec![foreground_config("researcher")]) .await .unwrap(); assert_eq!(results.len(), 1); let run_id = &results[0].task_id; let run = coordinator.get_run(&caller, run_id).await.unwrap().unwrap(); assert_eq!(run.agent_id, "researcher"); assert_eq!(run.mode, AgentRunMode::Foreground); // Provider is unreachable in this test, so the run must fail — but it // must still be persisted with a terminal status and full audit row. assert!(run.status.is_terminal()); assert!(matches!(run.status, AgentRunStatus::Failed)); assert!(run.finished_at.is_some()); } #[tokio::test] async fn batch_creates_group_and_keeps_request_order() { let (coordinator, _dir) = coordinator().await; let caller = ToolExecutionContext::for_session("cli:test:dialog"); let results = coordinator .delegate_foreground( &caller, vec![ foreground_config("researcher"), foreground_config("researcher"), ], ) .await .unwrap(); assert_eq!(results.len(), 2); let first = coordinator .get_run(&caller, &results[0].task_id) .await .unwrap() .unwrap(); let second = coordinator .get_run(&caller, &results[1].task_id) .await .unwrap() .unwrap(); assert_ne!(first.id, second.id); } #[tokio::test] async fn other_session_cannot_read_or_cancel_runs() { let (coordinator, _dir) = coordinator().await; let caller = ToolExecutionContext::for_session("cli:test:dialog"); let results = coordinator .delegate_foreground(&caller, vec![foreground_config("researcher")]) .await .unwrap(); let run_id = results[0].task_id.clone(); let stranger = ToolExecutionContext::for_session("cli:other:dialog"); let error = coordinator.get_run(&stranger, &run_id).await.unwrap_err(); assert!(matches!(error, CoordinatorError::Rejected(_))); let error = coordinator .cancel_run(&stranger, &run_id, "x") .await .unwrap_err(); assert!(matches!(error, CoordinatorError::Rejected(_))); } #[tokio::test] async fn unknown_target_is_rejected_before_persistence() { let (coordinator, _dir) = coordinator().await; let caller = ToolExecutionContext::for_session("cli:test:dialog"); let error = coordinator .delegate_foreground(&caller, vec![foreground_config("missing")]) .await .unwrap_err(); assert!(matches!(error, CoordinatorError::SubAgent(_))); assert!( coordinator .list_runs(&caller, None, 10) .await .unwrap() .is_empty() ); } #[tokio::test] async fn background_admission_persists_and_reserves_completion() { let (coordinator, _dir) = coordinator().await; let caller = ToolExecutionContext::for_session("cli:test:dialog"); let admission = coordinator .delegate_background(&caller, vec![foreground_config("researcher")]) .await .unwrap(); assert_eq!(admission.run_ids.len(), 1); let run_id = &admission.run_ids[0]; let run = coordinator.get_run(&caller, run_id).await.unwrap().unwrap(); assert_eq!(run.mode, AgentRunMode::Background); assert!(run.completion_slot_reserved); // A second background run while the inbox is at capacity must be // rejected, not admitted silently. let error = coordinator .delegate_background(&caller, vec![foreground_config("researcher")]) .await .unwrap_err(); assert!(matches!(error, CoordinatorError::Rejected(_))); // Wait for the spawned runner to finish (provider is unreachable, so // it fails fast) and verify the terminal commit converted the // reservation into a durable completion event. let deadline = std::time::Instant::now() + std::time::Duration::from_secs(10); loop { let run = coordinator.get_run(&caller, run_id).await.unwrap().unwrap(); if run.status.is_terminal() { break; } assert!( std::time::Instant::now() < deadline, "background runner did not finish in time" ); tokio::time::sleep(std::time::Duration::from_millis(50)).await; } let events = coordinator .storage .list_agent_inbox_events("cli:test:dialog", 10) .await .unwrap(); assert_eq!(events.len(), 1); assert_eq!( events[0].status, crate::storage::agent_inbox::AgentEventStatus::Pending ); assert!(events[0].requires_continuation); let state: (i64, i64) = sqlx::query_as( "SELECT pending_event_count, reserved_completion_slots FROM agent_session_state \ WHERE root_session_id = 'cli:test:dialog'", ) .fetch_one(coordinator.storage.pool()) .await .unwrap(); // Reservation converted: nothing reserved, one event pending. assert_eq!(state, (1, 0)); } #[tokio::test] async fn background_batch_admits_all_runs_and_creates_a_group() { let (coordinator, _dir) = coordinator_with_inbox_limit(8).await; let caller = ToolExecutionContext::for_session("cli:test:dialog"); let configs = vec![ foreground_config("researcher"), foreground_config("researcher"), foreground_config("researcher"), ]; let admission = coordinator .delegate_background(&caller, configs) .await .unwrap(); assert_eq!(admission.run_ids.len(), 3); let runs = coordinator.list_runs(&caller, None, 10).await.unwrap(); assert_eq!(runs.len(), 3); assert!(runs.iter().all(|run| run.completion_slot_reserved)); } #[tokio::test] async fn background_batch_rejects_when_exceeding_run_quota() { let (coordinator, _dir) = coordinator_with_inbox_and_run_limit(16, Some(2)).await; let caller = ToolExecutionContext::for_session("cli:test:dialog"); let error = coordinator .delegate_background( &caller, vec![ foreground_config("researcher"), foreground_config("researcher"), foreground_config("researcher"), ], ) .await .unwrap_err(); assert!(matches!(error, CoordinatorError::Rejected(_))); } async fn storage_accept_run_with_delivery( storage: &Arc, run_id: &str, session: &str, now: i64, slot_reserved: bool, signal_delivery: Option<&str>, ) { let _ = storage.ensure_agent_session_state(session, now).await; if slot_reserved { let _ = storage.reserve_completion_slots(session, 1, 1, now).await; } let run = NewAgentRun { id: run_id.to_string(), root_session_id: session.to_string(), root_turn_id: None, parent_run_id: None, caller_agent_id: "ROOT".to_string(), caller_scope_id: "turn-1".to_string(), idempotency_key: None, agent_id: "researcher".to_string(), definition_hash: "hash".to_string(), provider_profile: "research".to_string(), provider_name: "test".to_string(), model_id: "test-model".to_string(), mode: AgentRunMode::Background, depth: 1, plan_item_id: None, execution_id: run_id.to_string(), task: "work".to_string(), context_json: None, budget_json: "{}".to_string(), signal_contract_json: signal_delivery.map(|_| "{}".to_string()), signal_delivery: signal_delivery.map(str::to_string), deadline_at: now + 100_000, runtime_generation: 1, completion_slot_reserved: slot_reserved, }; let _ = storage .accept_agent_runs(AcceptAgentRequest { runs: vec![run], now, }) .await .unwrap(); } async fn storage_accept_run( storage: &Arc, run_id: &str, session: &str, now: i64, slot_reserved: bool, ) { storage_accept_run_with_delivery(storage, run_id, session, now, slot_reserved, None).await } #[tokio::test] async fn emit_signal_persists_wakes_and_respects_capacity_and_dedupe() { let (coordinator, _dir) = coordinator().await; let run_id = "run-sig-1"; let session = "cli:test:dialog"; storage_accept_run( &coordinator.storage, run_id, session, chrono::Utc::now().timestamp_millis(), false, ) .await; let context = crate::agent::AgentExecutionContext { root_session_id: session.to_string(), root_turn_id: None, run_id: run_id.to_string(), execution_id: run_id.to_string(), parent_run_id: None, caller_agent_id: "ROOT".to_string(), current_agent_id: "researcher".to_string(), ancestry: vec!["researcher".to_string()], depth: 1, plan_item_id: None, cancellation: tokio_util::sync::CancellationToken::new(), budget: crate::agent::AgentBudget { remaining_runs: 15, remaining_depth: 3, }, tree_runs: Arc::new(std::sync::atomic::AtomicUsize::new(1)), signal_contract: None, emitted_signals: Arc::new(std::sync::Mutex::new(Vec::new())), }; let signal = SignalInput { key: "err-rate".to_string(), severity: "warning".to_string(), summary: "error rate above threshold".to_string(), details: Some(serde_json::json!({ "current": 0.071 })), dedupe_key: Some("svc-a:err".to_string()), event_key: format!("signal:svc-a:err:{}", 0), }; let accepted = coordinator .emit_signal(&context, signal.clone()) .await .unwrap(); assert_eq!(accepted.status, SignalAcceptedStatus::Accepted); assert!(matches!( accepted.delivery, crate::storage::agent_inbox::AgentEventDelivery::Queue )); // Same dedupe key in the same window collapses to the same event. let mut duplicate = signal.clone(); duplicate.event_key = format!("signal:svc-a:err:{}", 0); let deduplicated = coordinator.emit_signal(&context, duplicate).await.unwrap(); assert_eq!(deduplicated.status, SignalAcceptedStatus::Deduplicated); assert_eq!(deduplicated.signal_id, accepted.signal_id); // A stale execution id is rejected. let mut stale = context.clone(); stale.execution_id = "other-exec".to_string(); assert!(matches!( coordinator.emit_signal(&stale, signal.clone()).await, Err(CoordinatorError::Rejected(_)) )); } #[tokio::test] async fn terminal_commit_carries_emitted_signal_ids_in_completion_payload() { let (coordinator, _dir) = coordinator_with_inbox_limit(8).await; let run_id = "run-sig-carrier"; let session = "cli:test:dialog"; let now = chrono::Utc::now().timestamp_millis(); storage_accept_run(&coordinator.storage, run_id, session, now, true).await; // Emit one signal, then commit the run terminal. let context = crate::agent::AgentExecutionContext { root_session_id: session.to_string(), root_turn_id: None, run_id: run_id.to_string(), execution_id: run_id.to_string(), parent_run_id: None, caller_agent_id: "ROOT".to_string(), current_agent_id: "researcher".to_string(), ancestry: vec!["researcher".to_string()], depth: 1, plan_item_id: None, cancellation: tokio_util::sync::CancellationToken::new(), budget: crate::agent::AgentBudget { remaining_runs: 15, remaining_depth: 3, }, tree_runs: Arc::new(std::sync::atomic::AtomicUsize::new(1)), signal_contract: None, emitted_signals: Arc::new(std::sync::Mutex::new(Vec::new())), }; let signal = SignalInput { key: "k".to_string(), severity: "warning".to_string(), summary: "s".to_string(), details: None, dedupe_key: None, event_key: format!("signal:{}", uuid::Uuid::new_v4()), }; let accepted = coordinator.emit_signal(&context, signal).await.unwrap(); context .emitted_signals .lock() .unwrap() .push(crate::agent::run::EmittedSignal { signal_id: accepted.signal_id.clone(), }); let signal_ids: Vec = context .emitted_signals .lock() .unwrap() .iter() .map(|signal| signal.signal_id.clone()) .collect(); coordinator .storage .commit_agent_terminal( run_id, run_id, 1, &AgentTerminalOutcome::Completed { result: "done".to_string(), prompt_tokens: None, completion_tokens: None, cost: None, tool_calls: 1, iterations: 1, signal_ids, }, None, now + 1, ) .await .unwrap(); let events = coordinator .storage .list_agent_inbox_events(session, 10) .await .unwrap(); assert_eq!(events.len(), 2); let completion = events .iter() .find(|event| { event.event_type == crate::storage::agent_inbox::AgentEventType::Completion }) .unwrap(); let payload: serde_json::Value = serde_json::from_str(&completion.payload_json).unwrap(); assert_eq!(payload["status"], "completed"); assert_eq!(payload["signal_ids"][0], accepted.signal_id); } #[tokio::test] async fn emit_signal_rejects_when_run_is_terminal_or_inbox_is_full() { let (coordinator, _dir) = coordinator().await; let run_id = "run-sig-2"; let session = "cli:test:dialog"; let now = chrono::Utc::now().timestamp_millis(); storage_accept_run(&coordinator.storage, run_id, session, now, false).await; let context = crate::agent::AgentExecutionContext { root_session_id: session.to_string(), root_turn_id: None, run_id: run_id.to_string(), execution_id: run_id.to_string(), parent_run_id: None, caller_agent_id: "ROOT".to_string(), current_agent_id: "researcher".to_string(), ancestry: vec!["researcher".to_string()], depth: 1, plan_item_id: None, cancellation: tokio_util::sync::CancellationToken::new(), budget: crate::agent::AgentBudget { remaining_runs: 15, remaining_depth: 3, }, tree_runs: Arc::new(std::sync::atomic::AtomicUsize::new(1)), signal_contract: None, emitted_signals: Arc::new(std::sync::Mutex::new(Vec::new())), }; let signal = SignalInput { key: "k".to_string(), severity: "info".to_string(), summary: "s".to_string(), details: None, dedupe_key: None, event_key: format!("signal:{}", uuid::Uuid::new_v4()), }; // Terminal run: rejected. coordinator .storage .commit_agent_terminal( run_id, run_id, 1, &AgentTerminalOutcome::Cancelled { reason: "test".to_string(), signal_ids: Vec::new(), }, None, now + 1, ) .await .unwrap(); assert!(matches!( coordinator.emit_signal(&context, signal.clone()).await, Err(CoordinatorError::Rejected(_)) )); // Capacity: a reserved completion slot exhausts the session limit. let coordinator2 = { let dir = tempfile::tempdir().unwrap(); let storage2 = Arc::new(Storage::new(&dir.path().join("c2.db")).await.unwrap()); let catalog = write_catalog(dir.path()); let manager = Arc::new( SubAgentManager::new( provider_config(), Arc::new(ToolRegistry::new()), Some(storage2.clone()), None, ) .with_catalog(Arc::new(catalog)), ); let work_manager = Arc::new(crate::work::WorkManager::new(storage2.clone())); let notifier = crate::agent::AgentInboxNotifier::new(); let supervisor = crate::task_supervisor::TaskSupervisor::new(); let orchestration = crate::config::AgentOrchestrationConfig { max_pending_inbox_events_per_session: 1, ..Default::default() }; AgentCoordinator::new( storage2, manager, work_manager, notifier, Arc::new(crate::agent::AgentProjectionHub::new()), crate::agent::gate::ExecutionGate::unbounded(), crate::gateway::reload::RuntimeAdmission::open(), supervisor, 1, &orchestration, ) }; storage_accept_run(&coordinator2.storage, "run-sig-3", session, now, true).await; let context2 = crate::agent::AgentExecutionContext { root_session_id: session.to_string(), root_turn_id: None, run_id: "run-sig-3".to_string(), execution_id: "run-sig-3".to_string(), parent_run_id: None, caller_agent_id: "ROOT".to_string(), current_agent_id: "researcher".to_string(), ancestry: vec!["researcher".to_string()], depth: 1, plan_item_id: None, cancellation: tokio_util::sync::CancellationToken::new(), budget: crate::agent::AgentBudget { remaining_runs: 15, remaining_depth: 3, }, tree_runs: Arc::new(std::sync::atomic::AtomicUsize::new(1)), signal_contract: None, emitted_signals: Arc::new(std::sync::Mutex::new(Vec::new())), }; assert!(matches!( coordinator2.emit_signal(&context2, signal).await, Err(CoordinatorError::Rejected(_)) )); } #[tokio::test] async fn emitted_signal_uses_the_runs_persisted_delivery_lane() { let (coordinator, _dir) = coordinator_with_inbox_limit(8).await; let run_id = "run-sig-steer"; let session = "cli:test:dialog"; let now = chrono::Utc::now().timestamp_millis(); storage_accept_run_with_delivery( &coordinator.storage, run_id, session, now, false, Some("steer"), ) .await; let run = coordinator .storage .get_agent_run(run_id) .await .unwrap() .unwrap(); assert_eq!(run.signal_delivery.as_deref(), Some("steer")); let context = crate::agent::AgentExecutionContext { root_session_id: session.to_string(), root_turn_id: None, run_id: run_id.to_string(), execution_id: run_id.to_string(), parent_run_id: None, caller_agent_id: "ROOT".to_string(), current_agent_id: "researcher".to_string(), ancestry: vec!["researcher".to_string()], depth: 1, plan_item_id: None, cancellation: tokio_util::sync::CancellationToken::new(), budget: crate::agent::AgentBudget { remaining_runs: 15, remaining_depth: 3, }, tree_runs: Arc::new(std::sync::atomic::AtomicUsize::new(1)), signal_contract: None, emitted_signals: Arc::new(std::sync::Mutex::new(Vec::new())), }; let signal = SignalInput { key: "k".to_string(), severity: "info".to_string(), summary: "s".to_string(), details: None, dedupe_key: None, event_key: format!("signal:{}", uuid::Uuid::new_v4()), }; let accepted = coordinator.emit_signal(&context, signal).await.unwrap(); assert!(matches!( accepted.delivery, crate::storage::agent_inbox::AgentEventDelivery::Steer )); let event = coordinator .storage .get_agent_inbox_event(&accepted.signal_id) .await .unwrap() .unwrap(); assert_eq!( event.delivery, crate::storage::agent_inbox::AgentEventDelivery::Steer ); } }