diff --git a/AGENTS.md b/AGENTS.md index 868a84aa..76872820 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -129,6 +129,7 @@ Key invariants: - Progressive summary (if present) is prepended to the messages array as a user/assistant pair in `translator.rs` - Loop prevention guardrail in `controller.rs` detects repeated tool failures (3+ identical) and injects corrective instructions - Direct-provider long-running shell follow-ups create unlinked CLI tasks under the root task with an empty subagent tool-call ID; `TaskStore` linearization must include their exchanges chronologically even though no parent `Subagent` output references them +- Direct-provider completed-command assessments are hidden, tool-free root-task turns; CLI monitor exchanges remain on the retained CLI task, while the root assessment output must survive CLI-task deactivation and restoration and its hidden input must remain available to future provider context - Orchestrated child conversations are leaf workers by default: nested `RunAgents` and legacy `StartAgent` calls must be rejected before autonomous or permission bypasses, and child requests must not advertise delegation tools ### Platform Setup diff --git a/app/src/ai/acp/prompt.rs b/app/src/ai/acp/prompt.rs index 79aa4b76..e1a020a3 100644 --- a/app/src/ai/acp/prompt.rs +++ b/app/src/ai/acp/prompt.rs @@ -153,6 +153,25 @@ fn append_hidden_input( )); } AIAgentInput::UserQuery { .. } | AIAgentInput::CreateNewProject { .. } => {} + AIAgentInput::CommandCompletionAssessment { + prompt, + completed_command, + .. + } => { + hidden_context.push(format!( + "A monitored command has completed.\n\ + Command: {}\n\ + Galaxy block_id: {}\n\ + Final output:\n{}\n\n{}", + completed_command.command, + completed_command.block_id, + tail_chars( + &completed_command.grid_contents, + MAX_RUNNING_COMMAND_OUTPUT_CHARS + ), + prompt, + )); + } AIAgentInput::AutoCodeDiffQuery { query, .. } => { hidden_context.push(format!( "Galaxy system request: create a code diff.\n{query}" diff --git a/app/src/ai/acp/prompt_tests.rs b/app/src/ai/acp/prompt_tests.rs index 37e14b28..5a8f4ce8 100644 --- a/app/src/ai/acp/prompt_tests.rs +++ b/app/src/ai/acp/prompt_tests.rs @@ -117,6 +117,46 @@ fn hidden_system_requests_still_reach_the_agent_without_a_user_bubble() { assert!(text.contains("hidden_from_transcript")); } +#[test] +fn completed_command_assessment_uses_hidden_context_without_monitor_guidance() { + let block_id = BlockId::from("completed-session-42".to_owned()); + let mut params = RequestParams::new_for_test(); + params.input = vec![AIAgentInput::CommandCompletionAssessment { + prompt: "Report whether the command succeeded.".to_owned(), + context: Arc::from([AIAgentContext::SelectedText("root context".to_owned())]), + completed_command: RunningCommand { + command: "script/run-soak-test".to_owned(), + block_id: block_id.clone(), + grid_contents: "completed successfully".to_owned(), + cursor: String::new(), + requested_command_id: None, + is_alt_screen_active: false, + }, + }]; + + let prompt = prompt_content( + ¶ms, + GalaxyTerminalTools { + status: true, + interrupt: true, + }, + ) + .expect("prompt"); + let text = prompt_text(&prompt); + + assert!(text.starts_with("Handle the Galaxy system request")); + assert!(text.contains("hidden_from_transcript")); + assert!(text.contains("A monitored command has completed.")); + assert!(text.contains("script/run-soak-test")); + assert!(text.contains(block_id.as_str())); + assert!(text.contains("Final output:\ncompleted successfully")); + assert!(text.contains("Report whether the command succeeded.")); + assert!(text.contains("Selected text:\nroot context")); + assert!(!text.contains("galaxy_terminal_status")); + assert!(!text.contains("galaxy_terminal_interrupt")); + assert!(!text.contains("running_for_ms")); +} + #[test] fn running_command_identity_and_output_are_sent_as_hidden_context() { let block_id = BlockId::from("session-42".to_owned()); diff --git a/app/src/ai/agent/api.rs b/app/src/ai/agent/api.rs index 28295f4e..7fc0f762 100644 --- a/app/src/ai/agent/api.rs +++ b/app/src/ai/agent/api.rs @@ -37,6 +37,21 @@ use crate::settings::AISettings; use crate::terminal::safe_mode_settings::get_secret_obfuscation_mode; use crate::workspaces::user_workspaces::UserWorkspaces; +const INTERNAL_COMMAND_COMPLETION_ASSESSMENT_MESSAGE_DATA: &str = + "galaxy:internal-command-completion-assessment:v1"; + +pub(crate) fn mark_internal_command_completion_assessment( + message: &mut warp_multi_agent_api::Message, +) { + message.server_message_data = INTERNAL_COMMAND_COMPLETION_ASSESSMENT_MESSAGE_DATA.to_string(); +} + +pub(crate) fn is_internal_command_completion_assessment( + message: &warp_multi_agent_api::Message, +) -> bool { + message.server_message_data == INTERNAL_COMMAND_COMPLETION_ASSESSMENT_MESSAGE_DATA +} + /// Unique, server-generated conversation-scoped token to be roundtripped to the API when sending /// requests that follow-up within a given conversation. #[derive(Serialize, Debug, Clone, PartialEq, Eq, Hash)] diff --git a/app/src/ai/agent/api/convert_conversation.rs b/app/src/ai/agent/api/convert_conversation.rs index 082763c7..4d118147 100644 --- a/app/src/ai/agent/api/convert_conversation.rs +++ b/app/src/ai/agent/api/convert_conversation.rs @@ -23,6 +23,7 @@ use crate::ai::agent::api::convert_from::{ convert_user_query_mode, ConversionParams, ConvertAPIMessageToClientOutputMessage, MaybeAIAgentOutputMessage, }; +use crate::ai::agent::api::is_internal_command_completion_assessment; use crate::ai::agent::conversation::{ update_todo_list_from_todo_op, AIConversation, AIConversationId, ServerAIConversationMetadata, }; @@ -389,17 +390,21 @@ impl ConvertToExchanges for &api::Task { let added_message_as_exchange_input = match message { api::message::Message::UserQuery(user_query) => { - // Add user query as input - current_inputs.push(AIAgentInput::UserQuery { + if is_internal_command_completion_assessment(api_message) { + false + } else { + // Add user query as input + current_inputs.push(AIAgentInput::UserQuery { query: user_query.query.clone(), context: convert_input_context(user_query.context.as_ref()), static_query_type: None, referenced_attachments: HashMap::new(), user_query_mode: convert_user_query_mode(user_query.mode.as_ref()), running_command: None, - intended_agent: Some(user_query.intended_agent()), - }); - true + intended_agent: Some(user_query.intended_agent()), + }); + true + } } api::message::Message::SystemQuery(query) => { let Some(query_type) = &query.r#type else { diff --git a/app/src/ai/agent/api/convert_conversation_tests.rs b/app/src/ai/agent/api/convert_conversation_tests.rs index 344815ca..59d1e26c 100644 --- a/app/src/ai/agent/api/convert_conversation_tests.rs +++ b/app/src/ai/agent/api/convert_conversation_tests.rs @@ -4,7 +4,7 @@ use chrono::Utc; use warp_multi_agent_api as api; use crate::ai::agent::api::convert_conversation::*; -use crate::ai::agent::api::ServerConversationToken; +use crate::ai::agent::api::{mark_internal_command_completion_assessment, ServerConversationToken}; use crate::ai::agent::conversation::{ AIAgentHarness, AIConversationId, ServerAIConversationMetadata, }; @@ -2129,6 +2129,61 @@ fn test_create_then_edit_then_create_version_tracking() { ); } +#[test] +fn test_internal_command_completion_assessment_restores_output_without_visible_input() { + let assessment_text = + "[Completed command: cargo test]\n[Final terminal output:\ntest result: ok\n]"; + let mut hidden_assessment = api::Message { + id: "msg_assessment".to_string(), + task_id: "task1".to_string(), + request_id: "req1".to_string(), + message: Some(api::message::Message::UserQuery(api::message::UserQuery { + query: assessment_text.to_string(), + ..Default::default() + })), + ..Default::default() + }; + mark_internal_command_completion_assessment(&mut hidden_assessment); + let provider_history = + crate::ai::bedrock::request_translator::convert_proto_message(&hidden_assessment) + .expect("hidden assessment should remain in provider history"); + assert_eq!( + provider_history.role, + crate::ai::provider::types::MessageRole::User + ); + assert!(matches!( + provider_history.content, + crate::ai::provider::types::MessageContent::Text(text) if text == assessment_text + )); + + let task = api::Task { + id: "task1".to_string(), + messages: vec![ + hidden_assessment, + api::Message { + id: "msg_output".to_string(), + task_id: "task1".to_string(), + request_id: "req1".to_string(), + message: Some(api::message::Message::AgentOutput( + api::message::AgentOutput { + text: "The command completed successfully.".to_string(), + }, + )), + ..Default::default() + }, + ], + ..Default::default() + }; + + let exchanges = task.into_exchanges(); + assert_eq!(exchanges.len(), 1); + assert!(exchanges[0].input.is_empty()); + assert_eq!( + exchanges[0].format_output_for_copy(None), + "The command completed successfully." + ); +} + /// Verify that a `SystemQuery::HandoffRehydration` message does not produce /// a displayed input when restoring a conversation. It must be treated as /// hidden, so the exchange should have zero user-visible inputs. diff --git a/app/src/ai/agent/api/convert_from.rs b/app/src/ai/agent/api/convert_from.rs index a8187493..f8714c8d 100644 --- a/app/src/ai/agent/api/convert_from.rs +++ b/app/src/ai/agent/api/convert_from.rs @@ -16,6 +16,7 @@ use warp_multi_agent_api as api; use crate::ai::agent::api::convert_conversation::{ convert_input_context, convert_tool_call_result_to_input, }; +use crate::ai::agent::api::is_internal_command_completion_assessment; use crate::ai::agent::comment::CodeReview; use crate::ai::agent::task::TaskId; use crate::ai::agent::todos::AIAgentTodoList; @@ -957,6 +958,9 @@ pub fn user_inputs_from_messages(messages: &[api::Message]) -> Vec let Some(inner) = &m.message else { continue }; match inner { api::message::Message::UserQuery(uq) => { + if is_internal_command_completion_assessment(m) { + continue; + } let context = convert_input_context(uq.context.as_ref()); let referenced_attachments = uq .referenced_attachments diff --git a/app/src/ai/agent/api/convert_from_tests.rs b/app/src/ai/agent/api/convert_from_tests.rs index a421e478..e2aa11c5 100644 --- a/app/src/ai/agent/api/convert_from_tests.rs +++ b/app/src/ai/agent/api/convert_from_tests.rs @@ -7,13 +7,14 @@ use warp_multi_agent_api as api; use warp_util::local_or_remote_path::LocalOrRemotePath; use super::{ - convert_api_question, ConversionParams, ConvertAPIMessageToClientOutputMessage, - MaybeAIAgentOutputMessage, + convert_api_question, user_inputs_from_messages, ConversionParams, + ConvertAPIMessageToClientOutputMessage, MaybeAIAgentOutputMessage, }; +use crate::ai::agent::api::mark_internal_command_completion_assessment; use crate::ai::agent::task::TaskId; use crate::ai::agent::{ - runtime_activity, AIAgentActionType, AIAgentOutputMessageType, LifecycleEventType, - StartAgentExecutionMode, + runtime_activity, AIAgentActionType, AIAgentInput, AIAgentOutputMessageType, + LifecycleEventType, StartAgentExecutionMode, }; fn start_agent_tool_call_message( @@ -617,6 +618,36 @@ fn converts_local_start_agent_v2_with_harness_type() { assert_eq!(lifecycle_subscription, None); } +#[test] +fn internal_command_completion_assessment_is_not_restored_as_shared_user_input() { + let mut hidden_assessment = api::Message { + id: "hidden-assessment".to_string(), + task_id: "task".to_string(), + message: Some(api::message::Message::UserQuery(api::message::UserQuery { + query: "[Completed command: cargo test]".to_string(), + ..Default::default() + })), + ..Default::default() + }; + mark_internal_command_completion_assessment(&mut hidden_assessment); + let visible_query = api::Message { + id: "visible-query".to_string(), + task_id: "task".to_string(), + message: Some(api::message::Message::UserQuery(api::message::UserQuery { + query: "What changed?".to_string(), + ..Default::default() + })), + ..Default::default() + }; + + let inputs = user_inputs_from_messages(&[hidden_assessment, visible_query]); + assert_eq!(inputs.len(), 1); + assert!(matches!( + &inputs[0], + AIAgentInput::UserQuery { query, .. } if query == "What changed?" + )); +} + #[test] fn transfer_control_tool_call_converts_to_action_message() { let task_id = TaskId::new("task".to_string()); diff --git a/app/src/ai/agent/api/convert_to.rs b/app/src/ai/agent/api/convert_to.rs index c6c19319..90f13b12 100644 --- a/app/src/ai/agent/api/convert_to.rs +++ b/app/src/ai/agent/api/convert_to.rs @@ -348,6 +348,43 @@ fn convert_input_to_user_input( } )) } + AIAgentInput::CommandCompletionAssessment { + prompt, + completed_command: + RunningCommand { + command, + block_id, + grid_contents: output, + cursor, + requested_command_id, + is_alt_screen_active, + }, + .. + } => Ok( + api::request::input::user_inputs::user_input::Input::CliAgentUserQuery( + api::request::input::CliAgentUserQuery { + user_query: Some(api::request::input::UserQuery { + query: prompt, + referenced_attachments: Default::default(), + mode: Some(UserQueryMode::Normal.into()), + intended_agent: api::AgentType::Primary.into(), + }), + running_command: Some(api::RunningShellCommand { + command, + snapshot: Some(api::LongRunningShellCommandSnapshot { + output, + cursor, + command_id: block_id.as_str().to_owned(), + is_alt_screen_active, + is_preempted: false, + }), + }), + run_shell_command_tool_call_id: requested_command_id + .map(|id| id.to_string()) + .unwrap_or_default(), + }, + ), + ), AIAgentInput::ActionResult { result, .. } => result.try_into(), AIAgentInput::MessagesReceivedFromAgents { messages } => Ok( api::request::input::user_inputs::user_input::Input::MessagesReceivedFromAgents( diff --git a/app/src/ai/agent/api/convert_to_tests.rs b/app/src/ai/agent/api/convert_to_tests.rs index c80ffa2f..fae1323c 100644 --- a/app/src/ai/agent/api/convert_to_tests.rs +++ b/app/src/ai/agent/api/convert_to_tests.rs @@ -1,11 +1,13 @@ +use std::sync::Arc; + use chrono::{DateTime, Utc}; use galaxy_core::command::ExitCode; use warp_multi_agent_api as api; use crate::ai::agent::task::TaskId; use crate::ai::agent::{ - AIAgentActionResult, AIAgentActionResultType, AIAgentContext, ImageContext, - TransferShellCommandControlToUserResult, + AIAgentActionResult, AIAgentActionResultType, AIAgentContext, AIAgentInput, ImageContext, + RunningCommand, TransferShellCommandControlToUserResult, UserQueryMode, }; use crate::terminal::model::block::BlockId; @@ -132,6 +134,84 @@ fn git_context_deserializes_legacy_string_pull_request_number() { assert_eq!(pull_request.number, 42); } +#[test] +fn command_completion_assessment_converts_to_primary_cli_query() { + let block_id = BlockId::from("completed-block".to_string()); + let converted = super::convert_input(vec![AIAgentInput::CommandCompletionAssessment { + prompt: "Summarize whether the command succeeded.".to_string(), + context: Arc::from([AIAgentContext::SelectedText("root context".to_string())]), + completed_command: RunningCommand { + command: "cargo test -p galaxy".to_string(), + block_id: block_id.clone(), + grid_contents: "test result: ok".to_string(), + cursor: "cursor".to_string(), + requested_command_id: Some("run-call".to_string().into()), + is_alt_screen_active: true, + }, + }]) + .unwrap(); + + let Some(api::request::input::Type::UserInputs(inputs)) = converted.r#type else { + panic!("expected user inputs"); + }; + let Some(api::request::input::user_inputs::user_input::Input::CliAgentUserQuery(query)) = + inputs.inputs[0].input.as_ref() + else { + panic!("expected CLI agent query"); + }; + let user_query = query + .user_query + .as_ref() + .expect("expected assessment prompt"); + assert_eq!(user_query.query, "Summarize whether the command succeeded."); + assert_eq!(user_query.intended_agent(), api::AgentType::Primary); + let command = query + .running_command + .as_ref() + .expect("expected completed command"); + assert_eq!(command.command, "cargo test -p galaxy"); + let snapshot = command.snapshot.as_ref().expect("expected final snapshot"); + assert_eq!(snapshot.command_id, block_id.as_str()); + assert_eq!(snapshot.output, "test result: ok"); + assert_eq!(snapshot.cursor, "cursor"); + assert!(snapshot.is_alt_screen_active); + assert_eq!(query.run_shell_command_tool_call_id, "run-call"); + + let active = super::convert_input(vec![AIAgentInput::UserQuery { + query: "Keep monitoring.".to_string(), + context: Arc::from([]), + static_query_type: None, + referenced_attachments: Default::default(), + user_query_mode: UserQueryMode::Normal, + running_command: Some(RunningCommand { + command: "cargo test -p galaxy".to_string(), + block_id, + grid_contents: "still running".to_string(), + cursor: String::new(), + requested_command_id: None, + is_alt_screen_active: false, + }), + intended_agent: None, + }]) + .unwrap(); + let Some(api::request::input::Type::UserInputs(inputs)) = active.r#type else { + panic!("expected active user inputs"); + }; + let Some(api::request::input::user_inputs::user_input::Input::CliAgentUserQuery(query)) = + inputs.inputs[0].input.as_ref() + else { + panic!("expected active CLI agent query"); + }; + assert_eq!( + query + .user_query + .as_ref() + .expect("expected active monitor prompt") + .intended_agent(), + api::AgentType::Cli + ); +} + #[test] fn transfer_control_snapshot_result_converts_to_tool_call_result_input() { let block_id = BlockId::default(); diff --git a/app/src/ai/agent/api/impl.rs b/app/src/ai/agent/api/impl.rs index e567a094..9106c32a 100644 --- a/app/src/ai/agent/api/impl.rs +++ b/app/src/ai/agent/api/impl.rs @@ -7,7 +7,7 @@ use warp_multi_agent_api as api; use super::convert_to::convert_input; use super::{ConvertToAPITypeError, RequestParams, ResponseStream, StreamEvent}; -use crate::ai::agent::redaction; +use crate::ai::agent::{redaction, AIAgentInput}; use crate::ai::openai::translator as openai_translator; use crate::ai::provider::ProviderConfig; use crate::server::server_api::AIApiError; @@ -97,6 +97,10 @@ pub async fn generate_multi_agent_output( ); } + let emit_user_query_message = !params + .input + .iter() + .any(|input| matches!(input, AIAgentInput::CommandCompletionAssessment { .. })); let mut request = api::Request { task_context: Some(api::request::TaskContext { tasks: params.tasks, @@ -189,6 +193,7 @@ pub async fn generate_multi_agent_output( progressive_summary: params.progressive_summary.clone(), messages_sent: params.messages_sent.clone(), global_rules: params.global_rules.clone(), + emit_user_query_message, }; match openai_translator::execute(translator_request, &mut request).await { @@ -223,6 +228,7 @@ pub async fn generate_multi_agent_output( bedrock_progressive_summary: params.progressive_summary.clone(), bedrock_messages_sent: params.messages_sent.clone(), global_rules: params.global_rules.clone(), + emit_user_query_message, }; match crate::ai::bedrock::translator::execute(translator_request, &mut request).await { diff --git a/app/src/ai/agent/mod.rs b/app/src/ai/agent/mod.rs index fabd9585..d27a0ea0 100644 --- a/app/src/ai/agent/mod.rs +++ b/app/src/ai/agent/mod.rs @@ -2705,6 +2705,13 @@ pub enum AIAgentInput { intended_agent: Option, }, + /// A hidden system turn that asks for the final assessment of a completed command. + CommandCompletionAssessment { + prompt: String, + context: Arc<[AIAgentContext]>, + completed_command: RunningCommand, + }, + AutoCodeDiffQuery { query: String, context: Arc<[AIAgentContext]>, @@ -2869,6 +2876,9 @@ impl Display for AIAgentInput { Self::UserQuery { .. } => { write!(f, "UserQuery: {}", self.display_query().unwrap_or_default()) } + Self::CommandCompletionAssessment { .. } => { + write!(f, "CommandCompletionAssessment") + } Self::AutoCodeDiffQuery { query, .. } => { write!(f, "AutoCodeDiffQuery: {query}") } @@ -2957,7 +2967,8 @@ impl AIAgentInput { suggestion: PassiveSuggestionResultType::Prompt { prompt }, .. } => Some(prompt.clone()), - Self::AutoCodeDiffQuery { .. } + Self::CommandCompletionAssessment { .. } + | Self::AutoCodeDiffQuery { .. } | Self::ActionResult { .. } | Self::TriggerPassiveSuggestion { .. } | Self::ResumeConversation { .. } @@ -3048,6 +3059,7 @@ impl AIAgentInput { pub fn context(&self) -> Option<&[AIAgentContext]> { match self { Self::UserQuery { context, .. } + | Self::CommandCompletionAssessment { context, .. } | Self::ActionResult { context, .. } | Self::AutoCodeDiffQuery { context, .. } | Self::ResumeConversation { context, .. } @@ -3081,7 +3093,8 @@ impl AIAgentInput { Some(res) } Self::TriggerPassiveSuggestion { attachments, .. } => Some(attachments.clone()), - Self::ActionResult { .. } + Self::CommandCompletionAssessment { .. } + | Self::ActionResult { .. } | Self::AutoCodeDiffQuery { .. } | Self::ResumeConversation { .. } | Self::InitProjectRules { .. } diff --git a/app/src/ai/agent/mod_tests.rs b/app/src/ai/agent/mod_tests.rs index 268d074d..44e48206 100644 --- a/app/src/ai/agent/mod_tests.rs +++ b/app/src/ai/agent/mod_tests.rs @@ -1,17 +1,22 @@ +use std::collections::HashSet; use std::ops::Range; use std::sync::Arc; use anyhow::anyhow; +use chrono::Local; use markdown_parser::{FormattedText, FormattedTextFragment, FormattedTextLine}; use warp_multi_agent_api::{FileContent, FileContentLineRange}; use crate::ai::agent::{ - AIAgentContext, AIAgentOutput, AIAgentOutputMessage, AIAgentOutputMessageType, AIAgentText, + AIAgentContext, AIAgentExchange, AIAgentExchangeId, AIAgentInput, AIAgentOutput, + AIAgentOutputMessage, AIAgentOutputMessageType, AIAgentOutputStatus, AIAgentText, AIAgentTextSection, AgentOutputImage, AgentOutputImageLayout, AgentOutputMermaidDiagram, AnyFileContent, FileContext, FormattedTextWrapper, MessageId, ProgrammingLanguage, - RenderableAIError, TransientNetworkErrorKind, + RenderableAIError, RunningCommand, TransientNetworkErrorKind, }; +use crate::ai::llms::LLMId; use crate::server::server_api::AIApiError; +use crate::terminal::model::block::BlockId; use crate::terminal::shell::ShellType; fn to_range(range: Range) -> Option { @@ -21,6 +26,51 @@ fn to_range(range: Range) -> Option { }) } +#[test] +fn command_completion_assessment_stays_hidden_from_user_transcript() { + let context: Arc<[AIAgentContext]> = + Arc::from([AIAgentContext::SelectedText("relevant context".to_string())]); + let input = AIAgentInput::CommandCompletionAssessment { + prompt: "Report the final result.".to_string(), + context: context.clone(), + completed_command: RunningCommand { + command: "cargo test -p galaxy".to_string(), + block_id: BlockId::from("completed-command".to_string()), + grid_contents: "test result: ok".to_string(), + cursor: String::new(), + requested_command_id: None, + is_alt_screen_active: false, + }, + }; + + assert_eq!(input.display_query(), None); + assert!(!input.is_user_query()); + assert!(!input.is_passive_request()); + assert_eq!(input.context(), Some(context.as_ref())); + assert_eq!(input.attachments(), None); + + let now = Local::now(); + let exchange = AIAgentExchange { + id: AIAgentExchangeId::new(), + input: vec![input], + output_status: AIAgentOutputStatus::Streaming { output: None }, + added_message_ids: HashSet::new(), + start_time: now, + finish_time: None, + time_to_first_token_ms: None, + working_directory: None, + model_id: LLMId::from("test-model"), + request_cost: None, + coding_model_id: LLMId::from("test-model"), + cli_agent_model_id: LLMId::from("test-model"), + computer_use_model_id: LLMId::from("test-model"), + response_initiator: None, + }; + assert_eq!(exchange.format_input_for_copy(), ""); + assert_eq!(exchange.format_for_copy(None), ""); + assert!(!exchange.has_user_query()); +} + #[test] fn formatted_text_wrapper_shares_arc_across_calls() { let text = FormattedText::new([FormattedTextLine::Line(vec![ diff --git a/app/src/ai/agent/redaction.rs b/app/src/ai/agent/redaction.rs index c99293d6..89e55371 100644 --- a/app/src/ai/agent/redaction.rs +++ b/app/src/ai/agent/redaction.rs @@ -47,6 +47,17 @@ pub(crate) fn redact_inputs(inputs: &mut [AIAgentInput]) { redact_secrets(&mut running_command.cursor); } } + AIAgentInput::CommandCompletionAssessment { + prompt, + context, + completed_command, + } => { + redact_secrets(prompt); + redact_context(Arc::make_mut(context)); + redact_secrets(&mut completed_command.command); + redact_secrets(&mut completed_command.grid_contents); + redact_secrets(&mut completed_command.cursor); + } AIAgentInput::AutoCodeDiffQuery { query, context, .. } => { redact_secrets(query); redact_context(Arc::make_mut(context)); diff --git a/app/src/ai/agent_sdk/driver/output.rs b/app/src/ai/agent_sdk/driver/output.rs index dcbb26de..58688e3f 100644 --- a/app/src/ai/agent_sdk/driver/output.rs +++ b/app/src/ai/agent_sdk/driver/output.rs @@ -26,6 +26,7 @@ pub mod text { pub fn format_input(input: &AIAgentInput, w: &mut W) -> io::Result<()> { match input { AIAgentInput::UserQuery { .. } + | AIAgentInput::CommandCompletionAssessment { .. } | AIAgentInput::AutoCodeDiffQuery { .. } | AIAgentInput::CreateNewProject { .. } | AIAgentInput::CloneRepository { .. } @@ -785,6 +786,7 @@ pub mod json { match input { // Do not include the user query, since it's already provided as input to the agent. AIAgentInput::UserQuery { .. } + | AIAgentInput::CommandCompletionAssessment { .. } | AIAgentInput::AutoCodeDiffQuery { .. } | AIAgentInput::CreateNewProject { .. } | AIAgentInput::CloneRepository { .. } diff --git a/app/src/ai/bedrock/request_translator.rs b/app/src/ai/bedrock/request_translator.rs index 832d5dbb..167afd9d 100644 --- a/app/src/ai/bedrock/request_translator.rs +++ b/app/src/ai/bedrock/request_translator.rs @@ -7,6 +7,7 @@ use warp_multi_agent_api as api; use super::convert::{ ContentPart, ConversationMessage, MessageContent, MessageRole, ToolDefinition, }; +use crate::ai::agent::api::mark_internal_command_completion_assessment; /// Command-monitor turns must wake often enough to react to steering and user-specified deadlines. /// @@ -91,32 +92,11 @@ pub fn extract_new_input_messages(request: &api::Request) -> Vec { if let Some(user_query) = &cli_query.user_query { if !user_query.query.is_empty() { - let query_text = - if let Some(running_cmd) = &cli_query.running_command { - let mut context = - format!("[Running command: {}]\n", running_cmd.command); - if let Some(snapshot) = &running_cmd.snapshot { - if !snapshot.command_id.is_empty() { - context.push_str(&format!( - "[Command ID: {}]\n", - snapshot.command_id - )); - } - if !snapshot.output.is_empty() { - context.push_str(&format!( - "[Terminal output:\n{}\n]\n", - snapshot.output - )); - } - } - context.push_str(&user_query.query); - context - } else { - user_query.query.clone() - }; user_queries.push(ConversationMessage { role: MessageRole::User, - content: MessageContent::Text(query_text), + content: MessageContent::Text(cli_query_text( + cli_query, user_query, + )), }); } } @@ -565,24 +545,7 @@ fn extract_input_messages(request: &api::Request) -> Vec { ) => { if let Some(user_query) = &cli_query.user_query { if !user_query.query.is_empty() { - let query_text = - if let Some(running_cmd) = &cli_query.running_command { - let mut context = - format!("[Running command: {}]\n", running_cmd.command); - if let Some(snapshot) = &running_cmd.snapshot { - if !snapshot.output.is_empty() { - context.push_str(&format!( - "[Terminal output:\n{}\n]\n", - snapshot.output - )); - } - } - context.push_str(&user_query.query); - context - } else { - user_query.query.clone() - }; - results.push(api::Message { + let mut message = api::Message { id: uuid::Uuid::new_v4().to_string(), task_id: task_id.clone(), request_id: String::new(), @@ -592,11 +555,15 @@ fn extract_input_messages(request: &api::Request) -> Vec { fetched_memories: vec![], message: Some(api::message::Message::UserQuery( api::message::UserQuery { - query: query_text, + query: cli_query_text(cli_query, user_query), ..Default::default() }, )), - }); + }; + if cli_query_is_completed_assessment(cli_query) { + mark_internal_command_completion_assessment(&mut message); + } + results.push(message); } } } @@ -1235,12 +1202,57 @@ fn collect_tool_result_ids(content: &MessageContent, ids: &mut std::collections: } } +fn cli_query_is_completed_assessment(cli_query: &api::request::input::CliAgentUserQuery) -> bool { + cli_query + .user_query + .as_ref() + .is_some_and(|query| query.intended_agent() == api::AgentType::Primary) +} + +fn cli_query_text( + cli_query: &api::request::input::CliAgentUserQuery, + user_query: &api::request::input::UserQuery, +) -> String { + let Some(command) = &cli_query.running_command else { + return user_query.query.clone(); + }; + let completed = cli_query_is_completed_assessment(cli_query); + let mut context = format!( + "[{}: {}]\n", + if completed { + "Completed command" + } else { + "Running command" + }, + command.command + ); + if let Some(snapshot) = &command.snapshot { + if !snapshot.command_id.is_empty() { + context.push_str(&format!("[Command ID: {}]\n", snapshot.command_id)); + } + if !snapshot.output.is_empty() { + context.push_str(&format!( + "[{}:\n{}\n]\n", + if completed { + "Final terminal output" + } else { + "Terminal output" + }, + snapshot.output + )); + } + } + context.push_str(&user_query.query); + context +} + #[derive(Clone, Copy, Debug, Eq, PartialEq)] enum AgentMode { Normal, Plan, Orchestrate, Cli, + CompletedCommandAssessment, } fn request_agent_mode(request: &api::Request) -> AgentMode { @@ -1252,6 +1264,17 @@ fn request_agent_mode(request: &api::Request) -> AgentMode { return AgentMode::Normal; }; + if user_inputs.inputs.iter().any(|user_input| { + matches!( + &user_input.input, + Some(api::request::input::user_inputs::user_input::Input::CliAgentUserQuery( + cli_query + )) if cli_query_is_completed_assessment(cli_query) + ) + }) { + return AgentMode::CompletedCommandAssessment; + } + let mut mode = AgentMode::Normal; for user_input in &user_inputs.inputs { match &user_input.input { @@ -1511,6 +1534,15 @@ pub fn extract_system_prompt( clear reason.\n\n", ); } + AgentMode::CompletedCommandAssessment => { + prompt.push_str("## Completed Command Assessment\n"); + prompt.push_str( + "The monitored command has finished. Use its command, command ID, final terminal \ + output, and the assessment instruction in the latest hidden input to provide the \ + final user-facing outcome. Do not continue polling, request more terminal output, \ + or call tools.\n\n", + ); + } } prompt.push_str("## Available Tools\n"); @@ -1559,6 +1591,10 @@ pub fn extract_system_prompt( } pub fn extract_tools(request: &api::Request) -> Vec { + if request_agent_mode(request) == AgentMode::CompletedCommandAssessment { + return Vec::new(); + } + let mut tools = default_tool_definitions(); let mut seen_names: std::collections::HashSet = tools.iter().map(|t| t.name.clone()).collect(); @@ -1620,10 +1656,12 @@ pub fn extract_tools(request: &api::Request) -> Vec { fn supported_tool_types(request: &api::Request) -> Option> { let settings = request.settings.as_ref()?; - let raw_tools = if request_agent_mode(request) == AgentMode::Cli { - &settings.supported_cli_agent_tools - } else { - &settings.supported_tools + let raw_tools = match request_agent_mode(request) { + AgentMode::Cli => &settings.supported_cli_agent_tools, + AgentMode::Normal + | AgentMode::Plan + | AgentMode::Orchestrate + | AgentMode::CompletedCommandAssessment => &settings.supported_tools, }; Some( raw_tools diff --git a/app/src/ai/bedrock/request_translator_tests.rs b/app/src/ai/bedrock/request_translator_tests.rs index 9ff97746..449f1c1d 100644 --- a/app/src/ai/bedrock/request_translator_tests.rs +++ b/app/src/ai/bedrock/request_translator_tests.rs @@ -5,6 +5,7 @@ use super::{ convert_proto_message_for_test, extract_new_input_messages, extract_system_prompt, extract_tools, inject_input_messages_into_task, sanitize_messages_for_bedrock, }; +use crate::ai::agent::api::is_internal_command_completion_assessment; use crate::ai::bedrock::convert::{ContentPart, ConversationMessage, MessageContent, MessageRole}; #[test] @@ -237,6 +238,117 @@ fn plan_mode_prompt_prohibits_mutation() { assert!(prompt.contains("do not edit files")); } +fn completed_command_request() -> api::Request { + api::Request { + task_context: Some(api::request::TaskContext { + tasks: vec![api::Task { + id: "root-task".to_string(), + ..Default::default() + }], + }), + input: Some(api::request::Input { + r#type: Some(api::request::input::Type::UserInputs( + api::request::input::UserInputs { + inputs: vec![api::request::input::user_inputs::UserInput { + input: Some( + api::request::input::user_inputs::user_input::Input::CliAgentUserQuery( + api::request::input::CliAgentUserQuery { + user_query: Some(api::request::input::UserQuery { + query: "Report the final outcome.".to_string(), + intended_agent: api::AgentType::Primary.into(), + ..Default::default() + }), + running_command: Some(api::RunningShellCommand { + command: "cargo test -p galaxy".to_string(), + snapshot: Some(api::LongRunningShellCommandSnapshot { + command_id: "completed-block-123".to_string(), + output: "test result: ok".to_string(), + cursor: "cursor".to_string(), + ..Default::default() + }), + }), + ..Default::default() + }, + ), + ), + }], + }, + )), + ..Default::default() + }), + settings: Some(api::request::Settings { + supported_tools: vec![ + api::ToolType::RunShellCommand.into(), + api::ToolType::ReadFiles.into(), + api::ToolType::CallMcpTool.into(), + ], + supported_cli_agent_tools: vec![api::ToolType::ReadShellCommandOutput.into()], + ..Default::default() + }), + mcp_context: Some(api::request::McpContext { + servers: vec![api::request::mcp_context::McpServer { + id: "server-id".to_string(), + name: "test-server".to_string(), + description: String::new(), + resources: Vec::new(), + tools: vec![api::request::mcp_context::McpTool { + name: "echo".to_string(), + description: "Echo input".to_string(), + input_schema: None, + }], + }], + ..Default::default() + }), + ..Default::default() + } +} + +#[test] +fn completed_command_assessment_is_tool_free_and_persists_hidden_provider_history() { + let mut request = completed_command_request(); + + let messages = extract_new_input_messages(&request); + assert_eq!(messages.len(), 1); + assert!(matches!( + &messages[0].content, + MessageContent::Text(text) + if text.contains("[Completed command: cargo test -p galaxy]") + && text.contains("[Command ID: completed-block-123]") + && text.contains("[Final terminal output:\ntest result: ok") + && text.contains("Report the final outcome.") + )); + + let prompt = extract_system_prompt(&request, &[]).expect("system prompt"); + assert!(prompt.contains("## Completed Command Assessment")); + assert!(prompt.contains("No tools are available for this request")); + assert!(!prompt.contains("## Running Command Monitor")); + assert!(!prompt.contains("next assistant output MUST be a tool call")); + assert!(extract_tools(&request).is_empty()); + + inject_input_messages_into_task(&mut request); + let persisted = &request.task_context.as_ref().expect("task context").tasks[0].messages; + assert_eq!(persisted.len(), 1); + assert!(is_internal_command_completion_assessment(&persisted[0])); + assert!(matches!( + persisted[0].message.as_ref(), + Some(api::message::Message::UserQuery(query)) + if query.query.contains("[Completed command: cargo test -p galaxy]") + && query.query.contains("[Command ID: completed-block-123]") + && query.query.contains("[Final terminal output:\ntest result: ok") + && query.query.contains("Report the final outcome.") + )); + + let restored = convert_proto_message_for_test(&persisted[0]) + .expect("hidden assessment should remain in provider history"); + assert_eq!(restored.role, MessageRole::User); + assert!(matches!( + restored.content, + MessageContent::Text(text) + if text.contains("[Completed command: cargo test -p galaxy]") + && text.contains("Report the final outcome.") + )); +} + #[test] fn running_command_turn_gets_monitor_prompt_and_cli_tools() { let request = api::Request { diff --git a/app/src/ai/bedrock/translator.rs b/app/src/ai/bedrock/translator.rs index 0f354437..07dd9c92 100644 --- a/app/src/ai/bedrock/translator.rs +++ b/app/src/ai/bedrock/translator.rs @@ -21,6 +21,8 @@ pub struct TranslatorRequest { pub bedrock_messages_sent: Arc>>, /// Global rules (name, content) from the local CloudModel. pub global_rules: Vec<(String, String)>, + /// Whether the native input should be emitted as a transcript-visible user query. + pub emit_user_query_message: bool, } pub async fn execute( @@ -131,7 +133,10 @@ pub async fn execute( ); } - let user_query_text = request_translator::extract_user_query_text(request); + let user_query_text = params + .emit_user_query_message + .then(|| request_translator::extract_user_query_text(request)) + .flatten(); let stream = bedrock .converse_stream( diff --git a/app/src/ai/blocklist/block/cli_controller.rs b/app/src/ai/blocklist/block/cli_controller.rs index d0113d19..67dc39f1 100644 --- a/app/src/ai/blocklist/block/cli_controller.rs +++ b/app/src/ai/blocklist/block/cli_controller.rs @@ -400,10 +400,10 @@ impl CLISubagentController { } fn advance_completed_subagent(&mut self, block_id: &BlockId, ctx: &mut ModelContext) { - let Some((task_id, completion)) = self + let Some(completion) = self .active_subagents_by_block .get(block_id) - .and_then(|state| Some((state.task_id.clone()?, state.completion.as_ref()?.clone()))) + .and_then(|state| state.completion.as_ref().cloned()) else { return; }; @@ -433,7 +433,6 @@ impl CLISubagentController { let sent = self.controller.update(ctx, |controller, ctx| { controller.send_command_completion_assessment( completion.conversation_id, - task_id, completion.prompt, completion.completed_command, ctx, diff --git a/app/src/ai/blocklist/block/view_impl.rs b/app/src/ai/blocklist/block/view_impl.rs index 52804d71..376f3738 100644 --- a/app/src/ai/blocklist/block/view_impl.rs +++ b/app/src/ai/blocklist/block/view_impl.rs @@ -1375,6 +1375,7 @@ impl AIAgentInput { app, )), AIAgentInput::UserQuery { .. } + | AIAgentInput::CommandCompletionAssessment { .. } | AIAgentInput::AutoCodeDiffQuery { .. } | AIAgentInput::ResumeConversation { .. } | AIAgentInput::InitProjectRules { .. } diff --git a/app/src/ai/blocklist/block/view_impl/common.rs b/app/src/ai/blocklist/block/view_impl/common.rs index 01c94ce5..7c963b67 100644 --- a/app/src/ai/blocklist/block/view_impl/common.rs +++ b/app/src/ai/blocklist/block/view_impl/common.rs @@ -3687,6 +3687,7 @@ pub(super) fn query_prefix_highlight_len( match input { AIAgentInput::InvokeSkill { skill, .. } => Some(1 + skill.name.len()), AIAgentInput::UserQuery { .. } + | AIAgentInput::CommandCompletionAssessment { .. } | AIAgentInput::AutoCodeDiffQuery { .. } | AIAgentInput::ResumeConversation { .. } | AIAgentInput::InitProjectRules { .. } diff --git a/app/src/ai/blocklist/controller.rs b/app/src/ai/blocklist/controller.rs index 0c2d8107..0df05e0e 100644 --- a/app/src/ai/blocklist/controller.rs +++ b/app/src/ai/blocklist/controller.rs @@ -1597,7 +1597,7 @@ impl BlocklistAIController { ); } - /// Sends one non-preemptive final assessment to a completed CLI-monitor task. + /// Sends one non-preemptive final assessment to the root task after a CLI monitor completes. /// /// This deliberately bypasses `send_query`: command completion must not cancel /// another conversation, drain unrelated action results, or replace a request @@ -1605,8 +1605,7 @@ impl BlocklistAIController { pub fn send_command_completion_assessment( &mut self, conversation_id: AIConversationId, - task_id: TaskId, - query: String, + prompt: String, completed_command: RunningCommand, ctx: &mut ModelContext, ) -> bool { @@ -1621,6 +1620,16 @@ impl BlocklistAIController { return false; } + let Some(root_task_id) = BlocklistAIHistoryModel::as_ref(ctx) + .conversation(&conversation_id) + .map(|conversation| conversation.get_root_task_id().clone()) + else { + log::warn!( + "Cannot send command completion assessment for missing conversation \ + {conversation_id:?}" + ); + return false; + }; let context = input_context_for_request( false, self.context_model.as_ref(ctx), @@ -1629,16 +1638,12 @@ impl BlocklistAIController { ctx, ); let request_input = RequestInput::for_task( - vec![AIAgentInput::UserQuery { - query, + vec![AIAgentInput::CommandCompletionAssessment { + prompt, context, - static_query_type: None, - referenced_attachments: HashMap::new(), - user_query_mode: UserQueryMode::Normal, - running_command: Some(completed_command), - intended_agent: None, + completed_command, }], - task_id, + root_task_id, &self.active_session, self.get_current_response_initiator(), conversation_id, @@ -4553,7 +4558,7 @@ impl BlocklistAIController { .as_ref() .map(|stream_cancellation| format!("{:?}", stream_cancellation.reason)), "queued_tools": remote_action_summaries(&actions_to_queue), - "proposed_tools": remote_action_summaries(&proposed_actions), + "proposed_tools": remote_action_summaries(proposed_actions), }), }, ); diff --git a/app/src/ai/blocklist/controller/response_stream.rs b/app/src/ai/blocklist/controller/response_stream.rs index 5bf0b480..a84ccc38 100644 --- a/app/src/ai/blocklist/controller/response_stream.rs +++ b/app/src/ai/blocklist/controller/response_stream.rs @@ -376,6 +376,7 @@ impl ResponseStream { } #[cfg(not(target_family = "wasm"))] + #[allow(clippy::too_many_arguments)] fn log_llm_request_started( ctx: &mut ModelContext, stream_id: &ResponseStreamId, diff --git a/app/src/ai/blocklist/history_model_tests.rs b/app/src/ai/blocklist/history_model_tests.rs index e1372c27..adc88436 100644 --- a/app/src/ai/blocklist/history_model_tests.rs +++ b/app/src/ai/blocklist/history_model_tests.rs @@ -21,8 +21,10 @@ use crate::ai::agent::conversation::{ ServerAIConversationMetadata, }; use crate::ai::agent::{ - AIAgentExchange, AIAgentExchangeId, AIAgentInput, AIAgentOutputStatus, FinishedAIAgentOutput, - RenderableAIError, Shared, TransientNetworkErrorKind, UserQueryMode, + AIAgentExchange, AIAgentExchangeId, AIAgentInput, AIAgentOutput, AIAgentOutputMessage, + AIAgentOutputMessageType, AIAgentOutputStatus, AIAgentText, AIAgentTextSection, + AgentOutputText, FinishedAIAgentOutput, MessageId, RenderableAIError, RunningCommand, Shared, + TransientNetworkErrorKind, UserQueryMode, }; use crate::ai::ambient_agents::{ conversation_output_status_from_conversation, AmbientAgentTaskId, AmbientConversationStatus, @@ -270,6 +272,125 @@ fn deactivating_cli_subtask_clears_activity_without_deleting_task() { }); } +#[test] +fn completed_command_assessment_survives_cli_subtask_deactivation_on_root() { + App::test((), |mut app| async move { + initialize_history_persistence_for_tests(&mut app); + let terminal_view_id = EntityId::new(); + let history_model = app.add_singleton_model(|_| BlocklistAIHistoryModel::new_for_test()); + let block_id = BlockId::new(); + let assessment_output = "The command completed successfully."; + + history_model.update(&mut app, |model, ctx| { + let conversation_id = + model.start_new_conversation(terminal_view_id, false, false, false, ctx); + let cli_task_id = model + .create_cli_subagent_task_for_conversation( + block_id.clone(), + conversation_id, + terminal_view_id, + ctx, + ) + .expect("CLI subtask should be created"); + + let monitor_exchange = + create_exchange_with_query("Check the command status.", Local::now(), None); + let monitor_exchange_id = monitor_exchange.id; + let conversation = model + .conversation_mut(&conversation_id) + .expect("conversation should exist"); + conversation + .append_task_exchange_for_test( + &cli_task_id, + monitor_exchange, + terminal_view_id, + ctx, + ) + .expect("monitor exchange should be appended to the CLI task"); + + let now = Local::now(); + let assessment_exchange = AIAgentExchange { + id: AIAgentExchangeId::new(), + input: vec![AIAgentInput::CommandCompletionAssessment { + prompt: "Assess the completed command.".to_string(), + context: Arc::from([]), + completed_command: RunningCommand { + command: "cargo test -p galaxy".to_string(), + block_id: block_id.clone(), + grid_contents: "test result: ok".to_string(), + cursor: String::new(), + requested_command_id: None, + is_alt_screen_active: false, + }, + }], + output_status: AIAgentOutputStatus::Finished { + finished_output: FinishedAIAgentOutput::Success { + output: Shared::new(AIAgentOutput { + messages: vec![AIAgentOutputMessage { + id: MessageId::new("assessment-output".to_string()), + message: AIAgentOutputMessageType::Text(AIAgentText { + sections: vec![AIAgentTextSection::PlainText { + text: AgentOutputText::from(assessment_output.to_string()), + }], + }), + citations: vec![], + }], + ..Default::default() + }), + }, + }, + added_message_ids: HashSet::new(), + start_time: now, + finish_time: Some(now), + time_to_first_token_ms: None, + working_directory: None, + model_id: LLMId::from("test-model"), + request_cost: None, + coding_model_id: LLMId::from("test-coding-model"), + cli_agent_model_id: LLMId::from("test-cli-agent-model"), + computer_use_model_id: LLMId::from("test-computer-use-model"), + response_initiator: None, + }; + let assessment_exchange_id = assessment_exchange.id; + model + .conversation_mut(&conversation_id) + .expect("conversation should exist") + .append_root_exchange_for_test(assessment_exchange); + + model + .deactivate_cli_subagent_task_for_conversation(&block_id, conversation_id) + .expect("CLI subtask should deactivate"); + + let conversation = model + .conversation(&conversation_id) + .expect("conversation should still exist"); + assert!(!conversation.has_active_subagent()); + let cli_task = conversation + .get_task(&cli_task_id) + .expect("CLI task should be retained after deactivation"); + assert_eq!(cli_task.exchanges_len(), 1); + assert_eq!( + cli_task.last_exchange().map(|exchange| exchange.id), + Some(monitor_exchange_id) + ); + + let root_exchange = conversation + .latest_visible_exchange() + .expect("root assessment output should remain visible"); + assert_eq!(root_exchange.id, assessment_exchange_id); + assert!(matches!( + root_exchange.input.as_slice(), + [AIAgentInput::CommandCompletionAssessment { .. }] + )); + assert!(root_exchange.input[0].display_query().is_none()); + assert_eq!( + root_exchange.format_output_for_copy(None), + assessment_output + ); + }); + }); +} + #[test] fn monitoring_a_different_block_preserves_completed_cli_task_history() { App::test((), |mut app| async move { diff --git a/app/src/ai/blocklist/persistence.rs b/app/src/ai/blocklist/persistence.rs index 8fdc6e7b..08b5acf3 100644 --- a/app/src/ai/blocklist/persistence.rs +++ b/app/src/ai/blocklist/persistence.rs @@ -77,7 +77,8 @@ impl TryFrom<&AIAgentInput> for PersistedAIInputType { AIAgentInput::PassiveSuggestionResult { suggestion: PassiveSuggestionResultType::CodeDiff { .. }, .. } => Err(anyhow!( "PassiveSuggestionResult::CodeDiff is not persisted as a query." )), - AIAgentInput::ActionResult { .. } + AIAgentInput::CommandCompletionAssessment { .. } + | AIAgentInput::ActionResult { .. } | AIAgentInput::ResumeConversation { .. } | AIAgentInput::InitProjectRules { .. } | AIAgentInput::CreateEnvironment { .. } diff --git a/app/src/ai/openai/translator.rs b/app/src/ai/openai/translator.rs index 08e08e79..0a720961 100644 --- a/app/src/ai/openai/translator.rs +++ b/app/src/ai/openai/translator.rs @@ -24,6 +24,8 @@ pub struct TranslatorRequest { pub messages_sent: Arc>>, /// Global rules (name, content) from the local CloudModel. pub global_rules: Vec<(String, String)>, + /// Whether the native input should be emitted as a transcript-visible user query. + pub emit_user_query_message: bool, } pub(crate) struct PreparedTurn { @@ -102,7 +104,10 @@ pub(crate) fn prepare_turn(params: &TranslatorRequest, request: &mut api::Reques PreparedTurn { task_id, needs_create_task, - user_query: request_translator::extract_user_query_text(request), + user_query: params + .emit_user_query_message + .then(|| request_translator::extract_user_query_text(request)) + .flatten(), messages, system_prompt: request_translator::extract_system_prompt(request, ¶ms.global_rules), tools, diff --git a/app/src/ai/runtime/rig_request.rs b/app/src/ai/runtime/rig_request.rs index 3a29da2a..7b4fcceb 100644 --- a/app/src/ai/runtime/rig_request.rs +++ b/app/src/ai/runtime/rig_request.rs @@ -110,18 +110,29 @@ fn prepare_rig_turn_for_provider( let mode = request_mode(&input); let available_tools = match mode { RigRequestMode::Cli => supported_cli_agent_tools, + RigRequestMode::CompletedCommandAssessment => Vec::new(), RigRequestMode::Normal | RigRequestMode::Plan | RigRequestMode::Orchestrate => { supported_tools } }; - let (mut tools, mcp_tool_aliases) = tool_definitions(&available_tools, mcp_context.as_ref()); - if matches!(mode, RigRequestMode::Cli) { - // History recall cannot advance a running command and is handled inline by the Rig - // adapter (without producing a client action that can trigger another turn). Keeping it - // in the CLI tool list lets the model spend its entire monitor turn recalling the prior - // snapshot instead of scheduling `read_shell_command_output`, so make polling the only - // way to inspect the active command here. - tools.retain(|tool| tool.name != "recall_tool_history"); + let (mut tools, mut mcp_tool_aliases) = + tool_definitions(&available_tools, mcp_context.as_ref()); + match mode { + RigRequestMode::Cli => { + // History recall cannot advance a running command and is handled inline by the Rig + // adapter (without producing a client action that can trigger another turn). Keeping it + // in the CLI tool list lets the model spend its entire monitor turn recalling the prior + // snapshot instead of scheduling `read_shell_command_output`, so make polling the only + // way to inspect the active command here. + tools.retain(|tool| tool.name != "recall_tool_history"); + } + RigRequestMode::CompletedCommandAssessment => { + // The caller deliberately disables tools for the final assessment. The inline history + // tool is added independently of supported tool types, so remove it explicitly too. + tools.clear(); + mcp_tool_aliases.clear(); + } + RigRequestMode::Normal | RigRequestMode::Plan | RigRequestMode::Orchestrate => {} } let system_prompt = build_system_prompt(&input, &tools, &global_rules, mode); @@ -243,6 +254,20 @@ fn input_message(input: AIAgentInput) -> Option { }; (text, image_parts(&context)) } + AIAgentInput::CommandCompletionAssessment { + prompt, + context, + completed_command, + } => ( + format!( + "[Completed command: {}]\n[Command ID: {}]\n[Final terminal output:\n{}\n]\n{}", + completed_command.command, + completed_command.block_id, + completed_command.grid_contents, + prompt + ), + image_parts(&context), + ), AIAgentInput::ActionResult { .. } => return None, AIAgentInput::AutoCodeDiffQuery { query, .. } => (query, Vec::new()), AIAgentInput::ResumeConversation { .. } => ( @@ -389,7 +414,8 @@ fn input_user_query(input: &AIAgentInput) -> Option { match input { AIAgentInput::UserQuery { query, .. } => Some(query.clone()), AIAgentInput::InvokeSkill { skill, .. } => Some(format!("/{}", skill.name)), - AIAgentInput::AutoCodeDiffQuery { .. } + AIAgentInput::CommandCompletionAssessment { .. } + | AIAgentInput::AutoCodeDiffQuery { .. } | AIAgentInput::ResumeConversation { .. } | AIAgentInput::InitProjectRules { .. } | AIAgentInput::CreateEnvironment { .. } @@ -414,9 +440,17 @@ enum RigRequestMode { Plan, Orchestrate, Cli, + CompletedCommandAssessment, } fn request_mode(inputs: &[AIAgentInput]) -> RigRequestMode { + if inputs + .iter() + .any(|input| matches!(input, AIAgentInput::CommandCompletionAssessment { .. })) + { + return RigRequestMode::CompletedCommandAssessment; + } + for input in inputs { // A direct-provider follow-up carries an LRC snapshot as an action result rather than // as a user query with `running_command`. Treat that result as a CLI-monitor turn so the @@ -747,6 +781,9 @@ fn build_system_prompt( RigRequestMode::Cli => prompt.push_str( "## Running Command Monitor\nThis turn concerns a running or just-finished shell command. Act as its dedicated monitor while still following the user's steering messages. Use the command ID from the tool result for every read/write operation. If the result says the command finished, report its outcome and stop polling. If it says the command is still running, the next assistant output MUST be a tool call: use `read_shell_command_output` with a short delay, or use `interrupt_shell_command` immediately when the user's explicit stop condition is met. Do not end a still-running monitor turn with prose, a status message, or a request for the user to say continue. Never choose a poll interval that crosses a user-specified deadline or stop condition. After an interrupt, poll briefly to verify the outcome. Never start a duplicate command merely to check its state, and never report completion while a result says it is still running.\n\n", ), + RigRequestMode::CompletedCommandAssessment => prompt.push_str( + "## Completed Command Assessment\nThe monitored command has finished. Use its command, command ID, final terminal output, and the assessment instruction in the latest hidden input to provide the final user-facing outcome. Do not continue polling, request more terminal output, or call tools.\n\n", + ), } prompt.push_str("## Available Tools\n"); if tools.is_empty() { diff --git a/app/src/ai/runtime/rig_request_tests.rs b/app/src/ai/runtime/rig_request_tests.rs index 8b6fb4ad..c7c16947 100644 --- a/app/src/ai/runtime/rig_request_tests.rs +++ b/app/src/ai/runtime/rig_request_tests.rs @@ -248,6 +248,77 @@ fn rig_prompt_requires_follow_through_without_manual_continue_prompts() { assert!(prompt.contains("After each tool result, choose and perform the next necessary step")); } +#[test] +#[allow(deprecated)] +fn completed_command_assessment_uses_root_history_without_tools_or_monitor_instructions() { + let block_id: galaxy_terminal::model::BlockId = "completed-lrc-test".to_string().into(); + let mcp_tool = serde_json::from_value(serde_json::json!({ + "name": "echo", + "description": "Echo input", + "inputSchema": {"type": "object"} + })) + .unwrap(); + let mut params = RequestParams::new_for_test(); + params.root_task_id = Some("root-task".to_string()); + params.message_history = vec![galaxy_agent_core::ConversationMessage { + role: MessageRole::User, + content: MessageContent::Text("Prior root conversation".to_string()), + }]; + params.mcp_context = Some(MCPContext { + resources: Vec::new(), + tools: Vec::new(), + servers: vec![MCPServer { + id: "11111111-1111-4111-8111-111111111111".to_string(), + name: "Echo".to_string(), + description: String::new(), + resources: Vec::new(), + tools: vec![mcp_tool], + }], + }); + params.input = vec![AIAgentInput::CommandCompletionAssessment { + prompt: "Report the final result to the user.".to_string(), + context: Arc::from([]), + completed_command: crate::ai::agent::RunningCommand { + command: "bash loop.sh".to_string(), + block_id, + grid_contents: "All 42 checks passed.".to_string(), + cursor: String::new(), + requested_command_id: None, + is_alt_screen_active: false, + }, + }]; + + let prepared = prepare_rig_turn( + &config(), + params, + vec![ToolType::RunShellCommand, ToolType::CallMcpTool], + vec![ToolType::ReadShellCommandOutput], + ); + let prompt = prepared.request.system_prompt.expect("system prompt"); + + assert_eq!(prepared.task_id, "root-task"); + assert_eq!(prepared.user_query, None); + assert!(prepared.request.tools.is_empty()); + assert!(prepared.mcp_tool_aliases.is_empty()); + assert!(prompt.contains("## Completed Command Assessment")); + assert!(prompt.contains("No tools are available")); + assert!(!prompt.contains("## Running Command Monitor")); + assert!(!prompt.contains("next assistant output MUST be a tool call")); + assert!(prepared.persistent_messages.iter().any(|message| matches!( + &message.content, + MessageContent::Text(text) if text == "Prior root conversation" + ))); + assert!(prepared.persistent_messages.iter().any(|message| matches!( + &message.content, + MessageContent::Text(text) + if text.contains("[Completed command: bash loop.sh]") + && text.contains("[Command ID: completed-lrc-test]") + && text.contains("[Final terminal output:\nAll 42 checks passed.") + && text.contains("Report the final result to the user.") + ))); + assert_eq!(prepared.request.messages, prepared.persistent_messages); +} + #[test] fn lrc_snapshot_follow_up_uses_the_cli_monitor_prompt_and_tools() { let block_id: galaxy_terminal::model::BlockId = "precmd-lrc-test".to_string().into(); diff --git a/app/src/server/telemetry/events.rs b/app/src/server/telemetry/events.rs index 1c4eca04..62c7b7bc 100644 --- a/app/src/server/telemetry/events.rs +++ b/app/src/server/telemetry/events.rs @@ -1021,6 +1021,7 @@ pub enum InputUXChangeOrigin { #[derive(Clone, Debug, Serialize)] pub enum AIAgentInput { UserQuery { query: String }, + CommandCompletionAssessment, AutoCodeDiffQuery { query: String }, ResumeConversation, InitProjectRules { display_query: Option }, @@ -1044,6 +1045,9 @@ impl From for AIAgentInput { fn from(input: FullAIAgentInput) -> Self { match input { FullAIAgentInput::UserQuery { query, .. } => Self::UserQuery { query }, + FullAIAgentInput::CommandCompletionAssessment { .. } => { + Self::CommandCompletionAssessment + } FullAIAgentInput::AutoCodeDiffQuery { query, .. } => Self::AutoCodeDiffQuery { query }, FullAIAgentInput::ResumeConversation { .. } => Self::ResumeConversation, FullAIAgentInput::InitProjectRules { display_query, .. } => {