Fix direct-provider command monitoring

This commit is contained in:
2026-08-15 15:49:07 -05:00
parent 2730712179
commit 4b20952cec
9 changed files with 152 additions and 56 deletions
+15 -12
View File
@@ -1520,18 +1520,21 @@ pub fn extract_system_prompt(
monitor while still following the user's steering messages. Use the command ID from \ monitor while still following the user's steering messages. Use the command ID from \
the running-command context or tool result for every read/write operation. If the \ the running-command context or tool result for every read/write operation. If the \
result says the command finished, report its outcome and stop polling. If it says the \ result says the command finished, report its outcome and stop polling. If it says the \
command is still running, the next assistant output MUST be a tool call: use \ command is still running, the next assistant output MUST be a tool call. Use \
`read_shell_command_output` with a short delay, or call `interrupt_shell_command` \ `read_shell_command_output` with a short delay for normal progress. If the snapshot \
immediately when the user's explicit stop condition is met. Do not end a still-running \ clearly shows an interactive pager or editor, do not keep polling: an alternate screen \
monitor turn with prose, a status message, or a request for the user to say continue. \ containing `(END)` is `less`, so call `write_to_long_running_shell_command` with input \
Never choose a poll interval that crosses a user-specified deadline or stop condition. \ `q` and mode `raw`; for a clearly identified Vim screen, send input `:q` with mode \
After an interrupt, poll briefly \ `line`. Poll briefly after sending quit input to verify the outcome. Call \
to verify the outcome. Never try to encode Ctrl+C as `C-c`, `^C`, `\\x03`, or \ `interrupt_shell_command` immediately when the user's explicit stop condition is met. \
`\\u0003` through `write_to_long_running_shell_command`; that tool is only for actual \ Do not end a still-running monitor turn with prose, a status message, or a request for \
process input. Never start a duplicate command merely to check its state, and never \ the user to say continue. Never choose a poll interval that crosses a user-specified \
report completion while a result says it is still running. If user interaction is \ deadline or stop condition. After an interrupt, poll briefly to verify the outcome. \
the right next step and the transfer tool is available, transfer control with a \ Never try to encode Ctrl+C as `C-c`, `^C`, `\\x03`, or `\\u0003` through \
clear reason.\n\n", `write_to_long_running_shell_command`; that tool is only for actual process input. \
Never start a duplicate command merely to check its state, and never report completion \
while a result says it is still running. If user interaction is the right next step and \
the transfer tool is available, transfer control with a clear reason.\n\n",
); );
} }
AgentMode::CompletedCommandAssessment => { AgentMode::CompletedCommandAssessment => {
@@ -413,6 +413,8 @@ fn running_command_turn_gets_monitor_prompt_and_cli_tools() {
assert!(prompt.contains("read_shell_command_output")); assert!(prompt.contains("read_shell_command_output"));
assert!(prompt.contains("interrupt_shell_command")); assert!(prompt.contains("interrupt_shell_command"));
assert!(prompt.contains("next assistant output MUST be a tool call")); assert!(prompt.contains("next assistant output MUST be a tool call"));
assert!(prompt.contains("alternate screen containing `(END)` is `less`"));
assert!(prompt.contains("`write_to_long_running_shell_command` with input `q` and mode `raw`"));
assert!(prompt.contains("Do not end a still-running monitor turn with prose")); assert!(prompt.contains("Do not end a still-running monitor turn with prose"));
assert!(prompt.contains("Never try to encode Ctrl+C")); assert!(prompt.contains("Never try to encode Ctrl+C"));
assert!(!prompt.contains("- Use `run_shell_command`")); assert!(!prompt.contains("- Use `run_shell_command`"));
@@ -190,23 +190,6 @@ impl ShellCommandExecutor {
} }
} }
/// Decorate the command so that we can turn off pager.
fn turn_off_pager_for_command(&self, command: &String, ctx: &mut ModelContext<Self>) -> String {
match self.active_session.as_ref(ctx).shell_type(ctx) {
// If it's a posix shell, we can use parentheses as the grouping character. Add command to
// avoid cases with aliases.
Some(ShellType::Zsh) | Some(ShellType::Bash) => format!("({command}) | command cat"),
// Fish doesn't have grouping characters. We need to use begin; and end; to ensure the command
// gets evaluated first.
Some(ShellType::Fish) => format!("begin; {command} ;end | command cat"),
// For powershell, we use Out-Host to send paged output to the
// console. Add a backslash to avoid executing an alias.
Some(ShellType::PowerShell) => format!("({command}) | \\Out-Host"),
// If we can't determine a shell type, run command as it is.
None => command.clone(),
}
}
pub(super) fn execute( pub(super) fn execute(
&mut self, &mut self,
input: ExecuteActionInput, input: ExecuteActionInput,
@@ -231,7 +214,6 @@ impl ShellCommandExecutor {
match &input.action.action { match &input.action.action {
AIAgentActionType::RequestCommandOutput { AIAgentActionType::RequestCommandOutput {
command, command,
uses_pager,
wait_until_completion, wait_until_completion,
.. ..
} => { } => {
@@ -266,15 +248,13 @@ impl ShellCommandExecutor {
RequestCommandOutputResult::CancelledBeforeExecution, RequestCommandOutputResult::CancelledBeforeExecution,
)); ));
} }
// If the command might use pager and can't be interacted with, // A command expected to finish must not enter an implicit pager. Do not trust the
// we pipe its output to cat so we can prevent activating the altscreen. // model-provided pager hint: commands such as `git log` can page implicitly.
// The parentheses here ensures the command always gets evaluated first. let decorated_command = command_for_execution(
let decorated_command = command,
if uses_pager.is_some_and(|uses_pager| uses_pager) && *wait_until_completion { self.active_session.as_ref(ctx).shell_type(ctx),
self.turn_off_pager_for_command(command, ctx) *wait_until_completion,
} else { );
command.clone()
};
ctx.emit(ShellCommandExecutorEvent::ExecuteCommand { ctx.emit(ShellCommandExecutorEvent::ExecuteCommand {
action_id: action_id.clone(), action_id: action_id.clone(),
command: decorated_command, command: decorated_command,
@@ -710,6 +690,30 @@ impl ShellCommandExecutor {
} }
} }
fn command_for_execution(
command: &str,
shell_type: Option<ShellType>,
wait_until_completion: bool,
) -> String {
if !wait_until_completion {
return command.to_string();
}
match shell_type {
// Pager environment variables preserve the command's output and exit status, unlike piping
// through `cat`. Tool-specific variables override user configuration for common pagers.
Some(ShellType::Zsh) | Some(ShellType::Bash) => format!(
"(export PAGER=cat GIT_PAGER=cat GH_PAGER=cat AWS_PAGER=cat SYSTEMD_PAGER=cat; {command})"
),
Some(ShellType::Fish) => format!(
"begin; set -lx PAGER cat; set -lx GIT_PAGER cat; set -lx GH_PAGER cat; set -lx AWS_PAGER cat; set -lx SYSTEMD_PAGER cat; {command}; end"
),
// PowerShell's pipeline host suppresses paging for commands that honor the host stream.
Some(ShellType::PowerShell) => format!("({command}) | \\Out-Host"),
None => command.to_string(),
}
}
#[derive(Debug, Clone, Hash, PartialEq, Eq)] #[derive(Debug, Clone, Hash, PartialEq, Eq)]
enum BlockSelector { enum BlockSelector {
Id(BlockId), Id(BlockId),
@@ -5,7 +5,7 @@ use futures::channel::oneshot;
use parking_lot::FairMutex; use parking_lot::FairMutex;
use warpui::{App, EntityId}; use warpui::{App, EntityId};
use super::{ActionResult, BlockSelector, ShellCommandExecutor}; use super::{command_for_execution, ActionResult, BlockSelector, ShellCommandExecutor};
use crate::ai::agent::ShellCommandDelay; use crate::ai::agent::ShellCommandDelay;
use crate::terminal::event::{BlockMetadataReceivedEvent, BlockWorkingDirectoryUpdatedEvent}; use crate::terminal::event::{BlockMetadataReceivedEvent, BlockWorkingDirectoryUpdatedEvent};
use crate::terminal::model::block::{BlockId, BlockMetadata}; use crate::terminal::model::block::{BlockId, BlockMetadata};
@@ -13,6 +13,23 @@ use crate::terminal::model::session::active_session::ActiveSession;
use crate::terminal::model::session::Sessions; use crate::terminal::model::session::Sessions;
use crate::terminal::model::terminal_model::{BlockIndex, TerminalModel}; use crate::terminal::model::terminal_model::{BlockIndex, TerminalModel};
use crate::terminal::model_events::{ModelEvent, ModelEventDispatcher}; use crate::terminal::model_events::{ModelEvent, ModelEventDispatcher};
use crate::terminal::shell::ShellType;
#[test]
fn wait_commands_disable_implicit_posix_pagers_without_masking_exit_status() {
let command = "git log -8 --oneline && false";
let decorated = command_for_execution(command, Some(ShellType::Zsh), true);
assert_eq!(
decorated,
"(export PAGER=cat GIT_PAGER=cat GH_PAGER=cat AWS_PAGER=cat SYSTEMD_PAGER=cat; git log -8 --oneline && false)"
);
assert!(!decorated.contains("| command cat"));
assert_eq!(
command_for_execution(command, Some(ShellType::Zsh), false),
command
);
}
/// Locks in the contract that `ShellCommandExecutor`'s requested-command finish /// Locks in the contract that `ShellCommandExecutor`'s requested-command finish
/// detector reacts only to `BlockMetadataReceived` (precmd) and not to /// detector reacts only to `BlockMetadataReceived` (precmd) and not to
+1
View File
@@ -4,6 +4,7 @@ use crate::ai::bedrock::client::BedrockClientConfig;
use crate::ai::openai::client::OpenAIClientConfig; use crate::ai::openai::client::OpenAIClientConfig;
#[allow(dead_code)] #[allow(dead_code)]
#[derive(Clone)]
pub enum ProviderConfig { pub enum ProviderConfig {
Bedrock(BedrockClientConfig), Bedrock(BedrockClientConfig),
OpenAI(OpenAIClientConfig), OpenAI(OpenAIClientConfig),
+25 -15
View File
@@ -80,7 +80,18 @@ pub(crate) async fn prepare_provider_run(
let skill_path_origin = params.session_context.skill_path_origin(); let skill_path_origin = params.session_context.skill_path_origin();
let max_context_tokens = params.context_window_limit; let max_context_tokens = params.context_window_limit;
let mut cli_params = params.clone(); let mut cli_params = params.clone();
cli_params.model = params.cli_agent_model.clone(); let cli_provider_config = match cli_provider_config {
crate::ai::provider::ProviderConfig::None => {
// The CLI model can be absent from a model-specific provider routing table even when
// the base model is usable. Keep monitoring available through the base provider/model.
cli_params.model = params.model.clone();
base_provider_config.clone()
}
provider_config => {
cli_params.model = params.cli_agent_model.clone();
provider_config
}
};
let (base_runtime, prepared) = prepare_provider_profile( let (base_runtime, prepared) = prepare_provider_profile(
base_provider_config, base_provider_config,
@@ -90,20 +101,15 @@ pub(crate) async fn prepare_provider_run(
None, None,
) )
.await?; .await?;
let cli_monitor_profile = match cli_provider_config { let (cli_runtime, cli_prepared) = prepare_provider_profile(
crate::ai::provider::ProviderConfig::None => None, cli_provider_config,
provider_config => { cli_params,
let (runtime, prepared) = prepare_provider_profile( supported_tools,
provider_config, supported_cli_agent_tools,
cli_params, Some(RigRequestMode::Cli),
supported_tools, )
supported_cli_agent_tools, .await?;
Some(RigRequestMode::Cli), let cli_monitor_profile = Some(ProviderRunProfile::new(cli_runtime, cli_prepared.request));
)
.await?;
Some(ProviderRunProfile::new(runtime, prepared.request))
}
};
let PreparedRigTurn { let PreparedRigTurn {
task_id, task_id,
@@ -250,3 +256,7 @@ pub(crate) async fn provider_runtime_for_request(
}; };
Ok(runtime) Ok(runtime)
} }
#[cfg(test)]
#[path = "rig_tests.rs"]
mod tests;
+1 -1
View File
@@ -816,7 +816,7 @@ fn build_system_prompt(
"## Orchestration Mode\nDelegate only independent, bounded work where parallelism materially helps, then synthesize the results.\n\n", "## Orchestration Mode\nDelegate only independent, bounded work where parallelism materially helps, then synthesize the results.\n\n",
), ),
RigRequestMode::Cli => prompt.push_str( RigRequestMode::Cli => prompt.push_str(
"## Running Command Monitor\nThis turn concerns a running or just-finished shell command. Act as its dedicated monitor while still following the user's steering messages. Use the command ID from the tool result for every read/write operation. If the result says the command finished, report its outcome and stop polling. If it says the command is still running, the next assistant output MUST be a tool call: use `read_shell_command_output` with a short delay, or use `interrupt_shell_command` immediately when the user's explicit stop condition is met. Do not end a still-running monitor turn with prose, a status message, or a request for the user to say continue. Never choose a poll interval that crosses a user-specified deadline or stop condition. After an interrupt, poll briefly to verify the outcome. Never start a duplicate command merely to check its state, and never report completion while a result says it is still running.\n\n", "## Running Command Monitor\nThis turn concerns a running or just-finished shell command. Act as its dedicated monitor while still following the user's steering messages. Use the command ID from the tool result for every read/write operation. If the result says the command finished, report its outcome and stop polling. If it says the command is still running, the next assistant output MUST be a tool call. Use `read_shell_command_output` with a short delay for normal progress. If the snapshot clearly shows an interactive pager or editor, do not keep polling: an alternate screen containing `(END)` is `less`, so call `write_to_long_running_shell_command` with input `q` and mode `raw`; for a clearly identified Vim screen, send input `:q` with mode `line`. Poll briefly after sending quit input to verify the outcome. Use `interrupt_shell_command` immediately when the user's explicit stop condition is met. Do not end a still-running monitor turn with prose, a status message, or a request for the user to say continue. Never choose a poll interval that crosses a user-specified deadline or stop condition. After an interrupt, poll briefly to verify the outcome. Never start a duplicate command merely to check its state, and never report completion while a result says it is still running.\n\n",
), ),
RigRequestMode::CompletedCommandAssessment => prompt.push_str( RigRequestMode::CompletedCommandAssessment => prompt.push_str(
"## Completed Command Assessment\nThe monitored command has finished. Use its command, command ID, final terminal output, and the assessment instruction in the latest hidden input to provide the final user-facing outcome. Do not continue polling, request more terminal output, or call tools.\n\n", "## Completed Command Assessment\nThe monitored command has finished. Use its command, command ID, final terminal output, and the assessment instruction in the latest hidden input to provide the final user-facing outcome. Do not continue polling, request more terminal output, or call tools.\n\n",
+2
View File
@@ -429,6 +429,8 @@ fn lrc_snapshot_follow_up_uses_the_cli_monitor_prompt_and_tools() {
assert!(prompt.contains("## Running Command Monitor")); assert!(prompt.contains("## Running Command Monitor"));
assert!(prompt.contains("`read_shell_command_output` with a short delay")); assert!(prompt.contains("`read_shell_command_output` with a short delay"));
assert!(prompt.contains("next assistant output MUST be a tool call")); assert!(prompt.contains("next assistant output MUST be a tool call"));
assert!(prompt.contains("alternate screen containing `(END)` is `less`"));
assert!(prompt.contains("`write_to_long_running_shell_command` with input `q` and mode `raw`"));
assert!(prompt.contains("Do not end a still-running monitor turn with prose")); assert!(prompt.contains("Do not end a still-running monitor turn with prose"));
assert!(prepared assert!(prepared
.request .request
+57
View File
@@ -0,0 +1,57 @@
use super::prepare_provider_run;
use crate::ai::agent::api::RequestParams;
use crate::ai::llms::LLMId;
use crate::ai::openai::client::OpenAIClientConfig;
use crate::ai::provider::ProviderConfig;
use crate::settings::OpenAIProviderKind;
fn openai_config(model: &str) -> OpenAIClientConfig {
OpenAIClientConfig {
kind: OpenAIProviderKind::LiteLLM,
base_url: "http://localhost:4000/v1".to_string(),
api_key: None,
project_id: None,
location: None,
model: Some(model.to_string()),
reasoning_effort: None,
max_input_tokens: Some(128_000),
max_output_tokens: Some(8_192),
use_rig: true,
supports_system_messages: true,
}
}
#[tokio::test]
async fn missing_cli_provider_route_falls_back_to_base_provider_profile() {
let mut params = RequestParams::new_for_test();
params.model = LLMId::from("base-selection");
params.cli_agent_model = LLMId::from("unroutable-cli-selection");
let prepared = prepare_provider_run(
ProviderConfig::OpenAI(openai_config("base-provider-model")),
ProviderConfig::None,
params,
)
.await
.unwrap();
let cli_profile = prepared
.cli_monitor_profile
.expect("base provider should supply the CLI monitor fallback");
assert_eq!(cli_profile.request.model.as_str(), "base-provider-model");
assert!(cli_profile
.request
.system_prompt
.as_deref()
.is_some_and(|prompt| prompt.contains("## Running Command Monitor")));
assert!(cli_profile
.request
.tools
.iter()
.any(|tool| tool.name == "write_to_long_running_shell_command"));
assert!(cli_profile
.request
.tools
.iter()
.any(|tool| tool.name == "read_shell_command_output"));
}