Initial public release of Warp.

Repo-Sync-Origin: warpdotdev/warp-internal@12af1d983b
This commit is contained in:
David Stern
2026-04-28 08:43:33 -05:00
commit 0dbd3d567a
4982 changed files with 1431549 additions and 0 deletions
+120
View File
@@ -0,0 +1,120 @@
use std::collections::HashSet;
use lazy_static::lazy_static;
use natural_language_detection::check_if_token_has_shell_syntax;
use warp_completer::ParsedTokensSnapshot;
/// The percentage of input tokens that can be described by our completion engine before
/// we consider the input as a shell command. This could be tuned.
const DETECT_AS_COMMAND_THRESHOLD: f32 = 0.5;
/// Threshold for the case when we have a low number of input tokens and require a higher
/// confidence level. This could be tuned.
const DETECT_AS_COMMAND_LOW_TOKEN_THRESHOLD: f32 = 0.7;
lazy_static! {
/// One-off commands / keywords that should trigger a shell command classification.
///
/// `claude`, `codex`, and `gemini` are not actually _really_ one-off shell command keywords,
/// but false-positive NL classifications for these inputs (where the user was trying to use
/// claude code, codex CLI, or gemini CLI) suck, because the user often thinks we're
/// intentionally trying to push them away from those CLIs into Agent Mode, so we mitigate the
/// risk by always treating as shell.
static ref ONE_OFF_SHELL_COMMAND_KEYWORDS: HashSet<&'static str> = HashSet::from(["#", "echo", "man", "sudo", "claude", "codex", "gemini"]);
static ref ONE_OFF_NATURAL_LANGUAGE_WORDS: HashSet<&'static str> = HashSet::from(["hello", "hi", "hey", "hola", "thanks", "explain", "yes", "no", "what", "nice", "1. "]);
/// A set of words that should trigger an AI classification if they are the entire input
/// and the input is a follow-up to an agent response.
static ref AGENT_FOLLOW_UP_INPUTS: HashSet<&'static str> = HashSet::from(["yes", "continue", "do it"]);
}
pub fn is_agent_follow_up_input(input: &str) -> bool {
AGENT_FOLLOW_UP_INPUTS.contains(input)
}
pub fn is_one_off_shell_command_keyword(word: &str) -> bool {
ONE_OFF_SHELL_COMMAND_KEYWORDS.contains(word)
}
/// Returns true if the word is a one-off natural language word or a prefix of a one-off natural language word.
pub fn is_one_off_natural_language_word_or_prefix(word: &str) -> bool {
is_one_off_natural_language_word(word) || is_prefix_of_natural_language_word(word)
}
// Returns true if the word is a one-off natural language word.
pub fn is_one_off_natural_language_word(word: &str) -> bool {
ONE_OFF_NATURAL_LANGUAGE_WORDS.contains(word)
}
/// Checks if the input string is a prefix of any word in the ONE_OFF_NATURAL_LANGUAGE_WORDS set.
/// This helps with progressive typing detection to avoid mode flipping.
pub fn is_prefix_of_natural_language_word(input: &str) -> bool {
// input is already lowercase from caller
ONE_OFF_NATURAL_LANGUAGE_WORDS
.iter()
.any(|word| word.starts_with(input))
}
pub async fn is_likely_shell_command(
input: &ParsedTokensSnapshot,
word_tokens_count: usize,
) -> bool {
const YIELD_BATCH_SIZE: usize = 5;
let mut likely_command_token_count = 0;
let total_token_count = input.parsed_tokens.len();
let mut is_first_token_command = false;
for (idx, token) in input.parsed_tokens.iter().enumerate() {
// Periodically, yield to the executor so this task can be aborted if
// requested.
if idx % YIELD_BATCH_SIZE == 0 {
futures_lite::future::yield_now().await;
}
// Early return if we encounter a one-off command / keyword at the beginning of the line.
if token.token_index == 0 && ONE_OFF_SHELL_COMMAND_KEYWORDS.contains(&token.token.as_str())
{
return true;
}
if token.token_description.is_some()
|| check_if_token_has_shell_syntax(token.token.as_str())
{
likely_command_token_count += 1;
}
if token.token_index == 0 {
is_first_token_command = token.token_description.is_some();
}
}
// When token count is lower than 2, we should make sure all tokens
// are matching the target classification category.
let command_threshold = if total_token_count <= 2 {
1.0
} else if total_token_count <= 4 {
DETECT_AS_COMMAND_LOW_TOKEN_THRESHOLD
} else {
DETECT_AS_COMMAND_THRESHOLD
};
// Classify as shell if:
// 1) We hit significant threshold of likely shell command tokens.
// 2) When there are fewer than 3 tokens, the first token is a valid top-level command.
if likely_command_token_count >= (total_token_count as f32 * command_threshold) as usize
|| (word_tokens_count < 3 && is_first_token_command)
{
return true;
}
false
}
/// Returns true if the first token is a command that is installed on the system.
pub fn is_installed_binary(input: &ParsedTokensSnapshot) -> bool {
input
.parsed_tokens
.first()
.map(|token| token.token_description.is_some())
.unwrap_or(false)
}