diff --git a/README.md b/README.md index 06b27af..9837d92 100644 --- a/README.md +++ b/README.md @@ -130,6 +130,8 @@ agent-browser stream status # Show runtime streaming state and bound p agent-browser stream disable # Stop runtime WebSocket streaming agent-browser close # Close browser (aliases: quit, exit) agent-browser close --all # Close all active sessions +agent-browser chat "" # AI chat: natural language browser control (single-shot) +agent-browser chat # AI chat: interactive REPL mode ``` ### Get Info @@ -626,6 +628,9 @@ This is useful for multimodal AI models that can reason about visual layout, unl | `--confirm-interactive` | Interactive confirmation prompts; auto-denies if stdin is not a TTY (or `AGENT_BROWSER_CONFIRM_INTERACTIVE` env) | | `--engine ` | Browser engine: `chrome` (default), `lightpanda` (or `AGENT_BROWSER_ENGINE` env) | | `--no-auto-dialog` | Disable automatic dismissal of `alert`/`beforeunload` dialogs (or `AGENT_BROWSER_NO_AUTO_DIALOG` env) | +| `--model ` | AI model for chat command (or `AI_GATEWAY_MODEL` env) | +| `-v`, `--verbose` | Show tool commands and their raw output (chat) | +| `-q`, `--quiet` | Show only AI text responses, hide tool calls (chat) | | `--config ` | Use a custom config file (or `AGENT_BROWSER_CONFIG` env) | | `--debug` | Debug output | @@ -659,7 +664,7 @@ The dashboard displays: ### AI Chat -The dashboard includes an optional AI chat panel powered by the Vercel AI Gateway. Set these environment variables to enable it: +The dashboard includes an optional AI chat panel powered by the Vercel AI Gateway. The same functionality is available directly from the CLI via the `chat` command. Set these environment variables to enable AI chat: ```bash export AI_GATEWAY_API_KEY=gw_your_key_here @@ -667,6 +672,20 @@ export AI_GATEWAY_MODEL=anthropic/claude-sonnet-4.6 # optional, this i export AI_GATEWAY_URL=https://ai-gateway.vercel.sh # optional, this is the default ``` +**CLI usage:** + +```bash +agent-browser chat "open google.com and search for cats" # Single-shot +agent-browser chat # Interactive REPL +agent-browser -q chat "summarize this page" # Quiet mode (text only) +agent-browser -v chat "fill in the login form" # Verbose (show command output) +agent-browser --model openai/gpt-4o chat "take a screenshot" # Override model +``` + +The `chat` command translates natural language instructions into agent-browser commands, executes them, and streams the AI response. In interactive mode, type `quit` to exit. Use `--json` for structured output suitable for agent consumption. + +**Dashboard usage:** + The Chat tab is always visible in the dashboard. When `AI_GATEWAY_API_KEY` is set, the Rust server proxies requests to the gateway and streams responses back using the Vercel AI SDK's UI Message Stream protocol. Without the key, sending a message shows an error inline. ## Configuration diff --git a/cli/src/chat.rs b/cli/src/chat.rs new file mode 100644 index 0000000..0d68223 --- /dev/null +++ b/cli/src/chat.rs @@ -0,0 +1,503 @@ +use std::io::Write as _; +use std::process::exit; + +use serde_json::{json, Value}; + +use crate::color; +use crate::flags::Flags; +use crate::native::stream::chat; + +const DEFAULT_MODEL: &str = "anthropic/claude-sonnet-4.6"; + +#[derive(Clone, Copy, PartialEq)] +enum Verbosity { + Quiet, + Normal, + Verbose, +} + +pub fn run_chat(flags: &Flags, message: Option) { + if !chat::is_chat_enabled() { + if flags.json { + println!( + "{}", + json!({"success": false, "error": "AI_GATEWAY_API_KEY not set. Set the AI_GATEWAY_API_KEY environment variable to enable chat."}) + ); + } else { + eprintln!( + "{} AI_GATEWAY_API_KEY not set. Set the AI_GATEWAY_API_KEY environment variable to enable chat.", + color::error_indicator() + ); + } + exit(1); + } + + let verbosity = if flags.quiet { + Verbosity::Quiet + } else if flags.verbose { + Verbosity::Verbose + } else { + Verbosity::Normal + }; + + let model = flags + .model + .clone() + .unwrap_or_else(|| DEFAULT_MODEL.to_string()); + + let rt = tokio::runtime::Runtime::new().expect("Failed to create tokio runtime"); + + let is_tty = std::io::IsTerminal::is_terminal(&std::io::stdin()); + + match message { + Some(msg) => { + rt.block_on(run_single_turn( + &flags.session, + &model, + &msg, + verbosity, + flags.json, + )); + } + None if !is_tty => { + let mut input = String::new(); + if let Err(e) = std::io::stdin().read_line(&mut input) { + if flags.json { + println!( + "{}", + json!({"success": false, "error": format!("Failed to read stdin: {}", e)}) + ); + } else { + eprintln!("{} Failed to read stdin: {}", color::error_indicator(), e); + } + exit(1); + } + let input = input.trim(); + if input.is_empty() { + if flags.json { + println!( + "{}", + json!({"success": false, "error": "No input provided"}) + ); + } else { + eprintln!("{} No input provided", color::error_indicator()); + } + exit(1); + } + rt.block_on(run_single_turn( + &flags.session, + &model, + input, + verbosity, + flags.json, + )); + } + None => { + rt.block_on(run_interactive( + &flags.session, + &model, + verbosity, + flags.json, + )); + } + } +} + +async fn run_single_turn( + session: &str, + model: &str, + message: &str, + verbosity: Verbosity, + json_mode: bool, +) { + let mut openai_messages: Vec = + vec![json!({"role": "system", "content": chat::get_system_prompt()})]; + openai_messages.push(json!({"role": "user", "content": message})); + + let result = run_chat_turn(session, model, &mut openai_messages, verbosity, json_mode).await; + if !result { + exit(1); + } +} + +async fn run_interactive(session: &str, model: &str, verbosity: Verbosity, json_mode: bool) { + let mut openai_messages: Vec = + vec![json!({"role": "system", "content": chat::get_system_prompt()})]; + + let gateway_url = std::env::var("AI_GATEWAY_URL") + .unwrap_or_else(|_| chat::DEFAULT_AI_GATEWAY_URL.to_string()) + .trim_end_matches('/') + .to_string(); + let api_key = std::env::var("AI_GATEWAY_API_KEY").unwrap_or_default(); + let url = format!("{}/v1/chat/completions", gateway_url); + let client = chat::http_client(); + + loop { + if !json_mode { + eprint!("{} ", color::cyan(">")); + let _ = std::io::stderr().flush(); + } + + let mut input = String::new(); + match std::io::stdin().read_line(&mut input) { + Ok(0) => break, + Err(_) => break, + Ok(_) => {} + } + + let input = input.trim(); + if input.is_empty() { + continue; + } + + if matches!(input, "quit" | "exit" | "q") { + break; + } + + openai_messages.push(json!({"role": "user", "content": input})); + + // Compaction check + let total_chars = chat::estimate_chars(&openai_messages); + if total_chars > chat::COMPACT_THRESHOLD_CHARS + && openai_messages.len() > chat::KEEP_RECENT_MESSAGES + 2 + { + let split = chat::find_safe_split(&openai_messages, chat::KEEP_RECENT_MESSAGES); + let to_summarize = &openai_messages[1..split]; + if let Some(summary) = + chat::summarize_for_compaction(client, &url, &api_key, model, to_summarize).await + { + let summary_msg = json!({ + "role": "system", + "content": format!("[Conversation summary]\n{}", summary) + }); + let recent = openai_messages[split..].to_vec(); + openai_messages = vec![openai_messages[0].clone(), summary_msg]; + openai_messages.extend(recent); + } + } + + let success = + run_chat_turn(session, model, &mut openai_messages, verbosity, json_mode).await; + + if !success && !json_mode { + // Continue the loop on error; don't exit interactive mode + } + + if !json_mode { + eprintln!(); + } + } +} + +/// Runs one chat turn: sends messages to the gateway, streams text/tool calls, +/// executes tools in a loop until the model is done. Appends assistant and tool +/// messages to `openai_messages`. Returns true on success. +async fn run_chat_turn( + session: &str, + model: &str, + openai_messages: &mut Vec, + verbosity: Verbosity, + json_mode: bool, +) -> bool { + let gateway_url = std::env::var("AI_GATEWAY_URL") + .unwrap_or_else(|_| chat::DEFAULT_AI_GATEWAY_URL.to_string()) + .trim_end_matches('/') + .to_string(); + let api_key = match std::env::var("AI_GATEWAY_API_KEY") { + Ok(k) => k, + Err(_) => { + if json_mode { + println!( + "{}", + json!({"success": false, "error": "AI_GATEWAY_API_KEY not set"}) + ); + } else { + eprintln!("{} AI_GATEWAY_API_KEY not set", color::error_indicator()); + } + return false; + } + }; + + let tools: Value = serde_json::from_str(chat::CHAT_TOOLS).unwrap(); + let url = format!("{}/v1/chat/completions", gateway_url); + let client = chat::http_client(); + + let total_deadline = tokio::time::Instant::now() + std::time::Duration::from_secs(300); + let tool_timeout = std::time::Duration::from_secs(60); + + let mut all_text = String::new(); + let mut all_tool_calls: Vec = Vec::new(); + let mut had_text = false; + + for _step in 0..50 { + if tokio::time::Instant::now() >= total_deadline { + if json_mode { + println!( + "{}", + json!({"success": false, "error": "Chat session timed out (5 minute limit)."}) + ); + } else { + eprintln!( + "\n{} Chat session timed out (5 minute limit).", + color::error_indicator() + ); + } + return false; + } + + let gateway_body = json!({ + "model": model, + "messages": openai_messages, + "tools": tools, + "stream": true, + }); + + let gw_response = match client + .post(&url) + .header("Authorization", format!("Bearer {}", api_key)) + .header("Content-Type", "application/json") + .body(gateway_body.to_string()) + .send() + .await + { + Ok(r) => r, + Err(e) => { + if json_mode { + println!( + "{}", + json!({"success": false, "error": format!("Gateway request failed: {}", e)}) + ); + } else { + eprintln!( + "\n{} Gateway request failed: {}", + color::error_indicator(), + e + ); + } + return false; + } + }; + + if !gw_response.status().is_success() { + let body_text = gw_response.text().await.unwrap_or_default(); + if json_mode { + println!("{}", json!({"success": false, "error": body_text})); + } else { + eprintln!("\n{} {}", color::error_indicator(), body_text); + } + return false; + } + + let (text_chunks, tool_calls) = + parse_gateway_stream(gw_response, verbosity, json_mode).await; + + if !text_chunks.is_empty() { + let text = text_chunks.join(""); + all_text.push_str(&text); + if !json_mode { + if !had_text && verbosity != Verbosity::Quiet { + // Add blank line before text if we showed tool calls + if !all_tool_calls.is_empty() { + println!(); + } + } + had_text = true; + } + + let mut content = json!(text); + if let Some(last) = openai_messages.last() { + if last.get("role").and_then(|r| r.as_str()) == Some("assistant") + && last.get("tool_calls").is_some() + { + content = json!(text); + } + } + openai_messages.push(json!({"role": "assistant", "content": content})); + } + + if tool_calls.is_empty() { + break; + } + + let tc_values: Vec = tool_calls + .iter() + .map(|(id, name, args)| { + json!({"id": id, "type": "function", "function": {"name": name, "arguments": args}}) + }) + .collect(); + + if text_chunks.is_empty() { + openai_messages.push(json!({"role": "assistant", "tool_calls": tc_values})); + } else { + // If we had both text and tool calls in the same response, merge them + if let Some(last) = openai_messages.last_mut() { + if last.get("role").and_then(|r| r.as_str()) == Some("assistant") + && last.get("tool_calls").is_none() + { + last["tool_calls"] = json!(tc_values); + } else { + openai_messages.push(json!({"role": "assistant", "tool_calls": tc_values})); + } + } + } + + for (tc_id, _tc_name, tc_args) in &tool_calls { + let input: Value = serde_json::from_str(tc_args).unwrap_or(json!({})); + let command = input.get("command").and_then(|c| c.as_str()).unwrap_or(""); + + if !json_mode && verbosity != Verbosity::Quiet { + eprintln!("{}", color::dim(&format!("> {}", command))); + } + + let result = + match tokio::time::timeout(tool_timeout, chat::execute_chat_tool(session, command)) + .await + { + Ok(r) => r, + Err(_) => "Tool execution timed out after 60 seconds.".to_string(), + }; + + if !json_mode && verbosity == Verbosity::Verbose { + for line in result.lines() { + eprintln!(" {}", color::dim(line)); + } + } + + all_tool_calls.push(json!({ + "command": command, + "output": result + })); + + openai_messages.push(json!({ + "role": "tool", + "tool_call_id": tc_id, + "content": result + })); + } + } + + if json_mode { + println!( + "{}", + json!({ + "success": true, + "text": all_text, + "tool_calls": all_tool_calls + }) + ); + } else if !had_text && !json_mode { + // Model returned only tool calls with no final text; print newline for clean output + println!(); + } + + true +} + +/// Parses the SSE stream from the AI gateway, printing text deltas to stdout in +/// real-time. Returns (collected_text_chunks, tool_calls). +async fn parse_gateway_stream( + gw_response: reqwest::Response, + verbosity: Verbosity, + json_mode: bool, +) -> (Vec, Vec<(String, String, String)>) { + use futures_util::StreamExt as _; + + let mut text_chunks: Vec = Vec::new(); + let mut tool_call_args: std::collections::HashMap = + std::collections::HashMap::new(); + let mut byte_stream = gw_response.bytes_stream(); + let mut buffer = String::new(); + + while let Some(chunk_result) = byte_stream.next().await { + let chunk = match chunk_result { + Ok(c) => c, + Err(_) => break, + }; + + buffer.push_str(&String::from_utf8_lossy(&chunk)); + + while let Some(newline_pos) = buffer.find('\n') { + let line = buffer[..newline_pos].trim_end_matches('\r').to_string(); + buffer = buffer[newline_pos + 1..].to_string(); + + if line.is_empty() { + continue; + } + let Some(data) = line.strip_prefix("data: ") else { + continue; + }; + if data == "[DONE]" { + let tool_calls = collect_tool_calls(&mut tool_call_args); + if !json_mode && !text_chunks.is_empty() { + // End the streamed text line + let _ = std::io::stdout().flush(); + } + return (text_chunks, tool_calls); + } + let Ok(sse_json) = serde_json::from_str::(data) else { + continue; + }; + let delta = sse_json + .get("choices") + .and_then(|c| c.get(0)) + .and_then(|c| c.get("delta")); + let Some(delta) = delta else { continue }; + + if let Some(text) = delta.get("content").and_then(|c| c.as_str()) { + if !text.is_empty() { + text_chunks.push(text.to_string()); + if !json_mode && verbosity != Verbosity::Quiet { + print!("{}", text); + let _ = std::io::stdout().flush(); + } + } + } + + if let Some(tcs) = delta.get("tool_calls").and_then(|t| t.as_array()) { + for tc in tcs { + let idx = tc.get("index").and_then(|i| i.as_u64()).unwrap_or(0) as usize; + if let std::collections::hash_map::Entry::Vacant(e) = tool_call_args.entry(idx) + { + let id = tc + .get("id") + .and_then(|i| i.as_str()) + .unwrap_or("") + .to_string(); + let name = tc + .get("function") + .and_then(|f| f.get("name")) + .and_then(|n| n.as_str()) + .unwrap_or("") + .to_string(); + e.insert((id, name, String::new())); + } + if let Some(arg_delta) = tc + .get("function") + .and_then(|f| f.get("arguments")) + .and_then(|a| a.as_str()) + { + let entry = tool_call_args.get_mut(&idx).unwrap(); + entry.2.push_str(arg_delta); + } + } + } + } + } + + if !json_mode && !text_chunks.is_empty() { + let _ = std::io::stdout().flush(); + } + let tool_calls = collect_tool_calls(&mut tool_call_args); + (text_chunks, tool_calls) +} + +fn collect_tool_calls( + map: &mut std::collections::HashMap, +) -> Vec<(String, String, String)> { + let mut indices: Vec = map.keys().copied().collect(); + indices.sort(); + indices + .into_iter() + .filter_map(|idx| map.remove(&idx)) + .collect() +} diff --git a/cli/src/commands.rs b/cli/src/commands.rs index 1c6e365..274f997 100644 --- a/cli/src/commands.rs +++ b/cli/src/commands.rs @@ -2382,6 +2382,9 @@ mod tests { idle_timeout: None, default_timeout: None, no_auto_dialog: false, + model: None, + verbose: false, + quiet: false, } } diff --git a/cli/src/flags.rs b/cli/src/flags.rs index 49af096..6e517c9 100644 --- a/cli/src/flags.rs +++ b/cli/src/flags.rs @@ -88,6 +88,7 @@ pub struct Config { pub screenshot_format: Option, pub idle_timeout: Option, pub no_auto_dialog: Option, + pub model: Option, } impl Config { @@ -134,6 +135,7 @@ impl Config { screenshot_format: other.screenshot_format.or(self.screenshot_format), idle_timeout: other.idle_timeout.or(self.idle_timeout), no_auto_dialog: other.no_auto_dialog.or(self.no_auto_dialog), + model: other.model.or(self.model), } } } @@ -219,6 +221,7 @@ fn extract_config_path(args: &[String]) -> Option> { "--screenshot-quality", "--screenshot-format", "--idle-timeout", + "--model", ]; let mut i = 0; while i < args.len() { @@ -302,6 +305,9 @@ pub struct Flags { pub idle_timeout: Option, // Canonical milliseconds string for AGENT_BROWSER_IDLE_TIMEOUT_MS pub default_timeout: Option, // AGENT_BROWSER_DEFAULT_TIMEOUT in ms pub no_auto_dialog: bool, + pub model: Option, + pub verbose: bool, + pub quiet: bool, // Track which launch-time options were explicitly passed via CLI // (as opposed to being set only via environment variables) @@ -438,6 +444,9 @@ pub fn parse_flags(args: &[String]) -> Flags { .and_then(|s| s.parse::().ok()), no_auto_dialog: env_var_is_truthy("AGENT_BROWSER_NO_AUTO_DIALOG") || config.no_auto_dialog.unwrap_or(false), + model: env::var("AI_GATEWAY_MODEL").ok().or(config.model), + verbose: false, + quiet: false, cli_executable_path: false, cli_extensions: false, cli_profile: false, @@ -719,6 +728,18 @@ pub fn parse_flags(args: &[String]) -> Flags { i += 1; } } + "--model" => { + if let Some(s) = args.get(i + 1) { + flags.model = Some(s.clone()); + i += 1; + } + } + "-v" | "--verbose" => { + flags.verbose = true; + } + "-q" | "--quiet" => { + flags.quiet = true; + } "--config" => { // Already handled by load_config(); skip the value i += 1; @@ -746,6 +767,10 @@ pub fn clean_args(args: &[String]) -> Vec { "--content-boundaries", "--confirm-interactive", "--no-auto-dialog", + "-v", + "--verbose", + "-q", + "--quiet", ]; // Global flags that always take a value (need to skip the next arg too) const GLOBAL_FLAGS_WITH_VALUE: &[&str] = &[ @@ -776,6 +801,7 @@ pub fn clean_args(args: &[String]) -> Vec { "--screenshot-quality", "--screenshot-format", "--idle-timeout", + "--model", ]; let mut i = 0; diff --git a/cli/src/main.rs b/cli/src/main.rs index a8fa038..31d4ab3 100644 --- a/cli/src/main.rs +++ b/cli/src/main.rs @@ -1,3 +1,4 @@ +mod chat; mod color; mod commands; mod connection; @@ -704,6 +705,17 @@ fn main() { return; } + // Handle chat command + if clean.first().map(|s| s.as_str()) == Some("chat") { + let message = if clean.len() > 1 { + Some(clean[1..].join(" ")) + } else { + None + }; + chat::run_chat(&flags, message); + return; + } + let mut cmd = match parse_command(&clean, &flags) { Ok(c) => c, Err(e) => { diff --git a/cli/src/native/stream/chat.rs b/cli/src/native/stream/chat.rs index e271735..884f22a 100644 --- a/cli/src/native/stream/chat.rs +++ b/cli/src/native/stream/chat.rs @@ -6,15 +6,15 @@ use tokio::io::AsyncWriteExt; use super::http::cors_headers_for_origin; -const DEFAULT_AI_GATEWAY_URL: &str = "https://ai-gateway.vercel.sh"; +pub(crate) const DEFAULT_AI_GATEWAY_URL: &str = "https://ai-gateway.vercel.sh"; static HTTP_CLIENT: OnceLock = OnceLock::new(); -fn http_client() -> &'static reqwest::Client { +pub(crate) fn http_client() -> &'static reqwest::Client { HTTP_CLIENT.get_or_init(reqwest::Client::new) } -fn is_chat_enabled() -> bool { +pub(crate) fn is_chat_enabled() -> bool { std::env::var("AI_GATEWAY_API_KEY").is_ok() } @@ -121,7 +121,7 @@ fn strip_frontmatter(s: &str) -> &str { } } -fn get_system_prompt() -> &'static str { +pub(crate) fn get_system_prompt() -> &'static str { static PROMPT: OnceLock = OnceLock::new(); PROMPT.get_or_init(|| { let skills = load_skills(); @@ -153,12 +153,12 @@ The following skill references describe agent-browser capabilities in detail. Us }) } -const CHAT_TOOLS: &str = r#"[{"type":"function","function":{"name":"agent_browser","description":"Execute an agent-browser command. Runs against the active session by default. Add --session to target or create a different session, and --engine to choose a browser engine.","parameters":{"type":"object","properties":{"command":{"type":"string","description":"The command to execute, e.g. 'agent-browser open https://google.com' or 'agent-browser --session new-session open https://example.com' or 'agent-browser snapshot -i' or 'agent-browser click @e3'"}},"required":["command"]}}}]"#; +pub(crate) const CHAT_TOOLS: &str = r#"[{"type":"function","function":{"name":"agent_browser","description":"Execute an agent-browser command. Runs against the active session by default. Add --session to target or create a different session, and --engine to choose a browser engine.","parameters":{"type":"object","properties":{"command":{"type":"string","description":"The command to execute, e.g. 'agent-browser open https://google.com' or 'agent-browser --session new-session open https://example.com' or 'agent-browser snapshot -i' or 'agent-browser click @e3'"}},"required":["command"]}}}]"#; -const COMPACT_THRESHOLD_CHARS: usize = 200_000; -const KEEP_RECENT_MESSAGES: usize = 6; +pub(crate) const COMPACT_THRESHOLD_CHARS: usize = 200_000; +pub(crate) const KEEP_RECENT_MESSAGES: usize = 6; -fn estimate_chars(messages: &[Value]) -> usize { +pub(crate) fn estimate_chars(messages: &[Value]) -> usize { messages .iter() .map(|m| { @@ -181,7 +181,7 @@ fn estimate_chars(messages: &[Value]) -> usize { .sum() } -fn find_safe_split(messages: &[Value], keep_recent: usize) -> usize { +pub(crate) fn find_safe_split(messages: &[Value], keep_recent: usize) -> usize { if messages.len() <= keep_recent + 1 { return 1; } @@ -230,7 +230,7 @@ fn build_summary_text(messages: &[Value]) -> String { text } -async fn summarize_for_compaction( +pub(crate) async fn summarize_for_compaction( client: &reqwest::Client, url: &str, api_key: &str, @@ -454,7 +454,7 @@ const ALLOWED_COMMANDS: &[&str] = &[ const ALLOWED_GLOBAL_FLAGS: &[&str] = &["--session", "--engine"]; -async fn execute_chat_tool(session: &str, command: &str) -> String { +pub(crate) async fn execute_chat_tool(session: &str, command: &str) -> String { let exe = match std::env::current_exe() { Ok(p) => p, Err(e) => return format!("Failed to resolve executable: {}", e), diff --git a/cli/src/native/stream/mod.rs b/cli/src/native/stream/mod.rs index 9fd433a..81aaa5f 100644 --- a/cli/src/native/stream/mod.rs +++ b/cli/src/native/stream/mod.rs @@ -1,5 +1,5 @@ mod cdp_loop; -mod chat; +pub(crate) mod chat; mod dashboard; mod discovery; mod http; diff --git a/cli/src/output.rs b/cli/src/output.rs index 4f17720..c444f65 100644 --- a/cli/src/output.rs +++ b/cli/src/output.rs @@ -2679,6 +2679,40 @@ Examples: "## } + "chat" => { + r##" +agent-browser chat - Natural language browser control via AI + +Usage: + agent-browser chat Single-shot: execute instruction and exit + agent-browser chat Interactive REPL (when stdin is a TTY) + echo "instruction" | agent-browser chat Piped input + +Sends natural language instructions to an AI model that translates them +into agent-browser commands and executes them against the active session. +Requires AI_GATEWAY_API_KEY to be set. + +In interactive mode, type "quit", "exit", or "q" to leave the REPL. + +Chat Options: + --model AI model (or AI_GATEWAY_MODEL env, default: anthropic/claude-sonnet-4.6) + -v, --verbose Show tool commands and their raw output + -q, --quiet Show only the AI text response (hide tool calls) + +Global Options: + --json Structured JSON output per turn + --session Target session for commands + +Examples: + agent-browser chat "open google.com and search for cats" + agent-browser chat "take a screenshot of the current page" + agent-browser -q chat "summarize this page" + agent-browser -v chat "fill in the login form with test@example.com" + agent-browser --model openai/gpt-4o chat "navigate to hacker news" + agent-browser chat +"## + } + _ => return false, }; println!("{}", help.trim()); @@ -2794,6 +2828,11 @@ Sessions: session Show current session name session list List active sessions +Chat (AI): + chat Send a natural language instruction (single-shot) + chat Start interactive chat (REPL mode when stdin is a TTY) + Options: --model , -v/--verbose, -q/--quiet + Dashboard: dashboard [start] Start the dashboard server (default port: 4848) dashboard start --port Start on a specific port @@ -2856,6 +2895,9 @@ Options: --confirm-interactive Interactive confirmation prompts; auto-denies if stdin is not a TTY (or AGENT_BROWSER_CONFIRM_INTERACTIVE) --engine Browser engine: chrome (default), lightpanda (or AGENT_BROWSER_ENGINE) --no-auto-dialog Disable automatic dismissal of alert/beforeunload dialogs (or AGENT_BROWSER_NO_AUTO_DIALOG) + --model AI model for chat (or AI_GATEWAY_MODEL env) + -v, --verbose Show tool commands and their raw output + -q, --quiet Show only AI text responses (hide tool calls) --config Use a custom config file (or AGENT_BROWSER_CONFIG env) --debug Debug output --version, -V Show version @@ -2920,8 +2962,8 @@ Environment: AGENT_BROWSER_SCREENSHOT_QUALITY JPEG quality 0-100 AGENT_BROWSER_SCREENSHOT_FORMAT Screenshot format: png, jpeg AI_GATEWAY_URL Vercel AI Gateway base URL (default: https://ai-gateway.vercel.sh) - AI_GATEWAY_API_KEY API key for the AI Gateway (enables dashboard AI chat) - AI_GATEWAY_MODEL Default AI model (default: anthropic/claude-sonnet-4.6) + AI_GATEWAY_API_KEY API key for the AI Gateway (enables chat command and dashboard AI chat) + AI_GATEWAY_MODEL Default AI model (default: anthropic/claude-sonnet-4.6, or --model flag) Install: npm install -g agent-browser # npm @@ -2948,6 +2990,9 @@ Examples: agent-browser --profile ~/.myapp open example.com # Persistent custom profile agent-browser profiles # List available Chrome profiles agent-browser --session-name myapp open example.com # Auto-save/restore state + agent-browser chat "open google.com and search for cats" # AI chat (single-shot) + agent-browser chat # AI chat (interactive REPL) + agent-browser -q chat "summarize this page" # Quiet mode (text only) Command Chaining: Chain commands with && in a single shell call (browser persists via daemon): diff --git a/docs/src/app/commands/page.mdx b/docs/src/app/commands/page.mdx index 54ef73f..de6dfce 100644 --- a/docs/src/app/commands/page.mdx +++ b/docs/src/app/commands/page.mdx @@ -346,6 +346,28 @@ agent-browser dashboard stop # Stop the dashboard server agent-browser dashboard install # Install the dashboard files ``` +## Chat + +Use natural language to control the browser via AI. The `chat` command translates instructions into agent-browser commands, executes them, and streams the AI response. Requires `AI_GATEWAY_API_KEY` to be set. + +```bash +agent-browser chat "open google.com and search for cats" # Single-shot instruction +agent-browser chat # Interactive REPL (type quit to exit) +echo "summarize this page" | agent-browser chat # Piped input +agent-browser -q chat "summarize this page" # Quiet: text only, no tool calls shown +agent-browser -v chat "fill in the login form" # Verbose: show commands and their output +agent-browser --model openai/gpt-4o chat "take a screenshot" # Override the default AI model +agent-browser --json chat "open example.com" # Structured JSON output +``` + +Chat-specific options: + +```bash +--model # AI model (or AI_GATEWAY_MODEL env, default: anthropic/claude-sonnet-4.6) +-v, --verbose # Show tool commands and their raw output +-q, --quiet # Show only the AI text response (hide tool calls) +``` + ## Navigation ```bash @@ -388,6 +410,9 @@ agent-browser reload # Reload page --action-policy # Path to action policy JSON file --confirm-actions # Action categories requiring confirmation --confirm-interactive # Interactive confirmation prompts (auto-denies if stdin is not a TTY) +--model # AI model for chat (or AI_GATEWAY_MODEL env) +-v, --verbose # Show tool commands and their raw output (chat) +-q, --quiet # Show only AI text responses (chat) --config # Use a custom config file --debug # Debug output ``` diff --git a/skills/agent-browser/SKILL.md b/skills/agent-browser/SKILL.md index a3188e5..19176ba 100644 --- a/skills/agent-browser/SKILL.md +++ b/skills/agent-browser/SKILL.md @@ -224,6 +224,13 @@ agent-browser diff screenshot --baseline before.png # Visual pixel diff agent-browser diff url # Compare two pages agent-browser diff url --wait-until networkidle # Custom wait strategy agent-browser diff url --selector "#main" # Scope to element + +# Chat (AI natural language control) +agent-browser chat "open google.com and search for cats" # Single-shot instruction +agent-browser chat # Interactive REPL mode +agent-browser -q chat "summarize this page" # Quiet (text only, no tool calls) +agent-browser -v chat "fill in the login form" # Verbose (show command output) +agent-browser --model openai/gpt-4o chat "take a screenshot" # Override model ``` ## Streaming