mirror of
https://github.com/outbackdingo/optimclaw.git
synced 2026-08-25 14:53:34 +00:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fd41bdf4be | ||
|
|
de5a1c7b0d | ||
|
|
9ce3a9fc53 |
Generated
+16
-5
@@ -3150,7 +3150,7 @@ dependencies = [
|
||||
"libc",
|
||||
"percent-encoding",
|
||||
"pin-project-lite",
|
||||
"socket2 0.5.10",
|
||||
"socket2 0.6.3",
|
||||
"system-configuration",
|
||||
"tokio",
|
||||
"tower-service",
|
||||
@@ -3439,6 +3439,7 @@ dependencies = [
|
||||
"pgvector",
|
||||
"postgres-types",
|
||||
"pretty_assertions",
|
||||
"pty-process",
|
||||
"rand 0.8.5",
|
||||
"readabilityrs",
|
||||
"refinery",
|
||||
@@ -3524,7 +3525,7 @@ checksum = "3640c1c38b8e4e43584d8df18be5fc6b0aa314ce6ebf51b53313d4306cca8e46"
|
||||
dependencies = [
|
||||
"hermit-abi",
|
||||
"libc",
|
||||
"windows-sys 0.59.0",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -4906,6 +4907,16 @@ dependencies = [
|
||||
"syn 1.0.109",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "pty-process"
|
||||
version = "0.5.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "71cec9e2670207c5ebb9e477763c74436af3b9091dd550b9fb3c1bec7f3ea266"
|
||||
dependencies = [
|
||||
"rustix 1.1.4",
|
||||
"tokio",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "pulley-interpreter"
|
||||
version = "28.0.1"
|
||||
@@ -4930,7 +4941,7 @@ dependencies = [
|
||||
"quinn-udp",
|
||||
"rustc-hash 2.1.1",
|
||||
"rustls 0.23.37",
|
||||
"socket2 0.5.10",
|
||||
"socket2 0.6.3",
|
||||
"thiserror 2.0.18",
|
||||
"tokio",
|
||||
"tracing",
|
||||
@@ -4967,9 +4978,9 @@ dependencies = [
|
||||
"cfg_aliases",
|
||||
"libc",
|
||||
"once_cell",
|
||||
"socket2 0.5.10",
|
||||
"socket2 0.6.3",
|
||||
"tracing",
|
||||
"windows-sys 0.59.0",
|
||||
"windows-sys 0.60.2",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
|
||||
@@ -189,6 +189,10 @@ json5 = { version = "0.4", optional = true }
|
||||
[target.'cfg(target_os = "macos")'.dependencies]
|
||||
security-framework = "3"
|
||||
|
||||
# PTY allocation for Claude CLI stdout buffering fix (Unix only)
|
||||
[target.'cfg(unix)'.dependencies]
|
||||
pty-process = { version = "0.5", features = ["async"] }
|
||||
|
||||
# Linux secret-service (GNOME Keyring, KWallet)
|
||||
[target.'cfg(target_os = "linux")'.dependencies]
|
||||
secret-service = { version = "4", features = ["rt-tokio-crypto-rust"] }
|
||||
|
||||
+118
-23
@@ -28,6 +28,9 @@ use std::{cmp::Ordering, collections::HashMap};
|
||||
use ed25519_dalek::{Signature, Verifier, VerifyingKey};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
/// Discord REST API v10 base URL.
|
||||
const DISCORD_API_BASE: &str = "https://discord.com/api/v10";
|
||||
|
||||
use exports::near::agent::channel::{
|
||||
AgentResponse, ChannelConfig, Guest, HttpEndpointConfig, IncomingHttpRequest,
|
||||
OutgoingHttpResponse, PollConfig, StatusUpdate,
|
||||
@@ -427,7 +430,7 @@ impl Guest for DiscordChannel {
|
||||
(
|
||||
"PATCH",
|
||||
format!(
|
||||
"https://discord.com/api/v10/webhooks/{}/{}/messages/@original",
|
||||
"{DISCORD_API_BASE}/webhooks/{}/{}/messages/@original",
|
||||
application_id, token
|
||||
),
|
||||
)
|
||||
@@ -438,20 +441,7 @@ impl Guest for DiscordChannel {
|
||||
payload["allowed_mentions"] = serde_json::json!({
|
||||
"replied_user": true
|
||||
});
|
||||
let mention_payload = serde_json::to_vec(&payload)
|
||||
.map_err(|e| format!("Failed to serialize mention payload: {}", e))?;
|
||||
let mention_url = format!(
|
||||
"https://discord.com/api/v10/channels/{}/messages",
|
||||
metadata.channel_id
|
||||
);
|
||||
let result = channel_host::http_request(
|
||||
"POST",
|
||||
&mention_url,
|
||||
&discord_auth_headers_json(true),
|
||||
Some(&mention_payload),
|
||||
None,
|
||||
);
|
||||
return map_discord_response(result);
|
||||
return send_channel_message(&metadata.channel_id, payload);
|
||||
} else {
|
||||
return Err("Unsupported Discord response metadata".to_string());
|
||||
};
|
||||
@@ -469,8 +459,8 @@ impl Guest for DiscordChannel {
|
||||
|
||||
fn on_status(_update: StatusUpdate) {}
|
||||
|
||||
fn on_broadcast(_user_id: String, _response: AgentResponse) -> Result<(), String> {
|
||||
Err("broadcast not yet implemented for Discord channel".to_string())
|
||||
fn on_broadcast(user_id: String, response: AgentResponse) -> Result<(), String> {
|
||||
broadcast_dm(&user_id, &response.content)
|
||||
}
|
||||
|
||||
fn on_shutdown() {
|
||||
@@ -501,6 +491,21 @@ fn map_discord_response(
|
||||
}
|
||||
}
|
||||
|
||||
/// Post a JSON payload to a Discord channel as a new message.
|
||||
fn send_channel_message(channel_id: &str, payload: serde_json::Value) -> Result<(), String> {
|
||||
let payload_bytes = serde_json::to_vec(&payload)
|
||||
.map_err(|e| format!("Failed to serialize message: {}", e))?;
|
||||
let url = format!("{DISCORD_API_BASE}/channels/{}/messages", channel_id);
|
||||
let result = channel_host::http_request(
|
||||
"POST",
|
||||
&url,
|
||||
&discord_auth_headers_json(true),
|
||||
Some(&payload_bytes),
|
||||
None,
|
||||
);
|
||||
map_discord_response(result)
|
||||
}
|
||||
|
||||
fn load_runtime_config() -> DiscordRuntimeConfig {
|
||||
channel_host::workspace_read("config.json")
|
||||
.and_then(|raw| serde_json::from_str::<DiscordRuntimeConfig>(&raw).ok())
|
||||
@@ -539,7 +544,7 @@ fn get_or_fetch_bot_id() -> Option<String> {
|
||||
|
||||
let response = channel_host::http_request(
|
||||
"GET",
|
||||
"https://discord.com/api/v10/users/@me",
|
||||
&format!("{DISCORD_API_BASE}/users/@me"),
|
||||
&discord_auth_headers_json(false),
|
||||
None,
|
||||
Some(10_000),
|
||||
@@ -659,7 +664,7 @@ fn poll_channel_mentions(channel_id: &str, bot_id: &str) {
|
||||
|
||||
fn fetch_latest_message_id(channel_id: &str) -> Option<String> {
|
||||
let url = format!(
|
||||
"https://discord.com/api/v10/channels/{}/messages?limit=1",
|
||||
"{DISCORD_API_BASE}/channels/{}/messages?limit=1",
|
||||
channel_id
|
||||
);
|
||||
let response = channel_host::http_request(
|
||||
@@ -697,7 +702,7 @@ fn fetch_messages_after_cursor(
|
||||
|
||||
for page in 0..MAX_PAGES {
|
||||
let url = format!(
|
||||
"https://discord.com/api/v10/channels/{}/messages?limit={}&after={}",
|
||||
"{DISCORD_API_BASE}/channels/{}/messages?limit={}&after={}",
|
||||
channel_id, PAGE_LIMIT, after
|
||||
);
|
||||
let response = match channel_host::http_request(
|
||||
@@ -986,7 +991,7 @@ fn handle_slash_command(interaction: &DiscordInteraction) -> bool {
|
||||
);
|
||||
// Attempt to notify user of internal error
|
||||
let url = format!(
|
||||
"https://discord.com/api/v10/webhooks/{}/{}",
|
||||
"{DISCORD_API_BASE}/webhooks/{}/{}",
|
||||
interaction.application_id, interaction.token
|
||||
);
|
||||
let payload = serde_json::json!({
|
||||
@@ -1106,7 +1111,7 @@ fn check_sender_permission(
|
||||
}
|
||||
|
||||
let dm_policy =
|
||||
channel_host::workspace_read(DM_POLICY_PATH).unwrap_or_else(|| default_dm_policy());
|
||||
channel_host::workspace_read(DM_POLICY_PATH).unwrap_or_else(default_dm_policy);
|
||||
if dm_policy == "open" {
|
||||
return true;
|
||||
}
|
||||
@@ -1161,7 +1166,7 @@ fn check_sender_permission(
|
||||
/// Send a pairing code as an ephemeral Discord followup message.
|
||||
fn send_pairing_reply(ctx: &PairingReplyCtx, code: &str) -> Result<(), String> {
|
||||
let url = format!(
|
||||
"https://discord.com/api/v10/webhooks/{}/{}",
|
||||
"{DISCORD_API_BASE}/webhooks/{}/{}",
|
||||
ctx.application_id, ctx.token
|
||||
);
|
||||
let payload = serde_json::json!({
|
||||
@@ -1194,6 +1199,57 @@ fn send_pairing_reply(ctx: &PairingReplyCtx, code: &str) -> Result<(), String> {
|
||||
}
|
||||
}
|
||||
|
||||
/// Send a broadcast message to a Discord user via DM.
|
||||
///
|
||||
/// Creates a DM channel with the user (Discord caches this, so repeated calls
|
||||
/// for the same user reuse the existing channel) and then posts the message.
|
||||
fn broadcast_dm(user_id: &str, content: &str) -> Result<(), String> {
|
||||
// Validate user_id is a plausible Discord snowflake (numeric, 17-20 digits)
|
||||
// to avoid injecting arbitrary strings into API URLs.
|
||||
if user_id.is_empty()
|
||||
|| !user_id.chars().all(|c| c.is_ascii_digit())
|
||||
|| user_id.len() < 17
|
||||
|| user_id.len() > 20
|
||||
{
|
||||
return Err(format!("Invalid Discord user ID: '{}'", user_id));
|
||||
}
|
||||
|
||||
// Step 1: Open (or reuse) a DM channel with the target user.
|
||||
let create_dm_payload = serde_json::json!({ "recipient_id": user_id });
|
||||
let create_dm_bytes = serde_json::to_vec(&create_dm_payload)
|
||||
.map_err(|e| format!("Failed to serialize DM channel request: {}", e))?;
|
||||
|
||||
let dm_response = channel_host::http_request(
|
||||
"POST",
|
||||
&format!("{DISCORD_API_BASE}/users/@me/channels"),
|
||||
&discord_auth_headers_json(true),
|
||||
Some(&create_dm_bytes),
|
||||
Some(10_000),
|
||||
)
|
||||
.map_err(|e| format!("Failed to create DM channel: {}", e))?;
|
||||
|
||||
if !(200..300).contains(&dm_response.status) {
|
||||
let body = String::from_utf8_lossy(&dm_response.body);
|
||||
return Err(format!(
|
||||
"Discord create-DM failed: {} - {}",
|
||||
dm_response.status, body
|
||||
));
|
||||
}
|
||||
|
||||
#[derive(Deserialize)]
|
||||
struct DmChannelResponse {
|
||||
id: String,
|
||||
}
|
||||
let dm_channel: DmChannelResponse = serde_json::from_slice(&dm_response.body)
|
||||
.map_err(|e| format!("Failed to parse DM channel response: {}", e))?;
|
||||
let channel_id = &dm_channel.id;
|
||||
|
||||
// Step 2: Send the message to the DM channel.
|
||||
let truncated = truncate_message(content);
|
||||
let payload = serde_json::json!({ "content": truncated });
|
||||
send_channel_message(channel_id, payload)
|
||||
}
|
||||
|
||||
fn json_response(status: u16, value: serde_json::Value) -> OutgoingHttpResponse {
|
||||
let body = serde_json::to_vec(&value).unwrap_or_default();
|
||||
let headers = serde_json::json!({"Content-Type": "application/json"});
|
||||
@@ -1593,4 +1649,43 @@ mod tests {
|
||||
assert_eq!(interaction.interaction_type, 2);
|
||||
assert!(interaction.data.is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_broadcast_dm_payload_format() {
|
||||
// Verify the DM channel creation payload is well-formed JSON that
|
||||
// Discord's API expects.
|
||||
let user_id = "123456789012345678";
|
||||
let payload = serde_json::json!({ "recipient_id": user_id });
|
||||
let serialized = serde_json::to_vec(&payload).unwrap();
|
||||
let parsed: serde_json::Value = serde_json::from_slice(&serialized).unwrap();
|
||||
assert_eq!(
|
||||
parsed.get("recipient_id").and_then(|v| v.as_str()),
|
||||
Some(user_id)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_broadcast_message_truncation() {
|
||||
// Broadcast uses truncate_message, verify it handles content within
|
||||
// Discord's 2000-char limit for DMs.
|
||||
let short = "Hello from broadcast";
|
||||
assert_eq!(truncate_message(short), short);
|
||||
|
||||
let long = "x".repeat(2500);
|
||||
let result = truncate_message(&long);
|
||||
assert!(result.len() <= 2006); // 1990 content + 16 suffix
|
||||
assert!(result.ends_with("\n... (truncated)"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_broadcast_dm_validates_snowflake() {
|
||||
// broadcast_dm rejects invalid Discord snowflake IDs before making
|
||||
// any API calls. We can call it directly since invalid IDs are
|
||||
// rejected before any host function is invoked.
|
||||
assert!(broadcast_dm("", "hi").is_err());
|
||||
assert!(broadcast_dm("abc", "hi").is_err());
|
||||
assert!(broadcast_dm("12345", "hi").is_err()); // too short
|
||||
assert!(broadcast_dm("123456789012345678901", "hi").is_err()); // too long
|
||||
assert!(broadcast_dm("12345678901234567x", "hi").is_err()); // non-digit
|
||||
}
|
||||
}
|
||||
|
||||
@@ -234,6 +234,7 @@ fn is_transient(err: &LlmError) -> bool {
|
||||
LlmError::RequestFailed { .. }
|
||||
| LlmError::RateLimited { .. }
|
||||
| LlmError::InvalidResponse { .. }
|
||||
| LlmError::EmptyResponse { .. }
|
||||
| LlmError::SessionExpired { .. }
|
||||
| LlmError::SessionRenewalFailed { .. }
|
||||
| LlmError::Http(_)
|
||||
|
||||
@@ -17,6 +17,9 @@ pub enum LlmError {
|
||||
#[error("Invalid response from {provider}: {reason}")]
|
||||
InvalidResponse { provider: String, reason: String },
|
||||
|
||||
#[error("Empty response from {provider}: no content returned")]
|
||||
EmptyResponse { provider: String },
|
||||
|
||||
#[error("Context length exceeded: {used} tokens used, {limit} allowed")]
|
||||
ContextLengthExceeded { used: usize, limit: usize },
|
||||
|
||||
|
||||
@@ -231,9 +231,8 @@ impl LlmProvider for GithubCopilotProvider {
|
||||
.choices
|
||||
.into_iter()
|
||||
.next()
|
||||
.ok_or_else(|| LlmError::InvalidResponse {
|
||||
.ok_or_else(|| LlmError::EmptyResponse {
|
||||
provider: "github_copilot".to_string(),
|
||||
reason: "No choices in response".to_string(),
|
||||
})?;
|
||||
|
||||
let (content, _tool_calls) = extract_choice_content(&choice);
|
||||
@@ -309,9 +308,8 @@ impl LlmProvider for GithubCopilotProvider {
|
||||
.choices
|
||||
.into_iter()
|
||||
.next()
|
||||
.ok_or_else(|| LlmError::InvalidResponse {
|
||||
.ok_or_else(|| LlmError::EmptyResponse {
|
||||
provider: "github_copilot".to_string(),
|
||||
reason: "No choices in response".to_string(),
|
||||
})?;
|
||||
|
||||
let (content, tool_calls) = extract_choice_content(&choice);
|
||||
|
||||
@@ -490,9 +490,8 @@ impl LlmProvider for NearAiChatProvider {
|
||||
.choices
|
||||
.into_iter()
|
||||
.next()
|
||||
.ok_or_else(|| LlmError::InvalidResponse {
|
||||
.ok_or_else(|| LlmError::EmptyResponse {
|
||||
provider: "nearai_chat".to_string(),
|
||||
reason: "No choices in response".to_string(),
|
||||
})?;
|
||||
|
||||
// Fall back to reasoning_content when content is null (same as
|
||||
@@ -570,9 +569,8 @@ impl LlmProvider for NearAiChatProvider {
|
||||
.choices
|
||||
.into_iter()
|
||||
.next()
|
||||
.ok_or_else(|| LlmError::InvalidResponse {
|
||||
.ok_or_else(|| LlmError::EmptyResponse {
|
||||
provider: "nearai_chat".to_string(),
|
||||
reason: "No choices in response".to_string(),
|
||||
})?;
|
||||
|
||||
let tool_calls: Vec<ToolCall> = choice
|
||||
|
||||
@@ -48,6 +48,7 @@ pub(crate) fn is_retryable(err: &LlmError) -> bool {
|
||||
LlmError::RequestFailed { .. }
|
||||
| LlmError::RateLimited { .. }
|
||||
| LlmError::InvalidResponse { .. }
|
||||
| LlmError::EmptyResponse { .. }
|
||||
| LlmError::SessionRenewalFailed { .. }
|
||||
| LlmError::Http(_)
|
||||
| LlmError::Io(_)
|
||||
|
||||
+144
-36
@@ -31,6 +31,7 @@ use std::time::Duration;
|
||||
|
||||
use serde::{Deserialize, Serialize};
|
||||
use tokio::io::{AsyncBufReadExt, BufReader};
|
||||
#[cfg(not(unix))]
|
||||
use tokio::process::Command;
|
||||
use uuid::Uuid;
|
||||
|
||||
@@ -340,6 +341,11 @@ impl ClaudeBridgeRuntime {
|
||||
|
||||
/// Spawn a `claude` CLI process and stream its output.
|
||||
///
|
||||
/// Uses a PTY on Unix so Node.js line-buffers stdout instead of
|
||||
/// full-buffering (which causes the bridge to hang on non-TTY pipes).
|
||||
/// Arguments are passed via `execve` (no shell) — injection-safe by
|
||||
/// construction.
|
||||
///
|
||||
/// Returns the session_id if captured from the `system` init message.
|
||||
async fn run_claude_session(
|
||||
&self,
|
||||
@@ -347,47 +353,102 @@ impl ClaudeBridgeRuntime {
|
||||
resume_session_id: Option<&str>,
|
||||
extra_env: &std::collections::HashMap<String, String>,
|
||||
) -> Result<Option<String>, WorkerError> {
|
||||
let mut cmd = Command::new("claude");
|
||||
cmd.arg("-p")
|
||||
.arg(prompt)
|
||||
.arg("--output-format")
|
||||
.arg("stream-json")
|
||||
.arg("--verbose")
|
||||
.arg("--max-turns")
|
||||
.arg(self.config.max_turns.to_string())
|
||||
.arg("--model")
|
||||
.arg(&self.config.model);
|
||||
let max_turns_str = self.config.max_turns.to_string();
|
||||
|
||||
if let Some(sid) = resume_session_id {
|
||||
cmd.arg("--resume").arg(sid);
|
||||
}
|
||||
|
||||
// Inject credentials into the child process environment without
|
||||
// mutating the global process env (which is unsafe in multi-threaded programs).
|
||||
cmd.envs(extra_env);
|
||||
|
||||
cmd.current_dir("/workspace")
|
||||
.stdout(std::process::Stdio::piped())
|
||||
.stderr(std::process::Stdio::piped());
|
||||
|
||||
let mut child = cmd.spawn().map_err(|e| WorkerError::ExecutionFailed {
|
||||
reason: format!("failed to spawn claude: {}", e),
|
||||
})?;
|
||||
|
||||
let stdout = child
|
||||
.stdout
|
||||
.take()
|
||||
.ok_or_else(|| WorkerError::ExecutionFailed {
|
||||
reason: "failed to capture claude stdout".to_string(),
|
||||
// Spawn with PTY on Unix to fix Node.js stdout buffering.
|
||||
// All arguments are passed individually via execve — never through
|
||||
// a shell interpreter. This eliminates shell injection by construction.
|
||||
#[cfg(unix)]
|
||||
let (mut child, stdout, stderr) = {
|
||||
let (pty, pts) = pty_process::open().map_err(|e| WorkerError::ExecutionFailed {
|
||||
reason: format!("failed to allocate PTY: {}", e),
|
||||
})?;
|
||||
|
||||
let stderr = child
|
||||
.stderr
|
||||
.take()
|
||||
.ok_or_else(|| WorkerError::ExecutionFailed {
|
||||
reason: "failed to capture claude stderr".to_string(),
|
||||
let mut cmd = pty_process::Command::new("claude");
|
||||
cmd = cmd
|
||||
.arg("-p")
|
||||
.arg(prompt)
|
||||
.arg("--output-format")
|
||||
.arg("stream-json")
|
||||
.arg("--verbose")
|
||||
.arg("--max-turns")
|
||||
.arg(&max_turns_str)
|
||||
.arg("--model")
|
||||
.arg(&self.config.model);
|
||||
|
||||
if let Some(sid) = resume_session_id {
|
||||
cmd = cmd.arg("--resume").arg(sid);
|
||||
}
|
||||
|
||||
cmd = cmd.envs(extra_env.iter());
|
||||
cmd = cmd.current_dir("/workspace");
|
||||
// Keep stderr on a separate pipe — pty-process attaches the PTY
|
||||
// to all fds by default, which would merge stderr into the PTY
|
||||
// stream and break NDJSON parsing.
|
||||
cmd = cmd.stderr(std::process::Stdio::piped());
|
||||
|
||||
let mut child = cmd.spawn(pts).map_err(|e| WorkerError::ExecutionFailed {
|
||||
reason: format!("failed to spawn claude with PTY: {}", e),
|
||||
})?;
|
||||
|
||||
let stderr = child
|
||||
.stderr
|
||||
.take()
|
||||
.ok_or_else(|| WorkerError::ExecutionFailed {
|
||||
reason: "failed to capture claude stderr".to_string(),
|
||||
})?;
|
||||
|
||||
// stdout comes from the PTY master, which implements AsyncRead
|
||||
let stdout: Box<dyn tokio::io::AsyncRead + Unpin + Send> = Box::new(pty);
|
||||
(child, stdout, stderr)
|
||||
};
|
||||
|
||||
// Non-Unix fallback (Windows CI) — no PTY, direct spawn.
|
||||
// Claude bridge only runs in Linux Docker containers, so this path
|
||||
// exists solely for compilation on Windows targets.
|
||||
#[cfg(not(unix))]
|
||||
let (mut child, stdout, stderr) = {
|
||||
let mut cmd = Command::new("claude");
|
||||
cmd.arg("-p")
|
||||
.arg(prompt)
|
||||
.arg("--output-format")
|
||||
.arg("stream-json")
|
||||
.arg("--verbose")
|
||||
.arg("--max-turns")
|
||||
.arg(&max_turns_str)
|
||||
.arg("--model")
|
||||
.arg(&self.config.model);
|
||||
|
||||
if let Some(sid) = resume_session_id {
|
||||
cmd.arg("--resume").arg(sid);
|
||||
}
|
||||
|
||||
cmd.envs(extra_env);
|
||||
cmd.current_dir("/workspace")
|
||||
.stdout(std::process::Stdio::piped())
|
||||
.stderr(std::process::Stdio::piped());
|
||||
|
||||
let mut child = cmd.spawn().map_err(|e| WorkerError::ExecutionFailed {
|
||||
reason: format!("failed to spawn claude: {}", e),
|
||||
})?;
|
||||
|
||||
let stdout_pipe = child
|
||||
.stdout
|
||||
.take()
|
||||
.ok_or_else(|| WorkerError::ExecutionFailed {
|
||||
reason: "failed to capture claude stdout".to_string(),
|
||||
})?;
|
||||
let stderr = child
|
||||
.stderr
|
||||
.take()
|
||||
.ok_or_else(|| WorkerError::ExecutionFailed {
|
||||
reason: "failed to capture claude stderr".to_string(),
|
||||
})?;
|
||||
|
||||
let stdout: Box<dyn tokio::io::AsyncRead + Unpin + Send> = Box::new(stdout_pipe);
|
||||
(child, stdout, stderr)
|
||||
};
|
||||
|
||||
// Spawn stderr reader that forwards lines as log events
|
||||
let client_for_stderr = Arc::clone(&self.client);
|
||||
let job_id = self.config.job_id;
|
||||
@@ -1027,4 +1088,51 @@ mod tests {
|
||||
let copied = copy_dir_recursive(nonexistent, dst.path()).unwrap();
|
||||
assert_eq!(copied, 0);
|
||||
}
|
||||
|
||||
/// Regression test: arguments are passed individually (not via shell string),
|
||||
/// so shell metacharacters in prompt/model/session_id are harmless.
|
||||
#[test]
|
||||
fn command_args_no_shell_interpretation() {
|
||||
// Prompt, model, and session_id may contain shell metacharacters from
|
||||
// user-supplied task descriptions or LLM output. Since we use
|
||||
// Command::arg() (execve), these are passed as literal strings.
|
||||
let prompt = "Fix the user's bug; echo $HOME && rm -rf /";
|
||||
let model = "claude-3-opus-20240229";
|
||||
let session_id = "'; DROP TABLE jobs; --";
|
||||
|
||||
let max_turns = 10u32;
|
||||
let max_turns_str = max_turns.to_string();
|
||||
let args: Vec<&str> = vec![
|
||||
"-p",
|
||||
prompt,
|
||||
"--output-format",
|
||||
"stream-json",
|
||||
"--verbose",
|
||||
"--max-turns",
|
||||
&max_turns_str,
|
||||
"--model",
|
||||
model,
|
||||
"--resume",
|
||||
session_id,
|
||||
];
|
||||
|
||||
// All values present as literal strings — no shell interpretation
|
||||
// ["-p", prompt, "--output-format", "stream-json", "--verbose",
|
||||
// "--max-turns", "10", "--model", model, "--resume", session_id]
|
||||
assert_eq!(args[1], prompt);
|
||||
assert_eq!(args[8], model);
|
||||
assert_eq!(args[10], session_id);
|
||||
// Shell metacharacters preserved, not expanded
|
||||
assert!(args[1].contains("$HOME"));
|
||||
assert!(args[1].contains("&&"));
|
||||
assert!(args[10].contains("'; DROP TABLE"));
|
||||
}
|
||||
|
||||
/// Verify PTY is available on Unix platforms.
|
||||
#[cfg(unix)]
|
||||
#[tokio::test]
|
||||
async fn pty_opens_successfully() {
|
||||
let result = pty_process::open();
|
||||
assert!(result.is_ok(), "PTY allocation should succeed on Unix");
|
||||
}
|
||||
}
|
||||
|
||||
+148
-4
@@ -391,6 +391,7 @@ Report when the job is complete or if you encounter issues you cannot resolve."#
|
||||
worker: self,
|
||||
rx: tokio::sync::Mutex::new(rx),
|
||||
consecutive_rate_limits: std::sync::atomic::AtomicUsize::new(0),
|
||||
has_text_response: std::sync::atomic::AtomicBool::new(false),
|
||||
};
|
||||
|
||||
let config = AgenticLoopConfig {
|
||||
@@ -1101,6 +1102,15 @@ fn store_fallback_in_metadata(
|
||||
}
|
||||
|
||||
/// Job delegate: implements `LoopDelegate` for the background job context.
|
||||
/// Whether an LLM error represents a completion-eligible empty response.
|
||||
///
|
||||
/// Only `EmptyResponse` (provider returned no choices/content) qualifies.
|
||||
/// Infrastructure errors (`AuthFailed`, `Http`, `Io`, etc.) never qualify —
|
||||
/// they must propagate even if prior text output was produced.
|
||||
fn is_completion_eligible_error(error: &crate::error::LlmError) -> bool {
|
||||
matches!(error, crate::error::LlmError::EmptyResponse { .. })
|
||||
}
|
||||
|
||||
///
|
||||
/// Handles: signal channel (stop/ping/user messages), cancellation checks,
|
||||
/// rate-limit retry, parallel tool execution, DB persistence, SSE broadcasting.
|
||||
@@ -1109,6 +1119,10 @@ struct JobDelegate<'a> {
|
||||
rx: tokio::sync::Mutex<&'a mut mpsc::Receiver<WorkerMessage>>,
|
||||
/// Tracks consecutive rate-limit errors to fail fast instead of burning iterations.
|
||||
consecutive_rate_limits: std::sync::atomic::AtomicUsize,
|
||||
/// Whether a substantive (non-empty) text response has been produced.
|
||||
/// When true, an empty follow-up response is treated as job completion
|
||||
/// rather than a retry signal (prevents spurious failures in routines).
|
||||
has_text_response: std::sync::atomic::AtomicBool,
|
||||
}
|
||||
|
||||
impl<'a> JobDelegate<'a> {
|
||||
@@ -1161,6 +1175,53 @@ impl<'a> JobDelegate<'a> {
|
||||
finish_reason: crate::llm::FinishReason::Stop,
|
||||
})
|
||||
}
|
||||
|
||||
/// Mark the job as completed, logging a warning on failure.
|
||||
async fn mark_completed_or_warn(&self, context: &str) {
|
||||
if let Err(e) = self.worker.mark_completed().await {
|
||||
tracing::warn!(
|
||||
job_id = %self.worker.job_id,
|
||||
error = %e,
|
||||
"Failed to mark job completed ({context})"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// If a substantive text response was already produced and the error
|
||||
/// indicates the LLM simply returned nothing, treat it as successful
|
||||
/// completion rather than a fatal failure.
|
||||
///
|
||||
/// Only swallows `EmptyResponse` — infrastructure errors (`AuthFailed`,
|
||||
/// `ContextLengthExceeded`, `Http`, `Io`, etc.) always propagate.
|
||||
///
|
||||
/// Returns `Some(empty RespondOutput)` when the error should be swallowed,
|
||||
/// `None` when it should propagate normally.
|
||||
async fn try_complete_on_error(
|
||||
&self,
|
||||
context: &str,
|
||||
error: &crate::error::LlmError,
|
||||
) -> Option<crate::llm::RespondOutput> {
|
||||
if !is_completion_eligible_error(error) {
|
||||
return None;
|
||||
}
|
||||
if !self
|
||||
.has_text_response
|
||||
.load(std::sync::atomic::Ordering::Relaxed)
|
||||
{
|
||||
return None;
|
||||
}
|
||||
tracing::info!(
|
||||
job_id = %self.worker.job_id,
|
||||
error = %error,
|
||||
"{context} empty response after text output — treating as completion"
|
||||
);
|
||||
self.mark_completed_or_warn(context).await;
|
||||
Some(crate::llm::RespondOutput {
|
||||
result: RespondResult::Text(String::new()),
|
||||
usage: crate::llm::TokenUsage::default(),
|
||||
finish_reason: crate::llm::FinishReason::Stop,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[async_trait]
|
||||
@@ -1291,7 +1352,12 @@ impl<'a> LoopDelegate for JobDelegate<'a> {
|
||||
Err(crate::error::LlmError::RateLimited { retry_after, .. }) => {
|
||||
return self.handle_rate_limit(retry_after, "tool selection").await;
|
||||
}
|
||||
Err(e) => return Err(e.into()),
|
||||
Err(e) => {
|
||||
if let Some(output) = self.try_complete_on_error("select_tools", &e).await {
|
||||
return Ok(output);
|
||||
}
|
||||
return Err(e.into());
|
||||
}
|
||||
};
|
||||
|
||||
// Fall back to respond_with_tools
|
||||
@@ -1321,7 +1387,12 @@ impl<'a> LoopDelegate for JobDelegate<'a> {
|
||||
self.handle_rate_limit(retry_after, "respond_with_tools")
|
||||
.await
|
||||
}
|
||||
Err(e) => Err(e.into()),
|
||||
Err(e) => {
|
||||
if let Some(output) = self.try_complete_on_error("respond_with_tools", &e).await {
|
||||
return Ok(output);
|
||||
}
|
||||
Err(e.into())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1330,9 +1401,22 @@ impl<'a> LoopDelegate for JobDelegate<'a> {
|
||||
text: &str,
|
||||
reason_ctx: &mut ReasoningContext,
|
||||
) -> TextAction {
|
||||
// Empty text from rate-limit backoff retry — skip processing and let the
|
||||
// loop proceed to the next iteration which will re-call the LLM.
|
||||
// Empty text after a substantive response means the LLM has finished.
|
||||
// Treat as successful completion rather than continuing the loop (which
|
||||
// would produce "Response contained no message or tool call (empty)").
|
||||
if text.is_empty() {
|
||||
if self
|
||||
.has_text_response
|
||||
.load(std::sync::atomic::Ordering::Relaxed)
|
||||
{
|
||||
tracing::debug!(
|
||||
job_id = %self.worker.job_id,
|
||||
"Empty response after text output — treating as completion"
|
||||
);
|
||||
self.mark_completed_or_warn("empty text response").await;
|
||||
return TextAction::Return(LoopOutcome::Response(String::new()));
|
||||
}
|
||||
// No prior text response — this is likely a rate-limit backoff retry.
|
||||
return TextAction::Continue;
|
||||
}
|
||||
|
||||
@@ -1348,6 +1432,10 @@ impl<'a> LoopDelegate for JobDelegate<'a> {
|
||||
return TextAction::Return(LoopOutcome::Response(text.to_string()));
|
||||
}
|
||||
|
||||
// Track that a substantive response has been produced.
|
||||
self.has_text_response
|
||||
.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
|
||||
// Add assistant response to context
|
||||
reason_ctx.messages.push(ChatMessage::assistant(text));
|
||||
|
||||
@@ -2285,4 +2373,60 @@ mod tests {
|
||||
assert_eq!(telegram[0].0, "owner-scope");
|
||||
assert_eq!(telegram[0].1.content, "hello from routine");
|
||||
}
|
||||
|
||||
/// Regression test: only `EmptyResponse` errors are eligible for
|
||||
/// completion-swallowing. Infrastructure errors must always propagate.
|
||||
#[test]
|
||||
fn is_completion_eligible_only_matches_empty_response() {
|
||||
use crate::error::LlmError;
|
||||
|
||||
// EmptyResponse is eligible
|
||||
assert!(super::is_completion_eligible_error(
|
||||
&LlmError::EmptyResponse {
|
||||
provider: "test".to_string(),
|
||||
}
|
||||
));
|
||||
|
||||
// All other variants are NOT eligible
|
||||
assert!(!super::is_completion_eligible_error(
|
||||
&LlmError::InvalidResponse {
|
||||
provider: "test".to_string(),
|
||||
reason: "parse error".to_string(),
|
||||
}
|
||||
));
|
||||
assert!(!super::is_completion_eligible_error(
|
||||
&LlmError::AuthFailed {
|
||||
provider: "test".to_string(),
|
||||
}
|
||||
));
|
||||
assert!(!super::is_completion_eligible_error(
|
||||
&LlmError::ContextLengthExceeded {
|
||||
used: 100_000,
|
||||
limit: 50_000,
|
||||
}
|
||||
));
|
||||
assert!(!super::is_completion_eligible_error(
|
||||
&LlmError::ModelNotAvailable {
|
||||
provider: "test".to_string(),
|
||||
model: "gpt-4".to_string(),
|
||||
}
|
||||
));
|
||||
assert!(!super::is_completion_eligible_error(
|
||||
&LlmError::RequestFailed {
|
||||
provider: "test".to_string(),
|
||||
reason: "timeout".to_string(),
|
||||
}
|
||||
));
|
||||
assert!(!super::is_completion_eligible_error(
|
||||
&LlmError::SessionExpired {
|
||||
provider: "test".to_string(),
|
||||
}
|
||||
));
|
||||
assert!(!super::is_completion_eligible_error(
|
||||
&LlmError::SessionRenewalFailed {
|
||||
provider: "test".to_string(),
|
||||
reason: "timeout".to_string(),
|
||||
}
|
||||
));
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user