Compare commits

..
Author SHA1 Message Date
Claude f61a0b747f fix(clippy): use any() instead of find().is_none() in user stats test
https://claude.ai/code/session_01Nm95eCjdrwxDjwkHTRieZs
2026-03-28 19:21:15 +00:00
ZakiandClaude 65c374f53c fix(routines): broaden strip_html_tags to cover all HTML forms
The whitelist-based regex missed self-closing tags without whitespace
(<br/>, <img/>), HTML comments (<!--...-->), SVG/MathML tags, and
custom elements (<custom-element>). This weakened the HTML stripping
guarantee for untrusted routine/job summaries in notifications.

Changes:
- Add separate regex for HTML comments (<!--...-->)
- Add SVG tags (svg, path, circle, etc.) and MathML tags (math, mrow,
  etc.) to the known tag list
- Add regex for custom elements (tags containing hyphens per web
  components spec)
- Fix self-closing tag matching to handle <br/> without whitespace
  by making the whitespace before /> optional
- Add regression tests for all four cases plus generics preservation
- Fix pre-existing compilation error in tunnel/mod.rs test helpers
  (missing GatewayConfig fields)

Co-Authored-By: Claude Opus 4.6 (1M context) <[email protected]>
2026-03-28 19:16:32 +00:00
Claude faf3012a36 fix: remove .expect() from strip_html_tags to pass no-panics check
Use Option<Regex> with graceful fallback instead of panicking on
regex compilation failure.

https://claude.ai/code/session_01CsP5wMZ2evEMHgGghjAfR1
2026-03-28 19:16:03 +00:00
ZakiandClaude ce193ff2d2 fix(routines): set run.job_id in-memory and revert Cargo.lock drift
Address PR #1470 review feedback:

- Set run.job_id = Some(job_id) after link_routine_run_to_job succeeds
  so send_notification reads the correct value instead of always None.
- Revert Cargo.lock to staging baseline: the PR had accumulated
  unrelated dependency changes (openssl, native-tls, crossterm 0.28.1
  downgrade, foreign-types, vcpkg) from a dirty lockfile resolution.

[skip-regression-check]

Co-Authored-By: Claude Opus 4.6 (1M context) <[email protected]>
2026-03-28 19:16:03 +00:00
ZakiandClaude 7881998d93 fix: preserve non-HTML angle brackets in sanitize_summary
The previous strip_html_tags() implementation blindly removed all content
between angle brackets, mangling legitimate text like Vec<String>, shell
redirects (cat < input.txt), and comparison operators in LLM/error output.

Replace the naive char-by-char scanner with a regex that only matches
known HTML tag names (div, script, img, a, b, etc.), preserving generic
angle-bracket content that appears in code snippets and error messages.

Co-Authored-By: Claude Opus 4.6 (1M context) <[email protected]>
2026-03-28 19:16:03 +00:00
Claude d967e620cc fix: resolve mutability mismatch in execute_full_job call after rebase
[skip-regression-check]

https://claude.ai/code/session_01ABGWibdKVQ3b6pEKtxPPkM
2026-03-28 19:16:03 +00:00
ZakiandClaude 7b884c4a24 fix(routines): propagate job_id to notification metadata (#1321)
Update run.job_id after execute_full_job() links the routine run to the
job, ensuring send_notification receives the actual job ID instead of None
on the normal completion path.

Co-Authored-By: Claude Opus 4.6 (1M context) <[email protected]>
2026-03-28 19:16:03 +00:00
Claude d45d5977a0 fix(deps): resolve RUSTSEC-2026-0049 rustls-webpki CRL advisory
Update rustls-webpki 0.103.9 -> 0.103.10 and ignore the advisory for
0.102.8 which is pinned by libsql's rustls 0.22.4 dependency chain.

https://claude.ai/code/session_01MmxvBgAMn4m45pZguFKBEX
2026-03-28 19:16:03 +00:00
Claude d29811db6b style: run cargo fmt to fix formatting
https://claude.ai/code/session_01Nv2TJ3so5WQqpRUcirAhT3
2026-03-28 19:16:03 +00:00
ZakiandClaude 25dcbf78b8 fix(routines): normalize notification summaries with truncation and metadata (#1321)
- Capitalize status labels in notifications (ok -> Completed, attention -> Needs attention)
- Sanitize and truncate long summaries to 500 chars with UTF-8-safe ellipsis
- Include job_id in notification metadata for full-job routines
- Move sanitize_summary/strip_html_tags out of #[cfg(test)] for production use
- Add regression tests for truncation, status labels, job_id metadata

Co-Authored-By: Claude Opus 4.6 (1M context) <[email protected]>
2026-03-28 19:16:03 +00:00
36 changed files with 560 additions and 1020 deletions
Generated
+125 -6
View File
@@ -1510,7 +1510,7 @@ version = "1.1.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "980c2afde4af43d6a05c5be738f9eae595cff86dce1f38f88b95058a98c027f3"
dependencies = [
"crossterm",
"crossterm 0.29.0",
]
[[package]]
@@ -1731,7 +1731,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "04a63daf06a168535c74ab97cdba3ed4fa5d4f32cb36e437dcceb83d66854b7c"
dependencies = [
"crokey-proc_macros",
"crossterm",
"crossterm 0.29.0",
"once_cell",
"serde",
"strict",
@@ -1743,7 +1743,7 @@ version = "1.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "847f11a14855fc490bd5d059821895c53e77eeb3c2b73ee3dded7ce77c93b231"
dependencies = [
"crossterm",
"crossterm 0.29.0",
"proc-macro2",
"quote",
"strict",
@@ -1817,6 +1817,22 @@ version = "0.8.21"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d0a5c400df2834b80a4c3327b3aad3a4c4cd4de0629063962b03235697506a28"
[[package]]
name = "crossterm"
version = "0.28.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "829d955a0bb380ef178a640b91779e3987da38c9aea133b20614cfed8cdea9c6"
dependencies = [
"bitflags 2.11.0",
"crossterm_winapi",
"mio",
"parking_lot",
"rustix 0.38.44",
"signal-hook",
"signal-hook-mio",
"winapi",
]
[[package]]
name = "crossterm"
version = "0.29.0"
@@ -2476,6 +2492,21 @@ version = "0.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "77ce24cb58228fbb8aa041425bb1050850ac19177686ea6e0f41a70416f56fdb"
[[package]]
name = "foreign-types"
version = "0.3.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f6f339eb8adc052cd2ca78910fda869aefa38d22d5cb648e6485e4d3fc06f3b1"
dependencies = [
"foreign-types-shared",
]
[[package]]
name = "foreign-types-shared"
version = "0.1.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "00b0228411908ca8685dba7fc2cdd70ec9990a6e753e89b6ac91a84c40fbaf4b"
[[package]]
name = "form_urlencoded"
version = "1.2.2"
@@ -3118,7 +3149,6 @@ dependencies = [
"tokio",
"tokio-rustls 0.26.4",
"tower-service",
"webpki-roots 1.0.6",
]
[[package]]
@@ -3133,6 +3163,22 @@ dependencies = [
"tokio-io-timeout",
]
[[package]]
name = "hyper-tls"
version = "0.6.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "70206fc6890eaca9fde8a0bf71caa2ddfc9fe045ac9e5c70df101a7dbde866e0"
dependencies = [
"bytes",
"http-body-util",
"hyper 1.8.1",
"hyper-util",
"native-tls",
"tokio",
"tokio-native-tls",
"tower-service",
]
[[package]]
name = "hyper-util"
version = "0.1.20"
@@ -3410,7 +3456,7 @@ dependencies = [
"clap_complete",
"criterion",
"cron",
"crossterm",
"crossterm 0.28.1",
"deadpool-postgres",
"dirs 6.0.0",
"dotenvy",
@@ -4089,6 +4135,23 @@ dependencies = [
"rand 0.8.5",
]
[[package]]
name = "native-tls"
version = "0.2.18"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "465500e14ea162429d264d44189adc38b199b62b1c21eea9f69e4b73cb03bbf2"
dependencies = [
"libc",
"log",
"openssl",
"openssl-probe 0.2.1",
"openssl-sys",
"schannel",
"security-framework 3.7.0",
"security-framework-sys",
"tempfile",
]
[[package]]
name = "new_debug_unreachable"
version = "1.0.6"
@@ -4311,6 +4374,32 @@ dependencies = [
"pathdiff",
]
[[package]]
name = "openssl"
version = "0.10.76"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "951c002c75e16ea2c65b8c7e4d3d51d5530d8dfa7d060b4776828c88cfb18ecf"
dependencies = [
"bitflags 2.11.0",
"cfg-if",
"foreign-types",
"libc",
"once_cell",
"openssl-macros",
"openssl-sys",
]
[[package]]
name = "openssl-macros"
version = "0.1.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a948666b637a0f465e8564c73e89d4dde00d72d4d473cc972f390fc3dcee7d9c"
dependencies = [
"proc-macro2",
"quote",
"syn 2.0.117",
]
[[package]]
name = "openssl-probe"
version = "0.1.6"
@@ -4323,6 +4412,18 @@ version = "0.2.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "7c87def4c32ab89d880effc9e097653c8da5d6ef28e6b539d313baaacfbafcbe"
[[package]]
name = "openssl-sys"
version = "0.9.112"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "57d55af3b3e226502be1526dfdba67ab0e9c96fc293004e79576b2b9edb0dbdb"
dependencies = [
"cc",
"libc",
"pkg-config",
"vcpkg",
]
[[package]]
name = "option-ext"
version = "0.2.0"
@@ -5312,11 +5413,13 @@ dependencies = [
"http-body-util",
"hyper 1.8.1",
"hyper-rustls 0.27.7",
"hyper-tls",
"hyper-util",
"js-sys",
"log",
"mime",
"mime_guess",
"native-tls",
"percent-encoding",
"pin-project-lite",
"quinn",
@@ -5328,6 +5431,7 @@ dependencies = [
"serde_urlencoded",
"sync_wrapper 1.0.2",
"tokio",
"tokio-native-tls",
"tokio-rustls 0.26.4",
"tokio-util",
"tower 0.5.3",
@@ -5338,7 +5442,6 @@ dependencies = [
"wasm-bindgen-futures",
"wasm-streams",
"web-sys",
"webpki-roots 1.0.6",
]
[[package]]
@@ -6671,6 +6774,16 @@ dependencies = [
"syn 2.0.117",
]
[[package]]
name = "tokio-native-tls"
version = "0.3.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bbae76ab933c85776efabc971569dd6119c580d8f5d448769dec1764bf796ef2"
dependencies = [
"native-tls",
"tokio",
]
[[package]]
name = "tokio-postgres"
version = "0.7.16"
@@ -7354,6 +7467,12 @@ version = "0.1.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ba73ea9cf16a25df0c8caa16c51acb937d5712a8429db78a3ee29d5dcacd3a65"
[[package]]
name = "vcpkg"
version = "0.2.15"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "accd4ea62f7bb7a82fe23066fb0957d48ef677f6eeb8215f372f52e48bb32426"
[[package]]
name = "version_check"
version = "0.9.5"
+2 -1
View File
@@ -252,7 +252,8 @@ strip = true # Remove debug symbols from release binaries
# The profile that 'cargo dist' will build with
[profile.dist]
inherits = "release"
lto = "thin"
lto = "fat" # Full cross-crate LTO (slow build, better codegen)
codegen-units = 1 # Single codegen unit for maximum optimization
# Config for 'dist'
[workspace.metadata.dist]
-20
View File
@@ -1269,14 +1269,6 @@ pub(crate) fn extract_suggestions(text: &str) -> (String, Vec<String>) {
(cleaned, suggestions)
}
/// Remove `<suggestions>` tags from a response, returning only the cleaned text.
///
/// Convenience wrapper around [`extract_suggestions`] for callers that don't
/// need the parsed suggestion list (e.g. job worker, plan completion check).
pub(crate) fn strip_suggestions(text: &str) -> String {
extract_suggestions(text).0
}
#[cfg(test)]
mod tests {
use std::sync::Arc;
@@ -2547,18 +2539,6 @@ mod tests {
assert_eq!(suggestions, vec!["ok"]); // safety: test
}
#[test]
fn test_strip_suggestions_removes_tags() {
let input = "The job is complete.\n<suggestions>[\"Check logs\"]</suggestions>";
assert_eq!(super::strip_suggestions(input), "The job is complete."); // safety: test
}
#[test]
fn test_strip_suggestions_no_tag_passthrough() {
let input = "Plain text without tags.";
assert_eq!(super::strip_suggestions(input), input); // safety: test
}
#[test]
fn test_tool_error_format_includes_tool_name() {
let tool_name = "http";
-1
View File
@@ -36,7 +36,6 @@ pub(crate) use agent_loop::truncate_for_preview;
pub use agent_loop::{Agent, AgentDeps};
pub use compaction::{CompactionResult, ContextCompactor};
pub use context_monitor::{CompactionStrategy, ContextBreakdown, ContextMonitor};
pub(crate) use dispatcher::strip_suggestions;
pub use heartbeat::{
HeartbeatConfig, HeartbeatResult, HeartbeatRunner, spawn_heartbeat, spawn_multi_user_heartbeat,
};
+1 -6
View File
@@ -265,13 +265,8 @@ fn default_max_tokens() -> u32 {
4096
}
/// Default max agentic loop iterations for full_job routines.
///
/// Raised from 10 to 25 to accommodate multi-step tool chains that
/// stalled at the old cap. Worst-case LLM cost is 2.5x higher per run;
/// callers needing tighter budgets should set `max_iterations` explicitly.
fn default_max_iterations() -> u32 {
25
10
}
fn default_max_tool_rounds() -> u32 {
+327 -38
View File
@@ -715,6 +715,7 @@ impl RoutineEngine {
status,
Some(summary),
thread_id.as_deref(),
run.job_id,
)
.await;
@@ -1085,7 +1086,7 @@ struct EngineContext {
}
/// Execute a routine run. Handles both lightweight and full_job modes.
async fn execute_routine(ctx: EngineContext, routine: Routine, run: RoutineRun) {
async fn execute_routine(ctx: EngineContext, routine: Routine, mut run: RoutineRun) {
// Increment running count (atomic: survives panics in the execution below)
ctx.running_count.fetch_add(1, Ordering::Relaxed);
@@ -1118,7 +1119,7 @@ async fn execute_routine(ctx: EngineContext, routine: Routine, run: RoutineRun)
description,
max_iterations: *max_iterations,
};
execute_full_job(&ctx, &routine, &run, &execution).await
execute_full_job(&ctx, &routine, &mut run, &execution).await
}
};
@@ -1218,6 +1219,7 @@ async fn execute_routine(ctx: EngineContext, routine: Routine, run: RoutineRun)
status,
summary.as_deref(),
thread_id.as_deref(),
run.job_id,
)
.await;
}
@@ -1253,7 +1255,7 @@ struct FullJobExecutionConfig<'a> {
async fn execute_full_job(
ctx: &EngineContext,
routine: &Routine,
run: &RoutineRun,
run: &mut RoutineRun,
execution: &FullJobExecutionConfig<'_>,
) -> Result<(RunStatus, Option<String>, Option<i32>), RoutineError> {
match ctx.sandbox_readiness {
@@ -1292,23 +1294,11 @@ async fn execute_full_job(
}
metadata["notify_user"] = serde_json::json!(&routine.notify.user);
// Prepend execution context so the LLM knows it's already inside a
// routine and should execute the task directly — not set up infrastructure.
let contextualized_description = format!(
"IMPORTANT: You are executing inside routine \"{routine_name}\". \
The routine and its schedule are already configured. \
Tools and credentials are already set up. \
Do NOT create routines, jobs, or try to discover/install/authenticate tools. \
Execute the task directly.\n\n{desc}",
routine_name = routine.name,
desc = execution.description,
);
let job_id = scheduler
.dispatch_job(
&routine.user_id,
execution.title,
&contextualized_description,
execution.description,
Some(metadata),
)
.await
@@ -1327,6 +1317,9 @@ async fn execute_full_job(
reason: format!("failed to link run to job: {e}"),
})?;
// Keep the in-memory struct in sync so send_notification can read run.job_id.
run.job_id = Some(job_id);
tracing::info!(
routine = %routine.name,
job_id = %job_id,
@@ -1839,7 +1832,18 @@ async fn execute_routine_tool(
Ok(result_str)
}
/// Human-readable label for a run status, suitable for user-facing notifications.
fn status_display_label(status: RunStatus) -> &'static str {
match status {
RunStatus::Ok => "Completed",
RunStatus::Attention => "Needs attention",
RunStatus::Failed => "Failed",
RunStatus::Running => "Running",
}
}
/// Send a notification based on the routine's notify config and run status.
#[allow(clippy::too_many_arguments)]
async fn send_notification(
tx: &mpsc::Sender<OutgoingResponse>,
notify: &NotifyConfig,
@@ -1848,6 +1852,7 @@ async fn send_notification(
status: RunStatus,
summary: Option<&str>,
thread_id: Option<&str>,
job_id: Option<Uuid>,
) {
let should_notify = match status {
RunStatus::Ok => notify.on_success,
@@ -1867,23 +1872,37 @@ async fn send_notification(
RunStatus::Running => "",
};
let label = status_display_label(status);
let message = match summary {
Some(s) => format!("{} *Routine '{}'*: {}\n\n{}", icon, routine_name, status, s),
None => format!("{} *Routine '{}'*: {}", icon, routine_name, status),
Some(s) => {
let sanitized = sanitize_summary(s);
format!(
"{} *Routine '{}'*: {}\n\n{}",
icon, routine_name, label, sanitized
)
}
None => format!("{} *Routine '{}'*: {}", icon, routine_name, label),
};
let mut metadata = serde_json::json!({
"source": "routine",
"routine_name": routine_name,
"status": status.to_string(),
"owner_id": owner_id,
"notify_user": notify.user,
"notify_channel": notify.channel,
});
if let Some(jid) = job_id {
metadata["job_id"] = serde_json::json!(jid.to_string());
}
let response = OutgoingResponse {
content: message,
thread_id: thread_id.map(String::from),
attachments: Vec::new(),
metadata: serde_json::json!({
"source": "routine",
"routine_name": routine_name,
"status": status.to_string(),
"owner_id": owner_id,
"notify_user": notify.user,
"notify_channel": notify.channel,
}),
metadata,
};
if let Err(e) = tx.send(response).await {
@@ -1946,7 +1965,6 @@ fn truncate(s: &str, max: usize) -> String {
/// 2. Strip HTML tags to prevent injection in web-rendered notifications
/// 3. Collapse multiple whitespace/newlines to single spaces for cleaner output
/// 4. Truncate to 500 chars to prevent oversized notifications
#[cfg(test)]
fn sanitize_summary(s: &str) -> String {
// Strip control characters (keep newline for now, collapse later)
let no_control: String = s
@@ -1973,19 +1991,59 @@ fn sanitize_summary(s: &str) -> String {
}
}
/// Remove HTML/XML tags from a string.
#[cfg(test)]
/// Remove actual HTML tags from a string while preserving non-HTML angle brackets.
///
/// Only strips patterns that look like real HTML/XML tags (e.g. `<div>`, `</p>`,
/// `<img src=...>`), not generic angle-bracket content like `Vec<String>`,
/// `cat < input.txt`, or comparison operators.
///
/// Also strips HTML comments (`<!--...-->`), SVG/MathML tags, and custom elements
/// (tags containing hyphens like `<custom-element>`).
fn strip_html_tags(s: &str) -> String {
let mut result = String::with_capacity(s.len());
let mut in_tag = false;
for c in s.chars() {
match c {
'<' => in_tag = true,
'>' if in_tag => in_tag = false,
_ if !in_tag => result.push(c),
_ => {}
}
use std::sync::LazyLock;
// HTML comment pattern: <!--...-->
static COMMENT_RE: LazyLock<Option<Regex>> =
LazyLock::new(|| Regex::new(r"<!--[\s\S]*?-->").ok());
// Known HTML/SVG/MathML tag names. Includes SVG tags (svg, path, circle, etc.)
// and MathML tags (math, mrow, etc.) that can carry event handlers.
static HTML_TAG_RE: LazyLock<Option<Regex>> = LazyLock::new(|| {
let tags = "a|abbr|address|area|article|aside|audio|b|base|bdi|bdo|blockquote|\
body|br|button|canvas|caption|cite|code|col|colgroup|data|datalist|dd|del|\
details|dfn|dialog|div|dl|dt|em|embed|fieldset|figcaption|figure|footer|\
form|h[1-6]|head|header|hgroup|hr|html|i|iframe|img|input|ins|kbd|label|\
legend|li|link|main|map|mark|meta|meter|nav|noscript|object|ol|optgroup|\
option|output|p|param|picture|pre|progress|q|rp|rt|ruby|s|samp|script|\
section|select|slot|small|source|span|strong|style|sub|summary|sup|table|\
tbody|td|template|textarea|tfoot|th|thead|time|title|tr|track|u|ul|var|\
video|wbr|\
svg|g|path|circle|ellipse|line|polyline|polygon|rect|text|tspan|defs|\
clippath|mask|pattern|image|use|symbol|marker|lineargradient|\
radialgradient|stop|filter|foreignobject|animate|animatetransform|\
math|mrow|mi|mo|mn|ms|mtext|mfrac|msqrt|mroot|msub|msup|msubsup|\
munder|mover|munderover|mtable|mtr|mtd|mspace|mpadded|mfenced|menclose";
// Handles: <tag>, </tag>, <tag/>, <tag />, <tag attr="val">, <tag attr="val"/>
Regex::new(&format!(r"(?i)</?(?:{})(?:\s[^>]*)?\s*/?>", tags)).ok()
});
// Custom elements: tags containing a hyphen (web components spec requires it).
// E.g. <custom-element>, <my-widget foo="bar">, </x-foo>
static CUSTOM_ELEMENT_RE: LazyLock<Option<Regex>> =
LazyLock::new(|| Regex::new(r"(?i)</?\w+-[\w-]*(?:\s[^>]*)?\s*/?>").ok());
let mut result = s.to_string();
if let Some(re) = COMMENT_RE.as_ref() {
result = re.replace_all(&result, "").into_owned();
}
if let Some(re) = HTML_TAG_RE.as_ref() {
result = re.replace_all(&result, "").into_owned();
}
if let Some(re) = CUSTOM_ELEMENT_RE.as_ref() {
result = re.replace_all(&result, "").into_owned();
}
result
}
@@ -2560,6 +2618,33 @@ mod tests {
assert_eq!(sanitize_summary("<img src=x onerror=alert(1)>"), "");
}
#[test]
fn test_sanitize_summary_preserves_non_html_angle_brackets() {
use super::sanitize_summary;
// Rust/Java generics must pass through unchanged
assert_eq!(
sanitize_summary("expected Vec<String>"),
"expected Vec<String>"
);
assert_eq!(
sanitize_summary("HashMap<String, Vec<u8>>"),
"HashMap<String, Vec<u8>>"
);
// Shell redirects must pass through unchanged
assert_eq!(sanitize_summary("cat < input.txt"), "cat < input.txt");
// Comparison operators must pass through unchanged
assert_eq!(sanitize_summary("x < 10 && y > 20"), "x < 10 && y > 20");
// Mixed: real HTML stripped but generics preserved
assert_eq!(
sanitize_summary("Error in Vec<String>: <b>failed</b>"),
"Error in Vec<String>: failed"
);
}
#[test]
fn test_sanitize_summary_multibyte_truncation() {
use super::sanitize_summary;
@@ -2570,4 +2655,208 @@ mod tests {
assert!(result.len() <= 503);
assert!(result.ends_with("..."));
}
#[test]
fn test_sanitize_summary_truncates_long_text() {
use super::sanitize_summary;
let short = "This is a short summary.";
assert_eq!(sanitize_summary(short), short);
let long = "x".repeat(600);
let result = sanitize_summary(&long);
assert!(
result.len() <= 503,
"Truncated summary should be at most 503 bytes (500 + '...')"
);
assert!(
result.ends_with("..."),
"Truncated summary should end with ellipsis"
);
}
#[test]
fn test_sanitize_summary_strips_all_html_forms() {
use super::sanitize_summary;
// Self-closing tags without whitespace: <br/>, <img/>
assert_eq!(sanitize_summary("line1<br/>line2"), "line1line2");
assert_eq!(sanitize_summary("text<img/>more"), "textmore");
assert_eq!(sanitize_summary("text<br />more"), "textmore");
// HTML comments
assert_eq!(sanitize_summary("before<!--x-->after"), "beforeafter");
assert_eq!(sanitize_summary("a<!-- multi\nline -->b"), "ab");
// SVG tags (can carry event handlers)
assert_eq!(
sanitize_summary("<svg onload=alert(1)>payload</svg>"),
"payload"
);
assert_eq!(sanitize_summary("<svg><circle r=10/></svg>"), "");
// MathML tags
assert_eq!(sanitize_summary("<math><mrow>x</mrow></math>"), "x");
// Custom elements (web components with hyphens)
assert_eq!(
sanitize_summary("before<custom-element>inner</custom-element>after"),
"beforeinnerafter"
);
assert_eq!(
sanitize_summary("<my-widget foo=\"bar\">content</my-widget>"),
"content"
);
// Generics must still be preserved
assert_eq!(
sanitize_summary("expected Vec<String>"),
"expected Vec<String>"
);
}
#[test]
fn test_status_display_label_readable() {
use super::status_display_label;
assert_eq!(status_display_label(RunStatus::Ok), "Completed");
assert_eq!(status_display_label(RunStatus::Failed), "Failed");
assert_eq!(
status_display_label(RunStatus::Attention),
"Needs attention"
);
assert_eq!(status_display_label(RunStatus::Running), "Running");
}
#[tokio::test]
async fn test_notification_message_uses_readable_status() {
use tokio::sync::mpsc;
let (tx, mut rx) = mpsc::channel(1);
let notify = NotifyConfig {
on_success: true,
on_failure: true,
on_attention: true,
..Default::default()
};
super::send_notification(
&tx,
&notify,
"user-1",
"my-routine",
RunStatus::Ok,
Some("All good"),
None,
None,
)
.await;
let msg = rx.recv().await.expect("should receive notification");
assert!(
msg.content.contains("Completed"),
"Notification should use readable label 'Completed', got: {}",
msg.content
);
assert!(
!msg.content.contains(": ok"),
"Notification should not contain raw lowercase status"
);
}
#[tokio::test]
async fn test_notification_includes_job_id_in_metadata() {
use tokio::sync::mpsc;
let (tx, mut rx) = mpsc::channel(1);
let notify = NotifyConfig {
on_failure: true,
..Default::default()
};
let job_id = uuid::Uuid::new_v4();
super::send_notification(
&tx,
&notify,
"user-1",
"my-routine",
RunStatus::Failed,
Some("something broke"),
None,
Some(job_id),
)
.await;
let msg = rx.recv().await.expect("should receive notification");
let meta_job_id = msg.metadata["job_id"]
.as_str()
.expect("metadata should contain job_id");
assert_eq!(meta_job_id, job_id.to_string());
}
#[tokio::test]
async fn test_notification_omits_job_id_when_none() {
use tokio::sync::mpsc;
let (tx, mut rx) = mpsc::channel(1);
let notify = NotifyConfig {
on_success: true,
..Default::default()
};
super::send_notification(
&tx,
&notify,
"user-1",
"my-routine",
RunStatus::Ok,
Some("done"),
None,
None,
)
.await;
let msg = rx.recv().await.expect("should receive notification");
assert!(
msg.metadata.get("job_id").is_none(),
"metadata should not contain job_id when None"
);
}
#[tokio::test]
async fn test_notification_truncates_long_summary() {
use tokio::sync::mpsc;
let (tx, mut rx) = mpsc::channel(1);
let notify = NotifyConfig {
on_failure: true,
..Default::default()
};
let long_summary = "z".repeat(1000);
super::send_notification(
&tx,
&notify,
"user-1",
"my-routine",
RunStatus::Failed,
Some(&long_summary),
None,
None,
)
.await;
let msg = rx.recv().await.expect("should receive notification");
// The sanitized summary should be truncated to ~500 chars + "..."
// The full message includes icon + routine name + label, so just check
// it doesn't contain the full 1000-char string.
assert!(
!msg.content.contains(&long_summary),
"Notification should truncate long summaries"
);
assert!(
msg.content.contains("..."),
"Truncated notification should contain ellipsis"
);
}
}
+24 -13
View File
@@ -492,12 +492,10 @@ impl Channel for ReplChannel {
async fn start(&self) -> Result<MessageStream, ChannelError> {
let (tx, rx) = mpsc::channel(32);
// Store tx so send_status can inject approval responses directly.
// Skip for single-message mode — no interactive approval is needed
// and the extra sender would keep the stream open after /quit.
if self.single_message.is_none()
&& let Ok(mut guard) = self.msg_tx.lock()
{
// Approval prompts inject responses back through this sender.
// In single-message mode we keep it until the turn finishes, then
// drop it after enqueuing /quit so the receiver stream can close.
if let Ok(mut guard) = self.msg_tx.lock() {
*guard = Some(tx.clone());
}
let single_message = self.single_message.clone();
@@ -916,10 +914,8 @@ mod tests {
use super::*;
/// Regression: single-message mode must close the stream after the one
/// message so callers (and tests) don't hang forever.
#[tokio::test]
async fn single_message_mode_sends_message_and_closes_stream() {
async fn single_message_mode_sends_message_then_quit() {
let repl = ReplChannel::with_message("hi".to_string());
let mut stream = repl.start().await.expect("repl start should succeed");
@@ -930,15 +926,30 @@ mod tests {
assert_eq!(first.channel, "repl");
assert_eq!(first.content, "hi");
// The spawned thread sent the message and returned, dropping its
// sender. Because we skip storing a clone in msg_tx for single-
// message mode, the stream should close immediately.
assert!(
timeout(Duration::from_millis(100), stream.next())
.await
.is_err(),
"single-message mode should wait for the turn to finish before quitting"
);
repl.respond(&first, OutgoingResponse::text("done"))
.await
.expect("respond should succeed");
let second = timeout(Duration::from_secs(1), stream.next())
.await
.expect("timed out waiting for quit message")
.expect("quit message missing");
assert_eq!(second.channel, "repl");
assert_eq!(second.content, "/quit");
assert!(
timeout(Duration::from_secs(1), stream.next())
.await
.expect("timed out waiting for stream to close")
.is_none(),
"stream should end after the single message"
"stream should end after /quit"
);
}
}
+11 -26
View File
@@ -236,18 +236,6 @@ pub async fn jobs_detail_handler(
(end - start).num_seconds().max(0) as u64
});
// Build transitions from the job's state transition history.
let transitions: Vec<TransitionInfo> = ctx
.transitions
.iter()
.map(|t| TransitionInfo {
from: t.from.to_string(),
to: t.to.to_string(),
timestamp: t.timestamp.to_rfc3339(),
reason: t.reason.clone(),
})
.collect();
// Only show prompt bar for jobs that have a running worker (Pending/InProgress).
// Stuck jobs have no active worker loop, so messages would be silently dropped.
let is_promptable = matches!(
@@ -267,7 +255,7 @@ pub async fn jobs_detail_handler(
project_dir: None,
browse_url: None,
job_mode: None,
transitions,
transitions: Vec::new(),
can_restart: state.scheduler.is_some(),
can_prompt: is_promptable && state.scheduler.is_some(),
job_kind: Some("agent".to_string()),
@@ -655,28 +643,25 @@ pub async fn jobs_events_handler(
.parse()
.map_err(|_| (StatusCode::BAD_REQUEST, "Invalid job ID".to_string()))?;
// Verify ownership before returning events (check both sandbox and agent jobs).
let is_owner = match store.get_sandbox_job(job_id).await {
Ok(Some(job)) => job.user_id == user.user_id,
Ok(None) => {
// Fall back to agent job ownership check.
match store.get_job(job_id).await {
Ok(Some(ctx)) => ctx.user_id == user.user_id,
_ => false,
// Verify ownership before returning events.
match store.get_sandbox_job(job_id).await {
Ok(Some(job)) => {
if job.user_id != user.user_id {
return Err((StatusCode::NOT_FOUND, "Job not found".to_string()));
}
}
Ok(None) => {
return Err((StatusCode::NOT_FOUND, "Job not found".to_string()));
}
Err(e) => {
return Err(db_error("jobs_events_handler", e));
return Err(db_error("jobs_handler", e));
}
};
if !is_owner {
return Err((StatusCode::NOT_FOUND, "Job not found".to_string()));
}
let events = store
.list_job_events(job_id, None)
.await
.map_err(|e| db_error("jobs_events_handler", e))?;
.map_err(|e| (StatusCode::INTERNAL_SERVER_ERROR, e.to_string()))?;
let events_json: Vec<serde_json::Value> = events
.into_iter()
-11
View File
@@ -122,16 +122,6 @@ pub async fn routines_detail_handler(
.collect();
let routine_info = RoutineInfo::from_routine(&routine);
// Read-only lookup — do not create a conversation on a GET request.
// The conversation is created lazily when the routine first executes.
let conversation_id = store
.find_routine_conversation(routine.id, &routine.user_id)
.await
.unwrap_or_else(|e| {
tracing::warn!(routine_id = %routine.id, error = %e, "Failed to look up routine conversation");
None
});
Ok(Json(RoutineDetailResponse {
id: routine.id,
name: routine.name.clone(),
@@ -149,7 +139,6 @@ pub async fn routines_detail_handler(
run_count: routine.run_count,
consecutive_failures: routine.consecutive_failures,
created_at: routine.created_at.to_rfc3339(),
conversation_id,
recent_runs,
}))
}
+8 -8
View File
@@ -347,10 +347,10 @@ impl Channel for GatewayChannel {
let thread_id = match &msg.thread_id {
Some(tid) => tid.clone(),
None => {
return Err(ChannelError::MissingRoutingTarget {
name: "gateway".to_string(),
reason: "respond() requires a thread_id on the incoming message".to_string(),
});
tracing::warn!(
"Gateway respond with no thread_id — skipping (clients would drop it)"
);
return Ok(());
}
};
@@ -507,10 +507,10 @@ impl Channel for GatewayChannel {
let thread_id = match response.thread_id {
Some(tid) => tid,
None => {
return Err(ChannelError::MissingRoutingTarget {
name: "gateway".to_string(),
reason: "broadcast() requires a thread_id on the response".to_string(),
});
tracing::warn!(
"Gateway broadcast with no thread_id — skipping (clients would drop it)"
);
return Ok(());
}
};
self.state.sse.broadcast_for_user(
-12
View File
@@ -4265,13 +4265,6 @@ function renderRoutineDetail(routine) {
html += '<div class="job-description"><h3>Action</h3>'
+ '<pre class="action-json">' + escapeHtml(JSON.stringify(routine.action, null, 2)) + '</pre></div>';
// Conversation thread link
if (routine.conversation_id) {
html += '<div class="job-description">'
+ '<a href="#" data-action="view-routine-thread" data-id="' + escapeHtml(routine.conversation_id) + '" class="btn-primary" style="display:inline-block;margin:0.5rem 0">'
+ 'View Execution Thread</a></div>';
}
// Recent runs
if (routine.recent_runs && routine.recent_runs.length > 0) {
html += '<div class="job-timeline-section"><h3>Recent Runs</h3>'
@@ -6197,11 +6190,6 @@ document.addEventListener('click', function(e) {
switchTab('jobs');
openJobDetail(el.dataset.id);
break;
case 'view-routine-thread':
e.preventDefault();
switchTab('chat');
switchThread(el.dataset.id);
break;
case 'copy-tee-report':
copyTeeReport();
break;
-1
View File
@@ -1,4 +1,3 @@
//! Integration tests for the web gateway module.
mod multi_tenant;
mod no_silent_drop;
-93
View File
@@ -1,93 +0,0 @@
//! Regression tests: the gateway channel must never silently drop messages.
//!
//! Previously, `respond()` and `broadcast()` returned `Ok(())` when thread_id
//! was missing, making callers believe the message was delivered when it wasn't.
//! These tests ensure that missing routing info produces an explicit error.
use crate::channels::channel::{Channel, IncomingMessage, OutgoingResponse};
use crate::channels::web::GatewayChannel;
use crate::config::GatewayConfig;
use crate::error::ChannelError;
fn test_gateway() -> GatewayChannel {
GatewayChannel::new(
GatewayConfig {
host: "127.0.0.1".to_string(),
port: 0,
auth_token: Some("test-token".to_string()),
workspace_read_scopes: vec![],
memory_layers: vec![],
},
"test-user".to_string(),
)
}
#[tokio::test]
async fn gateway_respond_without_thread_id_returns_error() {
let gw = test_gateway();
let msg = IncomingMessage::new("gateway", "test-user", "hello");
// msg has no thread_id by default
assert!(msg.thread_id.is_none());
let response = OutgoingResponse::text("reply");
let result = gw.respond(&msg, response).await;
assert!(
result.is_err(),
"respond() must not silently succeed without thread_id"
);
assert!(
matches!(result, Err(ChannelError::MissingRoutingTarget { .. })),
"Expected MissingRoutingTarget, got: {:?}",
result
);
}
#[tokio::test]
async fn gateway_respond_with_thread_id_succeeds() {
let gw = test_gateway();
let mut msg = IncomingMessage::new("gateway", "test-user", "hello");
msg.thread_id = Some("thread-123".to_string());
let response = OutgoingResponse::text("reply");
let result = gw.respond(&msg, response).await;
assert!(
result.is_ok(),
"respond() should succeed with thread_id: {:?}",
result
);
}
#[tokio::test]
async fn gateway_broadcast_without_thread_id_returns_error() {
let gw = test_gateway();
let response = OutgoingResponse::text("notification");
// response has no thread_id by default
let result = gw.broadcast("test-user", response).await;
assert!(
result.is_err(),
"broadcast() must not silently succeed without thread_id"
);
assert!(
matches!(result, Err(ChannelError::MissingRoutingTarget { .. })),
"Expected MissingRoutingTarget, got: {:?}",
result
);
}
#[tokio::test]
async fn gateway_broadcast_with_thread_id_succeeds() {
let gw = test_gateway();
let response = OutgoingResponse::text("notification").in_thread("thread-456".to_string());
let result = gw.broadcast("test-user", response).await;
assert!(
result.is_ok(),
"broadcast() should succeed with thread_id: {:?}",
result
);
}
-1
View File
@@ -768,7 +768,6 @@ pub struct RoutineDetailResponse {
pub run_count: u64,
pub consecutive_failures: u32,
pub created_at: String,
pub conversation_id: Option<Uuid>,
pub recent_runs: Vec<RoutineRunInfo>,
}
-35
View File
@@ -290,41 +290,6 @@ impl ConversationStore for LibSqlBackend {
result
}
async fn find_routine_conversation(
&self,
routine_id: Uuid,
user_id: &str,
) -> Result<Option<Uuid>, DatabaseError> {
let conn = self.connect().await?;
let rid = routine_id.to_string();
let mut rows = conn
.query(
r#"
SELECT id FROM conversations
WHERE user_id = ?1 AND json_extract(metadata, '$.routine_id') = ?2
LIMIT 1
"#,
params![user_id, rid],
)
.await
.map_err(|e| DatabaseError::Query(e.to_string()))?;
if let Some(row) = rows
.next()
.await
.map_err(|e| DatabaseError::Query(e.to_string()))?
{
let id_str: String = row.get(0).map_err(|e| {
DatabaseError::Query(format!("Failed to read conversation id: {e}"))
})?;
let id = id_str
.parse()
.map_err(|_| DatabaseError::Serialization("Invalid UUID".to_string()))?;
return Ok(Some(id));
}
Ok(None)
}
/// Uses BEGIN IMMEDIATE to serialize concurrent writers and prevent
/// duplicate heartbeat conversations (TOCTOU race).
async fn get_or_create_heartbeat_conversation(
+1 -1
View File
@@ -981,7 +981,7 @@ mod tests {
assert!(alice_stats.last_active_at.is_some());
// Bob has no LLM calls so doesn't appear in summary stats
assert!(stats.iter().find(|s| s.user_id == "bob").is_none());
assert!(!stats.iter().any(|s| s.user_id == "bob"));
// Filter to single user
let alice_only = db.user_summary_stats(Some("alice")).await.unwrap();
-7
View File
@@ -391,13 +391,6 @@ pub trait ConversationStore: Send + Sync {
routine_name: &str,
user_id: &str,
) -> Result<Uuid, DatabaseError>;
/// Read-only lookup for an existing routine conversation. Returns `None`
/// if the routine has never executed (no conversation created yet).
async fn find_routine_conversation(
&self,
routine_id: Uuid,
user_id: &str,
) -> Result<Option<Uuid>, DatabaseError>;
async fn get_or_create_heartbeat_conversation(
&self,
user_id: &str,
-10
View File
@@ -137,16 +137,6 @@ impl ConversationStore for PgBackend {
.await
}
async fn find_routine_conversation(
&self,
routine_id: Uuid,
user_id: &str,
) -> Result<Option<Uuid>, DatabaseError> {
self.store
.find_routine_conversation(routine_id, user_id)
.await
}
async fn get_or_create_heartbeat_conversation(
&self,
user_id: &str,
-21
View File
@@ -1771,27 +1771,6 @@ impl Store {
Ok(row.get("id"))
}
/// Read-only lookup for an existing routine conversation.
pub async fn find_routine_conversation(
&self,
routine_id: Uuid,
user_id: &str,
) -> Result<Option<Uuid>, DatabaseError> {
let conn = self.conn().await?;
let rid = routine_id.to_string();
let row = conn
.query_opt(
r#"
SELECT id FROM conversations
WHERE user_id = $1 AND metadata->>'routine_id' = $2
LIMIT 1
"#,
&[&user_id, &rid],
)
.await?;
Ok(row.map(|r| r.get("id")))
}
/// Get or create the singleton heartbeat conversation for a user.
///
/// Looks for a conversation where `metadata->>'thread_type' = 'heartbeat'`.
+3 -133
View File
@@ -276,33 +276,8 @@ impl LlmProvider for OpenAiCodexProvider {
&self,
request: ToolCompletionRequest,
) -> Result<ToolCompletionResponse, LlmError> {
// Build a reverse map so we can translate sanitized names back to originals.
// Only needed when sanitization actually changes a name (e.g. MCP tools with dots).
let name_map: std::collections::HashMap<String, String> = request
.tools
.iter()
.filter_map(|t| {
let sanitized = sanitize_tool_name(&t.name);
if sanitized != t.name {
Some((sanitized, t.name.clone()))
} else {
None
}
})
.collect();
let body = self.build_request_body(&request.messages, Some(&request.tools));
let mut parsed = self.send_request(body).await?;
// Reverse-map sanitized tool names back to originals so the caller
// can look them up in the tool registry.
if !name_map.is_empty() {
for tc in &mut parsed.tool_calls {
if let Some(original) = name_map.get(&tc.name) {
tc.name = original.clone();
}
}
}
let parsed = self.send_request(body).await?;
let finish_reason = if !parsed.tool_calls.is_empty() {
FinishReason::ToolUse
@@ -446,7 +421,7 @@ fn convert_message(msg: &ChatMessage, index: usize) -> Vec<serde_json::Value> {
serde_json::json!({
"type": "function_call",
"call_id": tc.id,
"name": sanitize_tool_name(&tc.name),
"name": tc.name,
"arguments": args_str,
})
})
@@ -477,20 +452,6 @@ fn convert_message(msg: &ChatMessage, index: usize) -> Vec<serde_json::Value> {
}
}
/// Sanitize a tool name to match the OpenAI Responses API pattern `^[a-zA-Z0-9_-]+$`.
/// Replaces any invalid character (e.g. dots in MCP tool names) with underscores.
fn sanitize_tool_name(name: &str) -> String {
name.chars()
.map(|c| {
if c.is_ascii_alphanumeric() || c == '_' || c == '-' {
c
} else {
'_'
}
})
.collect()
}
/// Convert a `ToolDefinition` to Responses API tool format.
///
/// Applies strict-mode schema normalization (same as OpenAI Chat Completions):
@@ -500,7 +461,7 @@ fn convert_tool_definition(tool: &ToolDefinition) -> serde_json::Value {
serde_json::json!({
"type": "function",
"name": sanitize_tool_name(&tool.name),
"name": tool.name,
"description": tool.description,
"parameters": normalize_schema_strict(&tool.parameters),
})
@@ -1132,95 +1093,4 @@ data: {"type":"response.completed","response":{"status":"completed","usage":{"in
assert_eq!(parsed.tool_calls[1].name, "read_file");
assert_eq!(parsed.finish_reason, FinishReason::ToolUse);
}
/// Regression test: tool names with dots (e.g. MCP tools) must be sanitized
/// to match OpenAI's `^[a-zA-Z0-9_-]+$` pattern.
#[test]
fn test_sanitize_tool_name_replaces_dots() {
assert_eq!(super::sanitize_tool_name("memory_search"), "memory_search");
assert_eq!(
super::sanitize_tool_name("mcp.server.tool"),
"mcp_server_tool"
);
assert_eq!(super::sanitize_tool_name("tool@v2"), "tool_v2");
assert_eq!(super::sanitize_tool_name("my-tool"), "my-tool");
}
/// Regression test: convert_tool_definition sanitizes the name.
#[test]
fn test_convert_tool_definition_sanitizes_name() {
let tool = ToolDefinition {
name: "mcp.server.search".to_string(),
description: "Search".to_string(),
parameters: serde_json::json!({"type": "object", "properties": {}}),
};
let json = super::convert_tool_definition(&tool);
assert_eq!(json["name"], "mcp_server_search");
}
/// Regression test: function_call items sanitize tool names.
#[test]
fn test_convert_message_sanitizes_tool_call_name() {
let tool_calls = vec![ToolCall {
id: "call_1".to_string(),
name: "mcp.server.search".to_string(),
arguments: serde_json::json!({"q": "test"}),
reasoning: None,
}];
let msg = ChatMessage::assistant_with_tool_calls(None, tool_calls);
let items = super::convert_message(&msg, 0);
assert_eq!(items[0]["name"], "mcp_server_search");
}
/// Regression: sanitized tool names in API responses must be reverse-mapped
/// back to original names so the tool registry can look them up.
#[test]
fn test_sanitized_name_reverse_mapping() {
use std::collections::HashMap;
let tools = [
ToolDefinition {
name: "mcp.server.search".to_string(),
description: "Search".to_string(),
parameters: serde_json::json!({"type": "object", "properties": {}}),
},
ToolDefinition {
name: "memory_search".to_string(),
description: "Memory".to_string(),
parameters: serde_json::json!({"type": "object", "properties": {}}),
},
];
// Build name map (same logic as complete_with_tools)
let name_map: HashMap<String, String> = tools
.iter()
.filter_map(|t| {
let sanitized = super::sanitize_tool_name(&t.name);
if sanitized != t.name {
Some((sanitized, t.name.clone()))
} else {
None
}
})
.collect();
// Only the MCP tool should appear (its name changed)
assert_eq!(name_map.len(), 1);
assert_eq!(
name_map.get("mcp_server_search"),
Some(&"mcp.server.search".to_string())
);
// Simulate a tool call coming back with the sanitized name
let mut tc = ToolCall {
id: "call_1".to_string(),
name: "mcp_server_search".to_string(),
arguments: serde_json::json!({}),
reasoning: None,
};
if let Some(original) = name_map.get(&tc.name) {
tc.name = original.clone();
}
assert_eq!(tc.name, "mcp.server.search");
}
}
+2 -10
View File
@@ -182,14 +182,8 @@ impl SkillCatalog {
/// Create a catalog with a custom registry URL (for testing).
#[cfg(test)]
pub fn with_url(url: &str) -> Self {
Self::with_url_and_timeout(url, REQUEST_TIMEOUT)
}
/// Create a catalog with a custom registry URL and timeout (for testing).
#[cfg(test)]
pub fn with_url_and_timeout(url: &str, timeout: Duration) -> Self {
let client = reqwest::Client::builder()
.timeout(timeout)
.timeout(REQUEST_TIMEOUT)
.user_agent(concat!("ironclaw/", env!("CARGO_PKG_VERSION")))
.build()
.unwrap_or_default();
@@ -464,9 +458,7 @@ mod tests {
#[tokio::test]
async fn test_search_returns_error_on_network_failure() {
// Use RFC 5737 TEST-NET-1 (192.0.2.0/24) for reliable failure even behind proxies.
// Short timeout so the test doesn't block for the full 10s REQUEST_TIMEOUT.
let catalog =
SkillCatalog::with_url_and_timeout("http://192.0.2.1:9999", Duration::from_secs(1));
let catalog = SkillCatalog::with_url("http://192.0.2.1:9999");
let outcome = catalog.search("test").await;
assert!(outcome.results.is_empty());
assert!(outcome.error.is_some());
+3 -37
View File
@@ -224,13 +224,7 @@ impl Tool for MessageTool {
) -> Result<ToolOutput, ToolError> {
let start = std::time::Instant::now();
// Accept "message" as an alias for "content" — LLMs frequently use
// the wrong parameter name in autonomous job execution.
let content = require_str(&params, "content").or_else(|_| {
require_str(&params, "message").map_err(|_| {
ToolError::InvalidParameters("missing 'content' parameter".to_string())
})
})?;
let content = require_str(&params, "content")?;
let explicit_channel = params
.get("channel")
@@ -329,11 +323,8 @@ impl Tool for MessageTool {
if !attachments.is_empty() {
response = response.with_attachments(attachments);
}
// Attach thread_id so the gateway can route the message into the
// correct conversation. Previously this only fired when channel was
// explicitly "gateway", which meant broadcast_all (channel=null) sent
// a response without a thread_id and the gateway silently dropped it.
if response.thread_id.is_none()
if channel.as_deref() == Some("gateway")
&& response.thread_id.is_none()
&& let Some(thread_id) = metadata_string(&ctx.metadata, "notify_thread_id")
{
response = response.in_thread(thread_id);
@@ -489,31 +480,6 @@ mod tests {
assert!(params.get("attachments").is_some());
}
/// Regression: LLMs frequently pass {"message": "..."} instead of
/// {"content": "..."}. The tool should accept both.
#[tokio::test]
async fn message_param_alias_accepted() {
let tool = MessageTool::new(Arc::new(ChannelManager::new()));
tool.set_context(Some("gateway".to_string()), Some("user".to_string()))
.await;
let ctx = crate::context::JobContext::new("test", "test");
// "message" alias should not produce InvalidParameters
let result = tool
.execute(serde_json::json!({"message": "hello from alias"}), &ctx)
.await;
// Execution may fail for other reasons (no real channel), but
// the error must NOT be about a missing 'content' parameter.
if let Err(ref e) = result {
let msg = e.to_string();
assert!(
!msg.contains("missing 'content'"),
"Should accept 'message' as alias for 'content', got: {msg}"
);
}
}
#[tokio::test]
async fn message_tool_set_context_updates_defaults() {
let tool = MessageTool::new(Arc::new(ChannelManager::new()));
+4 -68
View File
@@ -65,7 +65,6 @@ struct NormalizedExecutionRequest {
context_paths: Vec<String>,
use_tools: bool,
max_tool_rounds: u32,
max_iterations: u32,
}
#[derive(Debug, Clone, PartialEq, Eq)]
@@ -329,13 +328,6 @@ fn full_job_execution_variant() -> Value {
"type": "string",
"enum": ["full_job"],
"description": "Full-job execution mode."
},
"max_iterations": {
"type": "integer",
"description": "Maximum LLM iterations for the job (default: 25). Increase for complex multi-step tasks.",
"default": 25,
"minimum": 1,
"maximum": 200
}
},
"required": ["mode"]
@@ -652,12 +644,6 @@ pub(crate) fn routine_update_parameters_schema() -> Value {
"description": {
"type": "string",
"description": "New description"
},
"max_iterations": {
"type": "integer",
"description": "Maximum LLM iterations for full_job routines (1-200).",
"minimum": 1,
"maximum": 200
}
},
"required": ["name"]
@@ -901,16 +887,11 @@ fn parse_routine_execution(
.clamp(1, crate::agent::routine::MAX_TOOL_ROUNDS_LIMIT as u64)
as u32;
let max_iterations = u64_field(params, "execution", "max_iterations", &["max_iterations"])
.unwrap_or(25)
.clamp(1, 200) as u32;
Ok(NormalizedExecutionRequest {
mode,
context_paths,
use_tools,
max_tool_rounds,
max_iterations,
})
}
@@ -991,7 +972,7 @@ fn build_routine_action(
NormalizedExecutionMode::FullJob => RoutineAction::FullJob {
title: name.to_string(),
description: prompt.to_string(),
max_iterations: execution.max_iterations,
max_iterations: 10,
},
}
}
@@ -1336,12 +1317,6 @@ impl Tool for RoutineUpdateTool {
}
}
if let Some(iters) = params.get("max_iterations").and_then(|v| v.as_u64())
&& let RoutineAction::FullJob { max_iterations, .. } = &mut routine.action
{
*max_iterations = (iters.clamp(1, 200)) as u32;
}
// Validate timezone param if provided
let new_timezone = params
.get("timezone")
@@ -1569,7 +1544,6 @@ impl Tool for RoutineFireTool {
"name": name,
"run_id": run_id.to_string(),
"status": "fired",
"note": "Routine is executing asynchronously. Use routine_history to check the result.",
});
Ok(ToolOutput::success(result, start.elapsed()))
@@ -1668,47 +1642,10 @@ impl Tool for RoutineHistoryTool {
})
.collect();
// Look up the routine's conversation thread and fetch recent messages
// so the user can see the full output of routine runs.
let (conversation_id, recent_output) = match self
.store
.get_or_create_routine_conversation(routine.id, name, &ctx.user_id)
.await
{
Ok(conv_id) => {
let messages = self
.store
.list_conversation_messages_paginated(conv_id, None, limit)
.await
.map(|(msgs, _)| msgs)
.unwrap_or_default();
let msg_list: Vec<serde_json::Value> = messages
.iter()
.map(|m| {
serde_json::json!({
"role": m.role,
"content": m.content,
"timestamp": m.created_at.to_rfc3339(),
})
})
.collect();
(Some(conv_id.to_string()), msg_list)
}
Err(e) => {
tracing::warn!(
routine = %name,
"Failed to fetch routine conversation thread: {e}"
);
(None, Vec::new())
}
};
let result = serde_json::json!({
"routine": name,
"total_runs": routine.run_count,
"conversation_id": conversation_id,
"runs": run_list,
"recent_output": recent_output,
});
Ok(ToolOutput::success(result, start.elapsed()))
@@ -2345,8 +2282,8 @@ mod tests {
.and_then(Value::as_object)
.expect("full_job properties");
assert!(
full_job_props.contains_key("mode") && full_job_props.contains_key("max_iterations"),
"full_job variant should expose mode and max_iterations",
full_job_props.len() == 1 && full_job_props.contains_key("mode"),
"full_job variant should only expose the execution mode",
);
}
@@ -2554,7 +2491,6 @@ mod tests {
context_paths: Vec::new(),
use_tools: false,
max_tool_rounds: 3,
max_iterations: 25,
};
let action = build_routine_action("issue-1316", "Run it", &execution);
@@ -2567,7 +2503,7 @@ mod tests {
max_iterations,
} if title == "issue-1316"
&& description == "Run it"
&& max_iterations == 25
&& max_iterations == 10
));
}
}
+3 -6
View File
@@ -5,7 +5,7 @@ use chrono::{DateTime, LocalResult, NaiveDate, NaiveDateTime, TimeZone, Utc};
use chrono_tz::Tz;
use crate::context::JobContext;
use crate::tools::tool::{Tool, ToolError, ToolOutput};
use crate::tools::tool::{Tool, ToolError, ToolOutput, require_str};
/// Tool for getting current time and date operations.
pub struct TimeTool;
@@ -62,7 +62,7 @@ impl Tool for TimeTool {
"description": "Second timestamp for diff."
}
},
"required": []
"required": ["operation"]
})
}
@@ -73,10 +73,7 @@ impl Tool for TimeTool {
) -> Result<ToolOutput, ToolError> {
let start = std::time::Instant::now();
let operation = params
.get("operation")
.and_then(|v| v.as_str())
.unwrap_or("now");
let operation = require_str(&params, "operation")?;
let result = match operation {
"now" => execute_now(&params, ctx)?,
+7 -28
View File
@@ -954,22 +954,16 @@ pub async fn store_tokens(
server_config: &McpServerConfig,
token: &AccessToken,
) -> Result<(), AuthError> {
// Store access token (with expiry if provided)
let mut params =
CreateSecretParams::new(server_config.token_secret_name(), &token.access_token)
.with_provider(format!("mcp:{}", server_config.name));
if let Some(secs) = token.expires_in {
let expires_at = chrono::Utc::now() + chrono::Duration::seconds(secs as i64);
params = params.with_expiry(expires_at);
}
// Store access token
let params = CreateSecretParams::new(server_config.token_secret_name(), &token.access_token)
.with_provider(format!("mcp:{}", server_config.name));
secrets
.create(user_id, params)
.await
.map_err(|e| AuthError::Secrets(e.to_string()))?;
// Store refresh token if present (no expiry — long-lived)
// Store refresh token if present
if let Some(ref refresh_token) = token.refresh_token {
let params =
CreateSecretParams::new(server_config.refresh_token_secret_name(), refresh_token)
@@ -1070,26 +1064,11 @@ pub async fn refresh_access_token(
// Get client_id (from config or stored DCR)
let client_id = get_client_id(server_config, secrets, user_id).await?;
// Get the refresh token (try current name, fall back to legacy name for
// users who authenticated before the naming convention was fixed).
// Only fall back on NotFound/Expired — propagate real errors (DB, decryption).
let refresh_token = match secrets
// Get the refresh token
let refresh_token = secrets
.get_decrypted(user_id, &server_config.refresh_token_secret_name())
.await
{
Ok(token) => token,
Err(crate::secrets::SecretError::NotFound(_) | crate::secrets::SecretError::Expired) => {
secrets
.get_decrypted(user_id, &server_config.legacy_refresh_token_secret_name())
.await
.map_err(|e| AuthError::RefreshFailed(format!("No refresh token: {}", e)))?
}
Err(e) => {
return Err(AuthError::RefreshFailed(format!(
"Failed to read refresh token: {e}"
)));
}
};
.map_err(|e| AuthError::RefreshFailed(format!("No refresh token: {}", e)))?;
// Discover the token endpoint
let token_url = if let Some(ref oauth) = server_config.oauth {
-30
View File
@@ -259,9 +259,6 @@ impl McpClient {
}
/// Get the access token for this server (if authenticated).
///
/// If the stored token has expired, automatically attempts a refresh using
/// the stored refresh token before failing.
async fn get_access_token(&self) -> Result<Option<String>, ToolError> {
let Some(ref secrets) = self.secrets else {
return Ok(None);
@@ -275,33 +272,6 @@ impl McpClient {
{
Ok(token) => Ok(Some(token.expose().to_string())),
Err(crate::secrets::SecretError::NotFound(_)) => Ok(None),
Err(crate::secrets::SecretError::Expired) => {
// Token expired — attempt refresh before failing.
tracing::info!(
server = %self.server_name,
"Access token expired, attempting refresh"
);
match refresh_access_token(config, secrets, &self.user_id).await {
Ok(new_token) => {
tracing::info!(
server = %self.server_name,
"Access token refreshed successfully"
);
Ok(Some(new_token.access_token))
}
Err(e) => {
tracing::warn!(
server = %self.server_name,
"Token refresh failed: {}", e
);
Err(ToolError::ExternalService(format!(
"Failed to get access token: Secret has expired \
and refresh failed: {}",
e
)))
}
}
}
Err(e) => Err(ToolError::ExternalService(format!(
"Failed to get access token: {}",
e
-19
View File
@@ -250,19 +250,7 @@ impl McpServerConfig {
}
/// Get the secret name used to store the refresh token.
///
/// Matches the convention used by the hosted OAuth flow in
/// `store_oauth_tokens`: `{token_secret_name}_refresh_token`.
pub fn refresh_token_secret_name(&self) -> String {
format!("{}_refresh_token", self.token_secret_name())
}
/// Legacy secret name for refresh tokens (pre-v0.22).
///
/// Earlier versions stored refresh tokens as `mcp_{name}_refresh_token`
/// instead of `{token_secret_name}_refresh_token`. Used as a fallback
/// during lookup to avoid forcing re-auth on existing users.
pub fn legacy_refresh_token_secret_name(&self) -> String {
format!("mcp_{}_refresh_token", self.name)
}
@@ -762,15 +750,8 @@ mod tests {
fn test_token_secret_names() {
let config = McpServerConfig::new("notion", "https://mcp.notion.com");
assert_eq!(config.token_secret_name(), "mcp_notion_access_token");
// Refresh token name follows the hosted OAuth convention:
// {token_secret_name}_refresh_token
assert_eq!(
config.refresh_token_secret_name(),
"mcp_notion_access_token_refresh_token"
);
// Legacy name used before v0.22 — fallback lookup prevents forced re-auth
assert_eq!(
config.legacy_refresh_token_secret_name(),
"mcp_notion_refresh_token"
);
}
-22
View File
@@ -225,26 +225,4 @@ mod tests {
"The tool returned: TASK_COMPLETE signal"
));
}
#[test]
fn signals_completion_after_suggestions_stripped() {
// Regression: after stripping <suggestions> tags, the completion
// signal should still be detected in the cleaned text.
assert!(llm_signals_completion(
"The job is complete. All requested work has been finished."
));
}
#[test]
fn signals_completion_self_dialogue_pattern() {
// Regression: the "not complete" pattern that caused the self-dialogue
// loop when left in job context after plan completion.
assert!(!llm_signals_completion(
"No — the job is **not complete**.\n\n\
What still needs to be done:\n\
1. Fetch actual meeting note contents\n\
2. Create the Notion page\n\
3. Send the completion message"
));
}
}
+23 -123
View File
@@ -828,11 +828,14 @@ Report when the job is complete or if you encounter issues you cannot resolve."#
}),
);
// All tool errors (including AutonomousUnavailable) are
// recoverable — the error message is already recorded in
// reason_ctx so the LLM can see it and try a different
// approach. Returning Err here would kill the entire job.
Ok(())
if matches!(
&e,
Error::Tool(crate::error::ToolError::AutonomousUnavailable { .. })
) {
Err(e)
} else {
Ok(())
}
}
}
}
@@ -927,31 +930,17 @@ Report when the job is complete or if you encounter issues you cannot resolve."#
tokio::time::sleep(Duration::from_millis(100)).await;
}
// Plan completed — ask the LLM whether the job is done.
let msg_count_before = reason_ctx.messages.len();
// Plan completed, check with LLM if job is done
reason_ctx.messages.push(ChatMessage::user(
"All planned actions have been executed. Assess the results: \
if the job is fully complete, state that the job is complete. \
Otherwise, briefly list what remains.",
"All planned actions have been executed. Is the job complete? If not, what else needs to be done?",
));
let response = reasoning.respond(reason_ctx).await?;
let response = crate::agent::strip_suggestions(&response);
reason_ctx.messages.push(ChatMessage::assistant(&response));
if crate::util::llm_signals_completion(&response) {
reason_ctx.messages.push(ChatMessage::assistant(&response));
self.mark_completed().await?;
} else {
// Replace the completion-check exchange with an action-oriented
// continuation prompt. Leaving the "Is the job complete?" / "No"
// dialogue in context causes the agentic loop to repeat the same
// analysis instead of calling tools (self-dialogue loop).
reason_ctx.messages.truncate(msg_count_before);
reason_ctx.messages.push(ChatMessage::user(format!(
"The planned actions are done but the job is not yet complete. \
Remaining work:\n\n{response}\n\n\
Continue executing now use tools to finish the job."
)));
tracing::info!(
"Job {} plan completed but work remains, falling back to direct selection",
self.job_id
@@ -1431,20 +1420,16 @@ impl<'a> LoopDelegate for JobDelegate<'a> {
return TextAction::Continue;
}
// Jobs run autonomously — strip <suggestions> tags that are only
// meaningful for interactive chat sessions.
let text = crate::agent::strip_suggestions(text);
// A non-empty text response with no tool intent (already filtered
// by the agentic loop's nudge mechanism) is the LLM's final answer.
// Mark the job complete and stop the loop. Without this, the LLM
// restates its summary every iteration until the cap is hit.
if let Err(e) = self.worker.mark_completed().await {
tracing::warn!(
"Failed to mark job {} as completed: {}",
self.worker.job_id,
e
);
// Check for explicit completion
if crate::util::llm_signals_completion(text) {
if let Err(e) = self.worker.mark_completed().await {
tracing::warn!(
"Failed to mark job {} as completed: {}",
self.worker.job_id,
e
);
}
return TextAction::Return(LoopOutcome::Response(text.to_string()));
}
// Track that a substantive response has been produced.
@@ -1452,7 +1437,7 @@ impl<'a> LoopDelegate for JobDelegate<'a> {
.store(true, std::sync::atomic::Ordering::Relaxed);
// Add assistant response to context
reason_ctx.messages.push(ChatMessage::assistant(&text));
reason_ctx.messages.push(ChatMessage::assistant(text));
self.worker.log_event(
"message",
@@ -1462,7 +1447,7 @@ impl<'a> LoopDelegate for JobDelegate<'a> {
}),
);
TextAction::Return(LoopOutcome::Response(text))
TextAction::Continue
}
async fn execute_tool_calls(
@@ -1471,9 +1456,6 @@ impl<'a> LoopDelegate for JobDelegate<'a> {
content: Option<String>,
reason_ctx: &mut ReasoningContext,
) -> Result<Option<LoopOutcome>, crate::error::Error> {
// Strip suggestions from accompanying text (not useful in job context).
let content = content.map(|c| crate::agent::strip_suggestions(&c));
if let Some(ref text) = content {
self.worker.log_event(
"message",
@@ -2174,53 +2156,6 @@ mod tests {
);
}
/// Regression: a text response without rigid completion phrases (e.g.
/// "Weekly review completed and saved to Notion") must still terminate the
/// agentic loop and mark the job complete, rather than continuing until
/// max_iterations.
#[tokio::test]
async fn test_text_response_terminates_loop_without_explicit_completion_phrase() {
let worker = make_worker(vec![]).await;
worker
.context_manager()
.update_context(worker.job_id, |ctx| {
ctx.transition_to(JobState::InProgress, None)
})
.await
.unwrap() // safety: test
.unwrap(); // safety: test
let (_, mut rx) = tokio::sync::mpsc::channel(1);
let delegate = JobDelegate {
worker: &worker,
rx: tokio::sync::Mutex::new(&mut rx),
consecutive_rate_limits: std::sync::atomic::AtomicUsize::new(0),
has_text_response: std::sync::atomic::AtomicBool::new(false),
};
let mut reason_ctx = ReasoningContext::new();
// Text that a real LLM would produce but doesn't match llm_signals_completion
let action = delegate
.handle_text_response(
"Weekly review created in Notion and notification sent.",
&mut reason_ctx,
)
.await;
assert!(
matches!(action, TextAction::Return(_)),
"Text response should terminate the loop, got Continue"
); // safety: test
let ctx = worker
.context_manager()
.get_context(worker.job_id)
.await
.unwrap(); // safety: test
assert_eq!(ctx.state, JobState::Completed); // safety: test
}
/// Regression test: selections_to_tool_calls must preserve tool_call_id
/// so that tool_result messages match the assistant_with_tool_calls message
/// and are not treated as orphaned by sanitize_tool_messages.
@@ -2494,39 +2429,4 @@ mod tests {
}
));
}
/// Regression test: AutonomousUnavailable errors must be recoverable.
/// Previously the job worker treated them as fatal, killing the entire
/// job instead of feeding the error back to the LLM.
#[tokio::test]
async fn test_autonomous_unavailable_is_recoverable() {
let worker = make_worker(vec![]).await;
let mut reason_ctx = ReasoningContext::new();
let selection = ToolSelection {
tool_name: "secret_list".to_string(),
parameters: serde_json::json!({}),
reasoning: "list secrets".to_string(),
alternatives: vec![],
tool_call_id: "call_123".to_string(),
};
let err = Error::Tool(crate::error::ToolError::AutonomousUnavailable {
name: "secret_list".to_string(),
reason: "not available in autonomous jobs".to_string(),
});
let result = worker
.process_tool_result_job(&mut reason_ctx, &selection, Err(err))
.await;
assert!(
result.is_ok(),
"AutonomousUnavailable must be recoverable, not fatal: {:?}",
result
);
// The error should be fed back to the LLM as a message.
assert!(
!reason_ctx.messages.is_empty(),
"Error message should be added to reason_ctx for the LLM"
);
}
}
@@ -14,12 +14,9 @@ scenarios/test_extensions.py
scenarios/test_html_injection.py
scenarios/test_mcp_auth_flow.py
scenarios/test_oauth_credential_fallback.py
scenarios/test_oauth_refresh.py
scenarios/test_oauth_url_parameters.py
scenarios/test_owner_scope.py
scenarios/test_pairing.py
scenarios/test_routine_event_batch.py
scenarios/test_routine_full_job.py
scenarios/test_routine_oauth_credential_injection.py
scenarios/test_skills.py
scenarios/test_sse_reconnect.py
-82
View File
@@ -121,75 +121,6 @@ def _last_user_content(messages: list[dict]) -> str:
return ""
def _is_job_mode(messages: list[dict]) -> bool:
"""Detect if this conversation is a background job (not chat)."""
for msg in messages:
if msg.get("role") == "system":
content = msg.get("content", "")
if "autonomous agent working on a job" in content:
return True
return False
def _count_tool_results(messages: list[dict]) -> int:
"""Count how many tool result messages are in the conversation."""
return sum(1 for m in messages if m.get("role") == "tool")
def match_job_response(messages: list[dict], has_tools: bool) -> dict | None:
"""Handle background job conversations.
Returns a dict with either {"text": ...} or {"tool_call": ...},
or None if this isn't a job conversation.
"""
if not _is_job_mode(messages):
return None
last_user = _last_user_content(messages)
tool_result_count = _count_tool_results(messages)
# Planning call (no tools available = complete() not complete_with_tools())
if "create a plan" in last_user.lower():
return {"text": json.dumps({
"goal": "Complete the requested routine job",
"actions": [
{
"tool_name": "echo",
"parameters": {"message": "job-step-1"},
"reasoning": "First step: echo a test message",
"expected_outcome": "Echo returns the message",
},
{
"tool_name": "time",
"parameters": {"operation": "now"},
"reasoning": "Second step: get the current time",
"expected_outcome": "Returns current timestamp",
},
],
"estimated_cost": 0.001,
"estimated_time_secs": 5,
"confidence": 0.95,
})}
# Post-plan completion check: after tool results, say complete
if "planned actions" in last_user.lower() and tool_result_count >= 2:
return {"text": "The job is complete. All tasks are done."}
# Continuation prompt (from our fix): the plan didn't fully complete,
# now the agentic loop should call tools
if "continue executing now" in last_user.lower() and has_tools:
return {"tool_call": {
"tool_name": "echo",
"arguments": {"message": "continuation-step"},
}}
# After a tool result in the agentic loop, signal completion
if tool_result_count > 0 and has_tools:
return {"text": "The job is complete. All requested work has been finished."}
return None
def match_response(messages: list[dict]) -> str:
content = _last_user_content(messages)
for pattern, response in CANNED_RESPONSES:
@@ -262,19 +193,6 @@ async def chat_completions(request: web.Request) -> web.StreamResponse:
has_tools = bool(body.get("tools"))
cid = f"mock-{uuid.uuid4().hex[:8]}"
# Job-mode conversations (background routine/job execution)
job_resp = match_job_response(messages, has_tools)
if job_resp:
if "tool_call" in job_resp:
tc = job_resp["tool_call"]
if not stream:
return _tool_call_response(cid, tc)
return await _stream_tool_call(request, cid, tc)
text = job_resp["text"]
if not stream:
return _text_response(cid, text)
return await _stream_text(request, cid, text)
# Tool result in messages -> text summary
tr = _find_tool_result(messages)
if tr:
@@ -1,133 +0,0 @@
"""E2E tests for full_job routine execution.
Exercises the complete lifecycle: create a full_job routine via the
web UI, trigger it via the API, and verify the job runs tools and
completes without hitting the iteration cap.
Requires Playwright (browser-based tests).
"""
import asyncio
import uuid
from helpers import SEL, api_get, api_post
# -- Helpers ------------------------------------------------------------------
async def _send_chat_message(page, message: str) -> None:
"""Send a chat message and wait for the assistant turn to appear."""
chat_input = page.locator(SEL["chat_input"])
await chat_input.wait_for(state="visible", timeout=5000)
assistant_messages = page.locator(SEL["message_assistant"])
before_count = await assistant_messages.count()
await chat_input.fill(message)
await chat_input.press("Enter")
await page.wait_for_function(
"""({ selector, expectedCount }) => {
return document.querySelectorAll(selector).length >= expectedCount;
}""",
arg={
"selector": SEL["message_assistant"],
"expectedCount": before_count + 1,
},
timeout=30000,
)
async def _wait_for_routine(base_url: str, name: str, timeout: float = 20.0) -> dict:
"""Poll until the named routine exists."""
for _ in range(int(timeout * 2)):
resp = await api_get(base_url, "/api/routines")
resp.raise_for_status()
for routine in resp.json()["routines"]:
if routine["name"] == name:
return routine
await asyncio.sleep(0.5)
raise AssertionError(f"Routine '{name}' not created within {timeout}s")
async def _get_routine_runs(base_url: str, routine_id: str) -> list[dict]:
"""Fetch routine runs."""
resp = await api_get(base_url, f"/api/routines/{routine_id}/runs")
resp.raise_for_status()
return resp.json()["runs"]
async def _wait_for_completed_run(
base_url: str,
routine_id: str,
*,
timeout: float = 60.0,
) -> dict:
"""Poll until the newest run reaches a terminal state."""
for _ in range(int(timeout * 2)):
runs = await _get_routine_runs(base_url, routine_id)
if runs and runs[0]["status"].lower() not in ("running", "pending"):
return runs[0]
await asyncio.sleep(0.5)
raise AssertionError(
f"Routine '{routine_id}' did not complete within {timeout}s"
)
async def _wait_for_job_terminal(
base_url: str,
job_id: str,
*,
timeout: float = 60.0,
) -> dict:
"""Poll until a job reaches a terminal state."""
terminal = {"completed", "failed", "cancelled", "submitted", "accepted"}
for _ in range(int(timeout * 2)):
resp = await api_get(base_url, f"/api/jobs/{job_id}")
resp.raise_for_status()
detail = resp.json()
if detail.get("state", "").lower() in terminal:
return detail
await asyncio.sleep(0.5)
raise AssertionError(f"Job '{job_id}' did not reach terminal state within {timeout}s")
# -- Tests --------------------------------------------------------------------
async def test_full_job_routine_completes_with_tools(page, ironclaw_server):
"""A full_job routine should plan, execute tools, and complete."""
name = f"fjob-{uuid.uuid4().hex[:8]}"
# Step 1: Create full_job routine via chat
await _send_chat_message(page, f"create full-job owner routine {name}")
routine = await _wait_for_routine(ironclaw_server, name)
assert routine["id"]
assert routine["action_type"] == "full_job"
# Step 2: Trigger the routine
resp = await api_post(ironclaw_server, f"/api/routines/{routine['id']}/trigger")
resp.raise_for_status()
trigger_data = resp.json()
assert trigger_data["status"] == "triggered"
# Step 3: Wait for the run to complete
completed_run = await _wait_for_completed_run(
ironclaw_server, routine["id"], timeout=60
)
# The run should have succeeded (not failed)
assert completed_run["status"].lower() != "failed", (
f"Full job routine run failed: {completed_run}"
)
# Step 4: Verify the job reached a success state.
# Jobs may advance past "completed" to "submitted" or "accepted",
# so treat all post-completion states as success.
success_states = {"completed", "submitted", "accepted"}
if completed_run.get("job_id"):
job = await _wait_for_job_terminal(
ironclaw_server, completed_run["job_id"], timeout=30
)
assert job["state"].lower() in success_states, (
f"Expected job state in {success_states}, got '{job['state']}'"
)
+12 -10
View File
@@ -21,7 +21,9 @@ mod tests {
NotifyConfig, Routine, RoutineAction, RoutineGuardrails, RoutineRun, RunStatus, Trigger,
};
use ironclaw::agent::routine_engine::RoutineEngine;
use ironclaw::agent::{HeartbeatConfig, HeartbeatRunner, Scheduler, SchedulerDeps};
use ironclaw::agent::{
HeartbeatConfig, HeartbeatRunner, SandboxReadiness, Scheduler, SchedulerDeps,
};
use ironclaw::channels::IncomingMessage;
use ironclaw::config::{AgentConfig, RoutineConfig, SafetyConfig};
use ironclaw::context::{ContextManager, JobContext};
@@ -350,7 +352,7 @@ mod tests {
extension_manager,
registry,
safety,
ironclaw::agent::routine_engine::SandboxReadiness::DisabledByConfig,
SandboxReadiness::Available,
))
}
@@ -454,7 +456,7 @@ mod tests {
None,
tools,
safety,
ironclaw::agent::routine_engine::SandboxReadiness::DisabledByConfig,
SandboxReadiness::DisabledByConfig,
));
// Insert a cron routine with next_fire_at in the past.
@@ -533,7 +535,7 @@ mod tests {
None,
tools,
safety,
ironclaw::agent::routine_engine::SandboxReadiness::DisabledByConfig,
SandboxReadiness::DisabledByConfig,
));
// Insert an event routine matching "deploy.*production".
@@ -620,7 +622,7 @@ mod tests {
None,
tools,
safety,
ironclaw::agent::routine_engine::SandboxReadiness::DisabledByConfig,
SandboxReadiness::DisabledByConfig,
));
let routine = make_routine(
@@ -729,7 +731,7 @@ mod tests {
None,
tools,
safety,
ironclaw::agent::routine_engine::SandboxReadiness::DisabledByConfig,
SandboxReadiness::DisabledByConfig,
));
let mut filters = std::collections::HashMap::new();
@@ -872,7 +874,7 @@ mod tests {
None,
tools,
safety,
ironclaw::agent::routine_engine::SandboxReadiness::DisabledByConfig,
SandboxReadiness::DisabledByConfig,
));
// Insert an event routine with 1-hour cooldown.
@@ -1055,7 +1057,7 @@ mod tests {
None,
tools,
safety,
ironclaw::agent::routine_engine::SandboxReadiness::DisabledByConfig,
SandboxReadiness::DisabledByConfig,
));
(engine, db, dir)
@@ -1177,7 +1179,7 @@ mod tests {
None,
tools,
safety,
ironclaw::agent::routine_engine::SandboxReadiness::DisabledByConfig,
SandboxReadiness::DisabledByConfig,
));
// Create a full_job routine with max_concurrent = 1
@@ -1285,7 +1287,7 @@ mod tests {
None,
tools,
safety,
ironclaw::agent::routine_engine::SandboxReadiness::DisabledByConfig,
SandboxReadiness::DisabledByConfig,
));
// Insert a due cron routine
+1 -1
View File
@@ -198,7 +198,7 @@ mod tests {
http_interceptor: None,
transcription: None,
document_extraction: None,
sandbox_readiness: ironclaw::agent::routine_engine::SandboxReadiness::DisabledByConfig,
sandbox_readiness: ironclaw::agent::SandboxReadiness::DisabledByConfig,
builder: None,
llm_backend: "nearai".to_string(),
tenant_rates: std::sync::Arc::new(ironclaw::tenant::TenantRateRegistry::new(4, 3)),
+1 -2
View File
@@ -264,8 +264,7 @@ impl GatewayWorkflowHarness {
http_interceptor: None,
transcription: None,
document_extraction: None,
sandbox_readiness:
ironclaw::agent::routine_engine::SandboxReadiness::DisabledByConfig,
sandbox_readiness: ironclaw::agent::SandboxReadiness::DisabledByConfig,
builder: None,
llm_backend: "nearai".to_string(),
tenant_rates: std::sync::Arc::new(ironclaw::tenant::TenantRateRegistry::new(4, 3)),
+2 -2
View File
@@ -650,7 +650,7 @@ impl TestRigBuilder {
None,
components.tools.clone(),
components.safety.clone(),
ironclaw::agent::routine_engine::SandboxReadiness::DisabledByConfig,
ironclaw::agent::SandboxReadiness::Available, // tests don't use real Docker
));
components
.tools
@@ -759,7 +759,7 @@ impl TestRigBuilder {
http_interceptor,
transcription: None,
document_extraction: None,
sandbox_readiness: ironclaw::agent::routine_engine::SandboxReadiness::DisabledByConfig,
sandbox_readiness: ironclaw::agent::SandboxReadiness::Available, // tests don't use real Docker
builder: None,
llm_backend: "nearai".to_string(),
tenant_rates: std::sync::Arc::new(ironclaw::tenant::TenantRateRegistry::new(4, 3)),