Files
optimclaw/crates/ironclaw_engine/src/executor/trace.rs
T
[email protected]andClaude Opus 4.6 e82dcbd5e6 feat(bridge): implement tool approval flow for engine v2
Adds a complete approval flow that mirrors v1 behavior, using the
existing v1 security controls (Tool::requires_approval, auto-approve
sets, StatusUpdate::ApprovalNeeded).

## How it works

### Step 1: Tool blocked at execution
When the LLM's code calls a tool (e.g., `shell("ls")`):
1. EffectBridgeAdapter.execute_action() looks up the Tool object
2. Calls tool.requires_approval(&params) — returns ApprovalRequirement
3. If Always → EngineError::LeaseDenied (always blocks)
4. If UnlessAutoApproved → checks auto_approved HashSet → if not in set,
   returns EngineError::LeaseDenied
5. If Never → proceeds to execution

### Step 2: Engine returns NeedApproval
The LeaseDenied error propagates through:
- CodeAct path: becomes Python RuntimeError, code halts, thread returns
  NeedApproval with action_name + parameters
- Structured path: same via ActionResult.is_error

### Step 3: Router stores pending approval
- PendingApproval { action_name, original_content } stored on EngineState
- StatusUpdate::ApprovalNeeded sent to channel (shows approval card in
  CLI/web with tool name, parameters, yes/always/no buttons)
- Returns text: "Tool 'shell' requires approval. Reply yes/always/no."

### Step 4: User responds
handle_message() intercepts Submission::ApprovalResponse when ENGINE_V2:
- 'yes' → auto_approve_tool(name) on EffectBridgeAdapter, re-processes
  original message (tool now passes the approval check on second run)
- 'always' → same + logs for session persistence
- 'no' → returns "Denied: tool was not executed."

### Key design choice
Instead of pausing/resuming mid-execution (which needs engine changes
to freeze/restore the Monty VM state), we auto-approve the tool and
re-run the full message. The EffectBridgeAdapter's auto_approved set
persists across runs, so the second execution passes immediately.

This trades one extra LLM call for zero engine modifications.

## Files changed
- src/bridge/router.rs: PendingApproval struct, handle_approval(),
  NeedApproval → StatusUpdate::ApprovalNeeded conversion
- src/bridge/mod.rs: export handle_approval
- src/agent/agent_loop.rs: intercept ApprovalResponse for engine v2
- src/bridge/effect_adapter.rs: fmt fixes

151 tests passing, clippy + fmt clean.

Co-Authored-By: Claude Opus 4.6 (1M context) <[email protected]>
2026-03-23 21:10:40 -07:00

368 lines
12 KiB
Rust

//! Execution trace recording and analysis.
//!
//! Records full execution traces to JSON files for debugging. Optionally
//! runs a post-execution analysis to detect common issues.
//!
//! Enable with `ENGINE_V2_TRACE=1` env var. Traces are written to
//! `engine_trace_{timestamp}.json` in the current directory.
use std::path::PathBuf;
use chrono::Utc;
use serde::Serialize;
use tracing::{info, warn};
use crate::types::event::ThreadEvent;
use crate::types::thread::{Thread, ThreadId, ThreadState};
/// Check if trace recording is enabled.
pub fn is_trace_enabled() -> bool {
std::env::var("ENGINE_V2_TRACE")
.map(|v| v == "1" || v == "true")
.unwrap_or(false)
}
/// A complete execution trace for a single thread.
#[derive(Debug, Serialize)]
pub struct ExecutionTrace {
pub thread_id: ThreadId,
pub goal: String,
pub final_state: ThreadState,
pub step_count: usize,
pub total_tokens: u64,
pub messages: Vec<MessageRecord>,
pub events: Vec<ThreadEvent>,
pub issues: Vec<TraceIssue>,
pub reflection: Option<ReflectionTrace>,
pub timestamp: chrono::DateTime<Utc>,
}
/// Reflection results captured in the trace.
#[derive(Debug, Serialize)]
pub struct ReflectionTrace {
pub docs: Vec<ReflectionDocRecord>,
pub tokens_used: u64,
}
/// A single doc produced by reflection, for the trace.
#[derive(Debug, Serialize)]
pub struct ReflectionDocRecord {
pub doc_type: String,
pub title: String,
pub content: String,
}
/// A message in the trace with role labeling.
#[derive(Debug, Serialize)]
pub struct MessageRecord {
pub role: String,
pub content_length: usize,
pub content_preview: String,
pub full_content: String,
pub action_name: Option<String>,
pub action_call_id: Option<String>,
}
/// An issue detected by the retrospective analyzer.
#[derive(Debug, Serialize)]
pub struct TraceIssue {
pub severity: IssueSeverity,
pub category: String,
pub description: String,
pub step: Option<usize>,
}
#[derive(Debug, Serialize)]
pub enum IssueSeverity {
Error,
Warning,
Info,
}
/// Build a trace from a completed thread.
pub fn build_trace(thread: &Thread) -> ExecutionTrace {
let messages: Vec<MessageRecord> = thread
.messages
.iter()
.map(|m| {
let preview: String = m.content.chars().take(300).collect();
MessageRecord {
role: format!("{:?}", m.role),
content_length: m.content.chars().count(),
content_preview: if m.content.chars().count() > 300 {
format!("{preview}...")
} else {
preview
},
full_content: m.content.clone(),
action_name: m.action_name.clone(),
action_call_id: m.action_call_id.clone(),
}
})
.collect();
let issues = analyze_trace(thread);
ExecutionTrace {
thread_id: thread.id,
goal: thread.goal.clone(),
final_state: thread.state,
step_count: thread.step_count,
total_tokens: thread.total_tokens_used,
messages,
events: thread.events.clone(),
issues,
reflection: None,
timestamp: Utc::now(),
}
}
/// Write a trace to a JSON file.
pub fn write_trace(trace: &ExecutionTrace) -> Option<PathBuf> {
let filename = format!("engine_trace_{}.json", Utc::now().format("%Y%m%dT%H%M%S"));
let path = PathBuf::from(&filename);
match serde_json::to_string_pretty(trace) {
Ok(json) => match std::fs::write(&path, json) {
Ok(()) => {
info!(path = %path.display(), "Execution trace written");
Some(path)
}
Err(e) => {
warn!("Failed to write trace: {e}");
None
}
},
Err(e) => {
warn!("Failed to serialize trace: {e}");
None
}
}
}
/// Attach reflection results to a trace.
pub fn attach_reflection(trace: &mut ExecutionTrace, result: &crate::reflection::ReflectionResult) {
trace.reflection = Some(ReflectionTrace {
docs: result
.docs
.iter()
.map(|d| ReflectionDocRecord {
doc_type: format!("{:?}", d.doc_type),
title: d.title.clone(),
content: d.content.clone(),
})
.collect(),
tokens_used: result.tokens_used.total(),
});
}
/// Print a summary of the trace to the log.
pub fn log_trace_summary(trace: &ExecutionTrace) {
info!(
thread_id = %trace.thread_id,
goal = %trace.goal,
state = ?trace.final_state,
steps = trace.step_count,
tokens = trace.total_tokens,
messages = trace.messages.len(),
events = trace.events.len(),
issues = trace.issues.len(),
"=== Engine V2 Trace Summary ==="
);
for issue in &trace.issues {
match issue.severity {
IssueSeverity::Error => warn!(
category = %issue.category,
step = ?issue.step,
"ISSUE: {}",
issue.description
),
IssueSeverity::Warning => warn!(
category = %issue.category,
step = ?issue.step,
"WARNING: {}",
issue.description
),
IssueSeverity::Info => info!(
category = %issue.category,
step = ?issue.step,
"NOTE: {}",
issue.description
),
}
}
if let Some(ref refl) = trace.reflection {
info!(
thread_id = %trace.thread_id,
docs = refl.docs.len(),
tokens = refl.tokens_used,
"=== Reflection ==="
);
for doc in &refl.docs {
let preview: String = doc.content.chars().take(200).collect();
let truncated = if doc.content.chars().count() > 200 {
"..."
} else {
""
};
info!(
doc_type = %doc.doc_type,
title = %doc.title,
" {preview}{truncated}"
);
}
}
}
// ── Retrospective analysis ──────────────────────────────────
/// Analyze a completed thread for common issues.
fn analyze_trace(thread: &Thread) -> Vec<TraceIssue> {
let mut issues = Vec::new();
// 1. Check if the thread failed
if thread.state == ThreadState::Failed {
issues.push(TraceIssue {
severity: IssueSeverity::Error,
category: "thread_failure".into(),
description: "Thread ended in Failed state".into(),
step: None,
});
}
// 2. Check for empty response (no FINAL, no useful output)
let has_assistant_response = thread
.messages
.iter()
.any(|m| m.role == crate::types::message::MessageRole::Assistant && !m.content.is_empty());
if !has_assistant_response {
issues.push(TraceIssue {
severity: IssueSeverity::Warning,
category: "no_response".into(),
description: "No assistant message in thread — model may not have generated output"
.into(),
step: None,
});
}
// 3. Check for tool errors
let tool_errors: Vec<&ThreadEvent> = thread
.events
.iter()
.filter(|e| matches!(e.kind, crate::types::event::EventKind::ActionFailed { .. }))
.collect();
if !tool_errors.is_empty() {
for event in &tool_errors {
if let crate::types::event::EventKind::ActionFailed {
action_name, error, ..
} = &event.kind
{
issues.push(TraceIssue {
severity: IssueSeverity::Warning,
category: "tool_error".into(),
description: format!("Tool '{action_name}' failed: {error}"),
step: None,
});
}
}
}
// 4. Check for code execution errors (NameError, etc. in messages)
for (i, msg) in thread.messages.iter().enumerate() {
if msg.role == crate::types::message::MessageRole::System
&& (msg.content.contains("NameError")
|| msg.content.contains("SyntaxError")
|| msg.content.contains("TypeError")
|| msg.content.contains("Error:"))
{
let preview: String = msg.content.chars().take(200).collect();
issues.push(TraceIssue {
severity: IssueSeverity::Warning,
category: "code_error".into(),
description: format!("Code error in message {i}: {preview}"),
step: None,
});
}
}
// 5. Check for model ignoring tool results (hallucination risk)
let has_tool_results = thread
.messages
.iter()
.any(|m| m.role == crate::types::message::MessageRole::ActionResult);
// Check if tool outputs are visible in the message history (any role).
// The engine adds tool results as system messages with "[tool_name result]"
// or "[tool_name error]" prefixes.
let has_tool_output_in_messages = thread
.messages
.iter()
.any(|m| m.content.contains(" result]") || m.content.contains(" error]"));
if has_tool_results && !has_tool_output_in_messages {
issues.push(TraceIssue {
severity: IssueSeverity::Warning,
category: "missing_tool_output".into(),
description: "Tool results exist but no tool output in system messages — model may not see tool results".into(),
step: None,
});
}
// 6. Check for excessive iterations
if thread.step_count > 10 {
issues.push(TraceIssue {
severity: IssueSeverity::Warning,
category: "excessive_steps".into(),
description: format!(
"Thread took {} steps — may be stuck in a loop",
thread.step_count
),
step: None,
});
}
// 7. Check for text response without FINAL (model answered from memory)
let text_without_code = thread.events.iter().all(|e| {
!matches!(
e.kind,
crate::types::event::EventKind::ActionExecuted { .. }
)
});
if text_without_code && thread.step_count == 1 && has_assistant_response {
issues.push(TraceIssue {
severity: IssueSeverity::Info,
category: "no_tools_used".into(),
description: "Model answered in one step without using any tools — may be answering from training data".into(),
step: Some(1),
});
}
// 8. Check for LLM not producing code blocks
let code_steps = thread
.events
.iter()
.filter(|e| matches!(e.kind, crate::types::event::EventKind::StepStarted { .. }))
.count();
let text_responses_without_code = thread
.messages
.iter()
.filter(|m| {
m.role == crate::types::message::MessageRole::Assistant
&& !m.content.contains("```")
&& !m.content.contains("FINAL(")
})
.count();
if text_responses_without_code > 0 && code_steps > 0 {
issues.push(TraceIssue {
severity: IssueSeverity::Info,
category: "mixed_mode".into(),
description: format!(
"{text_responses_without_code} text response(s) without code blocks — model may not be following CodeAct prompt"
),
step: None,
});
}
issues
}