mirror of
https://github.com/outbackdingo/optimclaw.git
synced 2026-08-30 16:19:21 +00:00
feat(agent): thread per-tool reasoning through provider, session, and all surfaces (#1513)
* feat(agent): thread per-tool reasoning from LLM through to REPL, HTTP, SSE, and DB Add end-to-end agent reasoning summaries so users can see *why* the agent chose specific tools, not just what it did. - Add `reasoning: Option<String>` to `ToolCall` (all providers) - Populate from LLM response content in `Reasoning::respond_with_tools` and `select_tools`, with per-tool override when providers supply it - Extend `Turn` with `narrative` and `TurnToolCall` with `rationale` + `tool_call_id` for identity-based result matching - Persist reasoning in DB via existing tool_calls JSON (no migration) - Add `StatusUpdate::ReasoningUpdate` and `SseEvent::ReasoningUpdate` + `SseEvent::JobReasoning` for real-time streaming - Emit reasoning events in both chat dispatcher and worker job path - Add `/reasoning [N|all]` command for inspecting turn reasoning - Surface `narrative` and `rationale` in HTTP `/api/chat/history` Based on the design from #361 and #456, reconstructed cleanly with Option<String> to minimize blast radius (vs mandatory String that broke compilation in #456). Closes #456 Co-Authored-By: panosAthDBX <[email protected]> Co-Authored-By: Claude Opus 4.6 (1M context) <[email protected]> * fix: address PR review feedback from Gemini and Copilot - Fix `_ => Ok(None)` in agent_loop.rs to avoid accidental shutdown - Fix fallback in record_tool_result_for/record_tool_error_for to use first pending call instead of last_mut (parallel execution safety) - Include per-tool decisions in WASM channel reasoning messages - Apply truncate_at_tool_tags + clean_response to shared_reasoning in select_tools (parity with respond_with_tools) - Persist turn-level narrative to DB in tool_calls JSON wrapper - Parse both old (array) and new (object) tool_calls formats in build_turns_from_db_messages for backward compatibility - Populate reasoning from action.reasoning in execute_plan ToolCalls [skip-regression-check] Co-Authored-By: Claude Opus 4.6 (1M context) <[email protected]> * fix: address second round of review comments + merge fixes - Add reasoning: None to new github_copilot.rs ToolCall sites (from staging merge) - Run cargo fmt on 4 files with formatting diffs - Truncate narrative to 1000 chars before DB persistence - Clone turn data and drop session lock in /reasoning command - Extract ToolDecisionDto::from_json_array shared helper (deduplicate worker/job.rs and orchestrator/api.rs) - Add unit tests for wrapped tool_calls JSON format with narrative [skip-regression-check] Co-Authored-By: Claude Opus 4.6 (1M context) <[email protected]> * fix: address third round of review comments (Copilot + serrrfirat) - Reword ToolCall.reasoning docstring to reflect provider-supplied or fallback contract - Sanitize narrative through SafetyLayer before storage/emission - Clean per-tool reasoning via truncate_at_tool_tags + clean_response in select_tools (parity with shared reasoning) - Convert 4 approval-path recording sites in thread_ops.rs to identity-based record_tool_result_for/record_tool_error_for - Preserve tool_call_id and reasoning through restore_from_messages - Fix has_result/has_error to reject JSON null values - Truncate tool_call_id to 128 chars before DB persistence - Add 4 unit tests for record_tool_result_for/error_for edge cases Co-Authored-By: Claude Opus 4.6 (1M context) <[email protected]> * fix: address zmanian review — sanitize JobDelegate reasoning + warn on dropped results - Sanitize narrative and per-tool rationale through SafetyLayer in JobDelegate reasoning events (parity with ChatDelegate) - Add tracing::warn when record_tool_result_for/error_for drops a result because no matching or pending tool call exists - Add 3 unit tests for reasoning normalization (thinking tags, tool tags, empty-after-cleaning) Co-Authored-By: Claude Opus 4.6 (1M context) <[email protected]> * fix: address 4 remaining unreplied review comments - Clean per-tool reasoning in respond_with_tools via truncate_at_tool_tags + clean_response (parity with select_tools) - Handle wrapped JSON format in rebuild_chat_messages_from_db so cold hydration works after persist_tool_calls format change - Update persist_tool_calls doc comment to describe new JSON shape - Sanitize per-tool rationale through SafetyLayer in ChatDelegate before emission and storage (parity with JobDelegate) Co-Authored-By: Claude Opus 4.6 (1M context) <[email protected]> * fix: address zmanian review round 2 - Add tracing::debug on fallback-to-pending path in record_tool_result_for and record_tool_error_for (item 1) - Add comment explaining why /reasoning is special-cased in agent_loop.rs (item 4) - Items 2 (narrative persistence), 3 (rationale sanitization), and 5 (catch-all fix) were already addressed in prior commits Co-Authored-By: Claude Opus 4.6 (1M context) <[email protected]> --------- Co-authored-by: panosAthDBX <[email protected]> Co-authored-by: Claude Opus 4.6 (1M context) <[email protected]>
This commit is contained in:
co-authored by
panosAthDBX
Claude Opus 4.6
parent
6daa2f155f
commit
41ed0a0f98
@@ -575,6 +575,7 @@ fn extract_response_content(response: &AnthropicResponse) -> (Option<String>, Ve
|
||||
id: id.clone(),
|
||||
name: name.clone(),
|
||||
arguments: input.clone(),
|
||||
reasoning: None,
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -623,6 +624,7 @@ mod tests {
|
||||
id: "call_1".to_string(),
|
||||
name: "search".to_string(),
|
||||
arguments: serde_json::json!({"q": "test"}),
|
||||
reasoning: None,
|
||||
}];
|
||||
let messages = vec![
|
||||
ChatMessage::user("Search for test"),
|
||||
|
||||
@@ -522,6 +522,7 @@ fn extract_content_blocks(
|
||||
id: tu.tool_use_id().to_string(),
|
||||
name: tu.name().to_string(),
|
||||
arguments: document_to_json(tu.input()),
|
||||
reasoning: None,
|
||||
});
|
||||
}
|
||||
// Ignore reasoning, citations, images, etc.
|
||||
@@ -759,11 +760,13 @@ mod tests {
|
||||
id: "call_1".to_string(),
|
||||
name: "echo".to_string(),
|
||||
arguments: serde_json::json!({"text": "hi"}),
|
||||
reasoning: None,
|
||||
};
|
||||
let tc2 = crate::llm::provider::ToolCall {
|
||||
id: "call_2".to_string(),
|
||||
name: "time".to_string(),
|
||||
arguments: serde_json::json!({}),
|
||||
reasoning: None,
|
||||
};
|
||||
|
||||
let messages = vec![
|
||||
@@ -802,6 +805,7 @@ mod tests {
|
||||
id: "call_1".to_string(),
|
||||
name: "search".to_string(),
|
||||
arguments: serde_json::json!({"query": "test"}),
|
||||
reasoning: None,
|
||||
};
|
||||
|
||||
let messages = vec![
|
||||
@@ -825,6 +829,7 @@ mod tests {
|
||||
id: "call_1".to_string(),
|
||||
name: "echo".to_string(),
|
||||
arguments: serde_json::json!({}),
|
||||
reasoning: None,
|
||||
};
|
||||
|
||||
let messages = vec![
|
||||
@@ -989,11 +994,13 @@ mod tests {
|
||||
id: "call_abc".to_string(),
|
||||
name: "get_weather".to_string(),
|
||||
arguments: serde_json::json!({"city": "NYC"}),
|
||||
reasoning: None,
|
||||
};
|
||||
let tc2 = crate::llm::provider::ToolCall {
|
||||
id: "call_def".to_string(),
|
||||
name: "get_time".to_string(),
|
||||
arguments: serde_json::json!({"tz": "EST"}),
|
||||
reasoning: None,
|
||||
};
|
||||
|
||||
let messages = vec![
|
||||
|
||||
@@ -732,6 +732,7 @@ impl LlmProvider for CodexChatGptProvider {
|
||||
id: tc.call_id,
|
||||
name: tc.name,
|
||||
arguments: args,
|
||||
reasoning: None,
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
@@ -825,6 +826,7 @@ mod tests {
|
||||
id: "call_1".to_string(),
|
||||
name: "search".to_string(),
|
||||
arguments: json!({"query": "rust"}),
|
||||
reasoning: None,
|
||||
};
|
||||
let msg = ChatMessage::assistant_with_tool_calls(Some("thinking...".into()), vec![tc]);
|
||||
let items = CodexChatGptProvider::message_to_input_items(&msg);
|
||||
|
||||
@@ -1898,6 +1898,7 @@ impl GeminiOauthProvider {
|
||||
id,
|
||||
name,
|
||||
arguments: args,
|
||||
reasoning: None,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -596,6 +596,7 @@ fn extract_choice_content(choice: &OpenAiChoice) -> (Option<String>, Vec<ToolCal
|
||||
name: tc.function.name.clone(),
|
||||
arguments: serde_json::from_str(&tc.function.arguments)
|
||||
.unwrap_or(serde_json::Value::Object(serde_json::Map::new())),
|
||||
reasoning: None,
|
||||
})
|
||||
.collect()
|
||||
})
|
||||
@@ -628,6 +629,7 @@ mod tests {
|
||||
id: "call_1".to_string(),
|
||||
name: "search".to_string(),
|
||||
arguments: serde_json::json!({"q": "test"}),
|
||||
reasoning: None,
|
||||
}];
|
||||
let messages = vec![
|
||||
ChatMessage::user("Search"),
|
||||
|
||||
@@ -587,6 +587,7 @@ impl LlmProvider for NearAiChatProvider {
|
||||
id: tc.id,
|
||||
name: tc.function.name,
|
||||
arguments,
|
||||
reasoning: None,
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
@@ -1180,11 +1181,13 @@ mod tests {
|
||||
id: "call_1".to_string(),
|
||||
name: "list_issues".to_string(),
|
||||
arguments: serde_json::json!({"owner": "foo", "repo": "bar"}),
|
||||
reasoning: None,
|
||||
},
|
||||
ToolCall {
|
||||
id: "call_2".to_string(),
|
||||
name: "search".to_string(),
|
||||
arguments: serde_json::json!({"query": "test"}),
|
||||
reasoning: None,
|
||||
},
|
||||
];
|
||||
|
||||
@@ -1217,6 +1220,7 @@ mod tests {
|
||||
id: "call_1".to_string(),
|
||||
name: "test".to_string(),
|
||||
arguments: serde_json::json!({"key": "value"}),
|
||||
reasoning: None,
|
||||
};
|
||||
let msg = ChatMessage::assistant_with_tool_calls(None, vec![tc]);
|
||||
let chat_msg: ChatCompletionMessage = msg.into();
|
||||
@@ -1460,6 +1464,7 @@ mod tests {
|
||||
id: tc.id,
|
||||
name: tc.function.name,
|
||||
arguments,
|
||||
reasoning: None,
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
@@ -1509,6 +1514,7 @@ mod tests {
|
||||
id: tc.id,
|
||||
name: tc.function.name,
|
||||
arguments,
|
||||
reasoning: None,
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
@@ -2131,6 +2137,7 @@ mod tests {
|
||||
id: "call_1".to_string(),
|
||||
name: "test".to_string(),
|
||||
arguments: serde_json::json!({}),
|
||||
reasoning: None,
|
||||
}],
|
||||
);
|
||||
let chat_msg: ChatCompletionMessage = msg.into();
|
||||
|
||||
@@ -625,6 +625,7 @@ fn parse_sse_response(body: &str) -> Result<ParsedResponse, LlmError> {
|
||||
id: state.call_id,
|
||||
name: state.name,
|
||||
arguments,
|
||||
reasoning: None,
|
||||
});
|
||||
} else {
|
||||
// Fallback: extract directly from the item
|
||||
@@ -650,6 +651,7 @@ fn parse_sse_response(body: &str) -> Result<ParsedResponse, LlmError> {
|
||||
id: call_id,
|
||||
name,
|
||||
arguments,
|
||||
reasoning: None,
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -727,6 +729,7 @@ fn parse_sse_response(body: &str) -> Result<ParsedResponse, LlmError> {
|
||||
id: state.call_id,
|
||||
name: state.name,
|
||||
arguments,
|
||||
reasoning: None,
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -822,11 +825,13 @@ mod tests {
|
||||
id: "call_1".to_string(),
|
||||
name: "search".to_string(),
|
||||
arguments: serde_json::json!({"query": "test"}),
|
||||
reasoning: None,
|
||||
},
|
||||
ToolCall {
|
||||
id: "call_2".to_string(),
|
||||
name: "read".to_string(),
|
||||
arguments: serde_json::json!({"path": "/tmp"}),
|
||||
reasoning: None,
|
||||
},
|
||||
];
|
||||
let msg =
|
||||
|
||||
@@ -231,6 +231,10 @@ pub struct ToolCall {
|
||||
pub id: String,
|
||||
pub name: String,
|
||||
pub arguments: serde_json::Value,
|
||||
/// Optional reasoning for why this tool was chosen — supplied by the provider
|
||||
/// or derived from the shared response content as a fallback.
|
||||
#[serde(default, skip_serializing_if = "Option::is_none")]
|
||||
pub reasoning: Option<String>,
|
||||
}
|
||||
|
||||
/// Generate a tool-call ID that satisfies all providers.
|
||||
@@ -637,6 +641,7 @@ mod tests {
|
||||
id: "call_1".to_string(),
|
||||
name: "echo".to_string(),
|
||||
arguments: serde_json::json!({}),
|
||||
reasoning: None,
|
||||
};
|
||||
let mut messages = vec![
|
||||
ChatMessage::user("hello"),
|
||||
@@ -680,6 +685,7 @@ mod tests {
|
||||
id: "call_1".to_string(),
|
||||
name: "echo".to_string(),
|
||||
arguments: serde_json::json!({}),
|
||||
reasoning: None,
|
||||
};
|
||||
let mut messages = vec![
|
||||
ChatMessage::user("test"),
|
||||
@@ -705,11 +711,13 @@ mod tests {
|
||||
id: "call_sel_1".to_string(),
|
||||
name: "search".to_string(),
|
||||
arguments: serde_json::json!({"q": "test"}),
|
||||
reasoning: None,
|
||||
};
|
||||
let tc2 = ToolCall {
|
||||
id: "call_sel_2".to_string(),
|
||||
name: "http".to_string(),
|
||||
arguments: serde_json::json!({"url": "https://example.com"}),
|
||||
reasoning: None,
|
||||
};
|
||||
let mut messages = vec![
|
||||
ChatMessage::system("You are a helpful assistant."),
|
||||
|
||||
+85
-12
@@ -525,17 +525,35 @@ impl Reasoning {
|
||||
|
||||
let response = self.llm.complete_with_tools(request).await?;
|
||||
|
||||
let reasoning = response.content.unwrap_or_default();
|
||||
let shared_reasoning = response
|
||||
.content
|
||||
.map(|c| {
|
||||
let pre_truncated = truncate_at_tool_tags(&c);
|
||||
clean_response(&pre_truncated)
|
||||
})
|
||||
.unwrap_or_default();
|
||||
|
||||
let selections: Vec<ToolSelection> = response
|
||||
.tool_calls
|
||||
.into_iter()
|
||||
.map(|tool_call| ToolSelection {
|
||||
tool_name: tool_call.name,
|
||||
parameters: tool_call.arguments,
|
||||
reasoning: reasoning.clone(),
|
||||
alternatives: vec![],
|
||||
tool_call_id: tool_call.id,
|
||||
.map(|tool_call| {
|
||||
// Prefer per-tool reasoning if the provider supplied it,
|
||||
// otherwise fall back to the shared response content.
|
||||
let rationale = tool_call
|
||||
.reasoning
|
||||
.map(|r| {
|
||||
let pre_truncated = truncate_at_tool_tags(&r);
|
||||
clean_response(&pre_truncated)
|
||||
})
|
||||
.filter(|r| !r.trim().is_empty())
|
||||
.unwrap_or_else(|| shared_reasoning.clone());
|
||||
ToolSelection {
|
||||
tool_name: tool_call.name,
|
||||
parameters: tool_call.arguments,
|
||||
reasoning: rationale,
|
||||
alternatives: vec![],
|
||||
tool_call_id: tool_call.id,
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
|
||||
@@ -664,13 +682,36 @@ Respond in JSON format:
|
||||
|
||||
// If there were tool calls, return them for execution
|
||||
if !response.tool_calls.is_empty() {
|
||||
let narrative = response.content.map(|c| {
|
||||
let pre_truncated = truncate_at_tool_tags(&c);
|
||||
clean_response(&pre_truncated)
|
||||
});
|
||||
// Populate per-tool reasoning from the shared narrative when the
|
||||
// provider did not supply per-tool rationale.
|
||||
let tool_calls: Vec<ToolCall> = response
|
||||
.tool_calls
|
||||
.into_iter()
|
||||
.map(|mut tc| {
|
||||
if tc.reasoning.as_ref().is_none_or(|r| r.trim().is_empty()) {
|
||||
tc.reasoning = narrative.as_ref().filter(|n| !n.is_empty()).cloned();
|
||||
} else {
|
||||
// Clean provider-supplied per-tool reasoning the same way
|
||||
// we clean the shared narrative (strip thinking/tool tags).
|
||||
tc.reasoning = tc
|
||||
.reasoning
|
||||
.map(|r| {
|
||||
let pre_truncated = truncate_at_tool_tags(&r);
|
||||
clean_response(&pre_truncated)
|
||||
})
|
||||
.filter(|r| !r.trim().is_empty());
|
||||
}
|
||||
tc
|
||||
})
|
||||
.collect();
|
||||
return Ok(RespondOutput {
|
||||
result: RespondResult::ToolCalls {
|
||||
tool_calls: response.tool_calls,
|
||||
content: response.content.map(|c| {
|
||||
let pre_truncated = truncate_at_tool_tags(&c);
|
||||
clean_response(&pre_truncated)
|
||||
}),
|
||||
tool_calls,
|
||||
content: narrative,
|
||||
},
|
||||
usage,
|
||||
});
|
||||
@@ -1350,6 +1391,7 @@ fn recover_tool_calls_from_content(
|
||||
),
|
||||
name: name.to_string(),
|
||||
arguments,
|
||||
reasoning: None,
|
||||
});
|
||||
continue;
|
||||
}
|
||||
@@ -1364,6 +1406,7 @@ fn recover_tool_calls_from_content(
|
||||
),
|
||||
name: name.to_string(),
|
||||
arguments: serde_json::Value::Object(Default::default()),
|
||||
reasoning: None,
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -1401,6 +1444,7 @@ fn recover_tool_calls_from_content(
|
||||
),
|
||||
name: name.to_string(),
|
||||
arguments,
|
||||
reasoning: None,
|
||||
});
|
||||
remaining = &args_start[bracket_end + 1..];
|
||||
continue;
|
||||
@@ -1412,6 +1456,7 @@ fn recover_tool_calls_from_content(
|
||||
id: super::provider::generate_tool_call_id(calls.len(), RECOVERED_TOOL_CALL_SEED),
|
||||
name: name.to_string(),
|
||||
arguments: serde_json::Value::Object(Default::default()),
|
||||
reasoning: None,
|
||||
});
|
||||
remaining = after_name;
|
||||
}
|
||||
@@ -3145,4 +3190,32 @@ That's my plan."#;
|
||||
"Text <function_call>{}</function_call> middle "
|
||||
);
|
||||
}
|
||||
|
||||
/// Verify that reasoning normalization strips thinking tags and tool tags
|
||||
/// from per-tool reasoning, matching the cleaning applied to shared reasoning.
|
||||
#[test]
|
||||
fn test_reasoning_normalization_strips_thinking_tags() {
|
||||
let raw = "<thinking>Let me consider...</thinking>Search memory for prior context";
|
||||
let pre_truncated = truncate_at_tool_tags(raw);
|
||||
let cleaned = clean_response(&pre_truncated);
|
||||
assert!(!cleaned.contains("<thinking>"));
|
||||
assert!(cleaned.contains("Search memory"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_reasoning_normalization_strips_tool_tags() {
|
||||
let raw = "Calling search <tool_call>{\"name\": \"search\"}";
|
||||
let pre_truncated = truncate_at_tool_tags(raw);
|
||||
let cleaned = clean_response(&pre_truncated);
|
||||
assert!(!cleaned.contains("<tool_call>"));
|
||||
assert!(cleaned.contains("Calling search"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_reasoning_normalization_empty_after_cleaning() {
|
||||
let raw = "<thinking>internal only</thinking>";
|
||||
let pre_truncated = truncate_at_tool_tags(raw);
|
||||
let cleaned = clean_response(&pre_truncated);
|
||||
assert!(cleaned.trim().is_empty());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -490,6 +490,7 @@ fn extract_response(
|
||||
id: tc.id.clone(),
|
||||
name: tc.function.name.clone(),
|
||||
arguments: tc.function.arguments.clone(),
|
||||
reasoning: None,
|
||||
});
|
||||
}
|
||||
// Reasoning and Image variants are not mapped to IronClaw types
|
||||
@@ -880,6 +881,7 @@ mod tests {
|
||||
id: "Xt7mK9pQ2".to_string(),
|
||||
name: "search".to_string(),
|
||||
arguments: serde_json::json!({"query": "test"}),
|
||||
reasoning: None,
|
||||
};
|
||||
let msg = ChatMessage::assistant_with_tool_calls(Some("thinking".to_string()), vec![tc]);
|
||||
let messages = vec![msg];
|
||||
@@ -997,6 +999,7 @@ mod tests {
|
||||
id: "".to_string(),
|
||||
name: "search".to_string(),
|
||||
arguments: serde_json::json!({"query": "test"}),
|
||||
reasoning: None,
|
||||
};
|
||||
let messages = vec![ChatMessage::assistant_with_tool_calls(None, vec![tc])];
|
||||
let (_preamble, history) = convert_messages(&messages);
|
||||
@@ -1028,6 +1031,7 @@ mod tests {
|
||||
id: " ".to_string(),
|
||||
name: "search".to_string(),
|
||||
arguments: serde_json::json!({"query": "test"}),
|
||||
reasoning: None,
|
||||
};
|
||||
let messages = vec![ChatMessage::assistant_with_tool_calls(None, vec![tc])];
|
||||
let (_preamble, history) = convert_messages(&messages);
|
||||
@@ -1061,6 +1065,7 @@ mod tests {
|
||||
id: "".to_string(),
|
||||
name: "search".to_string(),
|
||||
arguments: serde_json::json!({"query": "test"}),
|
||||
reasoning: None,
|
||||
};
|
||||
let assistant_msg = ChatMessage::assistant_with_tool_calls(None, vec![tc]);
|
||||
let tool_result_msg = ChatMessage {
|
||||
@@ -1380,11 +1385,13 @@ mod tests {
|
||||
id: "call_a".to_string(),
|
||||
name: "search".to_string(),
|
||||
arguments: serde_json::json!({"q": "rust"}),
|
||||
reasoning: None,
|
||||
};
|
||||
let tc2 = IronToolCall {
|
||||
id: "call_b".to_string(),
|
||||
name: "fetch".to_string(),
|
||||
arguments: serde_json::json!({"url": "https://example.com"}),
|
||||
reasoning: None,
|
||||
};
|
||||
let assistant = ChatMessage::assistant_with_tool_calls(None, vec![tc1, tc2]);
|
||||
let result_a = ChatMessage::tool_result("call_a", "search", "search results");
|
||||
|
||||
Reference in New Issue
Block a user