mirror of
https://github.com/outbackdingo/optimclaw.git
synced 2026-08-25 14:53:34 +00:00
* feat: Add benchmarking harness for agent evaluation Introduces ironclaw-bench, a Rust-native benchmarking crate that drives the real agent loop headlessly. Supports standard benchmarks (GAIA, Tau-bench, SWE-bench Pro) and custom JSONL task sets with parallel execution, resume support, and incremental JSONL output. Key components: - BenchChannel: headless Channel impl with auto-approval and response capture - InstrumentedLlm: LlmProvider wrapper recording per-call token/cost metrics - BenchRunner: task orchestration with parallel execution and JSONL resume - Scoring utilities: exact match, contains, regex (all with normalization) - CLI: run, results, compare, list subcommands via clap - Four suite adapters: custom, gaia, tau_bench, swe_bench Also fixes a pre-existing missing SseEvent::ToolResult match arm in the web gateway and adds FinishReason to the LLM module's public re-exports. Co-Authored-By: Claude Opus 4.6 <[email protected]> * feat: Add spot benchmark suite for end-to-end agent verification Adds a "spot" suite with 13 scenarios across 4 categories (smoke, tool use, multi-tool chaining, robustness) using multi-criterion assertions instead of simple text matching. Also adds an `error` field to TaskSubmission so suites can hard-fail on agent errors. Co-Authored-By: Claude Opus 4.6 <[email protected]> * fix: address audit findings in benchmarks crate - Fix O(n²) scoring loop by indexing tasks in a HashMap (was re-parsing JSONL per result) - Add UTF-8-safe truncation to prevent panic on multi-byte chars in channel capture - Wire setup_task/teardown_task into both sequential and parallel runner paths - Convert BenchRunner.suite from Box to Arc for parallel task setup/teardown - Add tracing::warn for placeholder scores in custom, swe_bench, tau_bench adapters - Add spot suite to CLI help text - Add doc comment clarifying tools_used HashSet behavior in SpotAssertions - Reorder match arms in create_suite to match KNOWN_SUITES alphabetical order Co-Authored-By: Claude Opus 4.6 <[email protected]> * fix: rewrite tasks.jsonl with scored results after scoring The JSONL file was only written during execution (pre-scoring), so the `results` command showed "pending" scores even after scoring completed. Now the runner rewrites the JSONL with final scored results, keeping task-level and aggregate data consistent. Co-Authored-By: Claude Opus 4.6 <[email protected]> * feat: prefix benchmark runs with model name and commit hash Run logs and results table now show the base model and short git commit hash, making it easy to correlate results with code versions. The commit hash is also persisted in run.json for historical tracking. Co-Authored-By: Claude Opus 4.6 <[email protected]> * feat: add 8 memory benchmark scenarios to spot suite Tests save-and-recall workflows using file tools: - daily tasks, reminders, meeting notes, append logs - detail extraction, todo priorities, multi-file ops - context updates (write-read-rewrite-verify) Total spot scenarios: 13 -> 21 Co-Authored-By: Claude Opus 4.6 <[email protected]> * chore: fmt channel.rs and gitignore bench-results Co-Authored-By: Claude Opus 4.6 <[email protected]> * fix: address critical and high findings from PR review - Fix race condition: parallel mode now writes JSONL after all tasks complete instead of concurrent unsynchronized appends - Fix UTF-8 panic: use .chars().take(25) instead of byte slicing on task_id which could panic on multi-byte characters - Remove dead code: max_iterations (parsed but never used), tool_whitelist() (declared but never called), MatrixEntry.tools (declared but never applied) - Eliminate double load_tasks(): cache task list on first load and reuse the index for scoring instead of re-reading from disk Co-Authored-By: Claude Opus 4.6 <[email protected]> * fix: relax smoke-greeting assertion to not demand parrot greeting The LLM often introduces itself without echoing "hello" back. Use a regex that accepts any reasonable self-introduction (hello, hi, hey, assistant, agent, help) instead of demanding a specific word. Co-Authored-By: Claude Opus 4.6 <[email protected]> * feat: 100% spot baseline (GPT-5.2 @ 2c43b83, 21/21 pass) Relax two brittle assertions: - smoke-greeting: use regex for any reasonable self-intro instead of demanding the model parrot "hello" - memory-update-context: drop response_not_contains PST since the model correctly says "not PST" which triggers the literal check - memory-multifile: lower min_tool_calls from 4 to 3, the model can batch two writes in one LLM turn Baseline results committed to benchmarks/baselines/ for regression tracking. Local runs stay in bench-results/ (gitignored). Results: 100.0% pass, 1.000 avg, $0.31 cost, 111s wall time Co-Authored-By: Claude Opus 4.6 <[email protected]> * fix: address remaining PR review comments - Replace .expect("semaphore closed") with proper error handling - Derive PartialEq on BenchScore for cleaner test assertions - Use ToPrimitive::to_f64() instead of string roundtrip in estimated_cost() - Validate SWE-bench inputs: task_id (path traversal), repo (owner/repo format), base_commit (valid git ref) with 5 new tests - Skip "pending" (unscored) entries during resume so they get re-executed - Use run.json mtime for find_latest_run (falls back to tasks.jsonl, then dir) - Move additional_tools() outside parallel loop to share Arc<[Tool]> across tasks - Add doc comments documenting known limitations (single-turn, resources, conversation) Co-Authored-By: Claude Opus 4.6 <[email protected]> * fix: reject absolute paths in SWE-bench and validate matrix config - is_safe_path_component now rejects paths starting with '/' - BenchConfig::from_file validates matrix is non-empty - Added tests for both validations Co-Authored-By: Claude Opus 4.6 <[email protected]> * fix: fail tasks on setup_task error and compute git hash once - setup_task failure now records an error TaskResult instead of continuing to run the task (both sequential and parallel paths) - git_short_hash() computed once per run instead of twice Co-Authored-By: Claude Opus 4.6 <[email protected]> --------- Co-authored-by: Claude Opus 4.6 <[email protected]>
210 lines
5.7 KiB
TOML
210 lines
5.7 KiB
TOML
[workspace]
|
|
members = [".", "benchmarks"]
|
|
exclude = [
|
|
"channels-src/telegram",
|
|
"channels-src/slack",
|
|
"channels-src/whatsapp",
|
|
"tools-src/gmail",
|
|
]
|
|
|
|
[package]
|
|
name = "ironclaw"
|
|
version = "0.5.0"
|
|
edition = "2024"
|
|
rust-version = "1.92"
|
|
description = "Secure personal AI assistant that protects your data and expands its capabilities on the fly"
|
|
authors = ["NEAR AI <[email protected]>"]
|
|
license = "MIT OR Apache-2.0"
|
|
homepage = "https://github.com/nearai/ironclaw"
|
|
repository = "https://github.com/nearai/ironclaw"
|
|
|
|
[package.metadata.wix]
|
|
upgrade-guid = "D0156E61-BA37-451E-8AB9-1A2ECCCFA48F"
|
|
path-guid = "F90B6EA6-87F7-499B-BB19-CF55DE1EB339"
|
|
license = false
|
|
eula = false
|
|
|
|
[dependencies]
|
|
# Async runtime
|
|
tokio = { version = "1", features = ["full"] }
|
|
tokio-stream = { version = "0.1", features = ["sync"] }
|
|
futures = "0.3"
|
|
|
|
# HTTP client
|
|
reqwest = { version = "0.12", default-features = false, features = ["json", "rustls-tls-native-roots", "stream"] }
|
|
|
|
# Serialization
|
|
serde = { version = "1", features = ["derive"] }
|
|
serde_json = "1"
|
|
|
|
# Database - PostgreSQL (default, feature-gated)
|
|
deadpool-postgres = { version = "0.14", optional = true }
|
|
tokio-postgres = { version = "0.7", features = ["with-uuid-1", "with-chrono-0_4", "with-serde_json-1"], optional = true }
|
|
postgres-types = { version = "0.2", features = ["with-serde_json-1"], optional = true }
|
|
refinery = { version = "0.8", features = ["tokio-postgres"], optional = true }
|
|
|
|
# Database - libSQL/Turso (optional embedded database)
|
|
libsql = { version = "0.6", optional = true, default-features = false, features = ["core", "replication"] }
|
|
|
|
# Error handling
|
|
thiserror = "2"
|
|
anyhow = "1"
|
|
|
|
# Logging
|
|
tracing = "0.1"
|
|
tracing-subscriber = { version = "0.3", features = ["env-filter", "json"] }
|
|
|
|
# Configuration
|
|
dotenvy = "0.15"
|
|
toml = "0.8"
|
|
|
|
# Core types
|
|
uuid = { version = "1", features = ["v4", "serde"] }
|
|
chrono = { version = "0.4", features = ["serde"] }
|
|
rust_decimal = { version = "1", features = ["serde", "serde-with-str", "maths"] }
|
|
rust_decimal_macros = "1"
|
|
|
|
# Async traits
|
|
async-trait = "0.1"
|
|
|
|
# CLI
|
|
clap = { version = "4", features = ["derive", "env"] }
|
|
|
|
# Terminal
|
|
crossterm = "0.28"
|
|
rustyline = { version = "17", features = ["derive", "with-file-history"] }
|
|
termimad = "0.34"
|
|
|
|
# Channel integrations
|
|
axum = { version = "0.8", features = ["ws"] }
|
|
tower = "0.5"
|
|
tower-http = { version = "0.6", features = ["trace", "cors"] }
|
|
|
|
# Cron scheduling for routines
|
|
cron = "0.13"
|
|
|
|
# Safety/sanitization
|
|
regex = "1"
|
|
aho-corasick = "1"
|
|
|
|
# Filesystem paths
|
|
dirs = "6"
|
|
fs4 = "0.6"
|
|
|
|
# Secrecy for sensitive values
|
|
secrecy = { version = "0.10", features = ["serde"] }
|
|
|
|
# URL parsing and encoding
|
|
url = "2"
|
|
urlencoding = "2"
|
|
|
|
# Open URLs in browser
|
|
open = "5"
|
|
|
|
# Vector embeddings for semantic search
|
|
# The postgres feature provides ToSql/FromSql for postgres-types (shared by tokio-postgres)
|
|
pgvector = { version = "0.4", features = ["postgres"], optional = true }
|
|
|
|
# WASM sandbox for untrusted tool execution
|
|
wasmtime = { version = "28", features = ["component-model"] }
|
|
wasmtime-wasi = "28" # WASI support for component model
|
|
wasmparser = "0.220" # WASM binary parsing for validation
|
|
|
|
# Cryptography for secrets management
|
|
aes-gcm = "0.10"
|
|
hkdf = "0.12"
|
|
sha2 = "0.10"
|
|
blake3 = "1"
|
|
rand = "0.8"
|
|
subtle = "2" # Constant-time comparisons for token validation
|
|
|
|
# Multi-provider LLM support
|
|
rig-core = "0.30"
|
|
|
|
# Docker sandbox
|
|
bollard = "0.18"
|
|
|
|
# HTTP proxy for sandboxed network access
|
|
hyper = { version = "1.5", features = ["server", "http1", "http2"] }
|
|
hyper-util = { version = "0.1", features = ["server", "tokio", "http1", "http2"] }
|
|
http-body-util = "0.1"
|
|
bytes = "1"
|
|
base64 = "0.22.1"
|
|
mime_guess = "2.0.5"
|
|
|
|
# macOS keychain
|
|
[target.'cfg(target_os = "macos")'.dependencies]
|
|
security-framework = "3"
|
|
|
|
# Linux secret-service (GNOME Keyring, KWallet)
|
|
[target.'cfg(target_os = "linux")'.dependencies]
|
|
secret-service = { version = "4", features = ["rt-tokio-crypto-rust"] }
|
|
zbus = "4"
|
|
|
|
[dev-dependencies]
|
|
tokio-test = "0.4"
|
|
tokio-tungstenite = "0.26"
|
|
testcontainers-modules = { version = "0.11", features = ["postgres"] }
|
|
pretty_assertions = "1"
|
|
tempfile = "3"
|
|
|
|
[features]
|
|
default = ["postgres", "libsql"]
|
|
postgres = [
|
|
"dep:deadpool-postgres",
|
|
"dep:tokio-postgres",
|
|
"dep:postgres-types",
|
|
"dep:refinery",
|
|
"dep:pgvector",
|
|
"rust_decimal/db-tokio-postgres",
|
|
]
|
|
libsql = ["dep:libsql"]
|
|
integration = []
|
|
|
|
[[example]]
|
|
name = "test_heartbeat"
|
|
required-features = ["postgres"]
|
|
|
|
# The profile that 'cargo dist' will build with
|
|
[profile.dist]
|
|
inherits = "release"
|
|
lto = "thin"
|
|
|
|
# Config for 'dist'
|
|
[workspace.metadata.dist]
|
|
# The preferred dist version to use in CI (Cargo.toml SemVer syntax)
|
|
cargo-dist-version = "0.30.3"
|
|
# CI backends to support
|
|
ci = "github"
|
|
# The installers to generate for each app
|
|
installers = ["shell", "powershell", "npm", "msi"]
|
|
# Publish jobs to run in CI
|
|
publish-jobs = []
|
|
# Target platforms to build apps for (Rust target-triple syntax)
|
|
targets = [
|
|
"aarch64-apple-darwin",
|
|
"aarch64-unknown-linux-gnu",
|
|
"x86_64-apple-darwin",
|
|
"x86_64-unknown-linux-gnu",
|
|
"x86_64-pc-windows-msvc",
|
|
]
|
|
# The archive format to use for windows builds (defaults .zip)
|
|
windows-archive = ".tar.gz"
|
|
# The archive format to use for non-windows builds (defaults .tar.xz)
|
|
unix-archive = ".tar.gz"
|
|
# Which actions to run on pull requests
|
|
pr-run-mode = "skip"
|
|
# Path that installers should place binaries in
|
|
install-path = "CARGO_HOME"
|
|
# Whether to install an updater program
|
|
install-updater = true
|
|
# Cache intermediate build artifacts to speed up the release pipelines
|
|
cache-builds = true
|
|
|
|
[workspace.metadata.dist.github-custom-runners]
|
|
aarch64-unknown-linux-gnu = "ubuntu-24.04-arm"
|
|
x86_64-unknown-linux-gnu = "ubuntu-22.04"
|
|
x86_64-pc-windows-msvc = "windows-2022"
|
|
x86_64-apple-darwin = "macos-15-intel"
|
|
aarch64-apple-darwin = "macos-14"
|