diff --git a/Cargo.lock b/Cargo.lock index 854cd017f..03d9322bc 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -11311,6 +11311,7 @@ dependencies = [ "serde_json", "sha3", "smallvec 1.15.2", + "tempfile", "thiserror 2.0.18", "tokio", "tracing", @@ -11400,6 +11401,7 @@ dependencies = [ "async-trait", "criterion 0.5.1", "glob", + "libc", "mockall", "rvagent-backends", "rvagent-core", diff --git a/crates/ruvector-core/Cargo.toml b/crates/ruvector-core/Cargo.toml index 144c706fe..e78f5befa 100644 --- a/crates/ruvector-core/Cargo.toml +++ b/crates/ruvector-core/Cargo.toml @@ -317,3 +317,7 @@ overly_complex_bool_expr = { level = "allow", priority = 1 } zombie_processes = { level = "allow", priority = 1 } repeat_vec_with_capacity = { level = "allow", priority = 1 } missing_transmute_annotations = { level = "allow", priority = 1 } + +[[bench]] +name = "pdx_vs_rowmajor" +harness = false diff --git a/crates/ruvector-core/benches/pdx_vs_rowmajor.rs b/crates/ruvector-core/benches/pdx_vs_rowmajor.rs new file mode 100644 index 000000000..77f2ce6a3 --- /dev/null +++ b/crates/ruvector-core/benches/pdx_vs_rowmajor.rs @@ -0,0 +1,59 @@ +//! Row-major batch distance vs PDX vertical layout (ADR-279 §5.1). +//! +//! Both compute the same thing over the same data. The only difference is +//! memory layout and whether a horizontal reduction happens per vector. + +use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion, Throughput}; +use ruvector_core::pdx::PdxIndex; +use ruvector_core::simd_intrinsics::batch_euclidean; + +fn corpus(n: usize, dim: usize) -> Vec> { + (0..n) + .map(|i| { + (0..dim) + .map(|d| ((i * 31 + d * 17) % 101) as f32 / 101.0) + .collect() + }) + .collect() +} + +fn bench(c: &mut Criterion) { + // 1536 is the OpenAI embedding width; 768 covers most sentence encoders. + // (n, dim) pairs: the first three are cache-resident, the rest stream. + for (n, dim) in [ + (256usize, 768usize), + (512, 768), + (1024, 768), + (4096, 768), + (4096, 1536), + ] { + let vecs = corpus(n, dim); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let query: Vec = (0..dim).map(|d| (d % 7) as f32 / 7.0).collect(); + + let index = PdxIndex::from_rows(&refs); + let mut out = vec![0.0f32; n]; + + let mut group = c.benchmark_group(format!("batch_euclidean_{n}x{dim}")); + // Bytes of vector data scanned per query — makes the two comparable + // in bandwidth terms rather than just wall time. + group.throughput(Throughput::Bytes((n * dim * 4) as u64)); + + group.bench_function(BenchmarkId::new("row_major", dim), |b| { + b.iter(|| { + batch_euclidean(black_box(&query), black_box(&refs), black_box(&mut out)); + }) + }); + + group.bench_function(BenchmarkId::new("pdx_vertical", dim), |b| { + b.iter(|| { + index.euclidean_sq_into(black_box(&query), black_box(&mut out)); + }) + }); + + group.finish(); + } +} + +criterion_group!(benches, bench); +criterion_main!(benches); diff --git a/crates/ruvector-core/src/lib.rs b/crates/ruvector-core/src/lib.rs index 5663b9f96..ea301f08c 100644 --- a/crates/ruvector-core/src/lib.rs +++ b/crates/ruvector-core/src/lib.rs @@ -39,6 +39,7 @@ pub mod distance; pub mod embeddings; pub mod error; pub mod index; +pub mod pdx; pub mod quantization; // Storage backends - conditional compilation based on features diff --git a/crates/ruvector-core/src/pdx.rs b/crates/ruvector-core/src/pdx.rs new file mode 100644 index 000000000..9d7d5334e --- /dev/null +++ b/crates/ruvector-core/src/pdx.rs @@ -0,0 +1,394 @@ +//! PDX-style vertical (dimension-major) layout for bulk distance computation. +//! +//! Implements the layout change measured in *PDX: A Data Layout for Vector +//! Similarity Search* (SIGMOD '25), which beat hand-written SIMD kernels in +//! SimSIMD and FAISS by ~2.0× on average — using **scalar** code — purely by +//! reorganizing memory. See ADR-279 §5.1: layout is the lever, not intrinsics. +//! +//! # Why the row-major batch loop leaves throughput on the table +//! +//! The conventional layout stores each vector contiguously and computes one +//! distance at a time: +//! +//! ```text +//! v0: [d0 d1 d2 … dD] → SIMD across dimensions → horizontal sum → result[0] +//! v1: [d0 d1 d2 … dD] → SIMD across dimensions → horizontal sum → result[1] +//! ``` +//! +//! Two costs are structural, not tuning problems: +//! +//! 1. **A horizontal reduction per vector.** Summing a SIMD accumulator into a +//! scalar is a serial dependency chain (`_mm512_reduce_add_ps` is several +//! dependent shuffles and adds) and it happens once per vector. +//! 2. **A tail per vector.** Any dimension count not a multiple of the vector +//! width runs a scalar remainder loop, once per vector. +//! +//! # The vertical layout +//! +//! Store a *block* of `LANES` vectors dimension-major: +//! +//! ```text +//! dim 0: [v0 v1 v2 … v15] +//! dim 1: [v0 v1 v2 … v15] +//! … +//! ``` +//! +//! Now one SIMD register holds *one dimension of sixteen different vectors*. +//! Iterating dimensions accumulates sixteen independent distances in parallel: +//! +//! - **No horizontal reduction at all** — lane `i` of the accumulator *is* +//! distance `i` when the loop ends. +//! - **No per-vector tail** — the only remainder is the final partial block. +//! - **Sequential access** across the whole block. +//! +//! The accumulator also has no loop-carried dependency between lanes, so the +//! CPU can keep several FMAs in flight. +//! +//! # Portability +//! +//! Written against `f32` chunks that LLVM autovectorizes, with an explicit +//! AVX-512 path where available. Per ADR-279 this is deliberately *not* C: the +//! same source compiles for x86-64, aarch64, and wasm32. + +use std::fmt; + +/// Vectors per block. 16 f32 fills one AVX-512 register; on narrower ISAs the +/// compiler splits it into multiple registers, which still vectorizes cleanly. +pub const LANES: usize = 16; + +/// A block of up to [`LANES`] vectors stored dimension-major. +#[derive(Clone, PartialEq)] +pub struct PdxBlock { + /// `dim * LANES + lane`. Unused lanes hold zeros. + data: Vec, + dim: usize, + /// How many lanes carry real vectors (≤ `LANES`). + len: usize, +} + +impl fmt::Debug for PdxBlock { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("PdxBlock") + .field("dim", &self.dim) + .field("len", &self.len) + .finish_non_exhaustive() + } +} + +impl PdxBlock { + /// Transpose up to [`LANES`] row-major vectors into one vertical block. + /// + /// # Panics + /// If `vectors` is longer than [`LANES`], or any vector's length differs + /// from `dim`. Both are programming errors rather than input errors — a + /// ragged block would silently produce wrong distances. + pub fn from_rows(vectors: &[&[f32]], dim: usize) -> Self { + assert!( + vectors.len() <= LANES, + "block holds at most {LANES} vectors, got {}", + vectors.len() + ); + for (i, v) in vectors.iter().enumerate() { + assert_eq!( + v.len(), + dim, + "vector {i} has length {}, expected {dim}", + v.len() + ); + } + + let mut data = vec![0.0f32; dim * LANES]; + for (lane, v) in vectors.iter().enumerate() { + for (d, &value) in v.iter().enumerate() { + data[d * LANES + lane] = value; + } + } + Self { + data, + dim, + len: vectors.len(), + } + } + + pub fn dim(&self) -> usize { + self.dim + } + + pub fn len(&self) -> usize { + self.len + } + + pub fn is_empty(&self) -> bool { + self.len == 0 + } + + /// Squared euclidean distance from `query` to every vector in the block. + /// + /// Writes `self.len()` results. Padding lanes are computed but discarded — + /// branchless, and they hold zeros so they cannot fault or produce NaN. + /// + /// # Panics + /// If `query.len() != self.dim()` or `out.len() < self.len()`. + pub fn euclidean_sq_into(&self, query: &[f32], out: &mut [f32]) { + assert_eq!(query.len(), self.dim, "query dimension mismatch"); + assert!(out.len() >= self.len, "output buffer too small"); + + let mut acc = [0.0f32; LANES]; + + for (d, &q) in query.iter().enumerate() { + let row = &self.data[d * LANES..d * LANES + LANES]; + // Fixed-size slice so LLVM knows the trip count and emits a single + // vector op per dimension with no remainder branch. + for lane in 0..LANES { + let diff = row[lane] - q; + acc[lane] = diff.mul_add(diff, acc[lane]); + } + } + + out[..self.len].copy_from_slice(&acc[..self.len]); + } + + /// Dot product from `query` to every vector in the block. + pub fn dot_into(&self, query: &[f32], out: &mut [f32]) { + assert_eq!(query.len(), self.dim, "query dimension mismatch"); + assert!(out.len() >= self.len, "output buffer too small"); + + let mut acc = [0.0f32; LANES]; + + for (d, &q) in query.iter().enumerate() { + let row = &self.data[d * LANES..d * LANES + LANES]; + for lane in 0..LANES { + acc[lane] = row[lane].mul_add(q, acc[lane]); + } + } + + out[..self.len].copy_from_slice(&acc[..self.len]); + } +} + +/// A full vector set in PDX layout: a sequence of vertical blocks. +#[derive(Debug, Clone)] +pub struct PdxIndex { + blocks: Vec, + dim: usize, + len: usize, +} + +impl PdxIndex { + /// Build from row-major vectors. + /// + /// # Panics + /// If `vectors` is empty, or any vector's length differs from the first. + pub fn from_rows(vectors: &[&[f32]]) -> Self { + assert!( + !vectors.is_empty(), + "cannot build a PdxIndex from no vectors" + ); + let dim = vectors[0].len(); + + let blocks = vectors + .chunks(LANES) + .map(|chunk| PdxBlock::from_rows(chunk, dim)) + .collect(); + + Self { + blocks, + dim, + len: vectors.len(), + } + } + + pub fn dim(&self) -> usize { + self.dim + } + + pub fn len(&self) -> usize { + self.len + } + + pub fn is_empty(&self) -> bool { + self.len == 0 + } + + pub fn block_count(&self) -> usize { + self.blocks.len() + } + + /// Squared euclidean distance from `query` to every indexed vector. + /// + /// This is the bulk API the row-major batch path lacks — the second lever + /// in ADR-279 §5.1, worth 1.85–3.04× independently of layout, because it + /// amortizes dispatch and lets the prefetcher see a sequential stream. + /// + /// # Panics + /// If `out.len() < self.len()`. + pub fn euclidean_sq_into(&self, query: &[f32], out: &mut [f32]) { + assert!(out.len() >= self.len, "output buffer too small"); + let mut written = 0; + for block in &self.blocks { + block.euclidean_sq_into(query, &mut out[written..]); + written += block.len(); + } + } + + /// Dot product from `query` to every indexed vector. + pub fn dot_into(&self, query: &[f32], out: &mut [f32]) { + assert!(out.len() >= self.len, "output buffer too small"); + let mut written = 0; + for block in &self.blocks { + block.dot_into(query, &mut out[written..]); + written += block.len(); + } + } + + /// Convenience allocating wrapper. + pub fn euclidean_sq(&self, query: &[f32]) -> Vec { + let mut out = vec![0.0; self.len]; + self.euclidean_sq_into(query, &mut out); + out + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Reference implementation. Deliberately naive — its job is to be + /// obviously correct, not fast. + fn euclidean_sq_ref(a: &[f32], b: &[f32]) -> f32 { + a.iter().zip(b).map(|(x, y)| (x - y) * (x - y)).sum() + } + + fn dot_ref(a: &[f32], b: &[f32]) -> f32 { + a.iter().zip(b).map(|(x, y)| x * y).sum() + } + + fn make(n: usize, dim: usize) -> Vec> { + (0..n) + .map(|i| { + (0..dim) + .map(|d| ((i * 31 + d * 17) % 101) as f32 / 101.0) + .collect() + }) + .collect() + } + + #[test] + fn block_matches_reference_euclidean() { + let dim = 128; + let vecs = make(LANES, dim); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let block = PdxBlock::from_rows(&refs, dim); + let query: Vec = (0..dim).map(|d| (d % 7) as f32 / 7.0).collect(); + + let mut got = vec![0.0; LANES]; + block.euclidean_sq_into(&query, &mut got); + + for (i, v) in vecs.iter().enumerate() { + let want = euclidean_sq_ref(&query, v); + assert!( + (got[i] - want).abs() < 1e-4, + "lane {i}: got {}, want {want}", + got[i] + ); + } + } + + #[test] + fn block_matches_reference_dot() { + let dim = 96; + let vecs = make(LANES, dim); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let block = PdxBlock::from_rows(&refs, dim); + let query: Vec = (0..dim).map(|d| (d % 5) as f32 / 5.0).collect(); + + let mut got = vec![0.0; LANES]; + block.dot_into(&query, &mut got); + + for (i, v) in vecs.iter().enumerate() { + let want = dot_ref(&query, v); + assert!((got[i] - want).abs() < 1e-4, "lane {i}"); + } + } + + #[test] + fn partial_block_ignores_padding_lanes() { + // The case most likely to produce silently wrong results: a block that + // is not full. Padding lanes must not appear in the output. + let dim = 64; + let vecs = make(5, dim); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let block = PdxBlock::from_rows(&refs, dim); + assert_eq!(block.len(), 5); + + let query = vec![0.5f32; dim]; + let mut got = vec![f32::NAN; 5]; + block.euclidean_sq_into(&query, &mut got); + + for (i, v) in vecs.iter().enumerate() { + assert!((got[i] - euclidean_sq_ref(&query, v)).abs() < 1e-4); + } + } + + #[test] + fn index_matches_reference_across_block_boundaries() { + // 50 vectors = 3 full blocks + a partial, so boundary handling is + // exercised rather than assumed. + let dim = 129; // deliberately not a multiple of any vector width + let vecs = make(50, dim); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let index = PdxIndex::from_rows(&refs); + assert_eq!(index.len(), 50); + assert_eq!(index.block_count(), 4); + + let query: Vec = (0..dim).map(|d| (d % 11) as f32 / 11.0).collect(); + let got = index.euclidean_sq(&query); + + assert_eq!(got.len(), 50); + for (i, v) in vecs.iter().enumerate() { + let want = euclidean_sq_ref(&query, v); + assert!( + (got[i] - want).abs() < 1e-3, + "vector {i}: got {}, want {want}", + got[i] + ); + } + } + + #[test] + fn single_vector_index_works() { + let dim = 32; + let v: Vec = (0..dim).map(|d| d as f32).collect(); + let index = PdxIndex::from_rows(&[v.as_slice()]); + let got = index.euclidean_sq(&v); + assert_eq!(got.len(), 1); + assert!(got[0].abs() < 1e-5, "distance to self must be zero"); + } + + #[test] + #[should_panic(expected = "expected 4")] + fn ragged_block_is_rejected() { + let a = [1.0f32, 2.0, 3.0, 4.0]; + let b = [1.0f32, 2.0]; + PdxBlock::from_rows(&[&a[..], &b[..]], 4); + } + + #[test] + #[should_panic(expected = "at most")] + fn oversized_block_is_rejected() { + let v = vec![0.0f32; 8]; + let refs: Vec<&[f32]> = (0..LANES + 1).map(|_| v.as_slice()).collect(); + PdxBlock::from_rows(&refs, 8); + } + + #[test] + #[should_panic(expected = "output buffer too small")] + fn undersized_output_is_rejected() { + let dim = 16; + let vecs = make(20, dim); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let index = PdxIndex::from_rows(&refs); + let mut out = vec![0.0; 5]; + index.euclidean_sq_into(&vec![0.0; dim], &mut out); + } +} diff --git a/crates/rvAgent/README.md b/crates/rvAgent/README.md index fa06a6302..b85bf5cd5 100644 --- a/crates/rvAgent/README.md +++ b/crates/rvAgent/README.md @@ -50,8 +50,7 @@ let subagent_state = state.clone(); // No deep copy! | Feature | What It Does | Why It Matters | |---------|--------------|----------------| | **O(1) State Cloning** | Clone agent state instantly via Arc | Spawn subagents without copying gigabytes of context | -| **Parallel Tool Execution** | Run multiple tools simultaneously | 5-10x faster than sequential execution | -| **HNSW Semantic Search** | O(log n) memory retrieval | Find relevant context in millions of entries | +| **Parallel Tool Execution** | Bounded-concurrency JoinSet execution | Wall clock = slowest tool, not the sum; failures isolated per call | | **Single-Allocation Formatting** | Pre-calculated output buffers | No memory fragmentation under load | ### 🔒 Security @@ -120,17 +119,19 @@ How does rvAgent compare to other agent frameworks? ## Architecture -rvAgent is organized as 8 crates within the RuVector workspace: +rvAgent is organized as 10 crates within the RuVector workspace: ``` rvAgent/ - rvagent-core Core types, COW state, AGI containers, session encryption - rvagent-backends Backend protocol trait + sandbox security contracts - rvagent-middleware Middleware trait + 14 middleware implementations (incl. SONA, HNSW) - rvagent-tools Tool trait + 8 built-in tools (enum dispatch) + rvagent-core Core types, agent graph, COW state, AGI containers, session encryption + rvagent-backends Backend protocol trait + sandbox security + Anthropic/Gemini clients + rvagent-middleware Middleware trait + async pipeline + 14 middleware implementations + rvagent-tools Tool trait + 9 built-in tools (enum dispatch) rvagent-subagents SubAgent spec, CRDT merge, result validation, orchestration rvagent-cli Terminal coding agent (ratatui TUI) rvagent-acp Agent Communication Protocol server (axum) with auth + rvagent-mcp MCP server/client (JSON-RPC 2.0, stdio + SSE transports) + rvagent-a2a Agent2Agent peer protocol (signed cards, budgets, routing) rvagent-wasm WASM bindings for browser/Node.js ``` @@ -273,8 +274,7 @@ rvAgent solves these with Rust's zero-cost abstractions. |-----------|---------|-------------------|---------| | State cloning | <1μs (O(1)) | ~10ms (deep copy) | 10,000x | | Tool dispatch | No overhead (enum) | ~1ms (vtable lookup) | Direct | -| Parallel tools | True multi-threaded | Async (still serial) | Linear scaling | -| Memory search | O(log n) via HNSW | O(n) linear scan | 100-1000x on large sets | +| Parallel tools | Spawned tasks, bounded concurrency | Async (still serial) | Wall clock = slowest tool | ### Key Optimizations @@ -297,11 +297,11 @@ tools: ["read_file", "grep", "execute", "read_file", "glob"] let formatted = format_content_with_line_numbers(content); ``` -**HNSW Semantic Search** — Find relevant memories in massive datasets -```rust -// O(log n) retrieval instead of scanning everything -let relevant = hnsw.search("authentication bug", top_k=5); -``` +**Memory retrieval (experimental)** — The `hnsw` middleware ships a simplified +in-process index with a hash-based embedding placeholder. It is NOT semantic +search yet: integration with real RuVector embeddings is tracked in the +Hermes-class harness roadmap (`docs/research/rvagent-hermes-harness/`), and +no retrieval performance claims are made until it lands. ## Advanced Features diff --git a/crates/rvAgent/rvagent-acp/src/agent.rs b/crates/rvAgent/rvagent-acp/src/agent.rs index f9665361c..a96ffa55c 100644 --- a/crates/rvAgent/rvagent-acp/src/agent.rs +++ b/crates/rvAgent/rvagent-acp/src/agent.rs @@ -13,7 +13,7 @@ use rvagent_core::config::RvAgentConfig; use rvagent_core::error::Result as CoreResult; use rvagent_core::graph::{AgentGraph, GraphConfig, ToolExecutor}; use rvagent_core::messages::{Message, ToolCall}; -use rvagent_core::models::ChatModel; +use rvagent_core::models::{ChatModel, ToolDefinition}; use rvagent_core::state::AgentState; use crate::types::{ @@ -77,7 +77,11 @@ struct StubModel; #[async_trait] impl ChatModel for StubModel { - async fn complete(&self, messages: &[Message]) -> CoreResult { + async fn complete( + &self, + messages: &[Message], + _tools: &[ToolDefinition], + ) -> CoreResult { // Find the last human message and produce an intelligent echo. let user_text = messages .iter() @@ -95,8 +99,12 @@ impl ChatModel for StubModel { Ok(Message::ai(response)) } - async fn stream(&self, messages: &[Message]) -> CoreResult> { - let msg = self.complete(messages).await?; + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> CoreResult> { + let msg = self.complete(messages, tools).await?; Ok(vec![msg]) } } @@ -186,20 +194,48 @@ impl AcpAgent { let user_msg = Message::human(&user_text); - // Run the prompt through an AgentGraph with a stub model. + // Run the prompt through an AgentGraph with a stub model wrapped in + // the middleware pipeline (P0.3 wiring). // // In production, the model would be resolved from `self.config` - // and real tools/middleware would be wired in. The stub model - // allows the server to run without an API key. + // and real tools would be wired in. The stub model allows the + // server to run without an API key. let graph_config = GraphConfig { max_iterations: 10, parallel_tools: false, + ..GraphConfig::default() }; - let graph = AgentGraph::with_config(StubModel, AcpToolExecutor, graph_config); + + // Resolve the configured middleware names (an unknown name is an + // error); an empty configuration gets the default pipeline. + let pipeline_config = rvagent_middleware::PipelineConfig::default(); + let pipeline = if self.config.middleware.is_empty() { + rvagent_middleware::build_default_pipeline(&pipeline_config) + } else { + let names: Vec<&str> = self + .config + .middleware + .iter() + .map(|m| m.name.as_str()) + .collect(); + rvagent_middleware::build_pipeline_from_names(&names, &pipeline_config) + .map_err(|e| e.to_string())? + }; + let pipeline = Arc::new(pipeline); let mut agent_state = AgentState::new(); agent_state.push_message(user_msg.clone()); + // Run before_agent hooks over the initial state. + let mw_runtime = rvagent_middleware::Runtime::new(); + let run_config = rvagent_middleware::RunnableConfig::default(); + pipeline + .run_before_agent(&mut agent_state, &mw_runtime, &run_config) + .await; + + let model = rvagent_middleware::PipelineModel::new(StubModel, Arc::clone(&pipeline)); + let graph = AgentGraph::with_config(model, AcpToolExecutor, graph_config); + let final_state = graph .run(agent_state) .await diff --git a/crates/rvAgent/rvagent-backends/src/anthropic.rs b/crates/rvAgent/rvagent-backends/src/anthropic.rs index 43f502e93..0740da6f0 100644 --- a/crates/rvAgent/rvagent-backends/src/anthropic.rs +++ b/crates/rvAgent/rvagent-backends/src/anthropic.rs @@ -13,7 +13,7 @@ use tracing::{debug, warn}; use rvagent_core::error::{Result, RvAgentError}; use rvagent_core::messages::{AiMessage, Message, ToolCall}; -use rvagent_core::models::{ApiKeySource, ChatModel, ModelConfig}; +use rvagent_core::models::{ApiKeySource, ChatModel, ModelConfig, ToolDefinition}; // --------------------------------------------------------------------------- // Constants @@ -65,6 +65,24 @@ enum ContentBlock { }, } +/// A tool definition in the Anthropic Messages API format. +#[derive(Debug, Clone, Serialize, Deserialize)] +struct ApiTool { + name: String, + description: String, + input_schema: serde_json::Value, +} + +impl From<&ToolDefinition> for ApiTool { + fn from(def: &ToolDefinition) -> Self { + Self { + name: def.name.clone(), + description: def.description.clone(), + input_schema: def.input_schema.clone(), + } + } +} + /// The request body sent to the Anthropic Messages API. #[derive(Debug, Serialize)] struct ApiRequest { @@ -75,6 +93,8 @@ struct ApiRequest { #[serde(skip_serializing_if = "Option::is_none")] system: Option, messages: Vec, + #[serde(skip_serializing_if = "Vec::is_empty")] + tools: Vec, #[serde(skip_serializing_if = "Option::is_none")] stream: Option, } @@ -85,18 +105,14 @@ struct ApiResponse { content: Vec, #[allow(dead_code)] model: String, - #[allow(dead_code)] stop_reason: Option, - #[allow(dead_code)] usage: Option, } /// Token usage information. #[derive(Debug, Deserialize)] struct Usage { - #[allow(dead_code)] input_tokens: u64, - #[allow(dead_code)] output_tokens: u64, } @@ -129,7 +145,7 @@ struct ApiErrorDetail { /// # async fn example() -> rvagent_core::error::Result<()> { /// let config = resolve_model("anthropic:claude-sonnet-4-20250514"); /// let client = AnthropicClient::new(config)?; -/// let response = client.complete(&[Message::human("Hello!")]).await?; +/// let response = client.complete(&[Message::human("Hello!")], &[]).await?; /// println!("{}", response.content()); /// # Ok(()) /// # } @@ -168,12 +184,28 @@ impl AnthropicClient { } } - /// Build the API request body from rvAgent messages. - fn build_request(&self, messages: &[Message], stream: bool) -> ApiRequest { + /// Build the API request body from rvAgent messages and tool definitions. + fn build_request( + &self, + messages: &[Message], + tools: &[ToolDefinition], + stream: bool, + ) -> ApiRequest { let mut system_text: Option = None; let mut api_messages: Vec = Vec::new(); + // Anthropic expects every tool_result answering one assistant turn in + // a single user message; splitting them suppresses parallel tool + // calling. Buffer consecutive tool results and flush them together. + let mut pending_tool_results: Vec = Vec::new(); for msg in messages { + if !matches!(msg, Message::Tool(_)) && !pending_tool_results.is_empty() { + api_messages.push(ApiMessage { + role: "user".to_string(), + content: ApiContent::Blocks(std::mem::take(&mut pending_tool_results)), + }); + } + match msg { Message::System(s) => { // Anthropic uses a top-level `system` field; merge multiple system messages. @@ -219,17 +251,21 @@ impl AnthropicClient { } } Message::Tool(t) => { - api_messages.push(ApiMessage { - role: "user".to_string(), - content: ApiContent::Blocks(vec![ContentBlock::ToolResult { - tool_use_id: t.tool_call_id.clone(), - content: t.content.clone(), - }]), + pending_tool_results.push(ContentBlock::ToolResult { + tool_use_id: t.tool_call_id.clone(), + content: t.content.clone(), }); } } } + if !pending_tool_results.is_empty() { + api_messages.push(ApiMessage { + role: "user".to_string(), + content: ApiContent::Blocks(pending_tool_results), + }); + } + ApiRequest { model: self.config.model_id.clone(), max_tokens: self.config.max_tokens, @@ -240,6 +276,7 @@ impl AnthropicClient { }, system: system_text, messages: api_messages, + tools: tools.iter().map(ApiTool::from).collect(), stream: if stream { Some(true) } else { None }, } } @@ -330,6 +367,9 @@ impl AnthropicClient { } /// Convert an API response into an rvAgent [`Message`]. + /// + /// Token usage is attached to the message metadata under the `usage` key + /// so the agent loop and budget layers can account for it. fn parse_response(response: ApiResponse) -> Message { let mut text_parts: Vec = Vec::new(); let mut tool_calls: Vec = Vec::new(); @@ -352,23 +392,36 @@ impl AnthropicClient { let content = text_parts.join(""); - if tool_calls.is_empty() { - Message::ai(content) - } else { - Message::Ai(AiMessage { - content, - tool_calls, - metadata: HashMap::new(), - }) + let mut metadata = HashMap::new(); + if let Some(usage) = &response.usage { + metadata.insert( + "usage".to_string(), + serde_json::json!({ + "input_tokens": usage.input_tokens, + "output_tokens": usage.output_tokens, + }), + ); } + if let Some(stop_reason) = &response.stop_reason { + metadata.insert( + "stop_reason".to_string(), + serde_json::Value::String(stop_reason.clone()), + ); + } + + Message::Ai(AiMessage { + content, + tool_calls, + metadata, + }) } } #[async_trait] impl ChatModel for AnthropicClient { - /// Send messages and receive a complete response. - async fn complete(&self, messages: &[Message]) -> Result { - let request_body = self.build_request(messages, false); + /// Send messages and the active tool set, receive a complete response. + async fn complete(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result { + let request_body = self.build_request(messages, tools, false); let response = self .send_with_retry(&request_body, ANTHROPIC_API_URL) .await?; @@ -379,8 +432,8 @@ impl ChatModel for AnthropicClient { /// /// True SSE streaming is not yet implemented; this method calls the non-streaming /// endpoint and returns a single-element vector containing the complete message. - async fn stream(&self, messages: &[Message]) -> Result> { - let msg = self.complete(messages).await?; + async fn stream(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result> { + let msg = self.complete(messages, tools).await?; Ok(vec![msg]) } } @@ -452,7 +505,7 @@ mod tests { Message::system("You are helpful."), Message::human("Hello!"), ]; - let req = client.build_request(&messages, false); + let req = client.build_request(&messages, &[], false); assert_eq!(req.model, "claude-sonnet-4-20250514"); assert_eq!(req.max_tokens, 1024); @@ -471,7 +524,7 @@ mod tests { Message::system("Second instruction."), Message::human("Go."), ]; - let req = client.build_request(&messages, false); + let req = client.build_request(&messages, &[], false); assert_eq!( req.system, @@ -495,7 +548,7 @@ mod tests { ), Message::tool("tc_1", "file contents here"), ]; - let req = client.build_request(&messages, false); + let req = client.build_request(&messages, &[], false); assert_eq!(req.messages.len(), 3); assert_eq!(req.messages[0].role, "user"); @@ -511,12 +564,157 @@ mod tests { } } + #[test] + fn test_build_request_groups_parallel_tool_results() { + let client = + AnthropicClient::with_http(test_config(), reqwest::Client::new(), "key".to_string()); + let messages = vec![ + Message::human("Read both files."), + Message::ai_with_tools( + "Reading them.", + vec![ + ToolCall { + id: "tc_1".to_string(), + name: "read_file".to_string(), + args: json!({"path": "/tmp/a.txt"}), + }, + ToolCall { + id: "tc_2".to_string(), + name: "read_file".to_string(), + args: json!({"path": "/tmp/b.txt"}), + }, + ], + ), + Message::tool("tc_1", "a contents"), + Message::tool("tc_2", "b contents"), + ]; + let req = client.build_request(&messages, &[], false); + + // user + assistant + ONE user message holding both tool_results. + assert_eq!(req.messages.len(), 3); + assert_eq!(req.messages[2].role, "user"); + match &req.messages[2].content { + ApiContent::Blocks(blocks) => { + assert_eq!(blocks.len(), 2); + match (&blocks[0], &blocks[1]) { + ( + ContentBlock::ToolResult { + tool_use_id: first, .. + }, + ContentBlock::ToolResult { + tool_use_id: second, + .. + }, + ) => { + assert_eq!(first, "tc_1"); + assert_eq!(second, "tc_2"); + } + _ => panic!("expected two tool_result blocks"), + } + } + _ => panic!("expected Blocks content for grouped tool results"), + } + } + + #[test] + fn test_build_request_tool_results_flush_before_next_turn() { + let client = + AnthropicClient::with_http(test_config(), reqwest::Client::new(), "key".to_string()); + // Results answering distinct assistant turns must stay separate. + let messages = vec![ + Message::ai_with_tools( + "", + vec![ToolCall { + id: "tc_1".to_string(), + name: "first".to_string(), + args: json!({}), + }], + ), + Message::tool("tc_1", "one"), + Message::ai_with_tools( + "", + vec![ToolCall { + id: "tc_2".to_string(), + name: "second".to_string(), + args: json!({}), + }], + ), + Message::tool("tc_2", "two"), + ]; + let req = client.build_request(&messages, &[], false); + + assert_eq!(req.messages.len(), 4); + assert_eq!(req.messages[0].role, "assistant"); + assert_eq!(req.messages[1].role, "user"); + assert_eq!(req.messages[2].role, "assistant"); + assert_eq!(req.messages[3].role, "user"); + for idx in [1usize, 3] { + match &req.messages[idx].content { + ApiContent::Blocks(blocks) => assert_eq!(blocks.len(), 1), + _ => panic!("expected Blocks content at index {idx}"), + } + } + } + + #[test] + fn test_build_request_with_tool_definitions() { + let client = + AnthropicClient::with_http(test_config(), reqwest::Client::new(), "key".to_string()); + let tools = vec![ToolDefinition { + name: "read_file".to_string(), + description: "Read a file from the workspace".to_string(), + input_schema: json!({ + "type": "object", + "properties": {"path": {"type": "string"}}, + "required": ["path"] + }), + }]; + let req = client.build_request(&[Message::human("read it")], &tools, false); + + assert_eq!(req.tools.len(), 1); + assert_eq!(req.tools[0].name, "read_file"); + + // Wire format: tools must serialize with name/description/input_schema. + let json = serde_json::to_value(&req).unwrap(); + assert_eq!(json["tools"][0]["name"], "read_file"); + assert!(json["tools"][0]["input_schema"]["properties"]["path"].is_object()); + + // Empty tool set must omit the field entirely. + let req_no_tools = client.build_request(&[Message::human("hi")], &[], false); + let json_no_tools = serde_json::to_value(&req_no_tools).unwrap(); + assert!(json_no_tools.get("tools").is_none()); + } + + #[test] + fn test_parse_response_attaches_usage_metadata() { + let response = ApiResponse { + content: vec![ContentBlock::Text { + text: "hi".to_string(), + }], + model: "claude-sonnet-4-20250514".to_string(), + stop_reason: Some("end_turn".to_string()), + usage: Some(Usage { + input_tokens: 11, + output_tokens: 7, + }), + }; + let msg = AnthropicClient::parse_response(response); + if let Message::Ai(ai) = &msg { + let usage = ai.metadata.get("usage").expect("usage metadata"); + assert_eq!(usage["input_tokens"], 11); + assert_eq!(usage["output_tokens"], 7); + assert_eq!(ai.metadata["stop_reason"], "end_turn"); + } else { + panic!("expected Ai message"); + } + } + #[test] fn test_build_request_stream_flag() { let client = AnthropicClient::with_http(test_config(), reqwest::Client::new(), "key".to_string()); let messages = vec![Message::human("Hi")]; - let req = client.build_request(&messages, true); + let req = client.build_request(&messages, &[], true); assert_eq!(req.stream, Some(true)); } @@ -642,14 +840,14 @@ mod tests { fn test_temperature_serialization() { let client = AnthropicClient::with_http(test_config(), reqwest::Client::new(), "key".to_string()); - let req = client.build_request(&[Message::human("Hi")], false); + let req = client.build_request(&[Message::human("Hi")], &[], false); // temperature=0.0 => None (omitted) assert!(req.temperature.is_none()); let mut config = test_config(); config.temperature = 0.7; let client2 = AnthropicClient::with_http(config, reqwest::Client::new(), "key".to_string()); - let req2 = client2.build_request(&[Message::human("Hi")], false); + let req2 = client2.build_request(&[Message::human("Hi")], &[], false); assert_eq!(req2.temperature, Some(0.7)); } @@ -664,6 +862,7 @@ mod tests { role: "user".to_string(), content: ApiContent::Text("Hello".to_string()), }], + tools: Vec::new(), stream: None, }; let json = serde_json::to_value(&req).expect("serialization failed"); @@ -736,7 +935,7 @@ mod tests { let client = test_client(&server.url()); let url = format!("{}/v1/messages", server.url()); - let req = client.build_request(&[Message::human("Hello")], false); + let req = client.build_request(&[Message::human("Hello")], &[], false); let resp = client.send_with_retry(&req, &url).await; assert!(resp.is_ok()); @@ -768,7 +967,7 @@ mod tests { let client = test_client(&server.url()); let url = format!("{}/v1/messages", server.url()); - let req = client.build_request(&[Message::human("Search for Rust")], false); + let req = client.build_request(&[Message::human("Search for Rust")], &[], false); let resp = client.send_with_retry(&req, &url).await; assert!(resp.is_ok()); @@ -794,7 +993,7 @@ mod tests { let client = test_client(&server.url()); let url = format!("{}/v1/messages", server.url()); - let req = client.build_request(&[Message::human("Hi")], false); + let req = client.build_request(&[Message::human("Hi")], &[], false); let result = client.send_with_retry(&req, &url).await; assert!(result.is_err()); @@ -839,7 +1038,7 @@ mod tests { let client = test_client(&server.url()); let url = format!("{}/v1/messages", server.url()); - let req = client.build_request(&[Message::human("Hi")], false); + let req = client.build_request(&[Message::human("Hi")], &[], false); let result = client.send_with_retry(&req, &url).await; assert!(result.is_ok()); @@ -869,7 +1068,7 @@ mod tests { let client = test_client(&server.url()); let url = format!("{}/v1/messages", server.url()); - let req = client.build_request(&[Message::human("Hi")], false); + let req = client.build_request(&[Message::human("Hi")], &[], false); let result = client.send_with_retry(&req, &url).await; assert!(result.is_err()); @@ -902,7 +1101,7 @@ mod tests { let client = test_client(&server.url()); let url = format!("{}/v1/messages", server.url()); - let req = client.build_request(&[Message::human("Hi")], false); + let req = client.build_request(&[Message::human("Hi")], &[], false); let result = client.send_with_retry(&req, &url).await; assert!(result.is_ok()); diff --git a/crates/rvAgent/rvagent-backends/src/gemini.rs b/crates/rvAgent/rvagent-backends/src/gemini.rs index 980d5da1f..d82f4b208 100644 --- a/crates/rvAgent/rvagent-backends/src/gemini.rs +++ b/crates/rvAgent/rvagent-backends/src/gemini.rs @@ -9,9 +9,11 @@ use async_trait::async_trait; use serde::{Deserialize, Serialize}; use tracing::{debug, warn}; +use std::collections::HashMap; + use rvagent_core::error::{Result, RvAgentError}; -use rvagent_core::messages::{AiMessage, Message}; -use rvagent_core::models::{ApiKeySource, ChatModel, ModelConfig}; +use rvagent_core::messages::{AiMessage, Message, ToolCall}; +use rvagent_core::models::{ApiKeySource, ChatModel, ModelConfig, ToolDefinition}; // --------------------------------------------------------------------------- // Constants @@ -28,16 +30,69 @@ const RETRYABLE_STATUS_CODES: &[u16] = &[429, 500, 502, 503]; // Gemini API request / response types // --------------------------------------------------------------------------- -/// Content part in a Gemini message. +/// A function call emitted by the model. #[derive(Debug, Clone, Serialize, Deserialize)] +struct FunctionCall { + name: String, + #[serde(default)] + args: serde_json::Value, +} + +/// A function result sent back to the model. +#[derive(Debug, Clone, Serialize, Deserialize)] +struct FunctionResponse { + name: String, + response: serde_json::Value, +} + +/// Content part in a Gemini message: text, a function call, or a function response. +#[derive(Debug, Clone, Serialize, Deserialize, Default)] +#[serde(rename_all = "camelCase")] struct Part { - text: String, + #[serde(skip_serializing_if = "Option::is_none")] + text: Option, + #[serde(skip_serializing_if = "Option::is_none")] + function_call: Option, + #[serde(skip_serializing_if = "Option::is_none")] + function_response: Option, +} + +impl Part { + fn text(text: impl Into) -> Self { + Self { + text: Some(text.into()), + ..Self::default() + } + } +} + +/// A function declaration advertised to the model. +#[derive(Debug, Clone, Serialize, Deserialize)] +struct FunctionDeclaration { + name: String, + description: String, + /// Omitted entirely for parameterless tools: Gemini rejects an OBJECT + /// schema with an empty property map, and `parameters` is optional. + #[serde(default, skip_serializing_if = "Option::is_none")] + parameters: Option, +} + +/// A tool group in the Gemini API format. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +struct GeminiTool { + function_declarations: Vec, } /// A single message in the Gemini API format. -#[derive(Debug, Clone, Serialize, Deserialize)] +/// +/// Both fields default on deserialization: Gemini omits `parts` entirely on a +/// truncated or filtered candidate, and omits `role` on some content blocks. +#[derive(Debug, Clone, Serialize, Deserialize, Default)] struct GeminiContent { + #[serde(default)] role: String, + #[serde(default)] parts: Vec, } @@ -58,33 +113,47 @@ struct GeminiRequest { generation_config: GenerationConfig, #[serde(skip_serializing_if = "Option::is_none")] system_instruction: Option, + #[serde(skip_serializing_if = "Vec::is_empty")] + tools: Vec, } /// A candidate response from Gemini. #[derive(Debug, Deserialize)] +#[serde(rename_all = "camelCase")] struct Candidate { - content: GeminiContent, - #[allow(dead_code)] + /// Absent when Gemini blocks the candidate outright (e.g. a SAFETY stop + /// yields `{"finishReason": "SAFETY"}` with no content at all). + #[serde(default)] + content: Option, #[serde(default)] finish_reason: Option, } +/// Feedback about the prompt itself, present when Gemini refuses to answer. +#[derive(Debug, Deserialize)] +#[serde(rename_all = "camelCase")] +struct PromptFeedback { + #[serde(default)] + block_reason: Option, +} + /// The response body from the Gemini API. #[derive(Debug, Deserialize)] +#[serde(rename_all = "camelCase")] struct GeminiResponse { + #[serde(default)] candidates: Vec, - #[allow(dead_code)] #[serde(default)] usage_metadata: Option, + #[serde(default)] + prompt_feedback: Option, } /// Token usage information. #[derive(Debug, Deserialize)] #[serde(rename_all = "camelCase")] struct UsageMetadata { - #[allow(dead_code)] prompt_token_count: Option, - #[allow(dead_code)] candidates_token_count: Option, } @@ -117,7 +186,7 @@ struct GeminiErrorDetail { /// # async fn example() -> rvagent_core::error::Result<()> { /// let config = resolve_model("google:gemini-2.5-pro-preview-06-05"); /// let client = GeminiClient::new(config)?; -/// let response = client.complete(&[Message::human("Hello!")]).await?; +/// let response = client.complete(&[Message::human("Hello!")], &[]).await?; /// println!("{}", response.content()); /// # Ok(()) /// # } @@ -143,49 +212,106 @@ impl GeminiClient { }) } - /// Build the API request body from rvAgent messages. - fn build_request(&self, messages: &[Message]) -> GeminiRequest { + /// Build the API request body from rvAgent messages and tool definitions. + fn build_request(&self, messages: &[Message], tools: &[ToolDefinition]) -> GeminiRequest { let mut system_instruction: Option = None; let mut contents: Vec = Vec::new(); + // Gemini has no tool-call IDs on the wire; map our synthesized IDs + // back to function names so tool results can be sent as + // functionResponse parts. + let mut call_id_to_name: HashMap = HashMap::new(); + // Gemini requires every functionResponse answering one model turn to + // live in a single user Content. Buffer consecutive tool results and + // flush them together. + let mut pending_tool_parts: Vec = Vec::new(); for msg in messages { + if !matches!(msg, Message::Tool(_)) && !pending_tool_parts.is_empty() { + contents.push(GeminiContent { + role: "user".to_string(), + parts: std::mem::take(&mut pending_tool_parts), + }); + } + match msg { Message::System(s) => { system_instruction = Some(GeminiContent { role: "user".to_string(), - parts: vec![Part { - text: s.content.clone(), - }], + parts: vec![Part::text(s.content.clone())], }); } Message::Human(h) => { contents.push(GeminiContent { role: "user".to_string(), - parts: vec![Part { - text: h.content.clone(), - }], + parts: vec![Part::text(h.content.clone())], }); } Message::Ai(ai) => { + let mut parts: Vec = Vec::new(); + if !ai.content.is_empty() { + parts.push(Part::text(ai.content.clone())); + } + for tc in &ai.tool_calls { + call_id_to_name.insert(tc.id.clone(), tc.name.clone()); + parts.push(Part { + function_call: Some(FunctionCall { + name: tc.name.clone(), + args: tc.args.clone(), + }), + ..Part::default() + }); + } + if parts.is_empty() { + parts.push(Part::text(String::new())); + } contents.push(GeminiContent { role: "model".to_string(), - parts: vec![Part { - text: ai.content.clone(), - }], + parts, }); } Message::Tool(t) => { - // Tool results go as user messages - contents.push(GeminiContent { - role: "user".to_string(), - parts: vec![Part { - text: format!("Tool result: {}", t.content), - }], + // Prefer the tool name recorded on the message itself, + // then the id→name map from the preceding AI turn. The + // raw id is a last resort only (it is not a valid + // function name and Gemini may reject it). + let name = t + .tool_name + .clone() + .or_else(|| call_id_to_name.get(&t.tool_call_id).cloned()) + .unwrap_or_else(|| t.tool_call_id.clone()); + pending_tool_parts.push(Part { + function_response: Some(FunctionResponse { + name, + response: serde_json::json!({ "result": t.content }), + }), + ..Part::default() }); } } } + if !pending_tool_parts.is_empty() { + contents.push(GeminiContent { + role: "user".to_string(), + parts: pending_tool_parts, + }); + } + + let gemini_tools = if tools.is_empty() { + Vec::new() + } else { + vec![GeminiTool { + function_declarations: tools + .iter() + .map(|t| FunctionDeclaration { + name: t.name.clone(), + description: t.description.clone(), + parameters: tool_parameters(&t.input_schema), + }) + .collect(), + }] + }; + GeminiRequest { contents, generation_config: GenerationConfig { @@ -197,14 +323,18 @@ impl GeminiClient { }, }, system_instruction, + tools: gemini_tools, } } /// Send a request to the API with retry logic. async fn send_with_retry(&self, request_body: &GeminiRequest) -> Result { + // The API key travels in the `x-goog-api-key` header, never in the + // URL query string — URLs end up in error messages, proxy logs, and + // tracing output, which must never contain credentials. let url = format!( - "{}/{}:generateContent?key={}", - GEMINI_API_BASE, self.config.model_id, self.api_key + "{}/{}:generateContent", + GEMINI_API_BASE, self.config.model_id ); let mut last_err: Option = None; @@ -225,6 +355,7 @@ impl GeminiClient { let result = self .http .post(&url) + .header("x-goog-api-key", &self.api_key) .header("content-type", "application/json") .body(body_json) .send() @@ -279,32 +410,96 @@ impl GeminiClient { Err(last_err .unwrap_or_else(|| RvAgentError::model("Gemini API request failed after all retries"))) } + + /// Convert an API response into an rvAgent [`Message`]. + fn parse_response(response: GeminiResponse) -> Result { + // Collect text and function-call parts from the first candidate. + // Gemini does not assign tool-call IDs, so synthesize unique ones: + // a process-wide counter prevents id collisions across turns (the + // per-part index alone repeats every turn, which would let one + // turn's tool result satisfy another turn's call in id-keyed logic). + static CALL_COUNTER: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0); + let mut text_parts: Vec = Vec::new(); + let mut tool_calls: Vec = Vec::new(); + // An empty candidate list means Gemini produced nothing — typically a + // safety block, which carries a promptFeedback.blockReason. Surfacing + // it as an error keeps the agent loop from treating a refusal as an + // ordinary empty reply. + let candidate = response.candidates.first().ok_or_else(|| { + match response + .prompt_feedback + .as_ref() + .and_then(|f| f.block_reason.as_deref()) + { + Some(reason) => RvAgentError::model(format!( + "Gemini returned no candidates: prompt blocked (blockReason: {reason})" + )), + None => RvAgentError::model( + "Gemini returned no candidates (response may have been blocked)", + ), + } + })?; + + // A candidate can arrive with no content (SAFETY block) or with + // content but no parts (MAX_TOKENS truncation on the 2.5 series). + // Both are failures, not empty replies — report the finishReason so + // the caller can tell truncation from filtering. + let finish_reason = candidate.finish_reason.as_deref().unwrap_or("unspecified"); + let content = candidate.content.as_ref().ok_or_else(|| { + RvAgentError::model(format!( + "Gemini candidate had no content (finishReason: {finish_reason})" + )) + })?; + if content.parts.is_empty() { + return Err(RvAgentError::model(format!( + "Gemini candidate had no content parts (finishReason: {finish_reason})" + ))); + } + + for part in &content.parts { + if let Some(text) = &part.text { + text_parts.push(text.clone()); + } + if let Some(fc) = &part.function_call { + let n = CALL_COUNTER.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + tool_calls.push(ToolCall { + id: format!("gemini_call_{n}_{}", fc.name), + name: fc.name.clone(), + args: fc.args.clone(), + }); + } + } + + let mut metadata = HashMap::new(); + if let Some(usage) = &response.usage_metadata { + metadata.insert( + "usage".to_string(), + serde_json::json!({ + "input_tokens": usage.prompt_token_count.unwrap_or(0), + "output_tokens": usage.candidates_token_count.unwrap_or(0), + }), + ); + } + + Ok(Message::Ai(AiMessage { + content: text_parts.join(""), + tool_calls, + metadata, + })) + } } #[async_trait] impl ChatModel for GeminiClient { - async fn complete(&self, messages: &[Message]) -> Result { - let request = self.build_request(messages); + async fn complete(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result { + let request = self.build_request(messages, tools); let response = self.send_with_retry(&request).await?; - - // Extract text from first candidate - let text = response - .candidates - .first() - .and_then(|c| c.content.parts.first()) - .map(|p| p.text.clone()) - .unwrap_or_default(); - - Ok(Message::Ai(AiMessage { - content: text, - tool_calls: vec![], - metadata: std::collections::HashMap::new(), - })) + Self::parse_response(response) } - async fn stream(&self, messages: &[Message]) -> Result> { + async fn stream(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result> { // For now, use non-streaming completion - let msg = self.complete(messages).await?; + let msg = self.complete(messages, tools).await?; Ok(vec![msg]) } } @@ -313,6 +508,160 @@ impl ChatModel for GeminiClient { // Helpers // --------------------------------------------------------------------------- +/// JSON Schema keywords Gemini's restricted OpenAPI subset accepts. +const GEMINI_SCHEMA_KEYS: &[&str] = &[ + "type", + "format", + "description", + "nullable", + "enum", + "items", + "properties", + "required", + "minimum", + "maximum", + "minItems", + "maxItems", + "minLength", + "maxLength", + "pattern", + "anyOf", +]; + +/// Sanitize a tool's input schema into the `parameters` field of a +/// [`FunctionDeclaration`]. +/// +/// A parameterless tool sanitizes down to a bare `{"type": "object"}` (the +/// in-tree rvagent-mcp registry `ping` tool declares `{"type": "object", +/// "properties": {}}`). `parameters` is optional in the Gemini API, so the +/// field is omitted outright rather than sent as a property-less object. +fn tool_parameters(input_schema: &serde_json::Value) -> Option { + let sanitized = sanitize_schema(input_schema); + let is_bare_object = sanitized.get("type").and_then(serde_json::Value::as_str) + == Some("object") + && sanitized.get("properties").is_none(); + if is_bare_object { + None + } else { + Some(sanitized) + } +} + +/// `format` values Gemini accepts, keyed by the schema's resolved type. +/// Anything else (uri, email, uuid, date, binary, …) is a 400. +const GEMINI_STRING_FORMATS: &[&str] = &["enum", "date-time"]; +const GEMINI_INTEGER_FORMATS: &[&str] = &["int32", "int64"]; +const GEMINI_NUMBER_FORMATS: &[&str] = &["float", "double"]; + +/// Rewrite a JSON Schema into Gemini's restricted OpenAPI subset. +/// +/// Beyond stripping keywords Gemini rejects outright (`default`, `$ref`, +/// `oneOf`, `additionalProperties`, `$schema`, `examples`, …), this repairs +/// four shapes that are legal JSON Schema but 400 on Gemini: +/// +/// - an OBJECT with an empty (or emptied) property map — the `properties` key +/// is dropped, leaving a bare `{"type": "object"}`; +/// - the OpenAPI 3.1 / pydantic array type form `{"type": ["string", "null"]}` +/// — collapsed to the first non-null entry plus `"nullable": true`; +/// - a `format` the resolved type does not support — dropped; +/// - a subschema left type-less by stripping (it carried only `$ref`/`oneOf`) +/// or an `enum` with no type — defaulted to `"string"`, since a best-effort +/// type beats a guaranteed 400. +/// +/// Applied recursively so nested `properties`, `items`, and `anyOf` +/// subschemas are repaired too. +fn sanitize_schema(schema: &serde_json::Value) -> serde_json::Value { + use serde_json::Value; + + let Value::Object(map) = schema else { + return schema.clone(); + }; + + let mut out = serde_json::Map::new(); + for (key, value) in map { + if !GEMINI_SCHEMA_KEYS.contains(&key.as_str()) { + continue; + } + let cleaned = match key.as_str() { + "properties" => match value { + Value::Object(props) => Value::Object( + props + .iter() + .map(|(name, sub)| (name.clone(), sanitize_schema(sub))) + .collect(), + ), + other => other.clone(), + }, + // `items` is a subschema, or an array of subschemas in tuple form. + "items" => match value { + Value::Array(subs) => Value::Array(subs.iter().map(sanitize_schema).collect()), + sub => sanitize_schema(sub), + }, + "anyOf" => match value { + Value::Array(subs) => Value::Array(subs.iter().map(sanitize_schema).collect()), + other => other.clone(), + }, + _ => value.clone(), + }; + out.insert(key.clone(), cleaned); + } + + // Collapse the array type form to a single type + nullable flag. + if let Some(Value::Array(variants)) = out.get("type").cloned() { + let has_null = variants.iter().any(|v| v.as_str() == Some("null")); + let primary = variants + .iter() + .find(|v| matches!(v.as_str(), Some(s) if s != "null")) + .cloned() + .unwrap_or_else(|| Value::String("string".to_string())); + out.insert("type".to_string(), primary); + if has_null { + out.insert("nullable".to_string(), Value::Bool(true)); + } + } + + // An enum is a string enum unless the schema says otherwise. + if out.contains_key("enum") && !out.contains_key("type") { + out.insert("type".to_string(), Value::String("string".to_string())); + } + + // Gemini rejects OBJECT with an empty property map; a bare + // `{"type": "object"}` is accepted. + let properties_empty = out + .get("properties") + .is_some_and(|p| p.as_object().is_none_or(|m| m.is_empty())); + if properties_empty { + out.remove("properties"); + } + + // `format` is only valid for the types listed above. + if let Some(format) = out + .get("format") + .and_then(Value::as_str) + .map(str::to_string) + { + let supported = match out.get("type").and_then(Value::as_str) { + Some("string") => GEMINI_STRING_FORMATS.contains(&format.as_str()), + Some("integer") => GEMINI_INTEGER_FORMATS.contains(&format.as_str()), + Some("number") => GEMINI_NUMBER_FORMATS.contains(&format.as_str()), + _ => false, + }; + if !supported { + out.remove("format"); + } + } + + // Nothing left to describe the shape (e.g. the schema was only a `$ref`). + if !["type", "anyOf", "enum", "properties", "items"] + .iter() + .any(|k| out.contains_key(*k)) + { + out.insert("type".to_string(), Value::String("string".to_string())); + } + + Value::Object(out) +} + fn resolve_api_key(source: &ApiKeySource) -> Result { match source { ApiKeySource::Env(var) => std::env::var(var).map_err(|_| { @@ -338,20 +687,21 @@ mod tests { let request = GeminiRequest { contents: vec![GeminiContent { role: "user".to_string(), - parts: vec![Part { - text: "Hello".to_string(), - }], + parts: vec![Part::text("Hello")], }], generation_config: GenerationConfig { max_output_tokens: 1024, temperature: Some(0.7), }, system_instruction: None, + tools: Vec::new(), }; let json = serde_json::to_string(&request).unwrap(); assert!(json.contains("\"role\":\"user\"")); assert!(json.contains("\"maxOutputTokens\":1024")); + // Empty tools must be omitted entirely. + assert!(!json.contains("\"tools\"")); } #[test] @@ -368,6 +718,562 @@ mod tests { let response: GeminiResponse = serde_json::from_str(json).unwrap(); assert_eq!(response.candidates.len(), 1); - assert_eq!(response.candidates[0].content.parts[0].text, "Hello there!"); + assert_eq!( + response.candidates[0].content.as_ref().unwrap().parts[0] + .text + .as_deref(), + Some("Hello there!") + ); + } + + #[test] + fn test_gemini_request_with_tools() { + let config = ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }; + let client = GeminiClient { + config, + http: reqwest::Client::new(), + api_key: "test".to_string(), + }; + + let tools = vec![ToolDefinition { + name: "read_file".to_string(), + description: "Read a file".to_string(), + input_schema: serde_json::json!({ + "type": "object", + "properties": {"path": {"type": "string"}}, + "required": ["path"] + }), + }]; + let req = client.build_request(&[Message::human("read it")], &tools); + assert_eq!(req.tools.len(), 1); + assert_eq!(req.tools[0].function_declarations.len(), 1); + assert_eq!(req.tools[0].function_declarations[0].name, "read_file"); + + let json = serde_json::to_string(&req).unwrap(); + assert!(json.contains("functionDeclarations")); + } + + #[test] + fn test_gemini_function_call_response_parsing() { + let json = r#"{ + "candidates": [{ + "content": { + "role": "model", + "parts": [ + {"text": "Let me read that."}, + {"functionCall": {"name": "read_file", "args": {"path": "/tmp/x"}}} + ] + }, + "finishReason": "STOP" + }], + "usageMetadata": {"promptTokenCount": 12, "candidatesTokenCount": 7} + }"#; + + let response: GeminiResponse = serde_json::from_str(json).unwrap(); + let parts = &response.candidates[0].content.as_ref().unwrap().parts; + assert_eq!(parts.len(), 2); + assert!(parts[0].text.is_some()); + let fc = parts[1].function_call.as_ref().unwrap(); + assert_eq!(fc.name, "read_file"); + assert_eq!(fc.args, serde_json::json!({"path": "/tmp/x"})); + let usage = response.usage_metadata.unwrap(); + assert_eq!(usage.prompt_token_count, Some(12)); + assert_eq!(usage.candidates_token_count, Some(7)); + } + + #[test] + fn test_gemini_tool_result_roundtrip() { + let config = ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }; + let client = GeminiClient { + config, + http: reqwest::Client::new(), + api_key: "test".to_string(), + }; + + let messages = vec![ + Message::human("read it"), + Message::ai_with_tools( + "", + vec![ToolCall { + id: "gemini_call_0_read_file".to_string(), + name: "read_file".to_string(), + args: serde_json::json!({"path": "/tmp/x"}), + }], + ), + Message::tool("gemini_call_0_read_file", "contents"), + ]; + let req = client.build_request(&messages, &[]); + + // AI turn carries the functionCall part; tool turn carries the + // functionResponse part with the recovered function name. + let model_turn = &req.contents[1]; + assert_eq!(model_turn.role, "model"); + assert!(model_turn.parts[0].function_call.is_some()); + + let tool_turn = &req.contents[2]; + assert_eq!(tool_turn.role, "user"); + let fr = tool_turn.parts[0].function_response.as_ref().unwrap(); + assert_eq!(fr.name, "read_file"); + assert_eq!(fr.response, serde_json::json!({"result": "contents"})); + } + + #[test] + fn test_gemini_tool_result_prefers_recorded_tool_name() { + let config = ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }; + let client = GeminiClient { + config, + http: reqwest::Client::new(), + api_key: "test".to_string(), + }; + + // Orphaned tool message (no preceding AI turn, e.g. after history + // trimming): the recorded tool_name must win over the raw id. + let messages = vec![ + Message::human("read it"), + Message::tool_with_name("some_opaque_id", "contents", "read_file"), + ]; + let req = client.build_request(&messages, &[]); + let fr = req.contents[1].parts[0].function_response.as_ref().unwrap(); + assert_eq!(fr.name, "read_file"); + } + + #[test] + fn test_gemini_parallel_tool_results_share_one_content() { + let config = ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }; + let client = GeminiClient { + config, + http: reqwest::Client::new(), + api_key: "test".to_string(), + }; + + let messages = vec![ + Message::human("read both"), + Message::ai_with_tools( + "", + vec![ + ToolCall { + id: "gemini_call_0_read_file".to_string(), + name: "read_file".to_string(), + args: serde_json::json!({"path": "/tmp/a"}), + }, + ToolCall { + id: "gemini_call_1_list_dir".to_string(), + name: "list_dir".to_string(), + args: serde_json::json!({"path": "/tmp"}), + }, + ], + ), + Message::tool("gemini_call_0_read_file", "a contents"), + Message::tool("gemini_call_1_list_dir", "a\nb"), + ]; + let req = client.build_request(&messages, &[]); + + // human + model + ONE grouped tool-result turn. + assert_eq!(req.contents.len(), 3); + let tool_turn = &req.contents[2]; + assert_eq!(tool_turn.role, "user"); + assert_eq!(tool_turn.parts.len(), 2); + assert_eq!( + tool_turn.parts[0] + .function_response + .as_ref() + .unwrap() + .name + .as_str(), + "read_file" + ); + assert_eq!( + tool_turn.parts[1] + .function_response + .as_ref() + .unwrap() + .name + .as_str(), + "list_dir" + ); + } + + #[test] + fn test_gemini_tool_results_flush_before_following_turn() { + let config = ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }; + let client = GeminiClient { + config, + http: reqwest::Client::new(), + api_key: "test".to_string(), + }; + + // Two separate model turns, each answered by its own tool result: + // grouping must not merge across the intervening model turn. + let messages = vec![ + Message::ai_with_tools( + "", + vec![ToolCall { + id: "c0".to_string(), + name: "first".to_string(), + args: serde_json::json!({}), + }], + ), + Message::tool("c0", "one"), + Message::ai_with_tools( + "", + vec![ToolCall { + id: "c1".to_string(), + name: "second".to_string(), + args: serde_json::json!({}), + }], + ), + Message::tool("c1", "two"), + ]; + let req = client.build_request(&messages, &[]); + + assert_eq!(req.contents.len(), 4); + assert_eq!(req.contents[0].role, "model"); + assert_eq!(req.contents[1].parts.len(), 1); + assert_eq!(req.contents[2].role, "model"); + assert_eq!(req.contents[3].parts.len(), 1); + } + + #[test] + fn test_gemini_candidate_finish_reason_deserializes() { + let json = r#"{ + "candidates": [{ + "content": { + "role": "model", + "parts": [{"text": "truncated..."}] + }, + "finishReason": "MAX_TOKENS", + "index": 0 + }], + "usageMetadata": {"promptTokenCount": 8, "candidatesTokenCount": 1024} + }"#; + + let response: GeminiResponse = serde_json::from_str(json).unwrap(); + assert_eq!( + response.candidates[0].finish_reason.as_deref(), + Some("MAX_TOKENS") + ); + } + + #[test] + fn test_gemini_empty_candidates_is_error() { + let json = r#"{ + "candidates": [], + "promptFeedback": {"blockReason": "SAFETY"} + }"#; + let response: GeminiResponse = serde_json::from_str(json).unwrap(); + let err = GeminiClient::parse_response(response).unwrap_err(); + let msg = err.to_string(); + assert!(msg.contains("no candidates"), "unexpected error: {msg}"); + assert!(msg.contains("SAFETY"), "block reason missing: {msg}"); + } + + #[test] + fn test_gemini_missing_candidates_is_error() { + // No candidates key at all and no promptFeedback. + let response: GeminiResponse = serde_json::from_str("{}").unwrap(); + let err = GeminiClient::parse_response(response).unwrap_err(); + assert!(err.to_string().contains("no candidates")); + } + + #[test] + fn test_sanitize_schema_strips_unsupported_keys() { + let schema = serde_json::json!({ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "additionalProperties": false, + "properties": { + "path": { + "type": "string", + "description": "file path", + "default": "/tmp", + "examples": ["/tmp/a"] + }, + "mode": { + "$ref": "#/definitions/Mode", + "oneOf": [{"type": "string"}], + "enum": ["read", "write"] + }, + "tags": { + "type": "array", + "items": {"type": "string", "default": "x"} + } + }, + "required": ["path"] + }); + + let cleaned = sanitize_schema(&schema); + + assert!(cleaned.get("$schema").is_none()); + assert!(cleaned.get("additionalProperties").is_none()); + assert_eq!(cleaned["type"], "object"); + assert_eq!(cleaned["required"], serde_json::json!(["path"])); + + let path = &cleaned["properties"]["path"]; + assert!(path.get("default").is_none()); + assert!(path.get("examples").is_none()); + assert_eq!(path["description"], "file path"); + + let mode = &cleaned["properties"]["mode"]; + assert!(mode.get("$ref").is_none()); + assert!(mode.get("oneOf").is_none()); + assert_eq!(mode["enum"], serde_json::json!(["read", "write"])); + // An enum stripped down to no type must not stay type-less. + assert_eq!(mode["type"], "string"); + + // Nested subschemas under `items` are sanitized too. + assert!(cleaned["properties"]["tags"]["items"] + .get("default") + .is_none()); + + // And nothing unsupported survives serialization into the request. + let config = ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }; + let client = GeminiClient { + config, + http: reqwest::Client::new(), + api_key: "test".to_string(), + }; + let tools = vec![ToolDefinition { + name: "read_file".to_string(), + description: "Read a file".to_string(), + input_schema: schema, + }]; + let req = client.build_request(&[Message::human("go")], &tools); + let json = serde_json::to_string(&req).unwrap(); + assert!(!json.contains("\"default\"")); + assert!(!json.contains("$ref")); + assert!(!json.contains("additionalProperties")); + } + + fn test_client() -> GeminiClient { + GeminiClient { + config: ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }, + http: reqwest::Client::new(), + api_key: "test".to_string(), + } + } + + #[test] + fn test_sanitize_schema_drops_empty_properties() { + // The rvagent-mcp registry `ping` tool declares exactly this. + let cleaned = sanitize_schema(&serde_json::json!({ + "type": "object", + "properties": {} + })); + assert_eq!(cleaned, serde_json::json!({"type": "object"})); + + // Same when the map is only emptied by sanitizing (map schemas whose + // sole key was additionalProperties). + let cleaned = sanitize_schema(&serde_json::json!({ + "type": "object", + "additionalProperties": {"type": "string"} + })); + assert_eq!(cleaned, serde_json::json!({"type": "object"})); + } + + #[test] + fn test_parameterless_tool_omits_parameters() { + let tools = vec![ToolDefinition { + name: "ping".to_string(), + description: "Ping the server".to_string(), + input_schema: serde_json::json!({"type": "object", "properties": {}}), + }]; + let req = test_client().build_request(&[Message::human("ping")], &tools); + assert!(req.tools[0].function_declarations[0].parameters.is_none()); + + // `parameters` must not appear on the wire at all. + let json = serde_json::to_value(&req).unwrap(); + assert!(json["tools"][0]["functionDeclarations"][0] + .get("parameters") + .is_none()); + // A tool that does take parameters still carries them. + let tools = vec![ToolDefinition { + name: "read_file".to_string(), + description: "Read a file".to_string(), + input_schema: serde_json::json!({ + "type": "object", + "properties": {"path": {"type": "string"}} + }), + }]; + let req = test_client().build_request(&[Message::human("go")], &tools); + assert!(req.tools[0].function_declarations[0].parameters.is_some()); + } + + #[test] + fn test_sanitize_schema_normalizes_array_type_form() { + // OpenAPI 3.1 / pydantic nullable form. + let cleaned = sanitize_schema(&serde_json::json!({ + "type": ["string", "null"], + "description": "maybe a name" + })); + assert_eq!(cleaned["type"], "string"); + assert_eq!(cleaned["nullable"], true); + assert_eq!(cleaned["description"], "maybe a name"); + + // No null present: collapse to the first entry, no nullable flag. + let cleaned = sanitize_schema(&serde_json::json!({"type": ["integer", "string"]})); + assert_eq!(cleaned["type"], "integer"); + assert!(cleaned.get("nullable").is_none()); + + // Degenerate null-only form still yields a usable type. + let cleaned = sanitize_schema(&serde_json::json!({"type": ["null"]})); + assert_eq!(cleaned["type"], "string"); + assert_eq!(cleaned["nullable"], true); + + // Nested under properties too. + let cleaned = sanitize_schema(&serde_json::json!({ + "type": "object", + "properties": {"nick": {"type": ["string", "null"]}} + })); + assert_eq!(cleaned["properties"]["nick"]["type"], "string"); + assert_eq!(cleaned["properties"]["nick"]["nullable"], true); + } + + #[test] + fn test_sanitize_schema_filters_unsupported_formats() { + // Unsupported string formats are dropped, the type survives. + for format in ["uri", "email", "uuid", "date", "binary", "hostname"] { + let cleaned = sanitize_schema(&serde_json::json!({"type": "string", "format": format})); + assert_eq!( + cleaned, + serde_json::json!({"type": "string"}), + "format {format} should have been dropped" + ); + } + + // Supported ones are kept. + for format in ["enum", "date-time"] { + let cleaned = sanitize_schema(&serde_json::json!({"type": "string", "format": format})); + assert_eq!(cleaned["format"], format); + } + for (ty, format) in [ + ("integer", "int32"), + ("integer", "int64"), + ("number", "float"), + ("number", "double"), + ] { + let cleaned = sanitize_schema(&serde_json::json!({"type": ty, "format": format})); + assert_eq!(cleaned["format"], format); + } + + // Right format, wrong type => dropped. + let cleaned = sanitize_schema(&serde_json::json!({"type": "string", "format": "int32"})); + assert!(cleaned.get("format").is_none()); + let cleaned = sanitize_schema(&serde_json::json!({"type": "integer", "format": "double"})); + assert!(cleaned.get("format").is_none()); + } + + #[test] + fn test_sanitize_schema_defaults_typeless_subschemas() { + // A schema carrying only unsupported keys must not end up type-less. + let cleaned = sanitize_schema(&serde_json::json!({"$ref": "#/definitions/Mode"})); + assert_eq!(cleaned, serde_json::json!({"type": "string"})); + + let cleaned = sanitize_schema(&serde_json::json!({ + "allOf": [{"type": "string"}], + "oneOf": [{"type": "integer"}] + })); + assert_eq!(cleaned, serde_json::json!({"type": "string"})); + + // Enum without a type gains one. + let cleaned = sanitize_schema(&serde_json::json!({"enum": ["a", "b"]})); + assert_eq!(cleaned["type"], "string"); + assert_eq!(cleaned["enum"], serde_json::json!(["a", "b"])); + + // Schemas that already describe a shape are left alone. + let cleaned = sanitize_schema(&serde_json::json!({"anyOf": [{"type": "string"}]})); + assert!(cleaned.get("type").is_none()); + let cleaned = sanitize_schema(&serde_json::json!({ + "items": {"type": "string"}, "type": "array" + })); + assert_eq!(cleaned["type"], "array"); + } + + #[test] + fn test_gemini_candidate_without_content_deserializes_and_errors() { + // SAFETY stop: candidate present, content absent entirely. + let json = r#"{"candidates": [{"finishReason": "SAFETY", "index": 0}]}"#; + let response: GeminiResponse = serde_json::from_str(json).unwrap(); + assert_eq!(response.candidates.len(), 1); + assert!(response.candidates[0].content.is_none()); + + let err = GeminiClient::parse_response(response) + .unwrap_err() + .to_string(); + assert!(err.contains("no content"), "unexpected error: {err}"); + assert!(err.contains("SAFETY"), "finishReason missing: {err}"); + } + + #[test] + fn test_gemini_candidate_with_empty_parts_deserializes_and_errors() { + // MAX_TOKENS truncation on the 2.5 series: content with no parts. + let json = r#"{ + "candidates": [{"content": {"role": "model"}, "finishReason": "MAX_TOKENS"}], + "usageMetadata": {"promptTokenCount": 9, "candidatesTokenCount": 1024} + }"#; + let response: GeminiResponse = serde_json::from_str(json).unwrap(); + assert!(response.candidates[0] + .content + .as_ref() + .unwrap() + .parts + .is_empty()); + + let err = GeminiClient::parse_response(response) + .unwrap_err() + .to_string(); + assert!(err.contains("no content parts"), "unexpected error: {err}"); + assert!(err.contains("MAX_TOKENS"), "finishReason missing: {err}"); + } + + #[test] + fn test_gemini_candidate_missing_finish_reason_reports_unspecified() { + let json = r#"{"candidates": [{}]}"#; + let response: GeminiResponse = serde_json::from_str(json).unwrap(); + let err = GeminiClient::parse_response(response) + .unwrap_err() + .to_string(); + assert!(err.contains("unspecified"), "unexpected error: {err}"); } } diff --git a/crates/rvAgent/rvagent-backends/tests/live_anthropic_test.rs b/crates/rvAgent/rvagent-backends/tests/live_anthropic_test.rs index a945205a9..7aa310263 100644 --- a/crates/rvAgent/rvagent-backends/tests/live_anthropic_test.rs +++ b/crates/rvAgent/rvagent-backends/tests/live_anthropic_test.rs @@ -15,7 +15,10 @@ async fn test_live_anthropic_call() { let messages = vec![Message::human("What is 2+2? Reply with just the number.")]; - let response = client.complete(&messages).await.expect("API call failed"); + let response = client + .complete(&messages, &[]) + .await + .expect("API call failed"); let content = response.content(); println!("Response: {}", content); assert!( diff --git a/crates/rvAgent/rvagent-cli/src/app.rs b/crates/rvAgent/rvagent-cli/src/app.rs index 06057dd5e..2238b4077 100644 --- a/crates/rvAgent/rvagent-cli/src/app.rs +++ b/crates/rvAgent/rvagent-cli/src/app.rs @@ -10,10 +10,11 @@ use anyhow::{Context, Result}; use async_trait::async_trait; use tracing::{info, warn}; +use rvagent_core::bootstrap::EnvironmentSnapshot; use rvagent_core::config::{BackendConfig, MiddlewareConfig, RvAgentConfig, SecurityPolicy}; use rvagent_core::graph::{AgentGraph, ToolExecutor}; use rvagent_core::messages::{Message, ToolCall as CoreToolCall}; -use rvagent_core::models::{resolve_model, ChatModel}; +use rvagent_core::models::{resolve_model, ChatModel, ToolDefinition}; use rvagent_core::prompt::BASE_AGENT_PROMPT; use rvagent_core::state::AgentState; @@ -36,7 +37,9 @@ const DEFAULT_MIDDLEWARE: &[&str] = &[ "skills", "filesystem", "subagent", - "summarization", + // "summarization" removed (ADR-274): observation masking in the agent loop + // is the default compaction strategy. Still available opt-in via + // PipelineConfig::enable_summarization. "prompt_caching", "patch_tool_calls", "witness", @@ -66,7 +69,11 @@ impl StubModel { #[async_trait] impl ChatModel for StubModel { - async fn complete(&self, _messages: &[Message]) -> rvagent_core::error::Result { + async fn complete( + &self, + _messages: &[Message], + _tools: &[ToolDefinition], + ) -> rvagent_core::error::Result { Ok(Message::ai(format!( "No API key configured for model '{}'. \ Set the appropriate environment variable (e.g. ANTHROPIC_API_KEY) \ @@ -75,8 +82,12 @@ impl ChatModel for StubModel { ))) } - async fn stream(&self, messages: &[Message]) -> rvagent_core::error::Result> { - let msg = self.complete(messages).await?; + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> rvagent_core::error::Result> { + let msg = self.complete(messages, tools).await?; Ok(vec![msg]) } } @@ -95,19 +106,27 @@ enum CliModel { #[async_trait] impl ChatModel for CliModel { - async fn complete(&self, messages: &[Message]) -> rvagent_core::error::Result { + async fn complete( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> rvagent_core::error::Result { match self { - CliModel::Stub(m) => m.complete(messages).await, - CliModel::Anthropic(m) => m.complete(messages).await, - CliModel::Gemini(m) => m.complete(messages).await, + CliModel::Stub(m) => m.complete(messages, tools).await, + CliModel::Anthropic(m) => m.complete(messages, tools).await, + CliModel::Gemini(m) => m.complete(messages, tools).await, } } - async fn stream(&self, messages: &[Message]) -> rvagent_core::error::Result> { + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> rvagent_core::error::Result> { match self { - CliModel::Stub(m) => m.stream(messages).await, - CliModel::Anthropic(m) => m.stream(messages).await, - CliModel::Gemini(m) => m.stream(messages).await, + CliModel::Stub(m) => m.stream(messages, tools).await, + CliModel::Anthropic(m) => m.stream(messages, tools).await, + CliModel::Gemini(m) => m.stream(messages, tools).await, } } } @@ -125,9 +144,8 @@ struct CliToolExecutor { impl CliToolExecutor { fn new(cwd: &Path) -> Self { - let backend: rvagent_tools::BackendRef = Arc::new(LocalFsBackend { - cwd: cwd.to_path_buf(), - }); + // Confined to `cwd`: tool-supplied paths cannot escape the workspace. + let backend: rvagent_tools::BackendRef = Arc::new(rvagent_tools::LocalFsBackend::new(cwd)); Self { tools: rvagent_tools::builtin_tools(), backend, @@ -151,342 +169,17 @@ impl ToolExecutor for CliToolExecutor { None => Ok(format!("Error: tool '{}' not found", call.name)), } } -} -// --------------------------------------------------------------------------- -// LocalFsBackend — adapts the local filesystem for rvagent_tools::Backend -// --------------------------------------------------------------------------- - -/// A minimal filesystem backend implementing `rvagent_tools::Backend` for CLI use. -/// -/// Provides real filesystem and shell operations rooted at a working directory. -struct LocalFsBackend { - cwd: PathBuf, -} - -impl rvagent_tools::Backend for LocalFsBackend { - fn ls_info(&self, path: &str) -> std::result::Result, String> { - let target = if path.is_empty() || path == "." { - self.cwd.clone() - } else { - PathBuf::from(path) - }; - let entries = std::fs::read_dir(&target) - .map_err(|e| format!("ls failed on '{}': {}", target.display(), e))?; - let mut infos = Vec::new(); - for entry in entries { - let entry = entry.map_err(|e| format!("read_dir entry error: {}", e))?; - let meta = entry - .metadata() - .map_err(|e| format!("metadata error: {}", e))?; - let file_type = if meta.is_dir() { - "directory" - } else if meta.is_symlink() { - "symlink" - } else { - "file" - }; - infos.push(rvagent_tools::FileInfo { - name: entry.file_name().to_string_lossy().into_owned(), - file_type: file_type.to_string(), - permissions: String::new(), - size: meta.len(), - }); - } - infos.sort_by(|a, b| a.name.cmp(&b.name)); - Ok(infos) + fn definitions(&self) -> Vec { + self.tools + .iter() + .map(|t| ToolDefinition { + name: t.name().to_string(), + description: t.description().to_string(), + input_schema: t.parameters_schema(), + }) + .collect() } - - fn read(&self, path: &str, offset: usize, limit: usize) -> std::result::Result { - let content = - std::fs::read_to_string(path).map_err(|e| format!("read '{}': {}", path, e))?; - let lines: Vec<&str> = content.lines().collect(); - if offset >= lines.len() { - return Ok(String::new()); - } - let end = (offset + limit).min(lines.len()); - Ok(lines[offset..end].join("\n")) - } - - fn write(&self, path: &str, content: &str) -> rvagent_tools::WriteResult { - if std::path::Path::new(path).exists() { - return rvagent_tools::WriteResult { - error: Some(format!( - "Error: file {} already exists. Use force flag to overwrite.", - path - )), - ..Default::default() - }; - } - if let Some(parent) = std::path::Path::new(path).parent() { - if let Err(e) = std::fs::create_dir_all(parent) { - return rvagent_tools::WriteResult { - error: Some(format!("mkdir failed: {}", e)), - ..Default::default() - }; - } - } - match std::fs::write(path, content) { - Ok(_) => rvagent_tools::WriteResult::default(), - Err(e) => rvagent_tools::WriteResult { - error: Some(format!("write '{}': {}", path, e)), - ..Default::default() - }, - } - } - - fn edit( - &self, - path: &str, - old_string: &str, - new_string: &str, - replace_all: bool, - ) -> rvagent_tools::WriteResult { - let content = match std::fs::read_to_string(path) { - Ok(c) => c, - Err(e) => { - return rvagent_tools::WriteResult { - error: Some(format!("read '{}': {}", path, e)), - ..Default::default() - } - } - }; - let count = content.matches(old_string).count(); - if count == 0 { - return rvagent_tools::WriteResult { - error: Some(format!("Error: old_string not found in {}", path)), - ..Default::default() - }; - } - if count > 1 && !replace_all { - return rvagent_tools::WriteResult { - error: Some(format!( - "Error: old_string is not unique in {} ({} occurrences). Use replace_all=true.", - path, count - )), - ..Default::default() - }; - } - let new_content = if replace_all { - content.replace(old_string, new_string) - } else { - content.replacen(old_string, new_string, 1) - }; - match std::fs::write(path, &new_content) { - Ok(_) => rvagent_tools::WriteResult { - error: None, - occurrences: Some(if replace_all { count } else { 1 }), - ..Default::default() - }, - Err(e) => rvagent_tools::WriteResult { - error: Some(format!("write '{}': {}", path, e)), - ..Default::default() - }, - } - } - - fn glob_info(&self, pattern: &str, path: &str) -> std::result::Result, String> { - let base = if path.is_empty() || path == "." { - self.cwd.clone() - } else { - PathBuf::from(path) - }; - // Simple glob: walk directory and match by extension or name suffix. - // This handles common patterns like "*.rs", "**/*.toml" without - // requiring the `glob` crate. - let suffix = pattern - .trim_start_matches('*') - .trim_start_matches('/') - .trim_start_matches('*'); - let mut results = Vec::new(); - collect_glob_matches(&base, suffix, &mut results); - results.sort(); - Ok(results) - } - - fn grep_raw( - &self, - pattern: &str, - path: Option<&str>, - _include: Option<&str>, - ) -> std::result::Result, String> { - // Simple in-process grep implementation. - let search_dir = match path { - Some(p) if !p.is_empty() => PathBuf::from(p), - _ => self.cwd.clone(), - }; - let mut matches = Vec::new(); - if search_dir.is_file() { - grep_file(&search_dir, pattern, &mut matches)?; - } else if search_dir.is_dir() { - grep_dir(&search_dir, pattern, &mut matches)?; - } - Ok(matches) - } - - fn execute( - &self, - command: &str, - timeout_secs: u32, - ) -> std::result::Result { - use std::process::{Command, Stdio}; - use std::time::Duration; - - // Security: environment sanitization — strip sensitive variables (SEC-005 / ADR-103 C2). - // Only pass through a safe allowlist of environment variables. - const SAFE_ENV_VARS: &[&str] = &[ - "PATH", "HOME", "USER", "SHELL", "LANG", "LC_ALL", "LC_CTYPE", "TERM", "TMPDIR", "TZ", - ]; - // Patterns that identify sensitive env vars that must never reach child processes. - const SENSITIVE_PATTERNS: &[&str] = &[ - "SECRET", - "KEY", - "TOKEN", - "PASSWORD", - "CREDENTIAL", - "AWS_", - "AZURE_", - "GCP_", - "DATABASE_URL", - "PRIVATE", - "API_KEY", - "AUTH", - "BEARER", - "JWT", - "SESSION", - ]; - - let mut cmd = Command::new("sh"); - cmd.arg("-c").arg(command).current_dir(&self.cwd); - cmd.env_clear(); - for var in SAFE_ENV_VARS { - if let Ok(val) = std::env::var(var) { - let upper = var.to_uppercase(); - let sensitive = SENSITIVE_PATTERNS.iter().any(|pat| upper.contains(pat)); - if !sensitive { - cmd.env(var, val); - } - } - } - cmd.stdout(Stdio::piped()).stderr(Stdio::piped()); - - let timeout = if timeout_secs == 0 { 30 } else { timeout_secs }; - let deadline = std::time::Instant::now() + Duration::from_secs(timeout as u64); - - let mut child = cmd.spawn().map_err(|e| format!("execute failed: {}", e))?; - - // Poll for completion with a deadline to enforce the timeout. - loop { - match child - .try_wait() - .map_err(|e| format!("wait failed: {}", e))? - { - Some(_) => break, - None => { - if std::time::Instant::now() >= deadline { - let _ = child.kill(); - return Ok(rvagent_tools::ExecuteResponse { - output: format!("Command timed out after {} seconds", timeout), - exit_code: -1, - }); - } - std::thread::sleep(Duration::from_millis(50)); - } - } - } - - let output = child - .wait_with_output() - .map_err(|e| format!("output collection failed: {}", e))?; - - let stdout = String::from_utf8_lossy(&output.stdout); - let stderr = String::from_utf8_lossy(&output.stderr); - let mut combined = if stderr.is_empty() { - stdout.into_owned() - } else { - format!("{}\n{}", stdout, stderr) - }; - - // Security: cap output size to 1 MB to prevent memory exhaustion. - const MAX_OUTPUT_BYTES: usize = 1024 * 1024; - if combined.len() > MAX_OUTPUT_BYTES { - combined.truncate(MAX_OUTPUT_BYTES); - combined.push_str("\n... [output truncated at 1 MB]"); - } - - Ok(rvagent_tools::ExecuteResponse { - output: combined, - exit_code: output.status.code().unwrap_or(-1), - }) - } -} - -/// Recursively collect files matching a name suffix (simple glob substitute). -fn collect_glob_matches(dir: &Path, suffix: &str, results: &mut Vec) { - let entries = match std::fs::read_dir(dir) { - Ok(e) => e, - Err(_) => return, - }; - for entry in entries.flatten() { - let path = entry.path(); - let name = path - .file_name() - .map(|n| n.to_string_lossy().to_string()) - .unwrap_or_default(); - if path.is_file() && name.ends_with(suffix) { - results.push(path.to_string_lossy().into_owned()); - } else if path.is_dir() && !name.starts_with('.') { - collect_glob_matches(&path, suffix, results); - } - } -} - -/// Grep a single file for a pattern. -fn grep_file( - path: &Path, - pattern: &str, - matches: &mut Vec, -) -> std::result::Result<(), String> { - let content = match std::fs::read_to_string(path) { - Ok(c) => c, - Err(_) => return Ok(()), // skip binary / unreadable files - }; - for (i, line) in content.lines().enumerate() { - if line.contains(pattern) { - matches.push(rvagent_tools::GrepMatch { - file: path.to_string_lossy().into_owned(), - line_number: i + 1, - text: line.to_string(), - }); - } - } - Ok(()) -} - -/// Recursively grep a directory (limited depth). -fn grep_dir( - dir: &Path, - pattern: &str, - matches: &mut Vec, -) -> std::result::Result<(), String> { - let entries = std::fs::read_dir(dir).map_err(|e| format!("read_dir: {}", e))?; - for entry in entries { - let entry = entry.map_err(|e| format!("entry: {}", e))?; - let path = entry.path(); - if path.is_file() { - grep_file(&path, pattern, matches)?; - } else if path.is_dir() { - // Skip hidden directories. - let name = path - .file_name() - .map(|n| n.to_string_lossy().to_string()) - .unwrap_or_default(); - if !name.starts_with('.') { - grep_dir(&path, pattern, matches)?; - } - } - } - Ok(()) } // --------------------------------------------------------------------------- @@ -556,11 +249,18 @@ impl App { None => Session::new(model), }; + // Environment bootstrap (ADR-273 §3.5): hand the agent the workspace + // facts up front so it does not spend its first turns discovering + // them. Filesystem-only, so this costs nothing measurable. + let snapshot = + EnvironmentSnapshot::collect(cwd, &rvagent_core::bootstrap::BootstrapConfig::default()); + let system_prompt = snapshot.augment_prompt(BASE_AGENT_PROMPT); + Ok(Self { config, session, cwd: cwd.to_path_buf(), - system_prompt: BASE_AGENT_PROMPT.to_string(), + system_prompt, mcp_registry: McpRegistry::new(), }) } @@ -638,7 +338,8 @@ impl App { /// Invoke the agent pipeline with the given state. /// /// Creates the appropriate model (real Anthropic client or stub) and - /// tool executor, builds an `AgentGraph`, and runs it to completion. + /// tool executor, wraps the model in the configured middleware pipeline + /// (`PipelineModel`), builds an `AgentGraph`, and runs it to completion. /// Returns the final AI message from the completed state. async fn invoke_agent(&self, initial_state: &AgentState) -> Result { info!( @@ -707,9 +408,57 @@ impl App { CliModel::Stub(StubModel::new(&self.config.model)) }; + // Wire the middleware pipeline (P0.3): resolve the configured + // middleware names (DEFAULT_MIDDLEWARE) into instances — an unknown + // name is fatal — and run all model calls through it. + // + // The pipeline config carries the settings the middleware need to be + // built correctly; leaving `interrupt_on` unset gives HITL its + // conservative built-in gate rather than an empty (approve-everything) + // pattern list. The CLI has no interactive approval prompt yet, so + // gated calls fail closed; RVAGENT_AUTO_APPROVE=1 is the explicit, + // logged opt-out for unattended use. + let middleware_names: Vec<&str> = self + .config + .middleware + .iter() + .map(|m| m.name.as_str()) + .collect(); + let mut pipeline_config = rvagent_middleware::PipelineConfig::default(); + if matches!( + std::env::var("RVAGENT_AUTO_APPROVE").as_deref(), + Ok("1") | Ok("true") | Ok("yes") + ) { + // Straight to stderr, not just tracing: the TUI installs no + // subscriber and the non-TUI default is ERROR-only, so a `warn!` + // here is invisible in exactly the modes people run. A security + // downgrade the operator cannot see is one they cannot revoke. + eprintln!( + "warning: RVAGENT_AUTO_APPROVE set — HITL approval gate disabled; \ + all tool calls (including shell execution and file writes) run unattended" + ); + warn!("RVAGENT_AUTO_APPROVE set: HITL approval gate disabled; all tool calls run unattended"); + pipeline_config.interrupt_on = Some(Vec::new()); + } + let pipeline = Arc::new( + rvagent_middleware::build_pipeline_from_names(&middleware_names, &pipeline_config) + .context("failed to build middleware pipeline")?, + ); + info!(middlewares = ?pipeline.names(), "middleware pipeline wired"); + + // Run before_agent hooks (state patching, context injection). + let mut state = initial_state.clone(); + let mw_runtime = rvagent_middleware::Runtime::new(); + let run_config = rvagent_middleware::RunnableConfig::default(); + pipeline + .run_before_agent(&mut state, &mw_runtime, &run_config) + .await; + + let model = rvagent_middleware::PipelineModel::new(model, Arc::clone(&pipeline)); + let graph = AgentGraph::new(model, tool_executor); let completed_state = graph - .run(initial_state.clone()) + .run(state) .await .map_err(|e| anyhow::anyhow!("agent graph error: {}", e))?; @@ -767,7 +516,18 @@ mod tests { #[test] fn test_default_middleware_count() { - assert_eq!(DEFAULT_MIDDLEWARE.len(), 11); + // 10 since ADR-274 demoted summarization to opt-in. + assert_eq!(DEFAULT_MIDDLEWARE.len(), 10); + } + + #[test] + fn test_summarization_is_not_on_the_default_path() { + // The shipped default must match the decided strategy: masking in the + // agent loop, not LLM summarization. + assert!( + !DEFAULT_MIDDLEWARE.contains(&"summarization"), + "summarization is on the default path but ADR-274 decided against it" + ); } #[test] diff --git a/crates/rvAgent/rvagent-core/Cargo.toml b/crates/rvAgent/rvagent-core/Cargo.toml index b30e58960..365334ddc 100644 --- a/crates/rvAgent/rvagent-core/Cargo.toml +++ b/crates/rvAgent/rvagent-core/Cargo.toml @@ -24,6 +24,7 @@ sha3 = "0.10" rand = "0.8" [dev-dependencies] +tempfile = "3.14" criterion = { workspace = true } tokio = { workspace = true, features = ["test-util"] } proptest = { workspace = true } diff --git a/crates/rvAgent/rvagent-core/src/bootstrap.rs b/crates/rvAgent/rvagent-core/src/bootstrap.rs new file mode 100644 index 000000000..9281d2030 --- /dev/null +++ b/crates/rvAgent/rvagent-core/src/bootstrap.rs @@ -0,0 +1,333 @@ +//! Environment bootstrap — a workspace snapshot injected before the loop +//! starts (ADR-273 §3.5). +//! +//! Without it the agent spends its first turns discovering what it is looking +//! at: listing the directory, finding the test command, checking whether the +//! build is already broken. Those turns cost tokens, fill context, and produce +//! nothing the harness could not have supplied for free. +//! +//! The snapshot is deliberately small and factual. It is *not* a repo map or a +//! summary — those are lossy and expensive. It reports only what is cheap to +//! observe and expensive for the agent to discover. + +use std::fmt::Write as _; +use std::path::{Path, PathBuf}; + +/// A cheap, factual snapshot of the workspace. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct EnvironmentSnapshot { + /// Absolute working directory. + pub cwd: PathBuf, + /// Top-level entries, directories marked with a trailing `/`. + pub entries: Vec, + /// Detected project kind, e.g. "Rust (cargo)". + pub project_kind: Option, + /// Toolchain versions, as `(tool, version)`. + pub toolchain: Vec<(String, String)>, + /// The command to run tests, when it can be determined. + pub test_command: Option, + /// Whether the tree currently builds, when cheaply checkable. + /// + /// `None` means "not checked" — never guessed. Reporting a guess here + /// would be worse than reporting nothing, since the agent would trust it. + pub build_ok: Option, + /// Current VCS branch, when the workspace is a repository. + pub vcs_branch: Option, +} + +/// How much work the snapshot is allowed to do. +#[derive(Debug, Clone)] +pub struct BootstrapConfig { + /// Maximum top-level entries to list. + pub max_entries: usize, + /// Whether to include hidden entries. + pub include_hidden: bool, +} + +impl Default for BootstrapConfig { + fn default() -> Self { + Self { + max_entries: 50, + include_hidden: false, + } + } +} + +impl EnvironmentSnapshot { + /// Collect a snapshot of `root`. + /// + /// Filesystem-only: nothing here spawns a process, so it is fast and safe + /// to run unconditionally. `build_ok` is left `None` — a build check is a + /// caller decision, since it costs real time. + pub fn collect(root: &Path, config: &BootstrapConfig) -> Self { + let entries = list_entries(root, config); + let project_kind = detect_project_kind(&entries); + let test_command = detect_test_command(&entries); + let vcs_branch = detect_branch(root); + + Self { + cwd: root.to_path_buf(), + entries, + project_kind, + toolchain: Vec::new(), + test_command, + build_ok: None, + vcs_branch, + } + } + + /// Render as a prompt section. + /// + /// Returns `None` when there is nothing worth saying, so an empty or + /// unreadable workspace does not inject a misleading stub. + pub fn to_prompt_section(&self) -> Option { + if self.entries.is_empty() && self.project_kind.is_none() { + return None; + } + + let mut out = String::from("\n"); + let _ = writeln!(out, "cwd: {}", self.cwd.display()); + + if let Some(kind) = &self.project_kind { + let _ = writeln!(out, "project: {kind}"); + } + if let Some(branch) = &self.vcs_branch { + let _ = writeln!(out, "branch: {branch}"); + } + if let Some(cmd) = &self.test_command { + let _ = writeln!(out, "tests: {cmd}"); + } + match self.build_ok { + Some(true) => { + let _ = writeln!(out, "build: passing"); + } + Some(false) => { + let _ = writeln!(out, "build: FAILING before any of your changes"); + } + // Silence is correct here: an unchecked build must not read as passing. + None => {} + } + for (tool, version) in &self.toolchain { + let _ = writeln!(out, "{tool}: {version}"); + } + if !self.entries.is_empty() { + let _ = writeln!(out, "contents: {}", self.entries.join(", ")); + } + out.push_str(""); + Some(out) + } + + /// Append the environment section to a base system prompt. + /// + /// Returns `base` unchanged when there is nothing to report, so callers + /// need no conditional of their own. Kept here rather than at the call site + /// so the composition is covered by tests — the CLI is a binary crate and + /// anything assembled there is unreachable from a test. + pub fn augment_prompt(&self, base: &str) -> String { + match self.to_prompt_section() { + Some(section) => format!("{base}\n\n{section}"), + None => base.to_string(), + } + } +} + +fn list_entries(root: &Path, config: &BootstrapConfig) -> Vec { + let Ok(read) = std::fs::read_dir(root) else { + return Vec::new(); + }; + let mut names: Vec = read + .flatten() + .filter_map(|e| { + let name = e.file_name().to_string_lossy().into_owned(); + if !config.include_hidden && name.starts_with('.') { + return None; + } + let is_dir = e.file_type().map(|t| t.is_dir()).unwrap_or(false); + Some(if is_dir { format!("{name}/") } else { name }) + }) + .collect(); + names.sort(); + // Say so rather than silently showing a partial list: a truncated listing + // that looks complete invites the agent to conclude a file is absent. + if names.len() > config.max_entries { + let hidden = names.len() - config.max_entries; + names.truncate(config.max_entries); + names.push(format!("… and {hidden} more")); + } + names +} + +fn has(entries: &[String], name: &str) -> bool { + entries.iter().any(|e| e == name) +} + +fn detect_project_kind(entries: &[String]) -> Option { + if has(entries, "Cargo.toml") { + Some("Rust (cargo)".into()) + } else if has(entries, "package.json") { + Some("Node (npm)".into()) + } else if has(entries, "pyproject.toml") || has(entries, "setup.py") { + Some("Python".into()) + } else if has(entries, "go.mod") { + Some("Go".into()) + } else { + None + } +} + +fn detect_test_command(entries: &[String]) -> Option { + if has(entries, "Cargo.toml") { + Some("cargo test".into()) + } else if has(entries, "package.json") { + Some("npm test".into()) + } else if has(entries, "pyproject.toml") || has(entries, "setup.py") { + Some("pytest".into()) + } else if has(entries, "go.mod") { + Some("go test ./...".into()) + } else { + None + } +} + +/// Read the current branch from `.git/HEAD` without shelling out. +fn detect_branch(root: &Path) -> Option { + let head = std::fs::read_to_string(root.join(".git").join("HEAD")).ok()?; + let head = head.trim(); + head.strip_prefix("ref: refs/heads/") + .map(str::to_string) + // A detached HEAD is a raw sha; report a short form rather than nothing. + .or_else(|| Some(head.chars().take(12).collect())) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn write(dir: &Path, name: &str, body: &str) { + std::fs::write(dir.join(name), body).unwrap(); + } + + #[test] + fn detects_a_rust_project_and_its_test_command() { + let dir = tempfile::tempdir().unwrap(); + write(dir.path(), "Cargo.toml", "[package]"); + std::fs::create_dir(dir.path().join("src")).unwrap(); + + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + assert_eq!(snap.project_kind.as_deref(), Some("Rust (cargo)")); + assert_eq!(snap.test_command.as_deref(), Some("cargo test")); + assert!(snap.entries.contains(&"src/".to_string())); + assert!(snap.entries.contains(&"Cargo.toml".to_string())); + } + + #[test] + fn renders_a_prompt_section() { + let dir = tempfile::tempdir().unwrap(); + write(dir.path(), "package.json", "{}"); + + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + let section = snap.to_prompt_section().unwrap(); + assert!(section.starts_with("")); + assert!(section.ends_with("")); + assert!(section.contains("project: Node (npm)")); + assert!(section.contains("tests: npm test")); + } + + #[test] + fn unchecked_build_is_silent_not_passing() { + let dir = tempfile::tempdir().unwrap(); + write(dir.path(), "Cargo.toml", "[package]"); + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + assert_eq!(snap.build_ok, None); + let section = snap.to_prompt_section().unwrap(); + assert!( + !section.contains("build:"), + "an unchecked build must not be reported at all: {section}" + ); + } + + #[test] + fn failing_build_is_stated_plainly() { + let dir = tempfile::tempdir().unwrap(); + write(dir.path(), "Cargo.toml", "[package]"); + let mut snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + snap.build_ok = Some(false); + let section = snap.to_prompt_section().unwrap(); + assert!(section.contains("build: FAILING before any of your changes")); + } + + #[test] + fn empty_workspace_yields_no_section() { + let dir = tempfile::tempdir().unwrap(); + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + assert!(snap.to_prompt_section().is_none()); + } + + #[test] + fn truncation_is_announced() { + let dir = tempfile::tempdir().unwrap(); + for i in 0..30 { + write(dir.path(), &format!("f{i:02}.txt"), ""); + } + let config = BootstrapConfig { + max_entries: 10, + ..BootstrapConfig::default() + }; + let snap = EnvironmentSnapshot::collect(dir.path(), &config); + assert_eq!(snap.entries.len(), 11); + assert!(snap.entries.last().unwrap().contains("and 20 more")); + } + + #[test] + fn hidden_entries_are_excluded_by_default() { + let dir = tempfile::tempdir().unwrap(); + write(dir.path(), ".secret", ""); + write(dir.path(), "visible.txt", ""); + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + assert!(!snap.entries.iter().any(|e| e.starts_with('.'))); + assert!(snap.entries.contains(&"visible.txt".to_string())); + } + + #[test] + fn reads_the_branch_from_git_head() { + let dir = tempfile::tempdir().unwrap(); + std::fs::create_dir(dir.path().join(".git")).unwrap(); + write( + &dir.path().join(".git"), + "HEAD", + "ref: refs/heads/feature/x\n", + ); + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + assert_eq!(snap.vcs_branch.as_deref(), Some("feature/x")); + } + + #[test] + fn augment_prompt_appends_the_section() { + let dir = tempfile::tempdir().unwrap(); + write(dir.path(), "Cargo.toml", "[package]"); + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + + let prompt = snap.augment_prompt("BASE PROMPT"); + assert!(prompt.starts_with("BASE PROMPT")); + assert!(prompt.contains("")); + assert!(prompt.contains("tests: cargo test")); + } + + #[test] + fn augment_prompt_is_a_noop_when_there_is_nothing_to_say() { + let dir = tempfile::tempdir().unwrap(); + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + // No trailing whitespace, no empty stub — byte-identical to the input. + assert_eq!(snap.augment_prompt("BASE PROMPT"), "BASE PROMPT"); + } + + #[test] + fn unreadable_workspace_does_not_panic() { + let snap = EnvironmentSnapshot::collect( + Path::new("/nonexistent/path/xyz"), + &BootstrapConfig::default(), + ); + assert!(snap.entries.is_empty()); + assert!(snap.to_prompt_section().is_none()); + } +} diff --git a/crates/rvAgent/rvagent-core/src/config.rs b/crates/rvAgent/rvagent-core/src/config.rs index 84f2a8f8e..b284c4134 100644 --- a/crates/rvAgent/rvagent-core/src/config.rs +++ b/crates/rvAgent/rvagent-core/src/config.rs @@ -169,6 +169,24 @@ fn default_backend_type() -> String { "local_shell".into() } +// --------------------------------------------------------------------------- +// Runnable config (per-run context — canonical definition, ADR-103 A1) +// --------------------------------------------------------------------------- + +/// Configuration for a single runnable invocation (thread/run IDs, metadata). +/// +/// This is the canonical definition shared by the middleware pipeline and any +/// other layer that needs per-run context. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct RunnableConfig { + #[serde(default)] + pub thread_id: Option, + #[serde(default)] + pub run_id: Option, + #[serde(default)] + pub metadata: std::collections::HashMap, +} + // --------------------------------------------------------------------------- // Top-level config // --------------------------------------------------------------------------- diff --git a/crates/rvAgent/rvagent-core/src/graph.rs b/crates/rvAgent/rvagent-core/src/graph.rs index 2f45bad77..66e1fa9ab 100644 --- a/crates/rvAgent/rvagent-core/src/graph.rs +++ b/crates/rvAgent/rvagent-core/src/graph.rs @@ -2,13 +2,20 @@ //! //! Implements the core agent loop: Agent → check tool_calls → execute tools → loop. +use std::sync::Arc; + use async_trait::async_trait; use serde::{Deserialize, Serialize}; use tracing::{debug, info, instrument, warn}; use crate::error::{Result, RvAgentError}; +use crate::masking::{ + mask_observations, recall, recall_definition, recall_id_from_args, truncate_tool_result, + MaskConfig, RECALL_TOOL, +}; use crate::messages::{Message, ToolCall}; -use crate::models::ChatModel; +use crate::models::{ChatModel, ToolDefinition}; +use crate::parallel::parallel_execute_limited; use crate::state::AgentState; // --------------------------------------------------------------------------- @@ -47,6 +54,15 @@ pub struct Edge { pub trait ToolExecutor: Send + Sync { /// Execute a single tool call and return the result content. async fn execute(&self, call: &ToolCall, state: &AgentState) -> Result; + + /// Schemas for the tools this executor can dispatch. + /// + /// These are advertised to the model on every completion. The default is + /// empty (pure-chat agents), but any executor that dispatches real tools + /// MUST override this — otherwise the model can never call them. + fn definitions(&self) -> Vec { + Vec::new() + } } // --------------------------------------------------------------------------- @@ -60,6 +76,15 @@ pub struct GraphConfig { pub max_iterations: u32, /// Whether to execute tool calls in parallel (ADR-103 A2). pub parallel_tools: bool, + /// Maximum tool calls in flight at once when `parallel_tools` is set. + pub max_parallel_tools: usize, + /// How many times an identical `(tool, args)` pair may repeat + /// consecutively before the call is refused instead of executed. + /// + /// Set to 0 to disable loop detection entirely. + pub loop_repeat_threshold: usize, + /// Observation masking and tool-output caps (ADR-274). + pub mask: MaskConfig, } impl Default for GraphConfig { @@ -67,10 +92,91 @@ impl Default for GraphConfig { Self { max_iterations: 100, parallel_tools: true, + max_parallel_tools: 8, + loop_repeat_threshold: 3, + mask: MaskConfig::default(), } } } +/// Detects a stuck agent repeating the same tool call. +/// +/// A stuck agent repeats one call forever; raising `max_iterations` only makes +/// that more expensive. Refusing the repeat and telling the model *why* is what +/// breaks the cycle. +/// +/// Repeats are counted **consecutively**, not across a window. An agent that +/// re-runs `cargo test` between edits is doing legitimate work, and refusing +/// that would be worse than the loop it prevents — so only an unbroken run of +/// identical calls trips the detector. +/// +/// Known limitation: alternating cycles (A, B, A, B, ...) are not detected. +/// `max_iterations` remains the backstop for those. +#[derive(Debug)] +struct LoopDetector { + last: Option, + consecutive: usize, + threshold: usize, +} + +impl LoopDetector { + fn new(threshold: usize) -> Self { + Self { + last: None, + consecutive: 0, + threshold, + } + } + + fn enabled(&self) -> bool { + self.threshold > 0 + } + + /// Fingerprint a call by name and arguments. + /// + /// Arguments are hashed via their JSON string, which is canonical for key + /// ordering because `serde_json` maps are `BTreeMap` by default — so + /// logically identical calls collide regardless of the order the model + /// emitted the keys in. + fn fingerprint(call: &ToolCall) -> u64 { + use std::hash::{Hash, Hasher}; + let mut hasher = std::collections::hash_map::DefaultHasher::new(); + call.name.hash(&mut hasher); + call.args.to_string().hash(&mut hasher); + hasher.finish() + } + + /// Record a call and report whether it has now repeated too often. + fn observe(&mut self, call: &ToolCall) -> bool { + if !self.enabled() { + return false; + } + let fp = Self::fingerprint(call); + if self.last == Some(fp) { + self.consecutive += 1; + } else { + self.last = Some(fp); + self.consecutive = 1; + } + self.consecutive >= self.threshold + } +} + +/// The message substituted for a call that tripped loop detection. +/// +/// Tool errors must be actionable rather than opaque — the model has to be +/// told what to do differently, or it repeats the call again. +fn loop_break_message(call: &ToolCall, threshold: usize) -> String { + format!( + "Tool execution refused: this exact call to '{}' with identical \ + arguments has already been made {} times without progress, so it was \ + not executed again. Repeating it will not produce a different result. \ + Change the arguments, use a different tool, or explain what is \ + blocking you.", + call.name, threshold + ) +} + /// The agent execution graph. /// /// Implements the core loop: @@ -79,14 +185,16 @@ impl Default for GraphConfig { /// ├── yes → Tools → Agent (loop) /// └── no → End /// ``` -pub struct AgentGraph { +pub struct AgentGraph { model: M, - tool_executor: T, + // Arc so tool calls can be spawned onto the runtime for true parallel + // execution (ADR-103 A2) — spawned futures must be 'static. + tool_executor: Arc, config: GraphConfig, edges: Vec, } -impl AgentGraph { +impl AgentGraph { /// Create a new agent graph with the given model and tool executor. pub fn new(model: M, tool_executor: T) -> Self { Self::with_config(model, tool_executor, GraphConfig::default()) @@ -119,7 +227,7 @@ impl AgentGraph { Self { model, - tool_executor, + tool_executor: Arc::new(tool_executor), config, edges, } @@ -139,18 +247,24 @@ impl AgentGraph { pub async fn run(&self, mut state: AgentState) -> Result { let mut current_node = AgentNode::Start; let mut iterations: u32 = 0; + // Tool schemas advertised to the model on every completion. Without + // these the model can never emit a tool call. + let mut tool_definitions = self.tool_executor.definitions(); + // Masked observations are useless without a way to dereference them, + // so recall is advertised exactly when masking is active (ADR-274 §3.2). + if self.config.mask.masking_enabled() { + tool_definitions.push(recall_definition()); + } + // Cumulative token usage across the loop, aggregated from per-message + // usage metadata attached by provider backends. + let mut total_input_tokens: u64 = 0; + let mut total_output_tokens: u64 = 0; + // Spans the whole run: a loop is only visible across iterations. + let mut loop_detector = LoopDetector::new(self.config.loop_repeat_threshold); - info!(node = ?current_node, "graph: starting agent loop"); + info!(node = ?current_node, tools = tool_definitions.len(), "graph: starting agent loop"); loop { - if iterations >= self.config.max_iterations { - warn!(iterations, "graph: max iterations reached"); - return Err(RvAgentError::timeout(format!( - "agent loop exceeded {} iterations", - self.config.max_iterations - ))); - } - match current_node { AgentNode::Start => { debug!("graph: Start → Agent"); @@ -158,10 +272,38 @@ impl AgentGraph { } AgentNode::Agent => { + // The budget is checked here rather than at the top of the + // loop so that reaching End *within* the budget completes + // the run. Checking before every node discarded a finished + // run whose last allowed iteration produced the answer — + // with `max_iterations: 1`, every successful single-turn + // run failed. + if iterations >= self.config.max_iterations { + warn!(iterations, "graph: max iterations reached"); + return Err(RvAgentError::timeout(format!( + "agent loop exceeded {} iterations", + self.config.max_iterations + ))); + } iterations += 1; debug!(iteration = iterations, "graph: invoking model"); - let response = self.model.complete(&state.messages).await?; + // The model sees a masked projection; `state.messages` + // stays the complete log so elided content remains + // addressable by tool_call_id (ADR-274). + let outbound = mask_observations(&state.messages, &self.config.mask); + let response = self.model.complete(&outbound, &tool_definitions).await?; + if let Some((input, output)) = usage_from_message(&response) { + total_input_tokens += input; + total_output_tokens += output; + debug!( + input_tokens = input, + output_tokens = output, + total_input_tokens, + total_output_tokens, + "graph: turn usage" + ); + } let has_tool_calls = response.has_tool_calls(); state.push_message(response); @@ -180,31 +322,147 @@ impl AgentGraph { // Extract tool calls from the last AI message. let tool_calls = self.extract_tool_calls(&state)?; - if self.config.parallel_tools && tool_calls.len() > 1 { - // Parallel execution (ADR-103 A2). - let mut handles = Vec::with_capacity(tool_calls.len()); - for tc in &tool_calls { - let result = self.tool_executor.execute(tc, &state).await; - handles.push((tc.id.clone(), result)); + // Refuse calls that are repeating without progress. This + // happens before dispatch so a stuck agent stops burning + // tokens and side effects on the same call. + let mut looping: Vec> = Vec::with_capacity(tool_calls.len()); + for tc in &tool_calls { + looping.push(if loop_detector.observe(tc) { + warn!(tool = %tc.name, "graph: loop detected, refusing repeated call"); + Some(loop_break_message(tc, self.config.loop_repeat_threshold)) + } else { + None + }); + } + // Recall is served by the loop from the full log; it never + // reaches the executor, so a workspace tool cannot shadow it. + let mut handled: std::collections::HashMap = + std::collections::HashMap::new(); + for (tc, refused) in tool_calls.iter().zip(&looping) { + if refused.is_none() && tc.name == RECALL_TOOL { + let content = match recall_id_from_args(&tc.args) { + Ok(id) => recall(&state.messages, id), + Err(e) => e, + }; + handled.insert(tc.id.clone(), content); } - for (id, result) in handles { - let content = result?; - state.push_message(Message::tool(id, content)); + } + + // Only calls that cleared loop detection and were not + // handled in-loop reach the executor. + let dispatch: Vec = tool_calls + .iter() + .zip(&looping) + .filter(|(tc, refused)| refused.is_none() && !handled.contains_key(&tc.id)) + .map(|(tc, _)| tc.clone()) + .collect(); + + // Tool failures are fed back to the model as tool results + // rather than aborting the loop — the model must see the + // error to recover from it (execution alignment). + let mut executed: std::collections::HashMap = handled; + executed.reserve(dispatch.len()); + if self.config.parallel_tools && dispatch.len() > 1 { + // True parallel execution (ADR-103 A2): tasks are + // spawned onto the runtime with bounded concurrency; + // results are returned in input order. + let executor = Arc::clone(&self.tool_executor); + let exec_state = state.clone(); + let results = parallel_execute_limited( + dispatch, + move |tc: ToolCall| { + let executor = Arc::clone(&executor); + let exec_state = exec_state.clone(); + async move { + let id = tc.id.clone(); + let name = tc.name.clone(); + // Run the tool in its own task so a panicking + // tool surfaces as a tool error instead of + // crashing the whole agent loop. + let result = match tokio::spawn(async move { + executor.execute(&tc, &exec_state).await + }) + .await + { + Ok(res) => res, + Err(join_err) => Err(RvAgentError::tool(format!( + "tool '{name}' execution task failed: {join_err}" + ))), + }; + (id, name, result) + } + }, + self.config.max_parallel_tools.max(1), + ) + .await; + for (id, _name, result) in results { + executed.insert( + id, + truncate_tool_result( + tool_result_content(result), + self.config.mask.max_tool_result_bytes, + ), + ); } } else { - // Sequential execution. - for tc in &tool_calls { - let content = self.tool_executor.execute(tc, &state).await?; - state.push_message(Message::tool(&tc.id, content)); + // Sequential execution. Each call still runs in its own + // task, so a panicking tool becomes a tool error here + // exactly as it does on the parallel path — a panic on + // the single-call path would otherwise take down the + // whole agent loop. + for tc in &dispatch { + let executor = Arc::clone(&self.tool_executor); + let exec_state = state.clone(); + let call = tc.clone(); + let name = tc.name.clone(); + let result = match tokio::spawn(async move { + executor.execute(&call, &exec_state).await + }) + .await + { + Ok(res) => res, + Err(join_err) => Err(RvAgentError::tool(format!( + "tool '{name}' execution task failed: {join_err}" + ))), + }; + executed.insert( + tc.id.clone(), + truncate_tool_result( + tool_result_content(result), + self.config.mask.max_tool_result_bytes, + ), + ); } } + // Emit one tool result per call, in the model's original + // call order, substituting the refusal for looping calls. + for (tc, refused) in tool_calls.iter().zip(looping) { + let content = match refused { + Some(msg) => msg, + None => executed.remove(&tc.id).unwrap_or_else(|| { + // Defensive: a dispatched call must always + // produce a result. Report rather than drop it, + // since a missing tool result desyncs the + // provider's tool_use/tool_result pairing. + format!( + "Tool execution error: no result produced for '{}'", + tc.name + ) + }), + }; + state.push_message(Message::tool_with_name(&tc.id, content, &tc.name)); + } + debug!("graph: Tools → Agent"); current_node = AgentNode::Agent; } AgentNode::End => { - info!(iterations, "graph: agent loop complete"); + info!( + iterations, + total_input_tokens, total_output_tokens, "graph: agent loop complete" + ); return Ok(state); } } @@ -226,6 +484,30 @@ impl AgentGraph { } } +/// Convert a tool execution result into tool-message content. +/// +/// Errors become visible tool output instead of aborting the loop — feeding +/// the failure back to the model is the recovery path. +fn tool_result_content(result: Result) -> String { + match result { + Ok(content) => content, + Err(e) => format!("Tool execution error: {e}"), + } +} + +/// Extract `(input_tokens, output_tokens)` from a message's usage metadata, +/// as attached by provider backends under the `usage` key. +fn usage_from_message(msg: &Message) -> Option<(u64, u64)> { + if let Message::Ai(ai) = msg { + let usage = ai.metadata.get("usage")?; + let input = usage.get("input_tokens").and_then(|v| v.as_u64())?; + let output = usage.get("output_tokens").and_then(|v| v.as_u64())?; + Some((input, output)) + } else { + None + } +} + #[cfg(test)] mod tests { use super::*; @@ -246,7 +528,11 @@ mod tests { #[async_trait] impl ChatModel for MockModel { - async fn complete(&self, _messages: &[Message]) -> Result { + async fn complete( + &self, + _messages: &[Message], + _tools: &[ToolDefinition], + ) -> Result { let mut resps = self.responses.lock().unwrap(); if resps.is_empty() { Ok(Message::ai("done")) @@ -255,8 +541,12 @@ mod tests { } } - async fn stream(&self, messages: &[Message]) -> Result> { - let msg = self.complete(messages).await?; + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> Result> { + let msg = self.complete(messages, tools).await?; Ok(vec![msg]) } } @@ -308,6 +598,115 @@ mod tests { assert_eq!(result.message_count(), 4); } + /// A tool executor that panics on one tool and succeeds on others. + struct PanickyExecutor; + + #[async_trait] + impl ToolExecutor for PanickyExecutor { + async fn execute(&self, call: &ToolCall, _state: &AgentState) -> Result { + if call.name == "boom" { + panic!("tool panicked"); + } + Ok(format!("result of {}", call.name)) + } + } + + #[tokio::test] + async fn test_parallel_tool_panic_is_contained() { + let model = MockModel::new(vec![ + Message::ai_with_tools( + "", + vec![ + ToolCall { + id: "tc1".into(), + name: "boom".into(), + args: serde_json::json!({}), + }, + ToolCall { + id: "tc2".into(), + name: "ok_tool".into(), + args: serde_json::json!({}), + }, + ], + ), + Message::ai("done"), + ]); + let graph = AgentGraph::new(model, PanickyExecutor); + + let state = AgentState::new(); + // Must not panic: the panicking tool becomes an error tool-result. + let result = graph.run(state).await.unwrap(); + + let contents: Vec<&str> = result + .messages + .iter() + .filter_map(|m| match m { + Message::Tool(t) => Some(t.content.as_str()), + _ => None, + }) + .collect(); + assert_eq!(contents.len(), 2); + assert!(contents[0].contains("Tool execution error")); + assert!(contents[1].contains("result of ok_tool")); + } + + #[tokio::test] + async fn test_single_tool_panic_is_contained() { + // The single-call path runs sequentially even with parallel_tools on, + // so it needs its own containment — a panic here used to abort the run. + let model = MockModel::new(vec![ + Message::ai_with_tools( + "", + vec![ToolCall { + id: "tc1".into(), + name: "boom".into(), + args: serde_json::json!({}), + }], + ), + Message::ai("done"), + ]); + let graph = AgentGraph::with_config( + model, + PanickyExecutor, + GraphConfig { + parallel_tools: false, + ..GraphConfig::default() + }, + ); + + let result = graph.run(AgentState::new()).await.unwrap(); + + let contents: Vec<&str> = result + .messages + .iter() + .filter_map(|m| match m { + Message::Tool(t) => Some(t.content.as_str()), + _ => None, + }) + .collect(); + assert_eq!(contents.len(), 1); + assert!(contents[0].contains("Tool execution error")); + } + + #[tokio::test] + async fn test_max_iterations_of_one_allows_a_single_turn() { + // A run that answers on its last allowed iteration has not exceeded the + // budget; discarding it made max_iterations: 1 unusable. + let model = MockModel::new(vec![Message::ai("Hello!")]); + let graph = AgentGraph::with_config( + model, + MockToolExecutor, + GraphConfig { + max_iterations: 1, + ..GraphConfig::default() + }, + ); + + let result = graph.run(AgentState::with_system_message("sys")).await; + let state = result.expect("single-turn run within budget must succeed"); + assert!(matches!(state.messages.last(), Some(Message::Ai(_)))); + } + #[tokio::test] async fn test_max_iterations() { // Model always returns tool calls → should hit max iterations. @@ -328,6 +727,7 @@ mod tests { let config = GraphConfig { max_iterations: 3, parallel_tools: false, + ..GraphConfig::default() }; let graph = AgentGraph::with_config(model, executor, config); @@ -336,6 +736,202 @@ mod tests { assert!(matches!(err, RvAgentError::Timeout(_))); } + /// Counts how many times the executor was actually invoked. + struct CountingExecutor { + calls: Arc, + } + + #[async_trait] + impl ToolExecutor for CountingExecutor { + async fn execute(&self, call: &ToolCall, _state: &AgentState) -> Result { + self.calls.fetch_add(1, std::sync::atomic::Ordering::SeqCst); + Ok(format!("result of {}", call.name)) + } + } + + fn repeated_call_model(n: usize, args: serde_json::Value) -> MockModel { + let mut responses: Vec = (0..n) + .map(|i| { + Message::ai_with_tools( + "", + vec![ToolCall { + // Distinct ids, identical name+args — a real stuck loop + // looks exactly like this. + id: format!("tc{i}"), + name: "noop".into(), + args: args.clone(), + }], + ) + }) + .collect(); + responses.push(Message::ai("done")); + MockModel::new(responses) + } + + #[tokio::test] + async fn test_repeated_identical_call_is_refused() { + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let graph = AgentGraph::with_config( + repeated_call_model(5, serde_json::json!({"x": 1})), + CountingExecutor { + calls: Arc::clone(&calls), + }, + GraphConfig { + max_iterations: 20, + parallel_tools: false, + loop_repeat_threshold: 3, + ..GraphConfig::default() + }, + ); + let result = graph.run(AgentState::new()).await.unwrap(); + + // The 3rd identical call and everything after it must be refused, so + // the executor sees exactly 2 invocations. + assert_eq!( + calls.load(std::sync::atomic::Ordering::SeqCst), + 2, + "loop detection did not stop execution" + ); + + let tool_msgs: Vec<&str> = result + .messages + .iter() + .filter_map(|m| match m { + Message::Tool(t) => Some(t.content.as_str()), + _ => None, + }) + .collect(); + // Every call still gets exactly one result — dropping one would desync + // the provider's tool_use/tool_result pairing. + assert_eq!(tool_msgs.len(), 5); + assert!(tool_msgs[0].contains("result of noop")); + assert!(tool_msgs[1].contains("result of noop")); + for refused in &tool_msgs[2..] { + assert!( + refused.contains("Tool execution refused"), + "expected refusal, got: {refused}" + ); + // The refusal must tell the model what to do differently. + assert!(refused.contains("Change the arguments")); + } + } + + #[tokio::test] + async fn test_differing_args_are_not_treated_as_a_loop() { + // Same tool, different arguments each time: legitimate work. + let mut responses: Vec = (0..5) + .map(|i| { + Message::ai_with_tools( + "", + vec![ToolCall { + id: format!("tc{i}"), + name: "noop".into(), + args: serde_json::json!({ "x": i }), + }], + ) + }) + .collect(); + responses.push(Message::ai("done")); + + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let graph = AgentGraph::with_config( + MockModel::new(responses), + CountingExecutor { + calls: Arc::clone(&calls), + }, + GraphConfig { + max_iterations: 20, + parallel_tools: false, + ..GraphConfig::default() + }, + ); + graph.run(AgentState::new()).await.unwrap(); + assert_eq!(calls.load(std::sync::atomic::Ordering::SeqCst), 5); + } + + #[tokio::test] + async fn test_interleaved_repeats_are_not_refused() { + // The false positive that drove consecutive-only counting: re-running + // the same check between edits is legitimate and must not be blocked. + let check = ToolCall { + id: "c".into(), + name: "run_tests".into(), + args: serde_json::json!({}), + }; + let mut responses = Vec::new(); + for i in 0..4 { + responses.push(Message::ai_with_tools( + "", + vec![ToolCall { + id: format!("e{i}"), + name: "edit".into(), + args: serde_json::json!({ "line": i }), + }], + )); + responses.push(Message::ai_with_tools( + "", + vec![ToolCall { + id: format!("c{i}"), + ..check.clone() + }], + )); + } + responses.push(Message::ai("done")); + + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let graph = AgentGraph::with_config( + MockModel::new(responses), + CountingExecutor { + calls: Arc::clone(&calls), + }, + GraphConfig { + max_iterations: 30, + parallel_tools: false, + ..GraphConfig::default() + }, + ); + graph.run(AgentState::new()).await.unwrap(); + assert_eq!( + calls.load(std::sync::atomic::Ordering::SeqCst), + 8, + "legitimate interleaved re-runs were refused" + ); + } + + #[tokio::test] + async fn test_loop_detection_can_be_disabled() { + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let graph = AgentGraph::with_config( + repeated_call_model(4, serde_json::json!({})), + CountingExecutor { + calls: Arc::clone(&calls), + }, + GraphConfig { + max_iterations: 20, + parallel_tools: false, + loop_repeat_threshold: 0, + ..GraphConfig::default() + }, + ); + graph.run(AgentState::new()).await.unwrap(); + assert_eq!(calls.load(std::sync::atomic::Ordering::SeqCst), 4); + } + + #[test] + fn test_fingerprint_ignores_call_id_and_key_order() { + let a = ToolCall { + id: "one".into(), + name: "t".into(), + args: serde_json::json!({"a": 1, "b": 2}), + }; + let b = ToolCall { + id: "two".into(), + name: "t".into(), + args: serde_json::json!({"b": 2, "a": 1}), + }; + assert_eq!(LoopDetector::fingerprint(&a), LoopDetector::fingerprint(&b)); + } + #[test] fn test_graph_edges() { let model = MockModel::new(vec![]); diff --git a/crates/rvAgent/rvagent-core/src/invariants.rs b/crates/rvAgent/rvagent-core/src/invariants.rs new file mode 100644 index 000000000..d7c39b592 --- /dev/null +++ b/crates/rvAgent/rvagent-core/src/invariants.rs @@ -0,0 +1,184 @@ +//! Invariants that survive compaction verbatim (ADR-274 §2.2). +//! +//! Safety constraints and task statements **erode through successive compaction +//! cycles with no failure signal** — a documented mechanism ("governance +//! decay"), not a jailbreak. Each summarization pass paraphrases a little more +//! away until a rule that was explicit at turn 1 is gone by turn 200, and +//! nothing in the transcript marks the moment it disappeared. +//! +//! The fix is cheap and absolute: a small set of invariants is re-emitted +//! **byte-identical** after every compaction. They are never inputs to a +//! summarizer, never masked (ADR-274 §3.1), and never paraphrased. +//! +//! Keep this set small. Everything here is paid for on every turn after a +//! compaction, and a bloated invariant set recreates the context pressure +//! compaction exists to relieve. + +use crate::messages::Message; + +/// A rule that must never be summarized away. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Invariant { + /// Short stable label, for debugging and dedup. + pub id: String, + /// The exact text to re-emit. Reproduced byte-for-byte. + pub text: String, +} + +impl Invariant { + pub fn new(id: impl Into, text: impl Into) -> Self { + Self { + id: id.into(), + text: text.into(), + } + } +} + +/// The set of invariants carried across compaction boundaries. +#[derive(Debug, Clone, Default)] +pub struct InvariantSet { + invariants: Vec, +} + +impl InvariantSet { + pub fn new() -> Self { + Self::default() + } + + /// Add an invariant. Re-adding an existing `id` replaces it, so a caller + /// cannot accidentally accumulate near-duplicate copies of a rule. + pub fn insert(&mut self, invariant: Invariant) { + match self.invariants.iter_mut().find(|i| i.id == invariant.id) { + Some(existing) => *existing = invariant, + None => self.invariants.push(invariant), + } + } + + pub fn is_empty(&self) -> bool { + self.invariants.is_empty() + } + + pub fn len(&self) -> usize { + self.invariants.len() + } + + pub fn iter(&self) -> impl Iterator { + self.invariants.iter() + } + + /// Render the block re-emitted after a compaction. + /// + /// Returns `None` when empty so callers never inject an empty stub. + pub fn render(&self) -> Option { + if self.invariants.is_empty() { + return None; + } + let mut out = String::from( + "\nThese were established earlier and remain in force. \ + They are reproduced exactly and are not a summary.\n", + ); + for inv in &self.invariants { + out.push('\n'); + out.push_str(&inv.text); + out.push('\n'); + } + out.push_str(""); + Some(out) + } + + /// Append the invariant block to a compacted history. + /// + /// Call this immediately after any operation that drops or rewrites + /// history. Appending at the end — rather than restoring the original + /// position — is deliberate: recency is what survives a long context, and + /// the whole point is that these rules must not be the first thing lost. + pub fn reinject(&self, mut messages: Vec) -> Vec { + if let Some(block) = self.render() { + messages.push(Message::system(block)); + } + messages + } +} + +impl FromIterator for InvariantSet { + fn from_iter>(iter: T) -> Self { + let mut set = Self::new(); + for inv in iter { + set.insert(inv); + } + set + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn set() -> InvariantSet { + [ + Invariant::new("task", "TASK: Fix the failing parser test."), + Invariant::new("safety", "Never force-push to the default branch."), + ] + .into_iter() + .collect() + } + + #[test] + fn renders_text_byte_for_byte() { + let block = set().render().unwrap(); + // The exact strings must appear untransformed — no wrapping, no + // reflowing, no paraphrase. + assert!(block.contains("TASK: Fix the failing parser test.")); + assert!(block.contains("Never force-push to the default branch.")); + } + + #[test] + fn survives_a_round_trip_unchanged() { + let original = set(); + let block = original.render().unwrap(); + // Simulate several compaction cycles: each one re-renders from the + // same source, so the text can never drift. + for _ in 0..10 { + assert_eq!(original.render().unwrap(), block); + } + } + + #[test] + fn reinject_appends_a_system_message() { + let compacted = vec![Message::system("summary of earlier work")]; + let out = set().reinject(compacted); + assert_eq!(out.len(), 2); + match out.last().unwrap() { + Message::System(s) => { + assert!(s.content.contains("TASK: Fix the failing parser test.")); + assert!(s.content.contains("not a summary")); + } + other => panic!("expected a system message, got {other:?}"), + } + } + + #[test] + fn empty_set_injects_nothing() { + let empty = InvariantSet::new(); + assert!(empty.render().is_none()); + let msgs = vec![Message::human("hi")]; + assert_eq!(empty.reinject(msgs.clone()), msgs); + } + + #[test] + fn reinserting_an_id_replaces_rather_than_duplicates() { + let mut s = set(); + s.insert(Invariant::new("task", "TASK: Updated objective.")); + assert_eq!(s.len(), 2, "an updated rule must not accumulate copies"); + let block = s.render().unwrap(); + assert!(block.contains("TASK: Updated objective.")); + assert!(!block.contains("Fix the failing parser test")); + } + + #[test] + fn insertion_order_is_stable() { + let s = set(); + let ids: Vec<&str> = s.iter().map(|i| i.id.as_str()).collect(); + assert_eq!(ids, vec!["task", "safety"]); + } +} diff --git a/crates/rvAgent/rvagent-core/src/lib.rs b/crates/rvAgent/rvagent-core/src/lib.rs index 9f7d252b9..aba0196c6 100644 --- a/crates/rvAgent/rvagent-core/src/lib.rs +++ b/crates/rvAgent/rvagent-core/src/lib.rs @@ -19,20 +19,25 @@ pub mod agi_container; pub mod arena; +pub mod bootstrap; pub mod budget; pub mod config; pub mod cow_state; pub mod error; pub mod graph; +pub mod invariants; +pub mod masking; pub mod messages; pub mod metrics; pub mod models; pub mod parallel; +pub mod policy; pub mod prompt; pub mod rvf_bridge; pub mod session_crypto; pub mod state; pub mod string_pool; +pub mod subagent; // Re-export key types at crate root for convenience. pub use agi_container::{ @@ -40,7 +45,7 @@ pub use agi_container::{ OrchestratorConfig, ParsedContainer, SegmentType, SkillDefinition, ToolDefinition, }; pub use budget::{BudgetEnforcer, BudgetError, BudgetUtilization}; -pub use config::{BackendConfig, ResourceBudget, RvAgentConfig, SecurityPolicy}; +pub use config::{BackendConfig, ResourceBudget, RunnableConfig, RvAgentConfig, SecurityPolicy}; pub use cow_state::CowStateBackend; pub use error::{Result, RvAgentError}; pub use graph::{AgentGraph, AgentNode, GraphConfig, ToolExecutor}; diff --git a/crates/rvAgent/rvagent-core/src/masking.rs b/crates/rvAgent/rvagent-core/src/masking.rs new file mode 100644 index 000000000..acea64a48 --- /dev/null +++ b/crates/rvAgent/rvagent-core/src/masking.rs @@ -0,0 +1,343 @@ +//! Observation masking — the default context strategy (ADR-274). +//! +//! Old tool observations are replaced with compact placeholders before the +//! history is sent to the model. Reasoning steps and actions are kept verbatim; +//! only observations are masked. +//! +//! This is deliberately *not* summarization. Measured comparisons put simple +//! masking at or above LLM summarization on solve rate at roughly half the +//! cost, and show that summarization inflates trajectories 13–15% by destroying +//! the natural stopping signals an agent uses to notice it has finished. A +//! placeholder that says `[read_file output elided: 2847 bytes, recall id tc7]` +//! preserves the *shape* of history without fabricating its contents. +//! +//! Masking is a projection, not a mutation: `AgentState::messages` remains the +//! complete append-only log, which is what makes the elided content +//! addressable — the `tool_call_id` in each placeholder is the recall handle. + +use crate::messages::{Message, ToolMessage}; + +/// Configuration for the observation-masking projection. +#[derive(Debug, Clone)] +pub struct MaskConfig { + /// How many of the most recent observations to keep in full. + /// + /// Set to `usize::MAX` to disable masking. + pub keep_last_observations: usize, + /// Hard cap on a single tool result's size, applied at write time. + /// + /// An uncapped tool result can consume the whole context window in one + /// call. Truncation is explicit and marked so the model knows output was + /// cut rather than silently ending. + pub max_tool_result_bytes: usize, +} + +impl Default for MaskConfig { + fn default() -> Self { + Self { + keep_last_observations: 8, + // ~25k tokens at ~4 bytes/token, matching the industry default. + max_tool_result_bytes: 100_000, + } + } +} + +impl MaskConfig { + /// Whether masking is active at all. + pub fn masking_enabled(&self) -> bool { + self.keep_last_observations != usize::MAX + } +} + +/// Truncate `content` to at most `max_bytes`, on a character boundary, +/// appending an explicit marker when anything was removed. +/// +/// Operates on bytes rather than chars because the cap exists to bound memory +/// and context cost, but never splits a UTF-8 sequence. +pub fn truncate_tool_result(content: String, max_bytes: usize) -> String { + if content.len() <= max_bytes { + return content; + } + // Reserve room for the marker so the result still respects the budget. + let marker = "\n... [output truncated]"; + + // A cap smaller than the marker itself cannot carry the marker and stay + // within budget. The cap wins: it is what bounds context cost, and + // announcing the truncation is the part that can be given up. + if max_bytes < marker.len() { + return content[..floor_char_boundary(&content, max_bytes)].to_string(); + } + + let end = floor_char_boundary(&content, max_bytes - marker.len()); + let mut out = String::with_capacity(end + marker.len()); + out.push_str(&content[..end]); + out.push_str(marker); + out +} + +/// Largest index `<= max` that starts a character, so slicing there never +/// splits a multi-byte sequence. +fn floor_char_boundary(s: &str, max: usize) -> usize { + let mut n = max.min(s.len()); + while n > 0 && !s.is_char_boundary(n) { + n -= 1; + } + n +} + +/// The placeholder substituted for an elided observation. +fn placeholder(msg: &ToolMessage) -> String { + let name = msg.tool_name.as_deref().unwrap_or("tool"); + format!( + "[{} output elided: {} bytes, recall id {}]", + name, + msg.content.len(), + msg.tool_call_id + ) +} + +/// Project `messages` into the view sent to the model, masking all but the +/// most recent `keep_last_observations` tool results. +/// +/// Non-tool messages pass through untouched — masking reasoning or actions +/// would destroy exactly the trail the model needs to stay coherent. +pub fn mask_observations(messages: &[Message], config: &MaskConfig) -> Vec { + if !config.masking_enabled() { + return messages.to_vec(); + } + + let total_observations = messages + .iter() + .filter(|m| matches!(m, Message::Tool(_))) + .count(); + if total_observations <= config.keep_last_observations { + return messages.to_vec(); + } + let mask_before = total_observations - config.keep_last_observations; + + let mut seen = 0usize; + messages + .iter() + .map(|msg| match msg { + Message::Tool(tool) => { + let index = seen; + seen += 1; + if index < mask_before { + Message::Tool(ToolMessage { + tool_call_id: tool.tool_call_id.clone(), + content: placeholder(tool), + tool_name: tool.tool_name.clone(), + metadata: tool.metadata.clone(), + }) + } else { + msg.clone() + } + } + other => other.clone(), + }) + .collect() +} + +// --------------------------------------------------------------------------- +// Addressable recall (ADR-274 §3.2) +// --------------------------------------------------------------------------- + +/// The reserved tool name used to dereference a masked observation. +/// +/// Handled by the agent loop itself rather than a `ToolExecutor`: recall reads +/// the message log, which executors do not have, and reserving it in the loop +/// means a workspace tool cannot shadow it. +pub const RECALL_TOOL: &str = "recall"; + +/// Schema for the recall tool, advertised whenever masking is active. +pub fn recall_definition() -> crate::models::ToolDefinition { + crate::models::ToolDefinition { + name: RECALL_TOOL.to_string(), + description: + "Retrieve the full content of an earlier tool result that was elided from the \ + conversation. Pass the recall id shown in the placeholder, e.g. \ + '[read_file output elided: 2847 bytes, recall id tc7]' -> recall_id \"tc7\"." + .to_string(), + input_schema: serde_json::json!({ + "type": "object", + "properties": { + "recall_id": { + "type": "string", + "description": "The recall id from an elided observation placeholder." + } + }, + "required": ["recall_id"] + }), + } +} + +/// Dereference a recall id against the full message log. +/// +/// `messages` must be the complete log, never a masked projection — recalling +/// from a masked view would return the placeholder rather than the content. +pub fn recall(messages: &[Message], recall_id: &str) -> String { + for msg in messages { + if let Message::Tool(t) = msg { + if t.tool_call_id == recall_id { + return t.content.clone(); + } + } + } + // Actionable rather than opaque: tell the model where valid ids come from. + format!( + "Error: no observation found with recall id '{recall_id}'. Recall ids appear \ + in elided-output placeholders in this conversation; they are not tool names \ + or file paths." + ) +} + +/// Extract the `recall_id` argument from a recall tool call. +pub fn recall_id_from_args(args: &serde_json::Value) -> Result<&str, String> { + args.get("recall_id") + .and_then(|v| v.as_str()) + .ok_or_else(|| { + "Error: recall requires a string 'recall_id' argument, taken from an \ + elided-output placeholder." + .to_string() + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn tool(id: &str, name: &str, content: &str) -> Message { + Message::tool_with_name(id, content, name) + } + + #[test] + fn keeps_everything_when_under_the_limit() { + let msgs = vec![ + Message::ai("thinking"), + tool("t1", "read_file", "a"), + tool("t2", "read_file", "b"), + ]; + let out = mask_observations(&msgs, &MaskConfig::default()); + assert_eq!(out, msgs); + } + + #[test] + fn masks_all_but_the_last_n_observations() { + let config = MaskConfig { + keep_last_observations: 2, + ..MaskConfig::default() + }; + let msgs = vec![ + tool("t1", "read_file", "oldest"), + tool("t2", "grep", "older"), + tool("t3", "read_file", "recent"), + tool("t4", "ls", "newest"), + ]; + let out = mask_observations(&msgs, &config); + + let contents: Vec<&str> = out + .iter() + .map(|m| match m { + Message::Tool(t) => t.content.as_str(), + _ => unreachable!(), + }) + .collect(); + + assert!(contents[0].contains("read_file output elided")); + assert!(contents[0].contains("recall id t1")); + assert!(contents[1].contains("grep output elided")); + // The most recent two survive verbatim. + assert_eq!(contents[2], "recent"); + assert_eq!(contents[3], "newest"); + } + + #[test] + fn never_masks_reasoning_or_actions() { + let config = MaskConfig { + keep_last_observations: 0, + ..MaskConfig::default() + }; + let msgs = vec![ + Message::system("rules"), + Message::human("do the thing"), + Message::ai("here is my plan"), + tool("t1", "read_file", "contents"), + ]; + let out = mask_observations(&msgs, &config); + + assert_eq!(out[0], msgs[0]); + assert_eq!(out[1], msgs[1]); + assert_eq!(out[2], msgs[2], "AI reasoning must survive masking"); + match &out[3] { + Message::Tool(t) => assert!(t.content.contains("elided")), + _ => panic!("expected a tool message"), + } + } + + #[test] + fn masking_can_be_disabled() { + let config = MaskConfig { + keep_last_observations: usize::MAX, + ..MaskConfig::default() + }; + let msgs: Vec = (0..50) + .map(|i| tool(&format!("t{i}"), "read_file", "body")) + .collect(); + assert_eq!(mask_observations(&msgs, &config), msgs); + } + + #[test] + fn placeholder_preserves_the_recall_handle() { + let config = MaskConfig { + keep_last_observations: 0, + ..MaskConfig::default() + }; + let msgs = vec![tool("call-42", "grep", "many matches")]; + let out = mask_observations(&msgs, &config); + match &out[0] { + Message::Tool(t) => { + // The id must survive so the full content stays addressable, + // and must still pair with the model's tool_use block. + assert_eq!(t.tool_call_id, "call-42"); + assert!(t.content.contains("recall id call-42")); + assert!(t.content.contains("12 bytes")); + } + _ => panic!("expected a tool message"), + } + } + + #[test] + fn truncation_marks_what_it_removed() { + let out = truncate_tool_result("x".repeat(1000), 100); + assert!(out.len() <= 100); + assert!(out.ends_with("[output truncated]")); + } + + #[test] + fn truncation_leaves_short_content_alone() { + let out = truncate_tool_result("short".into(), 100); + assert_eq!(out, "short"); + } + + #[test] + fn truncation_never_splits_a_multibyte_char() { + // Every char is 4 bytes, so a naive byte cut would split one. + let content = "🙂".repeat(100); + for cap in [33usize, 50, 77, 99] { + let out = truncate_tool_result(content.clone(), cap); + // The result is valid UTF-8 by construction; the assertion is that + // this did not panic and stayed inside the cap. + assert!(out.ends_with("[output truncated]"), "cap {cap}"); + assert!(out.len() <= cap, "cap {cap} exceeded: {} bytes", out.len()); + } + } + + #[test] + fn truncation_respects_caps_too_small_for_the_marker() { + // The cap bounds context cost, so it wins over announcing the cut. + for cap in [0usize, 1, 5, 10, 22] { + let out = truncate_tool_result("🙂".repeat(100), cap); + assert!(out.len() <= cap, "cap {cap} exceeded: {} bytes", out.len()); + } + } +} diff --git a/crates/rvAgent/rvagent-core/src/messages.rs b/crates/rvAgent/rvagent-core/src/messages.rs index 2f2325e18..404b0a3ed 100644 --- a/crates/rvAgent/rvagent-core/src/messages.rs +++ b/crates/rvAgent/rvagent-core/src/messages.rs @@ -49,6 +49,9 @@ pub struct ToolMessage { pub tool_call_id: String, /// The tool's output content. pub content: String, + /// Name of the tool that produced this result (when known). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub tool_name: Option, #[serde(default, skip_serializing_if = "HashMap::is_empty")] pub metadata: HashMap, } @@ -103,6 +106,21 @@ impl Message { Self::Tool(ToolMessage { tool_call_id: tool_call_id.into(), content: content.into(), + tool_name: None, + metadata: HashMap::new(), + }) + } + + /// Create a tool result message that records the tool's name. + pub fn tool_with_name( + tool_call_id: impl Into, + content: impl Into, + tool_name: impl Into, + ) -> Self { + Self::Tool(ToolMessage { + tool_call_id: tool_call_id.into(), + content: content.into(), + tool_name: Some(tool_name.into()), metadata: HashMap::new(), }) } @@ -118,6 +136,17 @@ impl Message { } } + /// Get a mutable reference to the text content of any message variant. + #[inline] + pub fn content_mut(&mut self) -> &mut String { + match self { + Self::System(m) => &mut m.content, + Self::Human(m) => &mut m.content, + Self::Ai(m) => &mut m.content, + Self::Tool(m) => &mut m.content, + } + } + /// Returns true if this is an AI message with pending tool calls. #[inline] pub fn has_tool_calls(&self) -> bool { diff --git a/crates/rvAgent/rvagent-core/src/models.rs b/crates/rvAgent/rvagent-core/src/models.rs index 857518689..aceaf11fa 100644 --- a/crates/rvAgent/rvagent-core/src/models.rs +++ b/crates/rvAgent/rvagent-core/src/models.rs @@ -131,17 +131,36 @@ pub fn resolve_model(model_str: &str) -> ModelConfig { } } +/// A tool made available to the model for a completion request. +/// +/// Providers translate this into their wire format (Anthropic `tools`, +/// Gemini `functionDeclarations`, OpenAI `functions`). Without advertising +/// these schemas the model can never emit a tool call, so every completion +/// on the agent loop passes the active tool set. +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)] +pub struct ToolDefinition { + /// Tool name — must match `ToolCall::name` on the way back. + pub name: String, + /// Human-readable description shown to the model. + pub description: String, + /// JSON Schema for the tool's arguments. + pub input_schema: serde_json::Value, +} + /// Async trait for chat model implementations. /// /// Provider-specific crates implement this trait (e.g. `rvagent-anthropic`). #[async_trait] pub trait ChatModel: Send + Sync { - /// Send messages and receive a complete response. - async fn complete(&self, messages: &[Message]) -> Result; + /// Send messages and the active tool set, receive a complete response. + /// + /// `tools` may be empty for pure-chat completions; providers must omit + /// the tools field from the request in that case. + async fn complete(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result; /// Stream a response token-by-token. Returns a vector of incremental messages. /// The final element is the complete assembled message. - async fn stream(&self, messages: &[Message]) -> Result>; + async fn stream(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result>; } /// Extended trait for models that support chunk-based streaming. @@ -152,7 +171,11 @@ pub trait ChatModel: Send + Sync { #[async_trait] pub trait StreamingChatModel: ChatModel { /// Stream response chunks incrementally. - async fn stream_chunks(&self, messages: &[Message]) -> Result>; + async fn stream_chunks( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> Result>; } #[cfg(test)] diff --git a/crates/rvAgent/rvagent-core/src/policy.rs b/crates/rvAgent/rvagent-core/src/policy.rs new file mode 100644 index 000000000..d3152c547 --- /dev/null +++ b/crates/rvAgent/rvagent-core/src/policy.rs @@ -0,0 +1,471 @@ +//! Policy genome and score axes for the promotion flywheel (ADR-278). +//! +//! rvAgent does not implement a promotion engine. `@metaharness/flywheel` +//! already provides a frozen fingerprinted conjunctive gate, a holdout plus a +//! never-optimized-against anchor, Ed25519 receipts, independent replay +//! verification, and a compounding lineage DAG. This module is the Rust half of +//! that seam: the thing being evolved (a [`PolicyGenome`]) and the thing being +//! measured (a [`Score`]). +//! +//! The genome is deliberately shaped as the flywheel's `Policy` — +//! `Record` — so there is no adapter impedance. It is also what +//! GEPA-style optimizers consume, whose candidate is likewise a named set of +//! text components. +//! +//! # Why policy and not memory +//! +//! Self-learning splits into two objects with opposite evidence: policy text +//! (positive) and accumulated episodic memory (negative — an inverted-U where +//! utility eventually falls below no-memory). ADR-278 moves new effort here. + +use std::collections::BTreeMap; + +use serde::{Deserialize, Serialize}; + +use crate::graph::GraphConfig; +use crate::masking::MaskConfig; + +/// The levers this harness knows how to apply. +/// +/// A genome naming anything outside this set is rejected rather than ignored — +/// see [`PolicyGenome::apply_to`]. +pub const KNOWN_LEVERS: &[&str] = &[ + "max_iterations", + "parallel_tools", + "max_parallel_tools", + "loop_repeat_threshold", + "keep_last_observations", + "max_tool_result_bytes", + "system_prompt_suffix", + "compaction_rubric", +]; + +/// An operating policy: named string levers, the unit the flywheel evolves. +/// +/// `BTreeMap` rather than `HashMap` so serialization is deterministic — +/// iteration order feeds gate fingerprints and replay, and a nondeterministic +/// ordering would make identical genomes hash differently. +#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] +#[serde(transparent)] +pub struct PolicyGenome { + levers: BTreeMap, +} + +/// Why a genome could not be applied. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum PolicyError { + /// The genome names a lever this harness does not implement. + /// + /// This must be an error, not a silent skip. A mutation to an unapplied + /// lever produces a run identical to baseline, which the flywheel would + /// score as "no effect" and burn generations on — while a scoring artifact + /// could even promote it. Failing loudly keeps the search honest. + UnknownLever(String), + /// The value could not be parsed for that lever's type. + BadValue { lever: String, value: String }, +} + +impl std::fmt::Display for PolicyError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + PolicyError::UnknownLever(name) => write!( + f, + "unknown policy lever '{name}' — this harness would ignore it, so the \ + evaluation would be meaningless. Known levers: {}", + KNOWN_LEVERS.join(", ") + ), + PolicyError::BadValue { lever, value } => { + write!(f, "lever '{lever}' cannot take value {value:?}") + } + } + } +} + +impl std::error::Error for PolicyError {} + +impl PolicyGenome { + pub fn new() -> Self { + Self::default() + } + + /// Set a lever. Chainable. + pub fn with(mut self, lever: impl Into, value: impl Into) -> Self { + self.levers.insert(lever.into(), value.into()); + self + } + + pub fn get(&self, lever: &str) -> Option<&str> { + self.levers.get(lever).map(String::as_str) + } + + pub fn is_empty(&self) -> bool { + self.levers.is_empty() + } + + pub fn len(&self) -> usize { + self.levers.len() + } + + pub fn iter(&self) -> impl Iterator { + self.levers.iter() + } + + /// Free-text levers, which shape prompts rather than numeric config. + pub fn system_prompt_suffix(&self) -> Option<&str> { + self.get("system_prompt_suffix") + } + + pub fn compaction_rubric(&self) -> Option<&str> { + self.get("compaction_rubric") + } + + /// Apply the numeric and boolean levers onto a config. + /// + /// Rejects unknown levers (see [`PolicyError::UnknownLever`]). Text levers + /// are validated as known but applied by the caller, since they belong to + /// the prompt layer rather than the loop config. + pub fn apply_to(&self, base: GraphConfig) -> Result { + let mut config = base; + let mut mask: MaskConfig = config.mask.clone(); + + for (lever, value) in &self.levers { + match lever.as_str() { + "max_iterations" => config.max_iterations = parse(lever, value)?, + "parallel_tools" => config.parallel_tools = parse(lever, value)?, + "max_parallel_tools" => config.max_parallel_tools = parse(lever, value)?, + "loop_repeat_threshold" => config.loop_repeat_threshold = parse(lever, value)?, + "keep_last_observations" => mask.keep_last_observations = parse(lever, value)?, + "max_tool_result_bytes" => mask.max_tool_result_bytes = parse(lever, value)?, + // Known, but applied at the prompt layer. + "system_prompt_suffix" | "compaction_rubric" => {} + other => return Err(PolicyError::UnknownLever(other.to_string())), + } + } + + config.mask = mask; + Ok(config) + } +} + +fn parse(lever: &str, value: &str) -> Result { + value + .trim() + .parse::() + .map_err(|_| PolicyError::BadValue { + lever: lever.to_string(), + value: value.to_string(), + }) +} + +// --------------------------------------------------------------------------- +// Scoring +// --------------------------------------------------------------------------- + +/// Cost-per-win when a policy won nothing. +/// +/// **Not `f64::INFINITY`.** JSON has no infinity, so serde emits `null`, and the +/// gate's comparison `candidate.costPerWin > baseline.costPerWin` evaluates +/// `null > n` as `false` in JavaScript — meaning a policy that won nothing +/// would silently *pass* the cost clause. The largest finite double is the +/// honest encoding of "unboundedly bad" and compares correctly on both sides. +pub const COST_PER_WIN_NO_WINS: f64 = f64::MAX; + +/// The outcome of one evaluated run. +/// +/// Serializable so a headless run can emit it as JSON for the flywheel's +/// `Evaluator` to aggregate — that boundary is a process boundary, not a +/// function call. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct RunOutcome { + /// Which suite item this run covers, so an aggregator can detect a + /// missing or duplicated item rather than silently averaging fewer runs. + #[serde(default)] + pub item_id: String, + /// Did the run achieve the task (tests pass, issue resolved)? + pub succeeded: bool, + /// Did the run actually change anything? + /// + /// A run that ends without committing any change is a **no-op** even when + /// it reports success — the agent talked itself to a stop. This is the + /// signal `noop_rate` exists to catch. + #[serde(rename = "madeChanges")] + pub made_changes: bool, + /// Total cost in USD. + #[serde(rename = "costUsd")] + pub cost_usd: f64, + /// Hard safety or security regression. Any `true` blocks promotion. + #[serde(default)] + pub regressed: bool, +} + +/// The four axes the flywheel's gate decides over. +/// +/// Named generically on purpose — the gate is host- and benchmark-agnostic, and +/// projecting rvAgent's meaning onto these axes honestly is the trust boundary +/// for every downstream guarantee. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct Score { + /// Main quality signal — higher is better. + pub primary: f64, + /// Fraction of runs that committed nothing — lower is better. + /// + /// The default gate requires this to **strictly** improve: a policy earns + /// promotion by making the executor commit more, not merely score higher. + /// A policy that raises `primary` while leaving the agent more likely to + /// end empty has found a scoring artifact, not an improvement. + #[serde(rename = "noopRate")] + pub noop_rate: f64, + /// Resource cost per success — lower is better. + #[serde(rename = "costPerWin")] + pub cost_per_win: f64, + /// Hard safety/security stop. + pub regressed: bool, +} + +impl Score { + /// Aggregate run outcomes into the four axes. + /// + /// An empty set scores as maximally bad rather than perfect: zero runs must + /// never look like a clean sweep to the gate. + pub fn from_runs(runs: &[RunOutcome]) -> Self { + if runs.is_empty() { + return Self { + primary: 0.0, + noop_rate: 1.0, + cost_per_win: COST_PER_WIN_NO_WINS, + regressed: false, + }; + } + + let total = runs.len() as f64; + let wins = runs.iter().filter(|r| r.succeeded).count(); + let noops = runs.iter().filter(|r| !r.made_changes).count(); + let cost: f64 = runs.iter().map(|r| r.cost_usd).sum(); + + Self { + primary: wins as f64 / total, + noop_rate: noops as f64 / total, + cost_per_win: if wins == 0 { + COST_PER_WIN_NO_WINS + } else { + cost / wins as f64 + }, + regressed: runs.iter().any(|r| r.regressed), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn run(succeeded: bool, made_changes: bool, cost_usd: f64) -> RunOutcome { + RunOutcome { + item_id: String::new(), + succeeded, + made_changes, + cost_usd, + regressed: false, + } + } + + #[test] + fn genome_serializes_as_a_flat_string_map() { + let g = PolicyGenome::new() + .with("max_iterations", "50") + .with("system_prompt_suffix", "Be concise."); + let json = serde_json::to_string(&g).unwrap(); + // Must match the flywheel's Policy = Record exactly. + assert_eq!( + json, + r#"{"max_iterations":"50","system_prompt_suffix":"Be concise."}"# + ); + let back: PolicyGenome = serde_json::from_str(&json).unwrap(); + assert_eq!(back, g); + } + + #[test] + fn serialization_order_is_deterministic() { + let a = PolicyGenome::new().with("zebra", "1").with("alpha", "2"); + let b = PolicyGenome::new().with("alpha", "2").with("zebra", "1"); + // Identical genomes must serialize identically, or fingerprints and + // replay diverge for no reason. + assert_eq!( + serde_json::to_string(&a).unwrap(), + serde_json::to_string(&b).unwrap() + ); + } + + #[test] + fn applies_numeric_and_boolean_levers() { + let g = PolicyGenome::new() + .with("max_iterations", "42") + .with("parallel_tools", "false") + .with("loop_repeat_threshold", "5") + .with("keep_last_observations", "3"); + let config = g.apply_to(GraphConfig::default()).unwrap(); + + assert_eq!(config.max_iterations, 42); + assert!(!config.parallel_tools); + assert_eq!(config.loop_repeat_threshold, 5); + assert_eq!(config.mask.keep_last_observations, 3); + } + + #[test] + fn unknown_lever_is_rejected_not_ignored() { + let g = PolicyGenome::new().with("nonexistent_knob", "7"); + let err = g.apply_to(GraphConfig::default()).unwrap_err(); + assert_eq!(err, PolicyError::UnknownLever("nonexistent_knob".into())); + // The message must name the valid set, or the optimizer cannot recover. + assert!(err.to_string().contains("max_iterations")); + } + + #[test] + fn bad_value_is_reported_with_the_lever_name() { + let g = PolicyGenome::new().with("max_iterations", "not-a-number"); + let err = g.apply_to(GraphConfig::default()).unwrap_err(); + assert!(matches!(err, PolicyError::BadValue { .. })); + assert!(err.to_string().contains("max_iterations")); + } + + #[test] + fn text_levers_are_accepted_and_readable() { + let g = PolicyGenome::new() + .with("system_prompt_suffix", "Prefer small diffs.") + .with("compaction_rubric", "Preserve failing tests."); + assert!(g.apply_to(GraphConfig::default()).is_ok()); + assert_eq!(g.system_prompt_suffix(), Some("Prefer small diffs.")); + assert_eq!(g.compaction_rubric(), Some("Preserve failing tests.")); + } + + #[test] + fn empty_genome_leaves_config_untouched() { + let base = GraphConfig::default(); + let applied = PolicyGenome::new().apply_to(base.clone()).unwrap(); + assert_eq!(applied.max_iterations, base.max_iterations); + assert_eq!( + applied.mask.keep_last_observations, + base.mask.keep_last_observations + ); + } + + #[test] + fn score_computes_the_four_axes() { + let runs = vec![ + run(true, true, 1.0), + run(true, true, 3.0), + run(false, true, 2.0), + run(false, false, 0.5), + ]; + let s = Score::from_runs(&runs); + assert_eq!(s.primary, 0.5); + assert_eq!(s.noop_rate, 0.25); + assert_eq!(s.cost_per_win, 6.5 / 2.0); + assert!(!s.regressed); + } + + #[test] + fn a_successful_run_that_changed_nothing_still_counts_as_a_noop() { + // The case the axis exists for: the agent reports success but committed + // nothing. Scoring that as a win would reward talking over doing. + let s = Score::from_runs(&[run(true, false, 1.0)]); + assert_eq!(s.primary, 1.0); + assert_eq!(s.noop_rate, 1.0); + } + + #[test] + fn zero_wins_gives_maximally_bad_cost_per_win_not_zero() { + let s = Score::from_runs(&[run(false, true, 5.0)]); + assert_eq!(s.cost_per_win, COST_PER_WIN_NO_WINS); + } + + #[test] + fn score_never_serializes_a_non_finite_number() { + // JSON has no infinity: serde emits null, and the gate reads `null > n` + // as false in JS, so a zero-win policy would pass the cost clause. + for runs in [ + vec![], + vec![run(false, true, 5.0)], + vec![run(true, true, 1.0)], + ] { + let json = serde_json::to_value(Score::from_runs(&runs)).unwrap(); + let cost = json.get("costPerWin").unwrap(); + assert!( + cost.is_number(), + "costPerWin must serialize as a number, got {cost}" + ); + assert!(cost.as_f64().unwrap().is_finite()); + } + } + + #[test] + fn empty_run_set_scores_as_bad_not_perfect() { + let s = Score::from_runs(&[]); + assert_eq!(s.primary, 0.0); + assert_eq!(s.noop_rate, 1.0); + assert_eq!(s.cost_per_win, COST_PER_WIN_NO_WINS); + } + + #[test] + fn any_regression_sets_the_hard_stop() { + let mut bad = run(true, true, 1.0); + bad.regressed = true; + let s = Score::from_runs(&[run(true, true, 1.0), bad]); + assert!(s.regressed); + } + + #[test] + fn score_serializes_with_the_gate_field_names() { + let s = Score::from_runs(&[run(true, true, 2.0)]); + let json = serde_json::to_value(&s).unwrap(); + // These names are the gate's contract; renaming them silently breaks it. + assert!(json.get("primary").is_some()); + assert!(json.get("noopRate").is_some()); + assert!(json.get("costPerWin").is_some()); + assert!(json.get("regressed").is_some()); + } + + #[test] + fn run_outcome_round_trips_across_the_process_boundary() { + let outcome = RunOutcome { + item_id: "task-7".into(), + succeeded: true, + made_changes: true, + cost_usd: 0.42, + regressed: false, + }; + let json = serde_json::to_string(&outcome).unwrap(); + // Field names are the contract with the JS evaluator; renaming them + // silently would make every aggregated Score wrong rather than failing. + assert!(json.contains("\"madeChanges\"")); + assert!(json.contains("\"costUsd\"")); + assert!(json.contains("\"item_id\"")); + let back: RunOutcome = serde_json::from_str(&json).unwrap(); + assert_eq!(back, outcome); + } + + #[test] + fn run_outcome_tolerates_a_missing_regressed_flag() { + // An emitter that has nothing to report should not have to say so. + let json = r#"{"item_id":"a","succeeded":false,"madeChanges":false,"costUsd":0.0}"#; + let back: RunOutcome = serde_json::from_str(json).unwrap(); + assert!(!back.regressed); + } + + #[test] + fn every_known_lever_is_actually_applicable() { + // Guards against KNOWN_LEVERS drifting out of sync with apply_to. + for lever in KNOWN_LEVERS { + let value = match *lever { + "parallel_tools" => "true", + "system_prompt_suffix" | "compaction_rubric" => "text", + _ => "4", + }; + let g = PolicyGenome::new().with(*lever, value); + assert!( + g.apply_to(GraphConfig::default()).is_ok(), + "declared lever '{lever}' is not handled by apply_to" + ); + } + } +} diff --git a/crates/rvAgent/rvagent-core/src/subagent.rs b/crates/rvAgent/rvagent-core/src/subagent.rs new file mode 100644 index 000000000..fe5382d93 --- /dev/null +++ b/crates/rvAgent/rvagent-core/src/subagent.rs @@ -0,0 +1,203 @@ +//! Subagent boundary — a tool that spawns an isolated context and returns a +//! String (ADR-275). +//! +//! Deliberately *not* peer agents with a message bus, shared mutable state, or +//! a mergeable state type. One writer, auxiliary intelligence around it, never +//! parallel writes. +//! +//! The evidence for coding tasks is one-directional: at equal token budget, +//! single-agent matches or beats multi-agent on interdependent work, and the +//! team that shipped a parallel-writer architecture walked it back after a year +//! of production data. A CRDT can merge two edits to the same file without +//! textual conflict; it cannot make the *result* coherent. That is the failure +//! this boundary is shaped to make unrepresentable — a subagent that can only +//! return a `String` cannot write, so there is nothing to merge. + +use async_trait::async_trait; + +use crate::error::Result; + +/// What a subagent is for. +/// +/// Both adopted roles are read-only. The distinction is what context they get, +/// which is load-bearing rather than cosmetic. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum SubagentRole { + /// Evaluates an artifact with **no inherited conversation** — only the diff + /// and the task statement. + /// + /// The lack of shared context is the point, not an optimization: reviewers + /// measurably perform *better* without the parent's history, because a + /// shorter window means less context rot and none of the parent's + /// accumulated rationalizations. + /// + /// **Gated, not adopted** (ADR-278 §7). metaharness ADR-226 is a + /// gold-scored null on a closely related design — a read-only strong + /// advisor produced zero marginal resolves at 5.4× cost. That advisor saw + /// the full transcript where this reviewer sees only the diff, so it is not + /// refuted, but this role must demonstrate marginal lift over a + /// no-reviewer control before it reaches the default path. + Reviewer, + /// Explores, reads, greps; returns a summary string. + /// + /// Keeps exploration output out of the main window entirely, which is the + /// cleanest lever on wasted-context accumulation — the earliest and most + /// universal long-run failure. + Gatherer, +} + +impl SubagentRole { + /// Whether this role may see the parent's conversation. + /// + /// Always false. Encoded as a method rather than assumed, so adding a role + /// that inherits context is a visible decision rather than an oversight. + pub fn inherits_parent_context(&self) -> bool { + false + } + + /// Whether this role may use state-mutating tools. + /// + /// Always false — that is what makes this a single-writer architecture. + pub fn may_write(&self) -> bool { + false + } + + /// The model tier this role should run on. + /// + /// The gatherer runs cheap on measured grounds: a specialized small model + /// matched a frontier-mini in that slot, and putting a frontier model there + /// bought +0.4 pp at 5.8× cost — corroborated independently by ADR-226's + /// 5.4× null. Model tiering per role is part of the design, not a later + /// optimization. + pub fn model_tier(&self) -> ModelTier { + match self { + SubagentRole::Reviewer => ModelTier::Standard, + SubagentRole::Gatherer => ModelTier::Cheap, + } + } + + pub fn as_str(&self) -> &'static str { + match self { + SubagentRole::Reviewer => "reviewer", + SubagentRole::Gatherer => "gatherer", + } + } +} + +/// Which model tier a subagent runs on. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ModelTier { + Cheap, + Standard, +} + +/// A subagent invocation: everything it gets, and nothing more. +/// +/// There is no parent-state field by construction. A subagent cannot reach the +/// parent's conversation, files, or todos even by accident. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct SubagentRequest { + pub role: SubagentRole, + /// The complete input. For a reviewer this is the diff plus the task + /// statement; for a gatherer, the question to answer. + pub prompt: String, +} + +impl SubagentRequest { + pub fn new(role: SubagentRole, prompt: impl Into) -> Self { + Self { + role, + prompt: prompt.into(), + } + } +} + +/// The boundary itself: isolated context in, String out. +/// +/// The signature is the architecture. A subagent returns text and nothing else +/// — no state update, no file handle, no mergeable value — so a caller cannot +/// wire one up as a concurrent writer even if it wanted to. +#[async_trait] +pub trait Subagent: Send + Sync { + async fn run(&self, request: SubagentRequest) -> Result; +} + +/// Cap on a subagent's returned summary, in bytes. +/// +/// A subagent exists to *reduce* what reaches the parent's context. One that +/// returns its entire transcript has inverted its own purpose, so the boundary +/// enforces the budget rather than trusting the callee. +pub const MAX_SUMMARY_BYTES: usize = 8_000; + +/// Truncate a subagent's return value to the summary budget. +pub fn enforce_summary_budget(summary: String) -> String { + crate::masking::truncate_tool_result(summary, MAX_SUMMARY_BYTES) +} + +#[cfg(test)] +mod tests { + use super::*; + + struct EchoSubagent; + + #[async_trait] + impl Subagent for EchoSubagent { + async fn run(&self, request: SubagentRequest) -> Result { + Ok(format!("{}: {}", request.role.as_str(), request.prompt)) + } + } + + #[test] + fn no_role_may_write() { + // The single-writer invariant. If this ever fails, the architecture + // has changed and ADR-275 needs revisiting first. + for role in [SubagentRole::Reviewer, SubagentRole::Gatherer] { + assert!(!role.may_write(), "{} must not write", role.as_str()); + } + } + + #[test] + fn no_role_inherits_parent_context() { + for role in [SubagentRole::Reviewer, SubagentRole::Gatherer] { + assert!(!role.inherits_parent_context()); + } + } + + #[test] + fn gatherer_runs_on_the_cheap_tier() { + // A frontier model in this slot measured +0.4 pp at 5.8x cost. + assert_eq!(SubagentRole::Gatherer.model_tier(), ModelTier::Cheap); + } + + #[tokio::test] + async fn boundary_returns_only_a_string() { + let out = EchoSubagent + .run(SubagentRequest::new(SubagentRole::Gatherer, "where is X?")) + .await + .unwrap(); + assert_eq!(out, "gatherer: where is X?"); + } + + #[test] + fn request_carries_no_parent_state() { + // Compile-time property, asserted structurally: the request is exactly + // a role and a prompt. Adding a parent-state field would break this. + let r = SubagentRequest::new(SubagentRole::Reviewer, "diff"); + assert_eq!(r.role, SubagentRole::Reviewer); + assert_eq!(r.prompt, "diff"); + } + + #[test] + fn oversized_summaries_are_capped() { + let huge = "x".repeat(MAX_SUMMARY_BYTES * 2); + let capped = enforce_summary_budget(huge); + assert!(capped.len() <= MAX_SUMMARY_BYTES); + assert!(capped.ends_with("[output truncated]")); + } + + #[test] + fn short_summaries_pass_through_unchanged() { + let s = "found it in src/lib.rs".to_string(); + assert_eq!(enforce_summary_budget(s.clone()), s); + } +} diff --git a/crates/rvAgent/rvagent-core/tests/integration_tests.rs b/crates/rvAgent/rvagent-core/tests/integration_tests.rs index deee16e87..e4d85a5a2 100644 --- a/crates/rvAgent/rvagent-core/tests/integration_tests.rs +++ b/crates/rvAgent/rvagent-core/tests/integration_tests.rs @@ -10,7 +10,7 @@ use rvagent_core::config::RvAgentConfig; use rvagent_core::error::{Result, RvAgentError}; use rvagent_core::graph::{AgentGraph, GraphConfig, ToolExecutor}; use rvagent_core::messages::{Message, ToolCall}; -use rvagent_core::models::{ChatModel, Provider}; +use rvagent_core::models::{ChatModel, Provider, ToolDefinition}; use rvagent_core::state::AgentState; // --------------------------------------------------------------------------- @@ -32,7 +32,7 @@ impl MockModel { #[async_trait] impl ChatModel for MockModel { - async fn complete(&self, _messages: &[Message]) -> Result { + async fn complete(&self, _messages: &[Message], _tools: &[ToolDefinition]) -> Result { let mut resps = self.responses.lock().unwrap(); if resps.is_empty() { Ok(Message::ai("(no more responses)")) @@ -41,8 +41,8 @@ impl ChatModel for MockModel { } } - async fn stream(&self, messages: &[Message]) -> Result> { - let msg = self.complete(messages).await?; + async fn stream(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result> { + let msg = self.complete(messages, tools).await?; Ok(vec![msg]) } } @@ -183,6 +183,7 @@ async fn test_agent_graph_with_parallel_tool_calls() { let config = GraphConfig { max_iterations: 10, parallel_tools: true, + ..GraphConfig::default() }; let graph = AgentGraph::with_config(model, executor, config); @@ -255,26 +256,87 @@ fn test_config_to_graph_pipeline() { assert_eq!(edges.len(), 4); } -/// Tool execution failure propagates correctly through the graph. +/// Tool execution failure is fed back to the model as a tool result instead +/// of aborting the loop — the model must see the error to recover from it. #[tokio::test] -async fn test_agent_graph_tool_failure() { - let model = MockModel::new(vec![Message::ai_with_tools( - "", - vec![ToolCall { - id: "tc1".into(), - name: "dangerous_tool".into(), - args: serde_json::json!({}), - }], - )]); +async fn test_agent_graph_tool_failure_feeds_back() { + let model = MockModel::new(vec![ + Message::ai_with_tools( + "", + vec![ToolCall { + id: "tc1".into(), + name: "dangerous_tool".into(), + args: serde_json::json!({}), + }], + ), + // The model sees the error and answers without the tool. + Message::ai("The tool failed; here is a fallback answer."), + ]); let executor = FailingToolExecutor { fail_tool: "dangerous_tool".into(), }; let graph = AgentGraph::new(model, executor); let state = AgentState::new(); - let err = graph.run(state).await.unwrap_err(); - assert!(matches!(err, RvAgentError::Tool(_))); - assert!(err.to_string().contains("dangerous_tool failed")); + let result = graph.run(state).await.unwrap(); + + // The failure surfaced as a tool result, visible to the model. + let tool_msg = result + .messages + .iter() + .find(|m| matches!(m, Message::Tool(_))) + .expect("tool result message must be present"); + assert!(tool_msg.content().contains("Tool execution error")); + assert!(tool_msg.content().contains("dangerous_tool failed")); + + // The loop continued to a final answer instead of aborting. + assert_eq!( + result.messages.last().unwrap().content(), + "The tool failed; here is a fallback answer." + ); +} + +/// A failing tool in a parallel batch does not poison the other results. +#[tokio::test] +async fn test_parallel_tool_failure_isolated() { + let model = MockModel::new(vec![ + Message::ai_with_tools( + "", + vec![ + ToolCall { + id: "ok_call".into(), + name: "safe_tool".into(), + args: serde_json::json!({}), + }, + ToolCall { + id: "bad_call".into(), + name: "dangerous_tool".into(), + args: serde_json::json!({}), + }, + ], + ), + Message::ai("done"), + ]); + let executor = FailingToolExecutor { + fail_tool: "dangerous_tool".into(), + }; + let config = GraphConfig { + max_iterations: 10, + parallel_tools: true, + ..GraphConfig::default() + }; + let graph = AgentGraph::with_config(model, executor, config); + + let result = graph.run(AgentState::new()).await.unwrap(); + let tool_results: Vec<&str> = result + .messages + .iter() + .filter(|m| matches!(m, Message::Tool(_))) + .map(|m| m.content()) + .collect(); + assert_eq!(tool_results.len(), 2); + assert!(tool_results[0].contains("ok: safe_tool")); + assert!(tool_results[1].contains("Tool execution error")); } /// State mutations during graph execution use copy-on-write correctly. @@ -336,6 +398,7 @@ async fn test_max_iterations_terminates() { let config = GraphConfig { max_iterations: 5, parallel_tools: false, + ..GraphConfig::default() }; let graph = AgentGraph::with_config(model, executor, config); diff --git a/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs b/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs index 0173e76bc..5cd9aee86 100644 --- a/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs +++ b/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs @@ -7,6 +7,8 @@ use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion}; +use async_trait::async_trait; + use rvagent_core::rvf_bridge::{GovernanceMode, PolicyCheck, TaskOutcome}; use rvagent_middleware::skills::validate_skill_name; use rvagent_middleware::witness::{compute_arguments_hash, WitnessBuilder}; @@ -17,8 +19,10 @@ use rvagent_middleware::{ /// A no-op handler that returns immediately. struct NoOpHandler; + +#[async_trait] impl ModelHandler for NoOpHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { ModelResponse::text("ok") } } @@ -32,21 +36,26 @@ fn bench_full_pipeline(c: &mut Criterion) { enable_sona: false, enable_hnsw: false, enable_unicode_security: false, + enable_summarization: false, sona_config: None, hnsw_config: None, unicode_security_config: None, }; let pipeline = build_default_pipeline(&config); let handler = NoOpHandler; + let rt = tokio::runtime::Builder::new_current_thread() + .enable_time() + .build() + .unwrap(); c.bench_function("full_11_middleware_pipeline", |b| { b.iter(|| { let request = ModelRequest::new(vec![ - Message::user("Hello"), - Message::assistant("Hi there"), - Message::user("Write some code"), + Message::human("Hello"), + Message::ai("Hi there"), + Message::human("Write some code"), ]); - let response = pipeline.run_wrap_model_call(black_box(request), &handler); + let response = rt.block_on(pipeline.run_wrap_model_call(black_box(request), &handler)); black_box(response); }); }); @@ -117,6 +126,7 @@ fn bench_pipeline_modify_request(c: &mut Criterion) { enable_sona: false, enable_hnsw: false, enable_unicode_security: false, + enable_summarization: false, sona_config: None, hnsw_config: None, unicode_security_config: None, @@ -125,7 +135,7 @@ fn bench_pipeline_modify_request(c: &mut Criterion) { c.bench_function("pipeline_modify_request", |b| { b.iter(|| { - let request = ModelRequest::new(vec![Message::user("test")]) + let request = ModelRequest::new(vec![Message::human("test")]) .with_system(Some("You are helpful.".into())); let modified = pipeline.run_modify_request(black_box(request)); black_box(modified); diff --git a/crates/rvAgent/rvagent-middleware/src/filesystem.rs b/crates/rvAgent/rvagent-middleware/src/filesystem.rs index eb88bf5e6..63fd6332f 100644 --- a/crates/rvAgent/rvagent-middleware/src/filesystem.rs +++ b/crates/rvAgent/rvagent-middleware/src/filesystem.rs @@ -39,7 +39,7 @@ impl Middleware for FilesystemMiddleware { "filesystem" } - fn before_agent( + async fn before_agent( &self, _state: &AgentState, _runtime: &Runtime, @@ -77,6 +77,7 @@ impl Middleware for FilesystemMiddleware { macro_rules! fs_tool { ($name:ident, $tool_name:expr, $desc:expr, $schema:expr) => { struct $name; + #[async_trait] impl Tool for $name { fn name(&self) -> &str { $tool_name @@ -84,10 +85,10 @@ macro_rules! fs_tool { fn description(&self) -> &str { $desc } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { $schema } - fn invoke(&self, _args: serde_json::Value) -> Result { + async fn invoke(&self, _args: serde_json::Value) -> Result { Err("filesystem tool must be invoked through the agent runtime".into()) } } @@ -225,22 +226,22 @@ mod tests { assert!(names.contains(&"execute")); } - #[test] - fn test_before_agent_no_cwd() { + #[tokio::test] + async fn test_before_agent_no_cwd() { let mw = FilesystemMiddleware::new(); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } - #[test] - fn test_before_agent_with_cwd() { + #[tokio::test] + async fn test_before_agent_with_cwd() { let mw = FilesystemMiddleware::with_cwd("/tmp/test"); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let ext = &update.unwrap().extensions; assert_eq!( @@ -249,11 +250,11 @@ mod tests { ); } - #[test] - fn test_tools_return_error_without_runtime() { + #[tokio::test] + async fn test_tools_return_error_without_runtime() { let mw = FilesystemMiddleware::new(); for tool in mw.tools() { - let result = tool.invoke(serde_json::json!({})); + let result = tool.invoke(serde_json::json!({})).await; assert!(result.is_err()); } } @@ -262,7 +263,7 @@ mod tests { fn test_tool_schemas_are_objects() { let mw = FilesystemMiddleware::new(); for tool in mw.tools() { - let schema = tool.parameters_schema(); + let schema = tool.input_schema(); assert_eq!(schema["type"], "object"); } } diff --git a/crates/rvAgent/rvagent-middleware/src/hitl.rs b/crates/rvAgent/rvagent-middleware/src/hitl.rs index cf0e1fde8..65835eb11 100644 --- a/crates/rvAgent/rvagent-middleware/src/hitl.rs +++ b/crates/rvAgent/rvagent-middleware/src/hitl.rs @@ -1,5 +1,10 @@ -//! HumanInTheLoopMiddleware — intercepts tool calls matching interrupt patterns, -//! pausing execution awaiting human approval. +//! HumanInTheLoopMiddleware — intercepts tool calls matching interrupt +//! patterns and drops them, reporting the block to the model. +//! +//! There is no approval channel yet: a blocked call is not queued for a human +//! and never resumes. The middleware is a gate, not a pause, and the message it +//! injects says so — telling the model to wait for an approval that cannot +//! arrive is what turns a gate into a hang. use async_trait::async_trait; @@ -15,8 +20,9 @@ pub enum ApprovalDecision { /// Middleware that intercepts tool calls matching configurable interrupt patterns. /// -/// - `wrap_model_call`: after the model returns, checks if any tool calls match -/// the interrupt patterns. If so, pauses execution awaiting human approval. +/// - `wrap_model_call`: after the model returns, drops any tool calls matching +/// the interrupt patterns and appends a note naming them, so the model can +/// adapt rather than wait. pub struct HumanInTheLoopMiddleware { /// Tool name patterns that trigger human approval. interrupt_patterns: Vec, @@ -48,8 +54,12 @@ impl Middleware for HumanInTheLoopMiddleware { "hitl" } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { - let mut response = handler.call(request); + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { + let mut response = handler.call(request).await; // Filter out tool calls that require approval let (needs_approval, approved): (Vec, Vec) = response @@ -70,11 +80,20 @@ impl Middleware for HumanInTheLoopMiddleware { pending_names ); - if !response.message.content.is_empty() { - response.message.content.push_str("\n\n"); + let content = response.message.content_mut(); + if !content.is_empty() { + content.push_str("\n\n"); } - response.message.content.push_str(&format!( - "[HITL] Awaiting approval for: {}", + // State what actually happened: the calls were dropped, not queued. + // "Awaiting approval" implied something would come back for them and + // nothing does, which leaves the model waiting on a resolution that + // never arrives instead of adapting. + content.push_str(&format!( + "[HITL] Blocked tool call(s) requiring approval: {}. No approval \ + mechanism is wired in this runtime, so these calls were not \ + executed and will not be retried. Configure `interrupt_on` to \ + change which tools are gated (or set RVAGENT_AUTO_APPROVE=1 in \ + the CLI to run unattended).", pending_names.join(", ") )); } @@ -89,8 +108,10 @@ mod tests { use crate::Message; struct EchoHandler; + + #[async_trait] impl ModelHandler for EchoHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { let mut response = ModelResponse::text("response"); response.tool_calls = vec![ ToolCall { @@ -136,28 +157,53 @@ mod tests { assert!(!mw.should_interrupt("read_file")); } - #[test] - fn test_wrap_model_call_filters_tool_calls() { + #[tokio::test] + async fn test_wrap_model_call_filters_tool_calls() { let mw = HumanInTheLoopMiddleware::new(vec!["execute".into()]); - let request = ModelRequest::new(vec![Message::user("do something")]); + let request = ModelRequest::new(vec![Message::human("do something")]); let handler = EchoHandler; - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; assert_eq!(response.tool_calls.len(), 1); assert_eq!(response.tool_calls[0].name, "read_file"); - assert!(response.message.content.contains("[HITL]")); - assert!(response.message.content.contains("execute")); + assert!(response.content().contains("[HITL]")); + assert!(response.content().contains("execute")); } - #[test] - fn test_wrap_model_call_no_interrupt() { + #[tokio::test] + async fn test_block_message_states_reality_and_next_step() { + // The message is the model's only signal about what happened; if it + // says "awaiting" when nothing is coming, the model stalls. + let mw = HumanInTheLoopMiddleware::new(vec!["execute".into()]); + let response = mw + .wrap_model_call(ModelRequest::new(vec![Message::human("x")]), &EchoHandler) + .await; + let content = response.content().to_string(); + + assert!(content.contains("execute"), "must name the blocked call"); + assert!( + !content.contains("Awaiting"), + "must not imply a pending approval that never resolves: {content}" + ); + assert!( + content.contains("not be retried"), + "must tell the model the call is gone for good: {content}" + ); + assert!( + content.contains("interrupt_on"), + "must name the knob that changes gating: {content}" + ); + } + + #[tokio::test] + async fn test_wrap_model_call_no_interrupt() { let mw = HumanInTheLoopMiddleware::new(vec!["dangerous_tool".into()]); - let request = ModelRequest::new(vec![Message::user("safe")]); + let request = ModelRequest::new(vec![Message::human("safe")]); let handler = EchoHandler; - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; assert_eq!(response.tool_calls.len(), 2); - assert!(!response.message.content.contains("[HITL]")); + assert!(!response.content().contains("[HITL]")); } #[test] diff --git a/crates/rvAgent/rvagent-middleware/src/hnsw.rs b/crates/rvAgent/rvagent-middleware/src/hnsw.rs index 07394d089..73eac9bd0 100644 --- a/crates/rvAgent/rvagent-middleware/src/hnsw.rs +++ b/crates/rvAgent/rvagent-middleware/src/hnsw.rs @@ -12,12 +12,14 @@ //! //! # Performance //! -//! - 150x-12,500x faster than brute-force search +//! - EXPERIMENTAL: uses a hash-based embedding placeholder (not semantic); +//! no retrieval performance claims until real embeddings are integrated //! - O(log n) search complexity //! - Sub-millisecond latency for 10k vectors use crate::{ - AgentState, AgentStateUpdate, Middleware, ModelRequest, RunnableConfig, Runtime, ToolDefinition, + AgentState, AgentStateUpdate, Message, Middleware, ModelRequest, RunnableConfig, Runtime, + ToolDefinition, }; use async_trait::async_trait; use parking_lot::RwLock; @@ -722,7 +724,7 @@ impl HnswMiddleware { Some(ToolDefinition { name, description, - parameters, + input_schema: parameters, }) }) .collect() @@ -735,7 +737,7 @@ impl Middleware for HnswMiddleware { "hnsw" } - fn before_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, @@ -750,11 +752,11 @@ impl Middleware for HnswMiddleware { .messages .iter() .rev() - .find(|m| matches!(m.role, crate::Role::User))?; + .find(|m| matches!(m, Message::Human(_)))?; // Search for relevant memory let memory_results = self.search_memory( - &last_user.content, + last_user.content(), self.state.read().config.memory_retrieval_k, ); @@ -794,8 +796,8 @@ impl Middleware for HnswMiddleware { .messages .iter() .rev() - .find(|m| matches!(m.role, crate::Role::User)) - .map(|m| m.content.clone()); + .find(|m| matches!(m, Message::Human(_))) + .map(|m| m.content().to_string()); if let Some(query) = query { // Retrieve relevant skills as tools diff --git a/crates/rvAgent/rvagent-middleware/src/lib.rs b/crates/rvAgent/rvagent-middleware/src/lib.rs index 5153402e8..c706f5d3d 100644 --- a/crates/rvAgent/rvagent-middleware/src/lib.rs +++ b/crates/rvAgent/rvagent-middleware/src/lib.rs @@ -3,10 +3,22 @@ //! Provides the `Middleware` trait and `MiddlewarePipeline` for composing middleware //! in the DeepAgents architecture (ADR-095, ADR-103). //! +//! ## Unified type system (P0.1) +//! +//! All conversation/state types (`Message`, `ToolCall`, `AgentState`, `TodoItem`, +//! `TodoStatus`, `RunnableConfig`, `ToolDefinition`) are the canonical +//! `rvagent-core` definitions, re-exported here for convenience. +//! +//! ## Async model-call chain (P0.3) +//! +//! `Middleware::wrap_model_call` and `ModelHandler::call` are async, so a real +//! HTTP model call can run inside the pipeline. Use [`PipelineModel`] to wrap +//! any `rvagent_core::models::ChatModel` with a pipeline. +//! //! ## ADR-103 Learning Middleware (B5, B6) //! //! - [`sona`] — SONA Adaptive Learning with three loops (instant, background, deep) -//! - [`hnsw`] — HNSW Semantic Retrieval for skills and memory (150x-12,500x faster) +//! - [`hnsw`] — experimental in-process retrieval index (hash-embedding placeholder) pub mod filesystem; pub mod hitl; @@ -14,6 +26,8 @@ pub mod hnsw; pub mod mcp_bridge; pub mod memory; pub mod patch_tool_calls; +pub mod pipeline; +pub mod pipeline_model; pub mod prompt_caching; pub mod retry; pub mod rvf_manifest; @@ -23,276 +37,63 @@ pub mod subagents; pub mod summarization; pub mod todolist; pub mod tool_sanitizer; +pub mod types; pub mod unicode_security; pub mod unicode_security_middleware; pub mod utils; pub mod witness; use async_trait::async_trait; -use serde::{Deserialize, Serialize}; -use std::collections::HashMap; use std::fmt; // Re-exports +pub use pipeline::MiddlewarePipeline; +pub use pipeline_model::PipelineModel; +pub use types::{ + json_extension, AgentState, AgentStateUpdate, AiMessage, CacheControl, FileData, HumanMessage, + Message, ModelRequest, ModelResponse, RunnableConfig, Runtime, SystemMessage, TodoItem, + TodoStatus, ToolCall, ToolDefinition, ToolMessage, Usage, +}; pub use unicode_security::{UnicodeIssue, UnicodeSecurityChecker, UnicodeSecurityConfig}; pub use unicode_security_middleware::UnicodeSecurityMiddleware; pub use utils::{append_to_system_message, SystemPromptBuilder}; // --------------------------------------------------------------------------- -// Core types +// Model handler trait (async — P0.3) // --------------------------------------------------------------------------- -/// Message role in a conversation. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "lowercase")] -pub enum Role { - System, - User, - Assistant, - Tool, -} - -/// A single tool call within an assistant message. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct ToolCall { - pub id: String, - pub name: String, - pub args: serde_json::Value, -} - -/// A conversation message. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct Message { - pub role: Role, - pub content: String, - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub tool_calls: Vec, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub tool_call_id: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub tool_name: Option, -} - -impl Message { - pub fn system(content: impl Into) -> Self { - Self { - role: Role::System, - content: content.into(), - tool_calls: vec![], - tool_call_id: None, - tool_name: None, - } - } - - pub fn user(content: impl Into) -> Self { - Self { - role: Role::User, - content: content.into(), - tool_calls: vec![], - tool_call_id: None, - tool_name: None, - } - } - - pub fn assistant(content: impl Into) -> Self { - Self { - role: Role::Assistant, - content: content.into(), - tool_calls: vec![], - tool_call_id: None, - tool_name: None, - } - } - - pub fn tool( - content: impl Into, - tool_call_id: impl Into, - name: impl Into, - ) -> Self { - Self { - role: Role::Tool, - content: content.into(), - tool_calls: vec![], - tool_call_id: Some(tool_call_id.into()), - tool_name: Some(name.into()), - } - } -} - -/// Cache control hint for prompt caching (Anthropic). -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct CacheControl { - pub cache_type: String, -} - -/// Agent state — typed structure (ADR-103 A1) with extension map. -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -pub struct AgentState { - pub messages: Vec, - #[serde(default)] - pub todos: Vec, - #[serde(default)] - pub extensions: HashMap, -} - -/// A single todo item managed by TodoListMiddleware. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct TodoItem { - pub id: String, - pub content: String, - #[serde(default)] - pub status: TodoStatus, -} - -/// Status of a todo item. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "lowercase")] -pub enum TodoStatus { - Pending, - InProgress, - Completed, -} - -impl Default for TodoStatus { - fn default() -> Self { - Self::Pending - } -} - -/// State update returned by `before_agent`. Merged into `AgentState`. -#[derive(Debug, Clone, Default)] -pub struct AgentStateUpdate { - pub messages: Option>, - pub todos: Option>, - pub extensions: HashMap, -} - -/// Model request wrapping messages and configuration. -#[derive(Debug, Clone)] -pub struct ModelRequest { - pub system_message: Option, - pub messages: Vec, - pub tools: Vec, - pub cache_control: HashMap, - pub extensions: HashMap, -} - -impl ModelRequest { - /// Create a new model request. - pub fn new(messages: Vec) -> Self { - Self { - system_message: None, - messages, - tools: vec![], - cache_control: HashMap::new(), - extensions: HashMap::new(), - } - } - - /// Return a copy with a different system message. - pub fn with_system(mut self, system_message: Option) -> Self { - self.system_message = system_message; - self - } - - /// Return a copy with different messages. - pub fn with_messages(mut self, messages: Vec) -> Self { - self.messages = messages; - self - } -} - -/// Model response from an LLM call. -#[derive(Debug, Clone)] -pub struct ModelResponse { - pub message: Message, - pub tool_calls: Vec, - pub usage: Option, -} - -impl ModelResponse { - /// Create a simple text response. - pub fn text(content: impl Into) -> Self { - Self { - message: Message::assistant(content), - tool_calls: vec![], - usage: None, - } - } -} - -/// Token usage information. -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -pub struct Usage { - pub input_tokens: u64, - pub output_tokens: u64, - #[serde(default)] - pub cache_read_tokens: u64, - #[serde(default)] - pub cache_creation_tokens: u64, -} - -/// Tool definition for model requests. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct ToolDefinition { - pub name: String, - pub description: String, - pub parameters: serde_json::Value, -} - -/// Runtime context passed to middleware hooks. -pub struct Runtime { - pub context: serde_json::Value, - pub config: RunnableConfig, -} - -impl Runtime { - pub fn new() -> Self { - Self { - context: serde_json::Value::Null, - config: RunnableConfig::default(), - } - } -} - -impl Default for Runtime { - fn default() -> Self { - Self::new() - } -} - -/// Configuration for a runnable (thread/run IDs, metadata). -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -pub struct RunnableConfig { - #[serde(default)] - pub thread_id: Option, - #[serde(default)] - pub run_id: Option, - #[serde(default)] - pub metadata: HashMap, -} - -// --------------------------------------------------------------------------- -// Model handler traits -// --------------------------------------------------------------------------- - -/// Synchronous model handler — called by `wrap_model_call`. -pub trait ModelHandler: Send + Sync { - fn call(&self, request: ModelRequest) -> ModelResponse; -} - -/// Async model handler — called by `awrap_model_call`. +/// Async model handler — the "next" link called by `wrap_model_call`. #[async_trait] -pub trait AsyncModelHandler: Send + Sync { +pub trait ModelHandler: Send + Sync { async fn call(&self, request: ModelRequest) -> ModelResponse; } +// --------------------------------------------------------------------------- +// Tool trait (aligned with rvagent_core::models::ToolDefinition) +// --------------------------------------------------------------------------- + /// Tool trait — tools injected by middleware. +/// +/// Schema exposure aligns with `rvagent_core::models::ToolDefinition` +/// (`input_schema`); `definition()` produces the canonical form. +#[async_trait] pub trait Tool: Send + Sync { fn name(&self) -> &str; fn description(&self) -> &str; - fn parameters_schema(&self) -> serde_json::Value; - fn invoke(&self, args: serde_json::Value) -> Result; + /// JSON Schema for the tool's arguments (ToolDefinition::input_schema). + fn input_schema(&self) -> serde_json::Value; + + /// Canonical schema form advertised to models. + fn definition(&self) -> ToolDefinition { + ToolDefinition { + name: self.name().to_string(), + description: self.description().to_string(), + input_schema: self.input_schema(), + } + } + + /// Execute the tool asynchronously. + async fn invoke(&self, args: serde_json::Value) -> Result; } impl fmt::Debug for dyn Tool { @@ -302,17 +103,19 @@ impl fmt::Debug for dyn Tool { } // --------------------------------------------------------------------------- -// Middleware trait (ADR-095) +// Middleware trait (ADR-095) — async hooks (P0.3) // --------------------------------------------------------------------------- /// Core middleware trait — mirrors Python's `AgentMiddleware`. /// /// Each method has a default no-op implementation, so concrete middleware -/// only needs to override the hooks it uses. +/// only needs to override the hooks it uses. The former sync/async duplicate +/// hook pairs (`before_agent`/`abefore_agent`, `wrap_model_call`/ +/// `awrap_model_call`) are merged into single async hooks. #[async_trait] pub trait Middleware: Send + Sync { /// Called before agent execution. Returns state update or None. - fn before_agent( + async fn before_agent( &self, _state: &AgentState, _runtime: &Runtime, @@ -321,26 +124,12 @@ pub trait Middleware: Send + Sync { None } - /// Async version of `before_agent`. - async fn abefore_agent( - &self, - state: &AgentState, - runtime: &Runtime, - config: &RunnableConfig, - ) -> Option { - self.before_agent(state, runtime, config) - } - - /// Wrap a synchronous model call — intercept request/response. - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { - handler.call(request) - } - - /// Wrap an async model call. - async fn awrap_model_call( + /// Wrap a model call — intercept request/response. Async so real HTTP + /// calls can run inside the chain. + async fn wrap_model_call( &self, request: ModelRequest, - handler: &dyn AsyncModelHandler, + handler: &dyn ModelHandler, ) -> ModelResponse { handler.call(request).await } @@ -367,153 +156,6 @@ impl fmt::Debug for dyn Middleware { } } -// --------------------------------------------------------------------------- -// Middleware Pipeline (ADR-095) -// --------------------------------------------------------------------------- - -/// Executes the middleware pipeline in order. -/// Mirrors LangChain's `create_agent` middleware composition. -pub struct MiddlewarePipeline { - middlewares: Vec>, -} - -impl MiddlewarePipeline { - /// Create a new pipeline from an ordered list of middlewares. - pub fn new(middlewares: Vec>) -> Self { - Self { middlewares } - } - - /// Create an empty pipeline. - pub fn empty() -> Self { - Self { - middlewares: Vec::new(), - } - } - - /// Add a middleware to the end of the pipeline. - pub fn push(&mut self, middleware: Box) { - self.middlewares.push(middleware); - } - - /// Number of middlewares in the pipeline. - pub fn len(&self) -> usize { - self.middlewares.len() - } - - /// Whether the pipeline is empty. - pub fn is_empty(&self) -> bool { - self.middlewares.is_empty() - } - - /// Get middleware names in order. - pub fn names(&self) -> Vec<&str> { - self.middlewares.iter().map(|mw| mw.name()).collect() - } - - /// Run `before_agent` hooks in order, accumulating state updates. - pub async fn run_before_agent( - &self, - state: &mut AgentState, - runtime: &Runtime, - config: &RunnableConfig, - ) { - for mw in &self.middlewares { - if let Some(update) = mw.abefore_agent(state, runtime, config).await { - // Merge update into state - if let Some(messages) = update.messages { - state.messages = messages; - } - if let Some(todos) = update.todos { - state.todos = todos; - } - for (k, v) in update.extensions { - state.extensions.insert(k, v); - } - } - } - } - - /// Collect all tools from all middlewares. - pub fn collect_tools(&self) -> Vec> { - self.middlewares.iter().flat_map(|mw| mw.tools()).collect() - } - - /// Run `modify_request` through all middlewares in order. - pub fn run_modify_request(&self, mut request: ModelRequest) -> ModelRequest { - for mw in &self.middlewares { - request = mw.modify_request(request); - } - request - } - - /// Run `wrap_model_call` through the pipeline. - /// Middlewares are chained so the outermost (first) wraps the innermost (last). - pub fn run_wrap_model_call( - &self, - request: ModelRequest, - base_handler: &dyn ModelHandler, - ) -> ModelResponse { - if self.middlewares.is_empty() { - return base_handler.call(request); - } - - // Build chain from inside out using recursive approach. - fn chain_call<'a>( - middlewares: &'a [Box], - request: ModelRequest, - handler: &'a dyn ModelHandler, - ) -> ModelResponse { - if middlewares.is_empty() { - return handler.call(request); - } - let (first, rest) = middlewares.split_first().unwrap(); - let inner = ChainedInner { rest, handler }; - first.wrap_model_call(request, &inner) - } - - struct ChainedInner<'a> { - rest: &'a [Box], - handler: &'a dyn ModelHandler, - } - impl<'a> ModelHandler for ChainedInner<'a> { - fn call(&self, request: ModelRequest) -> ModelResponse { - chain_call(self.rest, request, self.handler) - } - } - - chain_call(&self.middlewares, request, base_handler) - } - - /// Full pipeline run: before_agent, collect tools, modify_request, wrap_model_call. - pub async fn run( - &self, - state: &mut AgentState, - runtime: &Runtime, - config: &RunnableConfig, - mut request: ModelRequest, - handler: &dyn ModelHandler, - ) -> ModelResponse { - // 1. Run before_agent hooks - self.run_before_agent(state, runtime, config).await; - - // 2. Collect tools from all middlewares - let tools: Vec> = self.collect_tools(); - for tool in &tools { - request.tools.push(ToolDefinition { - name: tool.name().to_string(), - description: tool.description().to_string(), - parameters: tool.parameters_schema(), - }); - } - - // 3. Run modify_request - request = self.run_modify_request(request); - - // 4. Run wrap_model_call chain - self.run_wrap_model_call(request, handler) - } -} - // --------------------------------------------------------------------------- // Default pipeline builder (ADR-095) // --------------------------------------------------------------------------- @@ -531,6 +173,16 @@ pub struct PipelineConfig { pub enable_hnsw: bool, /// Enable Unicode security middleware (C7 - CVE mitigation). pub enable_unicode_security: bool, + /// Enable LLM summarization as the compaction strategy (ADR-274). + /// + /// **Off by default.** Observation masking in the agent loop is the default + /// strategy; measured comparisons put simple masking at or above LLM + /// summarization on solve rate at roughly half the cost, and show + /// summarization inflating trajectories 13–15% by destroying the stopping + /// signals an agent uses to notice it has finished. Enable only as a + /// deliberate fallback, and supply a preservation rubric when you do — the + /// rubric is the load-bearing part, not the summarizer. + pub enable_summarization: bool, /// Custom SONA configuration. pub sona_config: Option, /// Custom HNSW configuration. @@ -539,9 +191,10 @@ pub struct PipelineConfig { pub unicode_security_config: Option, } -/// Build the default middleware pipeline per ADR-095 ordering: -/// Todo -> HNSW -> Memory -> Skills -> Filesystem -> SubAgent -> Summarization -/// -> PromptCaching -> PatchToolCalls -> UnicodeSecurityMiddleware -> SONA -> Witness -> ToolSanitizer -> HITL +/// Build the default middleware pipeline (ADR-095 ordering, amended by ADR-274): +/// Todo -> HNSW -> Memory -> Skills -> Filesystem -> SubAgent +/// -> [Summarization, opt-in] -> PromptCaching -> PatchToolCalls +/// -> UnicodeSecurityMiddleware -> SONA -> Witness -> ToolSanitizer -> HITL /// /// HNSW is early in the pipeline to augment context before other middleware. /// UnicodeSecurityMiddleware runs before SONA to sanitize inputs/outputs (C7). @@ -569,9 +222,18 @@ pub fn build_default_pipeline(config: &PipelineConfig) -> MiddlewarePipeline { middlewares.push(Box::new(filesystem::FilesystemMiddleware::new())); middlewares.push(Box::new(subagents::SubAgentMiddleware::new())); - middlewares.push(Box::new(summarization::SummarizationMiddleware::new( - 100_000, 0.85, 0.10, - ))); + + // Summarization is OFF by default (ADR-274). Observation masking in the + // agent loop is the default compaction strategy; this remains available as + // an explicit fallback. Trigger lowered from 0.85 to 0.75 when enabled: + // context degrades well before the nominal limit, so compacting at 85% + // leaves too little headroom to be selective rather than desperate. + if config.enable_summarization { + middlewares.push(Box::new(summarization::SummarizationMiddleware::new( + 100_000, 0.75, 0.10, + ))); + } + middlewares.push(Box::new(prompt_caching::PromptCachingMiddleware::new())); middlewares.push(Box::new(patch_tool_calls::PatchToolCallsMiddleware::new())); @@ -605,15 +267,168 @@ pub fn build_default_pipeline(config: &PipelineConfig) -> MiddlewarePipeline { tool_sanitizer::ToolResultSanitizerMiddleware::new(), )); - if let Some(patterns) = &config.interrupt_on { - middlewares.push(Box::new(hitl::HumanInTheLoopMiddleware::new( - patterns.clone(), - ))); - } + // Unconditional, and via the same fallback the by-name path uses: an + // unconfigured caller (the ACP agent reaches this with a default config) + // must not silently get an agent with no approval gate at all. An explicit + // `Some(vec![])` still opts out — it matches nothing. + middlewares.push(Box::new(hitl::HumanInTheLoopMiddleware::new( + interrupt_patterns(config), + ))); MiddlewarePipeline::new(middlewares) } +// --------------------------------------------------------------------------- +// Name-based middleware resolution (CLI wiring) +// --------------------------------------------------------------------------- + +/// Tool-name patterns gated by HITL when no `interrupt_on` is configured. +/// +/// Every pipeline builder falls back to this set: an approval gate that +/// approves everything — or a pipeline with no gate at all — is worse than a +/// real one, because it reads as protection that isn't there. Shell execution +/// and file mutation are the irreversible operations, so those are what a +/// caller who never configured `interrupt_on` gets gated on. +/// +/// `write_todos` is deliberately absent — gating the agent's own scratchpad +/// would interrupt every turn without protecting anything. +pub const DEFAULT_INTERRUPT_PATTERNS: &[&str] = &[ + // Shell / arbitrary command execution. + "execute", + "execute_command", + "shell*", + "bash*", + "run_command*", + // File mutation. + "write_file", + "edit_file", + "apply_patch", + "delete_file", + "move_file", +]; + +fn default_interrupt_patterns() -> Vec { + DEFAULT_INTERRUPT_PATTERNS + .iter() + .map(|s| (*s).to_string()) + .collect() +} + +/// The HITL patterns `config` asks for, or the conservative built-in set. +/// +/// Both pipeline builders resolve gating through here. The two paths having +/// separate fallbacks is what produced an ACP agent with no gate at all while +/// the CLI had one. +fn interrupt_patterns(config: &PipelineConfig) -> Vec { + config + .interrupt_on + .clone() + .unwrap_or_else(default_interrupt_patterns) +} + +/// A middleware name that could not be resolved to an implementation. +/// +/// Unknown names are fatal rather than skipped: a typo in the middleware list +/// silently drops whatever that entry was supposed to do, and the entries most +/// worth typo-ing are the security ones (`hitl`, `unicode_security`, +/// `tool_result_sanitizer`). +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct UnknownMiddlewareError { + pub name: String, +} + +impl fmt::Display for UnknownMiddlewareError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!( + f, + "unknown middleware '{}' (see rvagent_middleware::middleware_by_name for valid names)", + self.name + ) + } +} + +impl std::error::Error for UnknownMiddlewareError {} + +/// Resolve a middleware name (as used in `RvAgentConfig::middleware` / +/// the CLI `DEFAULT_MIDDLEWARE` list) into a middleware instance. +/// +/// `config` supplies the same settings [`build_default_pipeline`] uses, so the +/// two construction paths produce equivalently configured middleware. +/// +/// Returns `Err` for unknown names. +pub fn middleware_by_name( + name: &str, + config: &PipelineConfig, +) -> std::result::Result, UnknownMiddlewareError> { + let mw: Box = match name { + "todo" | "todos" | "todolist" => Box::new(todolist::TodoListMiddleware::new()), + "memory" => Box::new(memory::MemoryMiddleware::new( + config + .memory_sources + .clone() + .unwrap_or_else(|| vec!["AGENTS.md".into()]), + )), + "skills" => Box::new(skills::SkillsMiddleware::new( + config + .skill_sources + .clone() + .unwrap_or_else(|| vec![".skills".into()]), + )), + "filesystem" => Box::new(filesystem::FilesystemMiddleware::new()), + "subagent" | "subagents" => Box::new(subagents::SubAgentMiddleware::new()), + "summarization" => Box::new(summarization::SummarizationMiddleware::new( + 100_000, 0.75, 0.10, + )), + "prompt_caching" => Box::new(prompt_caching::PromptCachingMiddleware::new()), + "patch_tool_calls" => Box::new(patch_tool_calls::PatchToolCallsMiddleware::new()), + "witness" => Box::new(witness::WitnessMiddleware::new()), + "tool_result_sanitizer" | "tool_sanitizer" => { + Box::new(tool_sanitizer::ToolResultSanitizerMiddleware::new()) + } + "hitl" => Box::new(hitl::HumanInTheLoopMiddleware::new(interrupt_patterns( + config, + ))), + "retry" => Box::new(retry::RetryMiddleware::default()), + "hnsw" => Box::new(hnsw::HnswMiddleware::new( + config.hnsw_config.clone().unwrap_or_default(), + )), + "sona" => Box::new(sona::SonaMiddleware::new( + config.sona_config.clone().unwrap_or_default(), + )), + "unicode_security" => Box::new( + UnicodeSecurityMiddleware::new( + config + .unicode_security_config + .clone() + .unwrap_or_else(UnicodeSecurityConfig::strict), + ) + .with_input_sanitization(true) + .with_output_sanitization(false), // Log only by default + ), + "mcp_bridge" => Box::new(mcp_bridge::McpBridgeMiddleware::new()), + _ => { + return Err(UnknownMiddlewareError { + name: name.to_string(), + }) + } + }; + Ok(mw) +} + +/// Build a pipeline from an ordered list of middleware names. +/// +/// An unknown name is an error, not a skip — see [`UnknownMiddlewareError`]. +pub fn build_pipeline_from_names>( + names: &[S], + config: &PipelineConfig, +) -> std::result::Result { + let mut pipeline = MiddlewarePipeline::empty(); + for name in names { + pipeline.push(middleware_by_name(name.as_ref(), config)?); + } + Ok(pipeline) +} + // --------------------------------------------------------------------------- // Tests // --------------------------------------------------------------------------- @@ -624,8 +439,10 @@ mod tests { /// A passthrough test handler. struct EchoHandler; + + #[async_trait] impl ModelHandler for EchoHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(format!("echo: {}", request.messages.len())) } } @@ -646,19 +463,21 @@ mod tests { fn name(&self) -> &str { "prepend" } - fn wrap_model_call( + async fn wrap_model_call( &self, request: ModelRequest, handler: &dyn ModelHandler, ) -> ModelResponse { let new_sys = append_to_system_message(&request.system_message, &self.text); - handler.call(request.with_system(new_sys)) + handler.call(request.with_system(new_sys)).await } } /// A test middleware that injects a tool. struct ToolInjector; struct DummyTool; + + #[async_trait] impl Tool for DummyTool { fn name(&self) -> &str { "dummy_tool" @@ -666,10 +485,10 @@ mod tests { fn description(&self) -> &str { "A dummy tool" } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { serde_json::json!({}) } - fn invoke(&self, _args: serde_json::Value) -> Result { + async fn invoke(&self, _args: serde_json::Value) -> Result { Ok("ok".into()) } } @@ -683,27 +502,6 @@ mod tests { } } - #[test] - fn test_message_constructors() { - let sys = Message::system("sys"); - assert_eq!(sys.role, Role::System); - let usr = Message::user("hi"); - assert_eq!(usr.role, Role::User); - let asst = Message::assistant("resp"); - assert_eq!(asst.role, Role::Assistant); - let tool = Message::tool("result", "tc-1", "my_tool"); - assert_eq!(tool.role, Role::Tool); - assert_eq!(tool.tool_call_id.as_deref(), Some("tc-1")); - } - - #[test] - fn test_model_request_with_system() { - let req = ModelRequest::new(vec![Message::user("hi")]); - assert!(req.system_message.is_none()); - let req2 = req.with_system(Some("system".into())); - assert_eq!(req2.system_message, Some("system".into())); - } - #[test] fn test_empty_pipeline() { let pipeline = MiddlewarePipeline::empty(); @@ -722,29 +520,32 @@ mod tests { assert_eq!(pipeline.len(), 2); } - #[test] - fn test_pipeline_wrap_model_call_chaining() { + #[tokio::test] + async fn test_pipeline_wrap_model_call_chaining() { // Two prepend middlewares should chain: first wraps second wraps handler let pipeline = MiddlewarePipeline::new(vec![ Box::new(PrependMiddleware::new("A")), Box::new(PrependMiddleware::new("B")), ]); - let request = ModelRequest::new(vec![Message::user("hi")]).with_system(Some("base".into())); + let request = + ModelRequest::new(vec![Message::human("hi")]).with_system(Some("base".into())); // Track what system message the handler receives struct CaptureHandler; + + #[async_trait] impl ModelHandler for CaptureHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(request.system_message.unwrap_or_default()) } } - let response = pipeline.run_wrap_model_call(request, &CaptureHandler); + let response = pipeline.run_wrap_model_call(request, &CaptureHandler).await; // First middleware appends A, second appends B - assert!(response.message.content.contains("A")); - assert!(response.message.content.contains("B")); - assert!(response.message.content.contains("base")); + assert!(response.content().contains("A")); + assert!(response.content().contains("B")); + assert!(response.content().contains("base")); } #[test] @@ -754,6 +555,8 @@ mod tests { let tools = pipeline.collect_tools(); assert_eq!(tools.len(), 2); assert_eq!(tools[0].name(), "dummy_tool"); + let def = tools[0].definition(); + assert_eq!(def.name, "dummy_tool"); } #[tokio::test] @@ -766,21 +569,61 @@ mod tests { let mut state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let request = ModelRequest::new(vec![Message::user("test")]); + let request = ModelRequest::new(vec![Message::human("test")]); let response = pipeline .run(&mut state, &runtime, &config, request, &EchoHandler) .await; - assert!(response.message.content.contains("echo")); + assert!(response.content().contains("echo")); } #[test] fn test_build_default_pipeline_minimal() { let config = PipelineConfig::default(); let pipeline = build_default_pipeline(&config); - // Should have: todo, filesystem, subagent, summarization, prompt_caching, - // patch_tool_calls, tool_sanitizer = 7 + // todo, filesystem, subagent, prompt_caching, patch_tool_calls, + // tool_sanitizer, hitl = 7. Summarization is opt-in (ADR-274). assert!(pipeline.len() >= 7); + assert!(pipeline.names().contains(&"hitl")); + } + + #[tokio::test] + async fn test_build_default_pipeline_gates_execute_without_config() { + // rvagent-acp builds this pipeline with a default config whenever the + // agent config lists no middleware; it must not come out ungated. + let pipeline = build_default_pipeline(&PipelineConfig::default()); + let response = pipeline + .run_wrap_model_call( + ModelRequest::new(vec![Message::human("go")]), + &DangerousCallHandler, + ) + .await; + + let survived: Vec<&str> = response + .tool_calls + .iter() + .map(|c| c.name.as_str()) + .collect(); + assert_eq!(survived, vec!["read_file"]); + assert!(response.content().contains("[HITL]")); + } + + #[tokio::test] + async fn test_build_default_pipeline_explicit_empty_interrupt_on_disables_gating() { + // The documented opt-out for unattended runs must keep working. + let pipeline = build_default_pipeline(&PipelineConfig { + interrupt_on: Some(Vec::new()), + ..PipelineConfig::default() + }); + let response = pipeline + .run_wrap_model_call( + ModelRequest::new(vec![Message::human("go")]), + &DangerousCallHandler, + ) + .await; + + assert_eq!(response.tool_calls.len(), 3); + assert!(!response.content().contains("[HITL]")); } #[test] @@ -793,14 +636,34 @@ mod tests { enable_sona: false, enable_hnsw: false, enable_unicode_security: false, + enable_summarization: false, sona_config: None, hnsw_config: None, unicode_security_config: None, }; let pipeline = build_default_pipeline(&config); - // todo + memory + skills + filesystem + subagent + summarization + prompt_caching - // + patch_tool_calls + witness + tool_sanitizer + hitl = 11 - assert_eq!(pipeline.len(), 11); + // todo + memory + skills + filesystem + subagent + prompt_caching + // + patch_tool_calls + witness + tool_sanitizer + hitl = 10. + // Summarization is absent by default (ADR-274). + assert_eq!(pipeline.len(), 10); + } + + #[test] + fn test_summarization_is_absent_by_default_and_available_opt_in() { + // The default path must not carry summarization: masking in the agent + // loop is the decided strategy (ADR-274). + let default_len = build_default_pipeline(&PipelineConfig::default()).len(); + let opted_in = build_default_pipeline(&PipelineConfig { + enable_summarization: true, + ..PipelineConfig::default() + }); + assert_eq!( + opted_in.len(), + default_len + 1, + "enabling summarization must add exactly one middleware" + ); + // And it must still be constructible by name, so the fallback is real. + assert!(middleware_by_name("summarization", &PipelineConfig::default()).is_ok()); } #[test] @@ -808,26 +671,118 @@ mod tests { let state = AgentState::default(); assert!(state.messages.is_empty()); assert!(state.todos.is_empty()); - assert!(state.extensions.is_empty()); } #[test] - fn test_todo_status_default() { - let status = TodoStatus::default(); - assert_eq!(status, TodoStatus::Pending); + fn test_middleware_by_name_known() { + let config = PipelineConfig::default(); + for name in [ + "todo", + "memory", + "skills", + "filesystem", + "subagent", + "summarization", + "prompt_caching", + "patch_tool_calls", + "witness", + "tool_result_sanitizer", + "hitl", + "retry", + ] { + assert!( + middleware_by_name(name, &config).is_ok(), + "should resolve {name}" + ); + } } #[test] - fn test_model_response_text() { - let resp = ModelResponse::text("hello"); - assert_eq!(resp.message.content, "hello"); - assert_eq!(resp.message.role, Role::Assistant); - assert!(resp.tool_calls.is_empty()); + fn test_middleware_by_name_unknown() { + let err = middleware_by_name("does_not_exist", &PipelineConfig::default()).unwrap_err(); + assert_eq!(err.name, "does_not_exist"); } #[test] - fn test_runtime_default() { - let rt = Runtime::default(); - assert_eq!(rt.context, serde_json::Value::Null); + fn test_build_pipeline_from_names_rejects_unknown() { + // A typo must not silently drop a middleware — the ones most worth + // typo-ing are the security ones. + let err = + build_pipeline_from_names(&["todo", "bogus", "filesystem"], &PipelineConfig::default()) + .err() + .expect("an unknown middleware name must fail the build"); + assert_eq!(err.name, "bogus"); + } + + /// Returns one shell call and one read-only call, so an approval gate is + /// observable by which calls survive. + struct DangerousCallHandler; + + #[async_trait] + impl ModelHandler for DangerousCallHandler { + async fn call(&self, _request: ModelRequest) -> ModelResponse { + let mut response = ModelResponse::text(""); + response.tool_calls = vec![ + ToolCall { + id: "c1".into(), + name: "execute".into(), + args: serde_json::json!({"command": "rm -rf /"}), + }, + ToolCall { + id: "c2".into(), + name: "write_file".into(), + args: serde_json::json!({"path": "a.txt"}), + }, + ToolCall { + id: "c3".into(), + name: "read_file".into(), + args: serde_json::json!({"path": "a.txt"}), + }, + ]; + response + } + } + + #[tokio::test] + async fn test_by_name_hitl_gates_shell_and_writes_without_config() { + // The unconfigured by-name path is what the CLI runs; an approval gate + // that approves everything would be worse than no gate at all. + let pipeline = build_pipeline_from_names(&["hitl"], &PipelineConfig::default()).unwrap(); + let response = pipeline + .run_wrap_model_call( + ModelRequest::new(vec![Message::human("go")]), + &DangerousCallHandler, + ) + .await; + + let survived: Vec<&str> = response + .tool_calls + .iter() + .map(|c| c.name.as_str()) + .collect(); + assert_eq!(survived, vec!["read_file"]); + assert!(response.content().contains("[HITL]")); + } + + #[tokio::test] + async fn test_by_name_hitl_honours_configured_interrupt_on() { + let config = PipelineConfig { + interrupt_on: Some(vec!["read_file".into()]), + ..PipelineConfig::default() + }; + let pipeline = build_pipeline_from_names(&["hitl"], &config).unwrap(); + let response = pipeline + .run_wrap_model_call( + ModelRequest::new(vec![Message::human("go")]), + &DangerousCallHandler, + ) + .await; + + let survived: Vec<&str> = response + .tool_calls + .iter() + .map(|c| c.name.as_str()) + .collect(); + assert_eq!(survived, vec!["execute", "write_file"]); } } diff --git a/crates/rvAgent/rvagent-middleware/src/mcp_bridge.rs b/crates/rvAgent/rvagent-middleware/src/mcp_bridge.rs index 3b12e5f4e..63da89a97 100644 --- a/crates/rvAgent/rvagent-middleware/src/mcp_bridge.rs +++ b/crates/rvAgent/rvagent-middleware/src/mcp_bridge.rs @@ -6,10 +6,7 @@ use async_trait::async_trait; use serde::{Deserialize, Serialize}; -use crate::{ - AgentState, AgentStateUpdate, Middleware, ModelHandler, ModelRequest, ModelResponse, - RunnableConfig, Runtime, -}; +use crate::{AgentState, AgentStateUpdate, Middleware, ModelRequest, RunnableConfig, Runtime}; /// MCP tool call origin tracking. #[derive(Debug, Clone, Serialize, Deserialize)] @@ -81,7 +78,7 @@ impl Middleware for McpBridgeMiddleware { "mcp_bridge" } - fn before_agent( + async fn before_agent( &self, _state: &AgentState, _runtime: &Runtime, @@ -109,10 +106,6 @@ impl Middleware for McpBridgeMiddleware { request } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { - handler.call(request) - } - fn tools(&self) -> Vec> { if !self.config.enabled { return vec![]; @@ -128,6 +121,7 @@ struct McpStatusTool { config: McpBridgeConfig, } +#[async_trait] impl crate::Tool for McpStatusTool { fn name(&self) -> &str { "mcp_bridge_status" @@ -137,7 +131,7 @@ impl crate::Tool for McpStatusTool { "Returns the current MCP bridge configuration and status" } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { serde_json::json!({ "type": "object", "properties": {}, @@ -145,7 +139,7 @@ impl crate::Tool for McpStatusTool { }) } - fn invoke(&self, _args: serde_json::Value) -> Result { + async fn invoke(&self, _args: serde_json::Value) -> Result { Ok(serde_json::json!({ "enabled": self.config.enabled, "max_concurrent": self.config.max_concurrent, @@ -195,8 +189,8 @@ mod tests { assert!(!mw.is_transport_allowed("websocket")); } - #[test] - fn test_mcp_bridge_disabled() { + #[tokio::test] + async fn test_mcp_bridge_disabled() { let config = McpBridgeConfig { enabled: false, ..Default::default() @@ -207,17 +201,18 @@ mod tests { let runnable_config = RunnableConfig::default(); assert!(mw .before_agent(&state, &runtime, &runnable_config) + .await .is_none()); assert!(mw.tools().is_empty()); } - #[test] - fn test_mcp_bridge_enabled_injects_config() { + #[tokio::test] + async fn test_mcp_bridge_enabled_injects_config() { let mw = McpBridgeMiddleware::new(); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); assert!(update.unwrap().extensions.contains_key("mcp_bridge_config")); } @@ -230,13 +225,13 @@ mod tests { assert_eq!(tools[0].name(), "mcp_bridge_status"); } - #[test] - fn test_mcp_status_tool_invoke() { + #[tokio::test] + async fn test_mcp_status_tool_invoke() { use crate::Tool; let tool = McpStatusTool { config: McpBridgeConfig::default(), }; - let result = tool.invoke(serde_json::json!({})); + let result = tool.invoke(serde_json::json!({})).await; assert!(result.is_ok()); let json: serde_json::Value = serde_json::from_str(&result.unwrap()).unwrap(); assert_eq!(json["enabled"], true); diff --git a/crates/rvAgent/rvagent-middleware/src/memory.rs b/crates/rvAgent/rvagent-middleware/src/memory.rs index c73d692d9..de22ee470 100644 --- a/crates/rvAgent/rvagent-middleware/src/memory.rs +++ b/crates/rvAgent/rvagent-middleware/src/memory.rs @@ -201,13 +201,13 @@ impl Middleware for MemoryMiddleware { "memory" } - fn before_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, _config: &RunnableConfig, ) -> Option { - if state.extensions.contains_key("memory_contents") { + if crate::json_extension(state, "memory_contents").is_some() { return None; } @@ -231,7 +231,11 @@ impl Middleware for MemoryMiddleware { Some(update) } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { let contents: HashMap = request .extensions .get("memory_contents") @@ -239,12 +243,12 @@ impl Middleware for MemoryMiddleware { .unwrap_or_default(); if contents.is_empty() { - return handler.call(request); + return handler.call(request).await; } let memory_section = Self::format_agent_memory(&contents); let new_system = crate::append_to_system_message(&request.system_message, &memory_section); - handler.call(request.with_system(new_system)) + handler.call(request.with_system(new_system)).await } } @@ -252,9 +256,13 @@ impl Middleware for MemoryMiddleware { mod tests { use super::*; + use async_trait::async_trait; + struct PassthroughHandler; + + #[async_trait] impl ModelHandler for PassthroughHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(request.system_message.unwrap_or_default()) } } @@ -338,20 +346,18 @@ mod tests { assert!(mw.validate_content("any.md", "anything").is_some()); } - #[test] - fn test_before_agent_skip_if_loaded() { + #[tokio::test] + async fn test_before_agent_skip_if_loaded() { let mw = MemoryMiddleware::new(vec!["AGENTS.md".into()]); let mut state = AgentState::default(); - state - .extensions - .insert("memory_contents".into(), serde_json::json!({})); + state.set_extension("memory_contents", serde_json::json!({})); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } - #[test] - fn test_before_agent_loads() { + #[tokio::test] + async fn test_before_agent_loads() { let mut preloaded = HashMap::new(); preloaded.insert("AGENTS.md".into(), "Memory content".into()); @@ -360,7 +366,7 @@ mod tests { let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); assert!(update.unwrap().extensions.contains_key("memory_contents")); } @@ -375,13 +381,13 @@ mod tests { assert!(formatted.contains("")); } - #[test] - fn test_wrap_model_call_no_memory() { + #[tokio::test] + async fn test_wrap_model_call_no_memory() { let mw = MemoryMiddleware::new(vec![]); let request = ModelRequest::new(vec![]); let handler = PassthroughHandler; - let response = mw.wrap_model_call(request, &handler); - assert!(response.message.content.is_empty()); + let response = mw.wrap_model_call(request, &handler).await; + assert!(response.content().is_empty()); } #[test] diff --git a/crates/rvAgent/rvagent-middleware/src/patch_tool_calls.rs b/crates/rvAgent/rvagent-middleware/src/patch_tool_calls.rs index cee92f28b..89b4ce7fd 100644 --- a/crates/rvAgent/rvagent-middleware/src/patch_tool_calls.rs +++ b/crates/rvAgent/rvagent-middleware/src/patch_tool_calls.rs @@ -3,7 +3,7 @@ use async_trait::async_trait; -use crate::{AgentState, AgentStateUpdate, Message, Middleware, Role, RunnableConfig, Runtime}; +use crate::{AgentState, AgentStateUpdate, Message, Middleware, RunnableConfig, Runtime}; /// Maximum length for tool call IDs (ADR-103 C12). pub const MAX_TOOL_CALL_ID_LENGTH: usize = 128; @@ -50,7 +50,7 @@ impl Middleware for PatchToolCallsMiddleware { "patch_tool_calls" } - fn before_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, @@ -66,25 +66,25 @@ impl Middleware for PatchToolCallsMiddleware { for (i, msg) in state.messages.iter().enumerate() { patched.push(msg.clone()); - if msg.role == Role::Assistant && !msg.tool_calls.is_empty() { - for tc in &msg.tool_calls { + if let Message::Ai(ai) = msg { + for tc in &ai.tool_calls { // Validate tool call ID (ADR-103 C12) if let Err(err) = validate_tool_call_id(&tc.id) { tracing::warn!("Invalid tool call ID '{}': {}", tc.id, err); continue; } - let has_response = state.messages[i + 1..].iter().any(|m| { - m.role == Role::Tool && m.tool_call_id.as_deref() == Some(&*tc.id) - }); + let has_response = state.messages[i + 1..] + .iter() + .any(|m| matches!(m, Message::Tool(t) if t.tool_call_id == tc.id)); if !has_response { - patched.push(Message::tool( + patched.push(Message::tool_with_name( + &tc.id, format!( "Tool call {} with id {} was cancelled — another message came in before it could be completed.", tc.name, tc.id ), - &tc.id, &tc.name, )); modified = true; @@ -107,6 +107,13 @@ impl Middleware for PatchToolCallsMiddleware { mod tests { use super::*; use crate::ToolCall; + use std::sync::Arc; + + fn state_with_messages(messages: Vec) -> AgentState { + let mut state = AgentState::new(); + state.messages = Arc::new(messages); + state + } #[test] fn test_middleware_name() { @@ -147,109 +154,109 @@ mod tests { assert!(validate_tool_call_id("call/id").is_err()); } - #[test] - fn test_no_patch_needed() { + #[tokio::test] + async fn test_no_patch_needed() { let mw = PatchToolCallsMiddleware::new(); - let state = AgentState { - messages: vec![Message::user("hi"), Message::assistant("hello")], - ..Default::default() - }; + let state = state_with_messages(vec![Message::human("hi"), Message::ai("hello")]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } - #[test] - fn test_patch_dangling_tool_call() { + #[tokio::test] + async fn test_patch_dangling_tool_call() { let mw = PatchToolCallsMiddleware::new(); - let mut assistant_msg = Message::assistant("I'll use a tool"); - assistant_msg.tool_calls.push(ToolCall { - id: "call-1".into(), - name: "read_file".into(), - args: serde_json::json!({"path": "test.txt"}), - }); + let assistant_msg = Message::ai_with_tools( + "I'll use a tool", + vec![ToolCall { + id: "call-1".into(), + name: "read_file".into(), + args: serde_json::json!({"path": "test.txt"}), + }], + ); - let state = AgentState { - messages: vec![ - Message::user("help me"), - assistant_msg, - Message::user("never mind"), - ], - ..Default::default() - }; + let state = state_with_messages(vec![ + Message::human("help me"), + assistant_msg, + Message::human("never mind"), + ]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let messages = update.unwrap().messages.unwrap(); assert_eq!(messages.len(), 4); - assert_eq!(messages[2].role, Role::Tool); - assert!(messages[2].content.contains("cancelled")); - assert_eq!(messages[2].tool_call_id.as_deref(), Some("call-1")); + match &messages[2] { + Message::Tool(t) => { + assert!(t.content.contains("cancelled")); + assert_eq!(t.tool_call_id, "call-1"); + } + other => panic!("expected Tool message, got {:?}", other), + } } - #[test] - fn test_no_patch_when_response_exists() { + #[tokio::test] + async fn test_no_patch_when_response_exists() { let mw = PatchToolCallsMiddleware::new(); - let mut assistant_msg = Message::assistant("Using tool"); - assistant_msg.tool_calls.push(ToolCall { - id: "call-1".into(), - name: "read_file".into(), - args: serde_json::json!({}), - }); + let assistant_msg = Message::ai_with_tools( + "Using tool", + vec![ToolCall { + id: "call-1".into(), + name: "read_file".into(), + args: serde_json::json!({}), + }], + ); - let state = AgentState { - messages: vec![ - assistant_msg, - Message::tool("file content", "call-1", "read_file"), + let state = state_with_messages(vec![ + assistant_msg, + Message::tool_with_name("call-1", "file content", "read_file"), + ]); + let runtime = Runtime::new(); + let config = RunnableConfig::default(); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); + } + + #[tokio::test] + async fn test_patch_multiple_dangling() { + let mw = PatchToolCallsMiddleware::new(); + + let assistant_msg = Message::ai_with_tools( + "Using tools", + vec![ + ToolCall { + id: "call-1".into(), + name: "read_file".into(), + args: serde_json::json!({}), + }, + ToolCall { + id: "call-2".into(), + name: "write_file".into(), + args: serde_json::json!({}), + }, ], - ..Default::default() - }; + ); + + let state = state_with_messages(vec![assistant_msg]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); - } - - #[test] - fn test_patch_multiple_dangling() { - let mw = PatchToolCallsMiddleware::new(); - - let mut assistant_msg = Message::assistant("Using tools"); - assistant_msg.tool_calls.push(ToolCall { - id: "call-1".into(), - name: "read_file".into(), - args: serde_json::json!({}), - }); - assistant_msg.tool_calls.push(ToolCall { - id: "call-2".into(), - name: "write_file".into(), - args: serde_json::json!({}), - }); - - let state = AgentState { - messages: vec![assistant_msg], - ..Default::default() - }; - let runtime = Runtime::new(); - let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let messages = update.unwrap().messages.unwrap(); assert_eq!(messages.len(), 3); - assert_eq!(messages[1].role, Role::Tool); - assert_eq!(messages[2].role, Role::Tool); + assert!(matches!(&messages[1], Message::Tool(_))); + assert!(matches!(&messages[2], Message::Tool(_))); } - #[test] - fn test_empty_messages() { + #[tokio::test] + async fn test_empty_messages() { let mw = PatchToolCallsMiddleware::new(); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } } diff --git a/crates/rvAgent/rvagent-middleware/src/pipeline.rs b/crates/rvAgent/rvagent-middleware/src/pipeline.rs new file mode 100644 index 000000000..5f492cc49 --- /dev/null +++ b/crates/rvAgent/rvagent-middleware/src/pipeline.rs @@ -0,0 +1,142 @@ +//! Middleware pipeline execution (ADR-095) — fully async model-call chain (P0.3). + +use std::future::Future; +use std::pin::Pin; + +use crate::{ + AgentState, Middleware, ModelHandler, ModelRequest, ModelResponse, RunnableConfig, Runtime, + Tool, +}; + +/// Executes the middleware pipeline in order. +/// Mirrors LangChain's `create_agent` middleware composition. +pub struct MiddlewarePipeline { + middlewares: Vec>, +} + +impl MiddlewarePipeline { + /// Create a new pipeline from an ordered list of middlewares. + pub fn new(middlewares: Vec>) -> Self { + Self { middlewares } + } + + /// Create an empty pipeline. + pub fn empty() -> Self { + Self { + middlewares: Vec::new(), + } + } + + /// Add a middleware to the end of the pipeline. + pub fn push(&mut self, middleware: Box) { + self.middlewares.push(middleware); + } + + /// Number of middlewares in the pipeline. + pub fn len(&self) -> usize { + self.middlewares.len() + } + + /// Whether the pipeline is empty. + pub fn is_empty(&self) -> bool { + self.middlewares.is_empty() + } + + /// Get middleware names in order. + pub fn names(&self) -> Vec<&str> { + self.middlewares.iter().map(|mw| mw.name()).collect() + } + + /// Run `before_agent` hooks in order, accumulating state updates. + pub async fn run_before_agent( + &self, + state: &mut AgentState, + runtime: &Runtime, + config: &RunnableConfig, + ) { + for mw in &self.middlewares { + if let Some(update) = mw.before_agent(state, runtime, config).await { + update.apply_to(state); + } + } + } + + /// Collect all tools from all middlewares. + pub fn collect_tools(&self) -> Vec> { + self.middlewares.iter().flat_map(|mw| mw.tools()).collect() + } + + /// Run `modify_request` through all middlewares in order. + pub fn run_modify_request(&self, mut request: ModelRequest) -> ModelRequest { + for mw in &self.middlewares { + request = mw.modify_request(request); + } + request + } + + /// Run the async `wrap_model_call` chain. + /// Middlewares are chained so the outermost (first) wraps the innermost (last). + pub async fn run_wrap_model_call( + &self, + request: ModelRequest, + base_handler: &dyn ModelHandler, + ) -> ModelResponse { + chain_call(&self.middlewares, request, base_handler).await + } + + /// Full pipeline run: before_agent, collect tools, modify_request, wrap_model_call. + pub async fn run( + &self, + state: &mut AgentState, + runtime: &Runtime, + config: &RunnableConfig, + mut request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { + // 1. Run before_agent hooks + self.run_before_agent(state, runtime, config).await; + + // 2. Collect tools from all middlewares + for tool in self.collect_tools() { + request.tools.push(tool.definition()); + } + + // 3. Run modify_request + request = self.run_modify_request(request); + + // 4. Run the async wrap_model_call chain + self.run_wrap_model_call(request, handler).await + } +} + +type BoxResponseFuture<'a> = Pin + Send + 'a>>; + +/// Recursively chain `wrap_model_call` futures from the outside in. +fn chain_call<'a>( + middlewares: &'a [Box], + request: ModelRequest, + handler: &'a dyn ModelHandler, +) -> BoxResponseFuture<'a> { + Box::pin(async move { + match middlewares.split_first() { + None => handler.call(request).await, + Some((first, rest)) => { + let inner = ChainedHandler { rest, handler }; + first.wrap_model_call(request, &inner).await + } + } + }) +} + +/// Handler that forwards to the remainder of the middleware chain. +struct ChainedHandler<'a> { + rest: &'a [Box], + handler: &'a dyn ModelHandler, +} + +#[async_trait::async_trait] +impl ModelHandler for ChainedHandler<'_> { + async fn call(&self, request: ModelRequest) -> ModelResponse { + chain_call(self.rest, request, self.handler).await + } +} diff --git a/crates/rvAgent/rvagent-middleware/src/pipeline_model.rs b/crates/rvAgent/rvagent-middleware/src/pipeline_model.rs new file mode 100644 index 000000000..cfbf83731 --- /dev/null +++ b/crates/rvAgent/rvagent-middleware/src/pipeline_model.rs @@ -0,0 +1,308 @@ +//! `PipelineModel` — a `rvagent_core::models::ChatModel` adapter that runs +//! every model call through a `MiddlewarePipeline` (P0.3 wiring). +//! +//! This lets `AgentGraph` stay unchanged while gaining the full middleware +//! stack: the graph calls `ChatModel::complete`, and this adapter routes the +//! call through `modify_request` and the async `wrap_model_call` chain before +//! delegating to the wrapped inner model. + +use std::sync::{Arc, Mutex}; + +use async_trait::async_trait; + +use rvagent_core::error::{Result as CoreResult, RvAgentError}; +use rvagent_core::models::{ChatModel, ToolDefinition}; + +use crate::{Message, MiddlewarePipeline, ModelHandler, ModelRequest, ModelResponse}; + +/// A `ChatModel` that wraps an inner model with a middleware pipeline. +/// +/// Notes: +/// - A leading `Message::System` in the input is hoisted into the request's +/// `system_message` slot so middleware can append to it; the base handler +/// reassembles it before calling the inner model. +/// - Inner model errors surface to the chain as a `"error: …"` text response +/// (so e.g. `RetryMiddleware` can retry them); if the *final* inner call +/// still failed, the original error is propagated to the caller. +/// - Middleware-provided tools are NOT auto-advertised here; the tool set is +/// whatever the graph's `ToolExecutor` advertises. Use +/// `MiddlewarePipeline::collect_tools` to register middleware tools with an +/// executor if desired. +pub struct PipelineModel { + inner: M, + pipeline: Arc, +} + +impl PipelineModel { + /// Wrap `inner` with the given middleware pipeline. + pub fn new(inner: M, pipeline: Arc) -> Self { + Self { inner, pipeline } + } + + /// Access the pipeline. + pub fn pipeline(&self) -> &Arc { + &self.pipeline + } + + /// Access the wrapped inner model. + pub fn inner(&self) -> &M { + &self.inner + } +} + +/// Innermost handler: calls the real model. +struct BaseHandler<'a, M: ChatModel> { + model: &'a M, + /// Error from the most recent inner call (cleared at the start of each + /// call, so a successful retry clears a previous failure). + last_error: Mutex>, +} + +#[async_trait] +impl ModelHandler for BaseHandler<'_, M> { + async fn call(&self, request: ModelRequest) -> ModelResponse { + *self.last_error.lock().unwrap() = None; + + let ModelRequest { + system_message, + mut messages, + tools, + .. + } = request; + if let Some(sys) = system_message { + messages.insert(0, Message::system(sys)); + } + + match self.model.complete(&messages, &tools).await { + Ok(message) => { + let tool_calls = match &message { + Message::Ai(ai) => ai.tool_calls.clone(), + _ => Vec::new(), + }; + ModelResponse { + message, + tool_calls, + usage: None, + } + } + Err(e) => { + // Feed the failure into the chain as an "error: …" response so + // retry-style middleware can react; remember it for propagation. + let resp = ModelResponse::text(format!("error: {e}")); + *self.last_error.lock().unwrap() = Some(e); + resp + } + } + } +} + +#[async_trait] +impl ChatModel for PipelineModel { + async fn complete( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> CoreResult { + // Hoist a leading system message into the request's system slot. + let mut msgs = messages.to_vec(); + let system_message = match msgs.first() { + Some(Message::System(sys)) => { + let content = sys.content.clone(); + msgs.remove(0); + Some(content) + } + _ => None, + }; + + let mut request = ModelRequest::new(msgs).with_system(system_message); + request.tools = tools.to_vec(); + let request = self.pipeline.run_modify_request(request); + + let handler = BaseHandler { + model: &self.inner, + last_error: Mutex::new(None), + }; + let response = self.pipeline.run_wrap_model_call(request, &handler).await; + + // If the final inner call failed, propagate the original error. + if let Some(err) = handler.last_error.lock().unwrap().take() { + return Err(err); + } + + // Reattach the (possibly middleware-filtered) tool calls to the message. + let ModelResponse { + mut message, + tool_calls, + .. + } = response; + if let Message::Ai(ai) = &mut message { + ai.tool_calls = tool_calls; + } + Ok(message) + } + + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> CoreResult> { + let msg = self.complete(messages, tools).await?; + Ok(vec![msg]) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::{append_to_system_message, Middleware}; + use rvagent_core::messages::ToolCall; + + struct EchoModel; + + #[async_trait] + impl ChatModel for EchoModel { + async fn complete( + &self, + messages: &[Message], + _tools: &[ToolDefinition], + ) -> CoreResult { + // Echo the system message content (if any) for inspection. + let sys = messages + .iter() + .find_map(|m| match m { + Message::System(s) => Some(s.content.clone()), + _ => None, + }) + .unwrap_or_default(); + Ok(Message::ai(format!("sys:[{sys}] n={}", messages.len()))) + } + + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> CoreResult> { + Ok(vec![self.complete(messages, tools).await?]) + } + } + + struct FailingModel; + + #[async_trait] + impl ChatModel for FailingModel { + async fn complete( + &self, + _messages: &[Message], + _tools: &[ToolDefinition], + ) -> CoreResult { + Err(RvAgentError::model("boom")) + } + + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> CoreResult> { + Ok(vec![self.complete(messages, tools).await?]) + } + } + + struct AppendMw(&'static str); + + #[async_trait] + impl Middleware for AppendMw { + fn name(&self) -> &str { + "append" + } + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { + let sys = append_to_system_message(&request.system_message, self.0); + handler.call(request.with_system(sys)).await + } + } + + /// Middleware that drops all tool calls (HITL-style filtering). + struct DropToolCalls; + + #[async_trait] + impl Middleware for DropToolCalls { + fn name(&self) -> &str { + "drop_tool_calls" + } + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { + let mut response = handler.call(request).await; + response.tool_calls.clear(); + response + } + } + + struct ToolCallingModel; + + #[async_trait] + impl ChatModel for ToolCallingModel { + async fn complete( + &self, + _messages: &[Message], + _tools: &[ToolDefinition], + ) -> CoreResult { + Ok(Message::ai_with_tools( + "calling", + vec![ToolCall { + id: "tc1".into(), + name: "ls".into(), + args: serde_json::json!({}), + }], + )) + } + + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> CoreResult> { + Ok(vec![self.complete(messages, tools).await?]) + } + } + + #[tokio::test] + async fn test_pipeline_model_appends_system() { + let pipeline = Arc::new(MiddlewarePipeline::new(vec![Box::new(AppendMw("EXTRA"))])); + let model = PipelineModel::new(EchoModel, pipeline); + + let messages = vec![Message::system("base"), Message::human("hi")]; + let out = model.complete(&messages, &[]).await.unwrap(); + assert!(out.content().contains("base")); + assert!(out.content().contains("EXTRA")); + } + + #[tokio::test] + async fn test_pipeline_model_propagates_errors() { + let pipeline = Arc::new(MiddlewarePipeline::empty()); + let model = PipelineModel::new(FailingModel, pipeline); + let err = model.complete(&[Message::human("hi")], &[]).await; + assert!(err.is_err()); + } + + #[tokio::test] + async fn test_pipeline_model_middleware_filters_tool_calls() { + let pipeline = Arc::new(MiddlewarePipeline::new(vec![Box::new(DropToolCalls)])); + let model = PipelineModel::new(ToolCallingModel, pipeline); + let out = model.complete(&[Message::human("hi")], &[]).await.unwrap(); + assert!(!out.has_tool_calls(), "middleware must filter tool calls"); + } + + #[tokio::test] + async fn test_pipeline_model_empty_pipeline_passthrough() { + let pipeline = Arc::new(MiddlewarePipeline::empty()); + let model = PipelineModel::new(ToolCallingModel, pipeline); + let out = model.complete(&[Message::human("hi")], &[]).await.unwrap(); + assert!(out.has_tool_calls()); + } +} diff --git a/crates/rvAgent/rvagent-middleware/src/prompt_caching.rs b/crates/rvAgent/rvagent-middleware/src/prompt_caching.rs index 80f6ccb43..5f25a39ce 100644 --- a/crates/rvAgent/rvagent-middleware/src/prompt_caching.rs +++ b/crates/rvAgent/rvagent-middleware/src/prompt_caching.rs @@ -73,7 +73,7 @@ mod tests { #[test] fn test_modify_request_with_system() { let mw = PromptCachingMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hi")]) + let request = ModelRequest::new(vec![Message::human("hi")]) .with_system(Some("You are helpful.".into())); let modified = mw.modify_request(request); @@ -84,7 +84,7 @@ mod tests { #[test] fn test_modify_request_without_system() { let mw = PromptCachingMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hi")]); + let request = ModelRequest::new(vec![Message::human("hi")]); let modified = mw.modify_request(request); assert!(!modified.cache_control.contains_key("system")); @@ -93,11 +93,11 @@ mod tests { #[test] fn test_modify_request_with_tools() { let mw = PromptCachingMiddleware::new(); - let mut request = ModelRequest::new(vec![Message::user("hi")]); + let mut request = ModelRequest::new(vec![Message::human("hi")]); request.tools.push(crate::ToolDefinition { name: "test".into(), description: "test tool".into(), - parameters: serde_json::json!({}), + input_schema: serde_json::json!({}), }); let modified = mw.modify_request(request); diff --git a/crates/rvAgent/rvagent-middleware/src/retry.rs b/crates/rvAgent/rvagent-middleware/src/retry.rs index d08371e23..8bc6fb941 100644 --- a/crates/rvAgent/rvagent-middleware/src/retry.rs +++ b/crates/rvAgent/rvagent-middleware/src/retry.rs @@ -4,7 +4,6 @@ //! a transient error (e.g., content starts with `"error:"` or is empty). use std::sync::atomic::{AtomicU64, Ordering}; -use std::thread; use std::time::Duration; use async_trait::async_trait; @@ -13,13 +12,23 @@ use crate::{Middleware, ModelHandler, ModelRequest, ModelResponse}; /// Determines whether a `ModelResponse` represents a transient error worth retrying. /// -/// Heuristic: the response is considered an error if its content is empty or -/// starts with the prefix `"error:"` (case-insensitive). +/// Heuristic: the response is considered an error if its content starts with +/// the prefix `"error:"` (case-insensitive), or if it is completely empty — +/// no text AND no tool calls. A response with tool calls but no text is a +/// perfectly valid tool-use turn and must never be retried. fn is_transient_error(response: &ModelResponse) -> bool { - let content = &response.message.content; - content.is_empty() || content.to_ascii_lowercase().starts_with("error:") + let content = response.content(); + (content.is_empty() && response.tool_calls.is_empty()) + || content.to_ascii_lowercase().starts_with("error:") } +/// Upper bound on a single backoff delay (1 minute). +/// +/// Doubling is unbounded by nature, so a caller-supplied `max_retries` of 63 +/// yields a sleep of ~584 million years — indistinguishable from a hang. The +/// cap is what makes the backoff recoverable rather than terminal. +pub const MAX_BACKOFF_MS: u64 = 60_000; + /// Retry middleware that wraps model calls with exponential backoff. /// /// # Configuration @@ -29,7 +38,8 @@ fn is_transient_error(response: &ModelResponse) -> bool { /// | `max_retries` | 3 | Maximum number of retry attempts | /// | `initial_delay_ms` | 100 | Delay before the first retry (ms) | /// -/// The delay doubles after each attempt: `initial_delay_ms * 2^attempt`. +/// The delay doubles after each attempt: `initial_delay_ms * 2^attempt`, +/// capped at [`MAX_BACKOFF_MS`]. /// /// # Metrics /// @@ -65,6 +75,16 @@ impl RetryMiddleware { self.total_retries.load(Ordering::Relaxed) } + /// Backoff delay before retry `attempt` (0-based), capped and overflow-safe. + /// + /// Saturating arithmetic keeps `2^attempt` from panicking in debug builds; + /// the cap is what keeps the resulting sleep finite in practice. + fn backoff_ms(&self, attempt: u32) -> u64 { + self.initial_delay_ms + .saturating_mul(2u64.saturating_pow(attempt)) + .min(MAX_BACKOFF_MS) + } + /// Reset all counters to zero. pub fn reset_metrics(&self) { self.retry_count.store(0, Ordering::Relaxed); @@ -84,8 +104,12 @@ impl Middleware for RetryMiddleware { "retry" } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { - let mut response = handler.call(request.clone()); + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { + let mut response = handler.call(request.clone()).await; if !is_transient_error(&response) { return response; @@ -95,12 +119,12 @@ impl Middleware for RetryMiddleware { self.retry_count.fetch_add(1, Ordering::Relaxed); for attempt in 0..self.max_retries { - let delay_ms = self.initial_delay_ms * 2u64.pow(attempt); - thread::sleep(Duration::from_millis(delay_ms)); + let delay_ms = self.backoff_ms(attempt); + tokio::time::sleep(Duration::from_millis(delay_ms)).await; self.total_retries.fetch_add(1, Ordering::Relaxed); - response = handler.call(request.clone()); + response = handler.call(request.clone()).await; if !is_transient_error(&response) { return response; @@ -122,6 +146,25 @@ mod tests { use crate::{Message, ModelRequest, ModelResponse}; use std::sync::atomic::AtomicU32; + #[test] + fn test_backoff_doubles_then_caps() { + let mw = RetryMiddleware::new(3, 100); + assert_eq!(mw.backoff_ms(0), 100); + assert_eq!(mw.backoff_ms(1), 200); + assert_eq!(mw.backoff_ms(2), 400); + // Doubling past the cap flattens instead of running away. + assert_eq!(mw.backoff_ms(20), MAX_BACKOFF_MS); + } + + #[test] + fn test_backoff_survives_absurd_attempt_counts() { + // attempt 63 overflows `2^attempt` and, uncapped, sleeps ~584M years. + let mw = RetryMiddleware::new(u32::MAX, u64::MAX); + for attempt in [31u32, 63, 64, u32::MAX] { + assert_eq!(mw.backoff_ms(attempt), MAX_BACKOFF_MS, "attempt {attempt}"); + } + } + /// A handler that fails `n` times then succeeds. struct FailNHandler { remaining_failures: AtomicU32, @@ -135,8 +178,9 @@ mod tests { } } + #[async_trait] impl ModelHandler for FailNHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { let remaining = self.remaining_failures.load(Ordering::SeqCst); if remaining > 0 { self.remaining_failures.fetch_sub(1, Ordering::SeqCst); @@ -149,53 +193,57 @@ mod tests { /// A handler that always succeeds. struct SuccessHandler; + + #[async_trait] impl ModelHandler for SuccessHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { ModelResponse::text("ok") } } /// A handler that always fails with an error response. struct AlwaysFailHandler; + + #[async_trait] impl ModelHandler for AlwaysFailHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { ModelResponse::text("error: permanent failure") } } fn make_request() -> ModelRequest { - ModelRequest::new(vec![Message::user("hello")]) + ModelRequest::new(vec![Message::human("hello")]) } - #[test] - fn test_no_retry_on_success() { + #[tokio::test] + async fn test_no_retry_on_success() { let mw = RetryMiddleware::default(); let handler = SuccessHandler; - let resp = mw.wrap_model_call(make_request(), &handler); + let resp = mw.wrap_model_call(make_request(), &handler).await; - assert_eq!(resp.message.content, "ok"); + assert_eq!(resp.content(), "ok"); assert_eq!(mw.retry_count(), 0); assert_eq!(mw.total_retries(), 0); } - #[test] - fn test_retry_succeeds_after_failures() { + #[tokio::test] + async fn test_retry_succeeds_after_failures() { let mw = RetryMiddleware::new(3, 1); // 1ms delay for fast tests let handler = FailNHandler::new(2); // fails twice, then succeeds - let resp = mw.wrap_model_call(make_request(), &handler); + let resp = mw.wrap_model_call(make_request(), &handler).await; - assert_eq!(resp.message.content, "success"); + assert_eq!(resp.content(), "success"); assert_eq!(mw.retry_count(), 1); assert_eq!(mw.total_retries(), 2); } - #[test] - fn test_retries_exhausted() { + #[tokio::test] + async fn test_retries_exhausted() { let mw = RetryMiddleware::new(2, 1); let handler = AlwaysFailHandler; - let resp = mw.wrap_model_call(make_request(), &handler); + let resp = mw.wrap_model_call(make_request(), &handler).await; - assert!(resp.message.content.starts_with("error:")); + assert!(resp.content().starts_with("error:")); assert_eq!(mw.retry_count(), 1); assert_eq!(mw.total_retries(), 2); } @@ -207,11 +255,11 @@ mod tests { assert_eq!(mw.initial_delay_ms, 100); } - #[test] - fn test_reset_metrics() { + #[tokio::test] + async fn test_reset_metrics() { let mw = RetryMiddleware::new(3, 1); let handler = FailNHandler::new(1); - let _ = mw.wrap_model_call(make_request(), &handler); + let _ = mw.wrap_model_call(make_request(), &handler).await; assert!(mw.retry_count() > 0); mw.reset_metrics(); @@ -237,47 +285,59 @@ mod tests { assert!(is_transient_error(&resp)); } + #[test] + fn test_is_transient_error_empty_content_with_tool_calls() { + // A tool-use turn often has no text content — it is NOT an error. + let mut resp = ModelResponse::text(""); + resp.tool_calls = vec![crate::ToolCall { + id: "tc1".into(), + name: "read_file".into(), + args: serde_json::json!({}), + }]; + assert!(!is_transient_error(&resp)); + } + #[test] fn test_is_transient_error_normal_response() { let resp = ModelResponse::text("Here is the answer."); assert!(!is_transient_error(&resp)); } - #[test] - fn test_retry_first_attempt_succeeds() { + #[tokio::test] + async fn test_retry_first_attempt_succeeds() { // Edge case: handler fails on first call but succeeds on first retry (attempt 0). let mw = RetryMiddleware::new(5, 1); let handler = FailNHandler::new(1); - let resp = mw.wrap_model_call(make_request(), &handler); + let resp = mw.wrap_model_call(make_request(), &handler).await; - assert_eq!(resp.message.content, "success"); + assert_eq!(resp.content(), "success"); assert_eq!(mw.retry_count(), 1); assert_eq!(mw.total_retries(), 1); } - #[test] - fn test_zero_max_retries() { + #[tokio::test] + async fn test_zero_max_retries() { // With max_retries = 0, the initial call is made but no retries happen. let mw = RetryMiddleware::new(0, 1); let handler = AlwaysFailHandler; - let resp = mw.wrap_model_call(make_request(), &handler); + let resp = mw.wrap_model_call(make_request(), &handler).await; - assert!(resp.message.content.starts_with("error:")); + assert!(resp.content().starts_with("error:")); assert_eq!(mw.retry_count(), 1); assert_eq!(mw.total_retries(), 0); } - #[test] - fn test_metrics_accumulate_across_calls() { + #[tokio::test] + async fn test_metrics_accumulate_across_calls() { let mw = RetryMiddleware::new(3, 1); // First call: 1 failure then success let handler1 = FailNHandler::new(1); - let _ = mw.wrap_model_call(make_request(), &handler1); + let _ = mw.wrap_model_call(make_request(), &handler1).await; // Second call: 2 failures then success let handler2 = FailNHandler::new(2); - let _ = mw.wrap_model_call(make_request(), &handler2); + let _ = mw.wrap_model_call(make_request(), &handler2).await; assert_eq!(mw.retry_count(), 2); // two calls needed retries assert_eq!(mw.total_retries(), 3); // 1 + 2 retries diff --git a/crates/rvAgent/rvagent-middleware/src/rvf_manifest.rs b/crates/rvAgent/rvagent-middleware/src/rvf_manifest.rs index 1157cc062..850799bc9 100644 --- a/crates/rvAgent/rvagent-middleware/src/rvf_manifest.rs +++ b/crates/rvAgent/rvagent-middleware/src/rvf_manifest.rs @@ -115,6 +115,7 @@ struct RvfToolAdapter { parameters_schema: serde_json::Value, } +#[async_trait] impl Tool for RvfToolAdapter { fn name(&self) -> &str { &self.name @@ -124,11 +125,11 @@ impl Tool for RvfToolAdapter { &self.description } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { self.parameters_schema.clone() } - fn invoke(&self, args: serde_json::Value) -> Result { + async fn invoke(&self, args: serde_json::Value) -> Result { // Without rvf-runtime, return a stub response indicating the tool is available // but actual execution requires the rvf-compat feature. Ok(format!( @@ -148,7 +149,7 @@ impl Middleware for RvfManifestMiddleware { "rvf_manifest" } - fn before_agent( + async fn before_agent( &self, _state: &AgentState, _runtime: &Runtime, @@ -272,14 +273,16 @@ mod tests { assert!(tools.iter().any(|t| t.name() == "rvf:format")); } - #[test] - fn test_tool_invoke() { + #[tokio::test] + async fn test_tool_invoke() { let mw = RvfManifestMiddleware::new(sample_config()); mw.mount_package(sample_manifest()); let tools = mw.tools(); let lint = tools.iter().find(|t| t.name() == "rvf:lint").unwrap(); - let result = lint.invoke(serde_json::json!({"path": "src/main.rs"})); + let result = lint + .invoke(serde_json::json!({"path": "src/main.rs"})) + .await; assert!(result.is_ok()); assert!(result.unwrap().contains("rvf:lint")); } @@ -298,8 +301,8 @@ mod tests { assert!(tools.is_empty()); } - #[test] - fn test_before_agent_injects_state() { + #[tokio::test] + async fn test_before_agent_injects_state() { let mw = RvfManifestMiddleware::new(sample_config()); mw.mount_package(sample_manifest()); @@ -307,7 +310,7 @@ mod tests { let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let update = update.unwrap(); @@ -317,15 +320,15 @@ mod tests { assert_eq!(arr[0]["package"], "test-pkg"); } - #[test] - fn test_before_agent_empty_table() { + #[tokio::test] + async fn test_before_agent_empty_table() { let mw = RvfManifestMiddleware::new(sample_config()); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_none()); } diff --git a/crates/rvAgent/rvagent-middleware/src/skills.rs b/crates/rvAgent/rvagent-middleware/src/skills.rs index dd5afb61a..b842979ee 100644 --- a/crates/rvAgent/rvagent-middleware/src/skills.rs +++ b/crates/rvAgent/rvagent-middleware/src/skills.rs @@ -247,13 +247,13 @@ impl Middleware for SkillsMiddleware { "skills" } - fn before_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, _config: &RunnableConfig, ) -> Option { - if state.extensions.contains_key("skills_metadata") { + if crate::json_extension(state, "skills_metadata").is_some() { return None; } @@ -271,7 +271,11 @@ impl Middleware for SkillsMiddleware { Some(update) } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { let skills: Vec = request .extensions .get("skills_metadata") @@ -279,7 +283,7 @@ impl Middleware for SkillsMiddleware { .unwrap_or_default(); if skills.is_empty() { - return handler.call(request); + return handler.call(request).await; } let locations = self.format_skills_locations(); @@ -289,7 +293,7 @@ impl Middleware for SkillsMiddleware { .replace("{skills_list}", &skills_list); let new_system = crate::append_to_system_message(&request.system_message, §ion); - handler.call(request.with_system(new_system)) + handler.call(request.with_system(new_system)).await } } @@ -400,16 +404,14 @@ mod tests { assert_eq!(mw.name(), "skills"); } - #[test] - fn test_before_agent_skip_if_loaded() { + #[tokio::test] + async fn test_before_agent_skip_if_loaded() { let mw = SkillsMiddleware::new(vec![]); let mut state = AgentState::default(); - state - .extensions - .insert("skills_metadata".into(), serde_json::json!([])); + state.set_extension("skills_metadata", serde_json::json!([])); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } #[test] diff --git a/crates/rvAgent/rvagent-middleware/src/sona.rs b/crates/rvAgent/rvagent-middleware/src/sona.rs index 91bc0b908..fefd351d7 100644 --- a/crates/rvAgent/rvagent-middleware/src/sona.rs +++ b/crates/rvAgent/rvagent-middleware/src/sona.rs @@ -26,8 +26,8 @@ use ruvector_sona::{ }; use crate::{ - AgentState, AgentStateUpdate, AsyncModelHandler, Middleware, ModelHandler, ModelRequest, - ModelResponse, Role, RunnableConfig, Runtime, + AgentState, AgentStateUpdate, Message, Middleware, ModelHandler, ModelRequest, ModelResponse, + RunnableConfig, Runtime, }; use async_trait::async_trait; use parking_lot::RwLock; @@ -153,7 +153,7 @@ fn estimate_quality(_request: &ModelRequest, response: &ModelResponse) -> f32 { let mut quality = 0.5f32; // Longer responses often indicate more thorough answers - let response_len = response.message.content.len(); + let response_len = response.content().len(); if response_len > 100 { quality += 0.1; } @@ -167,7 +167,7 @@ fn estimate_quality(_request: &ModelRequest, response: &ModelResponse) -> f32 { } // Check for error indicators - let content_lower = response.message.content.to_lowercase(); + let content_lower = response.content().to_lowercase(); if content_lower.contains("error") || content_lower.contains("failed") { quality -= 0.2; } @@ -309,8 +309,8 @@ impl SonaState { let query_text = request .messages .iter() - .filter(|m| matches!(m.role, Role::User)) - .map(|m| m.content.as_str()) + .filter(|m| matches!(m, Message::Human(_))) + .map(|m| m.content()) .collect::>() .join(" "); @@ -321,8 +321,7 @@ impl SonaState { let mut builder = TrajectoryBuilder::new(id, query_embedding); // Add response as a step - let response_embedding = - generate_embedding(&response.message.content, self.config.embedding_dim); + let response_embedding = generate_embedding(response.content(), self.config.embedding_dim); let quality = estimate_quality(request, response); builder.add_step(response_embedding, vec![], quality); @@ -630,7 +629,7 @@ impl Middleware for SonaMiddleware { "sona" } - fn before_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, @@ -653,10 +652,10 @@ impl Middleware for SonaMiddleware { .messages .iter() .rev() - .find(|m| matches!(m.role, Role::User)); + .find(|m| matches!(m, Message::Human(_))); if let Some(msg) = last_user_message { - let patterns = self.state.read().find_similar_patterns(&msg.content); + let patterns = self.state.read().find_similar_patterns(msg.content()); if !patterns.is_empty() { // Store patterns in extensions for potential use @@ -674,29 +673,10 @@ impl Middleware for SonaMiddleware { None } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { - if !self.is_enabled() { - return handler.call(request); - } - - let start = Instant::now(); - - // Call the underlying handler - let response = handler.call(request.clone()); - - // Record trajectory (Loop A - Instant Learning) - let latency = start.elapsed(); - self.state - .read() - .record_trajectory(&request, &response, latency); - - response - } - - async fn awrap_model_call( + async fn wrap_model_call( &self, request: ModelRequest, - handler: &dyn AsyncModelHandler, + handler: &dyn ModelHandler, ) -> ModelResponse { if !self.is_enabled() { return handler.call(request).await; @@ -781,7 +761,7 @@ mod tests { #[test] fn test_estimate_quality() { - let request = ModelRequest::new(vec![Message::user("test")]); + let request = ModelRequest::new(vec![Message::human("test")]); // Short response let short_response = ModelResponse::text("ok"); @@ -835,21 +815,23 @@ mod tests { } struct TestHandler; + + #[async_trait] impl ModelHandler for TestHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { ModelResponse::text("Test response with some content for quality estimation") } } - #[test] - fn test_wrap_model_call() { + #[tokio::test] + async fn test_wrap_model_call() { let middleware = SonaMiddleware::default_config(); let handler = TestHandler; - let request = ModelRequest::new(vec![Message::user("Hello")]); + let request = ModelRequest::new(vec![Message::human("Hello")]); - let response = middleware.wrap_model_call(request, &handler); + let response = middleware.wrap_model_call(request, &handler).await; - assert!(response.message.content.contains("Test response")); + assert!(response.content().contains("Test response")); #[cfg(feature = "sona")] { @@ -858,17 +840,17 @@ mod tests { } } - #[test] - fn test_wrap_model_call_disabled() { + #[tokio::test] + async fn test_wrap_model_call_disabled() { let middleware = SonaMiddleware::default_config(); middleware.set_enabled(false); let handler = TestHandler; - let request = ModelRequest::new(vec![Message::user("Hello")]); + let request = ModelRequest::new(vec![Message::human("Hello")]); - let response = middleware.wrap_model_call(request, &handler); + let response = middleware.wrap_model_call(request, &handler).await; - assert!(response.message.content.contains("Test response")); + assert!(response.content().contains("Test response")); // No recording when disabled let stats = middleware.stats(); diff --git a/crates/rvAgent/rvagent-middleware/src/subagents.rs b/crates/rvAgent/rvagent-middleware/src/subagents.rs index 89c603ee7..1a2dc5100 100644 --- a/crates/rvAgent/rvagent-middleware/src/subagents.rs +++ b/crates/rvAgent/rvagent-middleware/src/subagents.rs @@ -59,7 +59,7 @@ impl Middleware for SubAgentMiddleware { "subagent" } - fn before_agent( + async fn before_agent( &self, _state: &AgentState, _runtime: &Runtime, @@ -77,14 +77,18 @@ impl Middleware for SubAgentMiddleware { Some(update) } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { if self.specs.is_empty() { - return handler.call(request); + return handler.call(request).await; } let descriptions = self.format_subagent_descriptions(); let new_system = crate::append_to_system_message(&request.system_message, &descriptions); - handler.call(request.with_system(new_system)) + handler.call(request.with_system(new_system)).await } fn tools(&self) -> Vec> { @@ -95,6 +99,7 @@ impl Middleware for SubAgentMiddleware { /// Tool for spawning subagents. struct TaskTool; +#[async_trait] impl Tool for TaskTool { fn name(&self) -> &str { "task" @@ -104,7 +109,7 @@ impl Tool for TaskTool { "Spawn a subagent to handle a specific task. The subagent runs independently and returns its result." } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { serde_json::json!({ "type": "object", "properties": { @@ -125,7 +130,7 @@ impl Tool for TaskTool { }) } - fn invoke(&self, _args: serde_json::Value) -> Result { + async fn invoke(&self, _args: serde_json::Value) -> Result { Err("task tool must be invoked through the agent runtime".into()) } } @@ -148,17 +153,17 @@ mod tests { assert_eq!(tools[0].name(), "task"); } - #[test] - fn test_before_agent_no_specs() { + #[tokio::test] + async fn test_before_agent_no_specs() { let mw = SubAgentMiddleware::new(); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } - #[test] - fn test_before_agent_with_specs() { + #[tokio::test] + async fn test_before_agent_with_specs() { let specs = vec![SubAgentSpec { name: "coder".into(), description: "A coding agent".into(), @@ -170,7 +175,7 @@ mod tests { let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); assert!(update.unwrap().extensions.contains_key("subagent_specs")); } @@ -202,7 +207,7 @@ mod tests { #[test] fn test_task_tool_schema() { let tool = TaskTool; - let schema = tool.parameters_schema(); + let schema = tool.input_schema(); assert_eq!(schema["type"], "object"); let required = schema["required"].as_array().unwrap(); assert!(required.contains(&serde_json::json!("description"))); diff --git a/crates/rvAgent/rvagent-middleware/src/summarization.rs b/crates/rvAgent/rvagent-middleware/src/summarization.rs index 586da0155..dbaaf7ae0 100644 --- a/crates/rvAgent/rvagent-middleware/src/summarization.rs +++ b/crates/rvAgent/rvagent-middleware/src/summarization.rs @@ -4,7 +4,21 @@ use async_trait::async_trait; use uuid::Uuid; -use crate::{Message, Middleware, ModelHandler, ModelRequest, ModelResponse, Role}; +use crate::{Message, Middleware, ModelHandler, ModelRequest, ModelResponse}; + +/// Bytes of each user message kept in a compaction summary preview. +const PREVIEW_BYTES: usize = 100; + +/// Largest index `<= max` that starts a character, so slicing there can never +/// split a multi-byte sequence. Conversation content is arbitrary user text, +/// so a byte-index slice is a panic waiting for the first non-ASCII message. +fn floor_char_boundary(s: &str, max: usize) -> usize { + let mut n = max.min(s.len()); + while n > 0 && !s.is_char_boundary(n) { + n -= 1; + } + n +} /// Trigger configuration for auto-compaction. pub enum TriggerConfig { @@ -44,7 +58,7 @@ impl SummarizationMiddleware { fn estimate_tokens(messages: &[Message]) -> u64 { messages .iter() - .map(|m| (m.content.len() as u64) / 4 + 1) + .map(|m| (m.content().len() as u64) / 4 + 1) .sum() } @@ -69,11 +83,14 @@ impl SummarizationMiddleware { )); for msg in messages { - if msg.role == Role::User { - let preview = if msg.content.len() > 100 { - format!("{}...", &msg.content[..100]) + if let Message::Human(h) = msg { + let preview = if h.content.len() > PREVIEW_BYTES { + format!( + "{}...", + &h.content[..floor_char_boundary(&h.content, PREVIEW_BYTES)] + ) } else { - msg.content.clone() + h.content.clone() }; summary.push_str(&format!("- User: {}\n", preview)); } @@ -91,13 +108,13 @@ impl SummarizationMiddleware { fn format_for_offload(messages: &[Message]) -> String { let mut out = String::new(); for msg in messages { - let role = match msg.role { - Role::System => "system", - Role::User => "user", - Role::Assistant => "assistant", - Role::Tool => "tool", + let role = match msg { + Message::System(_) => "system", + Message::Human(_) => "user", + Message::Ai(_) => "assistant", + Message::Tool(_) => "tool", }; - out.push_str(&format!("## {}\n\n{}\n\n---\n\n", role, msg.content)); + out.push_str(&format!("## {}\n\n{}\n\n---\n\n", role, msg.content())); } out } @@ -109,7 +126,11 @@ impl Middleware for SummarizationMiddleware { "summarization" } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { let token_count = Self::estimate_tokens(&request.messages); let threshold = self.threshold(); @@ -129,9 +150,9 @@ impl Middleware for SummarizationMiddleware { let mut compacted = vec![summary]; compacted.extend_from_slice(to_keep); - handler.call(request.with_messages(compacted)) + handler.call(request.with_messages(compacted)).await } else { - handler.call(request) + handler.call(request).await } } } @@ -140,9 +161,13 @@ impl Middleware for SummarizationMiddleware { mod tests { use super::*; + use async_trait::async_trait; + struct PassthroughHandler; + + #[async_trait] impl ModelHandler for PassthroughHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(format!("messages: {}", request.messages.len())) } } @@ -162,7 +187,7 @@ mod tests { #[test] fn test_estimate_tokens() { - let messages = vec![Message::user("hello world")]; + let messages = vec![Message::human("hello world")]; let tokens = SummarizationMiddleware::estimate_tokens(&messages); assert!(tokens > 0); } @@ -180,31 +205,30 @@ mod tests { assert_eq!(mw.keep_count(1), 1); } - #[test] - fn test_no_compaction_below_threshold() { + #[tokio::test] + async fn test_no_compaction_below_threshold() { let mw = SummarizationMiddleware::new(100_000, 0.85, 0.10); - let request = ModelRequest::new(vec![Message::user("short")]); + let request = ModelRequest::new(vec![Message::human("short")]); let handler = PassthroughHandler; - let response = mw.wrap_model_call(request, &handler); - assert!(response.message.content.contains("messages: 1")); + let response = mw.wrap_model_call(request, &handler).await; + assert!(response.content().contains("messages: 1")); } - #[test] - fn test_compaction_above_threshold() { + #[tokio::test] + async fn test_compaction_above_threshold() { let mw = SummarizationMiddleware::new(10, 0.5, 0.5); let mut messages = Vec::new(); for i in 0..20 { - messages.push(Message::user(format!( + messages.push(Message::human(format!( "message {} with enough content to trigger compaction when counted", i ))); } let request = ModelRequest::new(messages); let handler = PassthroughHandler; - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; let count: usize = response - .message - .content + .content() .strip_prefix("messages: ") .unwrap() .parse() @@ -224,18 +248,42 @@ mod tests { #[test] fn test_summarize() { let messages = vec![ - Message::user("What is Rust?"), - Message::assistant("Rust is a systems programming language."), + Message::human("What is Rust?"), + Message::ai("Rust is a systems programming language."), ]; let summary = SummarizationMiddleware::summarize(&messages); - assert_eq!(summary.role, Role::System); - assert!(summary.content.contains("2 messages")); - assert!(summary.content.contains("What is Rust?")); + assert!(matches!(summary, Message::System(_))); + assert!(summary.content().contains("2 messages")); + assert!(summary.content().contains("What is Rust?")); + } + + #[test] + fn test_summarize_does_not_split_multibyte_chars() { + // Byte 100 lands mid-character for a 3-byte-per-char message, which a + // plain `&content[..100]` would panic on. + let content = "日".repeat(200); + let messages = vec![Message::human(content.clone())]; + let summary = SummarizationMiddleware::summarize(&messages); + let text = summary.content().to_string(); + assert!(text.contains("...")); + // 100 / 3 = 33 whole characters fit. + assert!(text.contains(&"日".repeat(33))); + assert!(!text.contains(&"日".repeat(34))); + } + + #[test] + fn test_summarize_preview_boundary_cases() { + for len in [98usize, 99, 100, 101, 150] { + let messages = vec![Message::human("é".repeat(len))]; + // The assertion is that this does not panic and stays valid UTF-8. + let summary = SummarizationMiddleware::summarize(&messages); + assert!(summary.content().contains("User:")); + } } #[test] fn test_format_for_offload() { - let messages = vec![Message::user("test content")]; + let messages = vec![Message::human("test content")]; let offloaded = SummarizationMiddleware::format_for_offload(&messages); assert!(offloaded.contains("## user")); assert!(offloaded.contains("test content")); diff --git a/crates/rvAgent/rvagent-middleware/src/todolist.rs b/crates/rvAgent/rvagent-middleware/src/todolist.rs index 295b2ad5c..04b4119ce 100644 --- a/crates/rvAgent/rvagent-middleware/src/todolist.rs +++ b/crates/rvAgent/rvagent-middleware/src/todolist.rs @@ -31,7 +31,7 @@ impl Middleware for TodoListMiddleware { "todolist" } - fn before_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, @@ -67,8 +67,8 @@ fn format_todos(todos: &[TodoItem]) -> String { TodoStatus::Completed => "completed", }; out.push_str(&format!( - " {}\n", - todo.id, status_str, todo.content + " {}\n", + status_str, todo.content )); } out.push_str(""); @@ -78,16 +78,17 @@ fn format_todos(todos: &[TodoItem]) -> String { /// Tool for writing/updating todo items. struct WriteTodosTool; +#[async_trait] impl Tool for WriteTodosTool { fn name(&self) -> &str { "write_todos" } fn description(&self) -> &str { - "Create or update the todo list. Provide a complete list of todo items with id, content, and status (pending, in_progress, completed)." + "Create or update the todo list. Provide a complete list of todo items with content, status (pending, in_progress, completed), and optional active_form." } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { serde_json::json!({ "type": "object", "properties": { @@ -96,14 +97,14 @@ impl Tool for WriteTodosTool { "items": { "type": "object", "properties": { - "id": { "type": "string" }, "content": { "type": "string" }, "status": { "type": "string", "enum": ["pending", "in_progress", "completed"] - } + }, + "active_form": { "type": "string" } }, - "required": ["id", "content", "status"] + "required": ["content", "status"] } } }, @@ -111,7 +112,7 @@ impl Tool for WriteTodosTool { }) } - fn invoke(&self, args: serde_json::Value) -> Result { + async fn invoke(&self, args: serde_json::Value) -> Result { let todos = args .get("todos") .and_then(|v| v.as_array()) @@ -120,10 +121,6 @@ impl Tool for WriteTodosTool { let count = todos.len(); // Validate each item for item in todos { - let _id = item - .get("id") - .and_then(|v| v.as_str()) - .ok_or("each todo must have an 'id' string")?; let _content = item .get("content") .and_then(|v| v.as_str()) @@ -146,6 +143,14 @@ impl Tool for WriteTodosTool { mod tests { use super::*; + fn todo(content: &str, status: TodoStatus) -> TodoItem { + TodoItem { + content: content.into(), + status, + active_form: String::new(), + } + } + #[test] fn test_format_todos_empty() { let result = format_todos(&[]); @@ -155,16 +160,8 @@ mod tests { #[test] fn test_format_todos() { let todos = vec![ - TodoItem { - id: "1".into(), - content: "Do something".into(), - status: TodoStatus::Pending, - }, - TodoItem { - id: "2".into(), - content: "Done".into(), - status: TodoStatus::Completed, - }, + todo("Do something", TodoStatus::Pending), + todo("Done", TodoStatus::Completed), ]; let result = format_todos(&todos); assert!(result.contains("status=\"pending\"")); @@ -172,27 +169,23 @@ mod tests { assert!(result.contains("Do something")); } - #[test] - fn test_before_agent_empty_todos() { + #[tokio::test] + async fn test_before_agent_empty_todos() { let mw = TodoListMiddleware::new(); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } - #[test] - fn test_before_agent_with_todos() { + #[tokio::test] + async fn test_before_agent_with_todos() { let mw = TodoListMiddleware::new(); let mut state = AgentState::default(); - state.todos.push(TodoItem { - id: "1".into(), - content: "Test task".into(), - status: TodoStatus::InProgress, - }); + state.push_todo(todo("Test task", TodoStatus::InProgress)); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let update = update.unwrap(); assert!(update.extensions.contains_key("todo_context")); @@ -204,37 +197,37 @@ mod tests { assert_eq!(tool.name(), "write_todos"); } - #[test] - fn test_write_todos_invoke_valid() { + #[tokio::test] + async fn test_write_todos_invoke_valid() { let tool = WriteTodosTool; let args = serde_json::json!({ "todos": [ - {"id": "1", "content": "task 1", "status": "pending"}, - {"id": "2", "content": "task 2", "status": "completed"} + {"content": "task 1", "status": "pending"}, + {"content": "task 2", "status": "completed"} ] }); - let result = tool.invoke(args); + let result = tool.invoke(args).await; assert!(result.is_ok()); assert!(result.unwrap().contains("2 todo items")); } - #[test] - fn test_write_todos_invoke_invalid_status() { + #[tokio::test] + async fn test_write_todos_invoke_invalid_status() { let tool = WriteTodosTool; let args = serde_json::json!({ - "todos": [{"id": "1", "content": "task", "status": "invalid"}] + "todos": [{"content": "task", "status": "invalid"}] }); - let result = tool.invoke(args); + let result = tool.invoke(args).await; assert!(result.is_err()); } - #[test] - fn test_write_todos_invoke_missing_field() { + #[tokio::test] + async fn test_write_todos_invoke_missing_field() { let tool = WriteTodosTool; let args = serde_json::json!({ - "todos": [{"id": "1"}] + "todos": [{"status": "pending"}] }); - let result = tool.invoke(args); + let result = tool.invoke(args).await; assert!(result.is_err()); } diff --git a/crates/rvAgent/rvagent-middleware/src/tool_sanitizer.rs b/crates/rvAgent/rvagent-middleware/src/tool_sanitizer.rs index 83cadc397..f70789624 100644 --- a/crates/rvAgent/rvagent-middleware/src/tool_sanitizer.rs +++ b/crates/rvAgent/rvagent-middleware/src/tool_sanitizer.rs @@ -4,7 +4,7 @@ use async_trait::async_trait; -use crate::{Middleware, ModelHandler, ModelRequest, ModelResponse, Role}; +use crate::{Message, Middleware, ModelHandler, ModelRequest, ModelResponse}; /// Middleware that sanitizes tool results by wrapping them in XML-like delimiters. /// @@ -51,37 +51,38 @@ impl Middleware for ToolResultSanitizerMiddleware { "tool_result_sanitizer" } - fn wrap_model_call( + async fn wrap_model_call( &self, mut request: ModelRequest, handler: &dyn ModelHandler, ) -> ModelResponse { // Sanitize all tool messages in the request for msg in &mut request.messages { - if msg.role == Role::Tool { - let tool_name = msg.tool_name.as_deref().unwrap_or("unknown"); - let tool_call_id = msg.tool_call_id.as_deref().unwrap_or("unknown"); - msg.content = Self::sanitize_tool_result(tool_name, tool_call_id, &msg.content); + if let Message::Tool(t) = msg { + let tool_name = t.tool_name.as_deref().unwrap_or("unknown"); + t.content = Self::sanitize_tool_result(tool_name, &t.tool_call_id, &t.content); } } - handler.call(request) + handler.call(request).await } } #[cfg(test)] mod tests { use super::*; - use crate::Message; + use async_trait::async_trait; struct CaptureHandler; + + #[async_trait] impl ModelHandler for CaptureHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { let tool_content = request .messages .iter() - .find(|m| m.role == Role::Tool) - .map(|m| m.content.clone()) + .find(|m| matches!(m, Message::Tool(_))) + .map(|m| m.content().to_string()) .unwrap_or_default(); ModelResponse::text(tool_content) } @@ -130,39 +131,41 @@ mod tests { assert!(result.contains("id=\"id"val\"")); } - #[test] - fn test_wrap_model_call_sanitizes_tool_messages() { + #[tokio::test] + async fn test_wrap_model_call_sanitizes_tool_messages() { let mw = ToolResultSanitizerMiddleware::new(); let request = ModelRequest::new(vec![ - Message::user("help"), - Message::tool("raw tool output", "call-1", "read_file"), + Message::human("help"), + Message::tool_with_name("call-1", "raw tool output", "read_file"), ]); let handler = CaptureHandler; - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; - assert!(response.message.content.contains("")); + assert!(response.content().contains("")); } - #[test] - fn test_wrap_model_call_skips_non_tool_messages() { + #[tokio::test] + async fn test_wrap_model_call_skips_non_tool_messages() { let mw = ToolResultSanitizerMiddleware::new(); let request = ModelRequest::new(vec![ - Message::user("not a tool message"), - Message::assistant("also not a tool"), + Message::human("not a tool message"), + Message::ai("also not a tool"), ]); struct VerifyHandler; + + #[async_trait] impl ModelHandler for VerifyHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { - assert_eq!(request.messages[0].content, "not a tool message"); - assert_eq!(request.messages[1].content, "also not a tool"); + async fn call(&self, request: ModelRequest) -> ModelResponse { + assert_eq!(request.messages[0].content(), "not a tool message"); + assert_eq!(request.messages[1].content(), "also not a tool"); ModelResponse::text("ok") } } - mw.wrap_model_call(request, &VerifyHandler); + mw.wrap_model_call(request, &VerifyHandler).await; } #[test] diff --git a/crates/rvAgent/rvagent-middleware/src/types.rs b/crates/rvAgent/rvagent-middleware/src/types.rs new file mode 100644 index 000000000..2c48d35a2 --- /dev/null +++ b/crates/rvAgent/rvagent-middleware/src/types.rs @@ -0,0 +1,186 @@ +//! Middleware request/response types built on the canonical `rvagent-core` +//! type system (P0.1 — unified types). +//! +//! `Message`, `ToolCall`, `AgentState`, `TodoItem`, `TodoStatus`, +//! `RunnableConfig`, and `ToolDefinition` are re-exported from +//! `rvagent_core`; this module only defines the middleware-specific +//! envelope types (`ModelRequest`, `ModelResponse`, `AgentStateUpdate`, …). + +use std::collections::HashMap; +use std::sync::Arc; + +use serde::{Deserialize, Serialize}; + +// Canonical types (single source of truth: rvagent-core). +pub use rvagent_core::config::RunnableConfig; +pub use rvagent_core::messages::{ + AiMessage, HumanMessage, Message, SystemMessage, ToolCall, ToolMessage, +}; +pub use rvagent_core::models::ToolDefinition; +pub use rvagent_core::state::{AgentState, FileData, TodoItem, TodoStatus}; + +/// Cache control hint for prompt caching (Anthropic). +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct CacheControl { + pub cache_type: String, +} + +/// State update returned by `before_agent`. Merged into `AgentState`. +/// +/// Extensions are stored in the typed extension slot of the core +/// `AgentState` as `serde_json::Value` entries. +#[derive(Debug, Clone, Default)] +pub struct AgentStateUpdate { + pub messages: Option>, + pub todos: Option>, + pub extensions: HashMap, +} + +impl AgentStateUpdate { + /// Merge this update into an `AgentState`. + pub fn apply_to(self, state: &mut AgentState) { + if let Some(messages) = self.messages { + state.messages = Arc::new(messages); + } + if let Some(todos) = self.todos { + state.todos = Arc::new(todos); + } + for (k, v) in self.extensions { + state.set_extension(k, v); + } + } +} + +/// Read a JSON extension value stored on the core `AgentState`. +pub fn json_extension<'a>(state: &'a AgentState, key: &str) -> Option<&'a serde_json::Value> { + state.get_extension::(key) +} + +/// Model request wrapping messages and configuration. +#[derive(Debug, Clone)] +pub struct ModelRequest { + pub system_message: Option, + pub messages: Vec, + pub tools: Vec, + pub cache_control: HashMap, + pub extensions: HashMap, +} + +impl ModelRequest { + /// Create a new model request. + pub fn new(messages: Vec) -> Self { + Self { + system_message: None, + messages, + tools: vec![], + cache_control: HashMap::new(), + extensions: HashMap::new(), + } + } + + /// Return a copy with a different system message. + pub fn with_system(mut self, system_message: Option) -> Self { + self.system_message = system_message; + self + } + + /// Return a copy with different messages. + pub fn with_messages(mut self, messages: Vec) -> Self { + self.messages = messages; + self + } +} + +/// Model response from an LLM call. +#[derive(Debug, Clone)] +pub struct ModelResponse { + pub message: Message, + pub tool_calls: Vec, + pub usage: Option, +} + +impl ModelResponse { + /// Create a simple text response (an AI message). + pub fn text(content: impl Into) -> Self { + Self { + message: Message::ai(content), + tool_calls: vec![], + usage: None, + } + } + + /// Text content of the response message. + pub fn content(&self) -> &str { + self.message.content() + } +} + +/// Token usage information. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct Usage { + pub input_tokens: u64, + pub output_tokens: u64, + #[serde(default)] + pub cache_read_tokens: u64, + #[serde(default)] + pub cache_creation_tokens: u64, +} + +/// Runtime context passed to middleware hooks. +pub struct Runtime { + pub context: serde_json::Value, + pub config: RunnableConfig, +} + +impl Runtime { + pub fn new() -> Self { + Self { + context: serde_json::Value::Null, + config: RunnableConfig::default(), + } + } +} + +impl Default for Runtime { + fn default() -> Self { + Self::new() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_model_request_with_system() { + let req = ModelRequest::new(vec![Message::human("hi")]); + assert!(req.system_message.is_none()); + let req2 = req.with_system(Some("system".into())); + assert_eq!(req2.system_message, Some("system".into())); + } + + #[test] + fn test_model_response_text() { + let resp = ModelResponse::text("hello"); + assert_eq!(resp.content(), "hello"); + assert!(matches!(resp.message, Message::Ai(_))); + assert!(resp.tool_calls.is_empty()); + } + + #[test] + fn test_runtime_default() { + let rt = Runtime::default(); + assert_eq!(rt.context, serde_json::Value::Null); + } + + #[test] + fn test_agent_state_update_apply() { + let mut state = AgentState::default(); + let mut update = AgentStateUpdate::default(); + update.messages = Some(vec![Message::human("hi")]); + update.extensions.insert("k".into(), serde_json::json!("v")); + update.apply_to(&mut state); + assert_eq!(state.message_count(), 1); + assert_eq!(json_extension(&state, "k"), Some(&serde_json::json!("v"))); + } +} diff --git a/crates/rvAgent/rvagent-middleware/src/unicode_security_middleware.rs b/crates/rvAgent/rvagent-middleware/src/unicode_security_middleware.rs index d4976b118..3fa1817fd 100644 --- a/crates/rvAgent/rvagent-middleware/src/unicode_security_middleware.rs +++ b/crates/rvAgent/rvagent-middleware/src/unicode_security_middleware.rs @@ -3,7 +3,7 @@ //! Automatically checks tool inputs and outputs for Unicode-based security threats. use crate::unicode_security::{UnicodeIssue, UnicodeSecurityChecker, UnicodeSecurityConfig}; -use crate::{AgentState, AgentStateUpdate, Message, Middleware, Role, RunnableConfig, Runtime}; +use crate::{AgentState, AgentStateUpdate, Message, Middleware, RunnableConfig, Runtime}; use async_trait::async_trait; use tracing::{debug, warn}; @@ -69,11 +69,6 @@ impl UnicodeSecurityMiddleware { self } - /// Check a message for Unicode security issues. - fn check_message(&self, msg: &Message) -> Vec { - self.checker.check(&msg.content) - } - /// Log detected issues. fn log_issues(&self, issues: &[UnicodeIssue], context: &str) { if !issues.is_empty() { @@ -95,7 +90,7 @@ impl Middleware for UnicodeSecurityMiddleware { "unicode_security" } - async fn abefore_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, @@ -104,37 +99,37 @@ impl Middleware for UnicodeSecurityMiddleware { let mut modified = false; let mut new_messages = Vec::new(); - for msg in &state.messages { + for msg in state.messages.iter() { let mut msg_copy = msg.clone(); - match msg.role { - Role::User if self.check_user_input => { - let issues = self.check_message(msg); + match msg { + Message::Human(h) if self.check_user_input => { + let issues = self.checker.check(&h.content); if !issues.is_empty() { self.log_issues(&issues, "user message"); // Sanitize if configured if self.sanitize_inputs { - msg_copy.content = self.checker.sanitize(&msg.content); + *msg_copy.content_mut() = self.checker.sanitize(&h.content); modified = true; debug!("Sanitized user message"); } } } - Role::Tool => { - let issues = self.check_message(msg); + Message::Tool(t) => { + let issues = self.checker.check(&t.content); if !issues.is_empty() { self.log_issues( &issues, &format!( "tool result: {}", - msg.tool_name.as_deref().unwrap_or("unknown") + t.tool_name.as_deref().unwrap_or("unknown") ), ); // Sanitize if configured if self.sanitize_outputs { - msg_copy.content = self.checker.sanitize(&msg.content); + *msg_copy.content_mut() = self.checker.sanitize(&t.content); modified = true; debug!("Sanitized tool output"); } @@ -148,8 +143,8 @@ impl Middleware for UnicodeSecurityMiddleware { new_messages.push(msg_copy); // Check tool call arguments (in assistant messages) - if msg.role == Role::Assistant { - for tool_call in &msg.tool_calls { + if let Message::Ai(ai) = msg { + for tool_call in &ai.tool_calls { if let Some(args_str) = tool_call.args.as_str() { let issues = self.checker.check(args_str); if !issues.is_empty() { @@ -189,6 +184,13 @@ impl Middleware for UnicodeSecurityMiddleware { mod tests { use super::*; use crate::{Message, ToolCall}; + use std::sync::Arc; + + fn state_with_messages(messages: Vec) -> AgentState { + let mut state = AgentState::new(); + state.messages = Arc::new(messages); + state + } #[tokio::test] async fn test_strict_middleware() { @@ -200,30 +202,26 @@ mod tests { async fn test_detect_bidi_in_tool_result() { let mw = UnicodeSecurityMiddleware::strict(); - let state = AgentState { - messages: vec![Message::tool( - "evil\u{202E}txt.exe", // BiDi override - "tc-1", - "filesystem", - )], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::tool_with_name( + "tc-1", + "evil\u{202E}txt.exe", // BiDi override + "filesystem", + )]); let runtime = Runtime::new(); let config = RunnableConfig::default(); // Should detect but not modify (sanitize_outputs = false by default) - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_none()); // Enable sanitization let mw2 = UnicodeSecurityMiddleware::strict().with_output_sanitization(true); - let update2 = mw2.abefore_agent(&state, &runtime, &config).await; + let update2 = mw2.before_agent(&state, &runtime, &config).await; assert!(update2.is_some()); let new_msgs = update2.unwrap().messages.unwrap(); - assert_eq!(new_msgs[0].content, "eviltxt.exe"); // BiDi stripped + assert_eq!(new_msgs[0].content(), "eviltxt.exe"); // BiDi stripped } #[tokio::test] @@ -232,48 +230,39 @@ mod tests { .with_user_input_check(true) .with_input_sanitization(true); - let state = AgentState { - messages: vec![Message::user("Hello\u{200B}world")], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::human("Hello\u{200B}world")]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let new_msgs = update.unwrap().messages.unwrap(); - assert_eq!(new_msgs[0].content, "Helloworld"); + assert_eq!(new_msgs[0].content(), "Helloworld"); } #[tokio::test] async fn test_check_tool_call_arguments() { let mw = UnicodeSecurityMiddleware::strict(); - let state = AgentState { - messages: vec![{ - let mut msg = Message::assistant(""); - msg.tool_calls = vec![ToolCall { - id: "tc-1".to_string(), - name: "write_file".to_string(), - args: serde_json::json!({ - "path": "test.txt", - "content": "evil\u{202E}txt.exe" - }), - }]; - msg + let state = state_with_messages(vec![Message::ai_with_tools( + "", + vec![ToolCall { + id: "tc-1".to_string(), + name: "write_file".to_string(), + args: serde_json::json!({ + "path": "test.txt", + "content": "evil\u{202E}txt.exe" + }), }], - todos: vec![], - extensions: Default::default(), - }; + )]); let runtime = Runtime::new(); let config = RunnableConfig::default(); // Should detect (logs warning) but not modify - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_none()); } @@ -282,16 +271,16 @@ mod tests { // Without output sanitization, should only log warnings let mw = UnicodeSecurityMiddleware::strict().with_output_sanitization(false); - let state = AgentState { - messages: vec![Message::tool("pаypal.com", "tc-1", "browser")], // Cyrillic 'а' - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::tool_with_name( + "tc-1", + "pаypal.com", // Cyrillic 'а' + "browser", + )]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; // Should detect confusable and log, but not modify (sanitize_outputs = false) assert!(update.is_none()); } @@ -302,19 +291,15 @@ mod tests { .with_user_input_check(true) .with_input_sanitization(true); - let state = AgentState { - messages: vec![ - Message::user("Hello world"), - Message::tool("OK", "tc-1", "test"), - ], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![ + Message::human("Hello world"), + Message::tool_with_name("tc-1", "OK", "test"), + ]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_none()); // No modification needed } @@ -324,16 +309,12 @@ mod tests { .with_user_input_check(true) .with_input_sanitization(true); - let state = AgentState { - messages: vec![Message::system("System\u{202E}message")], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::system("System\u{202E}message")]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_none()); // System messages are never modified } @@ -342,26 +323,22 @@ mod tests { let mw = UnicodeSecurityMiddleware::new(UnicodeSecurityConfig::permissive()) .with_output_sanitization(true); - let state = AgentState { - messages: vec![ - Message::tool("pаypal.com", "tc-1", "test"), // Confusable (should pass) - Message::tool("evil\u{202E}txt.exe", "tc-2", "test"), // BiDi (should be caught) - ], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![ + Message::tool_with_name("tc-1", "pаypal.com", "test"), // Confusable (should pass) + Message::tool_with_name("tc-2", "evil\u{202E}txt.exe", "test"), // BiDi (should be caught) + ]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let new_msgs = update.unwrap().messages.unwrap(); // First message unchanged (confusables not checked in permissive mode) - assert_eq!(new_msgs[0].content, "pаypal.com"); + assert_eq!(new_msgs[0].content(), "pаypal.com"); // Second message sanitized (BiDi always checked) - assert_eq!(new_msgs[1].content, "eviltxt.exe"); + assert_eq!(new_msgs[1].content(), "eviltxt.exe"); } #[tokio::test] @@ -371,26 +348,22 @@ mod tests { .with_input_sanitization(true) .with_output_sanitization(true); - let state = AgentState { - messages: vec![ - Message::user("Hello\u{200B}world"), - Message::assistant("Response"), - Message::tool("evil\u{202E}txt.exe", "tc-1", "filesystem"), - ], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![ + Message::human("Hello\u{200B}world"), + Message::ai("Response"), + Message::tool_with_name("tc-1", "evil\u{202E}txt.exe", "filesystem"), + ]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let new_msgs = update.unwrap().messages.unwrap(); assert_eq!(new_msgs.len(), 3); - assert_eq!(new_msgs[0].content, "Helloworld"); // User message sanitized - assert_eq!(new_msgs[1].content, "Response"); // Assistant unchanged - assert_eq!(new_msgs[2].content, "eviltxt.exe"); // Tool result sanitized + assert_eq!(new_msgs[0].content(), "Helloworld"); // User message sanitized + assert_eq!(new_msgs[1].content(), "Response"); // Assistant unchanged + assert_eq!(new_msgs[2].content(), "eviltxt.exe"); // Tool result sanitized } } diff --git a/crates/rvAgent/rvagent-middleware/src/witness.rs b/crates/rvAgent/rvagent-middleware/src/witness.rs index 3ee855fd3..75b2ec939 100644 --- a/crates/rvAgent/rvagent-middleware/src/witness.rs +++ b/crates/rvAgent/rvagent-middleware/src/witness.rs @@ -298,8 +298,12 @@ impl Middleware for WitnessMiddleware { "witness" } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { - let response = handler.call(request); + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { + let response = handler.call(request).await; // Log each tool call to the witness chain if !response.tool_calls.is_empty() { @@ -318,11 +322,15 @@ mod tests { use super::*; use crate::{Message, ToolCall}; + use async_trait::async_trait; + struct ToolCallHandler { tool_calls: Vec, } + + #[async_trait] impl ModelHandler for ToolCallHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { let mut response = ModelResponse::text("done"); response.tool_calls = self.tool_calls.clone(); response @@ -362,8 +370,8 @@ mod tests { assert_eq!(builder.entries()[1].sequence, 1); } - #[test] - fn test_wrap_model_call_records_tool_calls() { + #[tokio::test] + async fn test_wrap_model_call_records_tool_calls() { let mw = WitnessMiddleware::new(); let handler = ToolCallHandler { tool_calls: vec![ @@ -380,8 +388,8 @@ mod tests { ], }; - let request = ModelRequest::new(vec![Message::user("test")]); - let _response = mw.wrap_model_call(request, &handler); + let request = ModelRequest::new(vec![Message::human("test")]); + let _response = mw.wrap_model_call(request, &handler).await; let builder = mw.builder().lock().unwrap(); assert_eq!(builder.len(), 2); @@ -389,20 +397,20 @@ mod tests { assert_eq!(builder.entries()[1].tool_name, "execute"); } - #[test] - fn test_wrap_model_call_no_tool_calls() { + #[tokio::test] + async fn test_wrap_model_call_no_tool_calls() { let mw = WitnessMiddleware::new(); let handler = ToolCallHandler { tool_calls: vec![] }; let request = ModelRequest::new(vec![]); - let _response = mw.wrap_model_call(request, &handler); + let _response = mw.wrap_model_call(request, &handler).await; let builder = mw.builder().lock().unwrap(); assert!(builder.is_empty()); } - #[test] - fn test_thread_safety() { + #[tokio::test] + async fn test_thread_safety() { let builder = Arc::new(Mutex::new(WitnessBuilder::new())); let mw1 = WitnessMiddleware::with_builder(builder.clone()); let mw2 = WitnessMiddleware::with_builder(builder.clone()); @@ -424,8 +432,8 @@ mod tests { let req1 = ModelRequest::new(vec![]); let req2 = ModelRequest::new(vec![]); - mw1.wrap_model_call(req1, &handler1); - mw2.wrap_model_call(req2, &handler2); + mw1.wrap_model_call(req1, &handler1).await; + mw2.wrap_model_call(req2, &handler2).await; let builder = builder.lock().unwrap(); assert_eq!(builder.len(), 2); diff --git a/crates/rvAgent/rvagent-middleware/tests/hitl_tests.rs b/crates/rvAgent/rvagent-middleware/tests/hitl_tests.rs index 5ebb37da5..320634650 100644 --- a/crates/rvAgent/rvagent-middleware/tests/hitl_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/hitl_tests.rs @@ -1,5 +1,6 @@ //! Integration tests for the Human-in-the-Loop (HITL) middleware. +use async_trait::async_trait; use rvagent_middleware::hitl::{ApprovalDecision, HumanInTheLoopMiddleware}; use rvagent_middleware::{ Message, Middleware, ModelHandler, ModelRequest, ModelResponse, ToolCall, @@ -33,8 +34,9 @@ impl ToolCallHandler { } } +#[async_trait] impl ModelHandler for ToolCallHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { let mut response = ModelResponse::text("model response"); response.tool_calls = self.tool_calls.clone(); response @@ -131,13 +133,13 @@ fn test_empty_patterns_interrupts_nothing() { // Tests: wrap_model_call // --------------------------------------------------------------------------- -#[test] -fn test_wrap_filters_matching_tool_calls() { +#[tokio::test] +async fn test_wrap_filters_matching_tool_calls() { let mw = HumanInTheLoopMiddleware::new(vec!["execute".into()]); let handler = ToolCallHandler::with_names(&["execute", "read_file"]); - let request = ModelRequest::new(vec![Message::user("do something")]); + let request = ModelRequest::new(vec![Message::human("do something")]); - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; // Only read_file should remain assert_eq!(response.tool_calls.len(), 1); @@ -145,90 +147,92 @@ fn test_wrap_filters_matching_tool_calls() { // HITL message should be appended assert!( - response.message.content.contains("[HITL]"), + response.content().contains("[HITL]"), "should contain HITL marker" ); assert!( - response.message.content.contains("execute"), + response.content().contains("execute"), "should mention the interrupted tool" ); } -#[test] -fn test_wrap_no_matching_tools_passes_all_through() { +#[tokio::test] +async fn test_wrap_no_matching_tools_passes_all_through() { let mw = HumanInTheLoopMiddleware::new(vec!["dangerous_tool".into()]); let handler = ToolCallHandler::with_names(&["read_file", "ls", "glob"]); - let request = ModelRequest::new(vec![Message::user("safe operation")]); + let request = ModelRequest::new(vec![Message::human("safe operation")]); - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; assert_eq!(response.tool_calls.len(), 3); assert!( - !response.message.content.contains("[HITL]"), + !response.content().contains("[HITL]"), "should not contain HITL marker when nothing is interrupted" ); } -#[test] -fn test_wrap_all_tools_interrupted() { +#[tokio::test] +async fn test_wrap_all_tools_interrupted() { let mw = HumanInTheLoopMiddleware::new(vec!["*".into()]); let handler = ToolCallHandler::with_names(&["execute", "write_file"]); - let request = ModelRequest::new(vec![Message::user("do things")]); + let request = ModelRequest::new(vec![Message::human("do things")]); - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; assert!( response.tool_calls.is_empty(), "all tool calls should be intercepted" ); - assert!(response.message.content.contains("[HITL]")); - assert!(response.message.content.contains("execute")); - assert!(response.message.content.contains("write_file")); + assert!(response.content().contains("[HITL]")); + assert!(response.content().contains("execute")); + assert!(response.content().contains("write_file")); } -#[test] -fn test_wrap_no_tool_calls_from_handler() { +#[tokio::test] +async fn test_wrap_no_tool_calls_from_handler() { let mw = HumanInTheLoopMiddleware::new(vec!["execute".into()]); struct NoToolHandler; + + #[async_trait] impl ModelHandler for NoToolHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { ModelResponse::text("just text, no tools") } } - let request = ModelRequest::new(vec![Message::user("question")]); - let response = mw.wrap_model_call(request, &NoToolHandler); + let request = ModelRequest::new(vec![Message::human("question")]); + let response = mw.wrap_model_call(request, &NoToolHandler).await; assert!(response.tool_calls.is_empty()); assert!( - !response.message.content.contains("[HITL]"), + !response.content().contains("[HITL]"), "should not add HITL marker when no tool calls" ); } -#[test] -fn test_wrap_preserves_original_response_content() { +#[tokio::test] +async fn test_wrap_preserves_original_response_content() { let mw = HumanInTheLoopMiddleware::new(vec!["dangerous".into()]); let handler = ToolCallHandler::with_names(&["read_file"]); - let request = ModelRequest::new(vec![Message::user("hi")]); + let request = ModelRequest::new(vec![Message::human("hi")]); - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; assert!( - response.message.content.contains("model response"), + response.content().contains("model response"), "should preserve original model response content" ); } -#[test] -fn test_wrap_prefix_pattern_filters_correctly() { +#[tokio::test] +async fn test_wrap_prefix_pattern_filters_correctly() { let mw = HumanInTheLoopMiddleware::new(vec!["write_*".into()]); let handler = ToolCallHandler::with_names(&["write_file", "write_todos", "read_file", "execute"]); - let request = ModelRequest::new(vec![Message::user("do writes")]); + let request = ModelRequest::new(vec![Message::human("do writes")]); - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; assert_eq!( response.tool_calls.len(), diff --git a/crates/rvAgent/rvagent-middleware/tests/mcp_bridge_tests.rs b/crates/rvAgent/rvagent-middleware/tests/mcp_bridge_tests.rs index 35ab4cd11..764aa6670 100644 --- a/crates/rvAgent/rvagent-middleware/tests/mcp_bridge_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/mcp_bridge_tests.rs @@ -1,5 +1,6 @@ //! Integration tests for the MCP bridge middleware. +use async_trait::async_trait; use rvagent_middleware::mcp_bridge::{McpBridgeConfig, McpBridgeMiddleware}; use rvagent_middleware::{ AgentState, Message, Middleware, ModelHandler, ModelRequest, ModelResponse, RunnableConfig, @@ -12,8 +13,9 @@ use rvagent_middleware::{ struct PassthroughHandler; +#[async_trait] impl ModelHandler for PassthroughHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(format!("handled:{}", request.messages.len())) } } @@ -109,14 +111,14 @@ fn test_middleware_name() { assert_eq!(mw.name(), "mcp_bridge"); } -#[test] -fn test_before_agent_when_enabled_injects_config() { +#[tokio::test] +async fn test_before_agent_when_enabled_injects_config() { let mw = McpBridgeMiddleware::new(); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!( update.is_some(), "enabled bridge should produce state update" @@ -129,8 +131,8 @@ fn test_before_agent_when_enabled_injects_config() { ); } -#[test] -fn test_before_agent_when_disabled_returns_none() { +#[tokio::test] +async fn test_before_agent_when_disabled_returns_none() { let config = McpBridgeConfig { enabled: false, ..Default::default() @@ -140,7 +142,7 @@ fn test_before_agent_when_disabled_returns_none() { let runtime = Runtime::new(); let runnable_config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &runnable_config); + let update = mw.before_agent(&state, &runtime, &runnable_config).await; assert!( update.is_none(), "disabled bridge should not produce update" @@ -150,7 +152,7 @@ fn test_before_agent_when_disabled_returns_none() { #[test] fn test_modify_request_when_enabled_sets_flag() { let mw = McpBridgeMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hello")]); + let request = ModelRequest::new(vec![Message::human("hello")]); let modified = mw.modify_request(request); assert_eq!( @@ -167,7 +169,7 @@ fn test_modify_request_when_disabled_does_not_set_flag() { ..Default::default() }; let mw = McpBridgeMiddleware::with_config(config); - let request = ModelRequest::new(vec![Message::user("hello")]); + let request = ModelRequest::new(vec![Message::human("hello")]); let modified = mw.modify_request(request); assert!( @@ -176,14 +178,14 @@ fn test_modify_request_when_disabled_does_not_set_flag() { ); } -#[test] -fn test_wrap_model_call_passes_through() { +#[tokio::test] +async fn test_wrap_model_call_passes_through() { let mw = McpBridgeMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hi")]); - let response = mw.wrap_model_call(request, &PassthroughHandler); + let request = ModelRequest::new(vec![Message::human("hi")]); + let response = mw.wrap_model_call(request, &PassthroughHandler).await; assert!( - response.message.content.contains("handled:1"), + response.content().contains("handled:1"), "wrap_model_call should pass through to handler" ); } @@ -211,13 +213,13 @@ fn test_tools_when_disabled_provides_no_tools() { assert!(tools.is_empty()); } -#[test] -fn test_status_tool_returns_config_values() { +#[tokio::test] +async fn test_status_tool_returns_config_values() { let mw = McpBridgeMiddleware::new(); let tools = mw.tools(); let status_tool = &tools[0]; - let result = status_tool.invoke(serde_json::json!({})); + let result = status_tool.invoke(serde_json::json!({})).await; assert!(result.is_ok()); let json: serde_json::Value = serde_json::from_str(&result.unwrap()).unwrap(); @@ -230,7 +232,7 @@ fn test_status_tool_returns_config_values() { fn test_status_tool_schema() { let mw = McpBridgeMiddleware::new(); let tools = mw.tools(); - let schema = tools[0].parameters_schema(); + let schema = tools[0].input_schema(); assert!(schema.is_object()); assert!(schema["properties"].is_object()); } diff --git a/crates/rvAgent/rvagent-middleware/tests/pipeline_tests.rs b/crates/rvAgent/rvagent-middleware/tests/pipeline_tests.rs index de8e83c34..35856e153 100644 --- a/crates/rvAgent/rvagent-middleware/tests/pipeline_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/pipeline_tests.rs @@ -3,9 +3,8 @@ use async_trait::async_trait; use rvagent_middleware::{ - append_to_system_message, AgentState, AgentStateUpdate, Message, Middleware, - MiddlewarePipeline, ModelHandler, ModelRequest, ModelResponse, Role, RunnableConfig, Runtime, - Tool, ToolDefinition, + append_to_system_message, json_extension, AgentState, AgentStateUpdate, Message, Middleware, + MiddlewarePipeline, ModelHandler, ModelRequest, ModelResponse, RunnableConfig, Runtime, Tool, }; // --------------------------------------------------------------------------- @@ -33,7 +32,7 @@ impl Middleware for RecordingMiddleware { &self.label } - fn before_agent( + async fn before_agent( &self, _state: &AgentState, _runtime: &Runtime, @@ -68,9 +67,13 @@ impl Middleware for SystemAppender { &self.label } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { let new_sys = append_to_system_message(&request.system_message, &self.text); - handler.call(request.with_system(new_sys)) + handler.call(request.with_system(new_sys)).await } } @@ -93,6 +96,7 @@ struct NamedTool { name: String, } +#[async_trait] impl Tool for NamedTool { fn name(&self) -> &str { &self.name @@ -100,10 +104,10 @@ impl Tool for NamedTool { fn description(&self) -> &str { "test tool" } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { serde_json::json!({"type": "object"}) } - fn invoke(&self, _args: serde_json::Value) -> Result { + async fn invoke(&self, _args: serde_json::Value) -> Result { Ok("ok".into()) } } @@ -124,21 +128,13 @@ impl Middleware for ToolInjectorMw { /// Handler that captures the final system message. struct CaptureSystemHandler; +#[async_trait] impl ModelHandler for CaptureSystemHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(request.system_message.unwrap_or_default()) } } -/// Handler that returns the number of tool definitions. -struct CountToolsHandler; - -impl ModelHandler for CountToolsHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { - ModelResponse::text(format!("tools:{}", request.tools.len())) - } -} - // --------------------------------------------------------------------------- // Tests // --------------------------------------------------------------------------- @@ -174,35 +170,37 @@ async fn test_pipeline_before_agent_chain() { // All three middlewares should have set their extension key. assert_eq!( - state.extensions.get("visited_first"), + json_extension(&state, "visited_first"), Some(&serde_json::json!(true)), "first middleware should have run" ); assert_eq!( - state.extensions.get("visited_second"), + json_extension(&state, "visited_second"), Some(&serde_json::json!(true)), "second middleware should have run" ); assert_eq!( - state.extensions.get("visited_third"), + json_extension(&state, "visited_third"), Some(&serde_json::json!(true)), "third middleware should have run" ); } -#[test] -fn test_pipeline_wrap_model_call_chain() { +#[tokio::test] +async fn test_pipeline_wrap_model_call_chain() { // Two appenders: "A" then "B". Both should appear in the final system message. let pipeline = MiddlewarePipeline::new(vec![ Box::new(SystemAppender::new("appender_a", "<>")), Box::new(SystemAppender::new("appender_b", "<>")), ]); - let request = ModelRequest::new(vec![Message::user("hi")]).with_system(Some("base".into())); + let request = ModelRequest::new(vec![Message::human("hi")]).with_system(Some("base".into())); - let response = pipeline.run_wrap_model_call(request, &CaptureSystemHandler); + let response = pipeline + .run_wrap_model_call(request, &CaptureSystemHandler) + .await; - let sys = response.message.content; + let sys = response.content(); assert!(sys.contains("base"), "should preserve base system message"); assert!(sys.contains("<>"), "should include appender A"); assert!(sys.contains("<>"), "should include appender B"); diff --git a/crates/rvAgent/rvagent-middleware/tests/prompt_caching_tests.rs b/crates/rvAgent/rvagent-middleware/tests/prompt_caching_tests.rs index d6cf3b28e..735f2cfbc 100644 --- a/crates/rvAgent/rvagent-middleware/tests/prompt_caching_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/prompt_caching_tests.rs @@ -17,7 +17,7 @@ fn test_middleware_name() { fn test_default_cache_type_is_ephemeral() { let mw = PromptCachingMiddleware::new(); let request = - ModelRequest::new(vec![Message::user("hi")]).with_system(Some("system prompt".into())); + ModelRequest::new(vec![Message::human("hi")]).with_system(Some("system prompt".into())); let modified = mw.modify_request(request); assert_eq!(modified.cache_control["system"].cache_type, "ephemeral"); @@ -27,7 +27,7 @@ fn test_default_cache_type_is_ephemeral() { fn test_custom_cache_type() { let mw = PromptCachingMiddleware::with_cache_type("persistent"); let request = - ModelRequest::new(vec![Message::user("hi")]).with_system(Some("system prompt".into())); + ModelRequest::new(vec![Message::human("hi")]).with_system(Some("system prompt".into())); let modified = mw.modify_request(request); assert_eq!(modified.cache_control["system"].cache_type, "persistent"); @@ -46,7 +46,7 @@ fn test_default_trait_implementation() { #[test] fn test_adds_cache_control_for_system_message() { let mw = PromptCachingMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hello")]) + let request = ModelRequest::new(vec![Message::human("hello")]) .with_system(Some("You are a helpful assistant.".into())); let modified = mw.modify_request(request); @@ -61,7 +61,7 @@ fn test_adds_cache_control_for_system_message() { #[test] fn test_no_cache_control_without_system_message() { let mw = PromptCachingMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hello")]); + let request = ModelRequest::new(vec![Message::human("hello")]); let modified = mw.modify_request(request); @@ -78,11 +78,11 @@ fn test_no_cache_control_without_system_message() { #[test] fn test_adds_cache_control_for_tools() { let mw = PromptCachingMiddleware::new(); - let mut request = ModelRequest::new(vec![Message::user("hello")]); + let mut request = ModelRequest::new(vec![Message::human("hello")]); request.tools.push(ToolDefinition { name: "read_file".into(), description: "Read a file".into(), - parameters: serde_json::json!({"type": "object"}), + input_schema: serde_json::json!({"type": "object"}), }); let modified = mw.modify_request(request); @@ -97,7 +97,7 @@ fn test_adds_cache_control_for_tools() { #[test] fn test_no_cache_control_without_tools() { let mw = PromptCachingMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hello")]); + let request = ModelRequest::new(vec![Message::human("hello")]); let modified = mw.modify_request(request); @@ -115,11 +115,11 @@ fn test_no_cache_control_without_tools() { fn test_both_system_and_tools_get_cache_control() { let mw = PromptCachingMiddleware::new(); let mut request = - ModelRequest::new(vec![Message::user("hello")]).with_system(Some("system".into())); + ModelRequest::new(vec![Message::human("hello")]).with_system(Some("system".into())); request.tools.push(ToolDefinition { name: "ls".into(), description: "List files".into(), - parameters: serde_json::json!({}), + input_schema: serde_json::json!({}), }); let modified = mw.modify_request(request); @@ -144,11 +144,11 @@ fn test_neither_system_nor_tools_no_cache_control() { #[test] fn test_custom_cache_type_applies_to_both() { let mw = PromptCachingMiddleware::with_cache_type("long_lived"); - let mut request = ModelRequest::new(vec![Message::user("hi")]).with_system(Some("sys".into())); + let mut request = ModelRequest::new(vec![Message::human("hi")]).with_system(Some("sys".into())); request.tools.push(ToolDefinition { name: "tool".into(), description: "desc".into(), - parameters: serde_json::json!({}), + input_schema: serde_json::json!({}), }); let modified = mw.modify_request(request); @@ -160,14 +160,14 @@ fn test_custom_cache_type_applies_to_both() { #[test] fn test_messages_are_preserved_after_modify() { let mw = PromptCachingMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("first"), Message::assistant("second")]) + let request = ModelRequest::new(vec![Message::human("first"), Message::ai("second")]) .with_system(Some("sys".into())); let modified = mw.modify_request(request); assert_eq!(modified.messages.len(), 2); - assert_eq!(modified.messages[0].content, "first"); - assert_eq!(modified.messages[1].content, "second"); + assert_eq!(modified.messages[0].content(), "first"); + assert_eq!(modified.messages[1].content(), "second"); assert_eq!(modified.system_message, Some("sys".to_string())); } @@ -178,12 +178,12 @@ fn test_multiple_tools_get_single_cache_entry() { request.tools.push(ToolDefinition { name: "tool_a".into(), description: "a".into(), - parameters: serde_json::json!({}), + input_schema: serde_json::json!({}), }); request.tools.push(ToolDefinition { name: "tool_b".into(), description: "b".into(), - parameters: serde_json::json!({}), + input_schema: serde_json::json!({}), }); let modified = mw.modify_request(request); diff --git a/crates/rvAgent/rvagent-middleware/tests/security_tests.rs b/crates/rvAgent/rvagent-middleware/tests/security_tests.rs index af2ea760d..aaaea662c 100644 --- a/crates/rvAgent/rvagent-middleware/tests/security_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/security_tests.rs @@ -11,6 +11,7 @@ use std::collections::HashMap; use std::sync::{Arc, Mutex}; +use async_trait::async_trait; use rvagent_middleware::memory::{ compute_sha3_256, MemoryMiddleware, SecurityPolicy, TrustManifest, TrustVerification, MAX_MEMORY_FILE_SIZE, @@ -20,8 +21,8 @@ use rvagent_middleware::skills::{parse_skill_metadata, validate_skill_name, MAX_ use rvagent_middleware::tool_sanitizer::ToolResultSanitizerMiddleware; use rvagent_middleware::witness::{WitnessBuilder, WitnessMiddleware}; use rvagent_middleware::{ - AgentState, Message, Middleware, ModelHandler, ModelRequest, ModelResponse, Role, - RunnableConfig, Runtime, ToolCall, + AgentState, Message, Middleware, ModelHandler, ModelRequest, ModelResponse, RunnableConfig, + Runtime, ToolCall, }; // --------------------------------------------------------------------------- @@ -30,14 +31,16 @@ use rvagent_middleware::{ /// Handler that captures the model request for inspection. struct CaptureHandler; + +#[async_trait] impl ModelHandler for CaptureHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { // Return the first tool message's content (for sanitizer tests) let tool_content = request .messages .iter() - .find(|m| m.role == Role::Tool) - .map(|m| m.content.clone()) + .find(|m| matches!(m, Message::Tool(_))) + .map(|m| m.content().to_string()) .unwrap_or_default(); ModelResponse::text(tool_content) } @@ -47,8 +50,10 @@ impl ModelHandler for CaptureHandler { struct ToolCallResponseHandler { tool_calls: Vec, } + +#[async_trait] impl ModelHandler for ToolCallResponseHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { let mut response = ModelResponse::text("done"); response.tool_calls = self.tool_calls.clone(); response @@ -59,20 +64,20 @@ impl ModelHandler for ToolCallResponseHandler { // test_tool_result_sanitizer_wraps_output // =========================================================================== -#[test] -fn test_tool_result_sanitizer_wraps_output() { +#[tokio::test] +async fn test_tool_result_sanitizer_wraps_output() { let mw = ToolResultSanitizerMiddleware::new(); // Build a request with a tool message let request = ModelRequest::new(vec![ - Message::user("read the file"), - Message::tool("fn main() { println!(\"hello\"); }", "call-42", "read_file"), + Message::human("read the file"), + Message::tool_with_name("call-42", "fn main() { println!(\"hello\"); }", "read_file"), ]); - let response = mw.wrap_model_call(request, &CaptureHandler); + let response = mw.wrap_model_call(request, &CaptureHandler).await; // The tool message content should now be wrapped in tags - let content = &response.message.content; + let content = response.content(); assert!( content.starts_with(") -> AgentState { + let mut state = AgentState::new(); + state.messages = std::sync::Arc::new(messages); + state + } + // Scenario 1: Valid tool call ID with no response → should be patched - let mut msg_valid = Message::assistant("Using tool"); - msg_valid.tool_calls.push(ToolCall { - id: "call-abc123".into(), - name: "read_file".into(), - args: serde_json::json!({"path": "test.txt"}), - }); + let msg_valid = Message::ai_with_tools( + "Using tool", + vec![ToolCall { + id: "call-abc123".into(), + name: "read_file".into(), + args: serde_json::json!({"path": "test.txt"}), + }], + ); - let state = AgentState { - messages: vec![ - Message::user("help"), - msg_valid, - Message::user("changed my mind"), - ], - ..Default::default() - }; + let state = state_with_messages(vec![ + Message::human("help"), + msg_valid, + Message::human("changed my mind"), + ]); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some(), "Dangling tool call must be patched"); let messages = update.unwrap().messages.unwrap(); // Should have: user, assistant, synthetic tool response, user assert_eq!(messages.len(), 4); - assert_eq!(messages[2].role, Role::Tool); - assert!(messages[2].content.contains("cancelled")); + assert!(matches!(&messages[2], Message::Tool(_))); + assert!(messages[2].content().contains("cancelled")); // Scenario 2: Tool call with existing response → no patching needed - let mut msg_with_response = Message::assistant("Using tool"); - msg_with_response.tool_calls.push(ToolCall { - id: "call-xyz".into(), - name: "read_file".into(), - args: serde_json::json!({}), - }); + let msg_with_response = Message::ai_with_tools( + "Using tool", + vec![ToolCall { + id: "call-xyz".into(), + name: "read_file".into(), + args: serde_json::json!({}), + }], + ); - let state2 = AgentState { - messages: vec![ - msg_with_response, - Message::tool("file contents", "call-xyz", "read_file"), - ], - ..Default::default() - }; + let state2 = state_with_messages(vec![ + msg_with_response, + Message::tool_with_name("call-xyz", "file contents", "read_file"), + ]); - let update2 = mw.before_agent(&state2, &runtime, &config); + let update2 = mw.before_agent(&state2, &runtime, &config).await; assert!( update2.is_none(), "Tool call with existing response must not be patched" @@ -292,15 +301,15 @@ fn test_patch_tool_calls_validates_ids() { // Scenario 3: Empty messages → no update let state3 = AgentState::default(); - assert!(mw.before_agent(&state3, &runtime, &config).is_none()); + assert!(mw.before_agent(&state3, &runtime, &config).await.is_none()); } // =========================================================================== // test_memory_trust_verification // =========================================================================== -#[test] -fn test_memory_trust_verification() { +#[tokio::test] +async fn test_memory_trust_verification() { // 1. Compute hash of known content let trusted_content = "# Agent Instructions\nBe helpful and accurate."; let hash = compute_sha3_256(trusted_content.as_bytes()); @@ -345,7 +354,7 @@ fn test_memory_trust_verification() { let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw_loaded.before_agent(&state, &runtime, &config); + let update = mw_loaded.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); // 7. Test content size limit @@ -356,7 +365,7 @@ fn test_memory_trust_verification() { .with_security_policy(SecurityPolicy::Permissive) .with_preloaded(oversized_preloaded); - let update_big = mw_big.before_agent(&state, &runtime, &config); + let update_big = mw_big.before_agent(&state, &runtime, &config).await; // The update should exist but the oversized content should be filtered out assert!(update_big.is_some()); let ext = &update_big.unwrap().extensions; diff --git a/crates/rvAgent/rvagent-middleware/tests/summarization_tests.rs b/crates/rvAgent/rvagent-middleware/tests/summarization_tests.rs index 924264ebd..765442980 100644 --- a/crates/rvAgent/rvagent-middleware/tests/summarization_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/summarization_tests.rs @@ -5,8 +5,9 @@ //! - UUID-based offload filenames (SEC-015) //! - File permission expectations (0600) +use async_trait::async_trait; use rvagent_middleware::summarization::SummarizationMiddleware; -use rvagent_middleware::{Message, Middleware, ModelHandler, ModelRequest, ModelResponse, Role}; +use rvagent_middleware::{Message, Middleware, ModelHandler, ModelRequest, ModelResponse}; // --------------------------------------------------------------------------- // Helpers @@ -14,8 +15,10 @@ use rvagent_middleware::{Message, Middleware, ModelHandler, ModelRequest, ModelR /// Handler that captures the number of messages in the request. struct MessageCountHandler; + +#[async_trait] impl ModelHandler for MessageCountHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(format!("count={}", request.messages.len())) } } @@ -23,7 +26,7 @@ impl ModelHandler for MessageCountHandler { /// Generate N user messages with enough content to exceed a token threshold. fn generate_messages(n: usize, content_size: usize) -> Vec { (0..n) - .map(|i| Message::user(format!("Message {} {}", i, "x".repeat(content_size)))) + .map(|i| Message::human(format!("Message {} {}", i, "x".repeat(content_size)))) .collect() } @@ -31,8 +34,8 @@ fn generate_messages(n: usize, content_size: usize) -> Vec { // test_auto_compact_triggers // =========================================================================== -#[test] -fn test_auto_compact_triggers() { +#[tokio::test] +async fn test_auto_compact_triggers() { // Create middleware with very low threshold: max_tokens=10, trigger at 50% // so trigger at 5 tokens. Even a single message will exceed this. let mw = SummarizationMiddleware::new(10, 0.5, 0.5); @@ -54,9 +57,9 @@ fn test_auto_compact_triggers() { // With many messages that exceed the threshold, compaction should reduce count let messages = generate_messages(20, 100); let request = ModelRequest::new(messages); - let response = mw.wrap_model_call(request, &MessageCountHandler); + let response = mw.wrap_model_call(request, &MessageCountHandler).await; - let count_str = response.message.content.clone(); + let count_str = response.content().to_string(); let count: usize = count_str.strip_prefix("count=").unwrap().parse().unwrap(); assert!( count < 20, @@ -68,19 +71,25 @@ fn test_auto_compact_triggers() { // With a single short message below threshold, no compaction let mw_high = SummarizationMiddleware::new(100_000, 0.85, 0.10); - let short_request = ModelRequest::new(vec![Message::user("hello")]); - let short_response = mw_high.wrap_model_call(short_request, &MessageCountHandler); + let short_request = ModelRequest::new(vec![Message::human("hello")]); + let short_response = mw_high + .wrap_model_call(short_request, &MessageCountHandler) + .await; assert_eq!( - short_response.message.content, "count=1", + short_response.content(), + "count=1", "Short conversation must not be compacted" ); // Edge case: single message above threshold should not compact (need >1 messages) let mw_tiny = SummarizationMiddleware::new(1, 0.1, 0.5); - let single_request = ModelRequest::new(vec![Message::user("a long message that exceeds")]); - let single_response = mw_tiny.wrap_model_call(single_request, &MessageCountHandler); + let single_request = ModelRequest::new(vec![Message::human("a long message that exceeds")]); + let single_response = mw_tiny + .wrap_model_call(single_request, &MessageCountHandler) + .await; assert_eq!( - single_response.message.content, "count=1", + single_response.content(), + "count=1", "Single message should not be compacted even above threshold" ); } @@ -182,8 +191,8 @@ fn test_offload_uses_uuid_filename() { // test_file_permissions // =========================================================================== -#[test] -fn test_file_permissions() { +#[tokio::test] +async fn test_file_permissions() { // This test validates the permission model at the design level. // The SummarizationMiddleware is expected to write offloaded history // with mode 0600 (owner read/write only) per SEC-015. @@ -204,14 +213,16 @@ fn test_file_permissions() { // Use a handler that returns the first message's role info struct FirstMessageHandler; + + #[async_trait] impl ModelHandler for FirstMessageHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { if let Some(first) = request.messages.first() { - let role = match first.role { - Role::System => "system", - Role::User => "user", - Role::Assistant => "assistant", - Role::Tool => "tool", + let role = match first { + Message::System(_) => "system", + Message::Human(_) => "user", + Message::Ai(_) => "assistant", + Message::Tool(_) => "tool", }; ModelResponse::text(format!("first_role={}", role)) } else { @@ -221,13 +232,15 @@ fn test_file_permissions() { } let request = ModelRequest::new(messages); - let response = mw_compact.wrap_model_call(request, &FirstMessageHandler); + let response = mw_compact + .wrap_model_call(request, &FirstMessageHandler) + .await; // When compaction triggers, the first message should be the summary (System role) assert!( - response.message.content.contains("first_role=system"), + response.content().contains("first_role=system"), "Compacted conversation must start with a system summary message, got: {}", - response.message.content + response.content() ); // Verify that keep_fraction and trigger_fraction are clamped diff --git a/crates/rvAgent/rvagent-middleware/tests/unicode_security_integration.rs b/crates/rvAgent/rvagent-middleware/tests/unicode_security_integration.rs index 09ef94784..06bc995c2 100644 --- a/crates/rvAgent/rvagent-middleware/tests/unicode_security_integration.rs +++ b/crates/rvAgent/rvagent-middleware/tests/unicode_security_integration.rs @@ -3,9 +3,16 @@ //! Demonstrates comprehensive security checks against Unicode-based attacks. use rvagent_middleware::{ - AgentState, Message, Middleware, PipelineConfig, RunnableConfig, Runtime, ToolCall, - UnicodeSecurityChecker, UnicodeSecurityConfig, UnicodeSecurityMiddleware, + AgentState, Message, Middleware, RunnableConfig, Runtime, ToolCall, UnicodeSecurityChecker, + UnicodeSecurityConfig, UnicodeSecurityMiddleware, }; +use std::sync::Arc; + +fn state_with_messages(messages: Vec) -> AgentState { + let mut state = AgentState::new(); + state.messages = Arc::new(messages); + state +} #[tokio::test] async fn test_real_world_bidi_attack() { @@ -14,29 +21,25 @@ async fn test_real_world_bidi_attack() { .with_input_sanitization(true) .with_output_sanitization(true); - let state = AgentState { - messages: vec![ - // Attacker tries to disguise evil.exe as safe.txt - Message::tool( - "Downloaded: safe\u{202E}exe.txt", // Displays as "safeexe.txt" (reversed) - "tc-1", - "filesystem", - ), - ], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![ + // Attacker tries to disguise evil.exe as safe.txt + Message::tool_with_name( + "tc-1", + "Downloaded: safe\u{202E}exe.txt", // Displays as "safeexe.txt" (reversed) + "filesystem", + ), + ]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let new_msgs = update.unwrap().messages.unwrap(); // BiDi should be stripped - assert!(!new_msgs[0].content.contains('\u{202E}')); - assert_eq!(new_msgs[0].content, "Downloaded: safeexe.txt"); + assert!(!new_msgs[0].content().contains('\u{202E}')); + assert_eq!(new_msgs[0].content(), "Downloaded: safeexe.txt"); } #[tokio::test] @@ -70,23 +73,19 @@ async fn test_real_world_zero_width_steganography() { .with_input_sanitization(true); // User input with hidden zero-width characters encoding secret data - let state = AgentState { - messages: vec![Message::user( - "Innocent\u{200B}text\u{200C}with\u{200D}hidden\u{200B}data", - )], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::human( + "Innocent\u{200B}text\u{200C}with\u{200D}hidden\u{200B}data", + )]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let new_msgs = update.unwrap().messages.unwrap(); // All zero-width should be stripped - assert_eq!(new_msgs[0].content, "Innocenttextwithhiddendata"); + assert_eq!(new_msgs[0].content(), "Innocenttextwithhiddendata"); } #[tokio::test] @@ -94,37 +93,32 @@ async fn test_tool_call_argument_sanitization() { // Test that tool call arguments are checked for Unicode attacks let mw = UnicodeSecurityMiddleware::strict(); - let state = AgentState { - messages: vec![{ - let mut msg = Message::assistant(""); - msg.tool_calls = vec![ - ToolCall { - id: "tc-1".to_string(), - name: "write_file".to_string(), - args: serde_json::json!({ - "path": "safe\u{202E}exe.txt", - "content": "malicious content" - }), - }, - ToolCall { - id: "tc-2".to_string(), - name: "browser_navigate".to_string(), - args: serde_json::json!({ - "url": "pаypal.com" // Cyrillic 'а' - }), - }, - ]; - msg - }], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::ai_with_tools( + "", + vec![ + ToolCall { + id: "tc-1".to_string(), + name: "write_file".to_string(), + args: serde_json::json!({ + "path": "safe\u{202E}exe.txt", + "content": "malicious content" + }), + }, + ToolCall { + id: "tc-2".to_string(), + name: "browser_navigate".to_string(), + args: serde_json::json!({ + "url": "pаypal.com" // Cyrillic 'а' + }), + }, + ], + )]); let runtime = Runtime::new(); let config = RunnableConfig::default(); // Should detect issues in tool call arguments (logs warnings) - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; // With sanitize_inputs = true by default, this should be None // because sanitize() is only applied to message content, not tool args assert!(update.is_none()); @@ -152,20 +146,16 @@ async fn test_safe_multilingual_content_unmodified() { let mw = UnicodeSecurityMiddleware::new(UnicodeSecurityConfig::permissive()) .with_output_sanitization(false); - let state = AgentState { - messages: vec![Message::tool( - "Hello, 世界! Привет! مرحبا", // Multi-script greeting - "tc-1", - "translator", - )], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::tool_with_name( + "tc-1", + "Hello, 世界! Привет! مرحبا", // Multi-script greeting + "translator", + )]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; // Permissive mode doesn't check mixed scripts or confusables assert!(update.is_none()); } @@ -193,30 +183,26 @@ async fn test_comprehensive_attack_scenario() { .with_user_input_check(true) .with_input_sanitization(true); - let state = AgentState { - messages: vec![ - Message::user("Visit pаypal.com\u{200B}now!"), // Homoglyph + zero-width - Message::tool( - "Downloaded: evil\u{202E}txt.exe", // BiDi override - "tc-1", - "filesystem", - ), - ], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![ + Message::human("Visit pаypal.com\u{200B}now!"), // Homoglyph + zero-width + Message::tool_with_name( + "tc-1", + "Downloaded: evil\u{202E}txt.exe", // BiDi override + "filesystem", + ), + ]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let new_msgs = update.unwrap().messages.unwrap(); // User message: zero-width stripped - assert_eq!(new_msgs[0].content, "Visit pаypal.comnow!"); // Confusable remains - // Tool message: BiDi stripped - assert_eq!(new_msgs[1].content, "Downloaded: eviltxt.exe"); + assert_eq!(new_msgs[0].content(), "Visit pаypal.comnow!"); // Confusable remains + // Tool message: BiDi stripped + assert_eq!(new_msgs[1].content(), "Downloaded: eviltxt.exe"); } #[test] diff --git a/crates/rvAgent/rvagent-subagents/src/lib.rs b/crates/rvAgent/rvagent-subagents/src/lib.rs index 52657f0f8..948501792 100644 --- a/crates/rvAgent/rvagent-subagents/src/lib.rs +++ b/crates/rvAgent/rvagent-subagents/src/lib.rs @@ -26,18 +26,20 @@ pub use crdt_merge::{merge_subagent_results, CrdtState, MergeError, VectorClock} pub use orchestrator::{spawn_parallel, SpawnError, SubAgentOrchestrator}; use serde::{Deserialize, Serialize}; -use std::collections::HashMap; +use std::sync::Arc; use std::time::Duration; +use rvagent_core::messages::Message; + // --------------------------------------------------------------------------- -// AgentState (simplified, JSON-based for cross-crate compatibility) +// AgentState — canonical typed state from rvagent-core (ADR-103 A1) // --------------------------------------------------------------------------- -/// Agent state represented as a JSON map. +/// Agent state — the canonical typed `rvagent_core::state::AgentState`. /// -/// Matches `HashMap` from ADR-097. -/// Future work (ADR-103 A1) will replace this with a typed struct. -pub type AgentState = HashMap; +/// Replaces the former `HashMap` alias (ADR-097) +/// per ADR-103 A1 / roadmap P0.1. +pub use rvagent_core::state::AgentState; // --------------------------------------------------------------------------- // RvAgentConfig @@ -226,39 +228,33 @@ pub const EXCLUDED_STATE_KEYS: &[&str] = &[ /// Prepare a filtered state for subagent invocation. /// -/// Strips excluded keys from the parent state, then injects a single -/// human message containing the task description. +/// State-isolation semantics (ADR-097) on the typed state: parent +/// `messages`, `todos`, `memory_contents`, and `skills_metadata` are +/// excluded; `files` pass through (O(1) Arc clone); a single human message +/// containing the task description is injected. pub fn prepare_subagent_state(parent_state: &AgentState, task_description: &str) -> AgentState { - let mut state: AgentState = parent_state - .iter() - .filter(|(k, _)| !EXCLUDED_STATE_KEYS.contains(&k.as_str())) - .map(|(k, v)| (k.clone(), v.clone())) - .collect(); - - state.insert( - "messages".to_string(), - serde_json::json!([{"type": "human", "content": task_description}]), - ); - + let mut state = AgentState::new(); + // Files are not in EXCLUDED_STATE_KEYS — they pass through to the child. + state.files = Arc::clone(&parent_state.files); + state.push_message(Message::human(task_description)); state } /// Extract the final message from a subagent's result state. pub fn extract_result_message(result_state: &AgentState) -> Option { - let messages = result_state.get("messages")?; - let arr = messages.as_array()?; - let last = arr.last()?; - last.get("content") - .and_then(|c| c.as_str()) - .map(|s| s.trim_end().to_string()) + result_state + .messages + .last() + .map(|m| m.content().trim_end().to_string()) } /// Merge non-excluded state from subagent result back into parent state. +/// +/// Only non-excluded state merges back: `files` (subagent wins on path +/// conflict). Parent `messages` and `todos` are never overwritten. pub fn merge_subagent_state(parent: &mut AgentState, subagent_result: &AgentState) { - for (k, v) in subagent_result { - if !EXCLUDED_STATE_KEYS.contains(&k.as_str()) { - parent.insert(k.clone(), v.clone()); - } + for (path, data) in subagent_result.files.iter() { + parent.set_file(path.clone(), data.clone()); } } @@ -309,48 +305,53 @@ mod tests { assert_eq!(back.tools.len(), 2); } + use rvagent_core::state::{FileData, TodoItem, TodoStatus}; + + fn file(content: &str) -> FileData { + FileData { + content: content.into(), + encoding: "utf-8".into(), + modified_at: None, + } + } + #[test] fn test_state_isolation_prepare() { let mut parent = AgentState::new(); - parent.insert( - "messages".into(), - serde_json::json!([{"type": "ai", "content": "secret"}]), - ); - parent.insert("remaining_steps".into(), serde_json::json!(5)); - parent.insert("task_completion".into(), serde_json::json!(false)); - parent.insert("custom_key".into(), serde_json::json!("visible")); - parent.insert("todos".into(), serde_json::json!([])); + parent.push_message(Message::ai("secret")); + parent.push_todo(TodoItem { + content: "parent todo".into(), + status: TodoStatus::Pending, + active_form: String::new(), + }); + parent.set_file("/src/main.rs", file("fn main() {}")); + parent.memory_contents = Some(std::sync::Arc::new( + [("AGENTS.md".to_string(), "secret memory".to_string())] + .into_iter() + .collect(), + )); let child = prepare_subagent_state(&parent, "Do X"); - // Parent messages must NOT leak - let msgs = child.get("messages").unwrap().as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["content"], "Do X"); - assert_eq!(msgs[0]["type"], "human"); + // Parent messages must NOT leak — child gets exactly one human message. + assert_eq!(child.message_count(), 1); + assert_eq!(child.messages[0].content(), "Do X"); + assert!(matches!(child.messages[0], Message::Human(_))); - // Excluded keys must not appear (except messages which is replaced) - assert!(child.get("remaining_steps").is_none()); - assert!(child.get("task_completion").is_none()); - assert!(child.get("todos").is_none()); + // Excluded state must not appear. + assert!(child.todos.is_empty()); + assert!(child.memory_contents.is_none()); + assert!(child.skills_metadata.is_none()); - // Non-excluded keys must pass through - assert_eq!( - child.get("custom_key").unwrap(), - &serde_json::json!("visible") - ); + // Files pass through (non-excluded state). + assert!(child.files.contains_key("/src/main.rs")); } #[test] fn test_extract_result_message() { let mut state = AgentState::new(); - state.insert( - "messages".into(), - serde_json::json!([ - {"type": "human", "content": "do X"}, - {"type": "ai", "content": "Done with X. "} - ]), - ); + state.push_message(Message::human("do X")); + state.push_message(Message::ai("Done with X. ")); let msg = extract_result_message(&state).unwrap(); assert_eq!(msg, "Done with X."); } @@ -358,25 +359,28 @@ mod tests { #[test] fn test_merge_subagent_state() { let mut parent = AgentState::new(); - parent.insert("messages".into(), serde_json::json!([])); - parent.insert("existing".into(), serde_json::json!(1)); + parent.push_message(Message::human("parent message")); + parent.set_file("/existing.rs", file("existing")); let mut child_result = AgentState::new(); - child_result.insert( - "messages".into(), - serde_json::json!([{"type": "ai", "content": "hi"}]), - ); - child_result.insert("new_key".into(), serde_json::json!("added")); - child_result.insert("todos".into(), serde_json::json!(["leaked"])); + child_result.push_message(Message::ai("hi")); + child_result.push_todo(TodoItem { + content: "leaked".into(), + status: TodoStatus::Pending, + active_form: String::new(), + }); + child_result.set_file("/new.rs", file("added")); merge_subagent_state(&mut parent, &child_result); // messages should NOT be overwritten (excluded) - assert_eq!(parent.get("messages").unwrap(), &serde_json::json!([])); + assert_eq!(parent.message_count(), 1); + assert_eq!(parent.messages[0].content(), "parent message"); // todos should NOT leak - assert!(parent.get("todos").is_none()); - // new non-excluded keys should merge - assert_eq!(parent.get("new_key").unwrap(), &serde_json::json!("added")); + assert!(parent.todos.is_empty()); + // new non-excluded state (files) should merge + assert!(parent.files.contains_key("/new.rs")); + assert!(parent.files.contains_key("/existing.rs")); } #[test] diff --git a/crates/rvAgent/rvagent-subagents/tests/integration_tests.rs b/crates/rvAgent/rvagent-subagents/tests/integration_tests.rs index 6e4b8244f..8a809b9d4 100644 --- a/crates/rvAgent/rvagent-subagents/tests/integration_tests.rs +++ b/crates/rvAgent/rvagent-subagents/tests/integration_tests.rs @@ -1,14 +1,22 @@ //! Integration tests for rvAgent subagents. -use std::collections::HashMap; - +use rvagent_core::messages::Message; +use rvagent_core::state::{FileData, TodoItem, TodoStatus}; use rvagent_subagents::builder::compile_subagents; use rvagent_subagents::orchestrator::{spawn_parallel, SubAgentOrchestrator}; use rvagent_subagents::{ extract_result_message, merge_subagent_state, prepare_subagent_state, AgentState, - CompiledSubAgent, RvAgentConfig, SubAgentSpec, EXCLUDED_STATE_KEYS, + CompiledSubAgent, RvAgentConfig, SubAgentSpec, }; +fn file_data(content: &str) -> FileData { + FileData { + content: content.into(), + encoding: "utf-8".into(), + modified_at: None, + } +} + fn test_config() -> RvAgentConfig { RvAgentConfig { default_model: Some("anthropic:claude-sonnet-4-20250514".into()), @@ -28,24 +36,14 @@ fn mock_compiled(name: &str) -> CompiledSubAgent { } fn parent_state_with_data() -> AgentState { - let mut state = AgentState::new(); - state.insert( - "messages".into(), - serde_json::json!([ - {"type": "system", "content": "You are helpful."}, - {"type": "human", "content": "Do something."}, - ]), - ); - state.insert("remaining_steps".into(), serde_json::json!(10)); - state.insert( - "task_completion".into(), - serde_json::json!({"status": "in_progress"}), - ); - state.insert( - "files".into(), - serde_json::json!({"main.rs": "fn main() {}"}), - ); - state.insert("custom_data".into(), serde_json::json!("value")); + let mut state = AgentState::with_system_message("You are helpful."); + state.push_message(Message::human("Do something.")); + state.push_todo(TodoItem { + content: "parent task".into(), + status: TodoStatus::InProgress, + active_form: String::new(), + }); + state.set_file("main.rs", file_data("fn main() {}")); state } @@ -73,39 +71,22 @@ fn test_state_isolation() { let parent = parent_state_with_data(); let child = prepare_subagent_state(&parent, "Do a subtask"); - // remaining_steps and task_completion should be excluded - assert!( - !child.contains_key("remaining_steps"), - "remaining_steps leaked" - ); - assert!( - !child.contains_key("task_completion"), - "task_completion leaked" - ); + // Parent todos should be excluded + assert!(child.todos.is_empty(), "todos leaked"); // messages is re-created with the task description, not the parent's messages - let child_msgs = child.get("messages").unwrap().as_array().unwrap(); - assert_eq!(child_msgs.len(), 1); - assert!(child_msgs[0]["content"] - .as_str() - .unwrap() - .contains("subtask")); + assert_eq!(child.message_count(), 1); + assert!(child.messages[0].content().contains("subtask")); - // Non-excluded keys should be present - assert!(child.contains_key("files")); - assert!(child.contains_key("custom_data")); + // Non-excluded state (files) should be present + assert!(child.files.contains_key("main.rs")); } #[test] fn test_extract_result_message() { let mut state = AgentState::new(); - state.insert( - "messages".into(), - serde_json::json!([ - {"type": "ai", "content": "Working..."}, - {"type": "ai", "content": "Done! Here is the result."} - ]), - ); + state.push_message(Message::ai("Working...")); + state.push_message(Message::ai("Done! Here is the result.")); let result = extract_result_message(&state); assert!(result.is_some()); @@ -115,24 +96,18 @@ fn test_extract_result_message() { #[test] fn test_merge_preserves_parent_messages() { let mut parent = parent_state_with_data(); - let parent_msgs = parent.get("messages").cloned(); + let parent_msg_count = parent.message_count(); let mut child_result = AgentState::new(); - child_result.insert( - "messages".into(), - serde_json::json!([{"type": "ai", "content": "child"}]), - ); - child_result.insert("new_key".into(), serde_json::json!("from child")); + child_result.push_message(Message::ai("child")); + child_result.set_file("child.rs", file_data("from child")); merge_subagent_state(&mut parent, &child_result); // Parent messages must not be overwritten - assert_eq!(parent.get("messages"), parent_msgs.as_ref()); - // New keys from child should be merged - assert_eq!( - parent.get("new_key"), - Some(&serde_json::json!("from child")) - ); + assert_eq!(parent.message_count(), parent_msg_count); + // New files from child should be merged + assert!(parent.files.contains_key("child.rs")); } #[test] @@ -184,8 +159,4 @@ fn test_compilation_respects_capabilities() { fn test_extract_result_empty_messages() { let state = AgentState::new(); assert!(extract_result_message(&state).is_none()); - - let mut state2 = AgentState::new(); - state2.insert("messages".into(), serde_json::json!([])); - assert!(extract_result_message(&state2).is_none()); } diff --git a/crates/rvAgent/rvagent-subagents/tests/orchestrator_tests.rs b/crates/rvAgent/rvagent-subagents/tests/orchestrator_tests.rs index 37261f025..0d40c5a42 100644 --- a/crates/rvAgent/rvagent-subagents/tests/orchestrator_tests.rs +++ b/crates/rvAgent/rvagent-subagents/tests/orchestrator_tests.rs @@ -6,14 +6,24 @@ //! - Result validation (max length, injection detection) //! - Parallel spawning +use rvagent_core::messages::Message; +use rvagent_core::state::{FileData, SkillMetadata, TodoItem, TodoStatus}; use rvagent_subagents::builder::compile_subagents; use rvagent_subagents::orchestrator::{spawn_parallel, SubAgentOrchestrator}; use rvagent_subagents::validator::{SubAgentResultValidator, DEFAULT_MAX_RESPONSE_LENGTH}; use rvagent_subagents::{ merge_subagent_state, prepare_subagent_state, AgentState, CompiledSubAgent, RvAgentConfig, - SubAgentSpec, EXCLUDED_STATE_KEYS, + SubAgentSpec, }; +fn file_data(content: &str) -> FileData { + FileData { + content: content.into(), + encoding: "utf-8".into(), + modified_at: None, + } +} + // --------------------------------------------------------------------------- // Helpers // --------------------------------------------------------------------------- @@ -42,38 +52,26 @@ fn mock_compiled(name: &str) -> CompiledSubAgent { } fn parent_state_with_secrets() -> AgentState { - let mut state = AgentState::new(); - state.insert( - "messages".into(), - serde_json::json!([ - {"type": "system", "content": "You are a helpful assistant."}, - {"type": "human", "content": "Help me refactor main.rs"}, - {"type": "ai", "content": "I'll help you refactor."}, - ]), - ); - state.insert("remaining_steps".into(), serde_json::json!(42)); - state.insert("task_completion".into(), serde_json::json!({"done": false})); - state.insert( - "todos".into(), - serde_json::json!([ - {"id": "1", "content": "Fix bug", "status": "in_progress"} - ]), - ); - state.insert( - "structured_response".into(), - serde_json::json!({"format": "markdown"}), - ); - state.insert( - "skills_metadata".into(), - serde_json::json!([{"name": "coder"}]), - ); - state.insert( - "memory_contents".into(), - serde_json::json!({"AGENTS.md": "secret"}), - ); - // Non-excluded keys - state.insert("cwd".into(), serde_json::json!("/home/user/project")); - state.insert("project_config".into(), serde_json::json!({"lang": "rust"})); + let mut state = AgentState::with_system_message("You are a helpful assistant."); + state.push_message(Message::human("Help me refactor main.rs")); + state.push_message(Message::ai("I'll help you refactor.")); + state.push_todo(TodoItem { + content: "Fix bug".into(), + status: TodoStatus::InProgress, + active_form: String::new(), + }); + state.skills_metadata = Some(std::sync::Arc::new(vec![SkillMetadata { + name: "coder".into(), + description: "Writes code".into(), + parameters: serde_json::json!({}), + }])); + state.memory_contents = Some(std::sync::Arc::new( + [("AGENTS.md".to_string(), "secret".to_string())] + .into_iter() + .collect(), + )); + // Non-excluded state + state.set_file("/home/user/project/main.rs", file_data("fn main() {}")); state } @@ -147,85 +145,62 @@ fn test_state_isolation() { // Prepare child state let child = prepare_subagent_state(&parent, "Refactor the auth module"); - // ALL excluded keys must not appear in child state (except messages which is replaced) - for key in EXCLUDED_STATE_KEYS { - if *key == "messages" { - // Messages is replaced, not excluded entirely - continue; - } - assert!( - !child.contains_key(*key), - "Excluded key '{}' must not appear in child state", - key - ); - } - - // Verify specific excluded keys - assert!(!child.contains_key("remaining_steps")); - assert!(!child.contains_key("task_completion")); - assert!(!child.contains_key("todos")); - assert!(!child.contains_key("structured_response")); - assert!(!child.contains_key("skills_metadata")); - assert!(!child.contains_key("memory_contents")); + // Excluded state must not appear in child state + assert!(child.todos.is_empty(), "todos must not leak"); + assert!( + child.skills_metadata.is_none(), + "skills_metadata must not leak" + ); + assert!( + child.memory_contents.is_none(), + "memory_contents must not leak" + ); // Messages must be replaced with task description - let child_msgs = child.get("messages").unwrap().as_array().unwrap(); - assert_eq!(child_msgs.len(), 1, "Child must have exactly 1 message"); - assert_eq!(child_msgs[0]["type"], "human"); - assert!(child_msgs[0]["content"] - .as_str() - .unwrap() + assert_eq!( + child.message_count(), + 1, + "Child must have exactly 1 message" + ); + assert!(matches!( + child.messages[0], + rvagent_core::messages::Message::Human(_) + )); + assert!(child.messages[0] + .content() .contains("Refactor the auth module")); - // Non-excluded keys must pass through - assert_eq!( - child.get("cwd").unwrap(), - &serde_json::json!("/home/user/project") - ); - assert_eq!( - child.get("project_config").unwrap(), - &serde_json::json!({"lang": "rust"}) - ); + // Non-excluded state (files) must pass through + assert!(child.files.contains_key("/home/user/project/main.rs")); - // Verify merge doesn't leak excluded keys back + // Verify merge doesn't leak excluded state back let mut parent_copy = parent_state_with_secrets(); - let parent_msgs_before = parent_copy.get("messages").cloned(); + let parent_msgs_before = parent_copy.message_count(); + let parent_todo_before = parent_copy.todos[0].content.clone(); let mut child_result = AgentState::new(); - child_result.insert( - "messages".into(), - serde_json::json!([ - {"type": "ai", "content": "Refactoring complete."} - ]), - ); - child_result.insert( - "todos".into(), - serde_json::json!([ - {"id": "child-1", "content": "leaked todo"} - ]), - ); - child_result.insert("new_discovery".into(), serde_json::json!("found a bug")); + child_result.push_message(Message::ai("Refactoring complete.")); + child_result.push_todo(TodoItem { + content: "leaked todo".into(), + status: TodoStatus::Pending, + active_form: String::new(), + }); + child_result.set_file("/new_discovery.md", file_data("found a bug")); merge_subagent_state(&mut parent_copy, &child_result); // Parent messages must NOT be overwritten by child - assert_eq!(parent_copy.get("messages"), parent_msgs_before.as_ref()); + assert_eq!(parent_copy.message_count(), parent_msgs_before); // Child's todos must NOT leak to parent - let parent_todos = parent_copy.get("todos").unwrap(); - assert!( - parent_todos.as_array().unwrap()[0]["content"] - .as_str() - .unwrap() - .contains("Fix bug"), + assert_eq!(parent_copy.todos.len(), 1); + assert_eq!( + parent_copy.todos[0].content, parent_todo_before, "Parent todos must not be overwritten by child" ); - // New non-excluded keys should merge - assert_eq!( - parent_copy.get("new_discovery"), - Some(&serde_json::json!("found a bug")) - ); + // New non-excluded state should merge + assert!(parent_copy.files.contains_key("/new_discovery.md")); } // =========================================================================== diff --git a/crates/rvAgent/rvagent-subagents/tests/security_validation.rs b/crates/rvAgent/rvagent-subagents/tests/security_validation.rs index 97d4ebdd2..fbb98f99e 100644 --- a/crates/rvAgent/rvagent-subagents/tests/security_validation.rs +++ b/crates/rvAgent/rvagent-subagents/tests/security_validation.rs @@ -6,7 +6,6 @@ use rvagent_subagents::{ spawn_parallel, AgentState, CompiledSubAgent, SpawnError, SubAgentOrchestrator, SubAgentSpec, ValidationConfig, ValidationError, }; -use std::collections::HashMap; fn create_test_orchestrator() -> SubAgentOrchestrator { let spec = SubAgentSpec::new("test-agent", "Do the thing"); @@ -33,7 +32,7 @@ fn create_test_orchestrator_with_config(config: ValidationConfig) -> SubAgentOrc } fn create_empty_state() -> AgentState { - HashMap::new() + AgentState::new() } #[test] diff --git a/crates/rvAgent/rvagent-tools/Cargo.toml b/crates/rvAgent/rvagent-tools/Cargo.toml index 8e9acb67c..4becdf4c1 100644 --- a/crates/rvAgent/rvagent-tools/Cargo.toml +++ b/crates/rvAgent/rvagent-tools/Cargo.toml @@ -20,6 +20,11 @@ async-trait = "0.1" glob = "0.3" walkdir = "2.5" +# Killing a timed-out command's whole process group needs kill(2) on a +# negative pid, which std does not expose. +[target.'cfg(unix)'.dependencies] +libc = "0.2" + [dev-dependencies] criterion = { workspace = true } tokio = { workspace = true, features = ["test-util"] } diff --git a/crates/rvAgent/rvagent-tools/src/edit_diag.rs b/crates/rvAgent/rvagent-tools/src/edit_diag.rs new file mode 100644 index 000000000..9accdd0d2 --- /dev/null +++ b/crates/rvAgent/rvagent-tools/src/edit_diag.rs @@ -0,0 +1,195 @@ +//! Diagnostics for failed string edits (ADR-273 §3.1, §3.3). +//! +//! "Error: old_string not found" is the single highest-frequency tool failure +//! in an editing agent, and it is nearly useless on its own: the model already +//! believed the string was there, so restating that it isn't gives it nothing +//! to change. It then retries a near-identical call, which is the input +//! condition for loop detection. +//! +//! Edit-tool ergonomics are load-bearing rather than incidental — in the +//! published reproductions, only the flavor with real failure diagnostics moved +//! the benchmark number; a plain `edit`/`write_file` pair gave no improvement +//! at all. So this module works out *why* a match failed and says so. + +/// Maximum characters of a candidate line echoed back in a diagnostic. +const SNIPPET_LEN: usize = 160; + +/// Explain why `old_string` did not match anything in `content`. +/// +/// Returns an actionable message naming the likely cause. Ordered by how +/// common the cause is in practice, so the first plausible explanation is the +/// most likely one. +pub fn diagnose_edit_failure(content: &str, old_string: &str, path: &str) -> String { + let base = format!("Error: old_string not found in {path}."); + + // 1. Line endings. Invisible, and it defeats an otherwise exact match. + if content.contains("\r\n") && !old_string.contains("\r\n") { + let normalized = old_string.replace('\n', "\r\n"); + if content.contains(&normalized) { + return format!( + "{base} The file uses CRLF line endings but old_string uses LF. \ + The text is present — re-read the file and copy the exact bytes." + ); + } + } + + // 2. Whitespace. Indentation drift is the classic cause: the model + // reconstructs the line from memory and gets the leading spaces wrong. + let squeeze = |s: &str| -> String { s.split_whitespace().collect::>().join(" ") }; + let squeezed_old = squeeze(old_string); + if !squeezed_old.is_empty() && squeeze(content).contains(&squeezed_old) { + return format!( + "{base} A match exists when whitespace is ignored, so the difference is \ + indentation, trailing spaces, or tabs-vs-spaces. Re-read the file and \ + copy the exact leading whitespace." + ); + } + + // 3. Case. + if content.to_lowercase().contains(&old_string.to_lowercase()) { + return format!( + "{base} A match exists ignoring case — the difference is capitalization only." + ); + } + + // 4. Partial match: locate the anchor line and show what is actually there. + // This is the most useful case, because it hands the model the real text. + if let Some(hint) = nearest_line_hint(content, old_string) { + return format!("{base} {hint}"); + } + + format!( + "{base} No similar text was found. The file may not contain this code at all — \ + read the file before editing it, rather than assuming its contents." + ) +} + +/// Find the line in `content` most similar to the first line of `old_string`, +/// and describe the mismatch. +fn nearest_line_hint(content: &str, old_string: &str) -> Option { + let needle = old_string.lines().next()?.trim(); + if needle.len() < 4 { + // Too short to attribute a near-match to anything meaningful. + return None; + } + + let mut best: Option<(usize, usize, &str)> = None; // (score, line_no, text) + for (i, line) in content.lines().enumerate() { + let score = shared_prefix_len(line.trim(), needle); + if score >= 4 && best.map(|(b, _, _)| score > b).unwrap_or(true) { + best = Some((score, i + 1, line)); + } + } + + let (_, line_no, text) = best?; + Some(format!( + "The closest line in the file is line {line_no}: {:?}. \ + Copy it exactly, including whitespace.", + truncate(text.trim_end(), SNIPPET_LEN) + )) +} + +/// Length of the common prefix of two strings, in characters. +fn shared_prefix_len(a: &str, b: &str) -> usize { + a.chars().zip(b.chars()).take_while(|(x, y)| x == y).count() +} + +/// Truncate on a character boundary, marking the cut. +fn truncate(s: &str, max: usize) -> String { + if s.chars().count() <= max { + return s.to_string(); + } + let cut: String = s.chars().take(max).collect(); + format!("{cut}…") +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn detects_indentation_mismatch() { + let content = "fn main() {\n let x = 1;\n}\n"; + // Wrong indentation, not missing: matching is substring-based, so an + // omitted indent still matches and never reaches this diagnostic. + let msg = diagnose_edit_failure(content, " let x = 1;", "a.rs"); + assert!(msg.contains("whitespace is ignored"), "got: {msg}"); + assert!(msg.contains("exact leading whitespace")); + } + + #[test] + fn detects_tabs_versus_spaces() { + let content = "fn main() {\n\tlet x = 1;\n}\n"; + let msg = diagnose_edit_failure(content, " let x = 1;", "a.rs"); + assert!(msg.contains("tabs-vs-spaces"), "got: {msg}"); + } + + #[test] + fn detects_crlf_mismatch() { + let content = "line one\r\nline two\r\n"; + let msg = diagnose_edit_failure(content, "line one\nline two", "a.txt"); + assert!(msg.contains("CRLF"), "got: {msg}"); + assert!(msg.contains("The text is present")); + } + + #[test] + fn detects_case_mismatch() { + let content = "let Value = 1;\n"; + let msg = diagnose_edit_failure(content, "let value = 1;", "a.rs"); + assert!(msg.contains("capitalization"), "got: {msg}"); + } + + #[test] + fn shows_the_nearest_line_when_content_drifted() { + let content = "fn compute(a: u32, b: u32) -> u32 {\n a + b\n}\n"; + // Same opening but a different signature — the most common real case. + let msg = diagnose_edit_failure(content, "fn compute(a: u32) -> u32 {", "a.rs"); + assert!( + msg.contains("closest line in the file is line 1"), + "got: {msg}" + ); + assert!(msg.contains("fn compute(a: u32, b: u32)"), "got: {msg}"); + } + + #[test] + fn says_so_plainly_when_nothing_is_close() { + let content = "completely unrelated file contents\n"; + let msg = diagnose_edit_failure(content, "fn transmogrify() {", "a.rs"); + assert!(msg.contains("No similar text was found"), "got: {msg}"); + assert!(msg.contains("read the file before editing")); + } + + #[test] + fn always_names_the_path() { + let msg = diagnose_edit_failure("x", "y", "src/lib.rs"); + assert!(msg.contains("src/lib.rs")); + } + + #[test] + fn handles_multibyte_content_without_panicking() { + let content = "let s = \"héllo 🙂 wörld\";\n"; + let msg = diagnose_edit_failure(content, "let s = \"hello world\";", "a.rs"); + assert!(!msg.is_empty()); + } + + #[test] + fn does_not_guess_from_a_trivially_short_needle() { + let content = "aaaa bbbb\ncccc dddd\n"; + // A 2-char needle would "nearly match" almost any line; it must not + // produce a confident and wrong nearest-line claim. + let msg = diagnose_edit_failure(content, "xy", "a.txt"); + assert!(msg.contains("No similar text was found"), "got: {msg}"); + } + + #[test] + fn truncates_a_very_long_candidate_line() { + let long = "x".repeat(500); + let content = format!("prefix_{long}\n"); + let msg = diagnose_edit_failure(&content, &format!("prefix_{}", "y".repeat(20)), "a.txt"); + assert!( + msg.len() < 400, + "diagnostic should stay compact: {}", + msg.len() + ); + } +} diff --git a/crates/rvAgent/rvagent-tools/src/lib.rs b/crates/rvAgent/rvagent-tools/src/lib.rs index 0952417d8..8d53ba5a3 100644 --- a/crates/rvAgent/rvagent-tools/src/lib.rs +++ b/crates/rvAgent/rvagent-tools/src/lib.rs @@ -3,10 +3,12 @@ //! Provides the `Tool` trait, `BuiltinTool`/`AnyTool` enum dispatch, //! `ToolRuntime` context, and parallel execution (ADR-103 A2). +pub mod edit_diag; pub mod edit_file; pub mod execute; pub mod glob; pub mod grep; +pub mod local; pub mod ls; pub mod read_file; pub mod task; @@ -19,10 +21,12 @@ use std::collections::HashMap; use std::fmt; use std::sync::Arc; +pub use edit_diag::diagnose_edit_failure; pub use edit_file::EditFileTool; pub use execute::ExecuteTool; pub use glob::GlobTool; pub use grep::GrepTool; +pub use local::LocalFsBackend; pub use ls::LsTool; pub use read_file::ReadFileTool; pub use task::TaskTool; @@ -475,16 +479,35 @@ pub fn format_content_with_line_numbers(content: &str, start_line: usize) -> Str if i > 0 { out.push('\n'); } - let truncated = &line[..line.len().min(MAX_LINE_LEN)]; use std::fmt::Write; - write!( - out, - "{:>width$}\t{}", - start_line + i, - truncated, - width = LINE_NUMBER_WIDTH - ) - .unwrap(); + if line.len() <= MAX_LINE_LEN { + write!( + out, + "{:>width$}\t{}", + start_line + i, + line, + width = LINE_NUMBER_WIDTH + ) + .unwrap(); + } else { + // Walk back to a character boundary: slicing at a fixed byte offset + // panics when a multi-byte character straddles it. + let mut end = MAX_LINE_LEN; + while end > 0 && !line.is_char_boundary(end) { + end -= 1; + } + // Mark the cut. Silent truncation leaves the model believing it has + // seen the whole line, which is worse than showing less. + write!( + out, + "{:>width$}\t{}… [line truncated, {} more bytes]", + start_line + i, + &line[..end], + line.len() - end, + width = LINE_NUMBER_WIDTH + ) + .unwrap(); + } } out } @@ -549,7 +572,7 @@ pub(crate) mod tests_common { if offset >= lines.len() { return Ok(String::new()); } - let end = (offset + limit).min(lines.len()); + let end = offset.saturating_add(limit).min(lines.len()); Ok(lines[offset..end].join("\n")) } None => Err(format!("File not found: {}", path)), @@ -944,9 +967,31 @@ mod tests { let result = format_content_with_line_numbers(&long_line, 1); let lines: Vec<&str> = result.lines().collect(); assert_eq!(lines.len(), 1); - // Extract the content after the line number and tab let content = lines[0].split('\t').nth(1).unwrap(); - assert_eq!(content.len(), MAX_LINE_LEN); + // The cut must be visible: silent truncation leaves the model believing + // it saw the whole line. + assert!(content.starts_with(&"a".repeat(MAX_LINE_LEN))); + assert!(content.contains("[line truncated, 100 more bytes]")); + } + + #[test] + fn test_format_line_truncation_is_char_boundary_safe() { + // A multi-byte character straddling the cut point used to panic on a + // raw byte slice. + let mut line = "a".repeat(MAX_LINE_LEN - 1); + line.push('é'); // 2 bytes, spanning MAX_LINE_LEN + line.push_str(&"b".repeat(50)); + let result = format_content_with_line_numbers(&line, 1); + assert!(result.contains("[line truncated")); + } + + #[test] + fn test_format_multibyte_line_does_not_panic() { + for pad in 0..4 { + let mut line = "a".repeat(MAX_LINE_LEN - pad); + line.push_str(&"🙂".repeat(10)); + let _ = format_content_with_line_numbers(&line, 1); + } } #[test] diff --git a/crates/rvAgent/rvagent-tools/src/local.rs b/crates/rvAgent/rvagent-tools/src/local.rs new file mode 100644 index 000000000..1b4db2b44 --- /dev/null +++ b/crates/rvAgent/rvagent-tools/src/local.rs @@ -0,0 +1,1006 @@ +//! Local filesystem backend — the real `Backend` used by the CLI and by the +//! end-to-end harness tests. +//! +//! # Path confinement +//! +//! Every path a tool supplies is model-controlled input, so it is resolved +//! against a fixed root and rejected if it escapes. Without this, +//! `read_file {"path": "/etc/passwd"}` — or a `write_file` anywhere on disk — +//! is a single tool call away. +//! +//! What confinement guarantees: +//! +//! - **Lexical escapes** (`..`, absolute paths) are normalized and then +//! rejected by a containment check against the canonical root. +//! - **Symlinks** cannot bridge out. The deepest existing ancestor is +//! canonicalized before the containment check, a dangling link is refused +//! outright rather than treated as a not-yet-existing path, and writes +//! refuse to travel through a link at all — including one pointing back +//! inside the root, which would otherwise pass containment. +//! - **Hard links** cannot bridge out on unix: a write to a file with +//! `nlink > 1` is refused, because the same inode may have another name +//! outside the root and nothing in the path tells us it does not. +//! +//! What it does not guarantee: +//! +//! - **TOCTOU races.** Resolution and the write are separate syscalls, so a +//! concurrent process that swaps a path component between them can still +//! redirect a write. Defending that needs `openat`-based traversal with +//! `O_NOFOLLOW`, which is out of scope here; the threat model is a +//! misbehaving model, not a local attacker racing the agent. +//! - **Non-unix hard links.** The `nlink` check is unix-only. +//! - **Anything below the root.** Confinement is a boundary, not a +//! permission system: every file under the root is writable. + +use std::path::{Component, Path, PathBuf}; + +use crate::{Backend, ExecuteResponse, FileInfo, GrepMatch, WriteResult}; + +/// A filesystem backend rooted at, and confined to, a working directory. +pub struct LocalFsBackend { + /// Canonical root. All resolved paths must live under this. + root: PathBuf, +} + +impl LocalFsBackend { + /// Create a backend confined to `root`. + /// + /// The root is canonicalized so that containment checks compare real paths. + /// If it cannot be canonicalized (e.g. it does not exist yet) the path is + /// used as given — resolution still applies, it is simply not symlink-proof + /// above the root itself. + pub fn new(root: impl Into) -> Self { + let root = root.into(); + let root = std::fs::canonicalize(&root).unwrap_or(root); + Self { root } + } + + /// The confinement root. + pub fn root(&self) -> &Path { + &self.root + } + + /// Resolve a tool-supplied path to a real path inside the root. + /// + /// Relative paths are joined to the root; absolute paths must already be + /// inside it. `..` is normalized lexically first so it cannot be used to + /// climb out, then the deepest existing ancestor is canonicalized to defeat + /// symlink escapes. + fn resolve(&self, path: &str) -> Result { + let normalized = self.literal_path(path); + let resolved = canonicalize_existing_prefix(&normalized) + .map_err(|e| format!("Error: path '{path}' is not resolvable: {e}"))?; + + if resolved.starts_with(&self.root) { + Ok(resolved) + } else { + Err(format!( + "Error: path '{path}' resolves outside the workspace root" + )) + } + } + + /// The lexically normalized path, before any symlink resolution. + /// + /// This is what the caller literally named. `resolve` canonicalizes it for + /// the containment check, which erases the fact that a component *was* a + /// symlink — so checks that care about that must run against this. + fn literal_path(&self, path: &str) -> PathBuf { + let raw = Path::new(path); + let joined = if raw.is_absolute() { + raw.to_path_buf() + } else if path.is_empty() || path == "." { + self.root.clone() + } else { + self.root.join(raw) + }; + lexical_normalize(&joined) + } + + /// Resolve for write-style operations, returning the error as a + /// `WriteResult` rather than a bare string. + /// + /// Beyond containment, a write must never travel *through* a link. + /// + /// `std::fs::write` follows a symlink, so a link inside the root is a + /// write primitive for wherever it points — and `resolve` cannot catch + /// it, because canonicalizing a link to an in-root file yields an in-root + /// path. The link is checked pre-resolution instead. + /// + /// A hard link is invisible to both checks: there is no link to follow and + /// `symlink_metadata` reports an ordinary file. The only signal is the + /// inode's link count, so a file with more than one name is refused — + /// another of those names may well be outside the root. + fn resolve_for_write(&self, path: &str) -> Result { + let fail = |e: String| WriteResult { + error: Some(e), + ..Default::default() + }; + let resolved = self.resolve(path).map_err(fail)?; + let literal = self.literal_path(path); + if let Ok(meta) = std::fs::symlink_metadata(&literal) { + if meta.file_type().is_symlink() { + return Err(fail(format!( + "Error: refusing to write through symlink '{}'. \ + Write to the link's target directly.", + literal.display() + ))); + } + } + #[cfg(unix)] + { + use std::os::unix::fs::MetadataExt; + if let Ok(meta) = std::fs::metadata(&resolved) { + // Directories always have several names (`.`, `..`, each + // child), so the count only means anything for a file. + if meta.is_file() && meta.nlink() > 1 { + return Err(fail(format!( + "Error: refusing to write through a hard link (nlink > 1): '{}'. \ + The file has another name, possibly outside the workspace, \ + which this write would also change.", + resolved.display() + ))); + } + } + } + Ok(resolved) + } +} + +/// Remove `.` and resolve `..` lexically, without touching the filesystem. +/// +/// Purely lexical so it works for paths that do not exist yet (`write_file` to +/// a new file). `..` at or above the root simply cannot climb past the prefix. +fn lexical_normalize(path: &Path) -> PathBuf { + let mut out = PathBuf::new(); + for comp in path.components() { + match comp { + Component::Prefix(p) => out.push(p.as_os_str()), + Component::RootDir => out.push(Component::RootDir.as_os_str()), + Component::CurDir => {} + Component::ParentDir => { + // Never pop past the root component itself. + if out.parent().is_some() { + out.pop(); + } + } + Component::Normal(seg) => out.push(seg), + } + } + out +} + +/// Canonicalize the deepest existing ancestor of `path`, re-appending the +/// non-existent tail. +/// +/// This is what makes the containment check symlink-aware: if any existing +/// component is a symlink out of the root, canonicalization exposes it before +/// the `starts_with` test. `path` must already be lexically normalized, so +/// re-appending the tail cannot reintroduce `..`. +/// +/// A *dangling* symlink is rejected outright. It cannot be canonicalized, so +/// treating it as part of the non-existent tail would leave it looking like a +/// fresh path under the canonical root — while an actual write through it +/// lands wherever the link points, anywhere on disk. +fn canonicalize_existing_prefix(path: &Path) -> Result { + let mut existing = path; + let mut tail: Vec<&std::ffi::OsStr> = Vec::new(); + + loop { + if let Ok(meta) = std::fs::symlink_metadata(existing) { + if meta.file_type().is_symlink() && std::fs::canonicalize(existing).is_err() { + return Err(format!( + "'{}' is a broken symlink; refusing to resolve through it", + existing.display() + )); + } + } + if let Ok(canonical) = std::fs::canonicalize(existing) { + let mut out = canonical; + for seg in tail.iter().rev() { + out.push(seg); + } + return Ok(out); + } + match (existing.file_name(), existing.parent()) { + (Some(name), Some(parent)) => { + tail.push(name); + existing = parent; + } + // Nothing along the chain exists; fall back to the lexical path. + _ => return Ok(path.to_path_buf()), + } + } +} + +impl Backend for LocalFsBackend { + fn ls_info(&self, path: &str) -> Result, String> { + let target = self.resolve(path)?; + let entries = std::fs::read_dir(&target) + .map_err(|e| format!("ls failed on '{}': {}", target.display(), e))?; + let mut infos = Vec::new(); + for entry in entries { + let entry = entry.map_err(|e| format!("read_dir entry error: {}", e))?; + let meta = entry + .metadata() + .map_err(|e| format!("metadata error: {}", e))?; + let file_type = if meta.is_dir() { + "directory" + } else if meta.is_symlink() { + "symlink" + } else { + "file" + }; + infos.push(FileInfo { + name: entry.file_name().to_string_lossy().into_owned(), + file_type: file_type.to_string(), + permissions: String::new(), + size: meta.len(), + }); + } + infos.sort_by(|a, b| a.name.cmp(&b.name)); + Ok(infos) + } + + fn read(&self, path: &str, offset: usize, limit: usize) -> Result { + let target = self.resolve(path)?; + let content = std::fs::read_to_string(&target) + .map_err(|e| format!("read '{}': {}", target.display(), e))?; + let lines: Vec<&str> = content.lines().collect(); + if offset >= lines.len() { + return Ok(String::new()); + } + // `limit` is model-supplied and arrives as an unbounded integer, so a + // plain `offset + limit` overflows and panics on values like u64::MAX. + // Saturating means "the rest of the file", which is what such a limit + // asks for anyway. + let end = offset.saturating_add(limit).min(lines.len()); + Ok(lines[offset..end].join("\n")) + } + + fn write(&self, path: &str, content: &str) -> WriteResult { + let target = match self.resolve_for_write(path) { + Ok(t) => t, + Err(e) => return e, + }; + if target.exists() { + return WriteResult { + error: Some(format!( + "Error: file {} already exists. Use force flag to overwrite.", + target.display() + )), + ..Default::default() + }; + } + if let Some(parent) = target.parent() { + if let Err(e) = std::fs::create_dir_all(parent) { + return WriteResult { + error: Some(format!("mkdir failed: {}", e)), + ..Default::default() + }; + } + } + if let Err(e) = std::fs::write(&target, content) { + return WriteResult { + error: Some(format!("write '{}': {}", target.display(), e)), + ..Default::default() + }; + } + match verify_written(&target, content) { + Ok(()) => WriteResult::default(), + Err(e) => WriteResult { + error: Some(e), + ..Default::default() + }, + } + } + + fn edit( + &self, + path: &str, + old_string: &str, + new_string: &str, + replace_all: bool, + ) -> WriteResult { + let target = match self.resolve_for_write(path) { + Ok(t) => t, + Err(e) => return e, + }; + let content = match std::fs::read_to_string(&target) { + Ok(c) => c, + Err(e) => { + return WriteResult { + error: Some(format!("read '{}': {}", target.display(), e)), + ..Default::default() + } + } + }; + let count = content.matches(old_string).count(); + if count == 0 { + // Say why it failed, not just that it did (ADR-273 §3.1/§3.3). + return WriteResult { + error: Some(crate::diagnose_edit_failure( + &content, + old_string, + &target.display().to_string(), + )), + ..Default::default() + }; + } + if count > 1 && !replace_all { + return WriteResult { + error: Some(format!( + "Error: old_string is not unique in {} ({} occurrences). Use replace_all=true.", + target.display(), + count + )), + ..Default::default() + }; + } + let new_content = if replace_all { + content.replace(old_string, new_string) + } else { + content.replacen(old_string, new_string, 1) + }; + if let Err(e) = std::fs::write(&target, &new_content) { + return WriteResult { + error: Some(format!("write '{}': {}", target.display(), e)), + ..Default::default() + }; + } + match verify_written(&target, &new_content) { + Ok(()) => WriteResult { + error: None, + occurrences: Some(if replace_all { count } else { 1 }), + ..Default::default() + }, + Err(e) => WriteResult { + error: Some(e), + ..Default::default() + }, + } + } + + fn glob_info(&self, pattern: &str, path: &str) -> Result, String> { + let base = self.resolve(path)?; + // Simple glob: walk the directory and match by name suffix. Handles the + // common `*.rs` / `**/*.toml` shapes without pulling in a glob crate. + let suffix = pattern + .trim_start_matches('*') + .trim_start_matches('/') + .trim_start_matches('*'); + let mut results = Vec::new(); + collect_glob_matches(&base, suffix, &mut results); + results.sort(); + Ok(results) + } + + fn grep_raw( + &self, + pattern: &str, + path: Option<&str>, + _include: Option<&str>, + ) -> Result, String> { + let search_dir = self.resolve(path.unwrap_or("."))?; + let mut matches = Vec::new(); + if search_dir.is_file() { + grep_file(&search_dir, pattern, &mut matches)?; + } else if search_dir.is_dir() { + grep_dir(&search_dir, pattern, &mut matches)?; + } + Ok(matches) + } + + fn execute(&self, command: &str, timeout_secs: u32) -> Result { + use std::process::{Command, Stdio}; + use std::time::Duration; + + // Security: environment sanitization — strip sensitive variables + // (SEC-005 / ADR-103 C2). The allowlist *is* the control: nothing + // outside it reaches the child, so no name-pattern denylist is needed. + const SAFE_ENV_VARS: &[&str] = &[ + "PATH", "HOME", "USER", "SHELL", "LANG", "LC_ALL", "LC_CTYPE", "TERM", "TMPDIR", "TZ", + ]; + + let mut cmd = Command::new("sh"); + cmd.arg("-c").arg(command).current_dir(&self.root); + cmd.env_clear(); + for var in SAFE_ENV_VARS { + if let Ok(val) = std::env::var(var) { + cmd.env(var, val); + } + } + cmd.stdin(Stdio::null()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); + + // Put the child in its own process group so a timeout can kill the + // whole tree. `child.kill()` signals only `sh`, leaving any grandchild + // running — and holding the pipe open — after we have given up on it. + #[cfg(unix)] + { + use std::os::unix::process::CommandExt; + cmd.process_group(0); + } + + let timeout = if timeout_secs == 0 { 30 } else { timeout_secs }; + let deadline = std::time::Instant::now() + Duration::from_secs(timeout as u64); + + let mut child = cmd.spawn().map_err(|e| format!("execute failed: {}", e))?; + + // Drain both pipes concurrently. A pipe holds only ~64 KB; if nobody + // reads while we poll, a chatty command blocks on write until the + // deadline and is then reported as a timeout it never actually hit. + // + // Each stream keeps one byte past the cap so that an oversized stream + // still trips the `> cap` test below and gets the truncation marker, + // rather than arriving as a silently shortened but cap-sized string. + let keep = MAX_OUTPUT_BYTES.saturating_add(1); + let out_pipe = child.stdout.take(); + let err_pipe = child.stderr.take(); + let out_reader = std::thread::spawn(move || match out_pipe { + Some(p) => drain_capped(p, keep), + None => Vec::new(), + }); + let err_reader = std::thread::spawn(move || match err_pipe { + Some(p) => drain_capped(p, keep), + None => Vec::new(), + }); + + let mut timed_out = false; + let status = loop { + match child + .try_wait() + .map_err(|e| format!("wait failed: {}", e))? + { + Some(status) => break Some(status), + None => { + if std::time::Instant::now() >= deadline { + kill_process_tree(&mut child); + // Reap, or the child lingers as a zombie for the life + // of the process. + let _ = child.wait(); + timed_out = true; + break None; + } + std::thread::sleep(Duration::from_millis(10)); + } + } + }; + + // The pipes are closed once the tree is gone, so these always finish. + let stdout_bytes = out_reader.join().unwrap_or_default(); + let stderr_bytes = err_reader.join().unwrap_or_default(); + + if timed_out { + return Ok(ExecuteResponse { + output: format!("Command timed out after {} seconds", timeout), + exit_code: -1, + }); + } + + let stdout = String::from_utf8_lossy(&stdout_bytes); + let stderr = String::from_utf8_lossy(&stderr_bytes); + let combined = if stderr.is_empty() { + stdout.into_owned() + } else { + format!("{}\n{}", stdout, stderr) + }; + + Ok(ExecuteResponse { + output: truncate_output(combined, MAX_OUTPUT_BYTES), + exit_code: status.and_then(|s| s.code()).unwrap_or(-1), + }) + } +} + +/// Security: cap on captured command output, to prevent memory exhaustion. +const MAX_OUTPUT_BYTES: usize = 1024 * 1024; + +/// Read a pipe to EOF, keeping at most `cap` bytes. +/// +/// Reading past the cap and discarding is deliberate: stopping early would +/// leave the writer blocked on a full pipe, which is the deadlock this exists +/// to avoid. +fn drain_capped(mut reader: R, cap: usize) -> Vec { + let mut kept = Vec::new(); + let mut buf = [0u8; 8192]; + loop { + match reader.read(&mut buf) { + Ok(0) => break, + Ok(n) => { + if kept.len() < cap { + let take = n.min(cap - kept.len()); + kept.extend_from_slice(&buf[..take]); + } + } + Err(ref e) if e.kind() == std::io::ErrorKind::Interrupted => {} + Err(_) => break, + } + } + kept +} + +/// Kill the child and everything it spawned. +#[cfg(unix)] +fn kill_process_tree(child: &mut std::process::Child) { + // The child leads its own group (see `process_group(0)`), so its pid is + // the pgid and a negative pid signals the whole group. + let pid = child.id() as i32; + unsafe { + libc::kill(-pid, libc::SIGKILL); + } + let _ = child.kill(); +} + +/// Kill the child. Without process groups only the direct child is reachable. +#[cfg(not(unix))] +fn kill_process_tree(child: &mut std::process::Child) { + let _ = child.kill(); +} + +/// Cap `s` at `cap` bytes, cutting on a character boundary. +/// +/// `String::truncate` panics when the cut lands inside a multi-byte character, +/// which is reachable whenever a command's output happens to straddle the cap. +fn truncate_output(mut s: String, cap: usize) -> String { + if s.len() <= cap { + return s; + } + let mut end = cap; + while end > 0 && !s.is_char_boundary(end) { + end -= 1; + } + s.truncate(end); + s.push_str(&format!("\n... [output truncated at {} bytes]", cap)); + s +} + +/// Confirm a write actually landed, by reading the file back (ADR-273 §3.1). +/// +/// `std::fs::write` returning `Ok` means the syscalls succeeded, not that the +/// bytes are on disk and readable: a full filesystem, a quota, a racing writer, +/// or an unusual mount can all produce a successful-looking write whose content +/// differs. Reporting success in that case is the worst outcome, because the +/// agent proceeds believing the edit is applied and every later step is built +/// on a false premise — the exact failure mode that dominates harness ablations. +/// +/// Costs one read per write, which is negligible against a model round trip. +fn verify_written(target: &Path, expected: &str) -> Result<(), String> { + match std::fs::read(target) { + Ok(actual) if actual == expected.as_bytes() => Ok(()), + Ok(actual) => Err(format!( + "Error: write to '{}' did not verify — expected {} bytes, file now holds {}. \ + The file may have been modified concurrently or the write was truncated. \ + Re-read the file before making further changes.", + target.display(), + expected.len(), + actual.len() + )), + Err(e) => Err(format!( + "Error: write to '{}' could not be verified: {e}. \ + Treat the file's contents as unknown and re-read it.", + target.display() + )), + } +} + +/// Recursively collect files matching a name suffix (simple glob substitute). +/// +/// Symlinks are skipped rather than followed. `Path::is_dir` follows them, and +/// a single `ln -s . loop` inside the tree then recurses until the stack +/// overflows — an unauthenticated crash from an ordinary directory listing. +fn collect_glob_matches(dir: &Path, suffix: &str, results: &mut Vec) { + let entries = match std::fs::read_dir(dir) { + Ok(e) => e, + Err(_) => return, + }; + for entry in entries.flatten() { + // `DirEntry::file_type` does not follow symlinks, so a link is neither + // a file nor a directory here and falls through untouched. + let file_type = match entry.file_type() { + Ok(t) => t, + Err(_) => continue, + }; + let path = entry.path(); + let name = path + .file_name() + .map(|n| n.to_string_lossy().to_string()) + .unwrap_or_default(); + if file_type.is_file() && name.ends_with(suffix) { + results.push(path.to_string_lossy().into_owned()); + } else if file_type.is_dir() && !name.starts_with('.') { + collect_glob_matches(&path, suffix, results); + } + } +} + +/// Grep a single file for a pattern. +fn grep_file(path: &Path, pattern: &str, matches: &mut Vec) -> Result<(), String> { + let content = match std::fs::read_to_string(path) { + Ok(c) => c, + Err(_) => return Ok(()), // skip binary / unreadable files + }; + for (i, line) in content.lines().enumerate() { + if line.contains(pattern) { + matches.push(GrepMatch { + file: path.to_string_lossy().into_owned(), + line_number: i + 1, + text: line.to_string(), + }); + } + } + Ok(()) +} + +/// Recursively grep a directory, skipping hidden directories and symlinks. +/// +/// Symlinks are skipped for the same reason as in `collect_glob_matches`: a +/// self-referencing link makes the descent unbounded. +fn grep_dir(dir: &Path, pattern: &str, matches: &mut Vec) -> Result<(), String> { + let entries = std::fs::read_dir(dir).map_err(|e| format!("read_dir: {}", e))?; + for entry in entries { + let entry = entry.map_err(|e| format!("entry: {}", e))?; + let file_type = match entry.file_type() { + Ok(t) => t, + Err(_) => continue, + }; + let path = entry.path(); + if file_type.is_file() { + grep_file(&path, pattern, matches)?; + } else if file_type.is_dir() { + let name = path + .file_name() + .map(|n| n.to_string_lossy().to_string()) + .unwrap_or_default(); + if !name.starts_with('.') { + grep_dir(&path, pattern, matches)?; + } + } + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn backend() -> (tempfile::TempDir, LocalFsBackend) { + let dir = tempfile::tempdir().unwrap(); + let backend = LocalFsBackend::new(dir.path()); + (dir, backend) + } + + #[test] + fn relative_paths_resolve_inside_root() { + let (dir, backend) = backend(); + std::fs::write(dir.path().join("a.txt"), "hello").unwrap(); + assert_eq!(backend.read("a.txt", 0, 10).unwrap(), "hello"); + } + + #[test] + fn absolute_path_outside_root_is_rejected() { + let (_dir, backend) = backend(); + let err = backend.read("/etc/passwd", 0, 10).unwrap_err(); + assert!(err.contains("outside the workspace root"), "got: {err}"); + } + + #[test] + fn parent_traversal_is_rejected() { + let (_dir, backend) = backend(); + let err = backend.read("../../../etc/passwd", 0, 10).unwrap_err(); + assert!(err.contains("outside the workspace root"), "got: {err}"); + } + + #[test] + fn write_outside_root_is_rejected() { + let (_dir, backend) = backend(); + let result = backend.write("/tmp/rvagent-escape-probe.txt", "pwned"); + assert!(result + .error + .as_deref() + .is_some_and(|e| e.contains("outside the workspace root"))); + assert!( + !Path::new("/tmp/rvagent-escape-probe.txt").exists(), + "escaping write must not touch the filesystem" + ); + } + + #[test] + fn edit_outside_root_is_rejected() { + let (_dir, backend) = backend(); + let result = backend.edit("/etc/hosts", "a", "b", false); + assert!(result + .error + .as_deref() + .is_some_and(|e| e.contains("outside the workspace root"))); + } + + #[cfg(unix)] + #[test] + fn symlink_escape_is_rejected() { + let (dir, backend) = backend(); + // A symlink inside the root pointing out of it must not be a bridge. + std::os::unix::fs::symlink("/etc", dir.path().join("escape")).unwrap(); + let err = backend.read("escape/passwd", 0, 10).unwrap_err(); + assert!(err.contains("outside the workspace root"), "got: {err}"); + } + + #[cfg(unix)] + #[test] + fn write_to_a_dangling_symlink_is_rejected() { + let (dir, backend) = backend(); + let outside = tempfile::tempdir().unwrap(); + let escaped = outside.path().join("escaped.txt"); + // The link target does not exist, so `canonicalize` fails and the path + // looks like a fresh file under the root — but `fs::write` would + // follow the link and create the file out there. + std::os::unix::fs::symlink(&escaped, dir.path().join("link")).unwrap(); + + let result = backend.write("link", "pwned"); + assert!( + result.error.is_some(), + "write through a broken link must fail" + ); + assert!( + !escaped.exists(), + "write must not have escaped to {}", + escaped.display() + ); + } + + #[cfg(unix)] + #[test] + fn write_through_a_dangling_symlink_directory_is_rejected() { + let (dir, backend) = backend(); + let outside = tempfile::tempdir().unwrap(); + let escaped_dir = outside.path().join("nope"); + std::os::unix::fs::symlink(&escaped_dir, dir.path().join("linkdir")).unwrap(); + + let result = backend.write("linkdir/f.txt", "pwned"); + assert!( + result.error.is_some(), + "write through a broken link must fail" + ); + assert!( + !escaped_dir.exists(), + "write must not have created the target" + ); + } + + #[cfg(unix)] + #[test] + fn write_through_a_symlink_to_an_inside_file_is_rejected() { + let (dir, backend) = backend(); + std::fs::write(dir.path().join("real.txt"), "original").unwrap(); + std::os::unix::fs::symlink(dir.path().join("real.txt"), dir.path().join("alias")).unwrap(); + + let result = backend.write("alias", "replaced"); + assert!( + result + .error + .as_deref() + .is_some_and(|e| e.contains("symlink")), + "got: {:?}", + result.error + ); + assert_eq!( + std::fs::read_to_string(dir.path().join("real.txt")).unwrap(), + "original" + ); + } + + #[cfg(unix)] + #[test] + fn write_or_edit_through_a_hard_link_is_rejected() { + let (dir, backend) = backend(); + let outside = tempfile::tempdir().unwrap(); + let outside_file = outside.path().join("secret.txt"); + std::fs::write(&outside_file, "original").unwrap(); + // A hard link has no target to canonicalize and looks like an ordinary + // file to `symlink_metadata`, so only the link count gives it away. + // Git cannot store one, but a prepared workspace can plant one. + std::fs::hard_link(&outside_file, dir.path().join("alias.txt")).unwrap(); + + let written = backend.write("alias.txt", "pwned"); + assert!( + written + .error + .as_deref() + .is_some_and(|e| e.contains("hard link")), + "got: {:?}", + written.error + ); + + let edited = backend.edit("alias.txt", "original", "pwned", false); + assert!( + edited + .error + .as_deref() + .is_some_and(|e| e.contains("hard link")), + "got: {:?}", + edited.error + ); + + assert_eq!( + std::fs::read_to_string(&outside_file).unwrap(), + "original", + "the file outside the root must be untouched" + ); + } + + #[cfg(unix)] + #[test] + fn an_ordinary_file_is_not_mistaken_for_a_hard_link() { + // The nlink check must not fire on the normal case, or every edit in + // the workspace breaks. + let (dir, backend) = backend(); + std::fs::write(dir.path().join("plain.txt"), "alpha").unwrap(); + let edited = backend.edit("plain.txt", "alpha", "beta", false); + assert!(edited.error.is_none(), "unexpected: {:?}", edited.error); + assert!(backend.write("fresh.txt", "new").error.is_none()); + } + + #[cfg(unix)] + #[test] + fn traversal_does_not_follow_a_self_referencing_symlink() { + let (dir, backend) = backend(); + std::fs::write(dir.path().join("a.rs"), "needle").unwrap(); + // Without a symlink check this recurses until the stack overflows. + std::os::unix::fs::symlink(dir.path(), dir.path().join("loop")).unwrap(); + + assert_eq!(backend.glob_info("*.rs", ".").unwrap().len(), 1); + assert_eq!( + backend.grep_raw("needle", Some("."), None).unwrap().len(), + 1 + ); + } + + #[test] + fn read_with_an_enormous_limit_does_not_overflow() { + let (dir, backend) = backend(); + std::fs::write(dir.path().join("big.txt"), "one\ntwo\nthree").unwrap(); + // A model can send any integer; `offset + limit` used to wrap and panic. + assert_eq!( + backend.read("big.txt", 0, u64::MAX as usize).unwrap(), + "one\ntwo\nthree" + ); + assert_eq!( + backend.read("big.txt", 1, usize::MAX).unwrap(), + "two\nthree" + ); + // Past EOF is an empty read, not a panic. + assert_eq!(backend.read("big.txt", 99, usize::MAX).unwrap(), ""); + } + + #[test] + fn output_truncation_cuts_on_a_char_boundary() { + // 'é' is two bytes, so a cap of 5 lands inside the third one — the + // case where `String::truncate` panics. + let out = truncate_output("ééé".to_string(), 5); + assert!(out.starts_with("éé"), "got: {out}"); + assert!(out.contains("[output truncated"), "got: {out}"); + let untouched = truncate_output("ééé".to_string(), 64); + assert_eq!(untouched, "ééé"); + } + + #[cfg(unix)] + #[test] + fn execute_captures_oversized_output_without_hanging() { + use std::time::{Duration, Instant}; + let (_dir, backend) = backend(); + let started = Instant::now(); + // 2 MB — far past a pipe buffer, so an undrained child would block + // until the deadline and be misreported as a timeout. + let result = backend + .execute("head -c 2000000 /dev/zero | tr '\\0' 'a'", 60) + .unwrap(); + assert_eq!(result.exit_code, 0, "got: {}", result.output); + assert!( + result.output.contains("[output truncated"), + "expected truncation, got {} bytes", + result.output.len() + ); + assert!( + started.elapsed() < Duration::from_secs(30), + "took {:?} — the pipe was not drained", + started.elapsed() + ); + } + + #[cfg(unix)] + #[test] + fn execute_timeout_kills_the_whole_process_tree() { + use std::time::{Duration, Instant}; + let (_dir, backend) = backend(); + let started = Instant::now(); + // `sh` waits on a grandchild; killing only `sh` would leave it running. + let result = backend.execute("sleep 30 & wait", 1).unwrap(); + assert_eq!(result.exit_code, -1); + assert!( + result.output.contains("timed out"), + "got: {}", + result.output + ); + assert!( + started.elapsed() < Duration::from_secs(15), + "timeout path hung for {:?}", + started.elapsed() + ); + } + + #[test] + fn write_then_read_roundtrip_inside_root() { + let (_dir, backend) = backend(); + let result = backend.write("nested/dir/new.txt", "content"); + assert!(result.error.is_none(), "unexpected: {:?}", result.error); + assert_eq!( + backend.read("nested/dir/new.txt", 0, 10).unwrap(), + "content" + ); + } + + #[test] + fn write_is_verified_by_reading_back() { + let (dir, backend) = backend(); + let result = backend.write("verified.txt", "exact contents"); + assert!(result.error.is_none()); + // The verification path must accept a correct write, not just reject + // bad ones — otherwise it would be a permanent false alarm. + assert_eq!( + std::fs::read_to_string(dir.path().join("verified.txt")).unwrap(), + "exact contents" + ); + } + + #[test] + fn edit_is_verified_by_reading_back() { + let (dir, backend) = backend(); + std::fs::write(dir.path().join("e.txt"), "alpha beta").unwrap(); + let result = backend.edit("e.txt", "alpha", "gamma", false); + assert!(result.error.is_none(), "unexpected: {:?}", result.error); + assert_eq!(result.occurrences, Some(1)); + assert_eq!( + std::fs::read_to_string(dir.path().join("e.txt")).unwrap(), + "gamma beta" + ); + } + + #[test] + fn verification_reports_a_content_mismatch() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("x.txt"); + std::fs::write(&path, "actual").unwrap(); + // Simulate the case the check exists for: what landed differs from + // what was asked for. + let err = verify_written(&path, "expected something longer").unwrap_err(); + assert!(err.contains("did not verify"), "got: {err}"); + assert!( + err.contains("Re-read the file"), + "must be actionable: {err}" + ); + } + + #[test] + fn verification_reports_an_unreadable_file() { + let dir = tempfile::tempdir().unwrap(); + let err = verify_written(&dir.path().join("missing.txt"), "anything").unwrap_err(); + assert!(err.contains("could not be verified"), "got: {err}"); + assert!(err.contains("re-read"), "must be actionable: {err}"); + } + + #[test] + fn verification_handles_empty_and_multibyte_content() { + let (_dir, backend) = backend(); + assert!(backend.write("empty.txt", "").error.is_none()); + assert!(backend.write("utf8.txt", "héllo 🙂 wörld").error.is_none()); + } + + #[test] + fn dotdot_inside_root_still_works() { + let (_dir, backend) = backend(); + assert!(backend.write("sub/file.txt", "x").error.is_none()); + // Climbs out of `sub` but stays under the root — legitimate. + assert_eq!(backend.read("sub/../sub/file.txt", 0, 10).unwrap(), "x"); + } +} diff --git a/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs b/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs new file mode 100644 index 000000000..f7f6752f2 --- /dev/null +++ b/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs @@ -0,0 +1,790 @@ +//! Phase 0 exit gate: end-to-end tool calling. +//! +//! These tests wire the *shipped* pieces together — the real builtin tool +//! registry, the real `LocalFsBackend`, and the real `AgentGraph` loop — against +//! a scripted model. Everything except the network call to the provider is +//! production code, so the gate fails if the loop, the schemas, or the tools +//! regress. +//! +//! What each P0 claim is verified by: +//! * P0.2 (schemas reach the model) — `schemas_reach_the_model` +//! * P0.4 (errors feed back, not abort) — `tool_error_feeds_back_and_loop_continues` +//! * P0.4 (parallel exec preserves order) — `parallel_tool_calls_preserve_order` +//! * P0.4 (usage accounting) — `usage_metadata_is_aggregated` +//! * real side effects on disk — `write_then_read_roundtrip_through_the_loop` +//! * confinement holds through the loop — `path_escape_is_refused_through_the_loop` + +use std::sync::{Arc, Mutex}; + +use async_trait::async_trait; +use rvagent_core::error::Result; +use rvagent_core::graph::{AgentGraph, GraphConfig, ToolExecutor}; +use rvagent_core::messages::{Message, ToolCall}; +use rvagent_core::models::{ChatModel, ToolDefinition}; +use rvagent_core::state::AgentState; +use rvagent_tools::Tool as _; + +// --------------------------------------------------------------------------- +// Test harness: the real tool executor, wired exactly as the CLI wires it +// --------------------------------------------------------------------------- + +/// Mirrors `CliToolExecutor`: real builtin tools over a real confined backend. +struct RealToolExecutor { + tools: Vec, + backend: rvagent_tools::BackendRef, +} + +impl RealToolExecutor { + fn new(root: &std::path::Path) -> Self { + Self { + tools: rvagent_tools::builtin_tools(), + backend: Arc::new(rvagent_tools::LocalFsBackend::new(root)), + } + } +} + +#[async_trait] +impl ToolExecutor for RealToolExecutor { + async fn execute(&self, call: &ToolCall, _state: &AgentState) -> Result { + let runtime = rvagent_tools::ToolRuntime::new(Arc::clone(&self.backend)); + match rvagent_tools::resolve_tool(&call.name, &self.tools) { + Some(tool) => Ok(tool.invoke(call.args.clone(), &runtime).to_string()), + None => Ok(format!("Error: tool '{}' not found", call.name)), + } + } + + fn definitions(&self) -> Vec { + self.tools + .iter() + .map(|t| ToolDefinition { + name: t.name().to_string(), + description: t.description().to_string(), + input_schema: t.parameters_schema(), + }) + .collect() + } +} + +/// A scripted model that records what the loop actually sent it. +struct ScriptedModel { + responses: Mutex>, + /// Tool schemas observed on each `complete` call. + seen_tools: Mutex>>, + /// Message history observed on the most recent `complete` call. + last_messages: Mutex>, +} + +impl ScriptedModel { + fn new(responses: Vec) -> Self { + Self { + responses: Mutex::new(responses), + seen_tools: Mutex::new(Vec::new()), + last_messages: Mutex::new(Vec::new()), + } + } +} + +#[async_trait] +impl ChatModel for ScriptedModel { + async fn complete(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result { + self.seen_tools.lock().unwrap().push(tools.to_vec()); + *self.last_messages.lock().unwrap() = messages.to_vec(); + let mut resps = self.responses.lock().unwrap(); + if resps.is_empty() { + Ok(Message::ai("done")) + } else { + Ok(resps.remove(0)) + } + } + + async fn stream(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result> { + Ok(vec![self.complete(messages, tools).await?]) + } +} + +fn call(id: &str, name: &str, args: serde_json::Value) -> ToolCall { + ToolCall { + id: id.into(), + name: name.into(), + args, + } +} + +/// Collect tool-result message contents, in order. +fn tool_results(state: &AgentState) -> Vec { + state + .messages + .iter() + .filter_map(|m| match m { + Message::Tool(t) => Some(t.content.clone()), + _ => None, + }) + .collect() +} + +// --------------------------------------------------------------------------- +// P0.2 — tool schemas reach the model +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn schemas_reach_the_model() { + let dir = tempfile::tempdir().unwrap(); + let model = ScriptedModel::new(vec![Message::ai("hi")]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + + let state = graph.run(AgentState::with_system_message("sys")).await; + assert!(state.is_ok(), "loop failed: {:?}", state.err()); + + // The graph owns the model, so re-derive the expectation from the registry: + // every builtin tool must have been advertised with a usable schema. + let executor = RealToolExecutor::new(dir.path()); + let defs = executor.definitions(); + assert_eq!(defs.len(), rvagent_tools::builtin_tools().len()); + assert!(defs.iter().any(|d| d.name == "read_file")); + assert!(defs.iter().any(|d| d.name == "write_file")); + for def in &defs { + assert!( + !def.description.is_empty(), + "{} has no description", + def.name + ); + assert_eq!( + def.input_schema.get("type").and_then(|v| v.as_str()), + Some("object"), + "{} schema is not a JSON-Schema object: {}", + def.name, + def.input_schema + ); + assert!( + def.input_schema.get("properties").is_some(), + "{} schema has no properties", + def.name + ); + } +} + +#[tokio::test] +async fn schemas_are_sent_on_every_turn_including_after_tools() { + let dir = tempfile::tempdir().unwrap(); + std::fs::write(dir.path().join("f.txt"), "content").unwrap(); + + let model = Arc::new(ScriptedModel::new(vec![ + Message::ai_with_tools( + "reading", + vec![call( + "t1", + "read_file", + serde_json::json!({"file_path": "f.txt"}), + )], + ), + Message::ai("read it"), + ])); + + let graph = AgentGraph::new( + SharedModel(Arc::clone(&model)), + RealToolExecutor::new(dir.path()), + ); + graph.run(AgentState::new()).await.unwrap(); + + let seen = model.seen_tools.lock().unwrap(); + assert_eq!(seen.len(), 2, "expected two model turns"); + for (turn, tools) in seen.iter().enumerate() { + assert!( + !tools.is_empty(), + "turn {turn} was sent no tool schemas — the model could not call a tool" + ); + } +} + +// --------------------------------------------------------------------------- +// Real side effects on disk +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn write_then_read_roundtrip_through_the_loop() { + let dir = tempfile::tempdir().unwrap(); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "writing", + vec![call( + "t1", + "write_file", + serde_json::json!({"file_path": "out.txt", "content": "hello from the agent"}), + )], + ), + Message::ai_with_tools( + "reading back", + vec![call( + "t2", + "read_file", + serde_json::json!({"file_path": "out.txt"}), + )], + ), + Message::ai("verified"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + let state = graph.run(AgentState::new()).await.unwrap(); + + // The file must actually exist on disk — this is the end-to-end claim. + let written = std::fs::read_to_string(dir.path().join("out.txt")) + .expect("agent's write_file did not produce a real file"); + assert_eq!(written, "hello from the agent"); + + let results = tool_results(&state); + assert_eq!(results.len(), 2, "expected one result per tool call"); + assert!( + results[1].contains("hello from the agent"), + "read_file did not return the written content: {}", + results[1] + ); +} + +// --------------------------------------------------------------------------- +// P0.4 — tool errors feed back as results instead of aborting +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn tool_error_feeds_back_and_loop_continues() { + let dir = tempfile::tempdir().unwrap(); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "reading a file that isn't there", + vec![call( + "t1", + "read_file", + serde_json::json!({"file_path": "does-not-exist.txt"}), + )], + ), + // The model gets to see the failure and recover. + Message::ai_with_tools( + "creating it instead", + vec![call( + "t2", + "write_file", + serde_json::json!({"file_path": "does-not-exist.txt", "content": "now it does"}), + )], + ), + Message::ai("recovered"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + + // The run must succeed: a failing tool is not a failing agent. + let state = graph + .run(AgentState::new()) + .await + .expect("a tool error must not abort the loop"); + + let results = tool_results(&state); + assert_eq!(results.len(), 2); + assert!( + results[0].to_lowercase().contains("error") + || results[0].to_lowercase().contains("no such file"), + "the failure was not reported back to the model: {}", + results[0] + ); + assert!( + dir.path().join("does-not-exist.txt").exists(), + "the recovery turn did not run" + ); +} + +#[tokio::test] +async fn unknown_tool_is_reported_not_fatal() { + let dir = tempfile::tempdir().unwrap(); + let model = ScriptedModel::new(vec![ + Message::ai_with_tools("", vec![call("t1", "no_such_tool", serde_json::json!({}))]), + Message::ai("ok"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + let state = graph.run(AgentState::new()).await.unwrap(); + + let results = tool_results(&state); + assert_eq!(results.len(), 1); + assert!(results[0].contains("not found"), "got: {}", results[0]); +} + +// --------------------------------------------------------------------------- +// P0.4 — parallel execution +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn parallel_tool_calls_preserve_order() { + let dir = tempfile::tempdir().unwrap(); + for i in 0..6 { + std::fs::write(dir.path().join(format!("f{i}.txt")), format!("body-{i}")).unwrap(); + } + + let calls: Vec = (0..6) + .map(|i| { + call( + &format!("t{i}"), + "read_file", + serde_json::json!({"file_path": format!("f{i}.txt")}), + ) + }) + .collect(); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools("reading all", calls), + Message::ai("done"), + ]); + let config = GraphConfig { + parallel_tools: true, + max_parallel_tools: 3, + ..GraphConfig::default() + }; + let graph = AgentGraph::with_config(model, RealToolExecutor::new(dir.path()), config); + let state = graph.run(AgentState::new()).await.unwrap(); + + let results = tool_results(&state); + assert_eq!(results.len(), 6); + // Results must come back in call order even though execution is concurrent + // and the concurrency limit is lower than the number of calls. + for (i, r) in results.iter().enumerate() { + assert!( + r.contains(&format!("body-{i}")), + "result {i} out of order or wrong: {r}" + ); + } +} + +#[tokio::test] +async fn parallel_and_sequential_agree() { + let dir = tempfile::tempdir().unwrap(); + for i in 0..4 { + std::fs::write(dir.path().join(format!("f{i}.txt")), format!("body-{i}")).unwrap(); + } + let calls: Vec = (0..4) + .map(|i| { + call( + &format!("t{i}"), + "read_file", + serde_json::json!({"file_path": format!("f{i}.txt")}), + ) + }) + .collect(); + + let mut outputs = Vec::new(); + for parallel in [true, false] { + let model = ScriptedModel::new(vec![ + Message::ai_with_tools("", calls.clone()), + Message::ai("done"), + ]); + let config = GraphConfig { + parallel_tools: parallel, + max_parallel_tools: 2, + ..GraphConfig::default() + }; + let graph = AgentGraph::with_config(model, RealToolExecutor::new(dir.path()), config); + let state = graph.run(AgentState::new()).await.unwrap(); + outputs.push(tool_results(&state)); + } + assert_eq!( + outputs[0], outputs[1], + "parallel and sequential execution disagree" + ); +} + +// --------------------------------------------------------------------------- +// P0.4 — usage accounting +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn usage_metadata_is_aggregated() { + let dir = tempfile::tempdir().unwrap(); + + // Two turns carrying provider usage metadata, as the backends attach it. + let mut first = + Message::ai_with_tools("", vec![call("t1", "ls", serde_json::json!({"path": "."}))]); + let mut second = Message::ai("done"); + for (msg, input, output) in [(&mut first, 100u64, 20u64), (&mut second, 150u64, 30u64)] { + if let Message::Ai(ai) = msg { + ai.metadata.insert( + "usage".into(), + serde_json::json!({"input_tokens": input, "output_tokens": output}), + ); + } + } + + let model = ScriptedModel::new(vec![first, second]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + let state = graph.run(AgentState::new()).await.unwrap(); + + // Usage metadata must survive on the messages so a caller can total it. + let totals: (u64, u64) = state + .messages + .iter() + .filter_map(|m| match m { + Message::Ai(ai) => ai.metadata.get("usage"), + _ => None, + }) + .fold((0, 0), |(i, o), usage| { + ( + i + usage + .get("input_tokens") + .and_then(|v| v.as_u64()) + .unwrap_or(0), + o + usage + .get("output_tokens") + .and_then(|v| v.as_u64()) + .unwrap_or(0), + ) + }); + assert_eq!(totals, (250, 50), "usage metadata was lost or miscounted"); +} + +// --------------------------------------------------------------------------- +// ADR-274 — observation masking reaches the model +// --------------------------------------------------------------------------- + +/// Shares one `ScriptedModel` so its recorded observations outlive the graph. +struct SharedModel(Arc); + +#[async_trait] +impl ChatModel for SharedModel { + async fn complete(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result { + self.0.complete(messages, tools).await + } + async fn stream(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result> { + self.0.stream(messages, tools).await + } +} + +#[tokio::test] +async fn old_observations_are_masked_before_reaching_the_model() { + let dir = tempfile::tempdir().unwrap(); + for i in 0..5 { + std::fs::write( + dir.path().join(format!("f{i}.txt")), + format!("UNIQUE-BODY-{i}"), + ) + .unwrap(); + } + + // Five sequential read turns, then a final answer. + let mut responses: Vec = (0..5) + .map(|i| { + Message::ai_with_tools( + "", + vec![call( + &format!("t{i}"), + "read_file", + serde_json::json!({ "file_path": format!("f{i}.txt") }), + )], + ) + }) + .collect(); + responses.push(Message::ai("done")); + + let model = Arc::new(ScriptedModel::new(responses)); + let config = GraphConfig { + parallel_tools: false, + mask: rvagent_core::masking::MaskConfig { + keep_last_observations: 2, + ..Default::default() + }, + ..GraphConfig::default() + }; + let graph = AgentGraph::with_config( + SharedModel(Arc::clone(&model)), + RealToolExecutor::new(dir.path()), + config, + ); + let state = graph.run(AgentState::new()).await.unwrap(); + + // What the model saw on the final turn: only the last 2 observations in + // full, the rest elided but still addressable. + let seen = model.last_messages.lock().unwrap().clone(); + let seen_tools: Vec<&Message> = seen + .iter() + .filter(|m| matches!(m, Message::Tool(_))) + .collect(); + assert_eq!(seen_tools.len(), 5, "every call still has a paired result"); + + for (i, msg) in seen_tools.iter().enumerate() { + let Message::Tool(t) = msg else { + unreachable!() + }; + if i < 3 { + assert!( + t.content.contains("output elided"), + "observation {i} should have been masked: {}", + t.content + ); + assert!( + t.content.contains(&format!("recall id t{i}")), + "masked observation {i} lost its recall handle: {}", + t.content + ); + assert!( + !t.content.contains(&format!("UNIQUE-BODY-{i}")), + "masked observation {i} still carried its full body" + ); + } else { + assert!( + t.content.contains(&format!("UNIQUE-BODY-{i}")), + "recent observation {i} must survive verbatim: {}", + t.content + ); + } + } + + // The stored log keeps everything — masking is a projection, not a + // mutation, which is what makes the elided content recoverable. + let stored = tool_results(&state); + assert_eq!(stored.len(), 5); + for (i, content) in stored.iter().enumerate() { + assert!( + content.contains(&format!("UNIQUE-BODY-{i}")), + "stored observation {i} was destroyed by masking: {content}" + ); + } +} + +#[tokio::test] +async fn recall_returns_the_full_content_of_a_masked_observation() { + let dir = tempfile::tempdir().unwrap(); + for i in 0..4 { + std::fs::write( + dir.path().join(format!("f{i}.txt")), + format!("SECRET-BODY-{i}"), + ) + .unwrap(); + } + + let mut responses: Vec = (0..4) + .map(|i| { + Message::ai_with_tools( + "", + vec![call( + &format!("t{i}"), + "read_file", + serde_json::json!({ "file_path": format!("f{i}.txt") }), + )], + ) + }) + .collect(); + // t0 has been masked out of the model's view by now; dereference it. + responses.push(Message::ai_with_tools( + "", + vec![call( + "r1", + "recall", + serde_json::json!({ "recall_id": "t0" }), + )], + )); + responses.push(Message::ai("done")); + + let model = Arc::new(ScriptedModel::new(responses)); + let config = GraphConfig { + parallel_tools: false, + mask: rvagent_core::masking::MaskConfig { + keep_last_observations: 2, + ..Default::default() + }, + ..GraphConfig::default() + }; + let graph = AgentGraph::with_config( + SharedModel(Arc::clone(&model)), + RealToolExecutor::new(dir.path()), + config, + ); + let state = graph.run(AgentState::new()).await.unwrap(); + + // The recall tool must have been advertised, or the model could not call it. + let seen = model.seen_tools.lock().unwrap(); + assert!( + seen.last().unwrap().iter().any(|d| d.name == "recall"), + "recall was not advertised while masking was active" + ); + drop(seen); + + // The recall result carries the content that was elided from the view. + let results = tool_results(&state); + let recalled = results.last().unwrap(); + assert!( + recalled.contains("SECRET-BODY-0"), + "recall did not return the elided content: {recalled}" + ); +} + +#[tokio::test] +async fn failed_edit_explains_why_it_failed() { + let dir = tempfile::tempdir().unwrap(); + std::fs::write( + dir.path().join("code.rs"), + "fn main() {\n let total = 1;\n}\n", + ) + .unwrap(); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "", + vec![call( + "t1", + "edit_file", + // Wrong indentation (8 spaces, file has 4). Omitting the + // indent entirely would still match as a substring; supplying + // the wrong amount is the failure that actually happens. + serde_json::json!({ + "file_path": "code.rs", + "old_string": " let total = 1;", + "new_string": " let total = 2;" + }), + )], + ), + Message::ai("done"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + let state = graph.run(AgentState::new()).await.unwrap(); + + let results = tool_results(&state); + assert_eq!(results.len(), 1); + // "not found" alone is useless — the model already believed it was there. + assert!( + results[0].contains("whitespace is ignored"), + "edit failure was not diagnosed: {}", + results[0] + ); + assert!(results[0].contains("exact leading whitespace")); +} + +#[tokio::test] +async fn recall_with_an_unknown_id_is_actionable_not_fatal() { + let dir = tempfile::tempdir().unwrap(); + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "", + vec![call( + "r1", + "recall", + serde_json::json!({ "recall_id": "nope" }), + )], + ), + Message::ai("done"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + let state = graph.run(AgentState::new()).await.unwrap(); + + let results = tool_results(&state); + assert_eq!(results.len(), 1); + assert!(results[0].contains("no observation found")); + // Must tell the model where valid ids come from, not just that it failed. + assert!(results[0].contains("placeholders")); +} + +#[tokio::test] +async fn oversized_tool_output_is_capped() { + let dir = tempfile::tempdir().unwrap(); + // Many lines, so the total far exceeds the cap. (A single very long line + // would not: read_file truncates individual lines at its own limit.) + let body: String = (0..20_000).map(|i| format!("line {i}\n")).collect(); + std::fs::write(dir.path().join("big.txt"), body).unwrap(); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "", + vec![call( + "t1", + "read_file", + serde_json::json!({"file_path": "big.txt", "limit": 100_000}), + )], + ), + Message::ai("done"), + ]); + let config = GraphConfig { + mask: rvagent_core::masking::MaskConfig { + max_tool_result_bytes: 4_000, + ..Default::default() + }, + ..GraphConfig::default() + }; + let graph = AgentGraph::with_config(model, RealToolExecutor::new(dir.path()), config); + let state = graph.run(AgentState::new()).await.unwrap(); + + let results = tool_results(&state); + assert_eq!(results.len(), 1); + assert!( + results[0].len() <= 4_000, + "tool output was not capped: {} bytes", + results[0].len() + ); + assert!( + results[0].ends_with("[output truncated]"), + "truncation must be explicit so the model knows output was cut; got {} bytes: {:?}", + results[0].len(), + &results[0][..results[0].len().min(200)] + ); +} + +// --------------------------------------------------------------------------- +// Confinement holds through the full loop +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn path_escape_is_refused_through_the_loop() { + let dir = tempfile::tempdir().unwrap(); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "exfiltrating", + vec![ + call( + "t1", + "read_file", + serde_json::json!({"file_path": "/etc/passwd"}), + ), + call( + "t2", + "read_file", + serde_json::json!({"file_path": "../../../../etc/passwd"}), + ), + ], + ), + Message::ai("done"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + let state = graph.run(AgentState::new()).await.unwrap(); + + for (i, result) in tool_results(&state).iter().enumerate() { + assert!( + !result.contains("root:"), + "escape {i} leaked /etc/passwd contents: {result}" + ); + // Must be refused *by the confinement check* specifically — a generic + // error (bad param, missing file) would pass vacuously and hide a + // regression in the boundary itself. + assert!( + result.contains("outside the workspace root"), + "escape {i} was not refused by path confinement: {result}" + ); + } +} + +#[tokio::test] +async fn write_escape_does_not_touch_the_filesystem() { + let dir = tempfile::tempdir().unwrap(); + let outside = dir.path().parent().unwrap().join("rvagent-e2e-escape.txt"); + let _ = std::fs::remove_file(&outside); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "", + vec![call( + "t1", + "write_file", + serde_json::json!({"file_path": outside.to_string_lossy(), "content": "pwned"}), + )], + ), + Message::ai("done"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + graph.run(AgentState::new()).await.unwrap(); + + assert!( + !outside.exists(), + "a write outside the workspace root reached the filesystem" + ); +} diff --git a/docs/adr/ADR-273-rvagent-harness-reliability-floor.md b/docs/adr/ADR-273-rvagent-harness-reliability-floor.md new file mode 100644 index 000000000..b82ae46d6 --- /dev/null +++ b/docs/adr/ADR-273-rvagent-harness-reliability-floor.md @@ -0,0 +1,182 @@ +--- +adr: 273 +title: "rvAgent Harness Reliability Floor" +status: accepted +date: 2026-08-01 +authors: [Reuven Cohen] +project: "rvAgent Harness" +related: [ADR-103, ADR-139, ADR-274, ADR-275, ADR-276, ADR-277] +tags: [rvagent, harness, reliability, agent-loop, tools, error-recovery, sota] +--- + +# ADR-273 — rvAgent Harness Reliability Floor + +## Status + +**Accepted.** Owner: Reuven Cohen. Project: rvAgent Harness. Date: 2026-08-01. + +Evidence base: `docs/research/rvagent-hermes-harness/04-sota-landscape.md`. + +## 1. Decision + +**Sequence harness work by measured reliability impact, not by architectural +ambition.** A defined "reliability floor" of seven mechanisms ships before +event-streaming, cache-tiering, evolution, or any other Phase 1+ item. + +This ADR reorders `03-roadmap.md` Phase 1. It does not remove any item from it. + +## 2. Context + +The cleanest harness ablation available (Claw-SWE-Bench, Jun 2026) rebuilds a +harness from bare model-emits-diff to full scaffolding: **19.1% → 73.4%, ++54.3 points**. Nearly all of that delta is patch-apply failures falling from +69.1% to under 1.5%. + +That result generalizes across the technique literature: **the dominant wins +come from eliminating mechanical failure modes, not from better reasoning.** +Patches that don't apply, tool-call loops, context rot, flaky tests. + +Two structural qualifiers: + +- **Harness value scales inversely with model strength.** Same harness set: + 12.5-point spread on GLM-5.1, 27.4-point spread on Qwen-3.6-flash. Targeting + frontier models only should be expected to yield roughly half the ROI the + literature reports. +- **Effect sizes are mostly single-source preprints.** The figures below are + hypotheses to re-measure in our own harness, not constants. What is + multiply-corroborated is the *direction* and the *ranking*. + +This also settles the project's positioning argument. The case for Rust is not +speed — it is that the failure modes dominating these ablations can be made +**type-unrepresentable** rather than merely rare. See ADR-277. + +## 3. The floor + +Seven mechanisms, ordered by measured effect per unit of effort. + +| # | Mechanism | Reported effect | Status | +|---|---|---|---| +| 1 | Reliable patch application | +54.3 pp | Partial | +| 2 | Observation-window management | +3 pp, prevents long-run collapse | Planned (ADR-274) | +| 3 | Loop / stuck detection | Removes most common catastrophic failure | **Done** | +| 4 | Actionable tool errors + response size caps | Part of the reliability delta | Partial | +| 5 | Tool surface held to 8–15 tools | Avoids −16 to −23 pt routing collapse | Holding at 9 | +| 6 | Environment bootstrap injection | Meta-Harness @ 76.4% TB2.0 | Planned | +| 7 | Persisted thinking across tool calls | +2.2 pp coding | Planned | + +### 3.1 Reliable patch application + +Real workspace, file-based edits, git-based diff extraction, and +**verify-after-write** — re-read the file and confirm the edit landed before +reporting success. Offer `str_replace` with fuzzy-failure diagnostics and +`write_file` side by side; **do not** offer unified diff, where line numbers, +hunk headers, and trailing newlines dominate apply failures. + +Edit-tool ergonomics are load-bearing, not incidental: in the Qwen-3.6 +reproduction only the SWE-agent `str_replace_editor` flavor moved the number, +while a different `edit`/`write_file` pair gave *zero* improvement. + +Post-edit gate: run `cargo check` (not a full build) inline. This is the +analogue of SWE-agent's linter guardrail, which its ablations found essential +for recovering from bad edits. + +### 3.2 Loop / stuck detection — **implemented** + +Fingerprint each tool call by `(name, args)`; refuse it once it repeats +consecutively past a threshold (default 3), substituting an actionable message. + +**Counting is consecutive, not windowed.** An agent re-running the same check +between edits is doing legitimate work; a windowed counter refuses it. Only an +unbroken run of identical calls trips the detector. Alternating cycles are not +caught — `max_iterations` remains the backstop, and the limitation is +documented on the type. + +Raising `max_iterations` does **not** fix loops; it makes them more expensive. +`max_iterations` is a cost cap, not a loop guard. + +Refused calls still emit exactly one tool result each, in the model's original +call order, so provider `tool_use`/`tool_result` pairing stays in sync. + +### 3.3 Actionable errors and output caps + +Tool errors must state what failed *and what to do differently*. An opaque +error code causes the model to retry the identical call, which is the input +condition for §3.2. + +Cap tool responses (~25k tokens, matching Claude Code's default) with explicit +truncation markers. An uncapped tool result can consume the context window in +one call. + +### 3.4 Tool surface budget + +**Hold the builtin surface at 8–15 tools.** Routing accuracy degrades 16–23 +points across large catalogs; vendors document degradation past 30–50 tools. +We currently ship 9 builtins — this is a constraint to *defend*, not a target +to grow toward. + +MCP servers must therefore be gated behind explicit per-session enablement. +Exposing an MCP firehose directly into the tool list forfeits this. + +### 3.5 Environment bootstrap injection + +Before the loop starts, snapshot the workspace and inject it into the initial +prompt: cwd, file listing, toolchain versions, `cargo metadata` summary, +workspace members, the test command, and whether `cargo check` currently +passes. This eliminates early exploration turns. ~100 lines of code. + +### 3.6 Persisted thinking + +Do not strip prior-turn thinking blocks from history. Pure protocol plumbing; ++2.2 pp on coding (the smallest of the reported deltas — coding benefits least +because tool results are self-explanatory — but free). + +## 4. Explicitly not in the floor + +Rejected for v1 on evidence, not on effort: + +- **Few-shot demonstrations and explicit CoT instructions** for reasoning + models — zero-shot ≥ few-shot; exemplars can contradict native reasoning. +- **Ungrounded self-reflection loops** — can degrade already-correct answers. + Only execution-grounded critique ("tests failed, here is the output") works. +- **Semantic/embedding code index** — vendor-only evidence, high maintenance, + and stale by construction on a repo the agent is actively editing. Layer + ripgrep → structural search → semantic, and only if a conceptual query + demands it. +- **Elaborate system-prompt frameworks** — the widely-quoted "20–30% + improvement" claims have no published methodology. Keep rulebooks under ~60 + lines. +- **Context windows beyond ~128k** — sweeps plateau around 114k; documented + ceilings sit at 96–112k. Buying more window buys nothing. +- **Learned/RL-trained components** — the hand-written 80% is available for 5% + of the effort. + +## 5. Consequences + +**Positive.** The largest measured deltas land first. Five of seven mechanisms +are days of work. The floor is testable end-to-end without a live provider, +which is how the P0 exit gate is already structured. + +**Negative.** Phase 1's architectural items (event-streaming loop, cache-first +prompt tiers) are deferred behind less glamorous work. This is deliberate: the +evidence does not support them being the biggest lever. + +**Risk.** Effect sizes are largely single-source. Mitigation: §6. + +## 6. Measurement obligation + +Before trusting any A/B of a harness change, **verify test determinism** — run +F2P/P2P repeatedly under gold and base patches. Weak tests inflate resolve +rates by ~6.4 pp, and 1 in 5 "solved" patches on a saturated benchmark are +semantically incorrect. A 3-point improvement sits inside the flaky-test noise +band and means nothing. + +Invest in eval-loop speed. Validation that cannot run in minutes will not get +run, and every mechanism in §3 needs it. + +## 7. Implementation status + +- §3.2 loop detection — **shipped** (`rvagent-core/src/graph.rs`, commit `e709e1a`) +- §3.1 partial — tools write real files; verify-after-write and git diff + extraction outstanding +- §3.4 — holding at 9 builtins +- §3.3, §3.5, §3.6 — outstanding diff --git a/docs/adr/ADR-274-rvagent-context-management.md b/docs/adr/ADR-274-rvagent-context-management.md new file mode 100644 index 000000000..85bcd98ea --- /dev/null +++ b/docs/adr/ADR-274-rvagent-context-management.md @@ -0,0 +1,196 @@ +--- +adr: 274 +title: "rvAgent Context Management: Masking over Summarization" +status: accepted +date: 2026-08-01 +authors: [Reuven Cohen] +project: "rvAgent Harness" +related: [ADR-103, ADR-252, ADR-273, ADR-275, ADR-276] +supersedes_parts_of: [ADR-252] +tags: [rvagent, harness, context, compaction, masking, long-horizon, sota] +--- + +# ADR-274 — rvAgent Context Management: Masking over Summarization + +## Status + +**Accepted.** Owner: Reuven Cohen. Project: rvAgent Harness. Date: 2026-08-01. + +Evidence base: `docs/research/rvagent-hermes-harness/04-sota-landscape.md` §4. + +**This ADR reverses a standing design bet.** `03-roadmap.md` Phase 1.3 +(middle-turn summarization), Phase 2.5 (ADR-252 coherence-weighted +compaction), and the shipped default middleware pipeline all rely on LLM +summarization as the primary context strategy. The evidence says that is close +to the worst available option. + +## 1. Decision + +1. **Observation masking is the default** context strategy, not summarization. +2. **Masked observations keep a dereferenceable ID** (addressable recall). +3. **Programmatic tool calling** is added as a first-class capability. +4. **Invariants are re-injected verbatim after every compaction**, never + summarized. +5. **Context features are capability-gated per model tier.** +6. LLM summarization is retained only as a **rubric-guided fallback**. + +## 2. Why the reversal + +**Observation masking matches or beats LLM summarization at roughly half the +cost** (JetBrains, 250-turn SWE-bench trajectories, NeurIPS 2025 workshop: ++2.6% solve rate at 52% lower cost on Qwen3-Coder 480B). + +The mechanism matters more than the number: **LLM summarization extended +trajectories 13–15%** by destroying natural stopping signals. The agent loses +the cue that it already finished something and keeps working. A hybrid cut +cost 7% below pure masking. + +A placeholder reading `[test output, 2,847 tokens, exit 1, elided]` is more +useful than a mediocre summary, because it preserves the *shape* of history +without fabricating its contents. + +**Addressable Recall Compaction** (arXiv 2607.25066) improves on plain masking +by leaving an ID the agent can dereference on demand. It beat full-context, +sliding window, LLM summary, structured state, *and* RAG memory — NIAH 99.40% +vs 88.12% for the best baseline. This removes the main objection to masking +(irreversible loss). + +### 2.1 The largest lever is not compaction at all + +On LOCA-bench, six context strategies were compared across four frontier +models at 128k context. **Programmatic tool calling — the model writes code +that orchestrates tools, so intermediate outputs never enter context — was the +only strategy positive on every model tested** (+6.0 to +13.3 points). + +Summarize-and-continue compaction was the *weakest* of the obvious strategies +(+2.6 points on the model where it helped most). + +This is absent from the current roadmap and is the single highest-value +context item. + +### 2.2 Compaction silently erases invariants + +"Governance Decay" (arXiv 2606.22528) shows safety constraints and +system-prompt instructions **erode through successive compaction cycles with no +failure signal.** This is not jailbreaking — it emerges from the compression +architecture itself. + +Therefore: the task statement, acceptance criteria, safety constraints, and +system prompt are **re-emitted byte-identical** after each compaction. They are +never inputs to a summarizer. Cheap, and directly counters a documented +mechanism. + +### 2.3 Context features can make weaker models worse + +On LOCA-bench, giving DeepSeek-V3.2 a memory tool dropped it from 10.7% to +**8.0%**, and telling it its remaining context budget dropped it to **4.0%** — +while both features *helped* GPT-5.2 and Gemini-3-Flash. + +Per-model-tier feature flags are therefore mandatory, not optional. A feature +that helps the frontier tier must not be enabled by default for weaker models. + +## 3. Design + +### 3.1 Masking + +Replace old tool observations with typed placeholders carrying: tool name, +elided byte/token count, exit status where applicable, and a recall ID. +**Reasoning steps and actions are kept verbatim** — only observations are +masked. Keep the last N observations in full. + +### 3.2 Addressable recall + +Append-only, ID-addressable log of full observations. The placeholder carries +the handle; a `recall` tool dereferences it. No re-execution, no similarity +search, no embedding index. + +### 3.3 Thresholds + +Compact at **70–80% of nominal context**, not 95%. Treat nominal window as +roughly **2× the reliable working budget** — degradation begins well under half +the advertised window, and it is silent. Anthropic's own API default compacts +at 150k on a 200k-class window, which is the most useful real-world datapoint. + +Do not pursue larger context windows as a substitute: sweeps plateau ~114k and +documented ceilings sit at 96–112k (ADR-273 §4). + +### 3.4 Fallback summarization + +When summarization is used, the **rubric is the load-bearing part**, not the +tool. Preserve explicitly: task statement, files touched, failing tests, +decisions made, unresolved issues. Offering a compaction tool without rubric +guidance produces uneven behavior; a paragraph of guidance closes the gap. + +Guard against recursive summarization of summaries — a documented and fixed bug +in Codex's rewrite. + +### 3.5 Prompt-cache interaction + +Compaction invalidates every cached prefix downstream of the break. Keep a +byte-stable system prefix and place cache breakpoints *before* the volatile +region. Design for 10+ compaction cycles per session. + +## 4. Retrieval policy + +Layered escalation, not a choice between grep and semantic: + +1. **ripgrep** — known symbol, exact identifier, error string +2. **Structural search** (ast-grep / LSP / `syn`) — callers of X, impls of Y +3. **Semantic** — only for conceptual cross-cutting queries + +**Never route a short keyword query to a dense retriever.** CoREB (arXiv +2605.04615) shows short keyword queries — "auth flow", "user service", exactly +the shape agents emit — collapse essentially every embedding model to near-zero +nDCG@10. This explains the 2025 industry migration away from vector search in +coding agents mechanistically. + +Summarize search output by default: return paths and match counts, require a +second call to read contents. + +Any index over code the agent is editing is **stale by construction**. If one +is added later it needs invalidation-on-write or explicit staleness bounds. + +## 5. What breaks first in long runs + +Design priority follows the observed failure order: + +1. **Wasted-context accumulation → attention dilution** — earliest, universal, + invisible. Unaided coding agents waste ~1-in-3 file reads. +2. **History error accumulation** — the largest single driver. Process-level + failures are **72.5%** of long-horizon failures (HORIZON, arXiv 2604.11978). + Errors compound *between* steps. Non-linear: sharp collapse past a + domain-specific threshold. +3. **Compaction-induced loss** — self-inflicted by the mitigation for #1. +4. **Goal / identity drift** — *downstream* of 1–3, not independent. +5. **Hallucinated state** — what #3 looks like when the summary is confidently + wrong rather than merely lossy. + +The evidence does **not** support treating goal drift as the primary problem. +It is the observable end-stage of context and error problems, which is why +this ADR targets 1–3 and ADR-273 targets error compounding. + +## 6. Consequences + +**Positive.** Cheaper and faster than summarization (no model call on the hot +path). Deterministic, so it does not itself become a source of nondeterminism +in replay (ADR-277). No fabrication risk. Preserves stopping signals. + +**Negative.** ADR-252 (coherence-weighted compaction) is demoted from the +default path to a fallback. Work already done there is not wasted — it becomes +the rubric-guided fallback of §3.4 — but it is no longer the primary strategy. + +The `summarization` middleware currently in the shipped default pipeline must +be reordered behind masking, or removed from the default set. + +**Risk.** Masking loses information that a good summary would have retained. +Mitigated by §3.2 addressable recall: the information is still there and still +reachable. + +## 7. Implementation order + +1. Observation masking with last-N-in-full (ADR-273 floor item 2) +2. Verbatim invariant re-injection (§2.2) — cheap, high consequence +3. Addressable recall log + `recall` tool (§3.2) +4. Programmatic tool calling (§2.1) — highest value, largest effort +5. Capability gating (§2.3) +6. Demote `summarization` middleware to fallback (§3.4) diff --git a/docs/adr/ADR-275-rvagent-subagent-topology.md b/docs/adr/ADR-275-rvagent-subagent-topology.md new file mode 100644 index 000000000..97077074d --- /dev/null +++ b/docs/adr/ADR-275-rvagent-subagent-topology.md @@ -0,0 +1,187 @@ +--- +adr: 275 +title: "rvAgent Subagent Topology: Single Writer with Auxiliary Intelligence" +status: accepted +date: 2026-08-01 +authors: [Reuven Cohen] +project: "rvAgent Harness" +related: [ADR-103, ADR-107, ADR-273, ADR-274, ADR-277] +tags: [rvagent, harness, subagents, multi-agent, concurrency, sota] +--- + +# ADR-275 — rvAgent Subagent Topology: Single Writer with Auxiliary Intelligence + +## Status + +**Accepted.** Owner: Reuven Cohen. Project: rvAgent Harness. Date: 2026-08-01. + +Evidence base: `docs/research/rvagent-hermes-harness/04-sota-landscape.md` §3. + +**This ADR reverses a standing design bet.** `03-roadmap.md` Phase 1.6 +specifies "real subagents (JoinSet spawn, semaphore, CoW fork/merge, CRDT +join)". CoW fork plus CRDT join *is* parallel-writer architecture — the one +multi-agent pattern with strong negative evidence for coding specifically. + +## 1. Decision + +**One writer. Auxiliary intelligence around it. Never parallel writes.** + +The subagent boundary is modelled as **a tool that spawns an isolated context +and returns a String** — not as peer agents with a message bus, shared mutable +state, or a mergeable state type. + +Two subagent roles are adopted; a third is deferred. + +| Role | Status | Shares context? | Writes? | +|---|---|---|---| +| Fresh-context reviewer | **Gated** (amended, see §3.1) | **No** — deliberately | No | +| Read-only context-gatherer | Adopt | No | No | +| Coordinator / manager | Defer | — | No | +| Parallel writers | **Rejected** | — | — | + +## 2. Why + +### 2.1 The positive multi-agent result does not transfer to coding + +Anthropic's multi-agent research system reports **+90.2%** over single-agent — +and simultaneously that **token usage alone explains 80% of the performance +variance** (95% with tool-call count and model added). Much of the gain is +*buying more compute*, not coordination; the missing control arm is a single +agent at the same 15× budget. + +Anthropic states directly that the architecture suits **breadth-first** +questions with independent paths and is **less effective for tightly +interdependent tasks such as coding.** + +### 2.2 The skeptical position was revised, not refuted + +Cognition's "Don't Build Multi-Agents" (Jun 2025) argued for single-threaded +linear agents because "actions carry implicit decisions" that conflict when +parallelized. + +Their April 2026 revision — after a year of production data — did not reverse +this. It refined it: + +- **One writer, augmented by auxiliary intelligence. Never parallel writes.** +- Code review loop works: Devin Review catches ~**2 bugs per PR, 58% severe**. +- **Reviewers perform better with NO shared context.** Shorter context → less + context rot → deeper analysis. This inverts the usual "share everything" + instinct and is the most actionable finding here. +- Manager delegation ships but "requires extensive context engineering; + managers default to over-prescription without deep codebase knowledge." + +### 2.3 At equal budget, single-agent wins on coding + +The 2026 consensus across sources: at **equal token budget**, single-agent +matches or beats multi-agent on multi-hop reasoning. Multi-agent earns its +overhead only on breadth-first, parallel-decomposable, low-state-sharing tasks. +Coding is the canonical *bad* fit — it is the case where sub-results are +interdependent and merge conflicts are semantic, not textual. + +A CRDT can merge two edits to the same file without conflict. It cannot make +the *result* coherent. That is precisely the failure Cognition describes. + +## 3. Adopted patterns + +### 3.1 Fresh-context reviewer — **GATED** (amended 2026-08-01, ADR-278 §7) + +Spawns with **no inherited conversation** — only the diff and the task +statement. Returns findings as a string. Does not write. + +The counterintuitive part is load-bearing: do **not** pass the parent's +context. The reviewer's value comes from evaluating the artifact without the +parent's accumulated rationalizations, and from having a short, clean window. + +> **Amendment.** This was originally written as *adopted* on the strength of +> Cognition's production data (~2 bugs/PR, 58% severe). That overstated the +> evidence: **metaharness ADR-226 is a gold-scored null on a closely related +> design** — a read-only strong advisor produced **zero marginal resolves at +> 5.4× cost**, while being genuinely active (33 advisories, 3 vetoes). It was +> not considered when this section was written. +> +> The distinction that may preserve this design: ADR-226's advisor received the +> **full transcript**, whereas this reviewer receives **only the diff** — and the +> Cognition finding is precisely that reviewers do better *without* shared +> context. So ADR-226 does not refute §3.1, but it is the strongest nearby +> negative result and cannot be ignored. +> +> **Status is therefore downgraded from adopted to gated.** The reviewer must +> show marginal lift over a no-reviewer control on the same instances before it +> reaches the default path, and ADR-226's configuration is the specific null it +> must beat. §3.2 below is unaffected — ADR-226 independently corroborates it. + +### 3.2 Read-only context-gatherer + +Explores, reads, greps; returns a summary string. No shared mutable state, no +write tools in its surface. + +Measured (SWE-Edit, Viewer + Editor split): **+2.1 pp resolve, −17.9% cost, +−34.5% main-agent input tokens.** + +**Use a cheap model here.** A specialized Qwen3-8B editor matched GPT-5-nano; +putting GPT-5 in that slot gave **+0.4 pp at 5.8× cost.** Model tiering per +subagent role is part of the design, not an optimization. + +This is now corroborated internally: metaharness ADR-226 measured **zero +marginal lift at 5.4× cost** for a frontier model in a read-only slot. Two +independent measurements, near-identical cost multiple — treat "no expensive +model in a read-only slot" as established, not provisional. + +This is also the cleanest lever on ADR-274 §5 failure #1 (wasted-context +accumulation): exploration output never enters the main window. + +### 3.3 Deferred: coordinator + +Only after single-agent is solid. Requires heavy context engineering to avoid +over-prescription. + +## 4. Rejected: parallel writers, CoW fork/merge, CRDT join + +Rejected on evidence for the coding domain. No rigorous positive coding result +exists; the negative evidence is production-scale and from a team that shipped +the architecture and walked it back. + +**What is kept from Phase 1.6:** `JoinSet` spawning and semaphore-bounded +concurrency remain — they are how §3.1 and §3.2 subagents run concurrently with +each other. What is dropped is CoW state forking and CRDT merge, because +nothing writes concurrently and therefore nothing needs merging. + +This is a substantial simplification: it removes a mergeable state type, the +merge-conflict semantics, and the entire class of bugs where two subagents +make locally-valid but jointly-incoherent edits. + +## 5. Concurrency model + +Retained from the existing loop and unchanged by this ADR: + +- **Read-only tools run concurrently; state-mutating tools run sequentially.** + This is the industry-convergent split and matches MCP's `readOnlyHint`. +- Bounded concurrency via semaphore; results returned in call order. +- Each tool runs in its own task so a panicking tool surfaces as a tool error + rather than crashing the loop (already shipped). + +## 6. Consequences + +**Positive.** Removes the most complex unbuilt subsystem in the roadmap. Buys +nearly all demonstrated multi-agent upside — the reviewer and gatherer are the +two patterns with real production numbers — at a fraction of the complexity. +A `Fn(prompt) -> String` boundary is trivially testable and trivially +replayable (ADR-277). + +**Negative.** Forecloses the "swarm of coders on one repo" demo. That demo has +no supporting evidence for coding tasks and would likely produce incoherent +results, so this is a cost worth paying — but it is a visible capability we are +choosing not to build. + +**Interaction with ADR-107** (rvagent native swarm/WASM): swarm topology +remains valid for *independent* tasks across separate workspaces. This ADR +constrains concurrent writers **within a single workspace on a single task**, +which is the case the evidence covers. + +## 7. Implementation + +1. Subagent-as-tool trait: isolated context in, String out +2. Fresh-context reviewer (no inherited history) +3. Read-only gatherer with a write-free tool surface and cheap-tier model +4. Per-role model override +5. Remove CoW fork/merge and CRDT join from the Phase 1.6 scope diff --git a/docs/adr/ADR-276-rvagent-learning-loop-gating.md b/docs/adr/ADR-276-rvagent-learning-loop-gating.md new file mode 100644 index 000000000..896c80d13 --- /dev/null +++ b/docs/adr/ADR-276-rvagent-learning-loop-gating.md @@ -0,0 +1,237 @@ +--- +adr: 276 +title: "rvAgent Learning Loop: Gating, Trust Tiers and Measurement" +status: accepted +date: 2026-08-01 +authors: [Reuven Cohen] +project: "rvAgent Harness" +related: [ADR-271, ADR-273, ADR-274, ADR-275, ADR-277, ADR-323] +tags: [rvagent, harness, memory, reasoningbank, sona, self-improvement, evaluation, security, sota] +--- + +# ADR-276 — rvAgent Learning Loop: Gating, Trust Tiers and Measurement + +## Status + +**Accepted.** Owner: Reuven Cohen. Project: rvAgent Harness. Date: 2026-08-01. + +Evidence base: `docs/research/rvagent-hermes-harness/04-sota-landscape.md` §5–6. + +**This ADR constrains a standing design bet.** `03-roadmap.md` Phase 2.4 puts +"SONA on the default path". The 2026 literature turned substantially against +this class of system. The component is not cancelled — it is gated behind +measurement it must earn. + +## 1. Decision + +1. **Trajectory-learning memory ships feature-gated OFF by default.** +2. **The measurement apparatus is a precondition for enabling it**, not a + follow-up. +3. **A permanent memory-off control arm** runs for the life of the system. +4. **Episodic storage is immutable**; distilled artifacts are derived. +5. **Consolidation is gated and delta-only** — never end-to-end rewrites. +6. **Trust tiers by verifier**, with untrusted-derived memories barred from + influencing permission or destructive-action decisions. +7. **Promotion uses anytime-valid sequential testing**, never greedy + accept-if-better. + +## 2. Why the constraint + +The case *for* is real and peer-reviewed: ReasoningBank reports +4.6 to +8.3 +points on WebArena across three backbones (ICLR 2026); ACE reports +10.6% on +agents. Distilling from *failures* as well as successes is a genuine +contribution over success-only baselines. + +The case *against* is now stronger: + +- **The gains are confound-sized.** MemDelta (arXiv 2606.29914): swapping the + embedding model alone shifts accuracy **±6.2 pp** — comparable to + ReasoningBank's entire headline gain. Without controlled ablation you cannot + distinguish "our memory design works" from "we picked a better embedder." In + the same work, **agent self-memory (42%) underperformed plain retrieval + (47%)**, and one system reached parity with cloud RAG at **50× the cost**. +- **Memory utility is an inverted U.** "Useful Memories Become Faulty" + (arXiv 2605.12978): utility rises, then degrades *below* the no-memory + baseline. GPT-5.4 failed **54% of previously-solved ARC-AGI problems** when + using consolidated memory. **Episodic-only management doubled accuracy** vs + forced consolidation — the consolidation step is the bug, not the storage. +- **Gains are benchmark-local.** MemoryArena specifically names ReasoningBank's + procedural memory as performing poorly on interdependent multi-session tasks + — the setting closest to real work. +- **No automatic self-evolution method sustains positive gain across settings** + (EvoAgentBench, arXiv 2607.05202). *Curated* ability content transfers across + model families; *automatic extraction* is the failure point. +- **Greedy acceptance is uncontrolled multiple testing.** PACE (arXiv + 2606.08106): "keep it if the score improved" committed **30–42% false + edits**, and made 13–21 spurious modifications when *no true gains existed*, + degrading one agent by 4.9 points. + +Meanwhile **Live-SWE-agent reaches 79.2% on SWE-bench Verified with zero +persistent memory** — on-the-fly tool synthesis from the current trajectory, +discarded after use. It sidesteps every failure mode above. That is not a +coincidence: nothing persistent means nothing to poison, stale, or collapse. + +**Conclusion.** Trajectory-learning memory is a nice-to-have with fragile +upside, not a differentiator (see ADR-277 for what the differentiators are). + +## 3. Design + +### 3.1 Two-tier storage + +**Episodic is immutable.** Append-only raw trajectory store; never rewritten, +never overwritten. Distilled playbook items are *derived artifacts* carrying +pointers back to their source episodes. + +Rationale: consolidation is the documented failure point, and episodic-only +management doubled accuracy against forced consolidation. Raw episodes remain +primary evidence. + +### 3.2 Gated, delta-only consolidation + +Never run consolidation automatically after each task. Never rewrite the +playbook end-to-end. Append or amend individual items with structured deltas. + +Hard per-item length cap (~1,500 chars) and a hard total cap. ACE documents a +single end-to-end rewrite collapsing **18,282 tokens → 122**, dropping +performance *below* the no-adaptation baseline. Production experience +independently shows unconstrained growth past 5,000 chars overfits, and that +length regularization is nearly free (4× compression for −0.8%). + +More data made it worse: 500 samples grew prompt length +75% and *dropped* +performance 2% versus a 20–100 sample sweet spot. + +### 3.3 Trust tiers + +Every candidate memory carries provenance: source episode, verdict source, +verifier type, timestamp. + +| Tier | Backed by | Retrieval | +|---|---|---| +| **A — active** | Programmatic verifier: tests pass, type check, schema validation, invariant assertion | Full weight | +| **B — quarantine** | LLM-judge verdict only | Reduced weight, or withheld until promoted | +| **C — tainted** | Derived from untrusted content: fetched pages, tool output, user-supplied text | Separate namespace. **Never** allowed to influence tool-permission or destructive-action decisions | + +Tier C is a security boundary, not a quality heuristic. See §5. + +### 3.4 Promotion by sequential testing + +Promotion B → A, and any prompt or scaffold edit, requires a **paired +anytime-valid sequential test** (e-process / testing-by-betting) against the +current version on identical held-out instances. Commit only when evidence is +decisive. + +PACE achieved comparable accuracy at **~18% lower evaluation cost** than greedy +acceptance while eliminating the false-commit rate. This is the single +highest-value component of the learning loop. + +### 3.5 Retrieval discipline + +- Inject retrieved memories framed explicitly as **"references, not rules"**. + Nearly free, and drops attack success **20.6% → 13.1%** while raising the + refusal rate 54.4% → 66.9%. +- Relevance-gated top-k with a threshold. **Never concatenate the whole bank** — + ExpeL's documented scaling failure. +- Hard token budget on injected memory; over budget, drop lowest-trust first. +- TTL, decay, and eviction on realized contribution. + +## 4. Measurement — the precondition + +The gate does not open until these run. + +**Primary metric — paired net lift:** +`lift = P(success | memory) − P(success | no memory)` on *the same instances*, +with PACE's e-process providing the stopping rule. Report a confidence +interval, not a point estimate. + +**The regression metric that matters most:** rate of **previously-solved tasks +that now fail with memory on**. This is the 54%-on-ARC-AGI signal and it is the +earliest warning that consolidation has gone bad. Track per consolidation +event; above threshold, roll back and quarantine the items it produced. + +**Permanent control arm.** A fraction of traffic — or a shadow run on a frozen +held-out suite — always executes with memory disabled. Not a one-time +ablation: without a live control the inverted-U crossover is undetectable. + +**Confound controls (run before believing any result):** + +- Hold the embedding model **fixed** across arms; report sensitivity separately. + A ±6.2 pp swing from the embedder alone masquerades as an architecture win. +- Hold the backbone LLM fixed; re-verify on a second backbone. +- **Log refusal rates per arm.** A 63%-refusal arm is not comparable to a 5% one. +- Always include a **plain-retrieval baseline** (BM25 or vanilla embedding RAG + over raw episodes). Failing to beat it is a 50×-cost parity result, not a win. + +**Cost-normalized:** tokens and dollars per *additional* success, not raw +accuracy. + +**Transfer holdout:** a task set from a *different distribution* than the +memories were written from. In-distribution gain is expected and tells you +almost nothing. + +**Per-item attribution:** track retrieval count and conditional lift when +retrieved; evict items with negative or non-significant contribution. This +makes the bank self-pruning and gives an audit trail when something poisons it. + +## 5. Security: the shared brain is the highest-risk surface + +Cross-agent shared memory multiplies the blast radius. Measured +memory-poisoning work reports **~50% attack success and ~41% relapse success**, +with **contextual assimilation** as the primary vector — poisoned entries work +best when they look like ordinary preferences, constraints, or workflow +requirements. Reported >90% of tested agents vulnerable, with **100% relapse** +when teams tried to fix it conversationally. + +Implications: + +- Tier C (§3.3) is mandatory and must be enforced structurally, not by prompt. +- Never write raw credentials, PHI, or secrets to shared memory (already policy). +- Sleeper entries may lie dormant until triggered — per-item attribution (§4) + is the detection mechanism. +- "Misevolution" affects top-tier models; the cheapest known mitigation is the + references-not-rules framing of §3.5. + +## 6. Verifier quality is the binding constraint + +Any learning loop is only as good as its verdict signal, and LLM-as-judge +evidence is poor: a judge surfaced **under 25%** of human-confirmed systematic +problems and **flagged zero** issues in a batch where humans confirmed 23 +distinct defects — implying a **3–6× undercount**. Blind spots are structural: +it catches turn-local problems and is severely blind to cross-turn state. +Calibration drifts — one judge at 0.91 agreement shifted four points after a +model update. + +**Rules:** + +- Programmatic and execution-grounded signals first (tests, type checks, schema + validation, invariants). +- LLM judge is a **secondary, quarantined** signal only (Tier B). +- Treat the judge as a **regression floor, never a promotion authority.** +- Re-anchor against a rolling human-labeled sample after **every** model update. + +## 7. Consequences + +**Positive.** The component ships honestly. If it works we can prove it; if it +degrades we detect the crossover instead of shipping a silent regression. The +measurement apparatus (§4) is reusable for every other harness change and +overlaps with the eval-loop investment ADR-273 §6 already requires. + +**Negative.** Slower to enable than "SONA on the default path". Significant +work lands before any measured benefit. This is the correct trade given §2 — +the alternative is enabling a component whose own literature says it may go +below baseline. + +**Relationship to ADR-271.** ADR-271's Darwin/SONA self-improvement direction +remains valid as *mechanism*. This ADR supplies the gating and acceptance +criteria it lacked, and replaces greedy fitness acceptance with §3.4. + +## 8. Implementation order + +1. Episodic append-only store (immutable, §3.1) +2. Measurement harness: paired lift, previously-solved regression rate, control + arm, plain-retrieval baseline (§4) +3. Trust tiers with structural Tier-C enforcement (§3.3, §5) +4. Retrieval discipline with references-not-rules framing (§3.5) +5. Gated delta-only consolidation with caps (§3.2) +6. PACE-style sequential-test promotion (§3.4) +7. Only then: consider default-on, if and only if §4 shows sustained positive + paired lift on the transfer holdout diff --git a/docs/adr/ADR-277-rvagent-positioning-and-claims.md b/docs/adr/ADR-277-rvagent-positioning-and-claims.md new file mode 100644 index 000000000..dd2e96946 --- /dev/null +++ b/docs/adr/ADR-277-rvagent-positioning-and-claims.md @@ -0,0 +1,307 @@ +--- +adr: 277 +title: "rvAgent Positioning, Protocols and Benchmark Claims" +status: accepted +date: 2026-08-01 +authors: [Reuven Cohen] +project: "rvAgent Harness" +related: [ADR-159, ADR-267, ADR-273, ADR-274, ADR-275, ADR-276] +tags: [rvagent, harness, positioning, mcp, acp, replay, benchmarks, sota, honesty] +--- + +# ADR-277 — rvAgent Positioning, Protocols and Benchmark Claims + +## Status + +**Accepted.** Owner: Reuven Cohen. Project: rvAgent Harness. Date: 2026-08-01. + +Evidence base: `docs/research/rvagent-hermes-harness/04-sota-landscape.md` §2, §7. + +**This ADR invalidates a premise.** The roadmap's implicit positioning — that a +Rust-native harness is itself differentiating — was true when the research +began and is false as of July 2026. + +## 1. Decision + +1. **Do not position on "Rust is fast."** That ground is taken. +2. **Position on two open gaps:** a stable embeddable **library API with open + governance**, and **deterministic replay** as a core primitive. +3. **Migrate to MCP 2026-07-28** before building further on the tool protocol. +4. **Treat ACP as a first-class target.** +5. **Retire SWE-bench Verified as a claim target.** +6. **No claim ships without harness disclosure.** + +## 2. The field is crowded at the top + +Three of the major 2026 harnesses are already Rust: + +| Harness | Scale | License | Notes | +|---|---|---|---| +| **Codex CLI** (OpenAI) | ~70–80 crates | Apache 2.0 | Rewritten *from* TypeScript | +| **Grok Build** (xAI) | ~844k LOC | Apache 2.0 | Open-sourced **2026-07-15** | +| **Goose** (Block) | — | Apache 2.0 | MCP-native | + +OpenAI's stated reasons for the rewrite are exactly our positioning: +zero-dependency install (Node 22+ blocked enterprise and air-gapped +deployments), no GC pauses in long-running agentic processes, memory-safe +sandbox bindings without FFI shims. + +**Tool-surface novelty is also unavailable.** Grok Build's tools are documented +ports — `apply_patch`, `grep_files`, `list_dir`, `read_file` from Codex; +`bash`, `edit`, `glob`, `grep`, `read`, `skill`, `todowrite`, `write` from +opencode. A frontier lab with a million lines of Rust ported the tool surface +rather than designing one. Convergence is complete. + +## 3. What remains open + +### 3.1 No Rust harness is usable as a library + +- Codex's own `AGENTS.md` **discourages** adding to `codex-core` — it is an + app, not a published SDK. +- **Grok Build has issues and PRs disabled.** Contributions explicitly + rejected; xAI develops internally and syncs a mirror. This is *source + transparency, not open governance.* +- Goose is app-first. +- **Anthropic's Agent SDK is Python and TypeScript only** — the docs instruct + other languages to shell out to the CLI with `-p --output-format json`. + +A stable, semver'd, embeddable harness crate with open governance is +unoccupied. This is the primary position. + +### 3.2 Deterministic replay + +A named gap in the Rust agent ecosystem and thin everywhere. Our existing +append-only witness/segment infrastructure is most of the way there, and +ADR-274's deterministic masking (no model call on the compaction path) and +ADR-275's `Fn(prompt) -> String` subagent boundary both make it tractable. + +**Honesty constraint — this is binding.** Even at temperature 0, hosted +inference is not reproducible: floating-point non-associativity and +batch-size-dependent kernels produce run-to-run variation, with reported +accuracy swings up to 15% across runs. + +Therefore the claim is **replay of the harness, not of the model**, and the +reported metric is **action-match rate**, never "reproducible" without +qualification. Record every LLM call, tool response, and timestamp; replay to +reproduce harness behavior; promote incidents to test fixtures. + +Claiming byte-exact reproducibility would be false and would be caught. + +### 3.3 Why Rust, restated honestly + +Not speed. The defensible argument is ADR-273's: the failure modes that +dominate every harness ablation — patch-apply failures, tool-call loops, +desynced `tool_use`/`tool_result` pairing, unbounded tool output — can be made +**type-unrepresentable** rather than merely rare. Plus structured concurrency +(`JoinSet` + `CancellationToken`) making mid-run interrupt and subagent +lifecycle nearly free, where they are hard in Python. + +**Where Rust is a liability — state these plainly:** + +- **Provider coverage.** LiteLLM's 100+ providers is a moat. Rust's best is + ~20. Permanent maintenance tax. +- **Iteration speed on what matters most.** Prompts, tool descriptions, and + compaction rubrics are where harness performance lives, and they want a + REPL. Mitigation: keep prompts and templates in **hot-reloadable external + files**, never `const &str`. +- **The eval ecosystem is Python.** We will shell out for evaluation. +- **Compile times** on a large workspace are a daily cost. +- **Extension authors don't write Rust.** Mitigation: **the extension language + is MCP, not Rust** — Goose's key insight. Pi has 2,143 third-party extensions + because they are TypeScript. + +## 4. Protocols + +### 4.1 MCP 2026-07-28 — migrate now + +Landed 2026-07-28 and is breaking: + +- **Stateless core.** Protocol-level sessions and `Mcp-Session-Id` **removed**. +- Protocol version, client info, and capabilities now travel in `_meta` on + **every** request. +- New `server/discover` method. +- **Tasks extension:** `tools/call` returns a task handle driven via + `tasks/get` / `tasks/update` / `tasks/cancel`. `tasks/list` removed. +- **Deprecated:** Roots, Sampling, Logging; HTTP+SSE reclassified deprecated. +- Auth aligns with real OAuth 2.0/OIDC; clients must validate `iss` per + RFC 9207 (mix-up attack mitigation). + +Building against the 2025-11-25 shape means a rewrite within months. The +official Rust SDK (`rmcp`) already implements the new spec while remaining +compatible with older ones — start there rather than hand-rolling. + +MCP servers remain gated behind per-session enablement (ADR-273 §3.4): the +tool-count ceiling is not negotiable for protocol convenience. + +### 4.2 ACP — first-class + +ACP went from Zed-only (Jun 2025) to headline feature of Zed 1.0 (2026-04-29), +built into JetBrains since Dec 2025, a public registry (2026-01-28), and 25+ +agents by March 2026. **Its reference implementation is Rust.** + +We already have an `rvagent-acp` crate. Being ACP-native rather than +ACP-bolted-on is closer than it looks and is the natural distribution channel +for a library-shaped harness. + +Division of labor as settled in 2026: **MCP = tools, A2A = agent discovery, +ACP = editor↔agent, AG-UI = agent↔UI.** ADR-159's A2A work sits in the third +slot and remains valid. + +## 5. Benchmark claims + +### 5.1 SWE-bench Verified is retired as a claim target + +- **UTBoost (ACL 2025):** resolve rates inflated **~6.4 pp** by weak tests; + **1 in 5** "solved" patches semantically incorrect; augmenting tests changed + leaderboard ranks in **24.4%** of Verified submissions (40.9% on Lite). +- **OpenAI's own audit:** **59.4%** of the hardest unsolved Verified problems + had flawed test cases. OpenAI stopped reporting Verified in early 2026. + +`03-roadmap.md` Phase 4's exit gate (`≥70% on a 350-instance fixed-model set`) +is stated against a benchmark whose **noise band exceeds the effect sizes we +would be claiming.** It is withdrawn pending replacement. + +Verified may still be used as an internal regression signal — with test +determinism verified per ADR-273 §6 — but not as a published claim. + +### 5.2 Claim conformance + +Two 2026 papers establish that harness choices substantially determine +benchmark results (arXiv 2605.23950) and that harnesses induce systematically +different agent *beliefs* on logically equivalent tasks (arXiv 2607.04528). + +Therefore every published claim must disclose: environment setup, tool +implementations, the full harness configuration, and the evaluation procedure. +**A result without harness disclosure is not a result.** + +Retain the existing honesty apparatus (ADR-267): fixed-model comparison, +conformant packaged submissions, Wilson confidence intervals, and retraction +discipline. Given §5.1, retraction discipline is a feature. + +### 5.3 Differentiator claims + +Claim the axes from §3, not throughput: + +- Stable embeddable library API with open governance (§3.1) +- Harness replay with **action-match rate** reported (§3.2) +- Startup latency, memory footprint, single-binary distribution — real, but + **already claimed by Codex and Grok Build.** Supporting evidence, not the + headline. + +## 6. Consequences + +**Positive.** Positioning now rests on gaps that are actually open and on +claims that survive scrutiny. The library-API framing also improves the +internal architecture — it forces a clean core/app split that ADR-275's +subagent-as-tool boundary and §3.2's replay both need. + +**Negative.** The "first fast Rust harness" story is gone. Phase 4's headline +gate is withdrawn without a replacement in hand (§7). + +**Risk — Grok Build's governance may change.** If xAI opens PRs, the "no +community Rust harness" gap closes quickly. The moat must be the library API +and replay, **not merely that a Rust harness exists.** + +## 7. The Phase 4 exit gate (resolved 2026-08-01) + +§5.1 withdrew the SWE-bench-Verified gate without a replacement. This section +supplies one. + +**The obvious successor is also gone.** SWE-bench **Pro** was **retracted by +OpenAI on 2026-07-08**: an audit of its 731 public tasks flagged **27.4% +broken** automatically and **34.1%** by five independent human reviewers, the +dominant failure being over-strict hidden tests enforcing unspecified +implementation details. Two benchmark retractions in six months is the context +every claim we publish now lands in. + +Also dead or unusable: SWE-Lancer (archived 2025-07-18), Aider polyglot (frozen +2025-11-20, no 2026 models), LiveCodeBench (a *model* benchmark — a harness +contributes nothing), OSWorld (self-reported rows, meeting-gated verification), +bare GAIA (a documented **30–50 point** spread on identical tasks purely from +scaffolding). + +### The gate — all four must pass + +**Gate 1 — Terminal-Bench 2.1 absolute.** +**≥ 78.0% pass rate, 5 trials, bootstrap 95% CI half-width ≤ 1.5 pp, on a +mid-tier model, team-verified.** + +Terminal-Bench 2.1 is the only board that is simultaneously team-verified, +CI-reporting, adversarially audited during construction, and *structurally a +harness comparison* (Claude Code, Codex, Terminus 2, Cursor CLI and +mini-SWE-agent all appear on shared models). + +78.0% is deliberate. It sits mid-board and CI-disjoint above the weakest +entries while explicitly **not** claiming to beat the leader at 83.8% ± 1.2. +**78% on a mid-tier model is a stronger result than 84% on a frontier model**, +and it is the honest version of our story. Claiming ≥84% would be overreach +given the CI widths and the reward-hackability base rate. + +**Gate 2 — Fixed-model cross-harness delta.** *This is the actual harness claim.* +**≥ +4.0 pp over Terminus 2 on the identical model, CI-disjoint, replicated on +≥3 models spanning ≥2 vendors.** + +Absolute pass rate is a joint model×harness measurement; only the fixed-model +delta is attributable to us. Cross-vendor is required — a single-vendor result +is indistinguishable from prompt-fitting to one tokenizer. + +Baselines: **Terminus 2** (the reference scaffold) and **mini-SWE-agent** (the +minimal-scaffold control). Both are on the current board, so the delta is +directly auditable. + +Report per-task pass rates, bootstrap CIs, and a **variance decomposition +separating harness-induced from model-induced variance** (arXiv 2605.23950, +which documents model-ranking *reversals* under different harnesses). + +**Publish the falsification:** the model tier where our delta vanishes or +inverts. A harness result with no stated inversion point reads as +cherry-picked, and per §2 of ADR-273 the effect should shrink as model +capability rises — if it does not, that is evidence something is wrong. + +**Gate 3 — Cost-normalized Pareto.** *Our differentiator.* +**Match or beat the best open-scaffold entry's pass rate at ≤ 40% of its +$/task; publish $/task, input+output tokens/task, and wall-clock/task for every +cell of the fixed-model matrix.** + +Terminal-Bench publishes **no cost column at all**, and HAL — the only board +that treats cost as a first-class axis — has **paused submissions**. There is +currently *no* operating cost-normalized agentic-coding leaderboard. Publishing +one in HAL's format is uncontested ground and is the natural claim for a Rust +harness. + +The framing anchor is HAL's own finding: *agents can be 100× more expensive +while being 1% better.* On SWE-bench-Verified-Mini its frontier runs from a +$65.31 cost-efficient knee to a $1,351 point that scores **11 points lower**. +A good harness on a cheap model dominating a mediocre harness on an expensive +one is an existence proof, not a hope. + +**Gate 4 — Contamination-resistant corroboration.** +**A SWE-rebench run on the current rolling window (not a frozen split), +reporting resolved% ± CI and pass@5 on ≥2 of the fixed models.** + +The rolling window makes memorization structurally impossible. This proves the +Terminal-Bench result is neither terminal-specific nor contaminated. + +### Required caveat language + +Every headline number ships with: model, harness, trial count, CI, verification +status, and the fixed-model delta vs Terminus 2 — plus an explicit statement +that absolute pass rate is a joint model×harness measurement and only the delta +is attributable to the harness. Cite arXiv 2605.23950. + +This is cheap, and it is the single thing separating a defensible claim from +the pattern that got two benchmarks retracted inside six months. + +### Data hygiene + +SEO aggregators are publishing leaderboard numbers that do not appear on +primary boards (e.g. inflated Terminal-Bench and SWE-bench figures). Some use +real model names, which makes them more dangerous rather than less. **Cite only +primary leaderboards.** + +## 8. Still open + +Read `codex-rs/core/src/` directly before adopting its `Op`/`EventMsg` +submit/event design — the survey's account is third-party. Note this session +cannot attach `openai/codex` (cross-owner adds unsupported); fetch the files +directly or read them in a session rooted on that repo. diff --git a/docs/adr/ADR-278-rvagent-flywheel-adoption.md b/docs/adr/ADR-278-rvagent-flywheel-adoption.md new file mode 100644 index 000000000..43e1b18bd --- /dev/null +++ b/docs/adr/ADR-278-rvagent-flywheel-adoption.md @@ -0,0 +1,205 @@ +--- +adr: 278 +title: "rvAgent Self-Learning: Adopt the metaharness Flywheel; Shift from Memory to Policy" +status: accepted +date: 2026-08-01 +authors: [Reuven Cohen] +project: "rvAgent Harness" +related: [ADR-271, ADR-273, ADR-275, ADR-276, ADR-277] +external: [metaharness ADR-226, metaharness ADR-228, metaharness ADR-236] +tags: [rvagent, self-learning, flywheel, gepa, policy-evolution, metaharness, promotion, sota] +--- + +# ADR-278 — rvAgent Self-Learning: Adopt the metaharness Flywheel; Shift from Memory to Policy + +## Status + +**Accepted.** Owner: Reuven Cohen. Project: rvAgent Harness. Date: 2026-08-01. + +Sources: `@metaharness/flywheel@0.1.7` (`/workspace/metaharness/packages/flywheel`), +metaharness ADR-226 / ADR-228 / ADR-236, and +`docs/research/rvagent-hermes-harness/04-sota-landscape.md`. + +## 1. Decision + +1. **Adopt `@metaharness/flywheel` as rvAgent's promotion engine.** Do not build + one. Do not port it to Rust. +2. **Shift self-learning investment from memory accumulation to policy + evolution.** ReasoningBank/SONA stays gated per ADR-276; GEPA-style policy + evolution becomes the primary self-learning mechanism. +3. **Adopt `noopRate` as a first-class score axis.** +4. **Contribute anytime-valid sequential testing back to the flywheel gate** + rather than only consuming it. +5. **Record the metaharness ADR-226 null as a binding constraint** on rvAgent's + subagent design (see ADR-275 amendment). + +## 2. Why not build our own + +ADR-276 §3.4 specified a promotion apparatus from scratch. `@metaharness/flywheel` +already implements it, and more rigorously. + +It is deliberately host-agnostic — a stated design rule forbids any host, model, +or benchmark from leaking into the package. Its entire vocabulary is +`Policy = Record`, `Score`, `PromotionEvidence`, and a +`PromotionRule`; everything rvAgent-specific enters through an injected +`Evaluator`. There is no adapter impedance to pay. + +What it already provides that ADR-276 was specifying: + +| ADR-276 requirement | Flywheel | +|---|---| +| Promotion gate | `meetsPromotionRule` — **frozen, conjunctive**, every clause load-bearing | +| Proof the gate did not move | `gateFingerprint()` — SHA-256 over the rule source | +| Transfer holdout | `HoldoutSuite` **plus** a frozen `AnchorSuite` never optimized against | +| Audit trail | Ed25519 `PromotionReceipt` + `verifyReceipt()` | +| Independent verification | `verifyReplayBundle()` — reviewer trusts the signature, not us | +| Compounding, not scattering | Lineage DAG re-basing on the promoted winner; `computeLiftCurve()` | + +The anchor deserves emphasis: ADR-276 §4 asked only for a transfer holdout. The +flywheel requires a candidate to clear a holdout **and** a frozen suite it is +never optimized against. That is a strictly stronger anti-Goodhart guard than we +specified. + +`verifyReplayBundle()` is also, in substance, the replay-verification story +ADR-277 §3.2 positions on — working, in JS, today. + +## 3. The `noopRate` clause + +The default gate's second clause requires the no-op rate to **strictly** improve: + +> a policy earns a promotion by making the executor COMMIT more, not just score higher + +This is non-obvious and load-bearing. A policy that raises the primary metric +while leaving the executor more likely to end empty has not improved the agent; +it has found a scoring artifact. + +**Adopt this axis in rvAgent's own scoring.** It pairs naturally with ADR-273's +reliability framing — "never end empty" is a reliability property, and the ++54.3-point patch-application result is the same phenomenon measured a different +way. + +## 4. Memory versus policy — the reframe + +These are different objects with different evidence: + +| | Object | 2026 evidence | +|---|---|---| +| Flywheel / GEPA | **Policy text** — a genome of named string levers | Positive; GEPA is the best-evidenced optimizer in the sweep | +| ReasoningBank / SONA | **Episodic memory** — accumulated trajectories | Negative; inverted-U, confound-sized gains, self-memory underperforming plain retrieval | + +RuVector's self-learning weight currently sits on the memory side, which is the +side the evidence argues against. **Move the weight to policy evolution.** + +metaharness ADR-228 reaches the same conclusion from its own measurements: +redirect strong-model judgment *offline into the executor's standing operating +policy* rather than injecting it as runtime advice. It notes GEPA's candidate is +a `dict[str,str]` of named text components, matching the flywheel's `Policy` +exactly — the same shape rvAgent would supply. + +ADR-276 is not repealed. Its gating, trust tiers, and inverted-U regression +metric remain the conditions under which memory may ever be enabled. This ADR +changes where *new* effort goes. + +## 5. Two nulls we must respect + +### 5.1 ADR-226 — the read-only advisor is dead + +A frontier read-only advisor over a cheap executor produced **zero marginal +gold-scored resolves at 5.4× cost**. The advisor was genuinely active — 33 +advisories and 3 vetoes across the slice — not silently disabled. The track was +killed. + +This **independently corroborates** the figure in ADR-275 §3.2 from the public +literature (a frontier model in the read-only slot: +0.4 pp at 5.8× cost). Two +independent measurements, near-identical cost multiple. Treat the conclusion as +established rather than provisional: **do not put an expensive model in a +read-only slot.** + +It also constrains ADR-275 §3.1 — see §7. + +### 5.2 ADR-236 — a promotion engine cannot rescue a weak loop + +The flywheel mechanism was proven end-to-end on real SWE-bench and still +produced **no compounding lift, because the base solver was too weak.** Recorded +as an honest null rather than buried. + +**Consequence for sequencing:** adopting the flywheel does not shorten ADR-273. +The reliability floor comes first; the flywheel amplifies a loop that already +works and does nothing for one that does not. This validates the ordering +ADR-273 already set. + +## 6. What we contribute back + +The flywheel's gate is a **single-shot** conjunctive comparison. Running many +generations against the same holdout is uncontrolled multiple testing — the +regime where PACE (arXiv 2606.08106) measured **30–42% false commits**, and 13–21 +spurious modifications even when no true gains existed. + +The conjunctive gate plus frozen anchor mitigates this with multiple hurdles, +which is real but is not anytime-valid. **Frozen conjunctive gate ∧ anytime-valid +sequential test is strictly stronger than either**, and PACE reported ~18% lower +evaluation cost as a side effect. + +This is an upstream contribution to `ruvnet/metaharness`, offered as an optional +`PromotionRule` plus a sequential-evidence accumulator — not a change to the +default gate, whose stability is itself the product. + +## 7. Amendment to ADR-275 + +ADR-275 §3.1 adopted a fresh-context reviewer subagent on the strength of +Cognition's production data (~2 bugs/PR, 58% severe). ADR-226 is the closest +*measured null* to that design and was not considered when §3.1 was written. + +The distinction that may preserve it: **ADR-226's advisor received the full +transcript**, whereas the fresh-context reviewer receives only the diff — and the +Cognition finding is specifically that reviewers perform *better* without shared +context. ADR-226 therefore does not refute §3.1, but it is the strongest nearby +negative result. + +**ADR-275 §3.1 is downgraded from adopted to gated.** The reviewer must +demonstrate marginal lift over a no-reviewer control on the same instances +before it goes on the default path, and ADR-226's design is the specific null it +must beat. §3.2 (read-only gatherer on a cheap model) is unaffected and is in +fact strengthened by §5.1. + +## 8. Integration + +**No Rust port.** Promotion is offline; the flywheel is not on the hot path. + +- Run it in CI at the existing ruflo/metaharness seam (roadmap Phase 3). +- rvAgent supplies an `Evaluator` mapping a run onto the four `Score` axes, and + a `Proposer` for the mutation seam. +- rvAgent's policy genome is the natural `Policy`: system-prompt components, + compaction rubric (ADR-274 §3.4), `loop_repeat_threshold`, masking + `keep_last_observations`, tool-surface composition, per-role model tiers. +- metaharness ships `crates/kernel-napi` as the in-process bridge pattern if we + later need it. We do not need it now. + +**Gap this closes:** ADR-271 (`metaharness-darwin-sona-self-improvement`) does +not reference the flywheel at all. This seam was previously unrecorded on our +side. + +## 9. Consequences + +**Positive.** Deletes the largest unbuilt subsystem in ADR-276 — we consume a +maintained engine instead. Inherits an audit and replay story that already +exists. Moves self-learning onto the side of the evidence. Turns a +one-directional dependency into a two-way exchange (§6). + +**Negative.** A cross-repo dependency on a package at `0.1.x`. Mitigated because +the flywheel is thin, runtime-dependency-free (Node `crypto` only), fully typed, +and — being offline — a version pin is low-risk. + +**Risk.** rvAgent's four `Score` axes must be projected honestly. `primary`, +`noopRate`, `costPerWin`, and `regressed` are where all host meaning lands, and +a dishonest projection defeats every downstream guarantee. The Evaluator is the +trust boundary. + +## 10. Implementation order + +1. Record the ADR-275 §3.1 downgrade (§7) — documentation only +2. Add `noopRate` to rvAgent's score axes (§3) +3. Define the rvAgent policy genome (§8) +4. Evaluator mapping a headless run onto the four axes +5. Wire `runFlywheelGenerations` into CI at the ruflo seam +6. Upstream the sequential-testing `PromotionRule` to metaharness (§6) diff --git a/docs/adr/ADR-279-no-c-and-the-sota-program.md b/docs/adr/ADR-279-no-c-and-the-sota-program.md new file mode 100644 index 000000000..e1cc635a8 --- /dev/null +++ b/docs/adr/ADR-279-no-c-and-the-sota-program.md @@ -0,0 +1,344 @@ +--- +adr: 279 +title: "No C in the Core; and the 2026 SOTA Program for Vector Search" +status: accepted +date: 2026-08-02 +authors: [Reuven Cohen] +project: "RuVector Core" +related: [ADR-264, ADR-267, ADR-268, ADR-272] +supersedes_parts_of: [ADR-267] +tags: [ruvector, performance, simd, ffi, wasm, ann, quantization, benchmarks, sota] +--- + +# ADR-279 — No C in the Core; and the 2026 SOTA Program + +## Status + +**Accepted.** Owner: Reuven Cohen. Date: 2026-08-02. + +Answers two questions asked together: *would adding C improve RuVector's +performance and capabilities?* and *what would actually make it state of the +art?* The answers turn out to be independent — and the second is far more +consequential than the first. + +## 1. Decisions + +1. **No C/C++ in the default build graph.** The premise does not survive 2026 + evidence, and the costs land precisely where RuVector is most exposed. +2. **Retarget the SOTA harness from ann-benchmarks to VIBE, and add + `1/Ratio@k`.** This is the highest-priority item in this ADR and blocks + every performance claim. +3. **Adopt a ranked SOTA program** (§5), led by 8-bit rotational quantization + and a SymphonyQG-class packed quantized graph. +4. **Bind, don't rewrite, for GPU.** `cuvs-sys` for CAGRA build; no pure-Rust + CAGRA. +5. **Differentiate on streaming stability**, which is the least crowded + frontier and happens to be our actual workload. + +## 2. The C question: no + +### 2.1 The premise fails + +Every capability commonly cited as C-only has a production Rust path in 2026: + +| Claimed C-only | 2026 reality | +|---|---| +| io_uring | `io-uring` crate is pure Rust; `bindgen` optional, bindings checked in | +| CUDA | `cudarc` defaults to dynamic loading — no toolkit at build time | +| BLAS/LAPACK | `faer` matches or surpasses OpenBLAS/LAPACK/Eigen | +| AVX-512 | Stable in `std::arch` since Rust **1.89**; **FP16 since 1.94** (Mar 2026) | +| NUMA / hugepages | Syscalls via `libc` — FFI *declarations*, not compiled C | +| RocksDB | Qdrant **removed** it | +| FAISS / hnswlib | Binding it forecloses filtered-search-inside-traversal | +| simdjson | `simd-json` supports AVX2/SSE4.2/NEON **and wasm simd128** natively | + +Two results are decisive rather than merely suggestive: + +- **`zlib-rs` is faster than zlib-ng in C on native, and is the fastest WASM + zlib in existence** (2× miniz-oxide). A direct counterexample to "C is + faster", on exactly the axis proposed. +- **Qdrant — the closest analogue to RuVector — spent two minor versions + removing its one C++ dependency**, naming *"interoperating with C++ slowed us + down"*, plus compaction latency spikes and tuning burden. The comparable + project went the opposite direction from the proposal. + +### 2.2 The cost lands on WASM, which is our largest commitment + +37 of 166 crates are WASM. The rustc platform-support book states plainly that +`wasm32-unknown-unknown` **has no C/C++ toolchain** — not "awkward", absent by +design. Consequences: + +- Any unconditional `cc::Build` in the graph breaks the browser build. +- Escape hatches don't help: emscripten produces an app, not a + `wasm-bindgen` library; `wasm32-wasip1` gives edge/server but not browser. +- The `extern "C"` ABI on wasm is *mid-migration* (future-incompat warning + since Rust 1.87), so any FFI boundary there stands on changing ground. + Rust-to-Rust is unaffected. +- Every C feature must be reimplemented in Rust for the browser anyway. **You + pay for each feature twice and get behavioral divergence for free.** + +### 2.3 Costs beyond WASM + +- **Miri cannot execute across an FFI boundary.** For a database with + concurrent index structures, losing Miri on the hot paths is a real + regression in verification capability. +- A 320-bug study of bindgen/cbindgen/CXX (ACM TOSEM, Feb 2026) found the + dominant failure mode is **not** crashes but *silently generating code + unfaithful to intent*, rooted in data-layout mismodeling. +- `cargo-audit`/`cargo-vet` give **no meaningful coverage of vendored C** + (cf. RUSTSEC-2023-0061, libwebp). +- The FFI boundary is a *pessimization* on fine-grained work: call overhead and + lost cross-module inlining make a C distance kernel slower than the + equivalent `std::arch` kernel. Distance kernels are the definition of + fine-grained. + +### 2.4 The headline "C is faster" number is a datatype artifact + +SimSIMD/NumKong is the most-cited evidence for C kernels, claiming **20–118×** +over autovectorized code. Decomposed, that table compares **NumKong f16 against +GCC's f32** — there is no f32-vs-f32 row, and the README states outright that +GCC "struggles with `_Float16`". Against the compiler's competent f32 output the +real effect is **1.15×–2.1×**. The 3–200× figures elsewhere are against +SciPy/NumPy Python, not compiled code. + +It is also independently contradicted. **PDX (SIGMOD '25) beat SimSIMD's and +FAISS's hand-written kernels by 2.0× on average using plain scalar C++ with no +intrinsics at all**, purely by changing data layout — 3–10× at low dimension. +And an independent harness found NumKong *loses* bulk scoring by **1.85–3.04×**, +because it has no bulk API and cannot hide memory latency. + +**The lever is layout and API shape, not the language the intrinsics are typed +in.** Two expert C teams differ from each other by more (0.72–1.20×) than Rust +differs from C. + +### 2.5 Qdrant ships zero C + +A code search for `simsimd` in `qdrant/qdrant` returns **0 hits**. Qdrant has +AVX-512 `vpopcntq`, `pshufb`/`maddubs`/`VPDPBUSD`, and NEON `SDOT` quantization +kernels — all in Rust `std::arch` — and shipped an ICLR-2026 quantizer +(TurboQuant, v1.18, May 2026) before most C++ engines had it. FAISS, in C++, +still lacks AVX-512 FastScan. + +Separately: plain **23-line Rust with no intrinsics beat the `simsimd` C crate +on Hamming distance** across three machines — autovectorization winning over +manual SIMD on precisely the binary-quantization inner loop. + +### 2.6 The one real gap closes in 18 days + +The largest historically-measured Rust-vs-C gap in this workload was **f32 +reduction reassociation: 8.4×** (84 µs vs 10 µs with Clang fast-math). +`float_algebraic` (`f32::algebraic_add`/`algebraic_mul`) stabilized in PR +#157029, **shipping in Rust 1.98 on 2026-08-20**. Adopt it the day it lands. + +What remains genuinely blocked on stable Rust is narrower than assumed: +**f16/bf16 arithmetic** (tracking #116909) and **ARM SVE/SME** (#145052). +Note that even LanceDB — the one major Rust vector DB linking C — uses it for +three files, and **two of them are pragma-autovectorized C, not hand-written +intrinsics**. What C bought Lance was `_Float16` and fast-math, and one of those +two arrives on 2026-08-20. + +For f16 the cheaper answer is to store f16 and upconvert with stable +`_mm512_cvtph_ps`/`vcvt_f32_f16`, accumulating in f32 at ~1.15–2.1× — or skip +f16 entirely for int8/binary quantization, which is faster *and* smaller. + +### 2.7 The existence proof is already in this repo + +`ruvector-rabitq/src/scan.rs` performs RaBitQ scanning with +`_mm512_popcnt_epi64` and `_mm512_xor_si512`, runtime-detected via +`avx512vpopcntdq`, with an AVX2 fallback — hand-written Rust `std::arch`. That +is precisely the work C would have been imported to do, already done, and it +compiles for WASM. + +Note a correction to an earlier survey of this repo: an initial pass concluded +there were "no binary/hamming popcount kernels". That was wrong — it +generalized from one file (`ruvector-core/simd_intrinsics.rs`) across a +166-crate workspace. The kernel exists. + +### 2.8 The narrow exceptions + +C is acceptable **only** when all hold: behind a non-default feature; in a +separate crate the core does not depend on; with a correctness-equivalent +pure-Rust fallback under differential test; genuinely unreachable in Rust; and +with WASM CI proving the core still builds `--no-default-features`. + +Realistically that means **FIPS-validated crypto** (a compliance case, not a +performance one) and **vendor accelerator SDKs**. `hailort-sys` is already +exactly this, correctly isolated. `cuvs-sys` (§5) qualifies under the same rule. + +## 3. The finding that matters more than C + + +**`ruvector-sota-bench` measures against `ann-benchmarks.com`, which is +deprecated.** Its README now reads: *"no longer actively maintained… consider +submitting your work to different benchmarks, such as VIBE."* Our dataset list +is SIFT-128, GloVe-25/100, Deep-image-96 — precisely the sets VIBE was built +because they are *"no longer representative of the current applications of ANN +search."* + +ADR-267 (SOTA Validation Protocol) does not mention VIBE. + +So: **every SOTA claim RuVector could make today rests on an unmaintained +artifact and non-representative data.** The instrument is pointed at the wrong +thing, and no amount of kernel work fixes that. This is the same failure mode as +withdrawing the SWE-bench-Verified gate in ADR-277 §5.1 — a benchmark can retire +underneath you, and continuing to cite it is how retractions happen. + +### 3.1 Recall@k is itself under credible attack + +Two 2026 papers argue the field has optimized the wrong objective. *ANN Search: +Recall What Matters* (arXiv:2606.04522) proposes **`1/Ratio@k`** and measures +**1.86×–9.36× fewer distance computations** to reach equal downstream quality +versus optimizing Recall@0.95. Downstream validation: image-classification +label precision held at 0.943–0.978 while Recall fell 1.0 → 0.4, and RAG answer +quality varied ≤5% across the same Recall range — `1/Ratio@k` tracked with MAD +0.5–2.6% against Recall's ~29%. + +If that reproduces, a large fraction of tuning effort industry-wide is spent +recovering near-equidistant, semantically irrelevant neighbors. + +## 4. What a defensible claim requires + +Adopted as an amendment to ADR-267: + +- **VIBE datasets, including the out-of-distribution splits.** OOD is a + first-class axis; HNSW's OOD gap is one of its known weaknesses. +- **Both `Recall@k` and `1/Ratio@k`.** +- **QPS at fixed recall on the Pareto frontier**, never single points. +- **Index build time and peak RSS**, reported alongside. +- **Full hardware disclosure** — exact CPU, cores, ISA level. AVX-512 vs AVX2 + vs NEON *changes rankings*. +- **Single-thread and multi-thread separately.** +- **Hyperparameter search budget disclosed for all baselines**, not just ours. +- **Self-published gists and unreproduced blog numbers are not evidence** — + including our own. The research sweep explicitly excluded ruvnet-authored + gists on SymphonyQG/MUVERA/RVQ for this reason, and that discipline stands. +- **Do not headline SIFT1M/GIST1M.** Leading there proves nothing in 2026. + +## 5. The SOTA program, ranked + +| # | Item | Measured effect | Effort | Needs C/CUDA | +|---|---|---|---|---| +| 1 | **Retarget to VIBE + `1/Ratio@k`** | Prerequisite for every claim; 1.86–9.36× wasted compute at stake | 2–3 wks | No | +| 2 | **8-bit rotational quantization** as default codec | >99% recall10@10, 4× compression, ~2.3× faster, **zero training** | **Days** | No | +| 3 | **SymphonyQG-class packed quantized graph** | 1.5–4.5× QPS vs best baselines, 3.5–17× vs hnswlib @95%; best hard-query robustness in VIBE | 2–3 mo | No — `std::arch` VPSHUFB | +| 4 | **Streaming-stable quantizer + dual hot/stable index** | Constant recall under drift vs progressive decay | 6–10 wks | No | +| 5 | **Adaptive filtered-query router** | Specialized-vs-general gap up to 10×, and it *inverts* with scale | 4–6 wks | No | +| 6 | **MUVERA FDE for multi-vector** | ~10% higher recall at 90% lower latency vs PLAID; 32× FDE compression | 3–4 wks | No | +| 7 | **CAGRA build → HNSW serve** via `cuvs-sys` | 12.3× build speedup; no GPU at serve time | 2 wks | **Bind cuVS** | + +**Item 2 has by far the best ratio in this document** — a rotation plus scalar +quantization, no clustering, no training, days of work. Do it while item 3 is +in flight. + +**Item 4 is where we can lead rather than catch up.** Every incumbent is weak: +SPFresh cannot update stably under contention, FreshDiskANN *"fails to maintain +a graph of great quality because streaming data destroys the navigability of the +original index"*, and **all** data-dependent quantizers — PQ, LVQ, RaBitQ, +ScaNN — are trained on a snapshot and go stale. For an agent memory database +this is not a nice-to-have, it is the workload. The relevant papers are from +Dec 2025 – Jun 2026 and none are reproduced. + +### 5.1 Kernel build order (where the SIMD evidence lands) + +Two independent research threads converged on the same conclusion: **layout is +the lever, not intrinsics.** PDX got 2.0× from layout alone with no intrinsics; +SymphonyQG's win is also fundamentally a layout change (RaBitQ codes +co-located with neighbor IDs, FastScan-packed). Order accordingly: + +1. **PDX-style vertical/blocked layout** — 2.0× average, 3–10× at low dimension +2. **Bulk kernel APIs** (1 query × N vectors) that amortize dispatch and hide + memory latency — worth 1.85–3.04× on its own, and the specific reason + NumKong loses bulk scoring despite better single-pair kernels +3. **Binary/int8 quantized kernels** in Rust `std::arch` (`vpopcntq`, `pshufb`, + `VPDPBUSD`, `SDOT`) — all stable since 1.89 +4. **`algebraic_*` f32 paths** — adopt on 2026-08-20 +5. **SVE** only if and when Rust stabilizes it + +Adding C appears nowhere in that list. + +### 5.2 Explicitly deprioritized + +- **TurboQuant** — deprioritized *as a RaBitQ replacement*: theoretically + dominated (bits scale `log log(1/δ)` vs `log(1/δ)`), with reported + quantization times ~2 orders of magnitude optimistic. + **Caveat, stated because the two research threads are in apparent tension:** + a separate thread reports Qdrant shipped TurboQuant in v1.18 (May 2026, ICLR + 2026) beating *binary quantization* by 9–24 pp recall at 16× compression. + These are different comparisons — TurboQuant-vs-RaBitQ on theory, and + TurboQuant-vs-BQ on measured recall — and conflating them would be an error. + The RaBitQ comparison is what governs this decision; the BQ result does not + contradict it. +- **LVQ / LeanVec** — closed-source, Intel-hardware-only, and measured as the + *worst* modern method on quantization error. +- **Pure-Rust CAGRA rewrite** — ~zero gain over binding `cuvs-sys`. +- **SAQ** — the measured frontier (1.8–5.4× lower error than Extended RaBitQ) + but unreproduced and same-community-authored. Revisit after item 1 makes it + measurable. + +## 6. Consequences + +**Positive.** Closes a live risk: we were positioned to publish claims against +a dead benchmark. The program is entirely Rust-native, so it composes with the +WASM story rather than fighting it. Items 2 and 7 are cheap and land early. + +**Negative.** Item 1 delays visible performance work by 2–3 weeks. That +sequencing is deliberate and matches the harness lesson in ADR-273 §6: a +measurement you cannot trust makes every subsequent optimization unfalsifiable. + +**Risk.** Several headline numbers (SAQ, CoDEQ, OctopusANN) are single-source +and unreproduced. They are treated as hypotheses to re-measure under item 1, +not as constants — the same evidence discipline applied in ADR-273. + +## 6.1 Measured: PDX layout did NOT reproduce (2026-08-02) + +A first implementation of the PDX vertical layout (`ruvector-core::pdx`) was +built and benchmarked against the existing row-major batch path on this host +(AVX-512, 4 cores). **The paper's ~2.0× did not reproduce.** Measured, both +paths compiled with `-C target-cpu=native`, 4096 vectors: + +| n × dim | working set | row-major | PDX vertical | ratio | +|---|---|---|---|---| +| 256 × 768 | 0.75 MB | 13.73 µs | 19.52 µs | **0.70×** | +| 512 × 768 | 1.5 MB | 29.78 µs | 45.59 µs | **0.65×** | +| 1024 × 768 | 3 MB | 123.5 µs | 140.4 µs | 0.88× | +| 4096 × 768 | 12 MB | 650.2 µs | 505.4 µs | 1.29× | +| 4096 × 1536 | 25 MB | 1158 µs | 1065 µs | 1.09× | + +PDX **loses** on cache-resident working sets and wins only when streaming. + +Three caveats, stated because they bound what this measurement proves: + +1. **The first run was invalid.** It showed PDX 14–18× slower, because the + row-major path runtime-dispatches to AVX-512 via `is_x86_feature_detected!` + while the new code compiled for baseline x86-64 (SSE2). That was an ISA + comparison wearing a layout comparison's clothes. Fixed by building both + with `target-cpu=native`. +2. **The remaining comparison is still confounded.** The row-major path takes + `Vec<&[f32]>` derived from `Vec>` — 4096 separate heap allocations, + so it pointer-chases — while `PdxIndex` is a single contiguous buffer. The + streaming win may be *allocation contiguity*, not vertical layout. A clean + experiment needs a contiguous row-major baseline. +3. **The PDX kernel is autovectorized generic Rust**, competing against + hand-written AVX-512 intrinsics. That the naive version reaches parity at + all is notable, but it is not a like-for-like layout test. + +**Conclusion for now: do not adopt PDX layout on this evidence.** The honest +reading is that at f32 precision these workloads are bandwidth-bound, so layout +cannot help much — which *strengthens* the case for §5 item 2 (8-bit rotational +quantization), since 4× less data to stream attacks the actual bottleneck. +Revisit PDX for quantized codes, where the working set shrinks enough to become +compute-bound. + +This is recorded rather than discarded because a negative result that cost a +day is worth more written down than repeated. + +## 7. Open + +- Third research thread (SIMD kernel evidence across FAISS/SimSIMD/usearch) + outstanding; it refines kernel specifics — VNNI int8, f16/bf16 now that + AVX-512 FP16 is stable — not the decisions above. +- Pull VIBE's `results/summary.parquet` to calibrate against the field before + writing index code. +- Audit whether `ruvector-rabitq` implements 1-bit only or extended/multi-bit + RaBitQ, and whether `rotation.rs` uses a fast Hadamard/Kac's walk or an + O(d²) dense rotation. diff --git a/docs/research/rvagent-hermes-harness/01-findings.md b/docs/research/rvagent-hermes-harness/01-findings.md new file mode 100644 index 000000000..d03549aa9 --- /dev/null +++ b/docs/research/rvagent-hermes-harness/01-findings.md @@ -0,0 +1,210 @@ +# Findings: Hermes, rvAgent, MetaHarness, RuFlo + +Research conducted 2026-08-01 across four parallel investigations: +web research on the Hermes harness, and deep code audits of +`crates/rvAgent/` (this repo), `ruvnet/metaharness`, and `ruvnet/ruflo`. + +--- + +## 1. The Hermes Harness (NousResearch/hermes-agent) + +MIT-licensed Python harness from Nous Research, released Feb 2026; ~175K +GitHub stars in four months, the most-used agent on OpenRouter by mid-2026. +Fully model-agnostic (18+ providers, 3 API modes, mid-session failover). +Sources: [repo](https://github.com/NousResearch/hermes-agent) · +[architecture docs](https://hermes-agent.nousresearch.com/docs/developer-guide/architecture) · +[self-evolution companion](https://github.com/NousResearch/hermes-agent-self-evolution). + +### Architecture + +- **Loop:** one synchronous `AIAgent` class (`run_agent.py`) serves CLI, + messaging gateway, ACP, batch, and API-server modes. Task ID → prompt + build → preflight compression check → provider resolution → API call → + tool dispatch loop → SQLite persistence. +- **Cache-first prompts:** ordered tiers `stable` (identity, tool guidance, + skill summaries) → `context` (user context files) → `volatile` + (memory/profile/timestamp). Invariant: *the system prompt never changes + mid-conversation*. Only model switch or memory/context file change breaks + the prefix cache → **91–97% measured cache hit rates**. +- **Tools:** 70+ tools/28 toolsets, self-registering; MCP for extension. + Benchmark configs run well with **only `terminal` + `file` enabled**. +- **Sandboxing:** 7 terminal backends (local, Docker, SSH, Singularity, + Modal, Daytona, Vercel) behind one interface; Docker defaults to + read-only root FS + dropped capabilities. Zero telemetry. +- **Context management:** a sentinel triggers compaction *before* hard + limits; an auxiliary model extracts durable facts into memory (hard + 3,575-char budget) and **summarizes middle turns instead of dropping + them**; compressed sessions keep parent/child **lineage** in SQLite. +- **Memory, four separated layers:** + 1. *Prompt*: MEMORY.md / USER.md, always injected, hard char budget + 2. *Episodic*: SQLite + FTS5 over all past sessions, retrieved on demand + 3. *Procedural*: **skills** — markdown in `~/.hermes/skills/` + (agentskills.io standard), progressive disclosure + 4. *User model*: optional passively-built profile +- **The learning loop (signature feature):** skill creation triggers on 5+ + tool calls, error recovery, or user correction → trajectory distilled + into a named skill; skills self-patch during use (string-patch default); + offline, GEPA (Genetic-Pareto, ICLR 2026 Oral, ~35× fewer rollouts than + GRPO) rewrites underperforming skills from execution traces. +- **Subagents:** `delegate_tool.py` spawns isolated subagents, multi-model + routing per subtask, programmatic tool-calling via `execute_code`. + +### Benchmark evidence + +- **Claw-SWE-Bench** (arXiv 2606.12344): harness choice alone swings pass@1 + by **12.5 pts** (strong model) to **27.4 pts** (weak model). Hermes 71.1% + with GLM 5.1. +- **Harness-Bench** (arXiv 2605.27922): Hermes 71.2% overall, 100% security; + paper's conclusion — "execution alignment" (model beliefs ↔ workspace + state ↔ tool feedback ↔ verification) is the dominant success factor. +- **Learning-loop payoff:** agents with 20+ self-created skills complete + similar tasks **~40% faster / 40% fewer tokens** (Nous internal, + independently corroborated; domain-specific). +- **Agentic Harness Engineering** (arXiv 2604.25850): *structure transfers, + prose doesn't* — tools, middleware, memory architecture generalize across + models; prompt wording tweaks don't. An observability-driven self-evolving + harness beat human-designed Codex-CLI on Terminal-Bench 2 (77.0% vs 71.9%). + +Hermes's reputation rests on **amortized performance via the learning loop +and ecosystem dominance**, not one-shot leaderboard wins — it is top-tier +but not #1 on frozen single-run benchmarks. + +--- + +## 2. rvAgent Current State (crates/rvAgent/, ~45K LoC, 10 crates) + +Declared as a 100%-fidelity Rust port of LangChain DeepAgents (ADR-093..103), +extended with MCP (ADR-104/105/112), A2A (ADR-159), RVF (ADR-106), WASM. + +### Production-grade parts + +- **`rvagent-a2a`** — the best crate: Ed25519-signed AgentCards, global + rolling budgets, per-task policy, peer routing with EWMA + circuit + breaker, recursion guard, W3C trace context, typed/versioned artifacts, + SSE + signed webhooks. 24 integration test files. *But no real runner — + `InMemoryRunner` echoes; not connected to the agent loop.* +- **`rvagent-mcp`** — complete JSON-RPC 2.0 MCP server/client, stdio/SSE, + tool groups, skills bridge (Claude Code + Codex formats). Implemented + per ADR-112. +- **Security primitives** — path confinement (`virtual_mode`), env + sanitization, Unicode/BiDi/homoglyph detection, tool-output sanitizer, + AES-256-GCM session encryption, SHA3-256 witness chains. + +### Blocking defects (the loop is scaffold-grade) + +| # | Defect | Evidence | +|---|---|---| +| D1 | **Tool schemas are never sent to the model.** `ChatModel::complete(&[Message])` has no tools param; Anthropic/Gemini request structs have no `tools`/`functionDeclarations` field. The loop's `Tools` node is unreachable in production — rvagent is currently a chat client, not an agent. | `rvagent-core/src/models.rs:140`, `rvagent-backends/src/anthropic.rs:69-80`, `gemini.rs:32-35` | +| D2 | **The 19-module middleware pipeline is never wired into the loop.** `build_default_pipeline` is called only from benches/tests; the CLI constructs `AgentGraph::new(model, tools)` directly. Memory, skills, summarization, prompt caching, witness, HITL, SONA, HNSW all dormant. | `rvagent-cli/src/app.rs:643-728` | +| D3 | **Three incompatible type systems.** `AgentState`, `Message`, `Tool`, `TodoItem`, `RunnableConfig` each defined 2–3× (core enum `Message` vs middleware struct vs subagents `HashMap`), no conversion layer. This is the structural blocker for D2. | `core/src/state.rs:81`, `middleware/src/lib.rs:130`, `subagents/src/lib.rs:39` | +| D4 | **Subagents, parallelism, streaming are stubs.** `spawn_sync` returns a formatted string; "parallel" tool execution awaits sequentially in a loop (README's "true concurrency" claim is false); `stream()` returns "not yet implemented"; middleware hooks are sync (can't make an HTTP call without blocking). | `subagents/src/orchestrator.rs:44-110`, `core/src/graph.rs:183-193`, `anthropic.rs:378-385` | + +Additional gaps: tool errors abort the loop via `?` instead of feeding back +as tool results (the single most important recovery behavior); session +persistence is messages-only (no todos/files/middleware state); no hooks +infrastructure at all; HITL has no approval transport; the "HNSW" middleware +uses **hash-based pseudo-embeddings with no semantic properties** and no +connection to real RuVector crates; prompt caching types exist but nothing +is emitted to the API; docs claim capabilities the code doesn't have. + +ADR-139 (decompiled Claude Code intelligence: async-generator loop yielding +13 event types, 6 permission modes, per-subagent model override) and ADR-107 +(`rvagent-swarm`) are proposed but unimplemented. + +--- + +## 3. MetaHarness (ruvnet/metaharness) + +A **harness factory + evolution lab**, not a harness runtime. Node/TS +(~50K LoC) + a small Rust kernel (2,259 LoC → WASM/NAPI). 19 published +`@metaharness/*` packages, 223 ADRs, exemplary CI/release engineering. + +- **Generator:** `npx metaharness` mints branded harnesses for **nine hosts** + — Claude Code, Codex, pi.dev, **Hermes**, OpenClaw, RVM, Copilot, + OpenCode, GitHub Actions — via a `HostAdapter.generateConfig(spec) → + {path: contents}` interface. The Hermes adapter is verified against the + real `cli-config.yaml` and mirrors ruflo's `scrubReasoningBlocks`. + **Generated harnesses contain no agent loop** (`init` + `doctor` only) — + the loop is provided by the host. +- **Rust kernel (`crates/kernel`):** MCP spec validation + `ToolRegistry`, + claim-checked `dispatch()`, **10-event hook taxonomy with + Allow/Deny/Ask/Defer decision-merge** (modeled on Claude Code), 3-tier + routing heuristic, Ed25519 witness, cost, federation. `#![forbid(unsafe_code)]`, + serde-typed, no async/no I/O — a library rvagent could consume or mirror. + (Memory module is a 34-line stub; real memory delegates to `@ruvector/*`.) +- **Darwin Mode (the credible asset):** a real DI-tested ReAct loop + (`bench/swebench/agentic-loop.mjs`: text-JSON and native function-calling + variants, anti-thrash state hashing, observation caps, escalation + cascades) plus the full measurement apparatus — official SWE-bench Docker + gold eval (**Verified 55.6%**, Lite 51.3%, conformant packaged + submissions), Terminal-Bench, LiveCodeBench, GAIA/FRAMES, DRACO, with + Wilson CIs and documented retractions of its own failed claims. +- **The flywheel policy seam:** `SWE_POLICY_SYSTEM` env appends an evolved + operating policy to the solver's system prompt without touching loop code + — the template for how Darwin should drive rvagent. +- **Key structural fact:** the marketed control plane (`@metaharness/harness` + `HarnessKernel`, ADR-047, still Proposed) and the loop that produced every + measured number (`agentic-loop.mjs`) **share no code**. The runtime slot + is open. + +RuVector's own ADR-256 already ruled: **borrow metaharness concepts, do not +take it as a runtime dependency** — and ruflo's ADR-150 enforces the same +invariant in the other direction (metaharness must stay removable). + +--- + +## 4. RuFlo (ruvnet/ruflo = claude-flow v3, TS, ~173K LoC in the CLI) + +Self-described *"agent meta-harness for Claude Code and Codex … Ruflo is the +harness — the execution layer around Claude Code."* `Agent = Model + Harness`. + +- **The execution split:** ruflo **coordinates; Claude Code executes.** + `agent_spawn` is registry metadata; multi-turn tool-use execution is + delegated to Claude Code's Task tool or `claude -p` subprocesses. Ruflo + itself executes only: single-turn LLM calls with bandit-fed model routing + (`agent-execute-core.ts`), headless Claude subprocesses (worker daemon, + fable-harness LLM-judge), and — the one real in-house turn loop — + **rvagent's WASM build** via `@ruvector/rvagent-wasm` + (`ruvector/agent-wasm.ts`, 27 `wasm_agent_*`/`wasm_gallery_*` MCP tools). +- **Memory substrate (battle-tested):** `.swarm/memory.db` (SQLite/AgentDB), + `.swarm/hnsw.index`, `.rvf` containers; hybrid retrieval (BM25 + + cross-encoder rerank + RaBitQ + graph edges); **ADR-323 provenance typing** + (`user_claim|agent_output|system_observation|tool_result`) that any + co-writing runtime must honor; ReasoningBank with HNSW-backed pattern + promotion; SONA/EWC++/LoRA distillation pipeline + (RETRIEVE→JUDGE→DISTILL→CONSOLIDATE). +- **Hooks bus:** all 8 Claude Code lifecycle events funnel through + `.claude/helpers/hook-handler.cjs` — a subprocess contract a Rust binary + could shim or replace. 17 hooks + 14 background workers. +- **Model routing:** 3-tier (codemods $0 / Haiku / Sonnet-Opus) with a + persisted Thompson-sampling bandit closed-loop from execution outcomes. +- **MCP:** hand-rolled stdio JSON-RPC server, 305 tools; the **Capability + Brain** (typed maturity/authority/risk/health metadata per domain) is the + best-designed integration seam for advertising a new runtime. +- **Governance:** `.harness/mcp-policy.json` default-deny; ADR-150's + invariant — metaharness may augment, never be required. +- **Honesty:** ruflo's docs audit their own claims (e.g. "150x-12,500x NOT + reproduced — was brute-force fallback"). Its IMPROVEMENT-ROADMAP names + **"skill synthesis vs Hermes-class agents"** the highest-leverage missing + capability — DISTILL exists but never emits shareable SKILL.md artifacts. +- **rvagent is already first-class:** `plugins/ruflo-agent` wires WASM + rvagent + `wasm_agent_compose` hands agents a **safety-gated allowlist of + ruflo MCP tools** (destructive-tool gate included); plugins opt in via a + `rvagent.exposeSkillsAsTools` manifest field. + +--- + +## 5. Synthesis: the opening + +| Harness | Loop | Learning loop | Evolution apparatus | Coordination plane | Native speed / WASM | +|---|---|---|---|---|---| +| Hermes | ✅ mature | ✅ (skills+GEPA) | partial (offline) | ❌ | ❌ (Python) | +| Claude Code | ✅ mature | partial (manual skills) | ❌ | partial (Task tool) | ❌ | +| metaharness | ❌ (bench-only) | ❌ | ✅ (Darwin/flywheel) | ❌ | kernel only | +| ruflo | ❌ (delegates) | partial (DISTILL, no skills) | partial (harness-loop) | ✅ (305 tools, memory, swarm) | via rvagent-wasm | +| **rvagent today** | ❌ broken | ❌ | ❌ | via A2A (no runner) | ✅ (Rust+WASM+NAPI) | +| **rvagent target** | ✅ Hermes-class | ✅ (skills+SONA+witness) | ✅ (via metaharness) | ✅ (via ruflo) | ✅ | + +Every column of the target row has real code behind it somewhere in the +rUv ecosystem. The work is repair (rvagent's loop) plus wiring (the seams), +not greenfield invention. See [02-target-architecture.md](02-target-architecture.md). diff --git a/docs/research/rvagent-hermes-harness/02-target-architecture.md b/docs/research/rvagent-hermes-harness/02-target-architecture.md new file mode 100644 index 000000000..9b6914b5e --- /dev/null +++ b/docs/research/rvagent-hermes-harness/02-target-architecture.md @@ -0,0 +1,217 @@ +# Target Architecture: rvAgent as a Hermes-Class, Self-Evolving Rust Harness + +Design principle (from the benchmark literature): **structure transfers, +prose doesn't**. Everything below is structural — tools, loop mechanics, +memory layers, seams — not prompt wording. + +--- + +## 1. The Loop (rvagent-core) + +Replace today's blocking 4-node state machine with an **event-streaming +loop** (the ADR-139 direction, matching Claude Code's decompiled design and +Hermes's single-class-many-modes pattern): + +```rust +// One loop, many frontends (CLI, TUI, ACP, A2A runner, WASM, headless) +pub trait AgentLoop { + fn run(&mut self, input: LoopInput) -> impl Stream; +} + +pub enum AgentEvent { + TurnStart { .. }, ModelDelta { text: String }, // streaming + ToolCallStart { id, name, args }, ToolCallEnd { id, result }, + PermissionRequest { .. }, // HITL surfaces here + CompactionStart { .. }, CompactionEnd { lineage: SessionLineage }, + SubagentSpawned { .. }, SubagentResult { .. }, + Checkpoint { id }, TurnEnd { usage: Usage }, Done { state: AgentState }, + Error { recoverable: bool, .. }, +} +``` + +Non-negotiable loop behaviors (each maps to a measured Hermes/Harness-Bench +lesson): + +1. **Tools go to the model.** `ChatModel::complete(&[Message], &[ToolDefinition])` + and `tools` on the Anthropic/Gemini request bodies. (Fixes D1.) +2. **Tool errors are tool results.** Never `?`-abort the loop on a failed + tool; feed the error text back as a `Message::tool` so the model can + recover. ("Execution alignment" — the dominant Harness-Bench factor.) +3. **Real parallel tool execution** via the existing (unused) + `parallel_execute` JoinSet+Semaphore path. (Fixes D4.) +4. **Real streaming**: SSE parsing in the backends, `impl Stream`, incremental TUI render. +5. **Few, high-fidelity core tools.** Keep the 9 built-ins; Hermes wins + benchmarks with `terminal` + `file` only. Everything else arrives via + MCP (rvagent-mcp client) and ruflo's `wasm_agent_compose` allowlist — + never as bespoke tool sprawl. +6. **One canonical type system.** `rvagent-core`'s `AgentState`/`Message`/ + `Tool` become the only definitions; middleware and subagents consume + them. (Fixes D3 — the blocker for everything else.) +7. **Async middleware.** `Middleware::wrap_model_call` becomes async; the + default pipeline is constructed and wired in every entrypoint. (Fixes D2.) + +## 2. Cache-First Prompt Assembly (new: `prompt_builder` in rvagent-core) + +Hermes's 91–97% cache hit rate comes from a discipline, not a feature: + +``` +[stable] identity + tool guidance + skill summaries ← changes only on config change +[context] AGENTS.md / project context files ← changes only on file change +[volatile] memory digest + todos + timestamp ← the ONLY tier that moves +``` + +- The stable tier is emitted with `cache_control` breakpoints (the existing + `PromptCachingMiddleware` + a new `cache_control` field on `ApiRequest`). +- Enumerate cache-breaking events (model switch, memory-file change, + context-file change) and log them — cache hit rate becomes a first-class + metric in `Metrics`. +- The system prompt never mutates mid-conversation; volatile data rides in + the tier boundary, and compaction respects the cache boundary. + +## 3. Compaction with Lineage (wire + upgrade `SummarizationMiddleware`) + +- Sentinel triggers **before** the hard limit (Hermes pattern): summarize + middle turns, keep head (stable prompt) and tail (recent turns) intact. +- Durable facts extracted to the memory layer under a **hard char budget** + (forcing curation, per Hermes's 3,575-char discipline). +- **Lineage**: compacted sessions record parent/child chains in the session + store so summaries trace back to raw turns (enables replay + Darwin's + trajectory harvesting). +- Upgrade path: ADR-252 coherence-weighted compaction + (`ruvector-agent-memory`) replaces naive char-window summarization — a + RuVector-native capability Hermes doesn't have. +- Replace the chars/4 token estimate with a real tokenizer + per-model + context-window table. + +## 4. Four-Layer Memory (mirror Hermes, back with RuVector/ruflo substrate) + +| Layer | Hermes | rvagent implementation | +|---|---|---| +| Prompt | MEMORY.md/USER.md, hard budget | `MemoryMiddleware` (AGENTS.md, exists) + write-back with budget enforcement | +| Episodic | SQLite+FTS5 over sessions | session store + **ruflo's `.swarm/memory.db`** via AgentDB bindings; honor ADR-323 provenance (`agent_output`, `tool_result`) and the WAL-sidecar guard | +| Procedural | skills (markdown, agentskills.io) | `SkillsMiddleware` + `skills_bridge` (exists; already speaks Claude Code + Codex formats) with progressive disclosure at runtime | +| User model | Honcho profile | optional; defer | + +**Kill the fake HNSW.** The current hash-based pseudo-embedding middleware +is worse than nothing (locality-insensitive, unfounded perf claims). Replace +with real `ruvector-core`/`@ruvector` embeddings behind the existing feature +gate, or use ruflo's hybrid retrieval over MCP. FTS-first, ANN-second — the +Hermes lesson is that deterministic local search beats a vector DB for a +local-first harness. + +## 5. Skills + The Learning Loop (the differentiator) + +This is where the three systems interlock, and it directly closes ruflo's +self-identified top gap ("DISTILL never emits shareable SKILL.md"): + +``` +witnessed trajectory (rvagent witness chain, SHA3-256 tool-call entries) + │ trigger: 5+ tool calls / error recovery / user correction + ▼ +skill synthesis (auxiliary model distills trajectory → SKILL.md) + │ progressive disclosure: summary in [stable] tier, body on demand + ▼ +in-use self-patching (string-patch edits, skill_manage tool) + │ + ▼ offline +metaharness Darwin/flywheel: GEPA-style evolution of skills + policy genome + - fitness from rvagent headless runs (SWE-bench/Terminal-Bench/TBLite) + - promotion via harness-loop gates (OBSERVE→QUALIFY→BENCHMARK→VERIFY→CANARY→ACCEPT) + - Ed25519-signed champion manifests (out-of-loop signing) + │ + ▼ +SONA/ReasoningBank consolidation (EWC++, ADR-271 genome recipe) + + optional weight-eft: SFT/DPO distillation of gold trajectories into cheap-tier LoRA +``` + +The **policy genome seam** is metaharness's `SWE_POLICY_SYSTEM` pattern: +rvagent accepts an evolved operating policy as an appended stable-tier +block (`--policy-file` / env), so Darwin mutates behavior without touching +loop code. Genome surfaces: policy text, temperature schedule, compaction +thresholds, tool config, skill set, `EwcConfig` (ADR-271). + +## 6. Subagents (make ADR-097 real) + +- `TaskTool` → `SubAgentOrchestrator` actually spawns `AgentLoop` instances + via JoinSet with a concurrency semaphore; results stream back as + `SubagentResult` events; CRDT merge + validators (already written) run on + join. +- Per-subagent model override (ADR-139's `CLAUDE_CODE_SUBAGENT_MODEL` + equivalent) — routing decision can come from ruflo's Thompson bandit via + MCP, or the local 3-tier heuristic. +- Subagent state isolation via the existing `CowStateBackend` fork/merge. +- Cross-*process/machine* delegation is already solved: the A2A crate. + Plug the real loop in as the A2A `TaskRunner` (replacing `InMemoryRunner`) + and rvagent instances federate with budgets, policy, recursion guards, + and signed identity for free — a capability Hermes does not have. + +## 7. Hooks, Permissions, Checkpoints + +- **Hooks:** adopt metaharness `crates/kernel`'s 10-event taxonomy + + `Allow/Deny/Ask/Defer` decision-merge (it's `#![forbid(unsafe_code)]`, + serde-typed, no-I/O — mirror it per ADR-256 "borrow concepts", or take it + as an *optional* dep consistent with ADR-150 symmetry). External hook + processes use the ruflo `hook-handler.cjs` subprocess contract so one + hook ecosystem serves both. +- **Permissions:** wire `HumanInTheLoopMiddleware` to the event stream + (`PermissionRequest` event ↔ approval reply), add permission modes and + per-tool allow/deny rules (ADR-139's 6-mode model). +- **Checkpoints:** full-state (messages + todos + files + middleware state + + in-flight tool calls) using the AGI container's unused + `SegmentType::Checkpoint`; resume/fork from any checkpoint. This is what + makes Darwin's population runs cheap (metaharness measured 39.3% cost + saved on resume, synthetic). +- **Sandbox depth:** keep env-sanitized local shell as the fast path, add a + Docker backend (read-only root, dropped caps — Hermes's secure default) + and honor ADR-140's WASM double-sandbox for untrusted agents. The + `SandboxBackend` trait already models this; it needs implementations. + +## 8. Headless Mode (the benchmark/evolution contract) + +`rvagent run -p "" --output-format json|stream-json --max-budget-usd +--policy-file --checkpoint-dir` — a stateless one-shot invocation emitting +the event stream as JSONL + a final result envelope (cost, tokens, tool +calls, patch). This single interface is what lets: +- metaharness Darwin use rvagent as a solver backend (exactly how + `handoff-solver.mjs` shells out to `claude -p` today), +- ruflo's worker daemon spawn rvagent instead of headless Claude, +- CI run frozen-eval gates. + +## 9. Integration Seams (explicit contracts) + +### rvagent ⇄ ruflo +| Seam | Mechanism | Status | +|---|---|---| +| Turn loop hosting | `@ruvector/rvagent-wasm` (`WasmAgent`, `JsModelProvider`) + a NAPI sibling for native speed | WASM exists; NAPI new | +| Tool surface | `wasm_agent_compose` safe-allowlist of ruflo's 305 MCP tools (destructive gate) | exists | +| Memory | shared `.swarm/memory.db` / `.rvf`; ADR-323 provenance mandatory | bindings exist (AgentDB/ruvector are Rust-origin) | +| Lifecycle | hook-handler.cjs subprocess contract; Capability Brain entry with maturity/authority/risk metadata | contract exists | +| Routing | consume `[TASK_MODEL_RECOMMENDATION]` signals; report outcomes to the bandit | signal exists | + +### rvagent ⇄ metaharness +| Seam | Mechanism | Status | +|---|---|---| +| Host adapter | `host-rvagent` (10th host): `generateConfig(spec)` emits rvagent config + skills + policy | new (adapter interface is trivial) | +| Evolution | headless JSON contract as Darwin solver; policy-genome env/file seam; flywheel promotion gates | pattern exists (`SWE_POLICY_SYSTEM`, `claude -p` precedent) | +| Governance | Ed25519 witness formats aligned (kernel witness ↔ rvagent witness chain ↔ RVF witness per ADR-106 Phase 4) | partial | +| Kernel reuse | mirror hooks/claims/dispatch types (ADR-256: concepts, not required dependency) | new | + +Both directions preserve the ADR-150 invariant: **every integration is +optional and degrades gracefully**. rvagent must run standalone; ruflo must +run without rvagent; metaharness must stay removable. + +## 10. What "best SOTA harness" concretely means here + +1. **Beat the Claw-SWE-Bench harness spread**: rvagent + a fixed open model + scores within the Hermes/OpenClaw band (≥70% on the 350-instance set), + validated with metaharness's conformant apparatus (Wilson CIs, gold + Docker eval, packaged submissions — no self-graded claims). +2. **Beat Hermes where it's weak**: native-speed kernel (startup, token + throughput, parallel tools), in-browser WASM deployment, federated + multi-agent execution with cryptographic identity/budgets (A2A), and + coherence-weighted compaction (ADR-252) instead of char windows. +3. **Match Hermes where it's strong**: the closed learning loop — measured + as ≥30% token reduction on repeat-task suites after 20+ synthesized + skills, with Darwin-evolved skill populations as the upgrade Hermes + only gets offline. diff --git a/docs/research/rvagent-hermes-harness/03-roadmap.md b/docs/research/rvagent-hermes-harness/03-roadmap.md new file mode 100644 index 000000000..0b7488250 --- /dev/null +++ b/docs/research/rvagent-hermes-harness/03-roadmap.md @@ -0,0 +1,146 @@ +# Implementation Roadmap + +Ordering rule: nothing in P1+ lands until P0 is green, because every +feature in P1+ is blocked by the type unification and the tools-to-model +fix. Each phase has a falsifiable exit gate. + +--- + +## Phase 0 — Foundation Repair (rvagent-core/backends/middleware) + +The loop must become a real agent before it can become a great harness. + +1. **Unify types.** `rvagent-core`'s `AgentState`/`Message`/`Tool`/ + `TodoItem`/`RunnableConfig` become canonical; delete the middleware and + subagents duplicates; add conversion impls only at the WASM/serde + boundary. (Blocks everything.) +2. **Send tools to the model.** `ChatModel::complete(&[Message], + &[ToolDefinition])`; `tools` field on Anthropic `ApiRequest`; + `functionDeclarations` for Gemini; parse `tool_use` into the canonical + `ToolCall`. +3. **Async middleware + wiring.** `wrap_model_call` → async; + `build_default_pipeline()` constructed and installed in CLI, ACP, and + the future A2A runner. Delete the CLI's duplicate `LocalFsBackend` + grep/glob/execute in favor of `rvagent-backends`. +4. **Loop correctness.** Tool errors feed back as tool results (no `?` + abort); real parallel execution via `parallel_execute`; per-turn `Usage` + flows into `BudgetEnforcer` and `Metrics`. +5. **Honest docs.** Remove unsupported claims (parallel exec, HNSW speedups, + streaming) from README/architecture.md until true; docs list all 10 + crates. + +**Exit gate:** `rvagent run "create and run a failing test, then fix it"` +completes end-to-end against the live Anthropic API with ≥2 tool round +trips, budget accounting, and a green `cargo test` across the workspace. + +## Phase 1 — Hermes-Class Loop Mechanics + +1. **Event-streaming loop** (`AgentEvent` stream; ADR-139 direction) with + SSE streaming backends and incremental TUI render. +2. **Cache-first prompt builder** (stable/context/volatile tiers, + `cache_control` emission, cache-hit-rate metric). +3. **Compaction with lineage** (sentinel pre-limit, middle-turn + summarization, durable-fact extraction under hard budget, parent/child + session chains, real tokenizer). +4. **Full-state checkpoints** (`SegmentType::Checkpoint`), resume/fork. +5. **Hooks + permissions** (10-event taxonomy with Allow/Deny/Ask/Defer + merge mirrored from metaharness kernel; HITL wired to + `PermissionRequest` events; permission modes; per-tool rules). +6. **Real subagents** (JoinSet spawn, semaphore, CoW fork/merge, CRDT join, + per-subagent model override). +7. **Headless contract** (`-p --output-format stream-json + --max-budget-usd --policy-file`). +8. **Docker sandbox backend** (read-only root, dropped caps) alongside the + local shell. + +**Exit gate:** rvagent completes a 20-instance SWE-bench-Lite smoke slice +via the headless contract inside metaharness's runner, with measured cache +hit rate >85% and zero loop aborts on tool errors. + +## Phase 2 — Memory & The Learning Loop + +1. **Kill fake HNSW**; episodic memory = session store + FTS, with real + `ruvector` embeddings behind the feature gate; optional shared + `.swarm/memory.db` (AgentDB bindings, ADR-323 provenance, WAL guard). +2. **Prompt-memory write-back** with hard char budget + periodic curation + nudges. +3. **Skill synthesis**: trajectory triggers (5+ tool calls / error recovery + / user correction) → SKILL.md emission via auxiliary model; progressive + disclosure; `skill_manage` tool with string-patch default. (Closes + ruflo's #1 roadmap gap; skills interop with Claude Code/Codex via the + existing `skills_bridge`.) +4. **SONA on the default path** (feature-gated on, trajectories from the + witness chain feeding ReasoningBank; ADR-271 EwcConfig as genome). +5. **ADR-252 coherence-weighted compaction** as the summarization upgrade. + +**Exit gate:** on a 3×-repeated task suite, the skilled agent shows ≥30% +token reduction vs a fresh instance (Hermes's measured ~40% is the bar). + +## Phase 3 — Ecosystem Integration + +1. **ruflo:** NAPI package (`@ruvector/rvagent-native`) as a sibling to + `rvagent-wasm` behind `agent-wasm.ts`; `wasm_agent_compose` tool + allowlist honored; Capability Brain registration; hook-handler + subprocess shim; bandit outcome reporting. ruflo's worker daemon gains + an `rvagent` executor option beside headless Claude. +2. **metaharness:** `packages/host-rvagent` adapter; rvagent registered as + a Darwin solver backend via the headless contract; witness format + alignment (kernel ↔ rvagent ↔ RVF per ADR-106 Phase 4). +3. **A2A goes live:** real `TaskRunner` backed by the loop replaces + `InMemoryRunner`; `rvagent a2a serve` advertises real skills; federated + rvagent↔rvagent task delegation demo with budgets + recursion guards. +4. **Eval in CI:** TBLite-style smoke slice + frozen-eval gate on PRs + (metaharness harness-loop gates: QUALIFY→BENCHMARK→VERIFY→CANARY). + +**Exit gate:** one command (`npx ruflo swarm ... --executor rvagent` or +equivalent) runs a swarm where rvagent instances execute, ruflo +coordinates/remembers, and the run emits signed witness manifests. + +## Phase 4 — Evolution & SOTA Claims + +1. **Policy-genome evolution**: Darwin mutates policy text, temperature + schedules, compaction thresholds, tool configs; flywheel promotion with + held-out benchmarks and signed champion manifests. +2. **Skill-population evolution** (GEPA-style over execution traces). +3. **weight-eft distillation** of gold trajectories into cheap-tier LoRA + (ADR-271's SFT/DPO on/off-policy recipe) → escalate to frontier models + less often; cost-Pareto tracking. +4. **Publish**: Claw-SWE-Bench-style fixed-model comparison vs Hermes/ + OpenClaw, Terminal-Bench 2.0, with conformant packaged submissions and + Wilson CIs. No claim ships without the metaharness honesty apparatus + (the ecosystem's retraction discipline is a feature — keep it). + +**Exit gate:** rvagent within the Hermes/OpenClaw band (≥70%) on the +350-instance fixed-model set, and demonstrably ahead on ≥2 of: startup +latency, token throughput, parallel-tool wall clock, federated multi-agent +tasks, in-browser deployment. + +--- + +## Proposed ADRs + +| ADR | Title | Decides | +|---|---|---| +| A | rvAgent Harness Core Repair (supersedes parts of ADR-095/097) | canonical types, tools-to-model, async middleware, loop recovery semantics | +| B | rvAgent Event-Streaming Loop & Headless Contract (implements ADR-139) | `AgentEvent` taxonomy, stream-json format, budget/policy flags | +| C | rvAgent Cache-First Prompts, Compaction & Lineage | prompt tiers, cache_control, sentinel compaction, session lineage | +| D | rvAgent Skills & Learning Loop | synthesis triggers, SKILL.md interop, SONA/witness wiring, char budgets | +| E | rvAgent ⇄ ruflo Execution Seam | NAPI sibling, memory co-tenancy rules (ADR-323/WAL), hook shim, Capability Brain entry | +| F | rvAgent ⇄ metaharness Evolution Seam | host adapter, Darwin solver contract, policy genome surfaces, witness alignment | + +## Risks + +- **Scope gravity.** The ecosystem's pattern (documented in all three + audits) is protocol/ADR surface outrunning the loop. Mitigation: P0/P1 + exit gates are executable, not documentary; no new crate until the gate + passes. +- **Type unification churn** touches every crate at once. Mitigation: land + as one PR series with the workspace green at each step; WASM API kept + stable via serde boundary. +- **Benchmark credibility.** Any SOTA claim without the gold-eval + + Wilson-CI + packaged-submission discipline damages the whole ecosystem's + (currently strong) honesty record. Mitigation: Phase 4 gates are + metaharness-conformant by construction. +- **Optionality invariants** (ADR-150/256) must hold in both directions or + the three projects become a distributed monolith. Mitigation: CI runs + each project's `--ignore-optional` path with rvagent absent/present. diff --git a/docs/research/rvagent-hermes-harness/04-sota-landscape.md b/docs/research/rvagent-hermes-harness/04-sota-landscape.md new file mode 100644 index 000000000..119802009 --- /dev/null +++ b/docs/research/rvagent-hermes-harness/04-sota-landscape.md @@ -0,0 +1,434 @@ +# SOTA Landscape (August 2026) — and what it changes + +Research sweep across five areas: benchmarks and credible-claim criteria, +harness technique literature with effect sizes, competing harness +architectures, long-horizon context methods, and self-improving harnesses. + +This document exists to correct the roadmap, not to decorate it. Where the +evidence contradicts `03-roadmap.md`, the contradiction is stated plainly and +the roadmap change is specified. + +**Evidence discipline.** Most 2026 material is single-source arXiv preprints +or vendor blogs. Individual percentage-point figures are hypotheses to +re-measure in our own harness, not constants. The findings below are ranked by +corroboration, and single-source claims are marked. Four findings are +multiply-corroborated and safe to build on: + +1. Harness choice moves results 10–30 points at fixed model. +2. Harness value scales *inversely* with model strength. +3. Context **management** beats context **size**. +4. Parallel writer agents fail on coding tasks specifically. + +--- + +## 1. The headline: most of the delta is reliability, not intelligence + +The cleanest ablation available (Claw-SWE-Bench, Jun 2026) strips a harness to +bare model-emits-diff and rebuilds it: **19.1% → 73.4%, +54.3 points**. Nearly +all of that is patch-apply failures going from 69.1% to under 1.5%. + +That reframes the entire project. The biggest measured wins in every ablation +come from eliminating mechanical failure modes — patches that don't apply, +tool-call loops, context rot, flaky tests — not from smarter reasoning. + +**This is the strongest argument for building the harness in Rust**, and it is +not the argument we have been making. The case is not "Rust is fast." It is +that the failure modes which dominate these ablations can be made +*type-unrepresentable* rather than merely rare. + +Two structural findings that should shape sequencing: + +- **Harness variance is larger for weaker models.** Same harness set, GLM-5.1: + 12.5-point spread. Qwen-3.6-flash: 27.4-point spread. If we target frontier + models only, we should expect roughly *half* the harness ROI the literature + reports. +- **The harness is now a disclosed experimental variable.** Two 2026 papers + (arXiv 2605.23950, 2607.04528) argue benchmark results are substantially + determined by undisclosed harness choices, and that harnesses induce + systematically different agent *beliefs* on logically equivalent tasks. Any + claim we publish must disclose the harness or it is not a claim. + +--- + +## 2. Competitive position: the Rust field is crowded at the top + +The premise that a Rust harness is differentiating is **false as of July 2026**. +Three of the major harnesses are already Rust: + +| Harness | Scale | License | Notes | +|---|---|---|---| +| **Codex CLI** (OpenAI) | ~70–80 crates | Apache 2.0 | Rewritten *from* TypeScript for single-binary distribution + native sandbox bindings | +| **Grok Build** (xAI) | ~844k LOC | Apache 2.0 | Open-sourced **2026-07-15**. ACP, checkpoints, TUI | +| **Goose** (Block) | — | Apache 2.0 | MCP-native, subagents via `Agent::new()` | + +OpenAI's stated reasons for the Rust rewrite are exactly the ones in our +positioning: zero-dependency install (Node 22+ blocked enterprise/air-gapped), +no GC pauses in long-running processes, memory-safe sandbox bindings without +FFI shims. That ground is taken. + +**Two positions remain genuinely open:** + +1. **No Rust harness is usable as a library.** Codex's own `AGENTS.md` + discourages extending `codex-core`; Grok Build has issues and PRs + *disabled* (source-visible, not open governance); Goose is app-first; and + Anthropic's Agent SDK is Python/TypeScript only — the docs instruct other + languages to shell out to the CLI. A stable, semver'd, embeddable harness + crate with open governance is unoccupied. +2. **Deterministic replay is a named gap** in the Rust agent ecosystem and + thin everywhere. Our append-only witness/segment infrastructure is already + most of the way there. + +**Honesty constraint on the replay claim.** Even at temperature 0, hosted +inference is not reproducible — floating-point non-associativity and +batch-size-dependent kernels produce run-to-run variation. The defensible +claim is *replay of the harness*, with **action-match rate** reported, not +byte-exact reproduction of the model. + +**Convergence warning.** Grok Build's tool implementations are documented ports +— `apply_patch`, `grep_files`, `list_dir`, `read_file` from Codex; `bash`, +`edit`, `glob`, `grep`, `read`, `skill`, `todowrite`, `write` from opencode. A +frontier lab with a million lines of Rust chose to port the tool surface rather +than design one. **Tool-surface novelty is not available as a differentiator.** + +--- + +## 3. Ranked technique priorities + +Ordered by measured effect per unit of engineering effort. Tier 1 items are +days of work each and carry the largest deltas in the literature. + +### Tier 1 — do first + +| # | Technique | Effect | Status in rvagent | +|---|---|---|---| +| 1 | Reliable patch application (real workspace, file-based edits, git diff extraction, verify-after-write) | **+54.3 pp** | Partial — tools write real files; no verify-after-write, no git extraction | +| 2 | Observation-window management (keep last N tool outputs in full, elide older) | +3 pp, prevents long-run collapse | **Missing** | +| 3 | Loop/stuck detection (3-strike tool-call fingerprint → inject warning, skip) | Removes most common catastrophic failure | **Missing** | +| 4 | Actionable structured tool errors + response size caps (~25k tokens) | Part of the reliability delta | Partial — errors feed back (P0.4), uncapped | +| 5 | Tool surface held to 8–15 tools | Avoids −16 to −23 pt routing collapse | **OK** — 9 builtins. Protect this. | +| 6 | Environment bootstrap injection (cwd, tree, toolchain, test command, current check status) | Stanford Meta-Harness @ 76.4% TB2.0 | **Missing** | +| 7 | Persist thinking blocks across tool calls | +2.2 pp coding | **Missing** | + +Item 3 deserves emphasis: raising max-iteration counts does **not** fix loops, +it makes them more expensive. Our `max_iterations: 100` is a cost cap, not a +loop guard. + +### Tier 2 — clear ROI, ~1–2 weeks each + +- **`str_replace` edit tool with fuzzy-failure diagnostics + `cargo check` lint + gate.** +10 to +23 pp for mid-tier models, ~0–5 pp frontier. The specific + ergonomics matter: in the Qwen reproduction only the SWE-agent + `str_replace_editor` flavor moved the number; a different `edit`/`write_file` + pair gave *zero* improvement. Offer `write_file` alongside (+2.1 pp, −17.9% + cost). Skip unified diff — apply failures dominate. +- **Reproduction-test-first loop** (+8 to +13% relative). Critical caveat: + adding "write tests first" to the prompt *without* targeted context made + regressions **worse** (6.08% → 9.94%). The gain is in *executing* + reproduction tests, not the TDD ritual. +- **Summarized grep/glob** — return paths + match counts, require a second call + for contents. `ripgrep` as a library crate. +- **Fresh-context reviewer subagent** — ~2 bugs/PR, 58% severe, in Cognition + production. Counterintuitively, reviewers perform **better with no shared + context**: shorter context, less rot, deeper analysis. +- **Read-only context-gathering subagent** returning a summary string. +2.1 pp, + −34.5% main-agent input tokens. Use a *cheap* model here — a frontier model + in this slot gave +0.4 pp at 5.8× cost. + +### Tier 3 — real but expensive or conditional + +Rubric-guided compaction (the rubric is load-bearing, not the tool); +best-of-N with test-based filtering then deterministic patch fusion (+7 to ++9.4 pp at N≈8, ≈8× cost); shadow-git checkpointing; coordinator delegation +(only after single-agent is solid). + +### Skip list — evidence says no + +- **Parallel writer swarms** for coding +- **Few-shot demos / explicit CoT instructions** for reasoning models — zero-shot ≥ few-shot; exemplars can contradict native reasoning +- **Ungrounded self-reflection loops** — can degrade already-correct answers; only execution-grounded critique works +- **Unified-diff edit format** +- **Semantic/embedding code index in v1** — vendor-only evidence, high maintenance, stale-by-construction on a repo the agent is editing +- **Elaborate system-prompt frameworks** — the "20–30% improvement" claims have no methodology; keep rulebooks under ~60 lines +- **Context windows beyond ~128k** — sweeps plateau ~114k; documented ceiling ~96–112k +- **Learned/RL-trained components** (adaptive edit-format selectors, RL compaction) — the hand-written 80% is available for 5% of the effort + +--- + +## 4. Long-horizon execution: what actually breaks + +Ranked by when it bites in a multi-hour run: + +1. **Wasted-context accumulation → attention dilution.** Earliest, most + universal, invisible. Unaided coding agents waste ~1-in-3 file reads. +2. **History error accumulation.** The largest single failure driver — + process-level failures are **72.5%** of long-horizon failures (HORIZON, + arXiv 2604.11978). Errors compound *between* steps, not within them. + Non-linear: sharp collapse past a domain-specific threshold. +3. **Compaction-induced information loss** — self-inflicted, caused by the + mitigation for #1. +4. **Goal/identity drift** — *downstream* of 1–3, not an independent disease. +5. **Hallucinated state** — phantom invoices, fabricated history. +6. **Memory staleness / negative transfer.** + +The 2026 evidence does **not** support treating goal drift as the primary +problem. It is the observable end-stage of context and error problems. + +### The compaction finding that contradicts our plan + +**Simple observation masking matches or beats LLM summarization at roughly +half the cost** (JetBrains, 250-turn SWE-bench trajectories, NeurIPS 2025 +workshop). Mechanism: LLM summarization **extended trajectories 13–15%** by +destroying natural stopping signals — the agent loses the cue that it already +finished something. + +Better still, **Addressable Recall Compaction** (arXiv 2607.25066): mask the +observation but leave an ID the agent can dereference on demand. Beat +full-context, sliding window, LLM summary, structured state, *and* RAG memory +(NIAH 99.40% vs 88.12% best baseline). + +And the single highest-leverage technique found anywhere in this sweep: +**programmatic tool calling** — the model writes code that orchestrates tools, +so intermediate outputs never enter context. On LOCA-bench it was the **only** +strategy positive across all four models tested (+6 to +13.3 points). It is +absent from our roadmap entirely. + +Two further hard requirements: + +- **Re-inject invariants verbatim after every compaction; never summarize + them.** "Governance Decay" (arXiv 2606.22528) shows safety constraints and + system-prompt instructions erode through successive compaction cycles with + no failure signal. This is architectural, not jailbreaking. +- **Capability-gate context features.** On LOCA-bench a memory tool made a + weaker model *worse* (10.7% → 8.0%), and context-budget awareness made it + much worse (10.7% → 4.0%), while both helped stronger models. Multi-model + support makes per-tier feature flags mandatory. + +--- + +## 5. Self-improvement: the literature turned against us + +RuVector already has ReasoningBank-style trajectory learning and a shared +brain. The 2026 evidence on this class of system is substantially negative. + +**The case for** is real: ReasoningBank reports +4.6 to +8.3 points on WebArena +across three backbones (ICLR 2026); ACE reports +10.6% on agents. Distilling +from *failures* as well as successes is a genuine contribution. + +**The case against is now stronger:** + +- **The gains are confound-sized.** MemDelta (arXiv 2606.29914) shows swapping + the embedding model alone shifts accuracy ±6.2pp — comparable to + ReasoningBank's entire headline gain. In the same work, **agent self-memory + (42%) underperformed plain retrieval (47%)**, and one system reached parity + with cloud RAG at **50× the cost**. +- **Memory utility is an inverted U.** "Useful Memories Become Faulty" + (arXiv 2605.12978): utility rises, then degrades *below* the no-memory + baseline. GPT-5.4 failed **54% of previously-solved ARC-AGI problems** when + using consolidated memory. Episodic-only management **doubled** accuracy vs + forced consolidation — **the consolidation step is the bug, not the storage.** +- **Benchmark-local.** MemoryArena specifically names ReasoningBank's + procedural memory as performing poorly on interdependent multi-session tasks + — the setting closest to real work. +- **No automatic self-evolution method sustains positive gain across settings** + (EvoAgentBench, arXiv 2607.05202). *Curated* ability content transfers + across model families; *automatic extraction* is the failure point. +- **Greedy acceptance is uncontrolled multiple testing.** PACE + (arXiv 2606.08106): "keep it if the score improved" committed **30–42% false + edits**, and made 13–21 spurious modifications when *no true gains existed*, + degrading one agent by 4.9 points. + +Meanwhile **Live-SWE-agent reaches 79.2% on SWE-bench Verified with zero +persistent memory** — on-the-fly tool synthesis from the current trajectory, +discarded after use. It sidesteps every failure mode above, which is not a +coincidence. + +**Verdict:** trajectory-learning memory is a nice-to-have with fragile upside, +not a differentiator. It must ship as a **gated optimization with a measured +contribution**, never on the default path unmeasured. + +Minimum viable discipline if we keep it: + +- Two-tier storage: **episodic immutable**, distilled artifacts *derived* with + pointers back to source episodes. +- Gated, delta-only consolidation with hard length caps. Never end-to-end + rewrites (ACE documents a single step collapsing 18,282 tokens → 122). +- Trust tiers by verifier: programmatic (tests/typecheck) = active; + LLM-judge-only = quarantined; **derived from untrusted content = never + allowed to influence tool-permission or destructive-action decisions.** +- **PACE-style anytime-valid sequential testing for promotion**, never greedy. +- Retrieved memories framed as **"references, not rules"** — nearly free, and + drops attack success 20.6% → 13.1%. +- **A permanent memory-off control arm.** Not a one-time ablation — without a + live control we cannot detect the inverted-U crossover. + +Primary metric: **paired net lift on the same instances**, with the regression +metric that matters most being *rate of previously-solved tasks that now fail +with memory on*. + +**Shared-brain caveat.** Cross-agent shared memory multiplies the poisoning +blast radius. Measured memory-poisoning work reports ~50% attack success and +~41% *relapse* success, with contextual assimilation as the primary vector — +poisoned entries work best when they look like ordinary preferences. The +shared brain needs the strongest gates in the system, not the weakest. + +--- + +## 6. Verifier quality is the binding constraint + +Any self-improving loop is only as good as its verdict signal, and the 2026 +evidence on LLM-as-judge is poor: + +- A judge surfaced **under 25%** of human-confirmed systematic problems, and + **flagged zero** issues in a batch where humans confirmed 23 distinct + defects. Implied **3–6× undercount** of true defect rates. +- Blind-spot structure is systematic: catches **turn-local** problems, severely + blind to **cross-turn state**. +- Calibration drifts: one judge at 0.91 agreement shifted four points after a + model update — the signal stopped meaning what it meant. + +**Design rule:** programmatic and execution-grounded signals first (tests, type +checks, schema validation, invariants). LLM judge as a *secondary, quarantined* +signal only. Treat the judge as a **regression floor, never a promotion +authority**. Re-anchor against a rolling human-labeled sample after every model +update. + +--- + +## 7. Benchmark hygiene + +**SWE-bench Verified is saturated and unreliable as a claim target.** + +- UTBoost (ACL 2025): resolve rates inflated **~6.4 pp** by weak tests; **1 in + 5** "solved" patches semantically incorrect; augmenting tests changed + leaderboard ranks in **24.4%** of Verified submissions. +- OpenAI's own audit found **59.4%** of the hardest unsolved Verified problems + had flawed test cases, and OpenAI stopped reporting Verified in early 2026. + +This directly threatens our Phase 4 exit gate (`≥70% on a 350-instance +fixed-model set`) — the gate is stated against a saturated benchmark whose +noise band exceeds the effect sizes we would be claiming. + +**Consequence for our own A/B testing:** before trusting any measurement of a +harness change, verify test determinism (run F2P/P2P repeatedly under gold and +base patches). A 3-point "improvement" sits inside the flaky-test noise band. +Invest in eval-loop speed — validation that cannot run in minutes will not get +run. + +--- + +## 8. Roadmap corrections + +Specific, and each traceable to a finding above. + +### Phase 1 — reorder and add + +**Add a new Phase 1a "reliability floor" ahead of everything else**, comprising +Tier 1 items 1–7 (§3). These are days of work with the largest measured +deltas, and five of seven are currently missing. The present Phase 1 leads with +event-streaming, cache-first prompts, and compaction — all defensible, none of +them the biggest lever. + +**Add programmatic tool calling** (§4). Highest-leverage single technique in +the sweep; absent from the plan. + +### Phase 1.3 / Phase 2.5 — change the compaction bet + +Both currently bet on summarization (middle-turn summarization; ADR-252 +coherence-weighted compaction). Evidence says summarization is close to the +*worst* measured option and inflates trajectories 13–15%. + +**Replace with:** observation masking as the default, plus addressable recall +(masked entries keep a dereferenceable ID). Keep summarization as a fallback +behind a rubric. Add mandatory verbatim invariant re-injection post-compaction. + +Note this also affects the shipped default pipeline, which currently includes a +`summarization` middleware. + +### Phase 1.6 — subagents: drop the CRDT merge + +Currently "CoW fork/merge, CRDT join." This is parallel-writer architecture, +which is the one multi-agent pattern with strong negative evidence for coding. +Cognition's 2026 revision — after a year of production data — is **one writer, +augmented by auxiliary intelligence; never parallel writes.** + +**Replace with** the two patterns that have production evidence: a +**fresh-context reviewer** (no shared context — it performs *better* without +it) and a **read-only context-gatherer** returning a summary string. Model the +subagent boundary as *a tool that spawns an isolated context and returns a +String*, not as peers with a message bus. That buys nearly all demonstrated +upside at a fraction of the complexity. + +**Amended (ADR-278 §7).** The reviewer is **gated, not adopted**. metaharness +ADR-226 is a gold-scored null on a closely related design. It gave its advisor +the *full transcript* where this reviewer sees *only the diff*, so it does not +refute the pattern — but it is the specific null the reviewer must beat before +reaching the default path. The gatherer is unaffected and corroborated. + +### Phase 2 — gate the learning loop, and shift memory → policy + +"SONA on the default path" contradicts §5. Move behind a feature gate with the +measurement apparatus (paired lift, previously-solved regression rate, control +arm) as the *precondition* for enabling it, not a follow-up. + +The existing exit gate (≥30% token reduction on a repeated task suite) is +well-formed — keep it, and add the control arm. + +**Update (ADR-278).** The promotion apparatus does not need building: +`@metaharness/flywheel` already implements a frozen fingerprinted conjunctive +gate, holdout **plus** a never-optimized-against anchor, Ed25519 receipts, +independent replay verification, and a compounding lineage DAG. Adopt it. + +More consequentially, metaharness's own measurements reframe the target. +Self-learning splits into two objects with opposite evidence: **policy text** +(GEPA-style; positive) and **episodic memory** (ReasoningBank/SONA; negative, +per §5). RuVector's weight currently sits on the memory side. Move new effort to +policy evolution. + +Two internal nulls to respect: + +- **ADR-226** — a read-only frontier advisor produced *zero* marginal + gold-scored resolves at **5.4× cost** while genuinely firing (33 advisories, + 3 vetoes). Independently corroborates this document's +0.4 pp / 5.8× figure. +- **ADR-236** — the flywheel mechanism was proven end-to-end on real SWE-bench + and still produced no compounding lift, because the base solver was too weak. + **A promotion engine cannot rescue an unreliable loop**, which confirms the + reliability-floor-first ordering above. + +### Phase 3 — MCP spec migration is now urgent + +MCP **2026-07-28** landed days ago and is breaking: protocol-level sessions and +`Mcp-Session-Id` removed, `_meta` on every request, `server/discover` added, +Tasks extension replaces long-running `tools/call`, and Roots/Sampling/Logging +deprecated. Building against the 2025-11-25 shape means a rewrite within +months. The official Rust SDK (`rmcp`) already implements the new spec. + +**Also add ACP as a first-class target.** It went from Zed-only to headline +feature of Zed 1.0, built into JetBrains, a public registry, and 25+ agents — +and its reference implementation is Rust. We have an `rvagent-acp` crate +already; this is closer than it looks. + +### Phase 4 — re-ground the SOTA claim + +Drop the SWE-bench-Verified-based gate. Retarget to non-saturated benchmarks +and to the axes where we can actually win. The differentiator claims should be +**library API + deterministic replay** (§2), not speed. + +Report action-match rate for replay, not reproducibility. Keep the honesty +apparatus — given §7, retraction discipline is the feature. + +--- + +## 9. Open items + +- ~~Benchmark/leaderboard sweep~~ **Resolved 2026-08-01.** Headline: **SWE-bench + Pro was retracted by OpenAI on 2026-07-08** (27.4% of public tasks flagged + broken automatically, 34.1% by human reviewers) — the obvious successor to + Verified is also gone. Terminal-Bench 2.1 is the only credible harness board; + no operating leaderboard reports cost, which makes a cost-normalized Pareto + uncontested ground. Full gate in ADR-277 §7. +- `codex-rs` internals in §2 come from a third-party architecture writeup, not + the repo. Worth reading `codex-rs/core/src/` directly before copying the + `Op`/`EventMsg` design. +- Grok Build's governance may change. If xAI opens PRs, the "no community Rust + harness" gap closes fast — the moat must be the library API and replay, not + merely that a Rust harness exists. diff --git a/docs/research/rvagent-hermes-harness/README.md b/docs/research/rvagent-hermes-harness/README.md new file mode 100644 index 000000000..fc9045815 --- /dev/null +++ b/docs/research/rvagent-hermes-harness/README.md @@ -0,0 +1,93 @@ +# rvAgent as a Hermes-Class Harness — Research & Architecture Proposal + +**Date:** 2026-08-01 +**Status:** Research complete, implementation proposed +**Related:** ADR-093..107 (rvAgent), ADR-139 (Claude Code intelligence), ADR-150 (ruflo metaharness surfaces), ADR-159 (A2A), ADR-211/252 (agent memory), ADR-256 (metaharness concepts), ADR-260/266/271 (Darwin evolution) + +## The Question + +> How can we implement rvagent more like the Hermes harness, integrated with +> ruvnet/metaharness and ruvnet/ruflo, to create the best SOTA harness in the world? + +## The Answer in One Paragraph + +The Hermes agent (NousResearch/hermes-agent, MIT, ~175K stars) proved two things: +(1) **the harness is worth up to 27 points of SWE-bench pass@1** — more than most +model upgrades — and (2) the winning differentiator is not one-shot benchmark +score but a **closed learning loop**: trajectories distilled into self-patching +skills, evolved offline by GEPA, yielding ~40% faster/cheaper repeat tasks. +RuVector already believes this thesis — ADR-260/266/271's "freeze the model, +evolve the harness" is the same idea Hermes ships. What no one ships yet is a +**native-speed, memory-safe, WASM-portable harness kernel with an evolutionary +optimizer and a swarm coordination plane attached**. That is exactly the seam +where rvagent (Rust execution kernel) + metaharness (Darwin evolution + eval +apparatus + governance) + ruflo (memory substrate, hooks bus, MCP surface, +model routing, swarm) combine into something none of the incumbent harnesses +— Hermes included — can match. The catch: rvagent's core loop is currently +scaffold-grade (it cannot even send tool schemas to the model), so the path +starts with foundation repair, not features. + +## Ecosystem Fit (who does what) + +``` +┌────────────────────────────────────────────────────────────────────┐ +│ metaharness — the harness FACTORY & EVOLVER │ +│ mints harnesses (9 hosts incl. Hermes), Darwin/flywheel evolves │ +│ policy genomes, Ed25519 witness governance, SWE/Terminal-Bench │ +│ apparatus. Generated harnesses ship NO agent loop today. │ +└───────────────▲────────────────────────────────────────────────────┘ + │ evolves genome / benchmarks / signs +┌───────────────┴────────────────────────────────────────────────────┐ +│ rvagent — the EXECUTION KERNEL (this proposal) │ +│ Rust agent loop: tools, streaming, compaction+lineage, skills, │ +│ subagents, checkpoints, hooks, budget, witness. Ships as native │ +│ CLI + NAPI + WASM. Fills the loop-shaped hole both siblings │ +│ currently outsource to Claude Code. │ +└───────────────▲────────────────────────────────────────────────────┘ + │ memory / hooks / routing / tool surface via MCP +┌───────────────┴────────────────────────────────────────────────────┐ +│ ruflo — the COORDINATION PLANE │ +│ .swarm/memory.db + hnsw.index + .rvf substrate (ADR-323 │ +│ provenance), 305 MCP tools, hooks lifecycle bus, Thompson-bandit │ +│ model router, swarm/hive-mind. Already hosts rvagent via │ +│ @ruvector/rvagent-wasm (27 wasm_agent_* tools). │ +└────────────────────────────────────────────────────────────────────┘ +``` + +## Documents + +| File | Contents | +|---|---| +| [01-findings.md](01-findings.md) | What we found: Hermes architecture & benchmark evidence; rvagent current-state audit (4 blocking defects); metaharness & ruflo capability maps | +| [02-target-architecture.md](02-target-architecture.md) | The Hermes-class rvagent design: loop, prompt tiers, memory layers, skills, subagents, and the exact integration seams into metaharness and ruflo | +| [03-roadmap.md](03-roadmap.md) | Phased implementation plan (P0 foundations → P4 evolution/SOTA), success gates, proposed ADRs | + +## Headline Findings + +1. **rvagent's protocol layers are production-grade; its loop is not.** + `rvagent-a2a` (signed cards, budgets, circuit breakers, recursion guards), + `rvagent-mcp`, and the security primitives are real and well-tested. But the + agent loop never sends tool schemas to the model (the `Tools` node is + unreachable in production), the 19-module middleware pipeline is never wired + into the CLI, subagents are stubs, streaming does not exist, and three + incompatible `AgentState`/`Message`/`Tool` type systems block assembly. + +2. **Hermes's edge is structural, and every piece has a RuVector-native analog.** + Cache-first tiered prompts → unwired `PromptCachingMiddleware`; layered + memory → AGENTS.md + ruflo's memory substrate + ADR-211/252; skills as + procedural memory → `SkillsMiddleware` + `skills_bridge`; trajectory→skill + distillation → SONA/ReasoningBank + witness chains; GEPA offline evolution → + metaharness Darwin/flywheel. The parts exist; nothing is connected. + +3. **Both siblings have a loop-shaped hole rvagent should fill.** metaharness + generates harness *configuration* for nine hosts but no runtime loop; ruflo + explicitly delegates multi-turn execution to Claude Code (`CLAUDE.md:908`) + and its only in-house turn loop is… rvagent's WASM build. ruflo's own + roadmap names "skill synthesis vs Hermes-class agents" its top gap. + +4. **The benchmark literature says invest here.** Harness choice swings + SWE-bench pass@1 by 12.5–27.4 points (Claw-SWE-Bench); "execution + alignment" and few high-fidelity tools beat sprawling toolsets + (Harness-Bench); structure transfers across models, prompt wording doesn't + (Agentic Harness Engineering). A Rust kernel + Darwin evolution attacks all + three levers at once. diff --git a/docs/rvagent/architecture.md b/docs/rvagent/architecture.md index 7c93770ff..333f107ef 100644 --- a/docs/rvagent/architecture.md +++ b/docs/rvagent/architecture.md @@ -1,6 +1,11 @@ # rvAgent Architecture -This document describes the internal architecture of the rvAgent crate family, covering the crate dependency graph, agent lifecycle, middleware pipeline, backend protocol hierarchy, security model, and performance characteristics. +This document describes the internal architecture of the rvAgent crate family (10 crates, including `rvagent-mcp` and `rvagent-a2a`), covering the crate dependency graph, agent lifecycle, middleware pipeline, backend protocol hierarchy, security model, and performance characteristics. + +> Current gaps (tracked in `docs/research/rvagent-hermes-harness/03-roadmap.md`): +> SSE streaming is not implemented (providers fall back to non-streaming +> completion); subagent orchestration spawning is stubbed; the `hnsw` +> middleware uses a hash-based embedding placeholder, not semantic search. ## Crate Dependency Graph @@ -28,6 +33,14 @@ rvagent-acp |-- rvagent-tools |-- rvagent-subagents | +rvagent-mcp +|-- rvagent-core +|-- rvagent-middleware (skills bridge) +| +rvagent-a2a +|-- rvagent-core +|-- rvagent-middleware +| rvagent-wasm |-- (standalone, no workspace deps except serde/wasm-bindgen) ``` diff --git a/scripts/rvagent-flywheel-evaluator.mjs b/scripts/rvagent-flywheel-evaluator.mjs new file mode 100644 index 000000000..2fc5175fb --- /dev/null +++ b/scripts/rvagent-flywheel-evaluator.mjs @@ -0,0 +1,115 @@ +#!/usr/bin/env node +// rvAgent's Evaluator for @metaharness/flywheel (ADR-278). +// +// The flywheel is deliberately host-agnostic — it knows only candidates, +// scores, gates, receipts, and lineage. Everything rvAgent-specific enters +// through this one seam, which is also the trust boundary: the four Score axes +// are where all host meaning lands, and a dishonest projection defeats every +// downstream guarantee the gate provides. +// +// Usage as a library: +// import { makeRvagentEvaluator } from './rvagent-flywheel-evaluator.mjs'; +// const evaluator = makeRvagentEvaluator({ runItem }); +// +// `runItem(policy, item) -> RunOutcome` is injected so this file stays +// testable without spawning real agent runs. + +/** Cost-per-win when a policy won nothing. + * + * Must match rvagent_core::policy::COST_PER_WIN_NO_WINS. NOT Infinity: JSON has + * no infinity, so it serializes to null, and the gate's + * `candidate.costPerWin > baseline.costPerWin` reads `null > n` as false — + * meaning a policy that won nothing would silently pass the cost clause. + */ +export const COST_PER_WIN_NO_WINS = Number.MAX_VALUE; + +/** + * Aggregate per-item run outcomes into the flywheel's four Score axes. + * + * @param {Array<{itemId?: string, succeeded: boolean, madeChanges: boolean, costUsd: number, regressed?: boolean}>} outcomes + * @returns {{primary: number, noopRate: number, costPerWin: number, regressed: boolean}} + */ +export function scoreFromOutcomes(outcomes) { + // Zero runs must never look like a clean sweep to the gate. + if (!Array.isArray(outcomes) || outcomes.length === 0) { + return { primary: 0, noopRate: 1, costPerWin: COST_PER_WIN_NO_WINS, regressed: false }; + } + + const total = outcomes.length; + const wins = outcomes.filter((o) => o.succeeded).length; + // A run that reports success while committing nothing is still a no-op — + // that is the whole point of the axis. A policy must not earn promotion by + // making the agent talk rather than act. + const noops = outcomes.filter((o) => !o.madeChanges).length; + const cost = outcomes.reduce((sum, o) => sum + (Number(o.costUsd) || 0), 0); + + return { + primary: wins / total, + noopRate: noops / total, + costPerWin: wins === 0 ? COST_PER_WIN_NO_WINS : cost / wins, + regressed: outcomes.some((o) => o.regressed === true), + }; +} + +/** Levers rvAgent knows how to apply. Must match rvagent_core::policy::KNOWN_LEVERS. */ +export const KNOWN_LEVERS = [ + 'max_iterations', + 'parallel_tools', + 'max_parallel_tools', + 'loop_repeat_threshold', + 'keep_last_observations', + 'max_tool_result_bytes', + 'system_prompt_suffix', + 'compaction_rubric', +]; + +/** + * Reject a policy naming a lever rvAgent does not apply. + * + * Throwing is deliberate. A mutation to an unapplied lever produces a run + * identical to baseline; the optimizer would read that as "no effect" and burn + * generations proposing more of them. Failing loudly keeps the search honest. + */ +export function assertKnownLevers(policy) { + const unknown = Object.keys(policy ?? {}).filter((k) => !KNOWN_LEVERS.includes(k)); + if (unknown.length > 0) { + throw new Error( + `policy names levers rvAgent does not apply: ${unknown.join(', ')}. ` + + `Known levers: ${KNOWN_LEVERS.join(', ')}`, + ); + } +} + +/** + * Build an Evaluator for `runFlywheelGenerations`. + * + * @param {{runItem: (policy: object, item: unknown) => Promise}} deps + * @returns {(policy: object, suite: {id: string, items: unknown[]}) => Promise} + */ +export function makeRvagentEvaluator({ runItem }) { + if (typeof runItem !== 'function') { + throw new TypeError('makeRvagentEvaluator requires a runItem function'); + } + + return async function evaluate(policy, suite) { + assertKnownLevers(policy); + + const items = suite?.items ?? []; + const outcomes = []; + for (const item of items) { + // Sequential on purpose: concurrent runs contend for the same workspace + // and would make cost and wall-clock unattributable per item. + outcomes.push(await runItem(policy, item)); + } + + // A dropped item would silently shrink the denominator and inflate every + // axis. Refuse rather than score a partial suite as if it were complete. + if (outcomes.length !== items.length) { + throw new Error( + `evaluator produced ${outcomes.length} outcomes for ${items.length} items`, + ); + } + + return scoreFromOutcomes(outcomes); + }; +}