From 8baf5acd478a06bd46d9b190e2e9ff957663196c Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 15:08:59 +0000 Subject: [PATCH 01/25] docs(research): rvagent Hermes-class harness architecture with metaharness + ruflo integration Research synthesis from four parallel investigations (Hermes harness web research; code audits of crates/rvAgent, ruvnet/metaharness, ruvnet/ruflo): - Findings: Hermes architecture/benchmarks, rvAgent's 4 blocking loop defects vs its production-grade A2A/MCP/security layers, metaharness Darwin/flywheel apparatus, ruflo coordination-plane seams - Target architecture: event-streaming loop, cache-first prompt tiers, compaction with lineage, 4-layer memory, skill synthesis learning loop, explicit rvagent<->ruflo and rvagent<->metaharness integration contracts - Phased roadmap (P0 foundation repair -> P4 evolution/SOTA) with falsifiable exit gates and six proposed ADRs Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- .../rvagent-hermes-harness/01-findings.md | 210 +++++++++++++++++ .../02-target-architecture.md | 217 ++++++++++++++++++ .../rvagent-hermes-harness/03-roadmap.md | 146 ++++++++++++ .../research/rvagent-hermes-harness/README.md | 93 ++++++++ 4 files changed, 666 insertions(+) create mode 100644 docs/research/rvagent-hermes-harness/01-findings.md create mode 100644 docs/research/rvagent-hermes-harness/02-target-architecture.md create mode 100644 docs/research/rvagent-hermes-harness/03-roadmap.md create mode 100644 docs/research/rvagent-hermes-harness/README.md diff --git a/docs/research/rvagent-hermes-harness/01-findings.md b/docs/research/rvagent-hermes-harness/01-findings.md new file mode 100644 index 0000000000..d03549aa9b --- /dev/null +++ b/docs/research/rvagent-hermes-harness/01-findings.md @@ -0,0 +1,210 @@ +# Findings: Hermes, rvAgent, MetaHarness, RuFlo + +Research conducted 2026-08-01 across four parallel investigations: +web research on the Hermes harness, and deep code audits of +`crates/rvAgent/` (this repo), `ruvnet/metaharness`, and `ruvnet/ruflo`. + +--- + +## 1. The Hermes Harness (NousResearch/hermes-agent) + +MIT-licensed Python harness from Nous Research, released Feb 2026; ~175K +GitHub stars in four months, the most-used agent on OpenRouter by mid-2026. +Fully model-agnostic (18+ providers, 3 API modes, mid-session failover). +Sources: [repo](https://github.com/NousResearch/hermes-agent) · +[architecture docs](https://hermes-agent.nousresearch.com/docs/developer-guide/architecture) · +[self-evolution companion](https://github.com/NousResearch/hermes-agent-self-evolution). + +### Architecture + +- **Loop:** one synchronous `AIAgent` class (`run_agent.py`) serves CLI, + messaging gateway, ACP, batch, and API-server modes. Task ID → prompt + build → preflight compression check → provider resolution → API call → + tool dispatch loop → SQLite persistence. +- **Cache-first prompts:** ordered tiers `stable` (identity, tool guidance, + skill summaries) → `context` (user context files) → `volatile` + (memory/profile/timestamp). Invariant: *the system prompt never changes + mid-conversation*. Only model switch or memory/context file change breaks + the prefix cache → **91–97% measured cache hit rates**. +- **Tools:** 70+ tools/28 toolsets, self-registering; MCP for extension. + Benchmark configs run well with **only `terminal` + `file` enabled**. +- **Sandboxing:** 7 terminal backends (local, Docker, SSH, Singularity, + Modal, Daytona, Vercel) behind one interface; Docker defaults to + read-only root FS + dropped capabilities. Zero telemetry. +- **Context management:** a sentinel triggers compaction *before* hard + limits; an auxiliary model extracts durable facts into memory (hard + 3,575-char budget) and **summarizes middle turns instead of dropping + them**; compressed sessions keep parent/child **lineage** in SQLite. +- **Memory, four separated layers:** + 1. *Prompt*: MEMORY.md / USER.md, always injected, hard char budget + 2. *Episodic*: SQLite + FTS5 over all past sessions, retrieved on demand + 3. *Procedural*: **skills** — markdown in `~/.hermes/skills/` + (agentskills.io standard), progressive disclosure + 4. *User model*: optional passively-built profile +- **The learning loop (signature feature):** skill creation triggers on 5+ + tool calls, error recovery, or user correction → trajectory distilled + into a named skill; skills self-patch during use (string-patch default); + offline, GEPA (Genetic-Pareto, ICLR 2026 Oral, ~35× fewer rollouts than + GRPO) rewrites underperforming skills from execution traces. +- **Subagents:** `delegate_tool.py` spawns isolated subagents, multi-model + routing per subtask, programmatic tool-calling via `execute_code`. + +### Benchmark evidence + +- **Claw-SWE-Bench** (arXiv 2606.12344): harness choice alone swings pass@1 + by **12.5 pts** (strong model) to **27.4 pts** (weak model). Hermes 71.1% + with GLM 5.1. +- **Harness-Bench** (arXiv 2605.27922): Hermes 71.2% overall, 100% security; + paper's conclusion — "execution alignment" (model beliefs ↔ workspace + state ↔ tool feedback ↔ verification) is the dominant success factor. +- **Learning-loop payoff:** agents with 20+ self-created skills complete + similar tasks **~40% faster / 40% fewer tokens** (Nous internal, + independently corroborated; domain-specific). +- **Agentic Harness Engineering** (arXiv 2604.25850): *structure transfers, + prose doesn't* — tools, middleware, memory architecture generalize across + models; prompt wording tweaks don't. An observability-driven self-evolving + harness beat human-designed Codex-CLI on Terminal-Bench 2 (77.0% vs 71.9%). + +Hermes's reputation rests on **amortized performance via the learning loop +and ecosystem dominance**, not one-shot leaderboard wins — it is top-tier +but not #1 on frozen single-run benchmarks. + +--- + +## 2. rvAgent Current State (crates/rvAgent/, ~45K LoC, 10 crates) + +Declared as a 100%-fidelity Rust port of LangChain DeepAgents (ADR-093..103), +extended with MCP (ADR-104/105/112), A2A (ADR-159), RVF (ADR-106), WASM. + +### Production-grade parts + +- **`rvagent-a2a`** — the best crate: Ed25519-signed AgentCards, global + rolling budgets, per-task policy, peer routing with EWMA + circuit + breaker, recursion guard, W3C trace context, typed/versioned artifacts, + SSE + signed webhooks. 24 integration test files. *But no real runner — + `InMemoryRunner` echoes; not connected to the agent loop.* +- **`rvagent-mcp`** — complete JSON-RPC 2.0 MCP server/client, stdio/SSE, + tool groups, skills bridge (Claude Code + Codex formats). Implemented + per ADR-112. +- **Security primitives** — path confinement (`virtual_mode`), env + sanitization, Unicode/BiDi/homoglyph detection, tool-output sanitizer, + AES-256-GCM session encryption, SHA3-256 witness chains. + +### Blocking defects (the loop is scaffold-grade) + +| # | Defect | Evidence | +|---|---|---| +| D1 | **Tool schemas are never sent to the model.** `ChatModel::complete(&[Message])` has no tools param; Anthropic/Gemini request structs have no `tools`/`functionDeclarations` field. The loop's `Tools` node is unreachable in production — rvagent is currently a chat client, not an agent. | `rvagent-core/src/models.rs:140`, `rvagent-backends/src/anthropic.rs:69-80`, `gemini.rs:32-35` | +| D2 | **The 19-module middleware pipeline is never wired into the loop.** `build_default_pipeline` is called only from benches/tests; the CLI constructs `AgentGraph::new(model, tools)` directly. Memory, skills, summarization, prompt caching, witness, HITL, SONA, HNSW all dormant. | `rvagent-cli/src/app.rs:643-728` | +| D3 | **Three incompatible type systems.** `AgentState`, `Message`, `Tool`, `TodoItem`, `RunnableConfig` each defined 2–3× (core enum `Message` vs middleware struct vs subagents `HashMap`), no conversion layer. This is the structural blocker for D2. | `core/src/state.rs:81`, `middleware/src/lib.rs:130`, `subagents/src/lib.rs:39` | +| D4 | **Subagents, parallelism, streaming are stubs.** `spawn_sync` returns a formatted string; "parallel" tool execution awaits sequentially in a loop (README's "true concurrency" claim is false); `stream()` returns "not yet implemented"; middleware hooks are sync (can't make an HTTP call without blocking). | `subagents/src/orchestrator.rs:44-110`, `core/src/graph.rs:183-193`, `anthropic.rs:378-385` | + +Additional gaps: tool errors abort the loop via `?` instead of feeding back +as tool results (the single most important recovery behavior); session +persistence is messages-only (no todos/files/middleware state); no hooks +infrastructure at all; HITL has no approval transport; the "HNSW" middleware +uses **hash-based pseudo-embeddings with no semantic properties** and no +connection to real RuVector crates; prompt caching types exist but nothing +is emitted to the API; docs claim capabilities the code doesn't have. + +ADR-139 (decompiled Claude Code intelligence: async-generator loop yielding +13 event types, 6 permission modes, per-subagent model override) and ADR-107 +(`rvagent-swarm`) are proposed but unimplemented. + +--- + +## 3. MetaHarness (ruvnet/metaharness) + +A **harness factory + evolution lab**, not a harness runtime. Node/TS +(~50K LoC) + a small Rust kernel (2,259 LoC → WASM/NAPI). 19 published +`@metaharness/*` packages, 223 ADRs, exemplary CI/release engineering. + +- **Generator:** `npx metaharness` mints branded harnesses for **nine hosts** + — Claude Code, Codex, pi.dev, **Hermes**, OpenClaw, RVM, Copilot, + OpenCode, GitHub Actions — via a `HostAdapter.generateConfig(spec) → + {path: contents}` interface. The Hermes adapter is verified against the + real `cli-config.yaml` and mirrors ruflo's `scrubReasoningBlocks`. + **Generated harnesses contain no agent loop** (`init` + `doctor` only) — + the loop is provided by the host. +- **Rust kernel (`crates/kernel`):** MCP spec validation + `ToolRegistry`, + claim-checked `dispatch()`, **10-event hook taxonomy with + Allow/Deny/Ask/Defer decision-merge** (modeled on Claude Code), 3-tier + routing heuristic, Ed25519 witness, cost, federation. `#![forbid(unsafe_code)]`, + serde-typed, no async/no I/O — a library rvagent could consume or mirror. + (Memory module is a 34-line stub; real memory delegates to `@ruvector/*`.) +- **Darwin Mode (the credible asset):** a real DI-tested ReAct loop + (`bench/swebench/agentic-loop.mjs`: text-JSON and native function-calling + variants, anti-thrash state hashing, observation caps, escalation + cascades) plus the full measurement apparatus — official SWE-bench Docker + gold eval (**Verified 55.6%**, Lite 51.3%, conformant packaged + submissions), Terminal-Bench, LiveCodeBench, GAIA/FRAMES, DRACO, with + Wilson CIs and documented retractions of its own failed claims. +- **The flywheel policy seam:** `SWE_POLICY_SYSTEM` env appends an evolved + operating policy to the solver's system prompt without touching loop code + — the template for how Darwin should drive rvagent. +- **Key structural fact:** the marketed control plane (`@metaharness/harness` + `HarnessKernel`, ADR-047, still Proposed) and the loop that produced every + measured number (`agentic-loop.mjs`) **share no code**. The runtime slot + is open. + +RuVector's own ADR-256 already ruled: **borrow metaharness concepts, do not +take it as a runtime dependency** — and ruflo's ADR-150 enforces the same +invariant in the other direction (metaharness must stay removable). + +--- + +## 4. RuFlo (ruvnet/ruflo = claude-flow v3, TS, ~173K LoC in the CLI) + +Self-described *"agent meta-harness for Claude Code and Codex … Ruflo is the +harness — the execution layer around Claude Code."* `Agent = Model + Harness`. + +- **The execution split:** ruflo **coordinates; Claude Code executes.** + `agent_spawn` is registry metadata; multi-turn tool-use execution is + delegated to Claude Code's Task tool or `claude -p` subprocesses. Ruflo + itself executes only: single-turn LLM calls with bandit-fed model routing + (`agent-execute-core.ts`), headless Claude subprocesses (worker daemon, + fable-harness LLM-judge), and — the one real in-house turn loop — + **rvagent's WASM build** via `@ruvector/rvagent-wasm` + (`ruvector/agent-wasm.ts`, 27 `wasm_agent_*`/`wasm_gallery_*` MCP tools). +- **Memory substrate (battle-tested):** `.swarm/memory.db` (SQLite/AgentDB), + `.swarm/hnsw.index`, `.rvf` containers; hybrid retrieval (BM25 + + cross-encoder rerank + RaBitQ + graph edges); **ADR-323 provenance typing** + (`user_claim|agent_output|system_observation|tool_result`) that any + co-writing runtime must honor; ReasoningBank with HNSW-backed pattern + promotion; SONA/EWC++/LoRA distillation pipeline + (RETRIEVE→JUDGE→DISTILL→CONSOLIDATE). +- **Hooks bus:** all 8 Claude Code lifecycle events funnel through + `.claude/helpers/hook-handler.cjs` — a subprocess contract a Rust binary + could shim or replace. 17 hooks + 14 background workers. +- **Model routing:** 3-tier (codemods $0 / Haiku / Sonnet-Opus) with a + persisted Thompson-sampling bandit closed-loop from execution outcomes. +- **MCP:** hand-rolled stdio JSON-RPC server, 305 tools; the **Capability + Brain** (typed maturity/authority/risk/health metadata per domain) is the + best-designed integration seam for advertising a new runtime. +- **Governance:** `.harness/mcp-policy.json` default-deny; ADR-150's + invariant — metaharness may augment, never be required. +- **Honesty:** ruflo's docs audit their own claims (e.g. "150x-12,500x NOT + reproduced — was brute-force fallback"). Its IMPROVEMENT-ROADMAP names + **"skill synthesis vs Hermes-class agents"** the highest-leverage missing + capability — DISTILL exists but never emits shareable SKILL.md artifacts. +- **rvagent is already first-class:** `plugins/ruflo-agent` wires WASM + rvagent + `wasm_agent_compose` hands agents a **safety-gated allowlist of + ruflo MCP tools** (destructive-tool gate included); plugins opt in via a + `rvagent.exposeSkillsAsTools` manifest field. + +--- + +## 5. Synthesis: the opening + +| Harness | Loop | Learning loop | Evolution apparatus | Coordination plane | Native speed / WASM | +|---|---|---|---|---|---| +| Hermes | ✅ mature | ✅ (skills+GEPA) | partial (offline) | ❌ | ❌ (Python) | +| Claude Code | ✅ mature | partial (manual skills) | ❌ | partial (Task tool) | ❌ | +| metaharness | ❌ (bench-only) | ❌ | ✅ (Darwin/flywheel) | ❌ | kernel only | +| ruflo | ❌ (delegates) | partial (DISTILL, no skills) | partial (harness-loop) | ✅ (305 tools, memory, swarm) | via rvagent-wasm | +| **rvagent today** | ❌ broken | ❌ | ❌ | via A2A (no runner) | ✅ (Rust+WASM+NAPI) | +| **rvagent target** | ✅ Hermes-class | ✅ (skills+SONA+witness) | ✅ (via metaharness) | ✅ (via ruflo) | ✅ | + +Every column of the target row has real code behind it somewhere in the +rUv ecosystem. The work is repair (rvagent's loop) plus wiring (the seams), +not greenfield invention. See [02-target-architecture.md](02-target-architecture.md). diff --git a/docs/research/rvagent-hermes-harness/02-target-architecture.md b/docs/research/rvagent-hermes-harness/02-target-architecture.md new file mode 100644 index 0000000000..9b6914b5e6 --- /dev/null +++ b/docs/research/rvagent-hermes-harness/02-target-architecture.md @@ -0,0 +1,217 @@ +# Target Architecture: rvAgent as a Hermes-Class, Self-Evolving Rust Harness + +Design principle (from the benchmark literature): **structure transfers, +prose doesn't**. Everything below is structural — tools, loop mechanics, +memory layers, seams — not prompt wording. + +--- + +## 1. The Loop (rvagent-core) + +Replace today's blocking 4-node state machine with an **event-streaming +loop** (the ADR-139 direction, matching Claude Code's decompiled design and +Hermes's single-class-many-modes pattern): + +```rust +// One loop, many frontends (CLI, TUI, ACP, A2A runner, WASM, headless) +pub trait AgentLoop { + fn run(&mut self, input: LoopInput) -> impl Stream; +} + +pub enum AgentEvent { + TurnStart { .. }, ModelDelta { text: String }, // streaming + ToolCallStart { id, name, args }, ToolCallEnd { id, result }, + PermissionRequest { .. }, // HITL surfaces here + CompactionStart { .. }, CompactionEnd { lineage: SessionLineage }, + SubagentSpawned { .. }, SubagentResult { .. }, + Checkpoint { id }, TurnEnd { usage: Usage }, Done { state: AgentState }, + Error { recoverable: bool, .. }, +} +``` + +Non-negotiable loop behaviors (each maps to a measured Hermes/Harness-Bench +lesson): + +1. **Tools go to the model.** `ChatModel::complete(&[Message], &[ToolDefinition])` + and `tools` on the Anthropic/Gemini request bodies. (Fixes D1.) +2. **Tool errors are tool results.** Never `?`-abort the loop on a failed + tool; feed the error text back as a `Message::tool` so the model can + recover. ("Execution alignment" — the dominant Harness-Bench factor.) +3. **Real parallel tool execution** via the existing (unused) + `parallel_execute` JoinSet+Semaphore path. (Fixes D4.) +4. **Real streaming**: SSE parsing in the backends, `impl Stream`, incremental TUI render. +5. **Few, high-fidelity core tools.** Keep the 9 built-ins; Hermes wins + benchmarks with `terminal` + `file` only. Everything else arrives via + MCP (rvagent-mcp client) and ruflo's `wasm_agent_compose` allowlist — + never as bespoke tool sprawl. +6. **One canonical type system.** `rvagent-core`'s `AgentState`/`Message`/ + `Tool` become the only definitions; middleware and subagents consume + them. (Fixes D3 — the blocker for everything else.) +7. **Async middleware.** `Middleware::wrap_model_call` becomes async; the + default pipeline is constructed and wired in every entrypoint. (Fixes D2.) + +## 2. Cache-First Prompt Assembly (new: `prompt_builder` in rvagent-core) + +Hermes's 91–97% cache hit rate comes from a discipline, not a feature: + +``` +[stable] identity + tool guidance + skill summaries ← changes only on config change +[context] AGENTS.md / project context files ← changes only on file change +[volatile] memory digest + todos + timestamp ← the ONLY tier that moves +``` + +- The stable tier is emitted with `cache_control` breakpoints (the existing + `PromptCachingMiddleware` + a new `cache_control` field on `ApiRequest`). +- Enumerate cache-breaking events (model switch, memory-file change, + context-file change) and log them — cache hit rate becomes a first-class + metric in `Metrics`. +- The system prompt never mutates mid-conversation; volatile data rides in + the tier boundary, and compaction respects the cache boundary. + +## 3. Compaction with Lineage (wire + upgrade `SummarizationMiddleware`) + +- Sentinel triggers **before** the hard limit (Hermes pattern): summarize + middle turns, keep head (stable prompt) and tail (recent turns) intact. +- Durable facts extracted to the memory layer under a **hard char budget** + (forcing curation, per Hermes's 3,575-char discipline). +- **Lineage**: compacted sessions record parent/child chains in the session + store so summaries trace back to raw turns (enables replay + Darwin's + trajectory harvesting). +- Upgrade path: ADR-252 coherence-weighted compaction + (`ruvector-agent-memory`) replaces naive char-window summarization — a + RuVector-native capability Hermes doesn't have. +- Replace the chars/4 token estimate with a real tokenizer + per-model + context-window table. + +## 4. Four-Layer Memory (mirror Hermes, back with RuVector/ruflo substrate) + +| Layer | Hermes | rvagent implementation | +|---|---|---| +| Prompt | MEMORY.md/USER.md, hard budget | `MemoryMiddleware` (AGENTS.md, exists) + write-back with budget enforcement | +| Episodic | SQLite+FTS5 over sessions | session store + **ruflo's `.swarm/memory.db`** via AgentDB bindings; honor ADR-323 provenance (`agent_output`, `tool_result`) and the WAL-sidecar guard | +| Procedural | skills (markdown, agentskills.io) | `SkillsMiddleware` + `skills_bridge` (exists; already speaks Claude Code + Codex formats) with progressive disclosure at runtime | +| User model | Honcho profile | optional; defer | + +**Kill the fake HNSW.** The current hash-based pseudo-embedding middleware +is worse than nothing (locality-insensitive, unfounded perf claims). Replace +with real `ruvector-core`/`@ruvector` embeddings behind the existing feature +gate, or use ruflo's hybrid retrieval over MCP. FTS-first, ANN-second — the +Hermes lesson is that deterministic local search beats a vector DB for a +local-first harness. + +## 5. Skills + The Learning Loop (the differentiator) + +This is where the three systems interlock, and it directly closes ruflo's +self-identified top gap ("DISTILL never emits shareable SKILL.md"): + +``` +witnessed trajectory (rvagent witness chain, SHA3-256 tool-call entries) + │ trigger: 5+ tool calls / error recovery / user correction + ▼ +skill synthesis (auxiliary model distills trajectory → SKILL.md) + │ progressive disclosure: summary in [stable] tier, body on demand + ▼ +in-use self-patching (string-patch edits, skill_manage tool) + │ + ▼ offline +metaharness Darwin/flywheel: GEPA-style evolution of skills + policy genome + - fitness from rvagent headless runs (SWE-bench/Terminal-Bench/TBLite) + - promotion via harness-loop gates (OBSERVE→QUALIFY→BENCHMARK→VERIFY→CANARY→ACCEPT) + - Ed25519-signed champion manifests (out-of-loop signing) + │ + ▼ +SONA/ReasoningBank consolidation (EWC++, ADR-271 genome recipe) + + optional weight-eft: SFT/DPO distillation of gold trajectories into cheap-tier LoRA +``` + +The **policy genome seam** is metaharness's `SWE_POLICY_SYSTEM` pattern: +rvagent accepts an evolved operating policy as an appended stable-tier +block (`--policy-file` / env), so Darwin mutates behavior without touching +loop code. Genome surfaces: policy text, temperature schedule, compaction +thresholds, tool config, skill set, `EwcConfig` (ADR-271). + +## 6. Subagents (make ADR-097 real) + +- `TaskTool` → `SubAgentOrchestrator` actually spawns `AgentLoop` instances + via JoinSet with a concurrency semaphore; results stream back as + `SubagentResult` events; CRDT merge + validators (already written) run on + join. +- Per-subagent model override (ADR-139's `CLAUDE_CODE_SUBAGENT_MODEL` + equivalent) — routing decision can come from ruflo's Thompson bandit via + MCP, or the local 3-tier heuristic. +- Subagent state isolation via the existing `CowStateBackend` fork/merge. +- Cross-*process/machine* delegation is already solved: the A2A crate. + Plug the real loop in as the A2A `TaskRunner` (replacing `InMemoryRunner`) + and rvagent instances federate with budgets, policy, recursion guards, + and signed identity for free — a capability Hermes does not have. + +## 7. Hooks, Permissions, Checkpoints + +- **Hooks:** adopt metaharness `crates/kernel`'s 10-event taxonomy + + `Allow/Deny/Ask/Defer` decision-merge (it's `#![forbid(unsafe_code)]`, + serde-typed, no-I/O — mirror it per ADR-256 "borrow concepts", or take it + as an *optional* dep consistent with ADR-150 symmetry). External hook + processes use the ruflo `hook-handler.cjs` subprocess contract so one + hook ecosystem serves both. +- **Permissions:** wire `HumanInTheLoopMiddleware` to the event stream + (`PermissionRequest` event ↔ approval reply), add permission modes and + per-tool allow/deny rules (ADR-139's 6-mode model). +- **Checkpoints:** full-state (messages + todos + files + middleware state + + in-flight tool calls) using the AGI container's unused + `SegmentType::Checkpoint`; resume/fork from any checkpoint. This is what + makes Darwin's population runs cheap (metaharness measured 39.3% cost + saved on resume, synthetic). +- **Sandbox depth:** keep env-sanitized local shell as the fast path, add a + Docker backend (read-only root, dropped caps — Hermes's secure default) + and honor ADR-140's WASM double-sandbox for untrusted agents. The + `SandboxBackend` trait already models this; it needs implementations. + +## 8. Headless Mode (the benchmark/evolution contract) + +`rvagent run -p "" --output-format json|stream-json --max-budget-usd +--policy-file --checkpoint-dir` — a stateless one-shot invocation emitting +the event stream as JSONL + a final result envelope (cost, tokens, tool +calls, patch). This single interface is what lets: +- metaharness Darwin use rvagent as a solver backend (exactly how + `handoff-solver.mjs` shells out to `claude -p` today), +- ruflo's worker daemon spawn rvagent instead of headless Claude, +- CI run frozen-eval gates. + +## 9. Integration Seams (explicit contracts) + +### rvagent ⇄ ruflo +| Seam | Mechanism | Status | +|---|---|---| +| Turn loop hosting | `@ruvector/rvagent-wasm` (`WasmAgent`, `JsModelProvider`) + a NAPI sibling for native speed | WASM exists; NAPI new | +| Tool surface | `wasm_agent_compose` safe-allowlist of ruflo's 305 MCP tools (destructive gate) | exists | +| Memory | shared `.swarm/memory.db` / `.rvf`; ADR-323 provenance mandatory | bindings exist (AgentDB/ruvector are Rust-origin) | +| Lifecycle | hook-handler.cjs subprocess contract; Capability Brain entry with maturity/authority/risk metadata | contract exists | +| Routing | consume `[TASK_MODEL_RECOMMENDATION]` signals; report outcomes to the bandit | signal exists | + +### rvagent ⇄ metaharness +| Seam | Mechanism | Status | +|---|---|---| +| Host adapter | `host-rvagent` (10th host): `generateConfig(spec)` emits rvagent config + skills + policy | new (adapter interface is trivial) | +| Evolution | headless JSON contract as Darwin solver; policy-genome env/file seam; flywheel promotion gates | pattern exists (`SWE_POLICY_SYSTEM`, `claude -p` precedent) | +| Governance | Ed25519 witness formats aligned (kernel witness ↔ rvagent witness chain ↔ RVF witness per ADR-106 Phase 4) | partial | +| Kernel reuse | mirror hooks/claims/dispatch types (ADR-256: concepts, not required dependency) | new | + +Both directions preserve the ADR-150 invariant: **every integration is +optional and degrades gracefully**. rvagent must run standalone; ruflo must +run without rvagent; metaharness must stay removable. + +## 10. What "best SOTA harness" concretely means here + +1. **Beat the Claw-SWE-Bench harness spread**: rvagent + a fixed open model + scores within the Hermes/OpenClaw band (≥70% on the 350-instance set), + validated with metaharness's conformant apparatus (Wilson CIs, gold + Docker eval, packaged submissions — no self-graded claims). +2. **Beat Hermes where it's weak**: native-speed kernel (startup, token + throughput, parallel tools), in-browser WASM deployment, federated + multi-agent execution with cryptographic identity/budgets (A2A), and + coherence-weighted compaction (ADR-252) instead of char windows. +3. **Match Hermes where it's strong**: the closed learning loop — measured + as ≥30% token reduction on repeat-task suites after 20+ synthesized + skills, with Darwin-evolved skill populations as the upgrade Hermes + only gets offline. diff --git a/docs/research/rvagent-hermes-harness/03-roadmap.md b/docs/research/rvagent-hermes-harness/03-roadmap.md new file mode 100644 index 0000000000..0b74882503 --- /dev/null +++ b/docs/research/rvagent-hermes-harness/03-roadmap.md @@ -0,0 +1,146 @@ +# Implementation Roadmap + +Ordering rule: nothing in P1+ lands until P0 is green, because every +feature in P1+ is blocked by the type unification and the tools-to-model +fix. Each phase has a falsifiable exit gate. + +--- + +## Phase 0 — Foundation Repair (rvagent-core/backends/middleware) + +The loop must become a real agent before it can become a great harness. + +1. **Unify types.** `rvagent-core`'s `AgentState`/`Message`/`Tool`/ + `TodoItem`/`RunnableConfig` become canonical; delete the middleware and + subagents duplicates; add conversion impls only at the WASM/serde + boundary. (Blocks everything.) +2. **Send tools to the model.** `ChatModel::complete(&[Message], + &[ToolDefinition])`; `tools` field on Anthropic `ApiRequest`; + `functionDeclarations` for Gemini; parse `tool_use` into the canonical + `ToolCall`. +3. **Async middleware + wiring.** `wrap_model_call` → async; + `build_default_pipeline()` constructed and installed in CLI, ACP, and + the future A2A runner. Delete the CLI's duplicate `LocalFsBackend` + grep/glob/execute in favor of `rvagent-backends`. +4. **Loop correctness.** Tool errors feed back as tool results (no `?` + abort); real parallel execution via `parallel_execute`; per-turn `Usage` + flows into `BudgetEnforcer` and `Metrics`. +5. **Honest docs.** Remove unsupported claims (parallel exec, HNSW speedups, + streaming) from README/architecture.md until true; docs list all 10 + crates. + +**Exit gate:** `rvagent run "create and run a failing test, then fix it"` +completes end-to-end against the live Anthropic API with ≥2 tool round +trips, budget accounting, and a green `cargo test` across the workspace. + +## Phase 1 — Hermes-Class Loop Mechanics + +1. **Event-streaming loop** (`AgentEvent` stream; ADR-139 direction) with + SSE streaming backends and incremental TUI render. +2. **Cache-first prompt builder** (stable/context/volatile tiers, + `cache_control` emission, cache-hit-rate metric). +3. **Compaction with lineage** (sentinel pre-limit, middle-turn + summarization, durable-fact extraction under hard budget, parent/child + session chains, real tokenizer). +4. **Full-state checkpoints** (`SegmentType::Checkpoint`), resume/fork. +5. **Hooks + permissions** (10-event taxonomy with Allow/Deny/Ask/Defer + merge mirrored from metaharness kernel; HITL wired to + `PermissionRequest` events; permission modes; per-tool rules). +6. **Real subagents** (JoinSet spawn, semaphore, CoW fork/merge, CRDT join, + per-subagent model override). +7. **Headless contract** (`-p --output-format stream-json + --max-budget-usd --policy-file`). +8. **Docker sandbox backend** (read-only root, dropped caps) alongside the + local shell. + +**Exit gate:** rvagent completes a 20-instance SWE-bench-Lite smoke slice +via the headless contract inside metaharness's runner, with measured cache +hit rate >85% and zero loop aborts on tool errors. + +## Phase 2 — Memory & The Learning Loop + +1. **Kill fake HNSW**; episodic memory = session store + FTS, with real + `ruvector` embeddings behind the feature gate; optional shared + `.swarm/memory.db` (AgentDB bindings, ADR-323 provenance, WAL guard). +2. **Prompt-memory write-back** with hard char budget + periodic curation + nudges. +3. **Skill synthesis**: trajectory triggers (5+ tool calls / error recovery + / user correction) → SKILL.md emission via auxiliary model; progressive + disclosure; `skill_manage` tool with string-patch default. (Closes + ruflo's #1 roadmap gap; skills interop with Claude Code/Codex via the + existing `skills_bridge`.) +4. **SONA on the default path** (feature-gated on, trajectories from the + witness chain feeding ReasoningBank; ADR-271 EwcConfig as genome). +5. **ADR-252 coherence-weighted compaction** as the summarization upgrade. + +**Exit gate:** on a 3×-repeated task suite, the skilled agent shows ≥30% +token reduction vs a fresh instance (Hermes's measured ~40% is the bar). + +## Phase 3 — Ecosystem Integration + +1. **ruflo:** NAPI package (`@ruvector/rvagent-native`) as a sibling to + `rvagent-wasm` behind `agent-wasm.ts`; `wasm_agent_compose` tool + allowlist honored; Capability Brain registration; hook-handler + subprocess shim; bandit outcome reporting. ruflo's worker daemon gains + an `rvagent` executor option beside headless Claude. +2. **metaharness:** `packages/host-rvagent` adapter; rvagent registered as + a Darwin solver backend via the headless contract; witness format + alignment (kernel ↔ rvagent ↔ RVF per ADR-106 Phase 4). +3. **A2A goes live:** real `TaskRunner` backed by the loop replaces + `InMemoryRunner`; `rvagent a2a serve` advertises real skills; federated + rvagent↔rvagent task delegation demo with budgets + recursion guards. +4. **Eval in CI:** TBLite-style smoke slice + frozen-eval gate on PRs + (metaharness harness-loop gates: QUALIFY→BENCHMARK→VERIFY→CANARY). + +**Exit gate:** one command (`npx ruflo swarm ... --executor rvagent` or +equivalent) runs a swarm where rvagent instances execute, ruflo +coordinates/remembers, and the run emits signed witness manifests. + +## Phase 4 — Evolution & SOTA Claims + +1. **Policy-genome evolution**: Darwin mutates policy text, temperature + schedules, compaction thresholds, tool configs; flywheel promotion with + held-out benchmarks and signed champion manifests. +2. **Skill-population evolution** (GEPA-style over execution traces). +3. **weight-eft distillation** of gold trajectories into cheap-tier LoRA + (ADR-271's SFT/DPO on/off-policy recipe) → escalate to frontier models + less often; cost-Pareto tracking. +4. **Publish**: Claw-SWE-Bench-style fixed-model comparison vs Hermes/ + OpenClaw, Terminal-Bench 2.0, with conformant packaged submissions and + Wilson CIs. No claim ships without the metaharness honesty apparatus + (the ecosystem's retraction discipline is a feature — keep it). + +**Exit gate:** rvagent within the Hermes/OpenClaw band (≥70%) on the +350-instance fixed-model set, and demonstrably ahead on ≥2 of: startup +latency, token throughput, parallel-tool wall clock, federated multi-agent +tasks, in-browser deployment. + +--- + +## Proposed ADRs + +| ADR | Title | Decides | +|---|---|---| +| A | rvAgent Harness Core Repair (supersedes parts of ADR-095/097) | canonical types, tools-to-model, async middleware, loop recovery semantics | +| B | rvAgent Event-Streaming Loop & Headless Contract (implements ADR-139) | `AgentEvent` taxonomy, stream-json format, budget/policy flags | +| C | rvAgent Cache-First Prompts, Compaction & Lineage | prompt tiers, cache_control, sentinel compaction, session lineage | +| D | rvAgent Skills & Learning Loop | synthesis triggers, SKILL.md interop, SONA/witness wiring, char budgets | +| E | rvAgent ⇄ ruflo Execution Seam | NAPI sibling, memory co-tenancy rules (ADR-323/WAL), hook shim, Capability Brain entry | +| F | rvAgent ⇄ metaharness Evolution Seam | host adapter, Darwin solver contract, policy genome surfaces, witness alignment | + +## Risks + +- **Scope gravity.** The ecosystem's pattern (documented in all three + audits) is protocol/ADR surface outrunning the loop. Mitigation: P0/P1 + exit gates are executable, not documentary; no new crate until the gate + passes. +- **Type unification churn** touches every crate at once. Mitigation: land + as one PR series with the workspace green at each step; WASM API kept + stable via serde boundary. +- **Benchmark credibility.** Any SOTA claim without the gold-eval + + Wilson-CI + packaged-submission discipline damages the whole ecosystem's + (currently strong) honesty record. Mitigation: Phase 4 gates are + metaharness-conformant by construction. +- **Optionality invariants** (ADR-150/256) must hold in both directions or + the three projects become a distributed monolith. Mitigation: CI runs + each project's `--ignore-optional` path with rvagent absent/present. diff --git a/docs/research/rvagent-hermes-harness/README.md b/docs/research/rvagent-hermes-harness/README.md new file mode 100644 index 0000000000..fc90458152 --- /dev/null +++ b/docs/research/rvagent-hermes-harness/README.md @@ -0,0 +1,93 @@ +# rvAgent as a Hermes-Class Harness — Research & Architecture Proposal + +**Date:** 2026-08-01 +**Status:** Research complete, implementation proposed +**Related:** ADR-093..107 (rvAgent), ADR-139 (Claude Code intelligence), ADR-150 (ruflo metaharness surfaces), ADR-159 (A2A), ADR-211/252 (agent memory), ADR-256 (metaharness concepts), ADR-260/266/271 (Darwin evolution) + +## The Question + +> How can we implement rvagent more like the Hermes harness, integrated with +> ruvnet/metaharness and ruvnet/ruflo, to create the best SOTA harness in the world? + +## The Answer in One Paragraph + +The Hermes agent (NousResearch/hermes-agent, MIT, ~175K stars) proved two things: +(1) **the harness is worth up to 27 points of SWE-bench pass@1** — more than most +model upgrades — and (2) the winning differentiator is not one-shot benchmark +score but a **closed learning loop**: trajectories distilled into self-patching +skills, evolved offline by GEPA, yielding ~40% faster/cheaper repeat tasks. +RuVector already believes this thesis — ADR-260/266/271's "freeze the model, +evolve the harness" is the same idea Hermes ships. What no one ships yet is a +**native-speed, memory-safe, WASM-portable harness kernel with an evolutionary +optimizer and a swarm coordination plane attached**. That is exactly the seam +where rvagent (Rust execution kernel) + metaharness (Darwin evolution + eval +apparatus + governance) + ruflo (memory substrate, hooks bus, MCP surface, +model routing, swarm) combine into something none of the incumbent harnesses +— Hermes included — can match. The catch: rvagent's core loop is currently +scaffold-grade (it cannot even send tool schemas to the model), so the path +starts with foundation repair, not features. + +## Ecosystem Fit (who does what) + +``` +┌────────────────────────────────────────────────────────────────────┐ +│ metaharness — the harness FACTORY & EVOLVER │ +│ mints harnesses (9 hosts incl. Hermes), Darwin/flywheel evolves │ +│ policy genomes, Ed25519 witness governance, SWE/Terminal-Bench │ +│ apparatus. Generated harnesses ship NO agent loop today. │ +└───────────────▲────────────────────────────────────────────────────┘ + │ evolves genome / benchmarks / signs +┌───────────────┴────────────────────────────────────────────────────┐ +│ rvagent — the EXECUTION KERNEL (this proposal) │ +│ Rust agent loop: tools, streaming, compaction+lineage, skills, │ +│ subagents, checkpoints, hooks, budget, witness. Ships as native │ +│ CLI + NAPI + WASM. Fills the loop-shaped hole both siblings │ +│ currently outsource to Claude Code. │ +└───────────────▲────────────────────────────────────────────────────┘ + │ memory / hooks / routing / tool surface via MCP +┌───────────────┴────────────────────────────────────────────────────┐ +│ ruflo — the COORDINATION PLANE │ +│ .swarm/memory.db + hnsw.index + .rvf substrate (ADR-323 │ +│ provenance), 305 MCP tools, hooks lifecycle bus, Thompson-bandit │ +│ model router, swarm/hive-mind. Already hosts rvagent via │ +│ @ruvector/rvagent-wasm (27 wasm_agent_* tools). │ +└────────────────────────────────────────────────────────────────────┘ +``` + +## Documents + +| File | Contents | +|---|---| +| [01-findings.md](01-findings.md) | What we found: Hermes architecture & benchmark evidence; rvagent current-state audit (4 blocking defects); metaharness & ruflo capability maps | +| [02-target-architecture.md](02-target-architecture.md) | The Hermes-class rvagent design: loop, prompt tiers, memory layers, skills, subagents, and the exact integration seams into metaharness and ruflo | +| [03-roadmap.md](03-roadmap.md) | Phased implementation plan (P0 foundations → P4 evolution/SOTA), success gates, proposed ADRs | + +## Headline Findings + +1. **rvagent's protocol layers are production-grade; its loop is not.** + `rvagent-a2a` (signed cards, budgets, circuit breakers, recursion guards), + `rvagent-mcp`, and the security primitives are real and well-tested. But the + agent loop never sends tool schemas to the model (the `Tools` node is + unreachable in production), the 19-module middleware pipeline is never wired + into the CLI, subagents are stubs, streaming does not exist, and three + incompatible `AgentState`/`Message`/`Tool` type systems block assembly. + +2. **Hermes's edge is structural, and every piece has a RuVector-native analog.** + Cache-first tiered prompts → unwired `PromptCachingMiddleware`; layered + memory → AGENTS.md + ruflo's memory substrate + ADR-211/252; skills as + procedural memory → `SkillsMiddleware` + `skills_bridge`; trajectory→skill + distillation → SONA/ReasoningBank + witness chains; GEPA offline evolution → + metaharness Darwin/flywheel. The parts exist; nothing is connected. + +3. **Both siblings have a loop-shaped hole rvagent should fill.** metaharness + generates harness *configuration* for nine hosts but no runtime loop; ruflo + explicitly delegates multi-turn execution to Claude Code (`CLAUDE.md:908`) + and its only in-house turn loop is… rvagent's WASM build. ruflo's own + roadmap names "skill synthesis vs Hermes-class agents" its top gap. + +4. **The benchmark literature says invest here.** Harness choice swings + SWE-bench pass@1 by 12.5–27.4 points (Claw-SWE-Bench); "execution + alignment" and few high-fidelity tools beat sprawling toolsets + (Harness-Bench); structure transfers across models, prompt wording doesn't + (Agentic Harness Engineering). A Rust kernel + Darwin evolution attacks all + three levers at once. From 0f63c04346087d7ac2dd515f0406e5ec5093e635 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 15:31:21 +0000 Subject: [PATCH 02/25] feat(rvagent): send tool schemas to the model; fix agent loop correctness (P0.2, P0.4) Roadmap: docs/research/rvagent-hermes-harness/03-roadmap.md Phase 0. - rvagent-core: add ToolDefinition; ChatModel::complete/stream and StreamingChatModel::stream_chunks now take the active tool set; ToolExecutor::definitions() advertises schemas to the loop - rvagent-core graph: tool errors feed back to the model as tool results instead of aborting the loop; parallel branch now uses parallel_execute_limited (JoinSet + semaphore, bounded by new GraphConfig::max_parallel_tools) instead of sequential awaits; per-turn usage aggregated from provider metadata and logged - rvagent-backends anthropic: tools field on ApiRequest; usage and stop_reason attached to response message metadata - rvagent-backends gemini: full function-calling support (functionDeclarations, functionCall parts -> ToolCalls with synthesized IDs, tool results as functionResponse parts with name recovery); usage metadata - rvagent-cli: CliToolExecutor advertises builtin tool schemas - tests: tool serialization, usage metadata, error feedback, parallel failure isolation, Gemini function-call roundtrips Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-acp/src/agent.rs | 17 +- .../rvAgent/rvagent-backends/src/anthropic.rs | 153 +++++++-- crates/rvAgent/rvagent-backends/src/gemini.rs | 292 +++++++++++++++--- .../tests/live_anthropic_test.rs | 5 +- crates/rvAgent/rvagent-cli/src/app.rs | 51 ++- crates/rvAgent/rvagent-core/src/graph.rs | 132 ++++++-- crates/rvAgent/rvagent-core/src/models.rs | 31 +- .../rvagent-core/tests/integration_tests.rs | 97 +++++- 8 files changed, 646 insertions(+), 132 deletions(-) diff --git a/crates/rvAgent/rvagent-acp/src/agent.rs b/crates/rvAgent/rvagent-acp/src/agent.rs index f9665361c2..e92379444f 100644 --- a/crates/rvAgent/rvagent-acp/src/agent.rs +++ b/crates/rvAgent/rvagent-acp/src/agent.rs @@ -13,7 +13,7 @@ use rvagent_core::config::RvAgentConfig; use rvagent_core::error::Result as CoreResult; use rvagent_core::graph::{AgentGraph, GraphConfig, ToolExecutor}; use rvagent_core::messages::{Message, ToolCall}; -use rvagent_core::models::ChatModel; +use rvagent_core::models::{ChatModel, ToolDefinition}; use rvagent_core::state::AgentState; use crate::types::{ @@ -77,7 +77,11 @@ struct StubModel; #[async_trait] impl ChatModel for StubModel { - async fn complete(&self, messages: &[Message]) -> CoreResult { + async fn complete( + &self, + messages: &[Message], + _tools: &[ToolDefinition], + ) -> CoreResult { // Find the last human message and produce an intelligent echo. let user_text = messages .iter() @@ -95,8 +99,12 @@ impl ChatModel for StubModel { Ok(Message::ai(response)) } - async fn stream(&self, messages: &[Message]) -> CoreResult> { - let msg = self.complete(messages).await?; + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> CoreResult> { + let msg = self.complete(messages, tools).await?; Ok(vec![msg]) } } @@ -194,6 +202,7 @@ impl AcpAgent { let graph_config = GraphConfig { max_iterations: 10, parallel_tools: false, + ..GraphConfig::default() }; let graph = AgentGraph::with_config(StubModel, AcpToolExecutor, graph_config); diff --git a/crates/rvAgent/rvagent-backends/src/anthropic.rs b/crates/rvAgent/rvagent-backends/src/anthropic.rs index 43f502e93e..01e1fc6f98 100644 --- a/crates/rvAgent/rvagent-backends/src/anthropic.rs +++ b/crates/rvAgent/rvagent-backends/src/anthropic.rs @@ -13,7 +13,7 @@ use tracing::{debug, warn}; use rvagent_core::error::{Result, RvAgentError}; use rvagent_core::messages::{AiMessage, Message, ToolCall}; -use rvagent_core::models::{ApiKeySource, ChatModel, ModelConfig}; +use rvagent_core::models::{ApiKeySource, ChatModel, ModelConfig, ToolDefinition}; // --------------------------------------------------------------------------- // Constants @@ -65,6 +65,24 @@ enum ContentBlock { }, } +/// A tool definition in the Anthropic Messages API format. +#[derive(Debug, Clone, Serialize, Deserialize)] +struct ApiTool { + name: String, + description: String, + input_schema: serde_json::Value, +} + +impl From<&ToolDefinition> for ApiTool { + fn from(def: &ToolDefinition) -> Self { + Self { + name: def.name.clone(), + description: def.description.clone(), + input_schema: def.input_schema.clone(), + } + } +} + /// The request body sent to the Anthropic Messages API. #[derive(Debug, Serialize)] struct ApiRequest { @@ -75,6 +93,8 @@ struct ApiRequest { #[serde(skip_serializing_if = "Option::is_none")] system: Option, messages: Vec, + #[serde(skip_serializing_if = "Vec::is_empty")] + tools: Vec, #[serde(skip_serializing_if = "Option::is_none")] stream: Option, } @@ -85,18 +105,14 @@ struct ApiResponse { content: Vec, #[allow(dead_code)] model: String, - #[allow(dead_code)] stop_reason: Option, - #[allow(dead_code)] usage: Option, } /// Token usage information. #[derive(Debug, Deserialize)] struct Usage { - #[allow(dead_code)] input_tokens: u64, - #[allow(dead_code)] output_tokens: u64, } @@ -129,7 +145,7 @@ struct ApiErrorDetail { /// # async fn example() -> rvagent_core::error::Result<()> { /// let config = resolve_model("anthropic:claude-sonnet-4-20250514"); /// let client = AnthropicClient::new(config)?; -/// let response = client.complete(&[Message::human("Hello!")]).await?; +/// let response = client.complete(&[Message::human("Hello!")], &[]).await?; /// println!("{}", response.content()); /// # Ok(()) /// # } @@ -168,8 +184,8 @@ impl AnthropicClient { } } - /// Build the API request body from rvAgent messages. - fn build_request(&self, messages: &[Message], stream: bool) -> ApiRequest { + /// Build the API request body from rvAgent messages and tool definitions. + fn build_request(&self, messages: &[Message], tools: &[ToolDefinition], stream: bool) -> ApiRequest { let mut system_text: Option = None; let mut api_messages: Vec = Vec::new(); @@ -240,6 +256,7 @@ impl AnthropicClient { }, system: system_text, messages: api_messages, + tools: tools.iter().map(ApiTool::from).collect(), stream: if stream { Some(true) } else { None }, } } @@ -330,6 +347,9 @@ impl AnthropicClient { } /// Convert an API response into an rvAgent [`Message`]. + /// + /// Token usage is attached to the message metadata under the `usage` key + /// so the agent loop and budget layers can account for it. fn parse_response(response: ApiResponse) -> Message { let mut text_parts: Vec = Vec::new(); let mut tool_calls: Vec = Vec::new(); @@ -352,23 +372,36 @@ impl AnthropicClient { let content = text_parts.join(""); - if tool_calls.is_empty() { - Message::ai(content) - } else { - Message::Ai(AiMessage { - content, - tool_calls, - metadata: HashMap::new(), - }) + let mut metadata = HashMap::new(); + if let Some(usage) = &response.usage { + metadata.insert( + "usage".to_string(), + serde_json::json!({ + "input_tokens": usage.input_tokens, + "output_tokens": usage.output_tokens, + }), + ); } + if let Some(stop_reason) = &response.stop_reason { + metadata.insert( + "stop_reason".to_string(), + serde_json::Value::String(stop_reason.clone()), + ); + } + + Message::Ai(AiMessage { + content, + tool_calls, + metadata, + }) } } #[async_trait] impl ChatModel for AnthropicClient { - /// Send messages and receive a complete response. - async fn complete(&self, messages: &[Message]) -> Result { - let request_body = self.build_request(messages, false); + /// Send messages and the active tool set, receive a complete response. + async fn complete(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result { + let request_body = self.build_request(messages, tools, false); let response = self .send_with_retry(&request_body, ANTHROPIC_API_URL) .await?; @@ -379,8 +412,8 @@ impl ChatModel for AnthropicClient { /// /// True SSE streaming is not yet implemented; this method calls the non-streaming /// endpoint and returns a single-element vector containing the complete message. - async fn stream(&self, messages: &[Message]) -> Result> { - let msg = self.complete(messages).await?; + async fn stream(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result> { + let msg = self.complete(messages, tools).await?; Ok(vec![msg]) } } @@ -452,7 +485,7 @@ mod tests { Message::system("You are helpful."), Message::human("Hello!"), ]; - let req = client.build_request(&messages, false); + let req = client.build_request(&messages, &[], false); assert_eq!(req.model, "claude-sonnet-4-20250514"); assert_eq!(req.max_tokens, 1024); @@ -471,7 +504,7 @@ mod tests { Message::system("Second instruction."), Message::human("Go."), ]; - let req = client.build_request(&messages, false); + let req = client.build_request(&messages, &[], false); assert_eq!( req.system, @@ -495,7 +528,7 @@ mod tests { ), Message::tool("tc_1", "file contents here"), ]; - let req = client.build_request(&messages, false); + let req = client.build_request(&messages, &[], false); assert_eq!(req.messages.len(), 3); assert_eq!(req.messages[0].role, "user"); @@ -511,12 +544,65 @@ mod tests { } } + #[test] + fn test_build_request_with_tool_definitions() { + let client = + AnthropicClient::with_http(test_config(), reqwest::Client::new(), "key".to_string()); + let tools = vec![ToolDefinition { + name: "read_file".to_string(), + description: "Read a file from the workspace".to_string(), + input_schema: json!({ + "type": "object", + "properties": {"path": {"type": "string"}}, + "required": ["path"] + }), + }]; + let req = client.build_request(&[Message::human("read it")], &tools, false); + + assert_eq!(req.tools.len(), 1); + assert_eq!(req.tools[0].name, "read_file"); + + // Wire format: tools must serialize with name/description/input_schema. + let json = serde_json::to_value(&req).unwrap(); + assert_eq!(json["tools"][0]["name"], "read_file"); + assert!(json["tools"][0]["input_schema"]["properties"]["path"].is_object()); + + // Empty tool set must omit the field entirely. + let req_no_tools = client.build_request(&[Message::human("hi")], &[], false); + let json_no_tools = serde_json::to_value(&req_no_tools).unwrap(); + assert!(json_no_tools.get("tools").is_none()); + } + + #[test] + fn test_parse_response_attaches_usage_metadata() { + let response = ApiResponse { + content: vec![ContentBlock::Text { + text: "hi".to_string(), + }], + model: "claude-sonnet-4-20250514".to_string(), + stop_reason: Some("end_turn".to_string()), + usage: Some(Usage { + input_tokens: 11, + output_tokens: 7, + }), + }; + let msg = AnthropicClient::parse_response(response); + if let Message::Ai(ai) = &msg { + let usage = ai.metadata.get("usage").expect("usage metadata"); + assert_eq!(usage["input_tokens"], 11); + assert_eq!(usage["output_tokens"], 7); + assert_eq!(ai.metadata["stop_reason"], "end_turn"); + } else { + panic!("expected Ai message"); + } + } + #[test] fn test_build_request_stream_flag() { let client = AnthropicClient::with_http(test_config(), reqwest::Client::new(), "key".to_string()); let messages = vec![Message::human("Hi")]; - let req = client.build_request(&messages, true); + let req = client.build_request(&messages, &[], true); assert_eq!(req.stream, Some(true)); } @@ -642,14 +728,14 @@ mod tests { fn test_temperature_serialization() { let client = AnthropicClient::with_http(test_config(), reqwest::Client::new(), "key".to_string()); - let req = client.build_request(&[Message::human("Hi")], false); + let req = client.build_request(&[Message::human("Hi")], &[], false); // temperature=0.0 => None (omitted) assert!(req.temperature.is_none()); let mut config = test_config(); config.temperature = 0.7; let client2 = AnthropicClient::with_http(config, reqwest::Client::new(), "key".to_string()); - let req2 = client2.build_request(&[Message::human("Hi")], false); + let req2 = client2.build_request(&[Message::human("Hi")], &[], false); assert_eq!(req2.temperature, Some(0.7)); } @@ -664,6 +750,7 @@ mod tests { role: "user".to_string(), content: ApiContent::Text("Hello".to_string()), }], + tools: Vec::new(), stream: None, }; let json = serde_json::to_value(&req).expect("serialization failed"); @@ -736,7 +823,7 @@ mod tests { let client = test_client(&server.url()); let url = format!("{}/v1/messages", server.url()); - let req = client.build_request(&[Message::human("Hello")], false); + let req = client.build_request(&[Message::human("Hello")], &[], false); let resp = client.send_with_retry(&req, &url).await; assert!(resp.is_ok()); @@ -768,7 +855,7 @@ mod tests { let client = test_client(&server.url()); let url = format!("{}/v1/messages", server.url()); - let req = client.build_request(&[Message::human("Search for Rust")], false); + let req = client.build_request(&[Message::human("Search for Rust")], &[], false); let resp = client.send_with_retry(&req, &url).await; assert!(resp.is_ok()); @@ -794,7 +881,7 @@ mod tests { let client = test_client(&server.url()); let url = format!("{}/v1/messages", server.url()); - let req = client.build_request(&[Message::human("Hi")], false); + let req = client.build_request(&[Message::human("Hi")], &[], false); let result = client.send_with_retry(&req, &url).await; assert!(result.is_err()); @@ -839,7 +926,7 @@ mod tests { let client = test_client(&server.url()); let url = format!("{}/v1/messages", server.url()); - let req = client.build_request(&[Message::human("Hi")], false); + let req = client.build_request(&[Message::human("Hi")], &[], false); let result = client.send_with_retry(&req, &url).await; assert!(result.is_ok()); @@ -869,7 +956,7 @@ mod tests { let client = test_client(&server.url()); let url = format!("{}/v1/messages", server.url()); - let req = client.build_request(&[Message::human("Hi")], false); + let req = client.build_request(&[Message::human("Hi")], &[], false); let result = client.send_with_retry(&req, &url).await; assert!(result.is_err()); @@ -902,7 +989,7 @@ mod tests { let client = test_client(&server.url()); let url = format!("{}/v1/messages", server.url()); - let req = client.build_request(&[Message::human("Hi")], false); + let req = client.build_request(&[Message::human("Hi")], &[], false); let result = client.send_with_retry(&req, &url).await; assert!(result.is_ok()); diff --git a/crates/rvAgent/rvagent-backends/src/gemini.rs b/crates/rvAgent/rvagent-backends/src/gemini.rs index 980d5da1fa..e65a4a73a7 100644 --- a/crates/rvAgent/rvagent-backends/src/gemini.rs +++ b/crates/rvAgent/rvagent-backends/src/gemini.rs @@ -9,9 +9,11 @@ use async_trait::async_trait; use serde::{Deserialize, Serialize}; use tracing::{debug, warn}; +use std::collections::HashMap; + use rvagent_core::error::{Result, RvAgentError}; -use rvagent_core::messages::{AiMessage, Message}; -use rvagent_core::models::{ApiKeySource, ChatModel, ModelConfig}; +use rvagent_core::messages::{AiMessage, Message, ToolCall}; +use rvagent_core::models::{ApiKeySource, ChatModel, ModelConfig, ToolDefinition}; // --------------------------------------------------------------------------- // Constants @@ -28,10 +30,55 @@ const RETRYABLE_STATUS_CODES: &[u16] = &[429, 500, 502, 503]; // Gemini API request / response types // --------------------------------------------------------------------------- -/// Content part in a Gemini message. +/// A function call emitted by the model. +#[derive(Debug, Clone, Serialize, Deserialize)] +struct FunctionCall { + name: String, + #[serde(default)] + args: serde_json::Value, +} + +/// A function result sent back to the model. #[derive(Debug, Clone, Serialize, Deserialize)] +struct FunctionResponse { + name: String, + response: serde_json::Value, +} + +/// Content part in a Gemini message: text, a function call, or a function response. +#[derive(Debug, Clone, Serialize, Deserialize, Default)] +#[serde(rename_all = "camelCase")] struct Part { - text: String, + #[serde(skip_serializing_if = "Option::is_none")] + text: Option, + #[serde(skip_serializing_if = "Option::is_none")] + function_call: Option, + #[serde(skip_serializing_if = "Option::is_none")] + function_response: Option, +} + +impl Part { + fn text(text: impl Into) -> Self { + Self { + text: Some(text.into()), + ..Self::default() + } + } +} + +/// A function declaration advertised to the model. +#[derive(Debug, Clone, Serialize, Deserialize)] +struct FunctionDeclaration { + name: String, + description: String, + parameters: serde_json::Value, +} + +/// A tool group in the Gemini API format. +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +struct GeminiTool { + function_declarations: Vec, } /// A single message in the Gemini API format. @@ -58,6 +105,8 @@ struct GeminiRequest { generation_config: GenerationConfig, #[serde(skip_serializing_if = "Option::is_none")] system_instruction: Option, + #[serde(skip_serializing_if = "Vec::is_empty")] + tools: Vec, } /// A candidate response from Gemini. @@ -71,9 +120,9 @@ struct Candidate { /// The response body from the Gemini API. #[derive(Debug, Deserialize)] +#[serde(rename_all = "camelCase")] struct GeminiResponse { candidates: Vec, - #[allow(dead_code)] #[serde(default)] usage_metadata: Option, } @@ -82,9 +131,7 @@ struct GeminiResponse { #[derive(Debug, Deserialize)] #[serde(rename_all = "camelCase")] struct UsageMetadata { - #[allow(dead_code)] prompt_token_count: Option, - #[allow(dead_code)] candidates_token_count: Option, } @@ -117,7 +164,7 @@ struct GeminiErrorDetail { /// # async fn example() -> rvagent_core::error::Result<()> { /// let config = resolve_model("google:gemini-2.5-pro-preview-06-05"); /// let client = GeminiClient::new(config)?; -/// let response = client.complete(&[Message::human("Hello!")]).await?; +/// let response = client.complete(&[Message::human("Hello!")], &[]).await?; /// println!("{}", response.content()); /// # Ok(()) /// # } @@ -143,49 +190,86 @@ impl GeminiClient { }) } - /// Build the API request body from rvAgent messages. - fn build_request(&self, messages: &[Message]) -> GeminiRequest { + /// Build the API request body from rvAgent messages and tool definitions. + fn build_request(&self, messages: &[Message], tools: &[ToolDefinition]) -> GeminiRequest { let mut system_instruction: Option = None; let mut contents: Vec = Vec::new(); + // Gemini has no tool-call IDs on the wire; map our synthesized IDs + // back to function names so tool results can be sent as + // functionResponse parts. + let mut call_id_to_name: HashMap = HashMap::new(); for msg in messages { match msg { Message::System(s) => { system_instruction = Some(GeminiContent { role: "user".to_string(), - parts: vec![Part { - text: s.content.clone(), - }], + parts: vec![Part::text(s.content.clone())], }); } Message::Human(h) => { contents.push(GeminiContent { role: "user".to_string(), - parts: vec![Part { - text: h.content.clone(), - }], + parts: vec![Part::text(h.content.clone())], }); } Message::Ai(ai) => { + let mut parts: Vec = Vec::new(); + if !ai.content.is_empty() { + parts.push(Part::text(ai.content.clone())); + } + for tc in &ai.tool_calls { + call_id_to_name.insert(tc.id.clone(), tc.name.clone()); + parts.push(Part { + function_call: Some(FunctionCall { + name: tc.name.clone(), + args: tc.args.clone(), + }), + ..Part::default() + }); + } + if parts.is_empty() { + parts.push(Part::text(String::new())); + } contents.push(GeminiContent { role: "model".to_string(), - parts: vec![Part { - text: ai.content.clone(), - }], + parts, }); } Message::Tool(t) => { - // Tool results go as user messages + let name = call_id_to_name + .get(&t.tool_call_id) + .cloned() + .unwrap_or_else(|| t.tool_call_id.clone()); contents.push(GeminiContent { role: "user".to_string(), parts: vec![Part { - text: format!("Tool result: {}", t.content), + function_response: Some(FunctionResponse { + name, + response: serde_json::json!({ "result": t.content }), + }), + ..Part::default() }], }); } } } + let gemini_tools = if tools.is_empty() { + Vec::new() + } else { + vec![GeminiTool { + function_declarations: tools + .iter() + .map(|t| FunctionDeclaration { + name: t.name.clone(), + description: t.description.clone(), + parameters: t.input_schema.clone(), + }) + .collect(), + }] + }; + GeminiRequest { contents, generation_config: GenerationConfig { @@ -197,6 +281,7 @@ impl GeminiClient { }, }, system_instruction, + tools: gemini_tools, } } @@ -283,28 +368,50 @@ impl GeminiClient { #[async_trait] impl ChatModel for GeminiClient { - async fn complete(&self, messages: &[Message]) -> Result { - let request = self.build_request(messages); + async fn complete(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result { + let request = self.build_request(messages, tools); let response = self.send_with_retry(&request).await?; - // Extract text from first candidate - let text = response - .candidates - .first() - .and_then(|c| c.content.parts.first()) - .map(|p| p.text.clone()) - .unwrap_or_default(); + // Collect text and function-call parts from the first candidate. + // Gemini does not assign tool-call IDs, so synthesize stable ones. + let mut text_parts: Vec = Vec::new(); + let mut tool_calls: Vec = Vec::new(); + if let Some(candidate) = response.candidates.first() { + for (idx, part) in candidate.content.parts.iter().enumerate() { + if let Some(text) = &part.text { + text_parts.push(text.clone()); + } + if let Some(fc) = &part.function_call { + tool_calls.push(ToolCall { + id: format!("gemini_call_{idx}_{}", fc.name), + name: fc.name.clone(), + args: fc.args.clone(), + }); + } + } + } + + let mut metadata = HashMap::new(); + if let Some(usage) = &response.usage_metadata { + metadata.insert( + "usage".to_string(), + serde_json::json!({ + "input_tokens": usage.prompt_token_count.unwrap_or(0), + "output_tokens": usage.candidates_token_count.unwrap_or(0), + }), + ); + } Ok(Message::Ai(AiMessage { - content: text, - tool_calls: vec![], - metadata: std::collections::HashMap::new(), + content: text_parts.join(""), + tool_calls, + metadata, })) } - async fn stream(&self, messages: &[Message]) -> Result> { + async fn stream(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result> { // For now, use non-streaming completion - let msg = self.complete(messages).await?; + let msg = self.complete(messages, tools).await?; Ok(vec![msg]) } } @@ -338,20 +445,21 @@ mod tests { let request = GeminiRequest { contents: vec![GeminiContent { role: "user".to_string(), - parts: vec![Part { - text: "Hello".to_string(), - }], + parts: vec![Part::text("Hello")], }], generation_config: GenerationConfig { max_output_tokens: 1024, temperature: Some(0.7), }, system_instruction: None, + tools: Vec::new(), }; let json = serde_json::to_string(&request).unwrap(); assert!(json.contains("\"role\":\"user\"")); assert!(json.contains("\"maxOutputTokens\":1024")); + // Empty tools must be omitted entirely. + assert!(!json.contains("\"tools\"")); } #[test] @@ -368,6 +476,112 @@ mod tests { let response: GeminiResponse = serde_json::from_str(json).unwrap(); assert_eq!(response.candidates.len(), 1); - assert_eq!(response.candidates[0].content.parts[0].text, "Hello there!"); + assert_eq!( + response.candidates[0].content.parts[0].text.as_deref(), + Some("Hello there!") + ); + } + + #[test] + fn test_gemini_request_with_tools() { + let config = ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }; + let client = GeminiClient { + config, + http: reqwest::Client::new(), + api_key: "test".to_string(), + }; + + let tools = vec![ToolDefinition { + name: "read_file".to_string(), + description: "Read a file".to_string(), + input_schema: serde_json::json!({ + "type": "object", + "properties": {"path": {"type": "string"}}, + "required": ["path"] + }), + }]; + let req = client.build_request(&[Message::human("read it")], &tools); + assert_eq!(req.tools.len(), 1); + assert_eq!(req.tools[0].function_declarations.len(), 1); + assert_eq!(req.tools[0].function_declarations[0].name, "read_file"); + + let json = serde_json::to_string(&req).unwrap(); + assert!(json.contains("functionDeclarations")); + } + + #[test] + fn test_gemini_function_call_response_parsing() { + let json = r#"{ + "candidates": [{ + "content": { + "role": "model", + "parts": [ + {"text": "Let me read that."}, + {"functionCall": {"name": "read_file", "args": {"path": "/tmp/x"}}} + ] + }, + "finishReason": "STOP" + }], + "usageMetadata": {"promptTokenCount": 12, "candidatesTokenCount": 7} + }"#; + + let response: GeminiResponse = serde_json::from_str(json).unwrap(); + let parts = &response.candidates[0].content.parts; + assert_eq!(parts.len(), 2); + assert!(parts[0].text.is_some()); + let fc = parts[1].function_call.as_ref().unwrap(); + assert_eq!(fc.name, "read_file"); + assert_eq!(fc.args, serde_json::json!({"path": "/tmp/x"})); + let usage = response.usage_metadata.unwrap(); + assert_eq!(usage.prompt_token_count, Some(12)); + assert_eq!(usage.candidates_token_count, Some(7)); + } + + #[test] + fn test_gemini_tool_result_roundtrip() { + let config = ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }; + let client = GeminiClient { + config, + http: reqwest::Client::new(), + api_key: "test".to_string(), + }; + + let messages = vec![ + Message::human("read it"), + Message::ai_with_tools( + "", + vec![ToolCall { + id: "gemini_call_0_read_file".to_string(), + name: "read_file".to_string(), + args: serde_json::json!({"path": "/tmp/x"}), + }], + ), + Message::tool("gemini_call_0_read_file", "contents"), + ]; + let req = client.build_request(&messages, &[]); + + // AI turn carries the functionCall part; tool turn carries the + // functionResponse part with the recovered function name. + let model_turn = &req.contents[1]; + assert_eq!(model_turn.role, "model"); + assert!(model_turn.parts[0].function_call.is_some()); + + let tool_turn = &req.contents[2]; + assert_eq!(tool_turn.role, "user"); + let fr = tool_turn.parts[0].function_response.as_ref().unwrap(); + assert_eq!(fr.name, "read_file"); + assert_eq!(fr.response, serde_json::json!({"result": "contents"})); } } diff --git a/crates/rvAgent/rvagent-backends/tests/live_anthropic_test.rs b/crates/rvAgent/rvagent-backends/tests/live_anthropic_test.rs index a945205a9c..7aa3102636 100644 --- a/crates/rvAgent/rvagent-backends/tests/live_anthropic_test.rs +++ b/crates/rvAgent/rvagent-backends/tests/live_anthropic_test.rs @@ -15,7 +15,10 @@ async fn test_live_anthropic_call() { let messages = vec![Message::human("What is 2+2? Reply with just the number.")]; - let response = client.complete(&messages).await.expect("API call failed"); + let response = client + .complete(&messages, &[]) + .await + .expect("API call failed"); let content = response.content(); println!("Response: {}", content); assert!( diff --git a/crates/rvAgent/rvagent-cli/src/app.rs b/crates/rvAgent/rvagent-cli/src/app.rs index 06057dd5e9..4c68be4675 100644 --- a/crates/rvAgent/rvagent-cli/src/app.rs +++ b/crates/rvAgent/rvagent-cli/src/app.rs @@ -13,7 +13,7 @@ use tracing::{info, warn}; use rvagent_core::config::{BackendConfig, MiddlewareConfig, RvAgentConfig, SecurityPolicy}; use rvagent_core::graph::{AgentGraph, ToolExecutor}; use rvagent_core::messages::{Message, ToolCall as CoreToolCall}; -use rvagent_core::models::{resolve_model, ChatModel}; +use rvagent_core::models::{resolve_model, ChatModel, ToolDefinition}; use rvagent_core::prompt::BASE_AGENT_PROMPT; use rvagent_core::state::AgentState; @@ -66,7 +66,11 @@ impl StubModel { #[async_trait] impl ChatModel for StubModel { - async fn complete(&self, _messages: &[Message]) -> rvagent_core::error::Result { + async fn complete( + &self, + _messages: &[Message], + _tools: &[ToolDefinition], + ) -> rvagent_core::error::Result { Ok(Message::ai(format!( "No API key configured for model '{}'. \ Set the appropriate environment variable (e.g. ANTHROPIC_API_KEY) \ @@ -75,8 +79,12 @@ impl ChatModel for StubModel { ))) } - async fn stream(&self, messages: &[Message]) -> rvagent_core::error::Result> { - let msg = self.complete(messages).await?; + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> rvagent_core::error::Result> { + let msg = self.complete(messages, tools).await?; Ok(vec![msg]) } } @@ -95,19 +103,27 @@ enum CliModel { #[async_trait] impl ChatModel for CliModel { - async fn complete(&self, messages: &[Message]) -> rvagent_core::error::Result { + async fn complete( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> rvagent_core::error::Result { match self { - CliModel::Stub(m) => m.complete(messages).await, - CliModel::Anthropic(m) => m.complete(messages).await, - CliModel::Gemini(m) => m.complete(messages).await, + CliModel::Stub(m) => m.complete(messages, tools).await, + CliModel::Anthropic(m) => m.complete(messages, tools).await, + CliModel::Gemini(m) => m.complete(messages, tools).await, } } - async fn stream(&self, messages: &[Message]) -> rvagent_core::error::Result> { + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> rvagent_core::error::Result> { match self { - CliModel::Stub(m) => m.stream(messages).await, - CliModel::Anthropic(m) => m.stream(messages).await, - CliModel::Gemini(m) => m.stream(messages).await, + CliModel::Stub(m) => m.stream(messages, tools).await, + CliModel::Anthropic(m) => m.stream(messages, tools).await, + CliModel::Gemini(m) => m.stream(messages, tools).await, } } } @@ -151,6 +167,17 @@ impl ToolExecutor for CliToolExecutor { None => Ok(format!("Error: tool '{}' not found", call.name)), } } + + fn definitions(&self) -> Vec { + self.tools + .iter() + .map(|t| ToolDefinition { + name: t.name().to_string(), + description: t.description().to_string(), + input_schema: t.parameters_schema(), + }) + .collect() + } } // --------------------------------------------------------------------------- diff --git a/crates/rvAgent/rvagent-core/src/graph.rs b/crates/rvAgent/rvagent-core/src/graph.rs index 2f45bad778..9bdec1accf 100644 --- a/crates/rvAgent/rvagent-core/src/graph.rs +++ b/crates/rvAgent/rvagent-core/src/graph.rs @@ -2,13 +2,16 @@ //! //! Implements the core agent loop: Agent → check tool_calls → execute tools → loop. +use std::sync::Arc; + use async_trait::async_trait; use serde::{Deserialize, Serialize}; use tracing::{debug, info, instrument, warn}; use crate::error::{Result, RvAgentError}; use crate::messages::{Message, ToolCall}; -use crate::models::ChatModel; +use crate::models::{ChatModel, ToolDefinition}; +use crate::parallel::parallel_execute_limited; use crate::state::AgentState; // --------------------------------------------------------------------------- @@ -47,6 +50,15 @@ pub struct Edge { pub trait ToolExecutor: Send + Sync { /// Execute a single tool call and return the result content. async fn execute(&self, call: &ToolCall, state: &AgentState) -> Result; + + /// Schemas for the tools this executor can dispatch. + /// + /// These are advertised to the model on every completion. The default is + /// empty (pure-chat agents), but any executor that dispatches real tools + /// MUST override this — otherwise the model can never call them. + fn definitions(&self) -> Vec { + Vec::new() + } } // --------------------------------------------------------------------------- @@ -60,6 +72,8 @@ pub struct GraphConfig { pub max_iterations: u32, /// Whether to execute tool calls in parallel (ADR-103 A2). pub parallel_tools: bool, + /// Maximum tool calls in flight at once when `parallel_tools` is set. + pub max_parallel_tools: usize, } impl Default for GraphConfig { @@ -67,6 +81,7 @@ impl Default for GraphConfig { Self { max_iterations: 100, parallel_tools: true, + max_parallel_tools: 8, } } } @@ -79,14 +94,16 @@ impl Default for GraphConfig { /// ├── yes → Tools → Agent (loop) /// └── no → End /// ``` -pub struct AgentGraph { +pub struct AgentGraph { model: M, - tool_executor: T, + // Arc so tool calls can be spawned onto the runtime for true parallel + // execution (ADR-103 A2) — spawned futures must be 'static. + tool_executor: Arc, config: GraphConfig, edges: Vec, } -impl AgentGraph { +impl AgentGraph { /// Create a new agent graph with the given model and tool executor. pub fn new(model: M, tool_executor: T) -> Self { Self::with_config(model, tool_executor, GraphConfig::default()) @@ -119,7 +136,7 @@ impl AgentGraph { Self { model, - tool_executor, + tool_executor: Arc::new(tool_executor), config, edges, } @@ -139,8 +156,15 @@ impl AgentGraph { pub async fn run(&self, mut state: AgentState) -> Result { let mut current_node = AgentNode::Start; let mut iterations: u32 = 0; + // Tool schemas advertised to the model on every completion. Without + // these the model can never emit a tool call. + let tool_definitions = self.tool_executor.definitions(); + // Cumulative token usage across the loop, aggregated from per-message + // usage metadata attached by provider backends. + let mut total_input_tokens: u64 = 0; + let mut total_output_tokens: u64 = 0; - info!(node = ?current_node, "graph: starting agent loop"); + info!(node = ?current_node, tools = tool_definitions.len(), "graph: starting agent loop"); loop { if iterations >= self.config.max_iterations { @@ -161,7 +185,21 @@ impl AgentGraph { iterations += 1; debug!(iteration = iterations, "graph: invoking model"); - let response = self.model.complete(&state.messages).await?; + let response = self + .model + .complete(&state.messages, &tool_definitions) + .await?; + if let Some((input, output)) = usage_from_message(&response) { + total_input_tokens += input; + total_output_tokens += output; + debug!( + input_tokens = input, + output_tokens = output, + total_input_tokens, + total_output_tokens, + "graph: turn usage" + ); + } let has_tool_calls = response.has_tool_calls(); state.push_message(response); @@ -180,22 +218,36 @@ impl AgentGraph { // Extract tool calls from the last AI message. let tool_calls = self.extract_tool_calls(&state)?; + // Tool failures are fed back to the model as tool results + // rather than aborting the loop — the model must see the + // error to recover from it (execution alignment). if self.config.parallel_tools && tool_calls.len() > 1 { - // Parallel execution (ADR-103 A2). - let mut handles = Vec::with_capacity(tool_calls.len()); - for tc in &tool_calls { - let result = self.tool_executor.execute(tc, &state).await; - handles.push((tc.id.clone(), result)); - } - for (id, result) in handles { - let content = result?; - state.push_message(Message::tool(id, content)); + // True parallel execution (ADR-103 A2): tasks are + // spawned onto the runtime with bounded concurrency; + // results are returned in input order. + let executor = Arc::clone(&self.tool_executor); + let exec_state = state.clone(); + let results = parallel_execute_limited( + tool_calls.clone(), + move |tc: ToolCall| { + let executor = Arc::clone(&executor); + let exec_state = exec_state.clone(); + async move { + let result = executor.execute(&tc, &exec_state).await; + (tc.id, result) + } + }, + self.config.max_parallel_tools.max(1), + ) + .await; + for (id, result) in results { + state.push_message(Message::tool(id, tool_result_content(result))); } } else { // Sequential execution. for tc in &tool_calls { - let content = self.tool_executor.execute(tc, &state).await?; - state.push_message(Message::tool(&tc.id, content)); + let result = self.tool_executor.execute(tc, &state).await; + state.push_message(Message::tool(&tc.id, tool_result_content(result))); } } @@ -204,7 +256,10 @@ impl AgentGraph { } AgentNode::End => { - info!(iterations, "graph: agent loop complete"); + info!( + iterations, + total_input_tokens, total_output_tokens, "graph: agent loop complete" + ); return Ok(state); } } @@ -226,6 +281,30 @@ impl AgentGraph { } } +/// Convert a tool execution result into tool-message content. +/// +/// Errors become visible tool output instead of aborting the loop — feeding +/// the failure back to the model is the recovery path. +fn tool_result_content(result: Result) -> String { + match result { + Ok(content) => content, + Err(e) => format!("Tool execution error: {e}"), + } +} + +/// Extract `(input_tokens, output_tokens)` from a message's usage metadata, +/// as attached by provider backends under the `usage` key. +fn usage_from_message(msg: &Message) -> Option<(u64, u64)> { + if let Message::Ai(ai) = msg { + let usage = ai.metadata.get("usage")?; + let input = usage.get("input_tokens").and_then(|v| v.as_u64())?; + let output = usage.get("output_tokens").and_then(|v| v.as_u64())?; + Some((input, output)) + } else { + None + } +} + #[cfg(test)] mod tests { use super::*; @@ -246,7 +325,11 @@ mod tests { #[async_trait] impl ChatModel for MockModel { - async fn complete(&self, _messages: &[Message]) -> Result { + async fn complete( + &self, + _messages: &[Message], + _tools: &[ToolDefinition], + ) -> Result { let mut resps = self.responses.lock().unwrap(); if resps.is_empty() { Ok(Message::ai("done")) @@ -255,8 +338,12 @@ mod tests { } } - async fn stream(&self, messages: &[Message]) -> Result> { - let msg = self.complete(messages).await?; + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> Result> { + let msg = self.complete(messages, tools).await?; Ok(vec![msg]) } } @@ -328,6 +415,7 @@ mod tests { let config = GraphConfig { max_iterations: 3, parallel_tools: false, + ..GraphConfig::default() }; let graph = AgentGraph::with_config(model, executor, config); diff --git a/crates/rvAgent/rvagent-core/src/models.rs b/crates/rvAgent/rvagent-core/src/models.rs index 8575186895..aceaf11fa6 100644 --- a/crates/rvAgent/rvagent-core/src/models.rs +++ b/crates/rvAgent/rvagent-core/src/models.rs @@ -131,17 +131,36 @@ pub fn resolve_model(model_str: &str) -> ModelConfig { } } +/// A tool made available to the model for a completion request. +/// +/// Providers translate this into their wire format (Anthropic `tools`, +/// Gemini `functionDeclarations`, OpenAI `functions`). Without advertising +/// these schemas the model can never emit a tool call, so every completion +/// on the agent loop passes the active tool set. +#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)] +pub struct ToolDefinition { + /// Tool name — must match `ToolCall::name` on the way back. + pub name: String, + /// Human-readable description shown to the model. + pub description: String, + /// JSON Schema for the tool's arguments. + pub input_schema: serde_json::Value, +} + /// Async trait for chat model implementations. /// /// Provider-specific crates implement this trait (e.g. `rvagent-anthropic`). #[async_trait] pub trait ChatModel: Send + Sync { - /// Send messages and receive a complete response. - async fn complete(&self, messages: &[Message]) -> Result; + /// Send messages and the active tool set, receive a complete response. + /// + /// `tools` may be empty for pure-chat completions; providers must omit + /// the tools field from the request in that case. + async fn complete(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result; /// Stream a response token-by-token. Returns a vector of incremental messages. /// The final element is the complete assembled message. - async fn stream(&self, messages: &[Message]) -> Result>; + async fn stream(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result>; } /// Extended trait for models that support chunk-based streaming. @@ -152,7 +171,11 @@ pub trait ChatModel: Send + Sync { #[async_trait] pub trait StreamingChatModel: ChatModel { /// Stream response chunks incrementally. - async fn stream_chunks(&self, messages: &[Message]) -> Result>; + async fn stream_chunks( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> Result>; } #[cfg(test)] diff --git a/crates/rvAgent/rvagent-core/tests/integration_tests.rs b/crates/rvAgent/rvagent-core/tests/integration_tests.rs index deee16e87d..e4d85a5a22 100644 --- a/crates/rvAgent/rvagent-core/tests/integration_tests.rs +++ b/crates/rvAgent/rvagent-core/tests/integration_tests.rs @@ -10,7 +10,7 @@ use rvagent_core::config::RvAgentConfig; use rvagent_core::error::{Result, RvAgentError}; use rvagent_core::graph::{AgentGraph, GraphConfig, ToolExecutor}; use rvagent_core::messages::{Message, ToolCall}; -use rvagent_core::models::{ChatModel, Provider}; +use rvagent_core::models::{ChatModel, Provider, ToolDefinition}; use rvagent_core::state::AgentState; // --------------------------------------------------------------------------- @@ -32,7 +32,7 @@ impl MockModel { #[async_trait] impl ChatModel for MockModel { - async fn complete(&self, _messages: &[Message]) -> Result { + async fn complete(&self, _messages: &[Message], _tools: &[ToolDefinition]) -> Result { let mut resps = self.responses.lock().unwrap(); if resps.is_empty() { Ok(Message::ai("(no more responses)")) @@ -41,8 +41,8 @@ impl ChatModel for MockModel { } } - async fn stream(&self, messages: &[Message]) -> Result> { - let msg = self.complete(messages).await?; + async fn stream(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result> { + let msg = self.complete(messages, tools).await?; Ok(vec![msg]) } } @@ -183,6 +183,7 @@ async fn test_agent_graph_with_parallel_tool_calls() { let config = GraphConfig { max_iterations: 10, parallel_tools: true, + ..GraphConfig::default() }; let graph = AgentGraph::with_config(model, executor, config); @@ -255,26 +256,87 @@ fn test_config_to_graph_pipeline() { assert_eq!(edges.len(), 4); } -/// Tool execution failure propagates correctly through the graph. +/// Tool execution failure is fed back to the model as a tool result instead +/// of aborting the loop — the model must see the error to recover from it. #[tokio::test] -async fn test_agent_graph_tool_failure() { - let model = MockModel::new(vec![Message::ai_with_tools( - "", - vec![ToolCall { - id: "tc1".into(), - name: "dangerous_tool".into(), - args: serde_json::json!({}), - }], - )]); +async fn test_agent_graph_tool_failure_feeds_back() { + let model = MockModel::new(vec![ + Message::ai_with_tools( + "", + vec![ToolCall { + id: "tc1".into(), + name: "dangerous_tool".into(), + args: serde_json::json!({}), + }], + ), + // The model sees the error and answers without the tool. + Message::ai("The tool failed; here is a fallback answer."), + ]); let executor = FailingToolExecutor { fail_tool: "dangerous_tool".into(), }; let graph = AgentGraph::new(model, executor); let state = AgentState::new(); - let err = graph.run(state).await.unwrap_err(); - assert!(matches!(err, RvAgentError::Tool(_))); - assert!(err.to_string().contains("dangerous_tool failed")); + let result = graph.run(state).await.unwrap(); + + // The failure surfaced as a tool result, visible to the model. + let tool_msg = result + .messages + .iter() + .find(|m| matches!(m, Message::Tool(_))) + .expect("tool result message must be present"); + assert!(tool_msg.content().contains("Tool execution error")); + assert!(tool_msg.content().contains("dangerous_tool failed")); + + // The loop continued to a final answer instead of aborting. + assert_eq!( + result.messages.last().unwrap().content(), + "The tool failed; here is a fallback answer." + ); +} + +/// A failing tool in a parallel batch does not poison the other results. +#[tokio::test] +async fn test_parallel_tool_failure_isolated() { + let model = MockModel::new(vec![ + Message::ai_with_tools( + "", + vec![ + ToolCall { + id: "ok_call".into(), + name: "safe_tool".into(), + args: serde_json::json!({}), + }, + ToolCall { + id: "bad_call".into(), + name: "dangerous_tool".into(), + args: serde_json::json!({}), + }, + ], + ), + Message::ai("done"), + ]); + let executor = FailingToolExecutor { + fail_tool: "dangerous_tool".into(), + }; + let config = GraphConfig { + max_iterations: 10, + parallel_tools: true, + ..GraphConfig::default() + }; + let graph = AgentGraph::with_config(model, executor, config); + + let result = graph.run(AgentState::new()).await.unwrap(); + let tool_results: Vec<&str> = result + .messages + .iter() + .filter(|m| matches!(m, Message::Tool(_))) + .map(|m| m.content()) + .collect(); + assert_eq!(tool_results.len(), 2); + assert!(tool_results[0].contains("ok: safe_tool")); + assert!(tool_results[1].contains("Tool execution error")); } /// State mutations during graph execution use copy-on-write correctly. @@ -336,6 +398,7 @@ async fn test_max_iterations_terminates() { let config = GraphConfig { max_iterations: 5, parallel_tools: false, + ..GraphConfig::default() }; let graph = AgentGraph::with_config(model, executor, config); From 084a30ed472249dc0d79ce43abcd638cc9bf035d Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 16:03:05 +0000 Subject: [PATCH 03/25] refactor(rvagent): unify type systems and wire async middleware pipeline (P0.1, P0.3) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Roadmap: docs/research/rvagent-hermes-harness/03-roadmap.md Phase 0. P0.1 — one canonical type system: - rvagent-core is now the only source of AgentState/Message/ToolCall/ TodoItem/TodoStatus/RunnableConfig; middleware's duplicate definitions deleted (new types.rs re-exports core + keeps middleware envelopes) - rvagent-subagents' HashMap-based AgentState alias replaced with the typed core AgentState (ADR-103 A1), preserving isolation semantics - ToolMessage gains optional tool_name; graph pushes tool results with the tool name so sanitizer middleware can label output - middleware Tool trait aligned with core ToolDefinition (input_schema(), definition()) P0.3 — middleware actually runs: - Middleware::before_agent/wrap_model_call now async; sync/async duplicate hooks (abefore_agent/awrap_model_call/AsyncModelHandler) deleted; MiddlewarePipeline drives the async chain so real HTTP model calls can run inside it; RetryMiddleware uses tokio sleep - new PipelineModel adapter runs modify_request + the wrap_model_call chain around any inner model - rvagent-cli invoke_agent and rvagent-acp prompt now build the default pipeline from configured middleware names and run all model calls through it (previously zero middleware executed on these paths) Net -242 lines. Verified: full rvAgent workspace test sweep green. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-acp/src/agent.rs | 33 +- crates/rvAgent/rvagent-cli/src/app.rs | 29 +- crates/rvAgent/rvagent-core/src/config.rs | 18 + crates/rvAgent/rvagent-core/src/graph.rs | 16 +- crates/rvAgent/rvagent-core/src/lib.rs | 2 +- crates/rvAgent/rvagent-core/src/messages.rs | 29 + .../benches/middleware_bench.rs | 21 +- .../rvagent-middleware/src/filesystem.rs | 27 +- crates/rvAgent/rvagent-middleware/src/hitl.rs | 41 +- crates/rvAgent/rvagent-middleware/src/hnsw.rs | 15 +- crates/rvAgent/rvagent-middleware/src/lib.rs | 617 +++++------------- .../rvagent-middleware/src/mcp_bridge.rs | 33 +- .../rvAgent/rvagent-middleware/src/memory.rs | 44 +- .../src/patch_tool_calls.rs | 159 ++--- .../rvagent-middleware/src/pipeline.rs | 142 ++++ .../rvagent-middleware/src/pipeline_model.rs | 308 +++++++++ .../rvagent-middleware/src/prompt_caching.rs | 8 +- .../rvAgent/rvagent-middleware/src/retry.rs | 82 +-- .../rvagent-middleware/src/rvf_manifest.rs | 25 +- .../rvAgent/rvagent-middleware/src/skills.rs | 24 +- crates/rvAgent/rvagent-middleware/src/sona.rs | 69 +- .../rvagent-middleware/src/subagents.rs | 31 +- .../rvagent-middleware/src/summarization.rs | 75 ++- .../rvagent-middleware/src/todolist.rs | 87 ++- .../rvagent-middleware/src/tool_sanitizer.rs | 57 +- .../rvAgent/rvagent-middleware/src/types.rs | 188 ++++++ .../src/unicode_security_middleware.rs | 169 ++--- .../rvAgent/rvagent-middleware/src/witness.rs | 36 +- .../rvagent-middleware/tests/hitl_tests.rs | 72 +- .../tests/mcp_bridge_tests.rs | 38 +- .../tests/pipeline_tests.rs | 50 +- .../tests/prompt_caching_tests.rs | 32 +- .../tests/security_tests.rs | 121 ++-- .../tests/summarization_tests.rs | 53 +- .../tests/unicode_security_integration.rs | 140 ++-- crates/rvAgent/rvagent-subagents/src/lib.rs | 148 +++-- .../tests/integration_tests.rs | 97 +-- .../tests/orchestrator_tests.rs | 161 ++--- .../tests/security_validation.rs | 3 +- 39 files changed, 1848 insertions(+), 1452 deletions(-) create mode 100644 crates/rvAgent/rvagent-middleware/src/pipeline.rs create mode 100644 crates/rvAgent/rvagent-middleware/src/pipeline_model.rs create mode 100644 crates/rvAgent/rvagent-middleware/src/types.rs diff --git a/crates/rvAgent/rvagent-acp/src/agent.rs b/crates/rvAgent/rvagent-acp/src/agent.rs index e92379444f..2f31a60c50 100644 --- a/crates/rvAgent/rvagent-acp/src/agent.rs +++ b/crates/rvAgent/rvagent-acp/src/agent.rs @@ -194,21 +194,46 @@ impl AcpAgent { let user_msg = Message::human(&user_text); - // Run the prompt through an AgentGraph with a stub model. + // Run the prompt through an AgentGraph with a stub model wrapped in + // the middleware pipeline (P0.3 wiring). // // In production, the model would be resolved from `self.config` - // and real tools/middleware would be wired in. The stub model - // allows the server to run without an API key. + // and real tools would be wired in. The stub model allows the + // server to run without an API key. let graph_config = GraphConfig { max_iterations: 10, parallel_tools: false, ..GraphConfig::default() }; - let graph = AgentGraph::with_config(StubModel, AcpToolExecutor, graph_config); + + // Resolve the configured middleware names (unknown names warn and + // are skipped); an empty configuration gets the default pipeline. + let pipeline = if self.config.middleware.is_empty() { + rvagent_middleware::build_default_pipeline(&rvagent_middleware::PipelineConfig::default()) + } else { + let names: Vec<&str> = self + .config + .middleware + .iter() + .map(|m| m.name.as_str()) + .collect(); + rvagent_middleware::build_pipeline_from_names(&names) + }; + let pipeline = Arc::new(pipeline); let mut agent_state = AgentState::new(); agent_state.push_message(user_msg.clone()); + // Run before_agent hooks over the initial state. + let mw_runtime = rvagent_middleware::Runtime::new(); + let run_config = rvagent_middleware::RunnableConfig::default(); + pipeline + .run_before_agent(&mut agent_state, &mw_runtime, &run_config) + .await; + + let model = rvagent_middleware::PipelineModel::new(StubModel, Arc::clone(&pipeline)); + let graph = AgentGraph::with_config(model, AcpToolExecutor, graph_config); + let final_state = graph .run(agent_state) .await diff --git a/crates/rvAgent/rvagent-cli/src/app.rs b/crates/rvAgent/rvagent-cli/src/app.rs index 4c68be4675..4d622e23a5 100644 --- a/crates/rvAgent/rvagent-cli/src/app.rs +++ b/crates/rvAgent/rvagent-cli/src/app.rs @@ -665,7 +665,8 @@ impl App { /// Invoke the agent pipeline with the given state. /// /// Creates the appropriate model (real Anthropic client or stub) and - /// tool executor, builds an `AgentGraph`, and runs it to completion. + /// tool executor, wraps the model in the configured middleware pipeline + /// (`PipelineModel`), builds an `AgentGraph`, and runs it to completion. /// Returns the final AI message from the completed state. async fn invoke_agent(&self, initial_state: &AgentState) -> Result { info!( @@ -734,9 +735,33 @@ impl App { CliModel::Stub(StubModel::new(&self.config.model)) }; + // Wire the middleware pipeline (P0.3): resolve the configured + // middleware names (DEFAULT_MIDDLEWARE) into instances — unknown + // names warn and are skipped — and run all model calls through it. + let middleware_names: Vec<&str> = self + .config + .middleware + .iter() + .map(|m| m.name.as_str()) + .collect(); + let pipeline = Arc::new(rvagent_middleware::build_pipeline_from_names( + &middleware_names, + )); + info!(middlewares = ?pipeline.names(), "middleware pipeline wired"); + + // Run before_agent hooks (state patching, context injection). + let mut state = initial_state.clone(); + let mw_runtime = rvagent_middleware::Runtime::new(); + let run_config = rvagent_middleware::RunnableConfig::default(); + pipeline + .run_before_agent(&mut state, &mw_runtime, &run_config) + .await; + + let model = rvagent_middleware::PipelineModel::new(model, Arc::clone(&pipeline)); + let graph = AgentGraph::new(model, tool_executor); let completed_state = graph - .run(initial_state.clone()) + .run(state) .await .map_err(|e| anyhow::anyhow!("agent graph error: {}", e))?; diff --git a/crates/rvAgent/rvagent-core/src/config.rs b/crates/rvAgent/rvagent-core/src/config.rs index 84f2a8f8ec..b284c4134e 100644 --- a/crates/rvAgent/rvagent-core/src/config.rs +++ b/crates/rvAgent/rvagent-core/src/config.rs @@ -169,6 +169,24 @@ fn default_backend_type() -> String { "local_shell".into() } +// --------------------------------------------------------------------------- +// Runnable config (per-run context — canonical definition, ADR-103 A1) +// --------------------------------------------------------------------------- + +/// Configuration for a single runnable invocation (thread/run IDs, metadata). +/// +/// This is the canonical definition shared by the middleware pipeline and any +/// other layer that needs per-run context. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct RunnableConfig { + #[serde(default)] + pub thread_id: Option, + #[serde(default)] + pub run_id: Option, + #[serde(default)] + pub metadata: std::collections::HashMap, +} + // --------------------------------------------------------------------------- // Top-level config // --------------------------------------------------------------------------- diff --git a/crates/rvAgent/rvagent-core/src/graph.rs b/crates/rvAgent/rvagent-core/src/graph.rs index 9bdec1accf..d50c00a50c 100644 --- a/crates/rvAgent/rvagent-core/src/graph.rs +++ b/crates/rvAgent/rvagent-core/src/graph.rs @@ -234,20 +234,28 @@ impl AgentGraph { let exec_state = exec_state.clone(); async move { let result = executor.execute(&tc, &exec_state).await; - (tc.id, result) + (tc.id, tc.name, result) } }, self.config.max_parallel_tools.max(1), ) .await; - for (id, result) in results { - state.push_message(Message::tool(id, tool_result_content(result))); + for (id, name, result) in results { + state.push_message(Message::tool_with_name( + id, + tool_result_content(result), + name, + )); } } else { // Sequential execution. for tc in &tool_calls { let result = self.tool_executor.execute(tc, &state).await; - state.push_message(Message::tool(&tc.id, tool_result_content(result))); + state.push_message(Message::tool_with_name( + &tc.id, + tool_result_content(result), + &tc.name, + )); } } diff --git a/crates/rvAgent/rvagent-core/src/lib.rs b/crates/rvAgent/rvagent-core/src/lib.rs index 9f7d252b9e..416660e17e 100644 --- a/crates/rvAgent/rvagent-core/src/lib.rs +++ b/crates/rvAgent/rvagent-core/src/lib.rs @@ -40,7 +40,7 @@ pub use agi_container::{ OrchestratorConfig, ParsedContainer, SegmentType, SkillDefinition, ToolDefinition, }; pub use budget::{BudgetEnforcer, BudgetError, BudgetUtilization}; -pub use config::{BackendConfig, ResourceBudget, RvAgentConfig, SecurityPolicy}; +pub use config::{BackendConfig, ResourceBudget, RunnableConfig, RvAgentConfig, SecurityPolicy}; pub use cow_state::CowStateBackend; pub use error::{Result, RvAgentError}; pub use graph::{AgentGraph, AgentNode, GraphConfig, ToolExecutor}; diff --git a/crates/rvAgent/rvagent-core/src/messages.rs b/crates/rvAgent/rvagent-core/src/messages.rs index 2f2325e18b..404b0a3ed5 100644 --- a/crates/rvAgent/rvagent-core/src/messages.rs +++ b/crates/rvAgent/rvagent-core/src/messages.rs @@ -49,6 +49,9 @@ pub struct ToolMessage { pub tool_call_id: String, /// The tool's output content. pub content: String, + /// Name of the tool that produced this result (when known). + #[serde(default, skip_serializing_if = "Option::is_none")] + pub tool_name: Option, #[serde(default, skip_serializing_if = "HashMap::is_empty")] pub metadata: HashMap, } @@ -103,6 +106,21 @@ impl Message { Self::Tool(ToolMessage { tool_call_id: tool_call_id.into(), content: content.into(), + tool_name: None, + metadata: HashMap::new(), + }) + } + + /// Create a tool result message that records the tool's name. + pub fn tool_with_name( + tool_call_id: impl Into, + content: impl Into, + tool_name: impl Into, + ) -> Self { + Self::Tool(ToolMessage { + tool_call_id: tool_call_id.into(), + content: content.into(), + tool_name: Some(tool_name.into()), metadata: HashMap::new(), }) } @@ -118,6 +136,17 @@ impl Message { } } + /// Get a mutable reference to the text content of any message variant. + #[inline] + pub fn content_mut(&mut self) -> &mut String { + match self { + Self::System(m) => &mut m.content, + Self::Human(m) => &mut m.content, + Self::Ai(m) => &mut m.content, + Self::Tool(m) => &mut m.content, + } + } + /// Returns true if this is an AI message with pending tool calls. #[inline] pub fn has_tool_calls(&self) -> bool { diff --git a/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs b/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs index 0173e76bce..ce7a5a9501 100644 --- a/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs +++ b/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs @@ -7,6 +7,8 @@ use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion}; +use async_trait::async_trait; + use rvagent_core::rvf_bridge::{GovernanceMode, PolicyCheck, TaskOutcome}; use rvagent_middleware::skills::validate_skill_name; use rvagent_middleware::witness::{compute_arguments_hash, WitnessBuilder}; @@ -17,8 +19,10 @@ use rvagent_middleware::{ /// A no-op handler that returns immediately. struct NoOpHandler; + +#[async_trait] impl ModelHandler for NoOpHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { ModelResponse::text("ok") } } @@ -38,15 +42,20 @@ fn bench_full_pipeline(c: &mut Criterion) { }; let pipeline = build_default_pipeline(&config); let handler = NoOpHandler; + let rt = tokio::runtime::Builder::new_current_thread() + .enable_time() + .build() + .unwrap(); c.bench_function("full_11_middleware_pipeline", |b| { b.iter(|| { let request = ModelRequest::new(vec![ - Message::user("Hello"), - Message::assistant("Hi there"), - Message::user("Write some code"), + Message::human("Hello"), + Message::ai("Hi there"), + Message::human("Write some code"), ]); - let response = pipeline.run_wrap_model_call(black_box(request), &handler); + let response = + rt.block_on(pipeline.run_wrap_model_call(black_box(request), &handler)); black_box(response); }); }); @@ -125,7 +134,7 @@ fn bench_pipeline_modify_request(c: &mut Criterion) { c.bench_function("pipeline_modify_request", |b| { b.iter(|| { - let request = ModelRequest::new(vec![Message::user("test")]) + let request = ModelRequest::new(vec![Message::human("test")]) .with_system(Some("You are helpful.".into())); let modified = pipeline.run_modify_request(black_box(request)); black_box(modified); diff --git a/crates/rvAgent/rvagent-middleware/src/filesystem.rs b/crates/rvAgent/rvagent-middleware/src/filesystem.rs index eb88bf5e61..63fd6332fd 100644 --- a/crates/rvAgent/rvagent-middleware/src/filesystem.rs +++ b/crates/rvAgent/rvagent-middleware/src/filesystem.rs @@ -39,7 +39,7 @@ impl Middleware for FilesystemMiddleware { "filesystem" } - fn before_agent( + async fn before_agent( &self, _state: &AgentState, _runtime: &Runtime, @@ -77,6 +77,7 @@ impl Middleware for FilesystemMiddleware { macro_rules! fs_tool { ($name:ident, $tool_name:expr, $desc:expr, $schema:expr) => { struct $name; + #[async_trait] impl Tool for $name { fn name(&self) -> &str { $tool_name @@ -84,10 +85,10 @@ macro_rules! fs_tool { fn description(&self) -> &str { $desc } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { $schema } - fn invoke(&self, _args: serde_json::Value) -> Result { + async fn invoke(&self, _args: serde_json::Value) -> Result { Err("filesystem tool must be invoked through the agent runtime".into()) } } @@ -225,22 +226,22 @@ mod tests { assert!(names.contains(&"execute")); } - #[test] - fn test_before_agent_no_cwd() { + #[tokio::test] + async fn test_before_agent_no_cwd() { let mw = FilesystemMiddleware::new(); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } - #[test] - fn test_before_agent_with_cwd() { + #[tokio::test] + async fn test_before_agent_with_cwd() { let mw = FilesystemMiddleware::with_cwd("/tmp/test"); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let ext = &update.unwrap().extensions; assert_eq!( @@ -249,11 +250,11 @@ mod tests { ); } - #[test] - fn test_tools_return_error_without_runtime() { + #[tokio::test] + async fn test_tools_return_error_without_runtime() { let mw = FilesystemMiddleware::new(); for tool in mw.tools() { - let result = tool.invoke(serde_json::json!({})); + let result = tool.invoke(serde_json::json!({})).await; assert!(result.is_err()); } } @@ -262,7 +263,7 @@ mod tests { fn test_tool_schemas_are_objects() { let mw = FilesystemMiddleware::new(); for tool in mw.tools() { - let schema = tool.parameters_schema(); + let schema = tool.input_schema(); assert_eq!(schema["type"], "object"); } } diff --git a/crates/rvAgent/rvagent-middleware/src/hitl.rs b/crates/rvAgent/rvagent-middleware/src/hitl.rs index cf0e1fde88..425ee9f683 100644 --- a/crates/rvAgent/rvagent-middleware/src/hitl.rs +++ b/crates/rvAgent/rvagent-middleware/src/hitl.rs @@ -48,8 +48,12 @@ impl Middleware for HumanInTheLoopMiddleware { "hitl" } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { - let mut response = handler.call(request); + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { + let mut response = handler.call(request).await; // Filter out tool calls that require approval let (needs_approval, approved): (Vec, Vec) = response @@ -70,10 +74,11 @@ impl Middleware for HumanInTheLoopMiddleware { pending_names ); - if !response.message.content.is_empty() { - response.message.content.push_str("\n\n"); + let content = response.message.content_mut(); + if !content.is_empty() { + content.push_str("\n\n"); } - response.message.content.push_str(&format!( + content.push_str(&format!( "[HITL] Awaiting approval for: {}", pending_names.join(", ") )); @@ -89,8 +94,10 @@ mod tests { use crate::Message; struct EchoHandler; + + #[async_trait] impl ModelHandler for EchoHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { let mut response = ModelResponse::text("response"); response.tool_calls = vec![ ToolCall { @@ -136,28 +143,28 @@ mod tests { assert!(!mw.should_interrupt("read_file")); } - #[test] - fn test_wrap_model_call_filters_tool_calls() { + #[tokio::test] + async fn test_wrap_model_call_filters_tool_calls() { let mw = HumanInTheLoopMiddleware::new(vec!["execute".into()]); - let request = ModelRequest::new(vec![Message::user("do something")]); + let request = ModelRequest::new(vec![Message::human("do something")]); let handler = EchoHandler; - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; assert_eq!(response.tool_calls.len(), 1); assert_eq!(response.tool_calls[0].name, "read_file"); - assert!(response.message.content.contains("[HITL]")); - assert!(response.message.content.contains("execute")); + assert!(response.content().contains("[HITL]")); + assert!(response.content().contains("execute")); } - #[test] - fn test_wrap_model_call_no_interrupt() { + #[tokio::test] + async fn test_wrap_model_call_no_interrupt() { let mw = HumanInTheLoopMiddleware::new(vec!["dangerous_tool".into()]); - let request = ModelRequest::new(vec![Message::user("safe")]); + let request = ModelRequest::new(vec![Message::human("safe")]); let handler = EchoHandler; - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; assert_eq!(response.tool_calls.len(), 2); - assert!(!response.message.content.contains("[HITL]")); + assert!(!response.content().contains("[HITL]")); } #[test] diff --git a/crates/rvAgent/rvagent-middleware/src/hnsw.rs b/crates/rvAgent/rvagent-middleware/src/hnsw.rs index 07394d0892..c5b70e3986 100644 --- a/crates/rvAgent/rvagent-middleware/src/hnsw.rs +++ b/crates/rvAgent/rvagent-middleware/src/hnsw.rs @@ -17,7 +17,8 @@ //! - Sub-millisecond latency for 10k vectors use crate::{ - AgentState, AgentStateUpdate, Middleware, ModelRequest, RunnableConfig, Runtime, ToolDefinition, + AgentState, AgentStateUpdate, Message, Middleware, ModelRequest, RunnableConfig, Runtime, + ToolDefinition, }; use async_trait::async_trait; use parking_lot::RwLock; @@ -722,7 +723,7 @@ impl HnswMiddleware { Some(ToolDefinition { name, description, - parameters, + input_schema: parameters, }) }) .collect() @@ -735,7 +736,7 @@ impl Middleware for HnswMiddleware { "hnsw" } - fn before_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, @@ -750,11 +751,11 @@ impl Middleware for HnswMiddleware { .messages .iter() .rev() - .find(|m| matches!(m.role, crate::Role::User))?; + .find(|m| matches!(m, Message::Human(_)))?; // Search for relevant memory let memory_results = self.search_memory( - &last_user.content, + last_user.content(), self.state.read().config.memory_retrieval_k, ); @@ -794,8 +795,8 @@ impl Middleware for HnswMiddleware { .messages .iter() .rev() - .find(|m| matches!(m.role, crate::Role::User)) - .map(|m| m.content.clone()); + .find(|m| matches!(m, Message::Human(_))) + .map(|m| m.content().to_string()); if let Some(query) = query { // Retrieve relevant skills as tools diff --git a/crates/rvAgent/rvagent-middleware/src/lib.rs b/crates/rvAgent/rvagent-middleware/src/lib.rs index 5153402e8a..f005f0ab32 100644 --- a/crates/rvAgent/rvagent-middleware/src/lib.rs +++ b/crates/rvAgent/rvagent-middleware/src/lib.rs @@ -3,6 +3,18 @@ //! Provides the `Middleware` trait and `MiddlewarePipeline` for composing middleware //! in the DeepAgents architecture (ADR-095, ADR-103). //! +//! ## Unified type system (P0.1) +//! +//! All conversation/state types (`Message`, `ToolCall`, `AgentState`, `TodoItem`, +//! `TodoStatus`, `RunnableConfig`, `ToolDefinition`) are the canonical +//! `rvagent-core` definitions, re-exported here for convenience. +//! +//! ## Async model-call chain (P0.3) +//! +//! `Middleware::wrap_model_call` and `ModelHandler::call` are async, so a real +//! HTTP model call can run inside the pipeline. Use [`PipelineModel`] to wrap +//! any `rvagent_core::models::ChatModel` with a pipeline. +//! //! ## ADR-103 Learning Middleware (B5, B6) //! //! - [`sona`] — SONA Adaptive Learning with three loops (instant, background, deep) @@ -14,6 +26,8 @@ pub mod hnsw; pub mod mcp_bridge; pub mod memory; pub mod patch_tool_calls; +pub mod pipeline; +pub mod pipeline_model; pub mod prompt_caching; pub mod retry; pub mod rvf_manifest; @@ -23,276 +37,63 @@ pub mod subagents; pub mod summarization; pub mod todolist; pub mod tool_sanitizer; +pub mod types; pub mod unicode_security; pub mod unicode_security_middleware; pub mod utils; pub mod witness; use async_trait::async_trait; -use serde::{Deserialize, Serialize}; -use std::collections::HashMap; use std::fmt; // Re-exports +pub use pipeline::MiddlewarePipeline; +pub use pipeline_model::PipelineModel; +pub use types::{ + json_extension, AgentState, AgentStateUpdate, AiMessage, CacheControl, FileData, HumanMessage, + Message, ModelRequest, ModelResponse, RunnableConfig, Runtime, SystemMessage, TodoItem, + TodoStatus, ToolCall, ToolDefinition, ToolMessage, Usage, +}; pub use unicode_security::{UnicodeIssue, UnicodeSecurityChecker, UnicodeSecurityConfig}; pub use unicode_security_middleware::UnicodeSecurityMiddleware; pub use utils::{append_to_system_message, SystemPromptBuilder}; // --------------------------------------------------------------------------- -// Core types +// Model handler trait (async — P0.3) // --------------------------------------------------------------------------- -/// Message role in a conversation. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "lowercase")] -pub enum Role { - System, - User, - Assistant, - Tool, -} - -/// A single tool call within an assistant message. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct ToolCall { - pub id: String, - pub name: String, - pub args: serde_json::Value, -} - -/// A conversation message. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct Message { - pub role: Role, - pub content: String, - #[serde(default, skip_serializing_if = "Vec::is_empty")] - pub tool_calls: Vec, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub tool_call_id: Option, - #[serde(default, skip_serializing_if = "Option::is_none")] - pub tool_name: Option, -} - -impl Message { - pub fn system(content: impl Into) -> Self { - Self { - role: Role::System, - content: content.into(), - tool_calls: vec![], - tool_call_id: None, - tool_name: None, - } - } - - pub fn user(content: impl Into) -> Self { - Self { - role: Role::User, - content: content.into(), - tool_calls: vec![], - tool_call_id: None, - tool_name: None, - } - } - - pub fn assistant(content: impl Into) -> Self { - Self { - role: Role::Assistant, - content: content.into(), - tool_calls: vec![], - tool_call_id: None, - tool_name: None, - } - } - - pub fn tool( - content: impl Into, - tool_call_id: impl Into, - name: impl Into, - ) -> Self { - Self { - role: Role::Tool, - content: content.into(), - tool_calls: vec![], - tool_call_id: Some(tool_call_id.into()), - tool_name: Some(name.into()), - } - } -} - -/// Cache control hint for prompt caching (Anthropic). -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct CacheControl { - pub cache_type: String, -} - -/// Agent state — typed structure (ADR-103 A1) with extension map. -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -pub struct AgentState { - pub messages: Vec, - #[serde(default)] - pub todos: Vec, - #[serde(default)] - pub extensions: HashMap, -} - -/// A single todo item managed by TodoListMiddleware. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct TodoItem { - pub id: String, - pub content: String, - #[serde(default)] - pub status: TodoStatus, -} - -/// Status of a todo item. -#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] -#[serde(rename_all = "lowercase")] -pub enum TodoStatus { - Pending, - InProgress, - Completed, -} - -impl Default for TodoStatus { - fn default() -> Self { - Self::Pending - } -} - -/// State update returned by `before_agent`. Merged into `AgentState`. -#[derive(Debug, Clone, Default)] -pub struct AgentStateUpdate { - pub messages: Option>, - pub todos: Option>, - pub extensions: HashMap, -} - -/// Model request wrapping messages and configuration. -#[derive(Debug, Clone)] -pub struct ModelRequest { - pub system_message: Option, - pub messages: Vec, - pub tools: Vec, - pub cache_control: HashMap, - pub extensions: HashMap, -} - -impl ModelRequest { - /// Create a new model request. - pub fn new(messages: Vec) -> Self { - Self { - system_message: None, - messages, - tools: vec![], - cache_control: HashMap::new(), - extensions: HashMap::new(), - } - } - - /// Return a copy with a different system message. - pub fn with_system(mut self, system_message: Option) -> Self { - self.system_message = system_message; - self - } - - /// Return a copy with different messages. - pub fn with_messages(mut self, messages: Vec) -> Self { - self.messages = messages; - self - } -} - -/// Model response from an LLM call. -#[derive(Debug, Clone)] -pub struct ModelResponse { - pub message: Message, - pub tool_calls: Vec, - pub usage: Option, -} - -impl ModelResponse { - /// Create a simple text response. - pub fn text(content: impl Into) -> Self { - Self { - message: Message::assistant(content), - tool_calls: vec![], - usage: None, - } - } -} - -/// Token usage information. -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -pub struct Usage { - pub input_tokens: u64, - pub output_tokens: u64, - #[serde(default)] - pub cache_read_tokens: u64, - #[serde(default)] - pub cache_creation_tokens: u64, -} - -/// Tool definition for model requests. -#[derive(Debug, Clone, Serialize, Deserialize)] -pub struct ToolDefinition { - pub name: String, - pub description: String, - pub parameters: serde_json::Value, -} - -/// Runtime context passed to middleware hooks. -pub struct Runtime { - pub context: serde_json::Value, - pub config: RunnableConfig, -} - -impl Runtime { - pub fn new() -> Self { - Self { - context: serde_json::Value::Null, - config: RunnableConfig::default(), - } - } -} - -impl Default for Runtime { - fn default() -> Self { - Self::new() - } -} - -/// Configuration for a runnable (thread/run IDs, metadata). -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -pub struct RunnableConfig { - #[serde(default)] - pub thread_id: Option, - #[serde(default)] - pub run_id: Option, - #[serde(default)] - pub metadata: HashMap, +/// Async model handler — the "next" link called by `wrap_model_call`. +#[async_trait] +pub trait ModelHandler: Send + Sync { + async fn call(&self, request: ModelRequest) -> ModelResponse; } // --------------------------------------------------------------------------- -// Model handler traits +// Tool trait (aligned with rvagent_core::models::ToolDefinition) // --------------------------------------------------------------------------- -/// Synchronous model handler — called by `wrap_model_call`. -pub trait ModelHandler: Send + Sync { - fn call(&self, request: ModelRequest) -> ModelResponse; -} - -/// Async model handler — called by `awrap_model_call`. -#[async_trait] -pub trait AsyncModelHandler: Send + Sync { - async fn call(&self, request: ModelRequest) -> ModelResponse; -} - /// Tool trait — tools injected by middleware. +/// +/// Schema exposure aligns with `rvagent_core::models::ToolDefinition` +/// (`input_schema`); `definition()` produces the canonical form. +#[async_trait] pub trait Tool: Send + Sync { fn name(&self) -> &str; fn description(&self) -> &str; - fn parameters_schema(&self) -> serde_json::Value; - fn invoke(&self, args: serde_json::Value) -> Result; + /// JSON Schema for the tool's arguments (ToolDefinition::input_schema). + fn input_schema(&self) -> serde_json::Value; + + /// Canonical schema form advertised to models. + fn definition(&self) -> ToolDefinition { + ToolDefinition { + name: self.name().to_string(), + description: self.description().to_string(), + input_schema: self.input_schema(), + } + } + + /// Execute the tool asynchronously. + async fn invoke(&self, args: serde_json::Value) -> Result; } impl fmt::Debug for dyn Tool { @@ -302,17 +103,19 @@ impl fmt::Debug for dyn Tool { } // --------------------------------------------------------------------------- -// Middleware trait (ADR-095) +// Middleware trait (ADR-095) — async hooks (P0.3) // --------------------------------------------------------------------------- /// Core middleware trait — mirrors Python's `AgentMiddleware`. /// /// Each method has a default no-op implementation, so concrete middleware -/// only needs to override the hooks it uses. +/// only needs to override the hooks it uses. The former sync/async duplicate +/// hook pairs (`before_agent`/`abefore_agent`, `wrap_model_call`/ +/// `awrap_model_call`) are merged into single async hooks. #[async_trait] pub trait Middleware: Send + Sync { /// Called before agent execution. Returns state update or None. - fn before_agent( + async fn before_agent( &self, _state: &AgentState, _runtime: &Runtime, @@ -321,26 +124,12 @@ pub trait Middleware: Send + Sync { None } - /// Async version of `before_agent`. - async fn abefore_agent( - &self, - state: &AgentState, - runtime: &Runtime, - config: &RunnableConfig, - ) -> Option { - self.before_agent(state, runtime, config) - } - - /// Wrap a synchronous model call — intercept request/response. - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { - handler.call(request) - } - - /// Wrap an async model call. - async fn awrap_model_call( + /// Wrap a model call — intercept request/response. Async so real HTTP + /// calls can run inside the chain. + async fn wrap_model_call( &self, request: ModelRequest, - handler: &dyn AsyncModelHandler, + handler: &dyn ModelHandler, ) -> ModelResponse { handler.call(request).await } @@ -367,153 +156,6 @@ impl fmt::Debug for dyn Middleware { } } -// --------------------------------------------------------------------------- -// Middleware Pipeline (ADR-095) -// --------------------------------------------------------------------------- - -/// Executes the middleware pipeline in order. -/// Mirrors LangChain's `create_agent` middleware composition. -pub struct MiddlewarePipeline { - middlewares: Vec>, -} - -impl MiddlewarePipeline { - /// Create a new pipeline from an ordered list of middlewares. - pub fn new(middlewares: Vec>) -> Self { - Self { middlewares } - } - - /// Create an empty pipeline. - pub fn empty() -> Self { - Self { - middlewares: Vec::new(), - } - } - - /// Add a middleware to the end of the pipeline. - pub fn push(&mut self, middleware: Box) { - self.middlewares.push(middleware); - } - - /// Number of middlewares in the pipeline. - pub fn len(&self) -> usize { - self.middlewares.len() - } - - /// Whether the pipeline is empty. - pub fn is_empty(&self) -> bool { - self.middlewares.is_empty() - } - - /// Get middleware names in order. - pub fn names(&self) -> Vec<&str> { - self.middlewares.iter().map(|mw| mw.name()).collect() - } - - /// Run `before_agent` hooks in order, accumulating state updates. - pub async fn run_before_agent( - &self, - state: &mut AgentState, - runtime: &Runtime, - config: &RunnableConfig, - ) { - for mw in &self.middlewares { - if let Some(update) = mw.abefore_agent(state, runtime, config).await { - // Merge update into state - if let Some(messages) = update.messages { - state.messages = messages; - } - if let Some(todos) = update.todos { - state.todos = todos; - } - for (k, v) in update.extensions { - state.extensions.insert(k, v); - } - } - } - } - - /// Collect all tools from all middlewares. - pub fn collect_tools(&self) -> Vec> { - self.middlewares.iter().flat_map(|mw| mw.tools()).collect() - } - - /// Run `modify_request` through all middlewares in order. - pub fn run_modify_request(&self, mut request: ModelRequest) -> ModelRequest { - for mw in &self.middlewares { - request = mw.modify_request(request); - } - request - } - - /// Run `wrap_model_call` through the pipeline. - /// Middlewares are chained so the outermost (first) wraps the innermost (last). - pub fn run_wrap_model_call( - &self, - request: ModelRequest, - base_handler: &dyn ModelHandler, - ) -> ModelResponse { - if self.middlewares.is_empty() { - return base_handler.call(request); - } - - // Build chain from inside out using recursive approach. - fn chain_call<'a>( - middlewares: &'a [Box], - request: ModelRequest, - handler: &'a dyn ModelHandler, - ) -> ModelResponse { - if middlewares.is_empty() { - return handler.call(request); - } - let (first, rest) = middlewares.split_first().unwrap(); - let inner = ChainedInner { rest, handler }; - first.wrap_model_call(request, &inner) - } - - struct ChainedInner<'a> { - rest: &'a [Box], - handler: &'a dyn ModelHandler, - } - impl<'a> ModelHandler for ChainedInner<'a> { - fn call(&self, request: ModelRequest) -> ModelResponse { - chain_call(self.rest, request, self.handler) - } - } - - chain_call(&self.middlewares, request, base_handler) - } - - /// Full pipeline run: before_agent, collect tools, modify_request, wrap_model_call. - pub async fn run( - &self, - state: &mut AgentState, - runtime: &Runtime, - config: &RunnableConfig, - mut request: ModelRequest, - handler: &dyn ModelHandler, - ) -> ModelResponse { - // 1. Run before_agent hooks - self.run_before_agent(state, runtime, config).await; - - // 2. Collect tools from all middlewares - let tools: Vec> = self.collect_tools(); - for tool in &tools { - request.tools.push(ToolDefinition { - name: tool.name().to_string(), - description: tool.description().to_string(), - parameters: tool.parameters_schema(), - }); - } - - // 3. Run modify_request - request = self.run_modify_request(request); - - // 4. Run wrap_model_call chain - self.run_wrap_model_call(request, handler) - } -} - // --------------------------------------------------------------------------- // Default pipeline builder (ADR-095) // --------------------------------------------------------------------------- @@ -614,6 +256,61 @@ pub fn build_default_pipeline(config: &PipelineConfig) -> MiddlewarePipeline { MiddlewarePipeline::new(middlewares) } +// --------------------------------------------------------------------------- +// Name-based middleware resolution (CLI wiring) +// --------------------------------------------------------------------------- + +/// Resolve a middleware name (as used in `RvAgentConfig::middleware` / +/// the CLI `DEFAULT_MIDDLEWARE` list) into a middleware instance. +/// +/// Returns `None` for unknown names — callers should warn and skip. +pub fn middleware_by_name(name: &str) -> Option> { + match name { + "todo" | "todos" | "todolist" => Some(Box::new(todolist::TodoListMiddleware::new())), + "memory" => Some(Box::new(memory::MemoryMiddleware::new(vec![ + "AGENTS.md".into() + ]))), + "skills" => Some(Box::new(skills::SkillsMiddleware::new(vec![ + ".skills".into() + ]))), + "filesystem" => Some(Box::new(filesystem::FilesystemMiddleware::new())), + "subagent" | "subagents" => Some(Box::new(subagents::SubAgentMiddleware::new())), + "summarization" => Some(Box::new(summarization::SummarizationMiddleware::new( + 100_000, 0.85, 0.10, + ))), + "prompt_caching" => Some(Box::new(prompt_caching::PromptCachingMiddleware::new())), + "patch_tool_calls" => Some(Box::new(patch_tool_calls::PatchToolCallsMiddleware::new())), + "witness" => Some(Box::new(witness::WitnessMiddleware::new())), + "tool_result_sanitizer" | "tool_sanitizer" => Some(Box::new( + tool_sanitizer::ToolResultSanitizerMiddleware::new(), + )), + // HITL with no interrupt patterns configured never interrupts. + "hitl" => Some(Box::new(hitl::HumanInTheLoopMiddleware::new(Vec::new()))), + "retry" => Some(Box::new(retry::RetryMiddleware::default())), + "hnsw" => Some(Box::new(hnsw::HnswMiddleware::default_config())), + "sona" => Some(Box::new(sona::SonaMiddleware::default_config())), + "unicode_security" => Some(Box::new(UnicodeSecurityMiddleware::strict())), + "mcp_bridge" => Some(Box::new(mcp_bridge::McpBridgeMiddleware::new())), + _ => None, + } +} + +/// Build a pipeline from an ordered list of middleware names. +/// Unknown names are logged (warn) and skipped. +pub fn build_pipeline_from_names>(names: &[S]) -> MiddlewarePipeline { + let mut pipeline = MiddlewarePipeline::empty(); + for name in names { + match middleware_by_name(name.as_ref()) { + Some(mw) => pipeline.push(mw), + None => tracing::warn!( + "unknown middleware '{}' — skipping (see rvagent_middleware::middleware_by_name)", + name.as_ref() + ), + } + } + pipeline +} + // --------------------------------------------------------------------------- // Tests // --------------------------------------------------------------------------- @@ -624,8 +321,10 @@ mod tests { /// A passthrough test handler. struct EchoHandler; + + #[async_trait] impl ModelHandler for EchoHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(format!("echo: {}", request.messages.len())) } } @@ -646,19 +345,21 @@ mod tests { fn name(&self) -> &str { "prepend" } - fn wrap_model_call( + async fn wrap_model_call( &self, request: ModelRequest, handler: &dyn ModelHandler, ) -> ModelResponse { let new_sys = append_to_system_message(&request.system_message, &self.text); - handler.call(request.with_system(new_sys)) + handler.call(request.with_system(new_sys)).await } } /// A test middleware that injects a tool. struct ToolInjector; struct DummyTool; + + #[async_trait] impl Tool for DummyTool { fn name(&self) -> &str { "dummy_tool" @@ -666,10 +367,10 @@ mod tests { fn description(&self) -> &str { "A dummy tool" } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { serde_json::json!({}) } - fn invoke(&self, _args: serde_json::Value) -> Result { + async fn invoke(&self, _args: serde_json::Value) -> Result { Ok("ok".into()) } } @@ -683,27 +384,6 @@ mod tests { } } - #[test] - fn test_message_constructors() { - let sys = Message::system("sys"); - assert_eq!(sys.role, Role::System); - let usr = Message::user("hi"); - assert_eq!(usr.role, Role::User); - let asst = Message::assistant("resp"); - assert_eq!(asst.role, Role::Assistant); - let tool = Message::tool("result", "tc-1", "my_tool"); - assert_eq!(tool.role, Role::Tool); - assert_eq!(tool.tool_call_id.as_deref(), Some("tc-1")); - } - - #[test] - fn test_model_request_with_system() { - let req = ModelRequest::new(vec![Message::user("hi")]); - assert!(req.system_message.is_none()); - let req2 = req.with_system(Some("system".into())); - assert_eq!(req2.system_message, Some("system".into())); - } - #[test] fn test_empty_pipeline() { let pipeline = MiddlewarePipeline::empty(); @@ -722,29 +402,32 @@ mod tests { assert_eq!(pipeline.len(), 2); } - #[test] - fn test_pipeline_wrap_model_call_chaining() { + #[tokio::test] + async fn test_pipeline_wrap_model_call_chaining() { // Two prepend middlewares should chain: first wraps second wraps handler let pipeline = MiddlewarePipeline::new(vec![ Box::new(PrependMiddleware::new("A")), Box::new(PrependMiddleware::new("B")), ]); - let request = ModelRequest::new(vec![Message::user("hi")]).with_system(Some("base".into())); + let request = + ModelRequest::new(vec![Message::human("hi")]).with_system(Some("base".into())); // Track what system message the handler receives struct CaptureHandler; + + #[async_trait] impl ModelHandler for CaptureHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(request.system_message.unwrap_or_default()) } } - let response = pipeline.run_wrap_model_call(request, &CaptureHandler); + let response = pipeline.run_wrap_model_call(request, &CaptureHandler).await; // First middleware appends A, second appends B - assert!(response.message.content.contains("A")); - assert!(response.message.content.contains("B")); - assert!(response.message.content.contains("base")); + assert!(response.content().contains("A")); + assert!(response.content().contains("B")); + assert!(response.content().contains("base")); } #[test] @@ -754,6 +437,8 @@ mod tests { let tools = pipeline.collect_tools(); assert_eq!(tools.len(), 2); assert_eq!(tools[0].name(), "dummy_tool"); + let def = tools[0].definition(); + assert_eq!(def.name, "dummy_tool"); } #[tokio::test] @@ -766,12 +451,12 @@ mod tests { let mut state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let request = ModelRequest::new(vec![Message::user("test")]); + let request = ModelRequest::new(vec![Message::human("test")]); let response = pipeline .run(&mut state, &runtime, &config, request, &EchoHandler) .await; - assert!(response.message.content.contains("echo")); + assert!(response.content().contains("echo")); } #[test] @@ -808,26 +493,36 @@ mod tests { let state = AgentState::default(); assert!(state.messages.is_empty()); assert!(state.todos.is_empty()); - assert!(state.extensions.is_empty()); } #[test] - fn test_todo_status_default() { - let status = TodoStatus::default(); - assert_eq!(status, TodoStatus::Pending); + fn test_middleware_by_name_known() { + for name in [ + "todo", + "memory", + "skills", + "filesystem", + "subagent", + "summarization", + "prompt_caching", + "patch_tool_calls", + "witness", + "tool_result_sanitizer", + "hitl", + "retry", + ] { + assert!(middleware_by_name(name).is_some(), "should resolve {name}"); + } } #[test] - fn test_model_response_text() { - let resp = ModelResponse::text("hello"); - assert_eq!(resp.message.content, "hello"); - assert_eq!(resp.message.role, Role::Assistant); - assert!(resp.tool_calls.is_empty()); + fn test_middleware_by_name_unknown() { + assert!(middleware_by_name("does_not_exist").is_none()); } #[test] - fn test_runtime_default() { - let rt = Runtime::default(); - assert_eq!(rt.context, serde_json::Value::Null); + fn test_build_pipeline_from_names_skips_unknown() { + let pipeline = build_pipeline_from_names(&["todo", "bogus", "filesystem"]); + assert_eq!(pipeline.len(), 2); } } diff --git a/crates/rvAgent/rvagent-middleware/src/mcp_bridge.rs b/crates/rvAgent/rvagent-middleware/src/mcp_bridge.rs index 3b12e5f4ea..63da89a973 100644 --- a/crates/rvAgent/rvagent-middleware/src/mcp_bridge.rs +++ b/crates/rvAgent/rvagent-middleware/src/mcp_bridge.rs @@ -6,10 +6,7 @@ use async_trait::async_trait; use serde::{Deserialize, Serialize}; -use crate::{ - AgentState, AgentStateUpdate, Middleware, ModelHandler, ModelRequest, ModelResponse, - RunnableConfig, Runtime, -}; +use crate::{AgentState, AgentStateUpdate, Middleware, ModelRequest, RunnableConfig, Runtime}; /// MCP tool call origin tracking. #[derive(Debug, Clone, Serialize, Deserialize)] @@ -81,7 +78,7 @@ impl Middleware for McpBridgeMiddleware { "mcp_bridge" } - fn before_agent( + async fn before_agent( &self, _state: &AgentState, _runtime: &Runtime, @@ -109,10 +106,6 @@ impl Middleware for McpBridgeMiddleware { request } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { - handler.call(request) - } - fn tools(&self) -> Vec> { if !self.config.enabled { return vec![]; @@ -128,6 +121,7 @@ struct McpStatusTool { config: McpBridgeConfig, } +#[async_trait] impl crate::Tool for McpStatusTool { fn name(&self) -> &str { "mcp_bridge_status" @@ -137,7 +131,7 @@ impl crate::Tool for McpStatusTool { "Returns the current MCP bridge configuration and status" } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { serde_json::json!({ "type": "object", "properties": {}, @@ -145,7 +139,7 @@ impl crate::Tool for McpStatusTool { }) } - fn invoke(&self, _args: serde_json::Value) -> Result { + async fn invoke(&self, _args: serde_json::Value) -> Result { Ok(serde_json::json!({ "enabled": self.config.enabled, "max_concurrent": self.config.max_concurrent, @@ -195,8 +189,8 @@ mod tests { assert!(!mw.is_transport_allowed("websocket")); } - #[test] - fn test_mcp_bridge_disabled() { + #[tokio::test] + async fn test_mcp_bridge_disabled() { let config = McpBridgeConfig { enabled: false, ..Default::default() @@ -207,17 +201,18 @@ mod tests { let runnable_config = RunnableConfig::default(); assert!(mw .before_agent(&state, &runtime, &runnable_config) + .await .is_none()); assert!(mw.tools().is_empty()); } - #[test] - fn test_mcp_bridge_enabled_injects_config() { + #[tokio::test] + async fn test_mcp_bridge_enabled_injects_config() { let mw = McpBridgeMiddleware::new(); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); assert!(update.unwrap().extensions.contains_key("mcp_bridge_config")); } @@ -230,13 +225,13 @@ mod tests { assert_eq!(tools[0].name(), "mcp_bridge_status"); } - #[test] - fn test_mcp_status_tool_invoke() { + #[tokio::test] + async fn test_mcp_status_tool_invoke() { use crate::Tool; let tool = McpStatusTool { config: McpBridgeConfig::default(), }; - let result = tool.invoke(serde_json::json!({})); + let result = tool.invoke(serde_json::json!({})).await; assert!(result.is_ok()); let json: serde_json::Value = serde_json::from_str(&result.unwrap()).unwrap(); assert_eq!(json["enabled"], true); diff --git a/crates/rvAgent/rvagent-middleware/src/memory.rs b/crates/rvAgent/rvagent-middleware/src/memory.rs index c73d692d9f..de22ee4705 100644 --- a/crates/rvAgent/rvagent-middleware/src/memory.rs +++ b/crates/rvAgent/rvagent-middleware/src/memory.rs @@ -201,13 +201,13 @@ impl Middleware for MemoryMiddleware { "memory" } - fn before_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, _config: &RunnableConfig, ) -> Option { - if state.extensions.contains_key("memory_contents") { + if crate::json_extension(state, "memory_contents").is_some() { return None; } @@ -231,7 +231,11 @@ impl Middleware for MemoryMiddleware { Some(update) } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { let contents: HashMap = request .extensions .get("memory_contents") @@ -239,12 +243,12 @@ impl Middleware for MemoryMiddleware { .unwrap_or_default(); if contents.is_empty() { - return handler.call(request); + return handler.call(request).await; } let memory_section = Self::format_agent_memory(&contents); let new_system = crate::append_to_system_message(&request.system_message, &memory_section); - handler.call(request.with_system(new_system)) + handler.call(request.with_system(new_system)).await } } @@ -252,9 +256,13 @@ impl Middleware for MemoryMiddleware { mod tests { use super::*; + use async_trait::async_trait; + struct PassthroughHandler; + + #[async_trait] impl ModelHandler for PassthroughHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(request.system_message.unwrap_or_default()) } } @@ -338,20 +346,18 @@ mod tests { assert!(mw.validate_content("any.md", "anything").is_some()); } - #[test] - fn test_before_agent_skip_if_loaded() { + #[tokio::test] + async fn test_before_agent_skip_if_loaded() { let mw = MemoryMiddleware::new(vec!["AGENTS.md".into()]); let mut state = AgentState::default(); - state - .extensions - .insert("memory_contents".into(), serde_json::json!({})); + state.set_extension("memory_contents", serde_json::json!({})); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } - #[test] - fn test_before_agent_loads() { + #[tokio::test] + async fn test_before_agent_loads() { let mut preloaded = HashMap::new(); preloaded.insert("AGENTS.md".into(), "Memory content".into()); @@ -360,7 +366,7 @@ mod tests { let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); assert!(update.unwrap().extensions.contains_key("memory_contents")); } @@ -375,13 +381,13 @@ mod tests { assert!(formatted.contains("")); } - #[test] - fn test_wrap_model_call_no_memory() { + #[tokio::test] + async fn test_wrap_model_call_no_memory() { let mw = MemoryMiddleware::new(vec![]); let request = ModelRequest::new(vec![]); let handler = PassthroughHandler; - let response = mw.wrap_model_call(request, &handler); - assert!(response.message.content.is_empty()); + let response = mw.wrap_model_call(request, &handler).await; + assert!(response.content().is_empty()); } #[test] diff --git a/crates/rvAgent/rvagent-middleware/src/patch_tool_calls.rs b/crates/rvAgent/rvagent-middleware/src/patch_tool_calls.rs index cee92f28bb..ab6b8b4de5 100644 --- a/crates/rvAgent/rvagent-middleware/src/patch_tool_calls.rs +++ b/crates/rvAgent/rvagent-middleware/src/patch_tool_calls.rs @@ -3,7 +3,7 @@ use async_trait::async_trait; -use crate::{AgentState, AgentStateUpdate, Message, Middleware, Role, RunnableConfig, Runtime}; +use crate::{AgentState, AgentStateUpdate, Message, Middleware, RunnableConfig, Runtime}; /// Maximum length for tool call IDs (ADR-103 C12). pub const MAX_TOOL_CALL_ID_LENGTH: usize = 128; @@ -50,7 +50,7 @@ impl Middleware for PatchToolCallsMiddleware { "patch_tool_calls" } - fn before_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, @@ -66,8 +66,8 @@ impl Middleware for PatchToolCallsMiddleware { for (i, msg) in state.messages.iter().enumerate() { patched.push(msg.clone()); - if msg.role == Role::Assistant && !msg.tool_calls.is_empty() { - for tc in &msg.tool_calls { + if let Message::Ai(ai) = msg { + for tc in &ai.tool_calls { // Validate tool call ID (ADR-103 C12) if let Err(err) = validate_tool_call_id(&tc.id) { tracing::warn!("Invalid tool call ID '{}': {}", tc.id, err); @@ -75,16 +75,16 @@ impl Middleware for PatchToolCallsMiddleware { } let has_response = state.messages[i + 1..].iter().any(|m| { - m.role == Role::Tool && m.tool_call_id.as_deref() == Some(&*tc.id) + matches!(m, Message::Tool(t) if t.tool_call_id == tc.id) }); if !has_response { - patched.push(Message::tool( + patched.push(Message::tool_with_name( + &tc.id, format!( "Tool call {} with id {} was cancelled — another message came in before it could be completed.", tc.name, tc.id ), - &tc.id, &tc.name, )); modified = true; @@ -107,6 +107,13 @@ impl Middleware for PatchToolCallsMiddleware { mod tests { use super::*; use crate::ToolCall; + use std::sync::Arc; + + fn state_with_messages(messages: Vec) -> AgentState { + let mut state = AgentState::new(); + state.messages = Arc::new(messages); + state + } #[test] fn test_middleware_name() { @@ -147,109 +154,109 @@ mod tests { assert!(validate_tool_call_id("call/id").is_err()); } - #[test] - fn test_no_patch_needed() { + #[tokio::test] + async fn test_no_patch_needed() { let mw = PatchToolCallsMiddleware::new(); - let state = AgentState { - messages: vec![Message::user("hi"), Message::assistant("hello")], - ..Default::default() - }; + let state = state_with_messages(vec![Message::human("hi"), Message::ai("hello")]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } - #[test] - fn test_patch_dangling_tool_call() { + #[tokio::test] + async fn test_patch_dangling_tool_call() { let mw = PatchToolCallsMiddleware::new(); - let mut assistant_msg = Message::assistant("I'll use a tool"); - assistant_msg.tool_calls.push(ToolCall { - id: "call-1".into(), - name: "read_file".into(), - args: serde_json::json!({"path": "test.txt"}), - }); - - let state = AgentState { - messages: vec![ - Message::user("help me"), - assistant_msg, - Message::user("never mind"), - ], - ..Default::default() - }; + let assistant_msg = Message::ai_with_tools( + "I'll use a tool", + vec![ToolCall { + id: "call-1".into(), + name: "read_file".into(), + args: serde_json::json!({"path": "test.txt"}), + }], + ); + + let state = state_with_messages(vec![ + Message::human("help me"), + assistant_msg, + Message::human("never mind"), + ]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let messages = update.unwrap().messages.unwrap(); assert_eq!(messages.len(), 4); - assert_eq!(messages[2].role, Role::Tool); - assert!(messages[2].content.contains("cancelled")); - assert_eq!(messages[2].tool_call_id.as_deref(), Some("call-1")); + match &messages[2] { + Message::Tool(t) => { + assert!(t.content.contains("cancelled")); + assert_eq!(t.tool_call_id, "call-1"); + } + other => panic!("expected Tool message, got {:?}", other), + } } - #[test] - fn test_no_patch_when_response_exists() { + #[tokio::test] + async fn test_no_patch_when_response_exists() { let mw = PatchToolCallsMiddleware::new(); - let mut assistant_msg = Message::assistant("Using tool"); - assistant_msg.tool_calls.push(ToolCall { - id: "call-1".into(), - name: "read_file".into(), - args: serde_json::json!({}), - }); - - let state = AgentState { - messages: vec![ - assistant_msg, - Message::tool("file content", "call-1", "read_file"), - ], - ..Default::default() - }; + let assistant_msg = Message::ai_with_tools( + "Using tool", + vec![ToolCall { + id: "call-1".into(), + name: "read_file".into(), + args: serde_json::json!({}), + }], + ); + + let state = state_with_messages(vec![ + assistant_msg, + Message::tool_with_name("call-1", "file content", "read_file"), + ]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } - #[test] - fn test_patch_multiple_dangling() { + #[tokio::test] + async fn test_patch_multiple_dangling() { let mw = PatchToolCallsMiddleware::new(); - let mut assistant_msg = Message::assistant("Using tools"); - assistant_msg.tool_calls.push(ToolCall { - id: "call-1".into(), - name: "read_file".into(), - args: serde_json::json!({}), - }); - assistant_msg.tool_calls.push(ToolCall { - id: "call-2".into(), - name: "write_file".into(), - args: serde_json::json!({}), - }); - - let state = AgentState { - messages: vec![assistant_msg], - ..Default::default() - }; + let assistant_msg = Message::ai_with_tools( + "Using tools", + vec![ + ToolCall { + id: "call-1".into(), + name: "read_file".into(), + args: serde_json::json!({}), + }, + ToolCall { + id: "call-2".into(), + name: "write_file".into(), + args: serde_json::json!({}), + }, + ], + ); + + let state = state_with_messages(vec![assistant_msg]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let messages = update.unwrap().messages.unwrap(); assert_eq!(messages.len(), 3); - assert_eq!(messages[1].role, Role::Tool); - assert_eq!(messages[2].role, Role::Tool); + assert!(matches!(&messages[1], Message::Tool(_))); + assert!(matches!(&messages[2], Message::Tool(_))); } - #[test] - fn test_empty_messages() { + #[tokio::test] + async fn test_empty_messages() { let mw = PatchToolCallsMiddleware::new(); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } } diff --git a/crates/rvAgent/rvagent-middleware/src/pipeline.rs b/crates/rvAgent/rvagent-middleware/src/pipeline.rs new file mode 100644 index 0000000000..5f492cc49b --- /dev/null +++ b/crates/rvAgent/rvagent-middleware/src/pipeline.rs @@ -0,0 +1,142 @@ +//! Middleware pipeline execution (ADR-095) — fully async model-call chain (P0.3). + +use std::future::Future; +use std::pin::Pin; + +use crate::{ + AgentState, Middleware, ModelHandler, ModelRequest, ModelResponse, RunnableConfig, Runtime, + Tool, +}; + +/// Executes the middleware pipeline in order. +/// Mirrors LangChain's `create_agent` middleware composition. +pub struct MiddlewarePipeline { + middlewares: Vec>, +} + +impl MiddlewarePipeline { + /// Create a new pipeline from an ordered list of middlewares. + pub fn new(middlewares: Vec>) -> Self { + Self { middlewares } + } + + /// Create an empty pipeline. + pub fn empty() -> Self { + Self { + middlewares: Vec::new(), + } + } + + /// Add a middleware to the end of the pipeline. + pub fn push(&mut self, middleware: Box) { + self.middlewares.push(middleware); + } + + /// Number of middlewares in the pipeline. + pub fn len(&self) -> usize { + self.middlewares.len() + } + + /// Whether the pipeline is empty. + pub fn is_empty(&self) -> bool { + self.middlewares.is_empty() + } + + /// Get middleware names in order. + pub fn names(&self) -> Vec<&str> { + self.middlewares.iter().map(|mw| mw.name()).collect() + } + + /// Run `before_agent` hooks in order, accumulating state updates. + pub async fn run_before_agent( + &self, + state: &mut AgentState, + runtime: &Runtime, + config: &RunnableConfig, + ) { + for mw in &self.middlewares { + if let Some(update) = mw.before_agent(state, runtime, config).await { + update.apply_to(state); + } + } + } + + /// Collect all tools from all middlewares. + pub fn collect_tools(&self) -> Vec> { + self.middlewares.iter().flat_map(|mw| mw.tools()).collect() + } + + /// Run `modify_request` through all middlewares in order. + pub fn run_modify_request(&self, mut request: ModelRequest) -> ModelRequest { + for mw in &self.middlewares { + request = mw.modify_request(request); + } + request + } + + /// Run the async `wrap_model_call` chain. + /// Middlewares are chained so the outermost (first) wraps the innermost (last). + pub async fn run_wrap_model_call( + &self, + request: ModelRequest, + base_handler: &dyn ModelHandler, + ) -> ModelResponse { + chain_call(&self.middlewares, request, base_handler).await + } + + /// Full pipeline run: before_agent, collect tools, modify_request, wrap_model_call. + pub async fn run( + &self, + state: &mut AgentState, + runtime: &Runtime, + config: &RunnableConfig, + mut request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { + // 1. Run before_agent hooks + self.run_before_agent(state, runtime, config).await; + + // 2. Collect tools from all middlewares + for tool in self.collect_tools() { + request.tools.push(tool.definition()); + } + + // 3. Run modify_request + request = self.run_modify_request(request); + + // 4. Run the async wrap_model_call chain + self.run_wrap_model_call(request, handler).await + } +} + +type BoxResponseFuture<'a> = Pin + Send + 'a>>; + +/// Recursively chain `wrap_model_call` futures from the outside in. +fn chain_call<'a>( + middlewares: &'a [Box], + request: ModelRequest, + handler: &'a dyn ModelHandler, +) -> BoxResponseFuture<'a> { + Box::pin(async move { + match middlewares.split_first() { + None => handler.call(request).await, + Some((first, rest)) => { + let inner = ChainedHandler { rest, handler }; + first.wrap_model_call(request, &inner).await + } + } + }) +} + +/// Handler that forwards to the remainder of the middleware chain. +struct ChainedHandler<'a> { + rest: &'a [Box], + handler: &'a dyn ModelHandler, +} + +#[async_trait::async_trait] +impl ModelHandler for ChainedHandler<'_> { + async fn call(&self, request: ModelRequest) -> ModelResponse { + chain_call(self.rest, request, self.handler).await + } +} diff --git a/crates/rvAgent/rvagent-middleware/src/pipeline_model.rs b/crates/rvAgent/rvagent-middleware/src/pipeline_model.rs new file mode 100644 index 0000000000..cfbf83731e --- /dev/null +++ b/crates/rvAgent/rvagent-middleware/src/pipeline_model.rs @@ -0,0 +1,308 @@ +//! `PipelineModel` — a `rvagent_core::models::ChatModel` adapter that runs +//! every model call through a `MiddlewarePipeline` (P0.3 wiring). +//! +//! This lets `AgentGraph` stay unchanged while gaining the full middleware +//! stack: the graph calls `ChatModel::complete`, and this adapter routes the +//! call through `modify_request` and the async `wrap_model_call` chain before +//! delegating to the wrapped inner model. + +use std::sync::{Arc, Mutex}; + +use async_trait::async_trait; + +use rvagent_core::error::{Result as CoreResult, RvAgentError}; +use rvagent_core::models::{ChatModel, ToolDefinition}; + +use crate::{Message, MiddlewarePipeline, ModelHandler, ModelRequest, ModelResponse}; + +/// A `ChatModel` that wraps an inner model with a middleware pipeline. +/// +/// Notes: +/// - A leading `Message::System` in the input is hoisted into the request's +/// `system_message` slot so middleware can append to it; the base handler +/// reassembles it before calling the inner model. +/// - Inner model errors surface to the chain as a `"error: …"` text response +/// (so e.g. `RetryMiddleware` can retry them); if the *final* inner call +/// still failed, the original error is propagated to the caller. +/// - Middleware-provided tools are NOT auto-advertised here; the tool set is +/// whatever the graph's `ToolExecutor` advertises. Use +/// `MiddlewarePipeline::collect_tools` to register middleware tools with an +/// executor if desired. +pub struct PipelineModel { + inner: M, + pipeline: Arc, +} + +impl PipelineModel { + /// Wrap `inner` with the given middleware pipeline. + pub fn new(inner: M, pipeline: Arc) -> Self { + Self { inner, pipeline } + } + + /// Access the pipeline. + pub fn pipeline(&self) -> &Arc { + &self.pipeline + } + + /// Access the wrapped inner model. + pub fn inner(&self) -> &M { + &self.inner + } +} + +/// Innermost handler: calls the real model. +struct BaseHandler<'a, M: ChatModel> { + model: &'a M, + /// Error from the most recent inner call (cleared at the start of each + /// call, so a successful retry clears a previous failure). + last_error: Mutex>, +} + +#[async_trait] +impl ModelHandler for BaseHandler<'_, M> { + async fn call(&self, request: ModelRequest) -> ModelResponse { + *self.last_error.lock().unwrap() = None; + + let ModelRequest { + system_message, + mut messages, + tools, + .. + } = request; + if let Some(sys) = system_message { + messages.insert(0, Message::system(sys)); + } + + match self.model.complete(&messages, &tools).await { + Ok(message) => { + let tool_calls = match &message { + Message::Ai(ai) => ai.tool_calls.clone(), + _ => Vec::new(), + }; + ModelResponse { + message, + tool_calls, + usage: None, + } + } + Err(e) => { + // Feed the failure into the chain as an "error: …" response so + // retry-style middleware can react; remember it for propagation. + let resp = ModelResponse::text(format!("error: {e}")); + *self.last_error.lock().unwrap() = Some(e); + resp + } + } + } +} + +#[async_trait] +impl ChatModel for PipelineModel { + async fn complete( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> CoreResult { + // Hoist a leading system message into the request's system slot. + let mut msgs = messages.to_vec(); + let system_message = match msgs.first() { + Some(Message::System(sys)) => { + let content = sys.content.clone(); + msgs.remove(0); + Some(content) + } + _ => None, + }; + + let mut request = ModelRequest::new(msgs).with_system(system_message); + request.tools = tools.to_vec(); + let request = self.pipeline.run_modify_request(request); + + let handler = BaseHandler { + model: &self.inner, + last_error: Mutex::new(None), + }; + let response = self.pipeline.run_wrap_model_call(request, &handler).await; + + // If the final inner call failed, propagate the original error. + if let Some(err) = handler.last_error.lock().unwrap().take() { + return Err(err); + } + + // Reattach the (possibly middleware-filtered) tool calls to the message. + let ModelResponse { + mut message, + tool_calls, + .. + } = response; + if let Message::Ai(ai) = &mut message { + ai.tool_calls = tool_calls; + } + Ok(message) + } + + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> CoreResult> { + let msg = self.complete(messages, tools).await?; + Ok(vec![msg]) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::{append_to_system_message, Middleware}; + use rvagent_core::messages::ToolCall; + + struct EchoModel; + + #[async_trait] + impl ChatModel for EchoModel { + async fn complete( + &self, + messages: &[Message], + _tools: &[ToolDefinition], + ) -> CoreResult { + // Echo the system message content (if any) for inspection. + let sys = messages + .iter() + .find_map(|m| match m { + Message::System(s) => Some(s.content.clone()), + _ => None, + }) + .unwrap_or_default(); + Ok(Message::ai(format!("sys:[{sys}] n={}", messages.len()))) + } + + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> CoreResult> { + Ok(vec![self.complete(messages, tools).await?]) + } + } + + struct FailingModel; + + #[async_trait] + impl ChatModel for FailingModel { + async fn complete( + &self, + _messages: &[Message], + _tools: &[ToolDefinition], + ) -> CoreResult { + Err(RvAgentError::model("boom")) + } + + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> CoreResult> { + Ok(vec![self.complete(messages, tools).await?]) + } + } + + struct AppendMw(&'static str); + + #[async_trait] + impl Middleware for AppendMw { + fn name(&self) -> &str { + "append" + } + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { + let sys = append_to_system_message(&request.system_message, self.0); + handler.call(request.with_system(sys)).await + } + } + + /// Middleware that drops all tool calls (HITL-style filtering). + struct DropToolCalls; + + #[async_trait] + impl Middleware for DropToolCalls { + fn name(&self) -> &str { + "drop_tool_calls" + } + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { + let mut response = handler.call(request).await; + response.tool_calls.clear(); + response + } + } + + struct ToolCallingModel; + + #[async_trait] + impl ChatModel for ToolCallingModel { + async fn complete( + &self, + _messages: &[Message], + _tools: &[ToolDefinition], + ) -> CoreResult { + Ok(Message::ai_with_tools( + "calling", + vec![ToolCall { + id: "tc1".into(), + name: "ls".into(), + args: serde_json::json!({}), + }], + )) + } + + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> CoreResult> { + Ok(vec![self.complete(messages, tools).await?]) + } + } + + #[tokio::test] + async fn test_pipeline_model_appends_system() { + let pipeline = Arc::new(MiddlewarePipeline::new(vec![Box::new(AppendMw("EXTRA"))])); + let model = PipelineModel::new(EchoModel, pipeline); + + let messages = vec![Message::system("base"), Message::human("hi")]; + let out = model.complete(&messages, &[]).await.unwrap(); + assert!(out.content().contains("base")); + assert!(out.content().contains("EXTRA")); + } + + #[tokio::test] + async fn test_pipeline_model_propagates_errors() { + let pipeline = Arc::new(MiddlewarePipeline::empty()); + let model = PipelineModel::new(FailingModel, pipeline); + let err = model.complete(&[Message::human("hi")], &[]).await; + assert!(err.is_err()); + } + + #[tokio::test] + async fn test_pipeline_model_middleware_filters_tool_calls() { + let pipeline = Arc::new(MiddlewarePipeline::new(vec![Box::new(DropToolCalls)])); + let model = PipelineModel::new(ToolCallingModel, pipeline); + let out = model.complete(&[Message::human("hi")], &[]).await.unwrap(); + assert!(!out.has_tool_calls(), "middleware must filter tool calls"); + } + + #[tokio::test] + async fn test_pipeline_model_empty_pipeline_passthrough() { + let pipeline = Arc::new(MiddlewarePipeline::empty()); + let model = PipelineModel::new(ToolCallingModel, pipeline); + let out = model.complete(&[Message::human("hi")], &[]).await.unwrap(); + assert!(out.has_tool_calls()); + } +} diff --git a/crates/rvAgent/rvagent-middleware/src/prompt_caching.rs b/crates/rvAgent/rvagent-middleware/src/prompt_caching.rs index 80f6ccb430..5f25a39cee 100644 --- a/crates/rvAgent/rvagent-middleware/src/prompt_caching.rs +++ b/crates/rvAgent/rvagent-middleware/src/prompt_caching.rs @@ -73,7 +73,7 @@ mod tests { #[test] fn test_modify_request_with_system() { let mw = PromptCachingMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hi")]) + let request = ModelRequest::new(vec![Message::human("hi")]) .with_system(Some("You are helpful.".into())); let modified = mw.modify_request(request); @@ -84,7 +84,7 @@ mod tests { #[test] fn test_modify_request_without_system() { let mw = PromptCachingMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hi")]); + let request = ModelRequest::new(vec![Message::human("hi")]); let modified = mw.modify_request(request); assert!(!modified.cache_control.contains_key("system")); @@ -93,11 +93,11 @@ mod tests { #[test] fn test_modify_request_with_tools() { let mw = PromptCachingMiddleware::new(); - let mut request = ModelRequest::new(vec![Message::user("hi")]); + let mut request = ModelRequest::new(vec![Message::human("hi")]); request.tools.push(crate::ToolDefinition { name: "test".into(), description: "test tool".into(), - parameters: serde_json::json!({}), + input_schema: serde_json::json!({}), }); let modified = mw.modify_request(request); diff --git a/crates/rvAgent/rvagent-middleware/src/retry.rs b/crates/rvAgent/rvagent-middleware/src/retry.rs index d08371e239..35c9365daa 100644 --- a/crates/rvAgent/rvagent-middleware/src/retry.rs +++ b/crates/rvAgent/rvagent-middleware/src/retry.rs @@ -4,7 +4,6 @@ //! a transient error (e.g., content starts with `"error:"` or is empty). use std::sync::atomic::{AtomicU64, Ordering}; -use std::thread; use std::time::Duration; use async_trait::async_trait; @@ -16,7 +15,7 @@ use crate::{Middleware, ModelHandler, ModelRequest, ModelResponse}; /// Heuristic: the response is considered an error if its content is empty or /// starts with the prefix `"error:"` (case-insensitive). fn is_transient_error(response: &ModelResponse) -> bool { - let content = &response.message.content; + let content = response.content(); content.is_empty() || content.to_ascii_lowercase().starts_with("error:") } @@ -84,8 +83,12 @@ impl Middleware for RetryMiddleware { "retry" } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { - let mut response = handler.call(request.clone()); + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { + let mut response = handler.call(request.clone()).await; if !is_transient_error(&response) { return response; @@ -96,11 +99,11 @@ impl Middleware for RetryMiddleware { for attempt in 0..self.max_retries { let delay_ms = self.initial_delay_ms * 2u64.pow(attempt); - thread::sleep(Duration::from_millis(delay_ms)); + tokio::time::sleep(Duration::from_millis(delay_ms)).await; self.total_retries.fetch_add(1, Ordering::Relaxed); - response = handler.call(request.clone()); + response = handler.call(request.clone()).await; if !is_transient_error(&response) { return response; @@ -135,8 +138,9 @@ mod tests { } } + #[async_trait] impl ModelHandler for FailNHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { let remaining = self.remaining_failures.load(Ordering::SeqCst); if remaining > 0 { self.remaining_failures.fetch_sub(1, Ordering::SeqCst); @@ -149,53 +153,57 @@ mod tests { /// A handler that always succeeds. struct SuccessHandler; + + #[async_trait] impl ModelHandler for SuccessHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { ModelResponse::text("ok") } } /// A handler that always fails with an error response. struct AlwaysFailHandler; + + #[async_trait] impl ModelHandler for AlwaysFailHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { ModelResponse::text("error: permanent failure") } } fn make_request() -> ModelRequest { - ModelRequest::new(vec![Message::user("hello")]) + ModelRequest::new(vec![Message::human("hello")]) } - #[test] - fn test_no_retry_on_success() { + #[tokio::test] + async fn test_no_retry_on_success() { let mw = RetryMiddleware::default(); let handler = SuccessHandler; - let resp = mw.wrap_model_call(make_request(), &handler); + let resp = mw.wrap_model_call(make_request(), &handler).await; - assert_eq!(resp.message.content, "ok"); + assert_eq!(resp.content(), "ok"); assert_eq!(mw.retry_count(), 0); assert_eq!(mw.total_retries(), 0); } - #[test] - fn test_retry_succeeds_after_failures() { + #[tokio::test] + async fn test_retry_succeeds_after_failures() { let mw = RetryMiddleware::new(3, 1); // 1ms delay for fast tests let handler = FailNHandler::new(2); // fails twice, then succeeds - let resp = mw.wrap_model_call(make_request(), &handler); + let resp = mw.wrap_model_call(make_request(), &handler).await; - assert_eq!(resp.message.content, "success"); + assert_eq!(resp.content(), "success"); assert_eq!(mw.retry_count(), 1); assert_eq!(mw.total_retries(), 2); } - #[test] - fn test_retries_exhausted() { + #[tokio::test] + async fn test_retries_exhausted() { let mw = RetryMiddleware::new(2, 1); let handler = AlwaysFailHandler; - let resp = mw.wrap_model_call(make_request(), &handler); + let resp = mw.wrap_model_call(make_request(), &handler).await; - assert!(resp.message.content.starts_with("error:")); + assert!(resp.content().starts_with("error:")); assert_eq!(mw.retry_count(), 1); assert_eq!(mw.total_retries(), 2); } @@ -207,11 +215,11 @@ mod tests { assert_eq!(mw.initial_delay_ms, 100); } - #[test] - fn test_reset_metrics() { + #[tokio::test] + async fn test_reset_metrics() { let mw = RetryMiddleware::new(3, 1); let handler = FailNHandler::new(1); - let _ = mw.wrap_model_call(make_request(), &handler); + let _ = mw.wrap_model_call(make_request(), &handler).await; assert!(mw.retry_count() > 0); mw.reset_metrics(); @@ -243,41 +251,41 @@ mod tests { assert!(!is_transient_error(&resp)); } - #[test] - fn test_retry_first_attempt_succeeds() { + #[tokio::test] + async fn test_retry_first_attempt_succeeds() { // Edge case: handler fails on first call but succeeds on first retry (attempt 0). let mw = RetryMiddleware::new(5, 1); let handler = FailNHandler::new(1); - let resp = mw.wrap_model_call(make_request(), &handler); + let resp = mw.wrap_model_call(make_request(), &handler).await; - assert_eq!(resp.message.content, "success"); + assert_eq!(resp.content(), "success"); assert_eq!(mw.retry_count(), 1); assert_eq!(mw.total_retries(), 1); } - #[test] - fn test_zero_max_retries() { + #[tokio::test] + async fn test_zero_max_retries() { // With max_retries = 0, the initial call is made but no retries happen. let mw = RetryMiddleware::new(0, 1); let handler = AlwaysFailHandler; - let resp = mw.wrap_model_call(make_request(), &handler); + let resp = mw.wrap_model_call(make_request(), &handler).await; - assert!(resp.message.content.starts_with("error:")); + assert!(resp.content().starts_with("error:")); assert_eq!(mw.retry_count(), 1); assert_eq!(mw.total_retries(), 0); } - #[test] - fn test_metrics_accumulate_across_calls() { + #[tokio::test] + async fn test_metrics_accumulate_across_calls() { let mw = RetryMiddleware::new(3, 1); // First call: 1 failure then success let handler1 = FailNHandler::new(1); - let _ = mw.wrap_model_call(make_request(), &handler1); + let _ = mw.wrap_model_call(make_request(), &handler1).await; // Second call: 2 failures then success let handler2 = FailNHandler::new(2); - let _ = mw.wrap_model_call(make_request(), &handler2); + let _ = mw.wrap_model_call(make_request(), &handler2).await; assert_eq!(mw.retry_count(), 2); // two calls needed retries assert_eq!(mw.total_retries(), 3); // 1 + 2 retries diff --git a/crates/rvAgent/rvagent-middleware/src/rvf_manifest.rs b/crates/rvAgent/rvagent-middleware/src/rvf_manifest.rs index 1157cc0620..4b60cb8244 100644 --- a/crates/rvAgent/rvagent-middleware/src/rvf_manifest.rs +++ b/crates/rvAgent/rvagent-middleware/src/rvf_manifest.rs @@ -115,6 +115,7 @@ struct RvfToolAdapter { parameters_schema: serde_json::Value, } +#[async_trait] impl Tool for RvfToolAdapter { fn name(&self) -> &str { &self.name @@ -124,11 +125,11 @@ impl Tool for RvfToolAdapter { &self.description } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { self.parameters_schema.clone() } - fn invoke(&self, args: serde_json::Value) -> Result { + async fn invoke(&self, args: serde_json::Value) -> Result { // Without rvf-runtime, return a stub response indicating the tool is available // but actual execution requires the rvf-compat feature. Ok(format!( @@ -148,7 +149,7 @@ impl Middleware for RvfManifestMiddleware { "rvf_manifest" } - fn before_agent( + async fn before_agent( &self, _state: &AgentState, _runtime: &Runtime, @@ -272,14 +273,14 @@ mod tests { assert!(tools.iter().any(|t| t.name() == "rvf:format")); } - #[test] - fn test_tool_invoke() { + #[tokio::test] + async fn test_tool_invoke() { let mw = RvfManifestMiddleware::new(sample_config()); mw.mount_package(sample_manifest()); let tools = mw.tools(); let lint = tools.iter().find(|t| t.name() == "rvf:lint").unwrap(); - let result = lint.invoke(serde_json::json!({"path": "src/main.rs"})); + let result = lint.invoke(serde_json::json!({"path": "src/main.rs"})).await; assert!(result.is_ok()); assert!(result.unwrap().contains("rvf:lint")); } @@ -298,8 +299,8 @@ mod tests { assert!(tools.is_empty()); } - #[test] - fn test_before_agent_injects_state() { + #[tokio::test] + async fn test_before_agent_injects_state() { let mw = RvfManifestMiddleware::new(sample_config()); mw.mount_package(sample_manifest()); @@ -307,7 +308,7 @@ mod tests { let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let update = update.unwrap(); @@ -317,15 +318,15 @@ mod tests { assert_eq!(arr[0]["package"], "test-pkg"); } - #[test] - fn test_before_agent_empty_table() { + #[tokio::test] + async fn test_before_agent_empty_table() { let mw = RvfManifestMiddleware::new(sample_config()); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_none()); } diff --git a/crates/rvAgent/rvagent-middleware/src/skills.rs b/crates/rvAgent/rvagent-middleware/src/skills.rs index dd5afb61a9..b842979eee 100644 --- a/crates/rvAgent/rvagent-middleware/src/skills.rs +++ b/crates/rvAgent/rvagent-middleware/src/skills.rs @@ -247,13 +247,13 @@ impl Middleware for SkillsMiddleware { "skills" } - fn before_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, _config: &RunnableConfig, ) -> Option { - if state.extensions.contains_key("skills_metadata") { + if crate::json_extension(state, "skills_metadata").is_some() { return None; } @@ -271,7 +271,11 @@ impl Middleware for SkillsMiddleware { Some(update) } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { let skills: Vec = request .extensions .get("skills_metadata") @@ -279,7 +283,7 @@ impl Middleware for SkillsMiddleware { .unwrap_or_default(); if skills.is_empty() { - return handler.call(request); + return handler.call(request).await; } let locations = self.format_skills_locations(); @@ -289,7 +293,7 @@ impl Middleware for SkillsMiddleware { .replace("{skills_list}", &skills_list); let new_system = crate::append_to_system_message(&request.system_message, §ion); - handler.call(request.with_system(new_system)) + handler.call(request.with_system(new_system)).await } } @@ -400,16 +404,14 @@ mod tests { assert_eq!(mw.name(), "skills"); } - #[test] - fn test_before_agent_skip_if_loaded() { + #[tokio::test] + async fn test_before_agent_skip_if_loaded() { let mw = SkillsMiddleware::new(vec![]); let mut state = AgentState::default(); - state - .extensions - .insert("skills_metadata".into(), serde_json::json!([])); + state.set_extension("skills_metadata", serde_json::json!([])); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } #[test] diff --git a/crates/rvAgent/rvagent-middleware/src/sona.rs b/crates/rvAgent/rvagent-middleware/src/sona.rs index 91bc0b9082..e90b5b1e37 100644 --- a/crates/rvAgent/rvagent-middleware/src/sona.rs +++ b/crates/rvAgent/rvagent-middleware/src/sona.rs @@ -26,8 +26,8 @@ use ruvector_sona::{ }; use crate::{ - AgentState, AgentStateUpdate, AsyncModelHandler, Middleware, ModelHandler, ModelRequest, - ModelResponse, Role, RunnableConfig, Runtime, + AgentState, AgentStateUpdate, Message, Middleware, ModelHandler, ModelRequest, ModelResponse, + RunnableConfig, Runtime, }; use async_trait::async_trait; use parking_lot::RwLock; @@ -153,7 +153,7 @@ fn estimate_quality(_request: &ModelRequest, response: &ModelResponse) -> f32 { let mut quality = 0.5f32; // Longer responses often indicate more thorough answers - let response_len = response.message.content.len(); + let response_len = response.content().len(); if response_len > 100 { quality += 0.1; } @@ -167,7 +167,7 @@ fn estimate_quality(_request: &ModelRequest, response: &ModelResponse) -> f32 { } // Check for error indicators - let content_lower = response.message.content.to_lowercase(); + let content_lower = response.content().to_lowercase(); if content_lower.contains("error") || content_lower.contains("failed") { quality -= 0.2; } @@ -309,8 +309,8 @@ impl SonaState { let query_text = request .messages .iter() - .filter(|m| matches!(m.role, Role::User)) - .map(|m| m.content.as_str()) + .filter(|m| matches!(m, Message::Human(_))) + .map(|m| m.content()) .collect::>() .join(" "); @@ -322,7 +322,7 @@ impl SonaState { // Add response as a step let response_embedding = - generate_embedding(&response.message.content, self.config.embedding_dim); + generate_embedding(response.content(), self.config.embedding_dim); let quality = estimate_quality(request, response); builder.add_step(response_embedding, vec![], quality); @@ -630,7 +630,7 @@ impl Middleware for SonaMiddleware { "sona" } - fn before_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, @@ -653,10 +653,10 @@ impl Middleware for SonaMiddleware { .messages .iter() .rev() - .find(|m| matches!(m.role, Role::User)); + .find(|m| matches!(m, Message::Human(_))); if let Some(msg) = last_user_message { - let patterns = self.state.read().find_similar_patterns(&msg.content); + let patterns = self.state.read().find_similar_patterns(msg.content()); if !patterns.is_empty() { // Store patterns in extensions for potential use @@ -674,29 +674,10 @@ impl Middleware for SonaMiddleware { None } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { - if !self.is_enabled() { - return handler.call(request); - } - - let start = Instant::now(); - - // Call the underlying handler - let response = handler.call(request.clone()); - - // Record trajectory (Loop A - Instant Learning) - let latency = start.elapsed(); - self.state - .read() - .record_trajectory(&request, &response, latency); - - response - } - - async fn awrap_model_call( + async fn wrap_model_call( &self, request: ModelRequest, - handler: &dyn AsyncModelHandler, + handler: &dyn ModelHandler, ) -> ModelResponse { if !self.is_enabled() { return handler.call(request).await; @@ -781,7 +762,7 @@ mod tests { #[test] fn test_estimate_quality() { - let request = ModelRequest::new(vec![Message::user("test")]); + let request = ModelRequest::new(vec![Message::human("test")]); // Short response let short_response = ModelResponse::text("ok"); @@ -835,21 +816,23 @@ mod tests { } struct TestHandler; + + #[async_trait] impl ModelHandler for TestHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { ModelResponse::text("Test response with some content for quality estimation") } } - #[test] - fn test_wrap_model_call() { + #[tokio::test] + async fn test_wrap_model_call() { let middleware = SonaMiddleware::default_config(); let handler = TestHandler; - let request = ModelRequest::new(vec![Message::user("Hello")]); + let request = ModelRequest::new(vec![Message::human("Hello")]); - let response = middleware.wrap_model_call(request, &handler); + let response = middleware.wrap_model_call(request, &handler).await; - assert!(response.message.content.contains("Test response")); + assert!(response.content().contains("Test response")); #[cfg(feature = "sona")] { @@ -858,17 +841,17 @@ mod tests { } } - #[test] - fn test_wrap_model_call_disabled() { + #[tokio::test] + async fn test_wrap_model_call_disabled() { let middleware = SonaMiddleware::default_config(); middleware.set_enabled(false); let handler = TestHandler; - let request = ModelRequest::new(vec![Message::user("Hello")]); + let request = ModelRequest::new(vec![Message::human("Hello")]); - let response = middleware.wrap_model_call(request, &handler); + let response = middleware.wrap_model_call(request, &handler).await; - assert!(response.message.content.contains("Test response")); + assert!(response.content().contains("Test response")); // No recording when disabled let stats = middleware.stats(); diff --git a/crates/rvAgent/rvagent-middleware/src/subagents.rs b/crates/rvAgent/rvagent-middleware/src/subagents.rs index 89c603ee74..1a2dc51003 100644 --- a/crates/rvAgent/rvagent-middleware/src/subagents.rs +++ b/crates/rvAgent/rvagent-middleware/src/subagents.rs @@ -59,7 +59,7 @@ impl Middleware for SubAgentMiddleware { "subagent" } - fn before_agent( + async fn before_agent( &self, _state: &AgentState, _runtime: &Runtime, @@ -77,14 +77,18 @@ impl Middleware for SubAgentMiddleware { Some(update) } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { if self.specs.is_empty() { - return handler.call(request); + return handler.call(request).await; } let descriptions = self.format_subagent_descriptions(); let new_system = crate::append_to_system_message(&request.system_message, &descriptions); - handler.call(request.with_system(new_system)) + handler.call(request.with_system(new_system)).await } fn tools(&self) -> Vec> { @@ -95,6 +99,7 @@ impl Middleware for SubAgentMiddleware { /// Tool for spawning subagents. struct TaskTool; +#[async_trait] impl Tool for TaskTool { fn name(&self) -> &str { "task" @@ -104,7 +109,7 @@ impl Tool for TaskTool { "Spawn a subagent to handle a specific task. The subagent runs independently and returns its result." } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { serde_json::json!({ "type": "object", "properties": { @@ -125,7 +130,7 @@ impl Tool for TaskTool { }) } - fn invoke(&self, _args: serde_json::Value) -> Result { + async fn invoke(&self, _args: serde_json::Value) -> Result { Err("task tool must be invoked through the agent runtime".into()) } } @@ -148,17 +153,17 @@ mod tests { assert_eq!(tools[0].name(), "task"); } - #[test] - fn test_before_agent_no_specs() { + #[tokio::test] + async fn test_before_agent_no_specs() { let mw = SubAgentMiddleware::new(); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } - #[test] - fn test_before_agent_with_specs() { + #[tokio::test] + async fn test_before_agent_with_specs() { let specs = vec![SubAgentSpec { name: "coder".into(), description: "A coding agent".into(), @@ -170,7 +175,7 @@ mod tests { let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); assert!(update.unwrap().extensions.contains_key("subagent_specs")); } @@ -202,7 +207,7 @@ mod tests { #[test] fn test_task_tool_schema() { let tool = TaskTool; - let schema = tool.parameters_schema(); + let schema = tool.input_schema(); assert_eq!(schema["type"], "object"); let required = schema["required"].as_array().unwrap(); assert!(required.contains(&serde_json::json!("description"))); diff --git a/crates/rvAgent/rvagent-middleware/src/summarization.rs b/crates/rvAgent/rvagent-middleware/src/summarization.rs index 586da0155d..82e60c2b8f 100644 --- a/crates/rvAgent/rvagent-middleware/src/summarization.rs +++ b/crates/rvAgent/rvagent-middleware/src/summarization.rs @@ -4,7 +4,7 @@ use async_trait::async_trait; use uuid::Uuid; -use crate::{Message, Middleware, ModelHandler, ModelRequest, ModelResponse, Role}; +use crate::{Message, Middleware, ModelHandler, ModelRequest, ModelResponse}; /// Trigger configuration for auto-compaction. pub enum TriggerConfig { @@ -44,7 +44,7 @@ impl SummarizationMiddleware { fn estimate_tokens(messages: &[Message]) -> u64 { messages .iter() - .map(|m| (m.content.len() as u64) / 4 + 1) + .map(|m| (m.content().len() as u64) / 4 + 1) .sum() } @@ -69,11 +69,11 @@ impl SummarizationMiddleware { )); for msg in messages { - if msg.role == Role::User { - let preview = if msg.content.len() > 100 { - format!("{}...", &msg.content[..100]) + if let Message::Human(h) = msg { + let preview = if h.content.len() > 100 { + format!("{}...", &h.content[..100]) } else { - msg.content.clone() + h.content.clone() }; summary.push_str(&format!("- User: {}\n", preview)); } @@ -91,13 +91,13 @@ impl SummarizationMiddleware { fn format_for_offload(messages: &[Message]) -> String { let mut out = String::new(); for msg in messages { - let role = match msg.role { - Role::System => "system", - Role::User => "user", - Role::Assistant => "assistant", - Role::Tool => "tool", + let role = match msg { + Message::System(_) => "system", + Message::Human(_) => "user", + Message::Ai(_) => "assistant", + Message::Tool(_) => "tool", }; - out.push_str(&format!("## {}\n\n{}\n\n---\n\n", role, msg.content)); + out.push_str(&format!("## {}\n\n{}\n\n---\n\n", role, msg.content())); } out } @@ -109,7 +109,11 @@ impl Middleware for SummarizationMiddleware { "summarization" } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { let token_count = Self::estimate_tokens(&request.messages); let threshold = self.threshold(); @@ -129,9 +133,9 @@ impl Middleware for SummarizationMiddleware { let mut compacted = vec![summary]; compacted.extend_from_slice(to_keep); - handler.call(request.with_messages(compacted)) + handler.call(request.with_messages(compacted)).await } else { - handler.call(request) + handler.call(request).await } } } @@ -140,9 +144,13 @@ impl Middleware for SummarizationMiddleware { mod tests { use super::*; + use async_trait::async_trait; + struct PassthroughHandler; + + #[async_trait] impl ModelHandler for PassthroughHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(format!("messages: {}", request.messages.len())) } } @@ -162,7 +170,7 @@ mod tests { #[test] fn test_estimate_tokens() { - let messages = vec![Message::user("hello world")]; + let messages = vec![Message::human("hello world")]; let tokens = SummarizationMiddleware::estimate_tokens(&messages); assert!(tokens > 0); } @@ -180,31 +188,30 @@ mod tests { assert_eq!(mw.keep_count(1), 1); } - #[test] - fn test_no_compaction_below_threshold() { + #[tokio::test] + async fn test_no_compaction_below_threshold() { let mw = SummarizationMiddleware::new(100_000, 0.85, 0.10); - let request = ModelRequest::new(vec![Message::user("short")]); + let request = ModelRequest::new(vec![Message::human("short")]); let handler = PassthroughHandler; - let response = mw.wrap_model_call(request, &handler); - assert!(response.message.content.contains("messages: 1")); + let response = mw.wrap_model_call(request, &handler).await; + assert!(response.content().contains("messages: 1")); } - #[test] - fn test_compaction_above_threshold() { + #[tokio::test] + async fn test_compaction_above_threshold() { let mw = SummarizationMiddleware::new(10, 0.5, 0.5); let mut messages = Vec::new(); for i in 0..20 { - messages.push(Message::user(format!( + messages.push(Message::human(format!( "message {} with enough content to trigger compaction when counted", i ))); } let request = ModelRequest::new(messages); let handler = PassthroughHandler; - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; let count: usize = response - .message - .content + .content() .strip_prefix("messages: ") .unwrap() .parse() @@ -224,18 +231,18 @@ mod tests { #[test] fn test_summarize() { let messages = vec![ - Message::user("What is Rust?"), - Message::assistant("Rust is a systems programming language."), + Message::human("What is Rust?"), + Message::ai("Rust is a systems programming language."), ]; let summary = SummarizationMiddleware::summarize(&messages); - assert_eq!(summary.role, Role::System); - assert!(summary.content.contains("2 messages")); - assert!(summary.content.contains("What is Rust?")); + assert!(matches!(summary, Message::System(_))); + assert!(summary.content().contains("2 messages")); + assert!(summary.content().contains("What is Rust?")); } #[test] fn test_format_for_offload() { - let messages = vec![Message::user("test content")]; + let messages = vec![Message::human("test content")]; let offloaded = SummarizationMiddleware::format_for_offload(&messages); assert!(offloaded.contains("## user")); assert!(offloaded.contains("test content")); diff --git a/crates/rvAgent/rvagent-middleware/src/todolist.rs b/crates/rvAgent/rvagent-middleware/src/todolist.rs index 295b2ad5c6..04b4119cef 100644 --- a/crates/rvAgent/rvagent-middleware/src/todolist.rs +++ b/crates/rvAgent/rvagent-middleware/src/todolist.rs @@ -31,7 +31,7 @@ impl Middleware for TodoListMiddleware { "todolist" } - fn before_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, @@ -67,8 +67,8 @@ fn format_todos(todos: &[TodoItem]) -> String { TodoStatus::Completed => "completed", }; out.push_str(&format!( - " {}\n", - todo.id, status_str, todo.content + " {}\n", + status_str, todo.content )); } out.push_str(""); @@ -78,16 +78,17 @@ fn format_todos(todos: &[TodoItem]) -> String { /// Tool for writing/updating todo items. struct WriteTodosTool; +#[async_trait] impl Tool for WriteTodosTool { fn name(&self) -> &str { "write_todos" } fn description(&self) -> &str { - "Create or update the todo list. Provide a complete list of todo items with id, content, and status (pending, in_progress, completed)." + "Create or update the todo list. Provide a complete list of todo items with content, status (pending, in_progress, completed), and optional active_form." } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { serde_json::json!({ "type": "object", "properties": { @@ -96,14 +97,14 @@ impl Tool for WriteTodosTool { "items": { "type": "object", "properties": { - "id": { "type": "string" }, "content": { "type": "string" }, "status": { "type": "string", "enum": ["pending", "in_progress", "completed"] - } + }, + "active_form": { "type": "string" } }, - "required": ["id", "content", "status"] + "required": ["content", "status"] } } }, @@ -111,7 +112,7 @@ impl Tool for WriteTodosTool { }) } - fn invoke(&self, args: serde_json::Value) -> Result { + async fn invoke(&self, args: serde_json::Value) -> Result { let todos = args .get("todos") .and_then(|v| v.as_array()) @@ -120,10 +121,6 @@ impl Tool for WriteTodosTool { let count = todos.len(); // Validate each item for item in todos { - let _id = item - .get("id") - .and_then(|v| v.as_str()) - .ok_or("each todo must have an 'id' string")?; let _content = item .get("content") .and_then(|v| v.as_str()) @@ -146,6 +143,14 @@ impl Tool for WriteTodosTool { mod tests { use super::*; + fn todo(content: &str, status: TodoStatus) -> TodoItem { + TodoItem { + content: content.into(), + status, + active_form: String::new(), + } + } + #[test] fn test_format_todos_empty() { let result = format_todos(&[]); @@ -155,16 +160,8 @@ mod tests { #[test] fn test_format_todos() { let todos = vec![ - TodoItem { - id: "1".into(), - content: "Do something".into(), - status: TodoStatus::Pending, - }, - TodoItem { - id: "2".into(), - content: "Done".into(), - status: TodoStatus::Completed, - }, + todo("Do something", TodoStatus::Pending), + todo("Done", TodoStatus::Completed), ]; let result = format_todos(&todos); assert!(result.contains("status=\"pending\"")); @@ -172,27 +169,23 @@ mod tests { assert!(result.contains("Do something")); } - #[test] - fn test_before_agent_empty_todos() { + #[tokio::test] + async fn test_before_agent_empty_todos() { let mw = TodoListMiddleware::new(); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - assert!(mw.before_agent(&state, &runtime, &config).is_none()); + assert!(mw.before_agent(&state, &runtime, &config).await.is_none()); } - #[test] - fn test_before_agent_with_todos() { + #[tokio::test] + async fn test_before_agent_with_todos() { let mw = TodoListMiddleware::new(); let mut state = AgentState::default(); - state.todos.push(TodoItem { - id: "1".into(), - content: "Test task".into(), - status: TodoStatus::InProgress, - }); + state.push_todo(todo("Test task", TodoStatus::InProgress)); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let update = update.unwrap(); assert!(update.extensions.contains_key("todo_context")); @@ -204,37 +197,37 @@ mod tests { assert_eq!(tool.name(), "write_todos"); } - #[test] - fn test_write_todos_invoke_valid() { + #[tokio::test] + async fn test_write_todos_invoke_valid() { let tool = WriteTodosTool; let args = serde_json::json!({ "todos": [ - {"id": "1", "content": "task 1", "status": "pending"}, - {"id": "2", "content": "task 2", "status": "completed"} + {"content": "task 1", "status": "pending"}, + {"content": "task 2", "status": "completed"} ] }); - let result = tool.invoke(args); + let result = tool.invoke(args).await; assert!(result.is_ok()); assert!(result.unwrap().contains("2 todo items")); } - #[test] - fn test_write_todos_invoke_invalid_status() { + #[tokio::test] + async fn test_write_todos_invoke_invalid_status() { let tool = WriteTodosTool; let args = serde_json::json!({ - "todos": [{"id": "1", "content": "task", "status": "invalid"}] + "todos": [{"content": "task", "status": "invalid"}] }); - let result = tool.invoke(args); + let result = tool.invoke(args).await; assert!(result.is_err()); } - #[test] - fn test_write_todos_invoke_missing_field() { + #[tokio::test] + async fn test_write_todos_invoke_missing_field() { let tool = WriteTodosTool; let args = serde_json::json!({ - "todos": [{"id": "1"}] + "todos": [{"status": "pending"}] }); - let result = tool.invoke(args); + let result = tool.invoke(args).await; assert!(result.is_err()); } diff --git a/crates/rvAgent/rvagent-middleware/src/tool_sanitizer.rs b/crates/rvAgent/rvagent-middleware/src/tool_sanitizer.rs index 83cadc397f..f707896244 100644 --- a/crates/rvAgent/rvagent-middleware/src/tool_sanitizer.rs +++ b/crates/rvAgent/rvagent-middleware/src/tool_sanitizer.rs @@ -4,7 +4,7 @@ use async_trait::async_trait; -use crate::{Middleware, ModelHandler, ModelRequest, ModelResponse, Role}; +use crate::{Message, Middleware, ModelHandler, ModelRequest, ModelResponse}; /// Middleware that sanitizes tool results by wrapping them in XML-like delimiters. /// @@ -51,37 +51,38 @@ impl Middleware for ToolResultSanitizerMiddleware { "tool_result_sanitizer" } - fn wrap_model_call( + async fn wrap_model_call( &self, mut request: ModelRequest, handler: &dyn ModelHandler, ) -> ModelResponse { // Sanitize all tool messages in the request for msg in &mut request.messages { - if msg.role == Role::Tool { - let tool_name = msg.tool_name.as_deref().unwrap_or("unknown"); - let tool_call_id = msg.tool_call_id.as_deref().unwrap_or("unknown"); - msg.content = Self::sanitize_tool_result(tool_name, tool_call_id, &msg.content); + if let Message::Tool(t) = msg { + let tool_name = t.tool_name.as_deref().unwrap_or("unknown"); + t.content = Self::sanitize_tool_result(tool_name, &t.tool_call_id, &t.content); } } - handler.call(request) + handler.call(request).await } } #[cfg(test)] mod tests { use super::*; - use crate::Message; + use async_trait::async_trait; struct CaptureHandler; + + #[async_trait] impl ModelHandler for CaptureHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { let tool_content = request .messages .iter() - .find(|m| m.role == Role::Tool) - .map(|m| m.content.clone()) + .find(|m| matches!(m, Message::Tool(_))) + .map(|m| m.content().to_string()) .unwrap_or_default(); ModelResponse::text(tool_content) } @@ -130,39 +131,41 @@ mod tests { assert!(result.contains("id=\"id"val\"")); } - #[test] - fn test_wrap_model_call_sanitizes_tool_messages() { + #[tokio::test] + async fn test_wrap_model_call_sanitizes_tool_messages() { let mw = ToolResultSanitizerMiddleware::new(); let request = ModelRequest::new(vec![ - Message::user("help"), - Message::tool("raw tool output", "call-1", "read_file"), + Message::human("help"), + Message::tool_with_name("call-1", "raw tool output", "read_file"), ]); let handler = CaptureHandler; - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; - assert!(response.message.content.contains("")); + assert!(response.content().contains("")); } - #[test] - fn test_wrap_model_call_skips_non_tool_messages() { + #[tokio::test] + async fn test_wrap_model_call_skips_non_tool_messages() { let mw = ToolResultSanitizerMiddleware::new(); let request = ModelRequest::new(vec![ - Message::user("not a tool message"), - Message::assistant("also not a tool"), + Message::human("not a tool message"), + Message::ai("also not a tool"), ]); struct VerifyHandler; + + #[async_trait] impl ModelHandler for VerifyHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { - assert_eq!(request.messages[0].content, "not a tool message"); - assert_eq!(request.messages[1].content, "also not a tool"); + async fn call(&self, request: ModelRequest) -> ModelResponse { + assert_eq!(request.messages[0].content(), "not a tool message"); + assert_eq!(request.messages[1].content(), "also not a tool"); ModelResponse::text("ok") } } - mw.wrap_model_call(request, &VerifyHandler); + mw.wrap_model_call(request, &VerifyHandler).await; } #[test] diff --git a/crates/rvAgent/rvagent-middleware/src/types.rs b/crates/rvAgent/rvagent-middleware/src/types.rs new file mode 100644 index 0000000000..797d517900 --- /dev/null +++ b/crates/rvAgent/rvagent-middleware/src/types.rs @@ -0,0 +1,188 @@ +//! Middleware request/response types built on the canonical `rvagent-core` +//! type system (P0.1 — unified types). +//! +//! `Message`, `ToolCall`, `AgentState`, `TodoItem`, `TodoStatus`, +//! `RunnableConfig`, and `ToolDefinition` are re-exported from +//! `rvagent_core`; this module only defines the middleware-specific +//! envelope types (`ModelRequest`, `ModelResponse`, `AgentStateUpdate`, …). + +use std::collections::HashMap; +use std::sync::Arc; + +use serde::{Deserialize, Serialize}; + +// Canonical types (single source of truth: rvagent-core). +pub use rvagent_core::config::RunnableConfig; +pub use rvagent_core::messages::{ + AiMessage, HumanMessage, Message, SystemMessage, ToolCall, ToolMessage, +}; +pub use rvagent_core::models::ToolDefinition; +pub use rvagent_core::state::{AgentState, FileData, TodoItem, TodoStatus}; + +/// Cache control hint for prompt caching (Anthropic). +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct CacheControl { + pub cache_type: String, +} + +/// State update returned by `before_agent`. Merged into `AgentState`. +/// +/// Extensions are stored in the typed extension slot of the core +/// `AgentState` as `serde_json::Value` entries. +#[derive(Debug, Clone, Default)] +pub struct AgentStateUpdate { + pub messages: Option>, + pub todos: Option>, + pub extensions: HashMap, +} + +impl AgentStateUpdate { + /// Merge this update into an `AgentState`. + pub fn apply_to(self, state: &mut AgentState) { + if let Some(messages) = self.messages { + state.messages = Arc::new(messages); + } + if let Some(todos) = self.todos { + state.todos = Arc::new(todos); + } + for (k, v) in self.extensions { + state.set_extension(k, v); + } + } +} + +/// Read a JSON extension value stored on the core `AgentState`. +pub fn json_extension<'a>(state: &'a AgentState, key: &str) -> Option<&'a serde_json::Value> { + state.get_extension::(key) +} + +/// Model request wrapping messages and configuration. +#[derive(Debug, Clone)] +pub struct ModelRequest { + pub system_message: Option, + pub messages: Vec, + pub tools: Vec, + pub cache_control: HashMap, + pub extensions: HashMap, +} + +impl ModelRequest { + /// Create a new model request. + pub fn new(messages: Vec) -> Self { + Self { + system_message: None, + messages, + tools: vec![], + cache_control: HashMap::new(), + extensions: HashMap::new(), + } + } + + /// Return a copy with a different system message. + pub fn with_system(mut self, system_message: Option) -> Self { + self.system_message = system_message; + self + } + + /// Return a copy with different messages. + pub fn with_messages(mut self, messages: Vec) -> Self { + self.messages = messages; + self + } +} + +/// Model response from an LLM call. +#[derive(Debug, Clone)] +pub struct ModelResponse { + pub message: Message, + pub tool_calls: Vec, + pub usage: Option, +} + +impl ModelResponse { + /// Create a simple text response (an AI message). + pub fn text(content: impl Into) -> Self { + Self { + message: Message::ai(content), + tool_calls: vec![], + usage: None, + } + } + + /// Text content of the response message. + pub fn content(&self) -> &str { + self.message.content() + } +} + +/// Token usage information. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +pub struct Usage { + pub input_tokens: u64, + pub output_tokens: u64, + #[serde(default)] + pub cache_read_tokens: u64, + #[serde(default)] + pub cache_creation_tokens: u64, +} + +/// Runtime context passed to middleware hooks. +pub struct Runtime { + pub context: serde_json::Value, + pub config: RunnableConfig, +} + +impl Runtime { + pub fn new() -> Self { + Self { + context: serde_json::Value::Null, + config: RunnableConfig::default(), + } + } +} + +impl Default for Runtime { + fn default() -> Self { + Self::new() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn test_model_request_with_system() { + let req = ModelRequest::new(vec![Message::human("hi")]); + assert!(req.system_message.is_none()); + let req2 = req.with_system(Some("system".into())); + assert_eq!(req2.system_message, Some("system".into())); + } + + #[test] + fn test_model_response_text() { + let resp = ModelResponse::text("hello"); + assert_eq!(resp.content(), "hello"); + assert!(matches!(resp.message, Message::Ai(_))); + assert!(resp.tool_calls.is_empty()); + } + + #[test] + fn test_runtime_default() { + let rt = Runtime::default(); + assert_eq!(rt.context, serde_json::Value::Null); + } + + #[test] + fn test_agent_state_update_apply() { + let mut state = AgentState::default(); + let mut update = AgentStateUpdate::default(); + update.messages = Some(vec![Message::human("hi")]); + update + .extensions + .insert("k".into(), serde_json::json!("v")); + update.apply_to(&mut state); + assert_eq!(state.message_count(), 1); + assert_eq!(json_extension(&state, "k"), Some(&serde_json::json!("v"))); + } +} diff --git a/crates/rvAgent/rvagent-middleware/src/unicode_security_middleware.rs b/crates/rvAgent/rvagent-middleware/src/unicode_security_middleware.rs index d4976b118c..3fa1817fd8 100644 --- a/crates/rvAgent/rvagent-middleware/src/unicode_security_middleware.rs +++ b/crates/rvAgent/rvagent-middleware/src/unicode_security_middleware.rs @@ -3,7 +3,7 @@ //! Automatically checks tool inputs and outputs for Unicode-based security threats. use crate::unicode_security::{UnicodeIssue, UnicodeSecurityChecker, UnicodeSecurityConfig}; -use crate::{AgentState, AgentStateUpdate, Message, Middleware, Role, RunnableConfig, Runtime}; +use crate::{AgentState, AgentStateUpdate, Message, Middleware, RunnableConfig, Runtime}; use async_trait::async_trait; use tracing::{debug, warn}; @@ -69,11 +69,6 @@ impl UnicodeSecurityMiddleware { self } - /// Check a message for Unicode security issues. - fn check_message(&self, msg: &Message) -> Vec { - self.checker.check(&msg.content) - } - /// Log detected issues. fn log_issues(&self, issues: &[UnicodeIssue], context: &str) { if !issues.is_empty() { @@ -95,7 +90,7 @@ impl Middleware for UnicodeSecurityMiddleware { "unicode_security" } - async fn abefore_agent( + async fn before_agent( &self, state: &AgentState, _runtime: &Runtime, @@ -104,37 +99,37 @@ impl Middleware for UnicodeSecurityMiddleware { let mut modified = false; let mut new_messages = Vec::new(); - for msg in &state.messages { + for msg in state.messages.iter() { let mut msg_copy = msg.clone(); - match msg.role { - Role::User if self.check_user_input => { - let issues = self.check_message(msg); + match msg { + Message::Human(h) if self.check_user_input => { + let issues = self.checker.check(&h.content); if !issues.is_empty() { self.log_issues(&issues, "user message"); // Sanitize if configured if self.sanitize_inputs { - msg_copy.content = self.checker.sanitize(&msg.content); + *msg_copy.content_mut() = self.checker.sanitize(&h.content); modified = true; debug!("Sanitized user message"); } } } - Role::Tool => { - let issues = self.check_message(msg); + Message::Tool(t) => { + let issues = self.checker.check(&t.content); if !issues.is_empty() { self.log_issues( &issues, &format!( "tool result: {}", - msg.tool_name.as_deref().unwrap_or("unknown") + t.tool_name.as_deref().unwrap_or("unknown") ), ); // Sanitize if configured if self.sanitize_outputs { - msg_copy.content = self.checker.sanitize(&msg.content); + *msg_copy.content_mut() = self.checker.sanitize(&t.content); modified = true; debug!("Sanitized tool output"); } @@ -148,8 +143,8 @@ impl Middleware for UnicodeSecurityMiddleware { new_messages.push(msg_copy); // Check tool call arguments (in assistant messages) - if msg.role == Role::Assistant { - for tool_call in &msg.tool_calls { + if let Message::Ai(ai) = msg { + for tool_call in &ai.tool_calls { if let Some(args_str) = tool_call.args.as_str() { let issues = self.checker.check(args_str); if !issues.is_empty() { @@ -189,6 +184,13 @@ impl Middleware for UnicodeSecurityMiddleware { mod tests { use super::*; use crate::{Message, ToolCall}; + use std::sync::Arc; + + fn state_with_messages(messages: Vec) -> AgentState { + let mut state = AgentState::new(); + state.messages = Arc::new(messages); + state + } #[tokio::test] async fn test_strict_middleware() { @@ -200,30 +202,26 @@ mod tests { async fn test_detect_bidi_in_tool_result() { let mw = UnicodeSecurityMiddleware::strict(); - let state = AgentState { - messages: vec![Message::tool( - "evil\u{202E}txt.exe", // BiDi override - "tc-1", - "filesystem", - )], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::tool_with_name( + "tc-1", + "evil\u{202E}txt.exe", // BiDi override + "filesystem", + )]); let runtime = Runtime::new(); let config = RunnableConfig::default(); // Should detect but not modify (sanitize_outputs = false by default) - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_none()); // Enable sanitization let mw2 = UnicodeSecurityMiddleware::strict().with_output_sanitization(true); - let update2 = mw2.abefore_agent(&state, &runtime, &config).await; + let update2 = mw2.before_agent(&state, &runtime, &config).await; assert!(update2.is_some()); let new_msgs = update2.unwrap().messages.unwrap(); - assert_eq!(new_msgs[0].content, "eviltxt.exe"); // BiDi stripped + assert_eq!(new_msgs[0].content(), "eviltxt.exe"); // BiDi stripped } #[tokio::test] @@ -232,48 +230,39 @@ mod tests { .with_user_input_check(true) .with_input_sanitization(true); - let state = AgentState { - messages: vec![Message::user("Hello\u{200B}world")], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::human("Hello\u{200B}world")]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let new_msgs = update.unwrap().messages.unwrap(); - assert_eq!(new_msgs[0].content, "Helloworld"); + assert_eq!(new_msgs[0].content(), "Helloworld"); } #[tokio::test] async fn test_check_tool_call_arguments() { let mw = UnicodeSecurityMiddleware::strict(); - let state = AgentState { - messages: vec![{ - let mut msg = Message::assistant(""); - msg.tool_calls = vec![ToolCall { - id: "tc-1".to_string(), - name: "write_file".to_string(), - args: serde_json::json!({ - "path": "test.txt", - "content": "evil\u{202E}txt.exe" - }), - }]; - msg + let state = state_with_messages(vec![Message::ai_with_tools( + "", + vec![ToolCall { + id: "tc-1".to_string(), + name: "write_file".to_string(), + args: serde_json::json!({ + "path": "test.txt", + "content": "evil\u{202E}txt.exe" + }), }], - todos: vec![], - extensions: Default::default(), - }; + )]); let runtime = Runtime::new(); let config = RunnableConfig::default(); // Should detect (logs warning) but not modify - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_none()); } @@ -282,16 +271,16 @@ mod tests { // Without output sanitization, should only log warnings let mw = UnicodeSecurityMiddleware::strict().with_output_sanitization(false); - let state = AgentState { - messages: vec![Message::tool("pаypal.com", "tc-1", "browser")], // Cyrillic 'а' - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::tool_with_name( + "tc-1", + "pаypal.com", // Cyrillic 'а' + "browser", + )]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; // Should detect confusable and log, but not modify (sanitize_outputs = false) assert!(update.is_none()); } @@ -302,19 +291,15 @@ mod tests { .with_user_input_check(true) .with_input_sanitization(true); - let state = AgentState { - messages: vec![ - Message::user("Hello world"), - Message::tool("OK", "tc-1", "test"), - ], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![ + Message::human("Hello world"), + Message::tool_with_name("tc-1", "OK", "test"), + ]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_none()); // No modification needed } @@ -324,16 +309,12 @@ mod tests { .with_user_input_check(true) .with_input_sanitization(true); - let state = AgentState { - messages: vec![Message::system("System\u{202E}message")], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::system("System\u{202E}message")]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_none()); // System messages are never modified } @@ -342,26 +323,22 @@ mod tests { let mw = UnicodeSecurityMiddleware::new(UnicodeSecurityConfig::permissive()) .with_output_sanitization(true); - let state = AgentState { - messages: vec![ - Message::tool("pаypal.com", "tc-1", "test"), // Confusable (should pass) - Message::tool("evil\u{202E}txt.exe", "tc-2", "test"), // BiDi (should be caught) - ], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![ + Message::tool_with_name("tc-1", "pаypal.com", "test"), // Confusable (should pass) + Message::tool_with_name("tc-2", "evil\u{202E}txt.exe", "test"), // BiDi (should be caught) + ]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let new_msgs = update.unwrap().messages.unwrap(); // First message unchanged (confusables not checked in permissive mode) - assert_eq!(new_msgs[0].content, "pаypal.com"); + assert_eq!(new_msgs[0].content(), "pаypal.com"); // Second message sanitized (BiDi always checked) - assert_eq!(new_msgs[1].content, "eviltxt.exe"); + assert_eq!(new_msgs[1].content(), "eviltxt.exe"); } #[tokio::test] @@ -371,26 +348,22 @@ mod tests { .with_input_sanitization(true) .with_output_sanitization(true); - let state = AgentState { - messages: vec![ - Message::user("Hello\u{200B}world"), - Message::assistant("Response"), - Message::tool("evil\u{202E}txt.exe", "tc-1", "filesystem"), - ], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![ + Message::human("Hello\u{200B}world"), + Message::ai("Response"), + Message::tool_with_name("tc-1", "evil\u{202E}txt.exe", "filesystem"), + ]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let new_msgs = update.unwrap().messages.unwrap(); assert_eq!(new_msgs.len(), 3); - assert_eq!(new_msgs[0].content, "Helloworld"); // User message sanitized - assert_eq!(new_msgs[1].content, "Response"); // Assistant unchanged - assert_eq!(new_msgs[2].content, "eviltxt.exe"); // Tool result sanitized + assert_eq!(new_msgs[0].content(), "Helloworld"); // User message sanitized + assert_eq!(new_msgs[1].content(), "Response"); // Assistant unchanged + assert_eq!(new_msgs[2].content(), "eviltxt.exe"); // Tool result sanitized } } diff --git a/crates/rvAgent/rvagent-middleware/src/witness.rs b/crates/rvAgent/rvagent-middleware/src/witness.rs index 3ee855fd30..75b2ec9397 100644 --- a/crates/rvAgent/rvagent-middleware/src/witness.rs +++ b/crates/rvAgent/rvagent-middleware/src/witness.rs @@ -298,8 +298,12 @@ impl Middleware for WitnessMiddleware { "witness" } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { - let response = handler.call(request); + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { + let response = handler.call(request).await; // Log each tool call to the witness chain if !response.tool_calls.is_empty() { @@ -318,11 +322,15 @@ mod tests { use super::*; use crate::{Message, ToolCall}; + use async_trait::async_trait; + struct ToolCallHandler { tool_calls: Vec, } + + #[async_trait] impl ModelHandler for ToolCallHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { let mut response = ModelResponse::text("done"); response.tool_calls = self.tool_calls.clone(); response @@ -362,8 +370,8 @@ mod tests { assert_eq!(builder.entries()[1].sequence, 1); } - #[test] - fn test_wrap_model_call_records_tool_calls() { + #[tokio::test] + async fn test_wrap_model_call_records_tool_calls() { let mw = WitnessMiddleware::new(); let handler = ToolCallHandler { tool_calls: vec![ @@ -380,8 +388,8 @@ mod tests { ], }; - let request = ModelRequest::new(vec![Message::user("test")]); - let _response = mw.wrap_model_call(request, &handler); + let request = ModelRequest::new(vec![Message::human("test")]); + let _response = mw.wrap_model_call(request, &handler).await; let builder = mw.builder().lock().unwrap(); assert_eq!(builder.len(), 2); @@ -389,20 +397,20 @@ mod tests { assert_eq!(builder.entries()[1].tool_name, "execute"); } - #[test] - fn test_wrap_model_call_no_tool_calls() { + #[tokio::test] + async fn test_wrap_model_call_no_tool_calls() { let mw = WitnessMiddleware::new(); let handler = ToolCallHandler { tool_calls: vec![] }; let request = ModelRequest::new(vec![]); - let _response = mw.wrap_model_call(request, &handler); + let _response = mw.wrap_model_call(request, &handler).await; let builder = mw.builder().lock().unwrap(); assert!(builder.is_empty()); } - #[test] - fn test_thread_safety() { + #[tokio::test] + async fn test_thread_safety() { let builder = Arc::new(Mutex::new(WitnessBuilder::new())); let mw1 = WitnessMiddleware::with_builder(builder.clone()); let mw2 = WitnessMiddleware::with_builder(builder.clone()); @@ -424,8 +432,8 @@ mod tests { let req1 = ModelRequest::new(vec![]); let req2 = ModelRequest::new(vec![]); - mw1.wrap_model_call(req1, &handler1); - mw2.wrap_model_call(req2, &handler2); + mw1.wrap_model_call(req1, &handler1).await; + mw2.wrap_model_call(req2, &handler2).await; let builder = builder.lock().unwrap(); assert_eq!(builder.len(), 2); diff --git a/crates/rvAgent/rvagent-middleware/tests/hitl_tests.rs b/crates/rvAgent/rvagent-middleware/tests/hitl_tests.rs index 5ebb37da5f..3206346505 100644 --- a/crates/rvAgent/rvagent-middleware/tests/hitl_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/hitl_tests.rs @@ -1,5 +1,6 @@ //! Integration tests for the Human-in-the-Loop (HITL) middleware. +use async_trait::async_trait; use rvagent_middleware::hitl::{ApprovalDecision, HumanInTheLoopMiddleware}; use rvagent_middleware::{ Message, Middleware, ModelHandler, ModelRequest, ModelResponse, ToolCall, @@ -33,8 +34,9 @@ impl ToolCallHandler { } } +#[async_trait] impl ModelHandler for ToolCallHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { let mut response = ModelResponse::text("model response"); response.tool_calls = self.tool_calls.clone(); response @@ -131,13 +133,13 @@ fn test_empty_patterns_interrupts_nothing() { // Tests: wrap_model_call // --------------------------------------------------------------------------- -#[test] -fn test_wrap_filters_matching_tool_calls() { +#[tokio::test] +async fn test_wrap_filters_matching_tool_calls() { let mw = HumanInTheLoopMiddleware::new(vec!["execute".into()]); let handler = ToolCallHandler::with_names(&["execute", "read_file"]); - let request = ModelRequest::new(vec![Message::user("do something")]); + let request = ModelRequest::new(vec![Message::human("do something")]); - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; // Only read_file should remain assert_eq!(response.tool_calls.len(), 1); @@ -145,90 +147,92 @@ fn test_wrap_filters_matching_tool_calls() { // HITL message should be appended assert!( - response.message.content.contains("[HITL]"), + response.content().contains("[HITL]"), "should contain HITL marker" ); assert!( - response.message.content.contains("execute"), + response.content().contains("execute"), "should mention the interrupted tool" ); } -#[test] -fn test_wrap_no_matching_tools_passes_all_through() { +#[tokio::test] +async fn test_wrap_no_matching_tools_passes_all_through() { let mw = HumanInTheLoopMiddleware::new(vec!["dangerous_tool".into()]); let handler = ToolCallHandler::with_names(&["read_file", "ls", "glob"]); - let request = ModelRequest::new(vec![Message::user("safe operation")]); + let request = ModelRequest::new(vec![Message::human("safe operation")]); - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; assert_eq!(response.tool_calls.len(), 3); assert!( - !response.message.content.contains("[HITL]"), + !response.content().contains("[HITL]"), "should not contain HITL marker when nothing is interrupted" ); } -#[test] -fn test_wrap_all_tools_interrupted() { +#[tokio::test] +async fn test_wrap_all_tools_interrupted() { let mw = HumanInTheLoopMiddleware::new(vec!["*".into()]); let handler = ToolCallHandler::with_names(&["execute", "write_file"]); - let request = ModelRequest::new(vec![Message::user("do things")]); + let request = ModelRequest::new(vec![Message::human("do things")]); - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; assert!( response.tool_calls.is_empty(), "all tool calls should be intercepted" ); - assert!(response.message.content.contains("[HITL]")); - assert!(response.message.content.contains("execute")); - assert!(response.message.content.contains("write_file")); + assert!(response.content().contains("[HITL]")); + assert!(response.content().contains("execute")); + assert!(response.content().contains("write_file")); } -#[test] -fn test_wrap_no_tool_calls_from_handler() { +#[tokio::test] +async fn test_wrap_no_tool_calls_from_handler() { let mw = HumanInTheLoopMiddleware::new(vec!["execute".into()]); struct NoToolHandler; + + #[async_trait] impl ModelHandler for NoToolHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { ModelResponse::text("just text, no tools") } } - let request = ModelRequest::new(vec![Message::user("question")]); - let response = mw.wrap_model_call(request, &NoToolHandler); + let request = ModelRequest::new(vec![Message::human("question")]); + let response = mw.wrap_model_call(request, &NoToolHandler).await; assert!(response.tool_calls.is_empty()); assert!( - !response.message.content.contains("[HITL]"), + !response.content().contains("[HITL]"), "should not add HITL marker when no tool calls" ); } -#[test] -fn test_wrap_preserves_original_response_content() { +#[tokio::test] +async fn test_wrap_preserves_original_response_content() { let mw = HumanInTheLoopMiddleware::new(vec!["dangerous".into()]); let handler = ToolCallHandler::with_names(&["read_file"]); - let request = ModelRequest::new(vec![Message::user("hi")]); + let request = ModelRequest::new(vec![Message::human("hi")]); - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; assert!( - response.message.content.contains("model response"), + response.content().contains("model response"), "should preserve original model response content" ); } -#[test] -fn test_wrap_prefix_pattern_filters_correctly() { +#[tokio::test] +async fn test_wrap_prefix_pattern_filters_correctly() { let mw = HumanInTheLoopMiddleware::new(vec!["write_*".into()]); let handler = ToolCallHandler::with_names(&["write_file", "write_todos", "read_file", "execute"]); - let request = ModelRequest::new(vec![Message::user("do writes")]); + let request = ModelRequest::new(vec![Message::human("do writes")]); - let response = mw.wrap_model_call(request, &handler); + let response = mw.wrap_model_call(request, &handler).await; assert_eq!( response.tool_calls.len(), diff --git a/crates/rvAgent/rvagent-middleware/tests/mcp_bridge_tests.rs b/crates/rvAgent/rvagent-middleware/tests/mcp_bridge_tests.rs index 35ab4cd111..764aa66704 100644 --- a/crates/rvAgent/rvagent-middleware/tests/mcp_bridge_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/mcp_bridge_tests.rs @@ -1,5 +1,6 @@ //! Integration tests for the MCP bridge middleware. +use async_trait::async_trait; use rvagent_middleware::mcp_bridge::{McpBridgeConfig, McpBridgeMiddleware}; use rvagent_middleware::{ AgentState, Message, Middleware, ModelHandler, ModelRequest, ModelResponse, RunnableConfig, @@ -12,8 +13,9 @@ use rvagent_middleware::{ struct PassthroughHandler; +#[async_trait] impl ModelHandler for PassthroughHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(format!("handled:{}", request.messages.len())) } } @@ -109,14 +111,14 @@ fn test_middleware_name() { assert_eq!(mw.name(), "mcp_bridge"); } -#[test] -fn test_before_agent_when_enabled_injects_config() { +#[tokio::test] +async fn test_before_agent_when_enabled_injects_config() { let mw = McpBridgeMiddleware::new(); let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!( update.is_some(), "enabled bridge should produce state update" @@ -129,8 +131,8 @@ fn test_before_agent_when_enabled_injects_config() { ); } -#[test] -fn test_before_agent_when_disabled_returns_none() { +#[tokio::test] +async fn test_before_agent_when_disabled_returns_none() { let config = McpBridgeConfig { enabled: false, ..Default::default() @@ -140,7 +142,7 @@ fn test_before_agent_when_disabled_returns_none() { let runtime = Runtime::new(); let runnable_config = RunnableConfig::default(); - let update = mw.before_agent(&state, &runtime, &runnable_config); + let update = mw.before_agent(&state, &runtime, &runnable_config).await; assert!( update.is_none(), "disabled bridge should not produce update" @@ -150,7 +152,7 @@ fn test_before_agent_when_disabled_returns_none() { #[test] fn test_modify_request_when_enabled_sets_flag() { let mw = McpBridgeMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hello")]); + let request = ModelRequest::new(vec![Message::human("hello")]); let modified = mw.modify_request(request); assert_eq!( @@ -167,7 +169,7 @@ fn test_modify_request_when_disabled_does_not_set_flag() { ..Default::default() }; let mw = McpBridgeMiddleware::with_config(config); - let request = ModelRequest::new(vec![Message::user("hello")]); + let request = ModelRequest::new(vec![Message::human("hello")]); let modified = mw.modify_request(request); assert!( @@ -176,14 +178,14 @@ fn test_modify_request_when_disabled_does_not_set_flag() { ); } -#[test] -fn test_wrap_model_call_passes_through() { +#[tokio::test] +async fn test_wrap_model_call_passes_through() { let mw = McpBridgeMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hi")]); - let response = mw.wrap_model_call(request, &PassthroughHandler); + let request = ModelRequest::new(vec![Message::human("hi")]); + let response = mw.wrap_model_call(request, &PassthroughHandler).await; assert!( - response.message.content.contains("handled:1"), + response.content().contains("handled:1"), "wrap_model_call should pass through to handler" ); } @@ -211,13 +213,13 @@ fn test_tools_when_disabled_provides_no_tools() { assert!(tools.is_empty()); } -#[test] -fn test_status_tool_returns_config_values() { +#[tokio::test] +async fn test_status_tool_returns_config_values() { let mw = McpBridgeMiddleware::new(); let tools = mw.tools(); let status_tool = &tools[0]; - let result = status_tool.invoke(serde_json::json!({})); + let result = status_tool.invoke(serde_json::json!({})).await; assert!(result.is_ok()); let json: serde_json::Value = serde_json::from_str(&result.unwrap()).unwrap(); @@ -230,7 +232,7 @@ fn test_status_tool_returns_config_values() { fn test_status_tool_schema() { let mw = McpBridgeMiddleware::new(); let tools = mw.tools(); - let schema = tools[0].parameters_schema(); + let schema = tools[0].input_schema(); assert!(schema.is_object()); assert!(schema["properties"].is_object()); } diff --git a/crates/rvAgent/rvagent-middleware/tests/pipeline_tests.rs b/crates/rvAgent/rvagent-middleware/tests/pipeline_tests.rs index de8e83c34a..35856e153a 100644 --- a/crates/rvAgent/rvagent-middleware/tests/pipeline_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/pipeline_tests.rs @@ -3,9 +3,8 @@ use async_trait::async_trait; use rvagent_middleware::{ - append_to_system_message, AgentState, AgentStateUpdate, Message, Middleware, - MiddlewarePipeline, ModelHandler, ModelRequest, ModelResponse, Role, RunnableConfig, Runtime, - Tool, ToolDefinition, + append_to_system_message, json_extension, AgentState, AgentStateUpdate, Message, Middleware, + MiddlewarePipeline, ModelHandler, ModelRequest, ModelResponse, RunnableConfig, Runtime, Tool, }; // --------------------------------------------------------------------------- @@ -33,7 +32,7 @@ impl Middleware for RecordingMiddleware { &self.label } - fn before_agent( + async fn before_agent( &self, _state: &AgentState, _runtime: &Runtime, @@ -68,9 +67,13 @@ impl Middleware for SystemAppender { &self.label } - fn wrap_model_call(&self, request: ModelRequest, handler: &dyn ModelHandler) -> ModelResponse { + async fn wrap_model_call( + &self, + request: ModelRequest, + handler: &dyn ModelHandler, + ) -> ModelResponse { let new_sys = append_to_system_message(&request.system_message, &self.text); - handler.call(request.with_system(new_sys)) + handler.call(request.with_system(new_sys)).await } } @@ -93,6 +96,7 @@ struct NamedTool { name: String, } +#[async_trait] impl Tool for NamedTool { fn name(&self) -> &str { &self.name @@ -100,10 +104,10 @@ impl Tool for NamedTool { fn description(&self) -> &str { "test tool" } - fn parameters_schema(&self) -> serde_json::Value { + fn input_schema(&self) -> serde_json::Value { serde_json::json!({"type": "object"}) } - fn invoke(&self, _args: serde_json::Value) -> Result { + async fn invoke(&self, _args: serde_json::Value) -> Result { Ok("ok".into()) } } @@ -124,21 +128,13 @@ impl Middleware for ToolInjectorMw { /// Handler that captures the final system message. struct CaptureSystemHandler; +#[async_trait] impl ModelHandler for CaptureSystemHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(request.system_message.unwrap_or_default()) } } -/// Handler that returns the number of tool definitions. -struct CountToolsHandler; - -impl ModelHandler for CountToolsHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { - ModelResponse::text(format!("tools:{}", request.tools.len())) - } -} - // --------------------------------------------------------------------------- // Tests // --------------------------------------------------------------------------- @@ -174,35 +170,37 @@ async fn test_pipeline_before_agent_chain() { // All three middlewares should have set their extension key. assert_eq!( - state.extensions.get("visited_first"), + json_extension(&state, "visited_first"), Some(&serde_json::json!(true)), "first middleware should have run" ); assert_eq!( - state.extensions.get("visited_second"), + json_extension(&state, "visited_second"), Some(&serde_json::json!(true)), "second middleware should have run" ); assert_eq!( - state.extensions.get("visited_third"), + json_extension(&state, "visited_third"), Some(&serde_json::json!(true)), "third middleware should have run" ); } -#[test] -fn test_pipeline_wrap_model_call_chain() { +#[tokio::test] +async fn test_pipeline_wrap_model_call_chain() { // Two appenders: "A" then "B". Both should appear in the final system message. let pipeline = MiddlewarePipeline::new(vec![ Box::new(SystemAppender::new("appender_a", "<>")), Box::new(SystemAppender::new("appender_b", "<>")), ]); - let request = ModelRequest::new(vec![Message::user("hi")]).with_system(Some("base".into())); + let request = ModelRequest::new(vec![Message::human("hi")]).with_system(Some("base".into())); - let response = pipeline.run_wrap_model_call(request, &CaptureSystemHandler); + let response = pipeline + .run_wrap_model_call(request, &CaptureSystemHandler) + .await; - let sys = response.message.content; + let sys = response.content(); assert!(sys.contains("base"), "should preserve base system message"); assert!(sys.contains("<>"), "should include appender A"); assert!(sys.contains("<>"), "should include appender B"); diff --git a/crates/rvAgent/rvagent-middleware/tests/prompt_caching_tests.rs b/crates/rvAgent/rvagent-middleware/tests/prompt_caching_tests.rs index d6cf3b28e2..735f2cfbc5 100644 --- a/crates/rvAgent/rvagent-middleware/tests/prompt_caching_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/prompt_caching_tests.rs @@ -17,7 +17,7 @@ fn test_middleware_name() { fn test_default_cache_type_is_ephemeral() { let mw = PromptCachingMiddleware::new(); let request = - ModelRequest::new(vec![Message::user("hi")]).with_system(Some("system prompt".into())); + ModelRequest::new(vec![Message::human("hi")]).with_system(Some("system prompt".into())); let modified = mw.modify_request(request); assert_eq!(modified.cache_control["system"].cache_type, "ephemeral"); @@ -27,7 +27,7 @@ fn test_default_cache_type_is_ephemeral() { fn test_custom_cache_type() { let mw = PromptCachingMiddleware::with_cache_type("persistent"); let request = - ModelRequest::new(vec![Message::user("hi")]).with_system(Some("system prompt".into())); + ModelRequest::new(vec![Message::human("hi")]).with_system(Some("system prompt".into())); let modified = mw.modify_request(request); assert_eq!(modified.cache_control["system"].cache_type, "persistent"); @@ -46,7 +46,7 @@ fn test_default_trait_implementation() { #[test] fn test_adds_cache_control_for_system_message() { let mw = PromptCachingMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hello")]) + let request = ModelRequest::new(vec![Message::human("hello")]) .with_system(Some("You are a helpful assistant.".into())); let modified = mw.modify_request(request); @@ -61,7 +61,7 @@ fn test_adds_cache_control_for_system_message() { #[test] fn test_no_cache_control_without_system_message() { let mw = PromptCachingMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hello")]); + let request = ModelRequest::new(vec![Message::human("hello")]); let modified = mw.modify_request(request); @@ -78,11 +78,11 @@ fn test_no_cache_control_without_system_message() { #[test] fn test_adds_cache_control_for_tools() { let mw = PromptCachingMiddleware::new(); - let mut request = ModelRequest::new(vec![Message::user("hello")]); + let mut request = ModelRequest::new(vec![Message::human("hello")]); request.tools.push(ToolDefinition { name: "read_file".into(), description: "Read a file".into(), - parameters: serde_json::json!({"type": "object"}), + input_schema: serde_json::json!({"type": "object"}), }); let modified = mw.modify_request(request); @@ -97,7 +97,7 @@ fn test_adds_cache_control_for_tools() { #[test] fn test_no_cache_control_without_tools() { let mw = PromptCachingMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("hello")]); + let request = ModelRequest::new(vec![Message::human("hello")]); let modified = mw.modify_request(request); @@ -115,11 +115,11 @@ fn test_no_cache_control_without_tools() { fn test_both_system_and_tools_get_cache_control() { let mw = PromptCachingMiddleware::new(); let mut request = - ModelRequest::new(vec![Message::user("hello")]).with_system(Some("system".into())); + ModelRequest::new(vec![Message::human("hello")]).with_system(Some("system".into())); request.tools.push(ToolDefinition { name: "ls".into(), description: "List files".into(), - parameters: serde_json::json!({}), + input_schema: serde_json::json!({}), }); let modified = mw.modify_request(request); @@ -144,11 +144,11 @@ fn test_neither_system_nor_tools_no_cache_control() { #[test] fn test_custom_cache_type_applies_to_both() { let mw = PromptCachingMiddleware::with_cache_type("long_lived"); - let mut request = ModelRequest::new(vec![Message::user("hi")]).with_system(Some("sys".into())); + let mut request = ModelRequest::new(vec![Message::human("hi")]).with_system(Some("sys".into())); request.tools.push(ToolDefinition { name: "tool".into(), description: "desc".into(), - parameters: serde_json::json!({}), + input_schema: serde_json::json!({}), }); let modified = mw.modify_request(request); @@ -160,14 +160,14 @@ fn test_custom_cache_type_applies_to_both() { #[test] fn test_messages_are_preserved_after_modify() { let mw = PromptCachingMiddleware::new(); - let request = ModelRequest::new(vec![Message::user("first"), Message::assistant("second")]) + let request = ModelRequest::new(vec![Message::human("first"), Message::ai("second")]) .with_system(Some("sys".into())); let modified = mw.modify_request(request); assert_eq!(modified.messages.len(), 2); - assert_eq!(modified.messages[0].content, "first"); - assert_eq!(modified.messages[1].content, "second"); + assert_eq!(modified.messages[0].content(), "first"); + assert_eq!(modified.messages[1].content(), "second"); assert_eq!(modified.system_message, Some("sys".to_string())); } @@ -178,12 +178,12 @@ fn test_multiple_tools_get_single_cache_entry() { request.tools.push(ToolDefinition { name: "tool_a".into(), description: "a".into(), - parameters: serde_json::json!({}), + input_schema: serde_json::json!({}), }); request.tools.push(ToolDefinition { name: "tool_b".into(), description: "b".into(), - parameters: serde_json::json!({}), + input_schema: serde_json::json!({}), }); let modified = mw.modify_request(request); diff --git a/crates/rvAgent/rvagent-middleware/tests/security_tests.rs b/crates/rvAgent/rvagent-middleware/tests/security_tests.rs index af2ea760d7..de499e0e6f 100644 --- a/crates/rvAgent/rvagent-middleware/tests/security_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/security_tests.rs @@ -19,9 +19,10 @@ use rvagent_middleware::patch_tool_calls::PatchToolCallsMiddleware; use rvagent_middleware::skills::{parse_skill_metadata, validate_skill_name, MAX_SKILL_FILE_SIZE}; use rvagent_middleware::tool_sanitizer::ToolResultSanitizerMiddleware; use rvagent_middleware::witness::{WitnessBuilder, WitnessMiddleware}; +use async_trait::async_trait; use rvagent_middleware::{ - AgentState, Message, Middleware, ModelHandler, ModelRequest, ModelResponse, Role, - RunnableConfig, Runtime, ToolCall, + AgentState, Message, Middleware, ModelHandler, ModelRequest, ModelResponse, RunnableConfig, + Runtime, ToolCall, }; // --------------------------------------------------------------------------- @@ -30,14 +31,16 @@ use rvagent_middleware::{ /// Handler that captures the model request for inspection. struct CaptureHandler; + +#[async_trait] impl ModelHandler for CaptureHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { // Return the first tool message's content (for sanitizer tests) let tool_content = request .messages .iter() - .find(|m| m.role == Role::Tool) - .map(|m| m.content.clone()) + .find(|m| matches!(m, Message::Tool(_))) + .map(|m| m.content().to_string()) .unwrap_or_default(); ModelResponse::text(tool_content) } @@ -47,8 +50,10 @@ impl ModelHandler for CaptureHandler { struct ToolCallResponseHandler { tool_calls: Vec, } + +#[async_trait] impl ModelHandler for ToolCallResponseHandler { - fn call(&self, _request: ModelRequest) -> ModelResponse { + async fn call(&self, _request: ModelRequest) -> ModelResponse { let mut response = ModelResponse::text("done"); response.tool_calls = self.tool_calls.clone(); response @@ -59,20 +64,20 @@ impl ModelHandler for ToolCallResponseHandler { // test_tool_result_sanitizer_wraps_output // =========================================================================== -#[test] -fn test_tool_result_sanitizer_wraps_output() { +#[tokio::test] +async fn test_tool_result_sanitizer_wraps_output() { let mw = ToolResultSanitizerMiddleware::new(); // Build a request with a tool message let request = ModelRequest::new(vec![ - Message::user("read the file"), - Message::tool("fn main() { println!(\"hello\"); }", "call-42", "read_file"), + Message::human("read the file"), + Message::tool_with_name("call-42", "fn main() { println!(\"hello\"); }", "read_file"), ]); - let response = mw.wrap_model_call(request, &CaptureHandler); + let response = mw.wrap_model_call(request, &CaptureHandler).await; // The tool message content should now be wrapped in tags - let content = &response.message.content; + let content = response.content(); assert!( content.starts_with(") -> AgentState { + let mut state = AgentState::new(); + state.messages = std::sync::Arc::new(messages); + state + } + // Scenario 1: Valid tool call ID with no response → should be patched - let mut msg_valid = Message::assistant("Using tool"); - msg_valid.tool_calls.push(ToolCall { - id: "call-abc123".into(), - name: "read_file".into(), - args: serde_json::json!({"path": "test.txt"}), - }); - - let state = AgentState { - messages: vec![ - Message::user("help"), - msg_valid, - Message::user("changed my mind"), - ], - ..Default::default() - }; + let msg_valid = Message::ai_with_tools( + "Using tool", + vec![ToolCall { + id: "call-abc123".into(), + name: "read_file".into(), + args: serde_json::json!({"path": "test.txt"}), + }], + ); + + let state = state_with_messages(vec![ + Message::human("help"), + msg_valid, + Message::human("changed my mind"), + ]); - let update = mw.before_agent(&state, &runtime, &config); + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some(), "Dangling tool call must be patched"); let messages = update.unwrap().messages.unwrap(); // Should have: user, assistant, synthetic tool response, user assert_eq!(messages.len(), 4); - assert_eq!(messages[2].role, Role::Tool); - assert!(messages[2].content.contains("cancelled")); + assert!(matches!(&messages[2], Message::Tool(_))); + assert!(messages[2].content().contains("cancelled")); // Scenario 2: Tool call with existing response → no patching needed - let mut msg_with_response = Message::assistant("Using tool"); - msg_with_response.tool_calls.push(ToolCall { - id: "call-xyz".into(), - name: "read_file".into(), - args: serde_json::json!({}), - }); - - let state2 = AgentState { - messages: vec![ - msg_with_response, - Message::tool("file contents", "call-xyz", "read_file"), - ], - ..Default::default() - }; + let msg_with_response = Message::ai_with_tools( + "Using tool", + vec![ToolCall { + id: "call-xyz".into(), + name: "read_file".into(), + args: serde_json::json!({}), + }], + ); + + let state2 = state_with_messages(vec![ + msg_with_response, + Message::tool_with_name("call-xyz", "file contents", "read_file"), + ]); - let update2 = mw.before_agent(&state2, &runtime, &config); + let update2 = mw.before_agent(&state2, &runtime, &config).await; assert!( update2.is_none(), "Tool call with existing response must not be patched" @@ -292,15 +301,15 @@ fn test_patch_tool_calls_validates_ids() { // Scenario 3: Empty messages → no update let state3 = AgentState::default(); - assert!(mw.before_agent(&state3, &runtime, &config).is_none()); + assert!(mw.before_agent(&state3, &runtime, &config).await.is_none()); } // =========================================================================== // test_memory_trust_verification // =========================================================================== -#[test] -fn test_memory_trust_verification() { +#[tokio::test] +async fn test_memory_trust_verification() { // 1. Compute hash of known content let trusted_content = "# Agent Instructions\nBe helpful and accurate."; let hash = compute_sha3_256(trusted_content.as_bytes()); @@ -345,7 +354,7 @@ fn test_memory_trust_verification() { let state = AgentState::default(); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw_loaded.before_agent(&state, &runtime, &config); + let update = mw_loaded.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); // 7. Test content size limit @@ -356,7 +365,7 @@ fn test_memory_trust_verification() { .with_security_policy(SecurityPolicy::Permissive) .with_preloaded(oversized_preloaded); - let update_big = mw_big.before_agent(&state, &runtime, &config); + let update_big = mw_big.before_agent(&state, &runtime, &config).await; // The update should exist but the oversized content should be filtered out assert!(update_big.is_some()); let ext = &update_big.unwrap().extensions; diff --git a/crates/rvAgent/rvagent-middleware/tests/summarization_tests.rs b/crates/rvAgent/rvagent-middleware/tests/summarization_tests.rs index 924264ebdf..fb7345a790 100644 --- a/crates/rvAgent/rvagent-middleware/tests/summarization_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/summarization_tests.rs @@ -5,8 +5,9 @@ //! - UUID-based offload filenames (SEC-015) //! - File permission expectations (0600) +use async_trait::async_trait; use rvagent_middleware::summarization::SummarizationMiddleware; -use rvagent_middleware::{Message, Middleware, ModelHandler, ModelRequest, ModelResponse, Role}; +use rvagent_middleware::{Message, Middleware, ModelHandler, ModelRequest, ModelResponse}; // --------------------------------------------------------------------------- // Helpers @@ -14,8 +15,10 @@ use rvagent_middleware::{Message, Middleware, ModelHandler, ModelRequest, ModelR /// Handler that captures the number of messages in the request. struct MessageCountHandler; + +#[async_trait] impl ModelHandler for MessageCountHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { ModelResponse::text(format!("count={}", request.messages.len())) } } @@ -23,7 +26,7 @@ impl ModelHandler for MessageCountHandler { /// Generate N user messages with enough content to exceed a token threshold. fn generate_messages(n: usize, content_size: usize) -> Vec { (0..n) - .map(|i| Message::user(format!("Message {} {}", i, "x".repeat(content_size)))) + .map(|i| Message::human(format!("Message {} {}", i, "x".repeat(content_size)))) .collect() } @@ -31,8 +34,8 @@ fn generate_messages(n: usize, content_size: usize) -> Vec { // test_auto_compact_triggers // =========================================================================== -#[test] -fn test_auto_compact_triggers() { +#[tokio::test] +async fn test_auto_compact_triggers() { // Create middleware with very low threshold: max_tokens=10, trigger at 50% // so trigger at 5 tokens. Even a single message will exceed this. let mw = SummarizationMiddleware::new(10, 0.5, 0.5); @@ -54,9 +57,9 @@ fn test_auto_compact_triggers() { // With many messages that exceed the threshold, compaction should reduce count let messages = generate_messages(20, 100); let request = ModelRequest::new(messages); - let response = mw.wrap_model_call(request, &MessageCountHandler); + let response = mw.wrap_model_call(request, &MessageCountHandler).await; - let count_str = response.message.content.clone(); + let count_str = response.content().to_string(); let count: usize = count_str.strip_prefix("count=").unwrap().parse().unwrap(); assert!( count < 20, @@ -68,19 +71,19 @@ fn test_auto_compact_triggers() { // With a single short message below threshold, no compaction let mw_high = SummarizationMiddleware::new(100_000, 0.85, 0.10); - let short_request = ModelRequest::new(vec![Message::user("hello")]); - let short_response = mw_high.wrap_model_call(short_request, &MessageCountHandler); + let short_request = ModelRequest::new(vec![Message::human("hello")]); + let short_response = mw_high.wrap_model_call(short_request, &MessageCountHandler).await; assert_eq!( - short_response.message.content, "count=1", + short_response.content(), "count=1", "Short conversation must not be compacted" ); // Edge case: single message above threshold should not compact (need >1 messages) let mw_tiny = SummarizationMiddleware::new(1, 0.1, 0.5); - let single_request = ModelRequest::new(vec![Message::user("a long message that exceeds")]); - let single_response = mw_tiny.wrap_model_call(single_request, &MessageCountHandler); + let single_request = ModelRequest::new(vec![Message::human("a long message that exceeds")]); + let single_response = mw_tiny.wrap_model_call(single_request, &MessageCountHandler).await; assert_eq!( - single_response.message.content, "count=1", + single_response.content(), "count=1", "Single message should not be compacted even above threshold" ); } @@ -182,8 +185,8 @@ fn test_offload_uses_uuid_filename() { // test_file_permissions // =========================================================================== -#[test] -fn test_file_permissions() { +#[tokio::test] +async fn test_file_permissions() { // This test validates the permission model at the design level. // The SummarizationMiddleware is expected to write offloaded history // with mode 0600 (owner read/write only) per SEC-015. @@ -204,14 +207,16 @@ fn test_file_permissions() { // Use a handler that returns the first message's role info struct FirstMessageHandler; + + #[async_trait] impl ModelHandler for FirstMessageHandler { - fn call(&self, request: ModelRequest) -> ModelResponse { + async fn call(&self, request: ModelRequest) -> ModelResponse { if let Some(first) = request.messages.first() { - let role = match first.role { - Role::System => "system", - Role::User => "user", - Role::Assistant => "assistant", - Role::Tool => "tool", + let role = match first { + Message::System(_) => "system", + Message::Human(_) => "user", + Message::Ai(_) => "assistant", + Message::Tool(_) => "tool", }; ModelResponse::text(format!("first_role={}", role)) } else { @@ -221,13 +226,13 @@ fn test_file_permissions() { } let request = ModelRequest::new(messages); - let response = mw_compact.wrap_model_call(request, &FirstMessageHandler); + let response = mw_compact.wrap_model_call(request, &FirstMessageHandler).await; // When compaction triggers, the first message should be the summary (System role) assert!( - response.message.content.contains("first_role=system"), + response.content().contains("first_role=system"), "Compacted conversation must start with a system summary message, got: {}", - response.message.content + response.content() ); // Verify that keep_fraction and trigger_fraction are clamped diff --git a/crates/rvAgent/rvagent-middleware/tests/unicode_security_integration.rs b/crates/rvAgent/rvagent-middleware/tests/unicode_security_integration.rs index 09ef947842..4051bc1705 100644 --- a/crates/rvAgent/rvagent-middleware/tests/unicode_security_integration.rs +++ b/crates/rvAgent/rvagent-middleware/tests/unicode_security_integration.rs @@ -3,9 +3,16 @@ //! Demonstrates comprehensive security checks against Unicode-based attacks. use rvagent_middleware::{ - AgentState, Message, Middleware, PipelineConfig, RunnableConfig, Runtime, ToolCall, - UnicodeSecurityChecker, UnicodeSecurityConfig, UnicodeSecurityMiddleware, + AgentState, Message, Middleware, RunnableConfig, Runtime, ToolCall, UnicodeSecurityChecker, + UnicodeSecurityConfig, UnicodeSecurityMiddleware, }; +use std::sync::Arc; + +fn state_with_messages(messages: Vec) -> AgentState { + let mut state = AgentState::new(); + state.messages = Arc::new(messages); + state +} #[tokio::test] async fn test_real_world_bidi_attack() { @@ -14,29 +21,25 @@ async fn test_real_world_bidi_attack() { .with_input_sanitization(true) .with_output_sanitization(true); - let state = AgentState { - messages: vec![ - // Attacker tries to disguise evil.exe as safe.txt - Message::tool( - "Downloaded: safe\u{202E}exe.txt", // Displays as "safeexe.txt" (reversed) - "tc-1", - "filesystem", - ), - ], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![ + // Attacker tries to disguise evil.exe as safe.txt + Message::tool_with_name( + "tc-1", + "Downloaded: safe\u{202E}exe.txt", // Displays as "safeexe.txt" (reversed) + "filesystem", + ), + ]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let new_msgs = update.unwrap().messages.unwrap(); // BiDi should be stripped - assert!(!new_msgs[0].content.contains('\u{202E}')); - assert_eq!(new_msgs[0].content, "Downloaded: safeexe.txt"); + assert!(!new_msgs[0].content().contains('\u{202E}')); + assert_eq!(new_msgs[0].content(), "Downloaded: safeexe.txt"); } #[tokio::test] @@ -70,23 +73,19 @@ async fn test_real_world_zero_width_steganography() { .with_input_sanitization(true); // User input with hidden zero-width characters encoding secret data - let state = AgentState { - messages: vec![Message::user( - "Innocent\u{200B}text\u{200C}with\u{200D}hidden\u{200B}data", - )], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::human( + "Innocent\u{200B}text\u{200C}with\u{200D}hidden\u{200B}data", + )]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let new_msgs = update.unwrap().messages.unwrap(); // All zero-width should be stripped - assert_eq!(new_msgs[0].content, "Innocenttextwithhiddendata"); + assert_eq!(new_msgs[0].content(), "Innocenttextwithhiddendata"); } #[tokio::test] @@ -94,37 +93,32 @@ async fn test_tool_call_argument_sanitization() { // Test that tool call arguments are checked for Unicode attacks let mw = UnicodeSecurityMiddleware::strict(); - let state = AgentState { - messages: vec![{ - let mut msg = Message::assistant(""); - msg.tool_calls = vec![ - ToolCall { - id: "tc-1".to_string(), - name: "write_file".to_string(), - args: serde_json::json!({ - "path": "safe\u{202E}exe.txt", - "content": "malicious content" - }), - }, - ToolCall { - id: "tc-2".to_string(), - name: "browser_navigate".to_string(), - args: serde_json::json!({ - "url": "pаypal.com" // Cyrillic 'а' - }), - }, - ]; - msg - }], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::ai_with_tools( + "", + vec![ + ToolCall { + id: "tc-1".to_string(), + name: "write_file".to_string(), + args: serde_json::json!({ + "path": "safe\u{202E}exe.txt", + "content": "malicious content" + }), + }, + ToolCall { + id: "tc-2".to_string(), + name: "browser_navigate".to_string(), + args: serde_json::json!({ + "url": "pаypal.com" // Cyrillic 'а' + }), + }, + ], + )]); let runtime = Runtime::new(); let config = RunnableConfig::default(); // Should detect issues in tool call arguments (logs warnings) - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; // With sanitize_inputs = true by default, this should be None // because sanitize() is only applied to message content, not tool args assert!(update.is_none()); @@ -152,20 +146,16 @@ async fn test_safe_multilingual_content_unmodified() { let mw = UnicodeSecurityMiddleware::new(UnicodeSecurityConfig::permissive()) .with_output_sanitization(false); - let state = AgentState { - messages: vec![Message::tool( - "Hello, 世界! Привет! مرحبا", // Multi-script greeting - "tc-1", - "translator", - )], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![Message::tool_with_name( + "tc-1", + "Hello, 世界! Привет! مرحبا", // Multi-script greeting + "translator", + )]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; // Permissive mode doesn't check mixed scripts or confusables assert!(update.is_none()); } @@ -193,30 +183,26 @@ async fn test_comprehensive_attack_scenario() { .with_user_input_check(true) .with_input_sanitization(true); - let state = AgentState { - messages: vec![ - Message::user("Visit pаypal.com\u{200B}now!"), // Homoglyph + zero-width - Message::tool( - "Downloaded: evil\u{202E}txt.exe", // BiDi override - "tc-1", - "filesystem", - ), - ], - todos: vec![], - extensions: Default::default(), - }; + let state = state_with_messages(vec![ + Message::human("Visit pаypal.com\u{200B}now!"), // Homoglyph + zero-width + Message::tool_with_name( + "tc-1", + "Downloaded: evil\u{202E}txt.exe", // BiDi override + "filesystem", + ), + ]); let runtime = Runtime::new(); let config = RunnableConfig::default(); - let update = mw.abefore_agent(&state, &runtime, &config).await; + let update = mw.before_agent(&state, &runtime, &config).await; assert!(update.is_some()); let new_msgs = update.unwrap().messages.unwrap(); // User message: zero-width stripped - assert_eq!(new_msgs[0].content, "Visit pаypal.comnow!"); // Confusable remains + assert_eq!(new_msgs[0].content(), "Visit pаypal.comnow!"); // Confusable remains // Tool message: BiDi stripped - assert_eq!(new_msgs[1].content, "Downloaded: eviltxt.exe"); + assert_eq!(new_msgs[1].content(), "Downloaded: eviltxt.exe"); } #[test] diff --git a/crates/rvAgent/rvagent-subagents/src/lib.rs b/crates/rvAgent/rvagent-subagents/src/lib.rs index 52657f0f8c..948501792a 100644 --- a/crates/rvAgent/rvagent-subagents/src/lib.rs +++ b/crates/rvAgent/rvagent-subagents/src/lib.rs @@ -26,18 +26,20 @@ pub use crdt_merge::{merge_subagent_results, CrdtState, MergeError, VectorClock} pub use orchestrator::{spawn_parallel, SpawnError, SubAgentOrchestrator}; use serde::{Deserialize, Serialize}; -use std::collections::HashMap; +use std::sync::Arc; use std::time::Duration; +use rvagent_core::messages::Message; + // --------------------------------------------------------------------------- -// AgentState (simplified, JSON-based for cross-crate compatibility) +// AgentState — canonical typed state from rvagent-core (ADR-103 A1) // --------------------------------------------------------------------------- -/// Agent state represented as a JSON map. +/// Agent state — the canonical typed `rvagent_core::state::AgentState`. /// -/// Matches `HashMap` from ADR-097. -/// Future work (ADR-103 A1) will replace this with a typed struct. -pub type AgentState = HashMap; +/// Replaces the former `HashMap` alias (ADR-097) +/// per ADR-103 A1 / roadmap P0.1. +pub use rvagent_core::state::AgentState; // --------------------------------------------------------------------------- // RvAgentConfig @@ -226,39 +228,33 @@ pub const EXCLUDED_STATE_KEYS: &[&str] = &[ /// Prepare a filtered state for subagent invocation. /// -/// Strips excluded keys from the parent state, then injects a single -/// human message containing the task description. +/// State-isolation semantics (ADR-097) on the typed state: parent +/// `messages`, `todos`, `memory_contents`, and `skills_metadata` are +/// excluded; `files` pass through (O(1) Arc clone); a single human message +/// containing the task description is injected. pub fn prepare_subagent_state(parent_state: &AgentState, task_description: &str) -> AgentState { - let mut state: AgentState = parent_state - .iter() - .filter(|(k, _)| !EXCLUDED_STATE_KEYS.contains(&k.as_str())) - .map(|(k, v)| (k.clone(), v.clone())) - .collect(); - - state.insert( - "messages".to_string(), - serde_json::json!([{"type": "human", "content": task_description}]), - ); - + let mut state = AgentState::new(); + // Files are not in EXCLUDED_STATE_KEYS — they pass through to the child. + state.files = Arc::clone(&parent_state.files); + state.push_message(Message::human(task_description)); state } /// Extract the final message from a subagent's result state. pub fn extract_result_message(result_state: &AgentState) -> Option { - let messages = result_state.get("messages")?; - let arr = messages.as_array()?; - let last = arr.last()?; - last.get("content") - .and_then(|c| c.as_str()) - .map(|s| s.trim_end().to_string()) + result_state + .messages + .last() + .map(|m| m.content().trim_end().to_string()) } /// Merge non-excluded state from subagent result back into parent state. +/// +/// Only non-excluded state merges back: `files` (subagent wins on path +/// conflict). Parent `messages` and `todos` are never overwritten. pub fn merge_subagent_state(parent: &mut AgentState, subagent_result: &AgentState) { - for (k, v) in subagent_result { - if !EXCLUDED_STATE_KEYS.contains(&k.as_str()) { - parent.insert(k.clone(), v.clone()); - } + for (path, data) in subagent_result.files.iter() { + parent.set_file(path.clone(), data.clone()); } } @@ -309,48 +305,53 @@ mod tests { assert_eq!(back.tools.len(), 2); } + use rvagent_core::state::{FileData, TodoItem, TodoStatus}; + + fn file(content: &str) -> FileData { + FileData { + content: content.into(), + encoding: "utf-8".into(), + modified_at: None, + } + } + #[test] fn test_state_isolation_prepare() { let mut parent = AgentState::new(); - parent.insert( - "messages".into(), - serde_json::json!([{"type": "ai", "content": "secret"}]), - ); - parent.insert("remaining_steps".into(), serde_json::json!(5)); - parent.insert("task_completion".into(), serde_json::json!(false)); - parent.insert("custom_key".into(), serde_json::json!("visible")); - parent.insert("todos".into(), serde_json::json!([])); + parent.push_message(Message::ai("secret")); + parent.push_todo(TodoItem { + content: "parent todo".into(), + status: TodoStatus::Pending, + active_form: String::new(), + }); + parent.set_file("/src/main.rs", file("fn main() {}")); + parent.memory_contents = Some(std::sync::Arc::new( + [("AGENTS.md".to_string(), "secret memory".to_string())] + .into_iter() + .collect(), + )); let child = prepare_subagent_state(&parent, "Do X"); - // Parent messages must NOT leak - let msgs = child.get("messages").unwrap().as_array().unwrap(); - assert_eq!(msgs.len(), 1); - assert_eq!(msgs[0]["content"], "Do X"); - assert_eq!(msgs[0]["type"], "human"); - - // Excluded keys must not appear (except messages which is replaced) - assert!(child.get("remaining_steps").is_none()); - assert!(child.get("task_completion").is_none()); - assert!(child.get("todos").is_none()); - - // Non-excluded keys must pass through - assert_eq!( - child.get("custom_key").unwrap(), - &serde_json::json!("visible") - ); + // Parent messages must NOT leak — child gets exactly one human message. + assert_eq!(child.message_count(), 1); + assert_eq!(child.messages[0].content(), "Do X"); + assert!(matches!(child.messages[0], Message::Human(_))); + + // Excluded state must not appear. + assert!(child.todos.is_empty()); + assert!(child.memory_contents.is_none()); + assert!(child.skills_metadata.is_none()); + + // Files pass through (non-excluded state). + assert!(child.files.contains_key("/src/main.rs")); } #[test] fn test_extract_result_message() { let mut state = AgentState::new(); - state.insert( - "messages".into(), - serde_json::json!([ - {"type": "human", "content": "do X"}, - {"type": "ai", "content": "Done with X. "} - ]), - ); + state.push_message(Message::human("do X")); + state.push_message(Message::ai("Done with X. ")); let msg = extract_result_message(&state).unwrap(); assert_eq!(msg, "Done with X."); } @@ -358,25 +359,28 @@ mod tests { #[test] fn test_merge_subagent_state() { let mut parent = AgentState::new(); - parent.insert("messages".into(), serde_json::json!([])); - parent.insert("existing".into(), serde_json::json!(1)); + parent.push_message(Message::human("parent message")); + parent.set_file("/existing.rs", file("existing")); let mut child_result = AgentState::new(); - child_result.insert( - "messages".into(), - serde_json::json!([{"type": "ai", "content": "hi"}]), - ); - child_result.insert("new_key".into(), serde_json::json!("added")); - child_result.insert("todos".into(), serde_json::json!(["leaked"])); + child_result.push_message(Message::ai("hi")); + child_result.push_todo(TodoItem { + content: "leaked".into(), + status: TodoStatus::Pending, + active_form: String::new(), + }); + child_result.set_file("/new.rs", file("added")); merge_subagent_state(&mut parent, &child_result); // messages should NOT be overwritten (excluded) - assert_eq!(parent.get("messages").unwrap(), &serde_json::json!([])); + assert_eq!(parent.message_count(), 1); + assert_eq!(parent.messages[0].content(), "parent message"); // todos should NOT leak - assert!(parent.get("todos").is_none()); - // new non-excluded keys should merge - assert_eq!(parent.get("new_key").unwrap(), &serde_json::json!("added")); + assert!(parent.todos.is_empty()); + // new non-excluded state (files) should merge + assert!(parent.files.contains_key("/new.rs")); + assert!(parent.files.contains_key("/existing.rs")); } #[test] diff --git a/crates/rvAgent/rvagent-subagents/tests/integration_tests.rs b/crates/rvAgent/rvagent-subagents/tests/integration_tests.rs index 6e4b8244f1..8a809b9d43 100644 --- a/crates/rvAgent/rvagent-subagents/tests/integration_tests.rs +++ b/crates/rvAgent/rvagent-subagents/tests/integration_tests.rs @@ -1,14 +1,22 @@ //! Integration tests for rvAgent subagents. -use std::collections::HashMap; - +use rvagent_core::messages::Message; +use rvagent_core::state::{FileData, TodoItem, TodoStatus}; use rvagent_subagents::builder::compile_subagents; use rvagent_subagents::orchestrator::{spawn_parallel, SubAgentOrchestrator}; use rvagent_subagents::{ extract_result_message, merge_subagent_state, prepare_subagent_state, AgentState, - CompiledSubAgent, RvAgentConfig, SubAgentSpec, EXCLUDED_STATE_KEYS, + CompiledSubAgent, RvAgentConfig, SubAgentSpec, }; +fn file_data(content: &str) -> FileData { + FileData { + content: content.into(), + encoding: "utf-8".into(), + modified_at: None, + } +} + fn test_config() -> RvAgentConfig { RvAgentConfig { default_model: Some("anthropic:claude-sonnet-4-20250514".into()), @@ -28,24 +36,14 @@ fn mock_compiled(name: &str) -> CompiledSubAgent { } fn parent_state_with_data() -> AgentState { - let mut state = AgentState::new(); - state.insert( - "messages".into(), - serde_json::json!([ - {"type": "system", "content": "You are helpful."}, - {"type": "human", "content": "Do something."}, - ]), - ); - state.insert("remaining_steps".into(), serde_json::json!(10)); - state.insert( - "task_completion".into(), - serde_json::json!({"status": "in_progress"}), - ); - state.insert( - "files".into(), - serde_json::json!({"main.rs": "fn main() {}"}), - ); - state.insert("custom_data".into(), serde_json::json!("value")); + let mut state = AgentState::with_system_message("You are helpful."); + state.push_message(Message::human("Do something.")); + state.push_todo(TodoItem { + content: "parent task".into(), + status: TodoStatus::InProgress, + active_form: String::new(), + }); + state.set_file("main.rs", file_data("fn main() {}")); state } @@ -73,39 +71,22 @@ fn test_state_isolation() { let parent = parent_state_with_data(); let child = prepare_subagent_state(&parent, "Do a subtask"); - // remaining_steps and task_completion should be excluded - assert!( - !child.contains_key("remaining_steps"), - "remaining_steps leaked" - ); - assert!( - !child.contains_key("task_completion"), - "task_completion leaked" - ); + // Parent todos should be excluded + assert!(child.todos.is_empty(), "todos leaked"); // messages is re-created with the task description, not the parent's messages - let child_msgs = child.get("messages").unwrap().as_array().unwrap(); - assert_eq!(child_msgs.len(), 1); - assert!(child_msgs[0]["content"] - .as_str() - .unwrap() - .contains("subtask")); - - // Non-excluded keys should be present - assert!(child.contains_key("files")); - assert!(child.contains_key("custom_data")); + assert_eq!(child.message_count(), 1); + assert!(child.messages[0].content().contains("subtask")); + + // Non-excluded state (files) should be present + assert!(child.files.contains_key("main.rs")); } #[test] fn test_extract_result_message() { let mut state = AgentState::new(); - state.insert( - "messages".into(), - serde_json::json!([ - {"type": "ai", "content": "Working..."}, - {"type": "ai", "content": "Done! Here is the result."} - ]), - ); + state.push_message(Message::ai("Working...")); + state.push_message(Message::ai("Done! Here is the result.")); let result = extract_result_message(&state); assert!(result.is_some()); @@ -115,24 +96,18 @@ fn test_extract_result_message() { #[test] fn test_merge_preserves_parent_messages() { let mut parent = parent_state_with_data(); - let parent_msgs = parent.get("messages").cloned(); + let parent_msg_count = parent.message_count(); let mut child_result = AgentState::new(); - child_result.insert( - "messages".into(), - serde_json::json!([{"type": "ai", "content": "child"}]), - ); - child_result.insert("new_key".into(), serde_json::json!("from child")); + child_result.push_message(Message::ai("child")); + child_result.set_file("child.rs", file_data("from child")); merge_subagent_state(&mut parent, &child_result); // Parent messages must not be overwritten - assert_eq!(parent.get("messages"), parent_msgs.as_ref()); - // New keys from child should be merged - assert_eq!( - parent.get("new_key"), - Some(&serde_json::json!("from child")) - ); + assert_eq!(parent.message_count(), parent_msg_count); + // New files from child should be merged + assert!(parent.files.contains_key("child.rs")); } #[test] @@ -184,8 +159,4 @@ fn test_compilation_respects_capabilities() { fn test_extract_result_empty_messages() { let state = AgentState::new(); assert!(extract_result_message(&state).is_none()); - - let mut state2 = AgentState::new(); - state2.insert("messages".into(), serde_json::json!([])); - assert!(extract_result_message(&state2).is_none()); } diff --git a/crates/rvAgent/rvagent-subagents/tests/orchestrator_tests.rs b/crates/rvAgent/rvagent-subagents/tests/orchestrator_tests.rs index 37261f0254..ec919abe18 100644 --- a/crates/rvAgent/rvagent-subagents/tests/orchestrator_tests.rs +++ b/crates/rvAgent/rvagent-subagents/tests/orchestrator_tests.rs @@ -6,14 +6,24 @@ //! - Result validation (max length, injection detection) //! - Parallel spawning +use rvagent_core::messages::Message; +use rvagent_core::state::{FileData, SkillMetadata, TodoItem, TodoStatus}; use rvagent_subagents::builder::compile_subagents; use rvagent_subagents::orchestrator::{spawn_parallel, SubAgentOrchestrator}; use rvagent_subagents::validator::{SubAgentResultValidator, DEFAULT_MAX_RESPONSE_LENGTH}; use rvagent_subagents::{ merge_subagent_state, prepare_subagent_state, AgentState, CompiledSubAgent, RvAgentConfig, - SubAgentSpec, EXCLUDED_STATE_KEYS, + SubAgentSpec, }; +fn file_data(content: &str) -> FileData { + FileData { + content: content.into(), + encoding: "utf-8".into(), + modified_at: None, + } +} + // --------------------------------------------------------------------------- // Helpers // --------------------------------------------------------------------------- @@ -42,38 +52,26 @@ fn mock_compiled(name: &str) -> CompiledSubAgent { } fn parent_state_with_secrets() -> AgentState { - let mut state = AgentState::new(); - state.insert( - "messages".into(), - serde_json::json!([ - {"type": "system", "content": "You are a helpful assistant."}, - {"type": "human", "content": "Help me refactor main.rs"}, - {"type": "ai", "content": "I'll help you refactor."}, - ]), - ); - state.insert("remaining_steps".into(), serde_json::json!(42)); - state.insert("task_completion".into(), serde_json::json!({"done": false})); - state.insert( - "todos".into(), - serde_json::json!([ - {"id": "1", "content": "Fix bug", "status": "in_progress"} - ]), - ); - state.insert( - "structured_response".into(), - serde_json::json!({"format": "markdown"}), - ); - state.insert( - "skills_metadata".into(), - serde_json::json!([{"name": "coder"}]), - ); - state.insert( - "memory_contents".into(), - serde_json::json!({"AGENTS.md": "secret"}), - ); - // Non-excluded keys - state.insert("cwd".into(), serde_json::json!("/home/user/project")); - state.insert("project_config".into(), serde_json::json!({"lang": "rust"})); + let mut state = AgentState::with_system_message("You are a helpful assistant."); + state.push_message(Message::human("Help me refactor main.rs")); + state.push_message(Message::ai("I'll help you refactor.")); + state.push_todo(TodoItem { + content: "Fix bug".into(), + status: TodoStatus::InProgress, + active_form: String::new(), + }); + state.skills_metadata = Some(std::sync::Arc::new(vec![SkillMetadata { + name: "coder".into(), + description: "Writes code".into(), + parameters: serde_json::json!({}), + }])); + state.memory_contents = Some(std::sync::Arc::new( + [("AGENTS.md".to_string(), "secret".to_string())] + .into_iter() + .collect(), + )); + // Non-excluded state + state.set_file("/home/user/project/main.rs", file_data("fn main() {}")); state } @@ -147,85 +145,58 @@ fn test_state_isolation() { // Prepare child state let child = prepare_subagent_state(&parent, "Refactor the auth module"); - // ALL excluded keys must not appear in child state (except messages which is replaced) - for key in EXCLUDED_STATE_KEYS { - if *key == "messages" { - // Messages is replaced, not excluded entirely - continue; - } - assert!( - !child.contains_key(*key), - "Excluded key '{}' must not appear in child state", - key - ); - } - - // Verify specific excluded keys - assert!(!child.contains_key("remaining_steps")); - assert!(!child.contains_key("task_completion")); - assert!(!child.contains_key("todos")); - assert!(!child.contains_key("structured_response")); - assert!(!child.contains_key("skills_metadata")); - assert!(!child.contains_key("memory_contents")); + // Excluded state must not appear in child state + assert!(child.todos.is_empty(), "todos must not leak"); + assert!( + child.skills_metadata.is_none(), + "skills_metadata must not leak" + ); + assert!( + child.memory_contents.is_none(), + "memory_contents must not leak" + ); // Messages must be replaced with task description - let child_msgs = child.get("messages").unwrap().as_array().unwrap(); - assert_eq!(child_msgs.len(), 1, "Child must have exactly 1 message"); - assert_eq!(child_msgs[0]["type"], "human"); - assert!(child_msgs[0]["content"] - .as_str() - .unwrap() + assert_eq!(child.message_count(), 1, "Child must have exactly 1 message"); + assert!(matches!( + child.messages[0], + rvagent_core::messages::Message::Human(_) + )); + assert!(child.messages[0] + .content() .contains("Refactor the auth module")); - // Non-excluded keys must pass through - assert_eq!( - child.get("cwd").unwrap(), - &serde_json::json!("/home/user/project") - ); - assert_eq!( - child.get("project_config").unwrap(), - &serde_json::json!({"lang": "rust"}) - ); + // Non-excluded state (files) must pass through + assert!(child.files.contains_key("/home/user/project/main.rs")); - // Verify merge doesn't leak excluded keys back + // Verify merge doesn't leak excluded state back let mut parent_copy = parent_state_with_secrets(); - let parent_msgs_before = parent_copy.get("messages").cloned(); + let parent_msgs_before = parent_copy.message_count(); + let parent_todo_before = parent_copy.todos[0].content.clone(); let mut child_result = AgentState::new(); - child_result.insert( - "messages".into(), - serde_json::json!([ - {"type": "ai", "content": "Refactoring complete."} - ]), - ); - child_result.insert( - "todos".into(), - serde_json::json!([ - {"id": "child-1", "content": "leaked todo"} - ]), - ); - child_result.insert("new_discovery".into(), serde_json::json!("found a bug")); + child_result.push_message(Message::ai("Refactoring complete.")); + child_result.push_todo(TodoItem { + content: "leaked todo".into(), + status: TodoStatus::Pending, + active_form: String::new(), + }); + child_result.set_file("/new_discovery.md", file_data("found a bug")); merge_subagent_state(&mut parent_copy, &child_result); // Parent messages must NOT be overwritten by child - assert_eq!(parent_copy.get("messages"), parent_msgs_before.as_ref()); + assert_eq!(parent_copy.message_count(), parent_msgs_before); // Child's todos must NOT leak to parent - let parent_todos = parent_copy.get("todos").unwrap(); - assert!( - parent_todos.as_array().unwrap()[0]["content"] - .as_str() - .unwrap() - .contains("Fix bug"), + assert_eq!(parent_copy.todos.len(), 1); + assert_eq!( + parent_copy.todos[0].content, parent_todo_before, "Parent todos must not be overwritten by child" ); - // New non-excluded keys should merge - assert_eq!( - parent_copy.get("new_discovery"), - Some(&serde_json::json!("found a bug")) - ); + // New non-excluded state should merge + assert!(parent_copy.files.contains_key("/new_discovery.md")); } // =========================================================================== diff --git a/crates/rvAgent/rvagent-subagents/tests/security_validation.rs b/crates/rvAgent/rvagent-subagents/tests/security_validation.rs index 97d4ebdd2b..fbb98f99e6 100644 --- a/crates/rvAgent/rvagent-subagents/tests/security_validation.rs +++ b/crates/rvAgent/rvagent-subagents/tests/security_validation.rs @@ -6,7 +6,6 @@ use rvagent_subagents::{ spawn_parallel, AgentState, CompiledSubAgent, SpawnError, SubAgentOrchestrator, SubAgentSpec, ValidationConfig, ValidationError, }; -use std::collections::HashMap; fn create_test_orchestrator() -> SubAgentOrchestrator { let spec = SubAgentSpec::new("test-agent", "Do the thing"); @@ -33,7 +32,7 @@ fn create_test_orchestrator_with_config(config: ValidationConfig) -> SubAgentOrc } fn create_empty_state() -> AgentState { - HashMap::new() + AgentState::new() } #[test] From 113bf404d54869351287326bed9b83bcf722d4dd Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 16:04:41 +0000 Subject: [PATCH 04/25] docs(rvagent): align docs with implementation reality (P0.5) - README: 10-crate map (adds rvagent-mcp, rvagent-a2a); parallel-tools claims now describe the actual bounded-concurrency JoinSet behavior; removed unverified 5-10x and HNSW O(log n) semantic-search claims (hnsw middleware is an experimental hash-embedding placeholder) - architecture.md: adds mcp/a2a to the dependency graph and an explicit current-gaps note (no SSE streaming, stubbed subagent spawning, placeholder hnsw embeddings) - middleware doc comments: drop inherited 150x-12,500x claim Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/README.md | 28 +++++++++---------- crates/rvAgent/rvagent-middleware/src/hnsw.rs | 3 +- crates/rvAgent/rvagent-middleware/src/lib.rs | 2 +- docs/rvagent/architecture.md | 15 +++++++++- 4 files changed, 31 insertions(+), 17 deletions(-) diff --git a/crates/rvAgent/README.md b/crates/rvAgent/README.md index fa06a6302a..b85bf5cd57 100644 --- a/crates/rvAgent/README.md +++ b/crates/rvAgent/README.md @@ -50,8 +50,7 @@ let subagent_state = state.clone(); // No deep copy! | Feature | What It Does | Why It Matters | |---------|--------------|----------------| | **O(1) State Cloning** | Clone agent state instantly via Arc | Spawn subagents without copying gigabytes of context | -| **Parallel Tool Execution** | Run multiple tools simultaneously | 5-10x faster than sequential execution | -| **HNSW Semantic Search** | O(log n) memory retrieval | Find relevant context in millions of entries | +| **Parallel Tool Execution** | Bounded-concurrency JoinSet execution | Wall clock = slowest tool, not the sum; failures isolated per call | | **Single-Allocation Formatting** | Pre-calculated output buffers | No memory fragmentation under load | ### 🔒 Security @@ -120,17 +119,19 @@ How does rvAgent compare to other agent frameworks? ## Architecture -rvAgent is organized as 8 crates within the RuVector workspace: +rvAgent is organized as 10 crates within the RuVector workspace: ``` rvAgent/ - rvagent-core Core types, COW state, AGI containers, session encryption - rvagent-backends Backend protocol trait + sandbox security contracts - rvagent-middleware Middleware trait + 14 middleware implementations (incl. SONA, HNSW) - rvagent-tools Tool trait + 8 built-in tools (enum dispatch) + rvagent-core Core types, agent graph, COW state, AGI containers, session encryption + rvagent-backends Backend protocol trait + sandbox security + Anthropic/Gemini clients + rvagent-middleware Middleware trait + async pipeline + 14 middleware implementations + rvagent-tools Tool trait + 9 built-in tools (enum dispatch) rvagent-subagents SubAgent spec, CRDT merge, result validation, orchestration rvagent-cli Terminal coding agent (ratatui TUI) rvagent-acp Agent Communication Protocol server (axum) with auth + rvagent-mcp MCP server/client (JSON-RPC 2.0, stdio + SSE transports) + rvagent-a2a Agent2Agent peer protocol (signed cards, budgets, routing) rvagent-wasm WASM bindings for browser/Node.js ``` @@ -273,8 +274,7 @@ rvAgent solves these with Rust's zero-cost abstractions. |-----------|---------|-------------------|---------| | State cloning | <1μs (O(1)) | ~10ms (deep copy) | 10,000x | | Tool dispatch | No overhead (enum) | ~1ms (vtable lookup) | Direct | -| Parallel tools | True multi-threaded | Async (still serial) | Linear scaling | -| Memory search | O(log n) via HNSW | O(n) linear scan | 100-1000x on large sets | +| Parallel tools | Spawned tasks, bounded concurrency | Async (still serial) | Wall clock = slowest tool | ### Key Optimizations @@ -297,11 +297,11 @@ tools: ["read_file", "grep", "execute", "read_file", "glob"] let formatted = format_content_with_line_numbers(content); ``` -**HNSW Semantic Search** — Find relevant memories in massive datasets -```rust -// O(log n) retrieval instead of scanning everything -let relevant = hnsw.search("authentication bug", top_k=5); -``` +**Memory retrieval (experimental)** — The `hnsw` middleware ships a simplified +in-process index with a hash-based embedding placeholder. It is NOT semantic +search yet: integration with real RuVector embeddings is tracked in the +Hermes-class harness roadmap (`docs/research/rvagent-hermes-harness/`), and +no retrieval performance claims are made until it lands. ## Advanced Features diff --git a/crates/rvAgent/rvagent-middleware/src/hnsw.rs b/crates/rvAgent/rvagent-middleware/src/hnsw.rs index c5b70e3986..73eac9bd01 100644 --- a/crates/rvAgent/rvagent-middleware/src/hnsw.rs +++ b/crates/rvAgent/rvagent-middleware/src/hnsw.rs @@ -12,7 +12,8 @@ //! //! # Performance //! -//! - 150x-12,500x faster than brute-force search +//! - EXPERIMENTAL: uses a hash-based embedding placeholder (not semantic); +//! no retrieval performance claims until real embeddings are integrated //! - O(log n) search complexity //! - Sub-millisecond latency for 10k vectors diff --git a/crates/rvAgent/rvagent-middleware/src/lib.rs b/crates/rvAgent/rvagent-middleware/src/lib.rs index f005f0ab32..c4e0163332 100644 --- a/crates/rvAgent/rvagent-middleware/src/lib.rs +++ b/crates/rvAgent/rvagent-middleware/src/lib.rs @@ -18,7 +18,7 @@ //! ## ADR-103 Learning Middleware (B5, B6) //! //! - [`sona`] — SONA Adaptive Learning with three loops (instant, background, deep) -//! - [`hnsw`] — HNSW Semantic Retrieval for skills and memory (150x-12,500x faster) +//! - [`hnsw`] — experimental in-process retrieval index (hash-embedding placeholder) pub mod filesystem; pub mod hitl; diff --git a/docs/rvagent/architecture.md b/docs/rvagent/architecture.md index 7c93770ff8..333f107efd 100644 --- a/docs/rvagent/architecture.md +++ b/docs/rvagent/architecture.md @@ -1,6 +1,11 @@ # rvAgent Architecture -This document describes the internal architecture of the rvAgent crate family, covering the crate dependency graph, agent lifecycle, middleware pipeline, backend protocol hierarchy, security model, and performance characteristics. +This document describes the internal architecture of the rvAgent crate family (10 crates, including `rvagent-mcp` and `rvagent-a2a`), covering the crate dependency graph, agent lifecycle, middleware pipeline, backend protocol hierarchy, security model, and performance characteristics. + +> Current gaps (tracked in `docs/research/rvagent-hermes-harness/03-roadmap.md`): +> SSE streaming is not implemented (providers fall back to non-streaming +> completion); subagent orchestration spawning is stubbed; the `hnsw` +> middleware uses a hash-based embedding placeholder, not semantic search. ## Crate Dependency Graph @@ -28,6 +33,14 @@ rvagent-acp |-- rvagent-tools |-- rvagent-subagents | +rvagent-mcp +|-- rvagent-core +|-- rvagent-middleware (skills bridge) +| +rvagent-a2a +|-- rvagent-core +|-- rvagent-middleware +| rvagent-wasm |-- (standalone, no workspace deps except serde/wasm-bindgen) ``` From 29e1805468fdfcc024df1ae4f0ae2af3cc7244c2 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 16:16:23 +0000 Subject: [PATCH 05/25] fix(rvagent): security hardening of the Phase 0 agent loop Findings from a security review of the Phase 0 diff: - MEDIUM: a panicking tool aborted the entire agent process. Tool args are model-controlled and indirectly attacker-influenced via prompt injection, so any panic on crafted input was a DoS. Each parallel tool call now runs in a nested spawn; JoinError becomes a normal tool-error result on the model's recovery path. Regression test added. - MEDIUM: RetryMiddleware treated empty content as a transient error. Now that tool schemas are actually sent, tool-call-only turns legitimately have empty text and would be retried up to max_retries (4x token cost, valid responses discarded). Empty content now only counts as an error when tool_calls is also empty. - MEDIUM: the Gemini API key was passed in the URL query string, where it lands in reqwest error Displays (which are logged and now also surface into the middleware chain) and proxy logs. Moved to the x-goog-api-key header. - LOW: synthesized Gemini call IDs (index+name) collided across turns, letting one turn's tool result satisfy another turn's call in id-keyed middleware. IDs now use a process-wide atomic counter; functionResponse name resolution prefers the recorded tool_name over the id fallback. Verified safe, no change needed: tool errors go through the same tool_result_sanitizer as success output on both CLI and ACP paths; HITL-denied tool calls stay denied through PipelineModel reattachment; COW state gives executors immutable snapshots (no torn reads); max_parallel_tools bounds concurrency; advertised tool schemas are static literals with no config or secret material. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-backends/src/gemini.rs | 56 ++++++++++++--- crates/rvAgent/rvagent-core/src/graph.rs | 70 ++++++++++++++++++- .../rvAgent/rvagent-middleware/src/retry.rs | 21 +++++- 3 files changed, 134 insertions(+), 13 deletions(-) diff --git a/crates/rvAgent/rvagent-backends/src/gemini.rs b/crates/rvAgent/rvagent-backends/src/gemini.rs index e65a4a73a7..28959ad922 100644 --- a/crates/rvAgent/rvagent-backends/src/gemini.rs +++ b/crates/rvAgent/rvagent-backends/src/gemini.rs @@ -237,9 +237,14 @@ impl GeminiClient { }); } Message::Tool(t) => { - let name = call_id_to_name - .get(&t.tool_call_id) - .cloned() + // Prefer the tool name recorded on the message itself, + // then the id→name map from the preceding AI turn. The + // raw id is a last resort only (it is not a valid + // function name and Gemini may reject it). + let name = t + .tool_name + .clone() + .or_else(|| call_id_to_name.get(&t.tool_call_id).cloned()) .unwrap_or_else(|| t.tool_call_id.clone()); contents.push(GeminiContent { role: "user".to_string(), @@ -287,9 +292,12 @@ impl GeminiClient { /// Send a request to the API with retry logic. async fn send_with_retry(&self, request_body: &GeminiRequest) -> Result { + // The API key travels in the `x-goog-api-key` header, never in the + // URL query string — URLs end up in error messages, proxy logs, and + // tracing output, which must never contain credentials. let url = format!( - "{}/{}:generateContent?key={}", - GEMINI_API_BASE, self.config.model_id, self.api_key + "{}/{}:generateContent", + GEMINI_API_BASE, self.config.model_id ); let mut last_err: Option = None; @@ -310,6 +318,7 @@ impl GeminiClient { let result = self .http .post(&url) + .header("x-goog-api-key", &self.api_key) .header("content-type", "application/json") .body(body_json) .send() @@ -373,17 +382,22 @@ impl ChatModel for GeminiClient { let response = self.send_with_retry(&request).await?; // Collect text and function-call parts from the first candidate. - // Gemini does not assign tool-call IDs, so synthesize stable ones. + // Gemini does not assign tool-call IDs, so synthesize unique ones: + // a process-wide counter prevents id collisions across turns (the + // per-part index alone repeats every turn, which would let one + // turn's tool result satisfy another turn's call in id-keyed logic). + static CALL_COUNTER: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0); let mut text_parts: Vec = Vec::new(); let mut tool_calls: Vec = Vec::new(); if let Some(candidate) = response.candidates.first() { - for (idx, part) in candidate.content.parts.iter().enumerate() { + for part in &candidate.content.parts { if let Some(text) = &part.text { text_parts.push(text.clone()); } if let Some(fc) = &part.function_call { + let n = CALL_COUNTER.fetch_add(1, std::sync::atomic::Ordering::Relaxed); tool_calls.push(ToolCall { - id: format!("gemini_call_{idx}_{}", fc.name), + id: format!("gemini_call_{n}_{}", fc.name), name: fc.name.clone(), args: fc.args.clone(), }); @@ -584,4 +598,30 @@ mod tests { assert_eq!(fr.name, "read_file"); assert_eq!(fr.response, serde_json::json!({"result": "contents"})); } + + #[test] + fn test_gemini_tool_result_prefers_recorded_tool_name() { + let config = ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }; + let client = GeminiClient { + config, + http: reqwest::Client::new(), + api_key: "test".to_string(), + }; + + // Orphaned tool message (no preceding AI turn, e.g. after history + // trimming): the recorded tool_name must win over the raw id. + let messages = vec![ + Message::human("read it"), + Message::tool_with_name("some_opaque_id", "contents", "read_file"), + ]; + let req = client.build_request(&messages, &[]); + let fr = req.contents[1].parts[0].function_response.as_ref().unwrap(); + assert_eq!(fr.name, "read_file"); + } } diff --git a/crates/rvAgent/rvagent-core/src/graph.rs b/crates/rvAgent/rvagent-core/src/graph.rs index d50c00a50c..7d5063bd8f 100644 --- a/crates/rvAgent/rvagent-core/src/graph.rs +++ b/crates/rvAgent/rvagent-core/src/graph.rs @@ -233,8 +233,22 @@ impl AgentGraph { let executor = Arc::clone(&executor); let exec_state = exec_state.clone(); async move { - let result = executor.execute(&tc, &exec_state).await; - (tc.id, tc.name, result) + let id = tc.id.clone(); + let name = tc.name.clone(); + // Run the tool in its own task so a panicking + // tool surfaces as a tool error instead of + // crashing the whole agent loop. + let result = match tokio::spawn(async move { + executor.execute(&tc, &exec_state).await + }) + .await + { + Ok(res) => res, + Err(join_err) => Err(RvAgentError::tool(format!( + "tool '{name}' execution task failed: {join_err}" + ))), + }; + (id, name, result) } }, self.config.max_parallel_tools.max(1), @@ -403,6 +417,58 @@ mod tests { assert_eq!(result.message_count(), 4); } + /// A tool executor that panics on one tool and succeeds on others. + struct PanickyExecutor; + + #[async_trait] + impl ToolExecutor for PanickyExecutor { + async fn execute(&self, call: &ToolCall, _state: &AgentState) -> Result { + if call.name == "boom" { + panic!("tool panicked"); + } + Ok(format!("result of {}", call.name)) + } + } + + #[tokio::test] + async fn test_parallel_tool_panic_is_contained() { + let model = MockModel::new(vec![ + Message::ai_with_tools( + "", + vec![ + ToolCall { + id: "tc1".into(), + name: "boom".into(), + args: serde_json::json!({}), + }, + ToolCall { + id: "tc2".into(), + name: "ok_tool".into(), + args: serde_json::json!({}), + }, + ], + ), + Message::ai("done"), + ]); + let graph = AgentGraph::new(model, PanickyExecutor); + + let state = AgentState::new(); + // Must not panic: the panicking tool becomes an error tool-result. + let result = graph.run(state).await.unwrap(); + + let contents: Vec<&str> = result + .messages + .iter() + .filter_map(|m| match m { + Message::Tool(t) => Some(t.content.as_str()), + _ => None, + }) + .collect(); + assert_eq!(contents.len(), 2); + assert!(contents[0].contains("Tool execution error")); + assert!(contents[1].contains("result of ok_tool")); + } + #[tokio::test] async fn test_max_iterations() { // Model always returns tool calls → should hit max iterations. diff --git a/crates/rvAgent/rvagent-middleware/src/retry.rs b/crates/rvAgent/rvagent-middleware/src/retry.rs index 35c9365daa..298d1b9d70 100644 --- a/crates/rvAgent/rvagent-middleware/src/retry.rs +++ b/crates/rvAgent/rvagent-middleware/src/retry.rs @@ -12,11 +12,14 @@ use crate::{Middleware, ModelHandler, ModelRequest, ModelResponse}; /// Determines whether a `ModelResponse` represents a transient error worth retrying. /// -/// Heuristic: the response is considered an error if its content is empty or -/// starts with the prefix `"error:"` (case-insensitive). +/// Heuristic: the response is considered an error if its content starts with +/// the prefix `"error:"` (case-insensitive), or if it is completely empty — +/// no text AND no tool calls. A response with tool calls but no text is a +/// perfectly valid tool-use turn and must never be retried. fn is_transient_error(response: &ModelResponse) -> bool { let content = response.content(); - content.is_empty() || content.to_ascii_lowercase().starts_with("error:") + (content.is_empty() && response.tool_calls.is_empty()) + || content.to_ascii_lowercase().starts_with("error:") } /// Retry middleware that wraps model calls with exponential backoff. @@ -245,6 +248,18 @@ mod tests { assert!(is_transient_error(&resp)); } + #[test] + fn test_is_transient_error_empty_content_with_tool_calls() { + // A tool-use turn often has no text content — it is NOT an error. + let mut resp = ModelResponse::text(""); + resp.tool_calls = vec![crate::ToolCall { + id: "tc1".into(), + name: "read_file".into(), + args: serde_json::json!({}), + }]; + assert!(!is_transient_error(&resp)); + } + #[test] fn test_is_transient_error_normal_response() { let resp = ModelResponse::text("Here is the answer."); From 848cbaeb9d6924151ced4ee1cf3b3bfafa41206b Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 16:46:33 +0000 Subject: [PATCH 06/25] =?UTF-8?q?feat(rvagent):=20P0=20exit=20gate=20?= =?UTF-8?q?=E2=80=94=20e2e=20tool-calling=20test=20+=20workspace=20path=20?= =?UTF-8?q?confinement?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds the Phase 0 exit-gate test: an end-to-end verification that wires the real builtin tool registry, the real filesystem backend, and the real AgentGraph loop against a scripted model. Only the provider network call is faked, so the gate fails if the loop, the schemas, or the tools regress. Writing the gate surfaced two problems. 1. The filesystem backend applied no path confinement. Tool-supplied paths were passed straight to std::fs, so a model-issued read_file or write_file with an absolute path (or ../ traversal) escaped the working directory entirely. Paths from the model are untrusted input and were treated as trusted. LocalFsBackend now resolves every path against a fixed root: lexical .. normalization first, then canonicalization of the deepest existing ancestor so a symlink inside the root cannot bridge out, then a containment check. Escaping reads and writes are refused before touching the filesystem. 2. Two of the gate's own assertions passed vacuously — a wrong parameter name produced "file_path is required", which satisfied a loose contains("error") check while never exercising the boundary under test. The confinement assertions now require the specific refusal message. The backend moves from the CLI binary into rvagent-tools so the gate exercises the shipped code path rather than a lookalike copy; the CLI now consumes it (-336 lines of duplication). Covers: schemas advertised on every turn including after tool results, real side effects on disk, tool errors feeding back as results without aborting the loop, unknown tools reported not fatal, parallel execution preserving call order and agreeing with sequential, usage metadata aggregation, and confinement holding through the full loop. Tests: 94 unit + 10 e2e in rvagent-tools; rvagent-cli/core/middleware/acp green. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-cli/src/app.rs | 342 +----------- crates/rvAgent/rvagent-tools/src/lib.rs | 2 + crates/rvAgent/rvagent-tools/src/local.rs | 524 ++++++++++++++++++ .../rvagent-tools/tests/e2e_tool_calling.rs | 516 +++++++++++++++++ 4 files changed, 1045 insertions(+), 339 deletions(-) create mode 100644 crates/rvAgent/rvagent-tools/src/local.rs create mode 100644 crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs diff --git a/crates/rvAgent/rvagent-cli/src/app.rs b/crates/rvAgent/rvagent-cli/src/app.rs index 4d622e23a5..47d513f97b 100644 --- a/crates/rvAgent/rvagent-cli/src/app.rs +++ b/crates/rvAgent/rvagent-cli/src/app.rs @@ -141,9 +141,9 @@ struct CliToolExecutor { impl CliToolExecutor { fn new(cwd: &Path) -> Self { - let backend: rvagent_tools::BackendRef = Arc::new(LocalFsBackend { - cwd: cwd.to_path_buf(), - }); + // Confined to `cwd`: tool-supplied paths cannot escape the workspace. + let backend: rvagent_tools::BackendRef = + Arc::new(rvagent_tools::LocalFsBackend::new(cwd)); Self { tools: rvagent_tools::builtin_tools(), backend, @@ -180,342 +180,6 @@ impl ToolExecutor for CliToolExecutor { } } -// --------------------------------------------------------------------------- -// LocalFsBackend — adapts the local filesystem for rvagent_tools::Backend -// --------------------------------------------------------------------------- - -/// A minimal filesystem backend implementing `rvagent_tools::Backend` for CLI use. -/// -/// Provides real filesystem and shell operations rooted at a working directory. -struct LocalFsBackend { - cwd: PathBuf, -} - -impl rvagent_tools::Backend for LocalFsBackend { - fn ls_info(&self, path: &str) -> std::result::Result, String> { - let target = if path.is_empty() || path == "." { - self.cwd.clone() - } else { - PathBuf::from(path) - }; - let entries = std::fs::read_dir(&target) - .map_err(|e| format!("ls failed on '{}': {}", target.display(), e))?; - let mut infos = Vec::new(); - for entry in entries { - let entry = entry.map_err(|e| format!("read_dir entry error: {}", e))?; - let meta = entry - .metadata() - .map_err(|e| format!("metadata error: {}", e))?; - let file_type = if meta.is_dir() { - "directory" - } else if meta.is_symlink() { - "symlink" - } else { - "file" - }; - infos.push(rvagent_tools::FileInfo { - name: entry.file_name().to_string_lossy().into_owned(), - file_type: file_type.to_string(), - permissions: String::new(), - size: meta.len(), - }); - } - infos.sort_by(|a, b| a.name.cmp(&b.name)); - Ok(infos) - } - - fn read(&self, path: &str, offset: usize, limit: usize) -> std::result::Result { - let content = - std::fs::read_to_string(path).map_err(|e| format!("read '{}': {}", path, e))?; - let lines: Vec<&str> = content.lines().collect(); - if offset >= lines.len() { - return Ok(String::new()); - } - let end = (offset + limit).min(lines.len()); - Ok(lines[offset..end].join("\n")) - } - - fn write(&self, path: &str, content: &str) -> rvagent_tools::WriteResult { - if std::path::Path::new(path).exists() { - return rvagent_tools::WriteResult { - error: Some(format!( - "Error: file {} already exists. Use force flag to overwrite.", - path - )), - ..Default::default() - }; - } - if let Some(parent) = std::path::Path::new(path).parent() { - if let Err(e) = std::fs::create_dir_all(parent) { - return rvagent_tools::WriteResult { - error: Some(format!("mkdir failed: {}", e)), - ..Default::default() - }; - } - } - match std::fs::write(path, content) { - Ok(_) => rvagent_tools::WriteResult::default(), - Err(e) => rvagent_tools::WriteResult { - error: Some(format!("write '{}': {}", path, e)), - ..Default::default() - }, - } - } - - fn edit( - &self, - path: &str, - old_string: &str, - new_string: &str, - replace_all: bool, - ) -> rvagent_tools::WriteResult { - let content = match std::fs::read_to_string(path) { - Ok(c) => c, - Err(e) => { - return rvagent_tools::WriteResult { - error: Some(format!("read '{}': {}", path, e)), - ..Default::default() - } - } - }; - let count = content.matches(old_string).count(); - if count == 0 { - return rvagent_tools::WriteResult { - error: Some(format!("Error: old_string not found in {}", path)), - ..Default::default() - }; - } - if count > 1 && !replace_all { - return rvagent_tools::WriteResult { - error: Some(format!( - "Error: old_string is not unique in {} ({} occurrences). Use replace_all=true.", - path, count - )), - ..Default::default() - }; - } - let new_content = if replace_all { - content.replace(old_string, new_string) - } else { - content.replacen(old_string, new_string, 1) - }; - match std::fs::write(path, &new_content) { - Ok(_) => rvagent_tools::WriteResult { - error: None, - occurrences: Some(if replace_all { count } else { 1 }), - ..Default::default() - }, - Err(e) => rvagent_tools::WriteResult { - error: Some(format!("write '{}': {}", path, e)), - ..Default::default() - }, - } - } - - fn glob_info(&self, pattern: &str, path: &str) -> std::result::Result, String> { - let base = if path.is_empty() || path == "." { - self.cwd.clone() - } else { - PathBuf::from(path) - }; - // Simple glob: walk directory and match by extension or name suffix. - // This handles common patterns like "*.rs", "**/*.toml" without - // requiring the `glob` crate. - let suffix = pattern - .trim_start_matches('*') - .trim_start_matches('/') - .trim_start_matches('*'); - let mut results = Vec::new(); - collect_glob_matches(&base, suffix, &mut results); - results.sort(); - Ok(results) - } - - fn grep_raw( - &self, - pattern: &str, - path: Option<&str>, - _include: Option<&str>, - ) -> std::result::Result, String> { - // Simple in-process grep implementation. - let search_dir = match path { - Some(p) if !p.is_empty() => PathBuf::from(p), - _ => self.cwd.clone(), - }; - let mut matches = Vec::new(); - if search_dir.is_file() { - grep_file(&search_dir, pattern, &mut matches)?; - } else if search_dir.is_dir() { - grep_dir(&search_dir, pattern, &mut matches)?; - } - Ok(matches) - } - - fn execute( - &self, - command: &str, - timeout_secs: u32, - ) -> std::result::Result { - use std::process::{Command, Stdio}; - use std::time::Duration; - - // Security: environment sanitization — strip sensitive variables (SEC-005 / ADR-103 C2). - // Only pass through a safe allowlist of environment variables. - const SAFE_ENV_VARS: &[&str] = &[ - "PATH", "HOME", "USER", "SHELL", "LANG", "LC_ALL", "LC_CTYPE", "TERM", "TMPDIR", "TZ", - ]; - // Patterns that identify sensitive env vars that must never reach child processes. - const SENSITIVE_PATTERNS: &[&str] = &[ - "SECRET", - "KEY", - "TOKEN", - "PASSWORD", - "CREDENTIAL", - "AWS_", - "AZURE_", - "GCP_", - "DATABASE_URL", - "PRIVATE", - "API_KEY", - "AUTH", - "BEARER", - "JWT", - "SESSION", - ]; - - let mut cmd = Command::new("sh"); - cmd.arg("-c").arg(command).current_dir(&self.cwd); - cmd.env_clear(); - for var in SAFE_ENV_VARS { - if let Ok(val) = std::env::var(var) { - let upper = var.to_uppercase(); - let sensitive = SENSITIVE_PATTERNS.iter().any(|pat| upper.contains(pat)); - if !sensitive { - cmd.env(var, val); - } - } - } - cmd.stdout(Stdio::piped()).stderr(Stdio::piped()); - - let timeout = if timeout_secs == 0 { 30 } else { timeout_secs }; - let deadline = std::time::Instant::now() + Duration::from_secs(timeout as u64); - - let mut child = cmd.spawn().map_err(|e| format!("execute failed: {}", e))?; - - // Poll for completion with a deadline to enforce the timeout. - loop { - match child - .try_wait() - .map_err(|e| format!("wait failed: {}", e))? - { - Some(_) => break, - None => { - if std::time::Instant::now() >= deadline { - let _ = child.kill(); - return Ok(rvagent_tools::ExecuteResponse { - output: format!("Command timed out after {} seconds", timeout), - exit_code: -1, - }); - } - std::thread::sleep(Duration::from_millis(50)); - } - } - } - - let output = child - .wait_with_output() - .map_err(|e| format!("output collection failed: {}", e))?; - - let stdout = String::from_utf8_lossy(&output.stdout); - let stderr = String::from_utf8_lossy(&output.stderr); - let mut combined = if stderr.is_empty() { - stdout.into_owned() - } else { - format!("{}\n{}", stdout, stderr) - }; - - // Security: cap output size to 1 MB to prevent memory exhaustion. - const MAX_OUTPUT_BYTES: usize = 1024 * 1024; - if combined.len() > MAX_OUTPUT_BYTES { - combined.truncate(MAX_OUTPUT_BYTES); - combined.push_str("\n... [output truncated at 1 MB]"); - } - - Ok(rvagent_tools::ExecuteResponse { - output: combined, - exit_code: output.status.code().unwrap_or(-1), - }) - } -} - -/// Recursively collect files matching a name suffix (simple glob substitute). -fn collect_glob_matches(dir: &Path, suffix: &str, results: &mut Vec) { - let entries = match std::fs::read_dir(dir) { - Ok(e) => e, - Err(_) => return, - }; - for entry in entries.flatten() { - let path = entry.path(); - let name = path - .file_name() - .map(|n| n.to_string_lossy().to_string()) - .unwrap_or_default(); - if path.is_file() && name.ends_with(suffix) { - results.push(path.to_string_lossy().into_owned()); - } else if path.is_dir() && !name.starts_with('.') { - collect_glob_matches(&path, suffix, results); - } - } -} - -/// Grep a single file for a pattern. -fn grep_file( - path: &Path, - pattern: &str, - matches: &mut Vec, -) -> std::result::Result<(), String> { - let content = match std::fs::read_to_string(path) { - Ok(c) => c, - Err(_) => return Ok(()), // skip binary / unreadable files - }; - for (i, line) in content.lines().enumerate() { - if line.contains(pattern) { - matches.push(rvagent_tools::GrepMatch { - file: path.to_string_lossy().into_owned(), - line_number: i + 1, - text: line.to_string(), - }); - } - } - Ok(()) -} - -/// Recursively grep a directory (limited depth). -fn grep_dir( - dir: &Path, - pattern: &str, - matches: &mut Vec, -) -> std::result::Result<(), String> { - let entries = std::fs::read_dir(dir).map_err(|e| format!("read_dir: {}", e))?; - for entry in entries { - let entry = entry.map_err(|e| format!("entry: {}", e))?; - let path = entry.path(); - if path.is_file() { - grep_file(&path, pattern, matches)?; - } else if path.is_dir() { - // Skip hidden directories. - let name = path - .file_name() - .map(|n| n.to_string_lossy().to_string()) - .unwrap_or_default(); - if !name.starts_with('.') { - grep_dir(&path, pattern, matches)?; - } - } - } - Ok(()) -} - // --------------------------------------------------------------------------- // App // --------------------------------------------------------------------------- diff --git a/crates/rvAgent/rvagent-tools/src/lib.rs b/crates/rvAgent/rvagent-tools/src/lib.rs index 0952417d85..00bfc60375 100644 --- a/crates/rvAgent/rvagent-tools/src/lib.rs +++ b/crates/rvAgent/rvagent-tools/src/lib.rs @@ -7,6 +7,7 @@ pub mod edit_file; pub mod execute; pub mod glob; pub mod grep; +pub mod local; pub mod ls; pub mod read_file; pub mod task; @@ -23,6 +24,7 @@ pub use edit_file::EditFileTool; pub use execute::ExecuteTool; pub use glob::GlobTool; pub use grep::GrepTool; +pub use local::LocalFsBackend; pub use ls::LsTool; pub use read_file::ReadFileTool; pub use task::TaskTool; diff --git a/crates/rvAgent/rvagent-tools/src/local.rs b/crates/rvAgent/rvagent-tools/src/local.rs new file mode 100644 index 0000000000..cea7b1a29f --- /dev/null +++ b/crates/rvAgent/rvagent-tools/src/local.rs @@ -0,0 +1,524 @@ +//! Local filesystem backend — the real `Backend` used by the CLI and by the +//! end-to-end harness tests. +//! +//! # Path confinement +//! +//! Every path a tool supplies is model-controlled input, so it is resolved +//! against a fixed root and rejected if it escapes. Resolution is symlink-aware: +//! the deepest existing ancestor is canonicalized before the containment check, +//! so a symlink pointing outside the root cannot be used as a bridge. Without +//! this, `read_file {"path": "/etc/passwd"}` — or a `write_file` anywhere on +//! disk — is a single tool call away. + +use std::path::{Component, Path, PathBuf}; + +use crate::{Backend, ExecuteResponse, FileInfo, GrepMatch, WriteResult}; + +/// A filesystem backend rooted at, and confined to, a working directory. +pub struct LocalFsBackend { + /// Canonical root. All resolved paths must live under this. + root: PathBuf, +} + +impl LocalFsBackend { + /// Create a backend confined to `root`. + /// + /// The root is canonicalized so that containment checks compare real paths. + /// If it cannot be canonicalized (e.g. it does not exist yet) the path is + /// used as given — resolution still applies, it is simply not symlink-proof + /// above the root itself. + pub fn new(root: impl Into) -> Self { + let root = root.into(); + let root = std::fs::canonicalize(&root).unwrap_or(root); + Self { root } + } + + /// The confinement root. + pub fn root(&self) -> &Path { + &self.root + } + + /// Resolve a tool-supplied path to a real path inside the root. + /// + /// Relative paths are joined to the root; absolute paths must already be + /// inside it. `..` is normalized lexically first so it cannot be used to + /// climb out, then the deepest existing ancestor is canonicalized to defeat + /// symlink escapes. + fn resolve(&self, path: &str) -> Result { + let raw = Path::new(path); + let joined = if raw.is_absolute() { + raw.to_path_buf() + } else if path.is_empty() || path == "." { + self.root.clone() + } else { + self.root.join(raw) + }; + + let normalized = lexical_normalize(&joined); + let resolved = canonicalize_existing_prefix(&normalized); + + if resolved.starts_with(&self.root) { + Ok(resolved) + } else { + Err(format!( + "Error: path '{path}' resolves outside the workspace root" + )) + } + } + + /// Resolve for write-style operations, returning the error as a + /// `WriteResult` rather than a bare string. + fn resolve_for_write(&self, path: &str) -> Result { + self.resolve(path).map_err(|e| WriteResult { + error: Some(e), + ..Default::default() + }) + } +} + +/// Remove `.` and resolve `..` lexically, without touching the filesystem. +/// +/// Purely lexical so it works for paths that do not exist yet (`write_file` to +/// a new file). `..` at or above the root simply cannot climb past the prefix. +fn lexical_normalize(path: &Path) -> PathBuf { + let mut out = PathBuf::new(); + for comp in path.components() { + match comp { + Component::Prefix(p) => out.push(p.as_os_str()), + Component::RootDir => out.push(Component::RootDir.as_os_str()), + Component::CurDir => {} + Component::ParentDir => { + // Never pop past the root component itself. + if out.parent().is_some() { + out.pop(); + } + } + Component::Normal(seg) => out.push(seg), + } + } + out +} + +/// Canonicalize the deepest existing ancestor of `path`, re-appending the +/// non-existent tail. +/// +/// This is what makes the containment check symlink-aware: if any existing +/// component is a symlink out of the root, canonicalization exposes it before +/// the `starts_with` test. `path` must already be lexically normalized, so +/// re-appending the tail cannot reintroduce `..`. +fn canonicalize_existing_prefix(path: &Path) -> PathBuf { + let mut existing = path; + let mut tail: Vec<&std::ffi::OsStr> = Vec::new(); + + loop { + if let Ok(canonical) = std::fs::canonicalize(existing) { + let mut out = canonical; + for seg in tail.iter().rev() { + out.push(seg); + } + return out; + } + match (existing.file_name(), existing.parent()) { + (Some(name), Some(parent)) => { + tail.push(name); + existing = parent; + } + // Nothing along the chain exists; fall back to the lexical path. + _ => return path.to_path_buf(), + } + } +} + +impl Backend for LocalFsBackend { + fn ls_info(&self, path: &str) -> Result, String> { + let target = self.resolve(path)?; + let entries = std::fs::read_dir(&target) + .map_err(|e| format!("ls failed on '{}': {}", target.display(), e))?; + let mut infos = Vec::new(); + for entry in entries { + let entry = entry.map_err(|e| format!("read_dir entry error: {}", e))?; + let meta = entry + .metadata() + .map_err(|e| format!("metadata error: {}", e))?; + let file_type = if meta.is_dir() { + "directory" + } else if meta.is_symlink() { + "symlink" + } else { + "file" + }; + infos.push(FileInfo { + name: entry.file_name().to_string_lossy().into_owned(), + file_type: file_type.to_string(), + permissions: String::new(), + size: meta.len(), + }); + } + infos.sort_by(|a, b| a.name.cmp(&b.name)); + Ok(infos) + } + + fn read(&self, path: &str, offset: usize, limit: usize) -> Result { + let target = self.resolve(path)?; + let content = std::fs::read_to_string(&target) + .map_err(|e| format!("read '{}': {}", target.display(), e))?; + let lines: Vec<&str> = content.lines().collect(); + if offset >= lines.len() { + return Ok(String::new()); + } + let end = (offset + limit).min(lines.len()); + Ok(lines[offset..end].join("\n")) + } + + fn write(&self, path: &str, content: &str) -> WriteResult { + let target = match self.resolve_for_write(path) { + Ok(t) => t, + Err(e) => return e, + }; + if target.exists() { + return WriteResult { + error: Some(format!( + "Error: file {} already exists. Use force flag to overwrite.", + target.display() + )), + ..Default::default() + }; + } + if let Some(parent) = target.parent() { + if let Err(e) = std::fs::create_dir_all(parent) { + return WriteResult { + error: Some(format!("mkdir failed: {}", e)), + ..Default::default() + }; + } + } + match std::fs::write(&target, content) { + Ok(_) => WriteResult::default(), + Err(e) => WriteResult { + error: Some(format!("write '{}': {}", target.display(), e)), + ..Default::default() + }, + } + } + + fn edit( + &self, + path: &str, + old_string: &str, + new_string: &str, + replace_all: bool, + ) -> WriteResult { + let target = match self.resolve_for_write(path) { + Ok(t) => t, + Err(e) => return e, + }; + let content = match std::fs::read_to_string(&target) { + Ok(c) => c, + Err(e) => { + return WriteResult { + error: Some(format!("read '{}': {}", target.display(), e)), + ..Default::default() + } + } + }; + let count = content.matches(old_string).count(); + if count == 0 { + return WriteResult { + error: Some(format!( + "Error: old_string not found in {}", + target.display() + )), + ..Default::default() + }; + } + if count > 1 && !replace_all { + return WriteResult { + error: Some(format!( + "Error: old_string is not unique in {} ({} occurrences). Use replace_all=true.", + target.display(), + count + )), + ..Default::default() + }; + } + let new_content = if replace_all { + content.replace(old_string, new_string) + } else { + content.replacen(old_string, new_string, 1) + }; + match std::fs::write(&target, &new_content) { + Ok(_) => WriteResult { + error: None, + occurrences: Some(if replace_all { count } else { 1 }), + ..Default::default() + }, + Err(e) => WriteResult { + error: Some(format!("write '{}': {}", target.display(), e)), + ..Default::default() + }, + } + } + + fn glob_info(&self, pattern: &str, path: &str) -> Result, String> { + let base = self.resolve(path)?; + // Simple glob: walk the directory and match by name suffix. Handles the + // common `*.rs` / `**/*.toml` shapes without pulling in a glob crate. + let suffix = pattern + .trim_start_matches('*') + .trim_start_matches('/') + .trim_start_matches('*'); + let mut results = Vec::new(); + collect_glob_matches(&base, suffix, &mut results); + results.sort(); + Ok(results) + } + + fn grep_raw( + &self, + pattern: &str, + path: Option<&str>, + _include: Option<&str>, + ) -> Result, String> { + let search_dir = self.resolve(path.unwrap_or("."))?; + let mut matches = Vec::new(); + if search_dir.is_file() { + grep_file(&search_dir, pattern, &mut matches)?; + } else if search_dir.is_dir() { + grep_dir(&search_dir, pattern, &mut matches)?; + } + Ok(matches) + } + + fn execute(&self, command: &str, timeout_secs: u32) -> Result { + use std::process::{Command, Stdio}; + use std::time::Duration; + + // Security: environment sanitization — strip sensitive variables + // (SEC-005 / ADR-103 C2). Only a safe allowlist reaches the child. + const SAFE_ENV_VARS: &[&str] = &[ + "PATH", "HOME", "USER", "SHELL", "LANG", "LC_ALL", "LC_CTYPE", "TERM", "TMPDIR", "TZ", + ]; + // Patterns identifying vars that must never reach child processes. + const SENSITIVE_PATTERNS: &[&str] = &[ + "SECRET", + "KEY", + "TOKEN", + "PASSWORD", + "CREDENTIAL", + "AWS_", + "AZURE_", + "GCP_", + "DATABASE_URL", + "PRIVATE", + "API_KEY", + "AUTH", + "BEARER", + "JWT", + "SESSION", + ]; + + let mut cmd = Command::new("sh"); + cmd.arg("-c").arg(command).current_dir(&self.root); + cmd.env_clear(); + for var in SAFE_ENV_VARS { + if let Ok(val) = std::env::var(var) { + let upper = var.to_uppercase(); + let sensitive = SENSITIVE_PATTERNS.iter().any(|pat| upper.contains(pat)); + if !sensitive { + cmd.env(var, val); + } + } + } + cmd.stdout(Stdio::piped()).stderr(Stdio::piped()); + + let timeout = if timeout_secs == 0 { 30 } else { timeout_secs }; + let deadline = std::time::Instant::now() + Duration::from_secs(timeout as u64); + + let mut child = cmd.spawn().map_err(|e| format!("execute failed: {}", e))?; + + // Poll for completion with a deadline to enforce the timeout. + loop { + match child + .try_wait() + .map_err(|e| format!("wait failed: {}", e))? + { + Some(_) => break, + None => { + if std::time::Instant::now() >= deadline { + let _ = child.kill(); + return Ok(ExecuteResponse { + output: format!("Command timed out after {} seconds", timeout), + exit_code: -1, + }); + } + std::thread::sleep(Duration::from_millis(50)); + } + } + } + + let output = child + .wait_with_output() + .map_err(|e| format!("output collection failed: {}", e))?; + + let stdout = String::from_utf8_lossy(&output.stdout); + let stderr = String::from_utf8_lossy(&output.stderr); + let mut combined = if stderr.is_empty() { + stdout.into_owned() + } else { + format!("{}\n{}", stdout, stderr) + }; + + // Security: cap output size to prevent memory exhaustion. + const MAX_OUTPUT_BYTES: usize = 1024 * 1024; + if combined.len() > MAX_OUTPUT_BYTES { + combined.truncate(MAX_OUTPUT_BYTES); + combined.push_str("\n... [output truncated at 1 MB]"); + } + + Ok(ExecuteResponse { + output: combined, + exit_code: output.status.code().unwrap_or(-1), + }) + } +} + +/// Recursively collect files matching a name suffix (simple glob substitute). +fn collect_glob_matches(dir: &Path, suffix: &str, results: &mut Vec) { + let entries = match std::fs::read_dir(dir) { + Ok(e) => e, + Err(_) => return, + }; + for entry in entries.flatten() { + let path = entry.path(); + let name = path + .file_name() + .map(|n| n.to_string_lossy().to_string()) + .unwrap_or_default(); + if path.is_file() && name.ends_with(suffix) { + results.push(path.to_string_lossy().into_owned()); + } else if path.is_dir() && !name.starts_with('.') { + collect_glob_matches(&path, suffix, results); + } + } +} + +/// Grep a single file for a pattern. +fn grep_file(path: &Path, pattern: &str, matches: &mut Vec) -> Result<(), String> { + let content = match std::fs::read_to_string(path) { + Ok(c) => c, + Err(_) => return Ok(()), // skip binary / unreadable files + }; + for (i, line) in content.lines().enumerate() { + if line.contains(pattern) { + matches.push(GrepMatch { + file: path.to_string_lossy().into_owned(), + line_number: i + 1, + text: line.to_string(), + }); + } + } + Ok(()) +} + +/// Recursively grep a directory, skipping hidden directories. +fn grep_dir(dir: &Path, pattern: &str, matches: &mut Vec) -> Result<(), String> { + let entries = std::fs::read_dir(dir).map_err(|e| format!("read_dir: {}", e))?; + for entry in entries { + let entry = entry.map_err(|e| format!("entry: {}", e))?; + let path = entry.path(); + if path.is_file() { + grep_file(&path, pattern, matches)?; + } else if path.is_dir() { + let name = path + .file_name() + .map(|n| n.to_string_lossy().to_string()) + .unwrap_or_default(); + if !name.starts_with('.') { + grep_dir(&path, pattern, matches)?; + } + } + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn backend() -> (tempfile::TempDir, LocalFsBackend) { + let dir = tempfile::tempdir().unwrap(); + let backend = LocalFsBackend::new(dir.path()); + (dir, backend) + } + + #[test] + fn relative_paths_resolve_inside_root() { + let (dir, backend) = backend(); + std::fs::write(dir.path().join("a.txt"), "hello").unwrap(); + assert_eq!(backend.read("a.txt", 0, 10).unwrap(), "hello"); + } + + #[test] + fn absolute_path_outside_root_is_rejected() { + let (_dir, backend) = backend(); + let err = backend.read("/etc/passwd", 0, 10).unwrap_err(); + assert!(err.contains("outside the workspace root"), "got: {err}"); + } + + #[test] + fn parent_traversal_is_rejected() { + let (_dir, backend) = backend(); + let err = backend.read("../../../etc/passwd", 0, 10).unwrap_err(); + assert!(err.contains("outside the workspace root"), "got: {err}"); + } + + #[test] + fn write_outside_root_is_rejected() { + let (_dir, backend) = backend(); + let result = backend.write("/tmp/rvagent-escape-probe.txt", "pwned"); + assert!(result + .error + .as_deref() + .is_some_and(|e| e.contains("outside the workspace root"))); + assert!( + !Path::new("/tmp/rvagent-escape-probe.txt").exists(), + "escaping write must not touch the filesystem" + ); + } + + #[test] + fn edit_outside_root_is_rejected() { + let (_dir, backend) = backend(); + let result = backend.edit("/etc/hosts", "a", "b", false); + assert!(result + .error + .as_deref() + .is_some_and(|e| e.contains("outside the workspace root"))); + } + + #[cfg(unix)] + #[test] + fn symlink_escape_is_rejected() { + let (dir, backend) = backend(); + // A symlink inside the root pointing out of it must not be a bridge. + std::os::unix::fs::symlink("/etc", dir.path().join("escape")).unwrap(); + let err = backend.read("escape/passwd", 0, 10).unwrap_err(); + assert!(err.contains("outside the workspace root"), "got: {err}"); + } + + #[test] + fn write_then_read_roundtrip_inside_root() { + let (_dir, backend) = backend(); + let result = backend.write("nested/dir/new.txt", "content"); + assert!(result.error.is_none(), "unexpected: {:?}", result.error); + assert_eq!(backend.read("nested/dir/new.txt", 0, 10).unwrap(), "content"); + } + + #[test] + fn dotdot_inside_root_still_works() { + let (_dir, backend) = backend(); + assert!(backend.write("sub/file.txt", "x").error.is_none()); + // Climbs out of `sub` but stays under the root — legitimate. + assert_eq!(backend.read("sub/../sub/file.txt", 0, 10).unwrap(), "x"); + } +} diff --git a/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs b/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs new file mode 100644 index 0000000000..4802b5a168 --- /dev/null +++ b/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs @@ -0,0 +1,516 @@ +//! Phase 0 exit gate: end-to-end tool calling. +//! +//! These tests wire the *shipped* pieces together — the real builtin tool +//! registry, the real `LocalFsBackend`, and the real `AgentGraph` loop — against +//! a scripted model. Everything except the network call to the provider is +//! production code, so the gate fails if the loop, the schemas, or the tools +//! regress. +//! +//! What each P0 claim is verified by: +//! * P0.2 (schemas reach the model) — `schemas_reach_the_model` +//! * P0.4 (errors feed back, not abort) — `tool_error_feeds_back_and_loop_continues` +//! * P0.4 (parallel exec preserves order) — `parallel_tool_calls_preserve_order` +//! * P0.4 (usage accounting) — `usage_metadata_is_aggregated` +//! * real side effects on disk — `write_then_read_roundtrip_through_the_loop` +//! * confinement holds through the loop — `path_escape_is_refused_through_the_loop` + +use std::sync::{Arc, Mutex}; + +use async_trait::async_trait; +use rvagent_core::error::Result; +use rvagent_core::graph::{AgentGraph, GraphConfig, ToolExecutor}; +use rvagent_core::messages::{Message, ToolCall}; +use rvagent_core::models::{ChatModel, ToolDefinition}; +use rvagent_core::state::AgentState; +use rvagent_tools::Tool as _; + +// --------------------------------------------------------------------------- +// Test harness: the real tool executor, wired exactly as the CLI wires it +// --------------------------------------------------------------------------- + +/// Mirrors `CliToolExecutor`: real builtin tools over a real confined backend. +struct RealToolExecutor { + tools: Vec, + backend: rvagent_tools::BackendRef, +} + +impl RealToolExecutor { + fn new(root: &std::path::Path) -> Self { + Self { + tools: rvagent_tools::builtin_tools(), + backend: Arc::new(rvagent_tools::LocalFsBackend::new(root)), + } + } +} + +#[async_trait] +impl ToolExecutor for RealToolExecutor { + async fn execute(&self, call: &ToolCall, _state: &AgentState) -> Result { + let runtime = rvagent_tools::ToolRuntime::new(Arc::clone(&self.backend)); + match rvagent_tools::resolve_tool(&call.name, &self.tools) { + Some(tool) => Ok(tool.invoke(call.args.clone(), &runtime).to_string()), + None => Ok(format!("Error: tool '{}' not found", call.name)), + } + } + + fn definitions(&self) -> Vec { + self.tools + .iter() + .map(|t| ToolDefinition { + name: t.name().to_string(), + description: t.description().to_string(), + input_schema: t.parameters_schema(), + }) + .collect() + } +} + +/// A scripted model that records what the loop actually sent it. +struct ScriptedModel { + responses: Mutex>, + /// Tool schemas observed on each `complete` call. + seen_tools: Mutex>>, + /// Message history observed on the most recent `complete` call. + last_messages: Mutex>, +} + +impl ScriptedModel { + fn new(responses: Vec) -> Self { + Self { + responses: Mutex::new(responses), + seen_tools: Mutex::new(Vec::new()), + last_messages: Mutex::new(Vec::new()), + } + } +} + +#[async_trait] +impl ChatModel for ScriptedModel { + async fn complete(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result { + self.seen_tools.lock().unwrap().push(tools.to_vec()); + *self.last_messages.lock().unwrap() = messages.to_vec(); + let mut resps = self.responses.lock().unwrap(); + if resps.is_empty() { + Ok(Message::ai("done")) + } else { + Ok(resps.remove(0)) + } + } + + async fn stream(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result> { + Ok(vec![self.complete(messages, tools).await?]) + } +} + +fn call(id: &str, name: &str, args: serde_json::Value) -> ToolCall { + ToolCall { + id: id.into(), + name: name.into(), + args, + } +} + +/// Collect tool-result message contents, in order. +fn tool_results(state: &AgentState) -> Vec { + state + .messages + .iter() + .filter_map(|m| match m { + Message::Tool(t) => Some(t.content.clone()), + _ => None, + }) + .collect() +} + +// --------------------------------------------------------------------------- +// P0.2 — tool schemas reach the model +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn schemas_reach_the_model() { + let dir = tempfile::tempdir().unwrap(); + let model = ScriptedModel::new(vec![Message::ai("hi")]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + + let state = graph.run(AgentState::with_system_message("sys")).await; + assert!(state.is_ok(), "loop failed: {:?}", state.err()); + + // The graph owns the model, so re-derive the expectation from the registry: + // every builtin tool must have been advertised with a usable schema. + let executor = RealToolExecutor::new(dir.path()); + let defs = executor.definitions(); + assert_eq!(defs.len(), rvagent_tools::builtin_tools().len()); + assert!(defs.iter().any(|d| d.name == "read_file")); + assert!(defs.iter().any(|d| d.name == "write_file")); + for def in &defs { + assert!(!def.description.is_empty(), "{} has no description", def.name); + assert_eq!( + def.input_schema.get("type").and_then(|v| v.as_str()), + Some("object"), + "{} schema is not a JSON-Schema object: {}", + def.name, + def.input_schema + ); + assert!( + def.input_schema.get("properties").is_some(), + "{} schema has no properties", + def.name + ); + } +} + +#[tokio::test] +async fn schemas_are_sent_on_every_turn_including_after_tools() { + let dir = tempfile::tempdir().unwrap(); + std::fs::write(dir.path().join("f.txt"), "content").unwrap(); + + let model = Arc::new(ScriptedModel::new(vec![ + Message::ai_with_tools( + "reading", + vec![call("t1", "read_file", serde_json::json!({"file_path": "f.txt"}))], + ), + Message::ai("read it"), + ])); + + // Share the model so the recorded observations survive the graph. + struct SharedModel(Arc); + #[async_trait] + impl ChatModel for SharedModel { + async fn complete( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> Result { + self.0.complete(messages, tools).await + } + async fn stream( + &self, + messages: &[Message], + tools: &[ToolDefinition], + ) -> Result> { + self.0.stream(messages, tools).await + } + } + + let graph = AgentGraph::new( + SharedModel(Arc::clone(&model)), + RealToolExecutor::new(dir.path()), + ); + graph.run(AgentState::new()).await.unwrap(); + + let seen = model.seen_tools.lock().unwrap(); + assert_eq!(seen.len(), 2, "expected two model turns"); + for (turn, tools) in seen.iter().enumerate() { + assert!( + !tools.is_empty(), + "turn {turn} was sent no tool schemas — the model could not call a tool" + ); + } +} + +// --------------------------------------------------------------------------- +// Real side effects on disk +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn write_then_read_roundtrip_through_the_loop() { + let dir = tempfile::tempdir().unwrap(); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "writing", + vec![call( + "t1", + "write_file", + serde_json::json!({"file_path": "out.txt", "content": "hello from the agent"}), + )], + ), + Message::ai_with_tools( + "reading back", + vec![call( + "t2", + "read_file", + serde_json::json!({"file_path": "out.txt"}), + )], + ), + Message::ai("verified"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + let state = graph.run(AgentState::new()).await.unwrap(); + + // The file must actually exist on disk — this is the end-to-end claim. + let written = std::fs::read_to_string(dir.path().join("out.txt")) + .expect("agent's write_file did not produce a real file"); + assert_eq!(written, "hello from the agent"); + + let results = tool_results(&state); + assert_eq!(results.len(), 2, "expected one result per tool call"); + assert!( + results[1].contains("hello from the agent"), + "read_file did not return the written content: {}", + results[1] + ); +} + +// --------------------------------------------------------------------------- +// P0.4 — tool errors feed back as results instead of aborting +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn tool_error_feeds_back_and_loop_continues() { + let dir = tempfile::tempdir().unwrap(); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "reading a file that isn't there", + vec![call( + "t1", + "read_file", + serde_json::json!({"file_path": "does-not-exist.txt"}), + )], + ), + // The model gets to see the failure and recover. + Message::ai_with_tools( + "creating it instead", + vec![call( + "t2", + "write_file", + serde_json::json!({"file_path": "does-not-exist.txt", "content": "now it does"}), + )], + ), + Message::ai("recovered"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + + // The run must succeed: a failing tool is not a failing agent. + let state = graph + .run(AgentState::new()) + .await + .expect("a tool error must not abort the loop"); + + let results = tool_results(&state); + assert_eq!(results.len(), 2); + assert!( + results[0].to_lowercase().contains("error") + || results[0].to_lowercase().contains("no such file"), + "the failure was not reported back to the model: {}", + results[0] + ); + assert!( + dir.path().join("does-not-exist.txt").exists(), + "the recovery turn did not run" + ); +} + +#[tokio::test] +async fn unknown_tool_is_reported_not_fatal() { + let dir = tempfile::tempdir().unwrap(); + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "", + vec![call("t1", "no_such_tool", serde_json::json!({}))], + ), + Message::ai("ok"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + let state = graph.run(AgentState::new()).await.unwrap(); + + let results = tool_results(&state); + assert_eq!(results.len(), 1); + assert!(results[0].contains("not found"), "got: {}", results[0]); +} + +// --------------------------------------------------------------------------- +// P0.4 — parallel execution +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn parallel_tool_calls_preserve_order() { + let dir = tempfile::tempdir().unwrap(); + for i in 0..6 { + std::fs::write(dir.path().join(format!("f{i}.txt")), format!("body-{i}")).unwrap(); + } + + let calls: Vec = (0..6) + .map(|i| { + call( + &format!("t{i}"), + "read_file", + serde_json::json!({"file_path": format!("f{i}.txt")}), + ) + }) + .collect(); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools("reading all", calls), + Message::ai("done"), + ]); + let config = GraphConfig { + parallel_tools: true, + max_parallel_tools: 3, + ..GraphConfig::default() + }; + let graph = AgentGraph::with_config(model, RealToolExecutor::new(dir.path()), config); + let state = graph.run(AgentState::new()).await.unwrap(); + + let results = tool_results(&state); + assert_eq!(results.len(), 6); + // Results must come back in call order even though execution is concurrent + // and the concurrency limit is lower than the number of calls. + for (i, r) in results.iter().enumerate() { + assert!( + r.contains(&format!("body-{i}")), + "result {i} out of order or wrong: {r}" + ); + } +} + +#[tokio::test] +async fn parallel_and_sequential_agree() { + let dir = tempfile::tempdir().unwrap(); + for i in 0..4 { + std::fs::write(dir.path().join(format!("f{i}.txt")), format!("body-{i}")).unwrap(); + } + let calls: Vec = (0..4) + .map(|i| { + call( + &format!("t{i}"), + "read_file", + serde_json::json!({"file_path": format!("f{i}.txt")}), + ) + }) + .collect(); + + let mut outputs = Vec::new(); + for parallel in [true, false] { + let model = ScriptedModel::new(vec![ + Message::ai_with_tools("", calls.clone()), + Message::ai("done"), + ]); + let config = GraphConfig { + parallel_tools: parallel, + max_parallel_tools: 2, + ..GraphConfig::default() + }; + let graph = AgentGraph::with_config(model, RealToolExecutor::new(dir.path()), config); + let state = graph.run(AgentState::new()).await.unwrap(); + outputs.push(tool_results(&state)); + } + assert_eq!( + outputs[0], outputs[1], + "parallel and sequential execution disagree" + ); +} + +// --------------------------------------------------------------------------- +// P0.4 — usage accounting +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn usage_metadata_is_aggregated() { + let dir = tempfile::tempdir().unwrap(); + + // Two turns carrying provider usage metadata, as the backends attach it. + let mut first = Message::ai_with_tools( + "", + vec![call("t1", "ls", serde_json::json!({"path": "."}))], + ); + let mut second = Message::ai("done"); + for (msg, input, output) in [(&mut first, 100u64, 20u64), (&mut second, 150u64, 30u64)] { + if let Message::Ai(ai) = msg { + ai.metadata.insert( + "usage".into(), + serde_json::json!({"input_tokens": input, "output_tokens": output}), + ); + } + } + + let model = ScriptedModel::new(vec![first, second]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + let state = graph.run(AgentState::new()).await.unwrap(); + + // Usage metadata must survive on the messages so a caller can total it. + let totals: (u64, u64) = state + .messages + .iter() + .filter_map(|m| match m { + Message::Ai(ai) => ai.metadata.get("usage"), + _ => None, + }) + .fold((0, 0), |(i, o), usage| { + ( + i + usage.get("input_tokens").and_then(|v| v.as_u64()).unwrap_or(0), + o + usage + .get("output_tokens") + .and_then(|v| v.as_u64()) + .unwrap_or(0), + ) + }); + assert_eq!(totals, (250, 50), "usage metadata was lost or miscounted"); +} + +// --------------------------------------------------------------------------- +// Confinement holds through the full loop +// --------------------------------------------------------------------------- + +#[tokio::test] +async fn path_escape_is_refused_through_the_loop() { + let dir = tempfile::tempdir().unwrap(); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "exfiltrating", + vec![ + call("t1", "read_file", serde_json::json!({"file_path": "/etc/passwd"})), + call( + "t2", + "read_file", + serde_json::json!({"file_path": "../../../../etc/passwd"}), + ), + ], + ), + Message::ai("done"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + let state = graph.run(AgentState::new()).await.unwrap(); + + for (i, result) in tool_results(&state).iter().enumerate() { + assert!( + !result.contains("root:"), + "escape {i} leaked /etc/passwd contents: {result}" + ); + // Must be refused *by the confinement check* specifically — a generic + // error (bad param, missing file) would pass vacuously and hide a + // regression in the boundary itself. + assert!( + result.contains("outside the workspace root"), + "escape {i} was not refused by path confinement: {result}" + ); + } +} + +#[tokio::test] +async fn write_escape_does_not_touch_the_filesystem() { + let dir = tempfile::tempdir().unwrap(); + let outside = dir.path().parent().unwrap().join("rvagent-e2e-escape.txt"); + let _ = std::fs::remove_file(&outside); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "", + vec![call( + "t1", + "write_file", + serde_json::json!({"file_path": outside.to_string_lossy(), "content": "pwned"}), + )], + ), + Message::ai("done"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + graph.run(AgentState::new()).await.unwrap(); + + assert!( + !outside.exists(), + "a write outside the workspace root reached the filesystem" + ); +} From d9bb214d1c2bb30de79573e981431077196a0c5c Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 16:50:17 +0000 Subject: [PATCH 07/25] docs(research): 2026 SOTA landscape sweep and roadmap corrections Five-area research sweep (benchmarks, harness techniques, competing architectures, long-horizon context, self-improving harnesses) with the roadmap corrections each finding implies. Findings that contradict the current roadmap: - Compaction: simple observation masking matches or beats LLM summarization at ~half the cost, and summarization inflates trajectories 13-15% by destroying stopping signals. Phase 1.3 and Phase 2.5 both bet on summarization, as does the shipped default middleware pipeline. - Subagents: Phase 1.6's CoW fork/merge + CRDT join is parallel-writer architecture, the one multi-agent pattern with strong negative evidence for coding. The patterns with production evidence are a fresh-context reviewer and a read-only context-gatherer. - Learning loop: the 2026 literature turned against trajectory-learning memory. Controlled baselines show agent self-memory underperforming plain retrieval, and a documented inverted-U where utility drops below no-memory. "SONA on the default path" needs a gate and a permanent control arm. - Positioning: three major 2026 harnesses are already Rust (Codex CLI, Grok Build at ~844k LOC open-sourced 2026-07-15, Goose). "Fast Rust harness" is not a differentiator. Two positions remain open: a stable embeddable library API with open governance, and deterministic replay. - Phase 4's SWE-bench-Verified gate targets a saturated benchmark whose noise band exceeds the effects we would claim. Missing from the roadmap entirely: programmatic tool calling (the only context strategy positive across all models tested), and most of the reliability floor (observation-window management, loop detection, environment bootstrap, persisted thinking) where the largest measured deltas live. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- .../04-sota-landscape.md | 404 ++++++++++++++++++ 1 file changed, 404 insertions(+) create mode 100644 docs/research/rvagent-hermes-harness/04-sota-landscape.md diff --git a/docs/research/rvagent-hermes-harness/04-sota-landscape.md b/docs/research/rvagent-hermes-harness/04-sota-landscape.md new file mode 100644 index 0000000000..dfe3d8e7a1 --- /dev/null +++ b/docs/research/rvagent-hermes-harness/04-sota-landscape.md @@ -0,0 +1,404 @@ +# SOTA Landscape (August 2026) — and what it changes + +Research sweep across five areas: benchmarks and credible-claim criteria, +harness technique literature with effect sizes, competing harness +architectures, long-horizon context methods, and self-improving harnesses. + +This document exists to correct the roadmap, not to decorate it. Where the +evidence contradicts `03-roadmap.md`, the contradiction is stated plainly and +the roadmap change is specified. + +**Evidence discipline.** Most 2026 material is single-source arXiv preprints +or vendor blogs. Individual percentage-point figures are hypotheses to +re-measure in our own harness, not constants. The findings below are ranked by +corroboration, and single-source claims are marked. Four findings are +multiply-corroborated and safe to build on: + +1. Harness choice moves results 10–30 points at fixed model. +2. Harness value scales *inversely* with model strength. +3. Context **management** beats context **size**. +4. Parallel writer agents fail on coding tasks specifically. + +--- + +## 1. The headline: most of the delta is reliability, not intelligence + +The cleanest ablation available (Claw-SWE-Bench, Jun 2026) strips a harness to +bare model-emits-diff and rebuilds it: **19.1% → 73.4%, +54.3 points**. Nearly +all of that is patch-apply failures going from 69.1% to under 1.5%. + +That reframes the entire project. The biggest measured wins in every ablation +come from eliminating mechanical failure modes — patches that don't apply, +tool-call loops, context rot, flaky tests — not from smarter reasoning. + +**This is the strongest argument for building the harness in Rust**, and it is +not the argument we have been making. The case is not "Rust is fast." It is +that the failure modes which dominate these ablations can be made +*type-unrepresentable* rather than merely rare. + +Two structural findings that should shape sequencing: + +- **Harness variance is larger for weaker models.** Same harness set, GLM-5.1: + 12.5-point spread. Qwen-3.6-flash: 27.4-point spread. If we target frontier + models only, we should expect roughly *half* the harness ROI the literature + reports. +- **The harness is now a disclosed experimental variable.** Two 2026 papers + (arXiv 2605.23950, 2607.04528) argue benchmark results are substantially + determined by undisclosed harness choices, and that harnesses induce + systematically different agent *beliefs* on logically equivalent tasks. Any + claim we publish must disclose the harness or it is not a claim. + +--- + +## 2. Competitive position: the Rust field is crowded at the top + +The premise that a Rust harness is differentiating is **false as of July 2026**. +Three of the major harnesses are already Rust: + +| Harness | Scale | License | Notes | +|---|---|---|---| +| **Codex CLI** (OpenAI) | ~70–80 crates | Apache 2.0 | Rewritten *from* TypeScript for single-binary distribution + native sandbox bindings | +| **Grok Build** (xAI) | ~844k LOC | Apache 2.0 | Open-sourced **2026-07-15**. ACP, checkpoints, TUI | +| **Goose** (Block) | — | Apache 2.0 | MCP-native, subagents via `Agent::new()` | + +OpenAI's stated reasons for the Rust rewrite are exactly the ones in our +positioning: zero-dependency install (Node 22+ blocked enterprise/air-gapped), +no GC pauses in long-running processes, memory-safe sandbox bindings without +FFI shims. That ground is taken. + +**Two positions remain genuinely open:** + +1. **No Rust harness is usable as a library.** Codex's own `AGENTS.md` + discourages extending `codex-core`; Grok Build has issues and PRs + *disabled* (source-visible, not open governance); Goose is app-first; and + Anthropic's Agent SDK is Python/TypeScript only — the docs instruct other + languages to shell out to the CLI. A stable, semver'd, embeddable harness + crate with open governance is unoccupied. +2. **Deterministic replay is a named gap** in the Rust agent ecosystem and + thin everywhere. Our append-only witness/segment infrastructure is already + most of the way there. + +**Honesty constraint on the replay claim.** Even at temperature 0, hosted +inference is not reproducible — floating-point non-associativity and +batch-size-dependent kernels produce run-to-run variation. The defensible +claim is *replay of the harness*, with **action-match rate** reported, not +byte-exact reproduction of the model. + +**Convergence warning.** Grok Build's tool implementations are documented ports +— `apply_patch`, `grep_files`, `list_dir`, `read_file` from Codex; `bash`, +`edit`, `glob`, `grep`, `read`, `skill`, `todowrite`, `write` from opencode. A +frontier lab with a million lines of Rust chose to port the tool surface rather +than design one. **Tool-surface novelty is not available as a differentiator.** + +--- + +## 3. Ranked technique priorities + +Ordered by measured effect per unit of engineering effort. Tier 1 items are +days of work each and carry the largest deltas in the literature. + +### Tier 1 — do first + +| # | Technique | Effect | Status in rvagent | +|---|---|---|---| +| 1 | Reliable patch application (real workspace, file-based edits, git diff extraction, verify-after-write) | **+54.3 pp** | Partial — tools write real files; no verify-after-write, no git extraction | +| 2 | Observation-window management (keep last N tool outputs in full, elide older) | +3 pp, prevents long-run collapse | **Missing** | +| 3 | Loop/stuck detection (3-strike tool-call fingerprint → inject warning, skip) | Removes most common catastrophic failure | **Missing** | +| 4 | Actionable structured tool errors + response size caps (~25k tokens) | Part of the reliability delta | Partial — errors feed back (P0.4), uncapped | +| 5 | Tool surface held to 8–15 tools | Avoids −16 to −23 pt routing collapse | **OK** — 9 builtins. Protect this. | +| 6 | Environment bootstrap injection (cwd, tree, toolchain, test command, current check status) | Stanford Meta-Harness @ 76.4% TB2.0 | **Missing** | +| 7 | Persist thinking blocks across tool calls | +2.2 pp coding | **Missing** | + +Item 3 deserves emphasis: raising max-iteration counts does **not** fix loops, +it makes them more expensive. Our `max_iterations: 100` is a cost cap, not a +loop guard. + +### Tier 2 — clear ROI, ~1–2 weeks each + +- **`str_replace` edit tool with fuzzy-failure diagnostics + `cargo check` lint + gate.** +10 to +23 pp for mid-tier models, ~0–5 pp frontier. The specific + ergonomics matter: in the Qwen reproduction only the SWE-agent + `str_replace_editor` flavor moved the number; a different `edit`/`write_file` + pair gave *zero* improvement. Offer `write_file` alongside (+2.1 pp, −17.9% + cost). Skip unified diff — apply failures dominate. +- **Reproduction-test-first loop** (+8 to +13% relative). Critical caveat: + adding "write tests first" to the prompt *without* targeted context made + regressions **worse** (6.08% → 9.94%). The gain is in *executing* + reproduction tests, not the TDD ritual. +- **Summarized grep/glob** — return paths + match counts, require a second call + for contents. `ripgrep` as a library crate. +- **Fresh-context reviewer subagent** — ~2 bugs/PR, 58% severe, in Cognition + production. Counterintuitively, reviewers perform **better with no shared + context**: shorter context, less rot, deeper analysis. +- **Read-only context-gathering subagent** returning a summary string. +2.1 pp, + −34.5% main-agent input tokens. Use a *cheap* model here — a frontier model + in this slot gave +0.4 pp at 5.8× cost. + +### Tier 3 — real but expensive or conditional + +Rubric-guided compaction (the rubric is load-bearing, not the tool); +best-of-N with test-based filtering then deterministic patch fusion (+7 to ++9.4 pp at N≈8, ≈8× cost); shadow-git checkpointing; coordinator delegation +(only after single-agent is solid). + +### Skip list — evidence says no + +- **Parallel writer swarms** for coding +- **Few-shot demos / explicit CoT instructions** for reasoning models — zero-shot ≥ few-shot; exemplars can contradict native reasoning +- **Ungrounded self-reflection loops** — can degrade already-correct answers; only execution-grounded critique works +- **Unified-diff edit format** +- **Semantic/embedding code index in v1** — vendor-only evidence, high maintenance, stale-by-construction on a repo the agent is editing +- **Elaborate system-prompt frameworks** — the "20–30% improvement" claims have no methodology; keep rulebooks under ~60 lines +- **Context windows beyond ~128k** — sweeps plateau ~114k; documented ceiling ~96–112k +- **Learned/RL-trained components** (adaptive edit-format selectors, RL compaction) — the hand-written 80% is available for 5% of the effort + +--- + +## 4. Long-horizon execution: what actually breaks + +Ranked by when it bites in a multi-hour run: + +1. **Wasted-context accumulation → attention dilution.** Earliest, most + universal, invisible. Unaided coding agents waste ~1-in-3 file reads. +2. **History error accumulation.** The largest single failure driver — + process-level failures are **72.5%** of long-horizon failures (HORIZON, + arXiv 2604.11978). Errors compound *between* steps, not within them. + Non-linear: sharp collapse past a domain-specific threshold. +3. **Compaction-induced information loss** — self-inflicted, caused by the + mitigation for #1. +4. **Goal/identity drift** — *downstream* of 1–3, not an independent disease. +5. **Hallucinated state** — phantom invoices, fabricated history. +6. **Memory staleness / negative transfer.** + +The 2026 evidence does **not** support treating goal drift as the primary +problem. It is the observable end-stage of context and error problems. + +### The compaction finding that contradicts our plan + +**Simple observation masking matches or beats LLM summarization at roughly +half the cost** (JetBrains, 250-turn SWE-bench trajectories, NeurIPS 2025 +workshop). Mechanism: LLM summarization **extended trajectories 13–15%** by +destroying natural stopping signals — the agent loses the cue that it already +finished something. + +Better still, **Addressable Recall Compaction** (arXiv 2607.25066): mask the +observation but leave an ID the agent can dereference on demand. Beat +full-context, sliding window, LLM summary, structured state, *and* RAG memory +(NIAH 99.40% vs 88.12% best baseline). + +And the single highest-leverage technique found anywhere in this sweep: +**programmatic tool calling** — the model writes code that orchestrates tools, +so intermediate outputs never enter context. On LOCA-bench it was the **only** +strategy positive across all four models tested (+6 to +13.3 points). It is +absent from our roadmap entirely. + +Two further hard requirements: + +- **Re-inject invariants verbatim after every compaction; never summarize + them.** "Governance Decay" (arXiv 2606.22528) shows safety constraints and + system-prompt instructions erode through successive compaction cycles with + no failure signal. This is architectural, not jailbreaking. +- **Capability-gate context features.** On LOCA-bench a memory tool made a + weaker model *worse* (10.7% → 8.0%), and context-budget awareness made it + much worse (10.7% → 4.0%), while both helped stronger models. Multi-model + support makes per-tier feature flags mandatory. + +--- + +## 5. Self-improvement: the literature turned against us + +RuVector already has ReasoningBank-style trajectory learning and a shared +brain. The 2026 evidence on this class of system is substantially negative. + +**The case for** is real: ReasoningBank reports +4.6 to +8.3 points on WebArena +across three backbones (ICLR 2026); ACE reports +10.6% on agents. Distilling +from *failures* as well as successes is a genuine contribution. + +**The case against is now stronger:** + +- **The gains are confound-sized.** MemDelta (arXiv 2606.29914) shows swapping + the embedding model alone shifts accuracy ±6.2pp — comparable to + ReasoningBank's entire headline gain. In the same work, **agent self-memory + (42%) underperformed plain retrieval (47%)**, and one system reached parity + with cloud RAG at **50× the cost**. +- **Memory utility is an inverted U.** "Useful Memories Become Faulty" + (arXiv 2605.12978): utility rises, then degrades *below* the no-memory + baseline. GPT-5.4 failed **54% of previously-solved ARC-AGI problems** when + using consolidated memory. Episodic-only management **doubled** accuracy vs + forced consolidation — **the consolidation step is the bug, not the storage.** +- **Benchmark-local.** MemoryArena specifically names ReasoningBank's + procedural memory as performing poorly on interdependent multi-session tasks + — the setting closest to real work. +- **No automatic self-evolution method sustains positive gain across settings** + (EvoAgentBench, arXiv 2607.05202). *Curated* ability content transfers + across model families; *automatic extraction* is the failure point. +- **Greedy acceptance is uncontrolled multiple testing.** PACE + (arXiv 2606.08106): "keep it if the score improved" committed **30–42% false + edits**, and made 13–21 spurious modifications when *no true gains existed*, + degrading one agent by 4.9 points. + +Meanwhile **Live-SWE-agent reaches 79.2% on SWE-bench Verified with zero +persistent memory** — on-the-fly tool synthesis from the current trajectory, +discarded after use. It sidesteps every failure mode above, which is not a +coincidence. + +**Verdict:** trajectory-learning memory is a nice-to-have with fragile upside, +not a differentiator. It must ship as a **gated optimization with a measured +contribution**, never on the default path unmeasured. + +Minimum viable discipline if we keep it: + +- Two-tier storage: **episodic immutable**, distilled artifacts *derived* with + pointers back to source episodes. +- Gated, delta-only consolidation with hard length caps. Never end-to-end + rewrites (ACE documents a single step collapsing 18,282 tokens → 122). +- Trust tiers by verifier: programmatic (tests/typecheck) = active; + LLM-judge-only = quarantined; **derived from untrusted content = never + allowed to influence tool-permission or destructive-action decisions.** +- **PACE-style anytime-valid sequential testing for promotion**, never greedy. +- Retrieved memories framed as **"references, not rules"** — nearly free, and + drops attack success 20.6% → 13.1%. +- **A permanent memory-off control arm.** Not a one-time ablation — without a + live control we cannot detect the inverted-U crossover. + +Primary metric: **paired net lift on the same instances**, with the regression +metric that matters most being *rate of previously-solved tasks that now fail +with memory on*. + +**Shared-brain caveat.** Cross-agent shared memory multiplies the poisoning +blast radius. Measured memory-poisoning work reports ~50% attack success and +~41% *relapse* success, with contextual assimilation as the primary vector — +poisoned entries work best when they look like ordinary preferences. The +shared brain needs the strongest gates in the system, not the weakest. + +--- + +## 6. Verifier quality is the binding constraint + +Any self-improving loop is only as good as its verdict signal, and the 2026 +evidence on LLM-as-judge is poor: + +- A judge surfaced **under 25%** of human-confirmed systematic problems, and + **flagged zero** issues in a batch where humans confirmed 23 distinct + defects. Implied **3–6× undercount** of true defect rates. +- Blind-spot structure is systematic: catches **turn-local** problems, severely + blind to **cross-turn state**. +- Calibration drifts: one judge at 0.91 agreement shifted four points after a + model update — the signal stopped meaning what it meant. + +**Design rule:** programmatic and execution-grounded signals first (tests, type +checks, schema validation, invariants). LLM judge as a *secondary, quarantined* +signal only. Treat the judge as a **regression floor, never a promotion +authority**. Re-anchor against a rolling human-labeled sample after every model +update. + +--- + +## 7. Benchmark hygiene + +**SWE-bench Verified is saturated and unreliable as a claim target.** + +- UTBoost (ACL 2025): resolve rates inflated **~6.4 pp** by weak tests; **1 in + 5** "solved" patches semantically incorrect; augmenting tests changed + leaderboard ranks in **24.4%** of Verified submissions. +- OpenAI's own audit found **59.4%** of the hardest unsolved Verified problems + had flawed test cases, and OpenAI stopped reporting Verified in early 2026. + +This directly threatens our Phase 4 exit gate (`≥70% on a 350-instance +fixed-model set`) — the gate is stated against a saturated benchmark whose +noise band exceeds the effect sizes we would be claiming. + +**Consequence for our own A/B testing:** before trusting any measurement of a +harness change, verify test determinism (run F2P/P2P repeatedly under gold and +base patches). A 3-point "improvement" sits inside the flaky-test noise band. +Invest in eval-loop speed — validation that cannot run in minutes will not get +run. + +--- + +## 8. Roadmap corrections + +Specific, and each traceable to a finding above. + +### Phase 1 — reorder and add + +**Add a new Phase 1a "reliability floor" ahead of everything else**, comprising +Tier 1 items 1–7 (§3). These are days of work with the largest measured +deltas, and five of seven are currently missing. The present Phase 1 leads with +event-streaming, cache-first prompts, and compaction — all defensible, none of +them the biggest lever. + +**Add programmatic tool calling** (§4). Highest-leverage single technique in +the sweep; absent from the plan. + +### Phase 1.3 / Phase 2.5 — change the compaction bet + +Both currently bet on summarization (middle-turn summarization; ADR-252 +coherence-weighted compaction). Evidence says summarization is close to the +*worst* measured option and inflates trajectories 13–15%. + +**Replace with:** observation masking as the default, plus addressable recall +(masked entries keep a dereferenceable ID). Keep summarization as a fallback +behind a rubric. Add mandatory verbatim invariant re-injection post-compaction. + +Note this also affects the shipped default pipeline, which currently includes a +`summarization` middleware. + +### Phase 1.6 — subagents: drop the CRDT merge + +Currently "CoW fork/merge, CRDT join." This is parallel-writer architecture, +which is the one multi-agent pattern with strong negative evidence for coding. +Cognition's 2026 revision — after a year of production data — is **one writer, +augmented by auxiliary intelligence; never parallel writes.** + +**Replace with** the two patterns that have production evidence: a +**fresh-context reviewer** (no shared context — it performs *better* without +it) and a **read-only context-gatherer** returning a summary string. Model the +subagent boundary as *a tool that spawns an isolated context and returns a +String*, not as peers with a message bus. That buys nearly all demonstrated +upside at a fraction of the complexity. + +### Phase 2 — gate the learning loop + +"SONA on the default path" contradicts §5. Move behind a feature gate with the +measurement apparatus (paired lift, previously-solved regression rate, control +arm) as the *precondition* for enabling it, not a follow-up. + +The existing exit gate (≥30% token reduction on a repeated task suite) is +well-formed — keep it, and add the control arm. + +### Phase 3 — MCP spec migration is now urgent + +MCP **2026-07-28** landed days ago and is breaking: protocol-level sessions and +`Mcp-Session-Id` removed, `_meta` on every request, `server/discover` added, +Tasks extension replaces long-running `tools/call`, and Roots/Sampling/Logging +deprecated. Building against the 2025-11-25 shape means a rewrite within +months. The official Rust SDK (`rmcp`) already implements the new spec. + +**Also add ACP as a first-class target.** It went from Zed-only to headline +feature of Zed 1.0, built into JetBrains, a public registry, and 25+ agents — +and its reference implementation is Rust. We have an `rvagent-acp` crate +already; this is closer than it looks. + +### Phase 4 — re-ground the SOTA claim + +Drop the SWE-bench-Verified-based gate. Retarget to non-saturated benchmarks +and to the axes where we can actually win. The differentiator claims should be +**library API + deterministic replay** (§2), not speed. + +Report action-match rate for replay, not reproducibility. Keep the honesty +apparatus — given §7, retraction discipline is the feature. + +--- + +## 9. Open items + +- Benchmark/leaderboard sweep (current top scores by model+harness, + cost-normalized Pareto, credible-claim conformance criteria) is still + outstanding; §7 covers hygiene but not target selection. +- `codex-rs` internals in §2 come from a third-party architecture writeup, not + the repo. Worth reading `codex-rs/core/src/` directly before copying the + `Op`/`EventMsg` design. +- Grok Build's governance may change. If xAI opens PRs, the "no community Rust + harness" gap closes fast — the moat must be the library API and replay, not + merely that a Rust harness exists. From e709e1a7443083d151d87c399700c95b038cd5fc Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 16:54:19 +0000 Subject: [PATCH 08/25] =?UTF-8?q?feat(rvagent):=20loop=20detection=20?= =?UTF-8?q?=E2=80=94=20refuse=20tool=20calls=20that=20repeat=20without=20p?= =?UTF-8?q?rogress?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A stuck agent repeats the same call forever, and max_iterations does not fix that — it only makes it more expensive. The loop now fingerprints each tool call by (name, args) and refuses it once it has repeated consecutively past a threshold (default 3), substituting an actionable message that tells the model to change approach rather than an opaque error. Evidence: the harness technique literature ranks loop/stuck detection among the highest value-per-effort items, identifying repeated near-identical actions as the clearest failure signal in agent trajectories, and notes explicitly that raising iteration caps does not help. Counting is consecutive rather than windowed, deliberately. An agent that re-runs the same check between edits is doing legitimate work; a windowed counter would refuse it. Only an unbroken run of identical calls trips the detector. Alternating cycles are not caught — max_iterations remains the backstop, and the limitation is documented on the type. Refused calls still emit exactly one tool result each, in the model's original call order, so the provider's tool_use/tool_result pairing stays in sync. Tests cover: refusal after threshold with execution actually stopping, differing args not treated as a loop, interleaved re-runs not refused (the false-positive case that drove the design), disabling via threshold 0, and fingerprint stability across call ids and JSON key order. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-core/src/graph.rs | 344 ++++++++++++++++++++++- 1 file changed, 330 insertions(+), 14 deletions(-) diff --git a/crates/rvAgent/rvagent-core/src/graph.rs b/crates/rvAgent/rvagent-core/src/graph.rs index 7d5063bd8f..5f5d93d9ef 100644 --- a/crates/rvAgent/rvagent-core/src/graph.rs +++ b/crates/rvAgent/rvagent-core/src/graph.rs @@ -74,6 +74,11 @@ pub struct GraphConfig { pub parallel_tools: bool, /// Maximum tool calls in flight at once when `parallel_tools` is set. pub max_parallel_tools: usize, + /// How many times an identical `(tool, args)` pair may repeat + /// consecutively before the call is refused instead of executed. + /// + /// Set to 0 to disable loop detection entirely. + pub loop_repeat_threshold: usize, } impl Default for GraphConfig { @@ -82,10 +87,89 @@ impl Default for GraphConfig { max_iterations: 100, parallel_tools: true, max_parallel_tools: 8, + loop_repeat_threshold: 3, } } } +/// Detects a stuck agent repeating the same tool call. +/// +/// A stuck agent repeats one call forever; raising `max_iterations` only makes +/// that more expensive. Refusing the repeat and telling the model *why* is what +/// breaks the cycle. +/// +/// Repeats are counted **consecutively**, not across a window. An agent that +/// re-runs `cargo test` between edits is doing legitimate work, and refusing +/// that would be worse than the loop it prevents — so only an unbroken run of +/// identical calls trips the detector. +/// +/// Known limitation: alternating cycles (A, B, A, B, ...) are not detected. +/// `max_iterations` remains the backstop for those. +#[derive(Debug)] +struct LoopDetector { + last: Option, + consecutive: usize, + threshold: usize, +} + +impl LoopDetector { + fn new(threshold: usize) -> Self { + Self { + last: None, + consecutive: 0, + threshold, + } + } + + fn enabled(&self) -> bool { + self.threshold > 0 + } + + /// Fingerprint a call by name and arguments. + /// + /// Arguments are hashed via their JSON string, which is canonical for key + /// ordering because `serde_json` maps are `BTreeMap` by default — so + /// logically identical calls collide regardless of the order the model + /// emitted the keys in. + fn fingerprint(call: &ToolCall) -> u64 { + use std::hash::{Hash, Hasher}; + let mut hasher = std::collections::hash_map::DefaultHasher::new(); + call.name.hash(&mut hasher); + call.args.to_string().hash(&mut hasher); + hasher.finish() + } + + /// Record a call and report whether it has now repeated too often. + fn observe(&mut self, call: &ToolCall) -> bool { + if !self.enabled() { + return false; + } + let fp = Self::fingerprint(call); + if self.last == Some(fp) { + self.consecutive += 1; + } else { + self.last = Some(fp); + self.consecutive = 1; + } + self.consecutive >= self.threshold + } +} + +/// The message substituted for a call that tripped loop detection. +/// +/// Tool errors must be actionable rather than opaque — the model has to be +/// told what to do differently, or it repeats the call again. +fn loop_break_message(call: &ToolCall, threshold: usize) -> String { + format!( + "Tool execution refused: this exact call to '{}' with identical \ + arguments has already been made {} times without progress, so it was \ + not executed again. Repeating it will not produce a different result. \ + Change the arguments, use a different tool, or explain what is \ + blocking you.", + call.name, threshold + ) +} + /// The agent execution graph. /// /// Implements the core loop: @@ -163,6 +247,8 @@ impl AgentGraph { // usage metadata attached by provider backends. let mut total_input_tokens: u64 = 0; let mut total_output_tokens: u64 = 0; + // Spans the whole run: a loop is only visible across iterations. + let mut loop_detector = LoopDetector::new(self.config.loop_repeat_threshold); info!(node = ?current_node, tools = tool_definitions.len(), "graph: starting agent loop"); @@ -218,17 +304,39 @@ impl AgentGraph { // Extract tool calls from the last AI message. let tool_calls = self.extract_tool_calls(&state)?; + // Refuse calls that are repeating without progress. This + // happens before dispatch so a stuck agent stops burning + // tokens and side effects on the same call. + let mut looping: Vec> = Vec::with_capacity(tool_calls.len()); + for tc in &tool_calls { + looping.push(if loop_detector.observe(tc) { + warn!(tool = %tc.name, "graph: loop detected, refusing repeated call"); + Some(loop_break_message(tc, self.config.loop_repeat_threshold)) + } else { + None + }); + } + // Only calls that cleared loop detection reach the executor. + let dispatch: Vec = tool_calls + .iter() + .zip(&looping) + .filter(|(_, refused)| refused.is_none()) + .map(|(tc, _)| tc.clone()) + .collect(); + // Tool failures are fed back to the model as tool results // rather than aborting the loop — the model must see the // error to recover from it (execution alignment). - if self.config.parallel_tools && tool_calls.len() > 1 { + let mut executed: std::collections::HashMap = + std::collections::HashMap::with_capacity(dispatch.len()); + if self.config.parallel_tools && dispatch.len() > 1 { // True parallel execution (ADR-103 A2): tasks are // spawned onto the runtime with bounded concurrency; // results are returned in input order. let executor = Arc::clone(&self.tool_executor); let exec_state = state.clone(); let results = parallel_execute_limited( - tool_calls.clone(), + dispatch, move |tc: ToolCall| { let executor = Arc::clone(&executor); let exec_state = exec_state.clone(); @@ -254,25 +362,36 @@ impl AgentGraph { self.config.max_parallel_tools.max(1), ) .await; - for (id, name, result) in results { - state.push_message(Message::tool_with_name( - id, - tool_result_content(result), - name, - )); + for (id, _name, result) in results { + executed.insert(id, tool_result_content(result)); } } else { // Sequential execution. - for tc in &tool_calls { + for tc in &dispatch { let result = self.tool_executor.execute(tc, &state).await; - state.push_message(Message::tool_with_name( - &tc.id, - tool_result_content(result), - &tc.name, - )); + executed.insert(tc.id.clone(), tool_result_content(result)); } } + // Emit one tool result per call, in the model's original + // call order, substituting the refusal for looping calls. + for (tc, refused) in tool_calls.iter().zip(looping) { + let content = match refused { + Some(msg) => msg, + None => executed.remove(&tc.id).unwrap_or_else(|| { + // Defensive: a dispatched call must always + // produce a result. Report rather than drop it, + // since a missing tool result desyncs the + // provider's tool_use/tool_result pairing. + format!( + "Tool execution error: no result produced for '{}'", + tc.name + ) + }), + }; + state.push_message(Message::tool_with_name(&tc.id, content, &tc.name)); + } + debug!("graph: Tools → Agent"); current_node = AgentNode::Agent; } @@ -498,6 +617,203 @@ mod tests { assert!(matches!(err, RvAgentError::Timeout(_))); } + /// Counts how many times the executor was actually invoked. + struct CountingExecutor { + calls: Arc, + } + + #[async_trait] + impl ToolExecutor for CountingExecutor { + async fn execute(&self, call: &ToolCall, _state: &AgentState) -> Result { + self.calls + .fetch_add(1, std::sync::atomic::Ordering::SeqCst); + Ok(format!("result of {}", call.name)) + } + } + + fn repeated_call_model(n: usize, args: serde_json::Value) -> MockModel { + let mut responses: Vec = (0..n) + .map(|i| { + Message::ai_with_tools( + "", + vec![ToolCall { + // Distinct ids, identical name+args — a real stuck loop + // looks exactly like this. + id: format!("tc{i}"), + name: "noop".into(), + args: args.clone(), + }], + ) + }) + .collect(); + responses.push(Message::ai("done")); + MockModel::new(responses) + } + + #[tokio::test] + async fn test_repeated_identical_call_is_refused() { + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let graph = AgentGraph::with_config( + repeated_call_model(5, serde_json::json!({"x": 1})), + CountingExecutor { + calls: Arc::clone(&calls), + }, + GraphConfig { + max_iterations: 20, + parallel_tools: false, + loop_repeat_threshold: 3, + ..GraphConfig::default() + }, + ); + let result = graph.run(AgentState::new()).await.unwrap(); + + // The 3rd identical call and everything after it must be refused, so + // the executor sees exactly 2 invocations. + assert_eq!( + calls.load(std::sync::atomic::Ordering::SeqCst), + 2, + "loop detection did not stop execution" + ); + + let tool_msgs: Vec<&str> = result + .messages + .iter() + .filter_map(|m| match m { + Message::Tool(t) => Some(t.content.as_str()), + _ => None, + }) + .collect(); + // Every call still gets exactly one result — dropping one would desync + // the provider's tool_use/tool_result pairing. + assert_eq!(tool_msgs.len(), 5); + assert!(tool_msgs[0].contains("result of noop")); + assert!(tool_msgs[1].contains("result of noop")); + for refused in &tool_msgs[2..] { + assert!( + refused.contains("Tool execution refused"), + "expected refusal, got: {refused}" + ); + // The refusal must tell the model what to do differently. + assert!(refused.contains("Change the arguments")); + } + } + + #[tokio::test] + async fn test_differing_args_are_not_treated_as_a_loop() { + // Same tool, different arguments each time: legitimate work. + let mut responses: Vec = (0..5) + .map(|i| { + Message::ai_with_tools( + "", + vec![ToolCall { + id: format!("tc{i}"), + name: "noop".into(), + args: serde_json::json!({ "x": i }), + }], + ) + }) + .collect(); + responses.push(Message::ai("done")); + + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let graph = AgentGraph::with_config( + MockModel::new(responses), + CountingExecutor { + calls: Arc::clone(&calls), + }, + GraphConfig { + max_iterations: 20, + parallel_tools: false, + ..GraphConfig::default() + }, + ); + graph.run(AgentState::new()).await.unwrap(); + assert_eq!(calls.load(std::sync::atomic::Ordering::SeqCst), 5); + } + + #[tokio::test] + async fn test_interleaved_repeats_are_not_refused() { + // The false positive that drove consecutive-only counting: re-running + // the same check between edits is legitimate and must not be blocked. + let check = ToolCall { + id: "c".into(), + name: "run_tests".into(), + args: serde_json::json!({}), + }; + let mut responses = Vec::new(); + for i in 0..4 { + responses.push(Message::ai_with_tools( + "", + vec![ToolCall { + id: format!("e{i}"), + name: "edit".into(), + args: serde_json::json!({ "line": i }), + }], + )); + responses.push(Message::ai_with_tools( + "", + vec![ToolCall { + id: format!("c{i}"), + ..check.clone() + }], + )); + } + responses.push(Message::ai("done")); + + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let graph = AgentGraph::with_config( + MockModel::new(responses), + CountingExecutor { + calls: Arc::clone(&calls), + }, + GraphConfig { + max_iterations: 30, + parallel_tools: false, + ..GraphConfig::default() + }, + ); + graph.run(AgentState::new()).await.unwrap(); + assert_eq!( + calls.load(std::sync::atomic::Ordering::SeqCst), + 8, + "legitimate interleaved re-runs were refused" + ); + } + + #[tokio::test] + async fn test_loop_detection_can_be_disabled() { + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let graph = AgentGraph::with_config( + repeated_call_model(4, serde_json::json!({})), + CountingExecutor { + calls: Arc::clone(&calls), + }, + GraphConfig { + max_iterations: 20, + parallel_tools: false, + loop_repeat_threshold: 0, + ..GraphConfig::default() + }, + ); + graph.run(AgentState::new()).await.unwrap(); + assert_eq!(calls.load(std::sync::atomic::Ordering::SeqCst), 4); + } + + #[test] + fn test_fingerprint_ignores_call_id_and_key_order() { + let a = ToolCall { + id: "one".into(), + name: "t".into(), + args: serde_json::json!({"a": 1, "b": 2}), + }; + let b = ToolCall { + id: "two".into(), + name: "t".into(), + args: serde_json::json!({"b": 2, "a": 1}), + }; + assert_eq!(LoopDetector::fingerprint(&a), LoopDetector::fingerprint(&b)); + } + #[test] fn test_graph_edges() { let model = MockModel::new(vec![]); From dcc907ba85c72187073b8533a47e0084da6bdfee Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 21:00:52 +0000 Subject: [PATCH 09/25] =?UTF-8?q?docs(adr):=20ADR-273..277=20=E2=80=94=20S?= =?UTF-8?q?OTA=20findings=20as=20decision=20records?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Five ADRs converting the 2026 SOTA sweep into decisions. Three of them reverse or constrain standing bets in the roadmap. ADR-273 Harness Reliability Floor. Sequences work by measured reliability impact. The cleanest ablation available moves +54.3 points, almost entirely from patch-apply failures dropping 69.1% -> <1.5%: the dominant wins are mechanical, not reasoning. Defines seven floor mechanisms and an explicit not-doing list (few-shot, ungrounded reflection, embedding code index, context beyond ~128k, learned components). ADR-274 Context Management: masking over summarization. Reverses Phase 1.3, Phase 2.5/ADR-252, and the shipped default pipeline. Observation masking matches or beats LLM summarization at ~half cost, and summarization inflates trajectories 13-15% by destroying stopping signals. Adds addressable recall, programmatic tool calling (the only context strategy positive on every model tested), verbatim invariant re-injection against documented governance decay, and per-model-tier capability gating. ADR-275 Subagent Topology: single writer with auxiliary intelligence. Reverses Phase 1.6's CoW fork/merge + CRDT join, which is parallel-writer architecture -- the one multi-agent pattern with strong negative evidence for coding. Adopts fresh-context reviewer (deliberately no shared context) and read-only context-gatherer on a cheap model tier. Keeps JoinSet/semaphore concurrency; drops the mergeable state type entirely. ADR-276 Learning Loop: gating, trust tiers, measurement. Constrains "SONA on the default path". Ships feature-gated off with the measurement apparatus as a precondition: paired lift, previously-solved regression rate, permanent memory-off control arm, plain-retrieval baseline, confound controls. Immutable episodic storage, delta-only consolidation, trust tiers with structural enforcement that untrusted-derived memories cannot influence permission decisions, and sequential-test promotion instead of greedy accept. ADR-277 Positioning, Protocols and Claims. Invalidates the premise that a Rust harness is itself differentiating -- Codex CLI, Grok Build (~844k LOC, open-sourced 2026-07-15) and Goose are already Rust. Repositions on the two open gaps: a stable embeddable library API with open governance, and deterministic replay reported as action-match rate rather than reproducibility, since hosted inference is not reproducible even at temperature 0. Mandates MCP 2026-07-28 migration, makes ACP first-class, and withdraws the SWE-bench-Verified exit gate as a claim target. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- ...R-273-rvagent-harness-reliability-floor.md | 182 ++++++++++++++ .../adr/ADR-274-rvagent-context-management.md | 196 +++++++++++++++ docs/adr/ADR-275-rvagent-subagent-topology.md | 164 ++++++++++++ .../ADR-276-rvagent-learning-loop-gating.md | 237 ++++++++++++++++++ .../ADR-277-rvagent-positioning-and-claims.md | 215 ++++++++++++++++ 5 files changed, 994 insertions(+) create mode 100644 docs/adr/ADR-273-rvagent-harness-reliability-floor.md create mode 100644 docs/adr/ADR-274-rvagent-context-management.md create mode 100644 docs/adr/ADR-275-rvagent-subagent-topology.md create mode 100644 docs/adr/ADR-276-rvagent-learning-loop-gating.md create mode 100644 docs/adr/ADR-277-rvagent-positioning-and-claims.md diff --git a/docs/adr/ADR-273-rvagent-harness-reliability-floor.md b/docs/adr/ADR-273-rvagent-harness-reliability-floor.md new file mode 100644 index 0000000000..b82ae46d63 --- /dev/null +++ b/docs/adr/ADR-273-rvagent-harness-reliability-floor.md @@ -0,0 +1,182 @@ +--- +adr: 273 +title: "rvAgent Harness Reliability Floor" +status: accepted +date: 2026-08-01 +authors: [Reuven Cohen] +project: "rvAgent Harness" +related: [ADR-103, ADR-139, ADR-274, ADR-275, ADR-276, ADR-277] +tags: [rvagent, harness, reliability, agent-loop, tools, error-recovery, sota] +--- + +# ADR-273 — rvAgent Harness Reliability Floor + +## Status + +**Accepted.** Owner: Reuven Cohen. Project: rvAgent Harness. Date: 2026-08-01. + +Evidence base: `docs/research/rvagent-hermes-harness/04-sota-landscape.md`. + +## 1. Decision + +**Sequence harness work by measured reliability impact, not by architectural +ambition.** A defined "reliability floor" of seven mechanisms ships before +event-streaming, cache-tiering, evolution, or any other Phase 1+ item. + +This ADR reorders `03-roadmap.md` Phase 1. It does not remove any item from it. + +## 2. Context + +The cleanest harness ablation available (Claw-SWE-Bench, Jun 2026) rebuilds a +harness from bare model-emits-diff to full scaffolding: **19.1% → 73.4%, ++54.3 points**. Nearly all of that delta is patch-apply failures falling from +69.1% to under 1.5%. + +That result generalizes across the technique literature: **the dominant wins +come from eliminating mechanical failure modes, not from better reasoning.** +Patches that don't apply, tool-call loops, context rot, flaky tests. + +Two structural qualifiers: + +- **Harness value scales inversely with model strength.** Same harness set: + 12.5-point spread on GLM-5.1, 27.4-point spread on Qwen-3.6-flash. Targeting + frontier models only should be expected to yield roughly half the ROI the + literature reports. +- **Effect sizes are mostly single-source preprints.** The figures below are + hypotheses to re-measure in our own harness, not constants. What is + multiply-corroborated is the *direction* and the *ranking*. + +This also settles the project's positioning argument. The case for Rust is not +speed — it is that the failure modes dominating these ablations can be made +**type-unrepresentable** rather than merely rare. See ADR-277. + +## 3. The floor + +Seven mechanisms, ordered by measured effect per unit of effort. + +| # | Mechanism | Reported effect | Status | +|---|---|---|---| +| 1 | Reliable patch application | +54.3 pp | Partial | +| 2 | Observation-window management | +3 pp, prevents long-run collapse | Planned (ADR-274) | +| 3 | Loop / stuck detection | Removes most common catastrophic failure | **Done** | +| 4 | Actionable tool errors + response size caps | Part of the reliability delta | Partial | +| 5 | Tool surface held to 8–15 tools | Avoids −16 to −23 pt routing collapse | Holding at 9 | +| 6 | Environment bootstrap injection | Meta-Harness @ 76.4% TB2.0 | Planned | +| 7 | Persisted thinking across tool calls | +2.2 pp coding | Planned | + +### 3.1 Reliable patch application + +Real workspace, file-based edits, git-based diff extraction, and +**verify-after-write** — re-read the file and confirm the edit landed before +reporting success. Offer `str_replace` with fuzzy-failure diagnostics and +`write_file` side by side; **do not** offer unified diff, where line numbers, +hunk headers, and trailing newlines dominate apply failures. + +Edit-tool ergonomics are load-bearing, not incidental: in the Qwen-3.6 +reproduction only the SWE-agent `str_replace_editor` flavor moved the number, +while a different `edit`/`write_file` pair gave *zero* improvement. + +Post-edit gate: run `cargo check` (not a full build) inline. This is the +analogue of SWE-agent's linter guardrail, which its ablations found essential +for recovering from bad edits. + +### 3.2 Loop / stuck detection — **implemented** + +Fingerprint each tool call by `(name, args)`; refuse it once it repeats +consecutively past a threshold (default 3), substituting an actionable message. + +**Counting is consecutive, not windowed.** An agent re-running the same check +between edits is doing legitimate work; a windowed counter refuses it. Only an +unbroken run of identical calls trips the detector. Alternating cycles are not +caught — `max_iterations` remains the backstop, and the limitation is +documented on the type. + +Raising `max_iterations` does **not** fix loops; it makes them more expensive. +`max_iterations` is a cost cap, not a loop guard. + +Refused calls still emit exactly one tool result each, in the model's original +call order, so provider `tool_use`/`tool_result` pairing stays in sync. + +### 3.3 Actionable errors and output caps + +Tool errors must state what failed *and what to do differently*. An opaque +error code causes the model to retry the identical call, which is the input +condition for §3.2. + +Cap tool responses (~25k tokens, matching Claude Code's default) with explicit +truncation markers. An uncapped tool result can consume the context window in +one call. + +### 3.4 Tool surface budget + +**Hold the builtin surface at 8–15 tools.** Routing accuracy degrades 16–23 +points across large catalogs; vendors document degradation past 30–50 tools. +We currently ship 9 builtins — this is a constraint to *defend*, not a target +to grow toward. + +MCP servers must therefore be gated behind explicit per-session enablement. +Exposing an MCP firehose directly into the tool list forfeits this. + +### 3.5 Environment bootstrap injection + +Before the loop starts, snapshot the workspace and inject it into the initial +prompt: cwd, file listing, toolchain versions, `cargo metadata` summary, +workspace members, the test command, and whether `cargo check` currently +passes. This eliminates early exploration turns. ~100 lines of code. + +### 3.6 Persisted thinking + +Do not strip prior-turn thinking blocks from history. Pure protocol plumbing; ++2.2 pp on coding (the smallest of the reported deltas — coding benefits least +because tool results are self-explanatory — but free). + +## 4. Explicitly not in the floor + +Rejected for v1 on evidence, not on effort: + +- **Few-shot demonstrations and explicit CoT instructions** for reasoning + models — zero-shot ≥ few-shot; exemplars can contradict native reasoning. +- **Ungrounded self-reflection loops** — can degrade already-correct answers. + Only execution-grounded critique ("tests failed, here is the output") works. +- **Semantic/embedding code index** — vendor-only evidence, high maintenance, + and stale by construction on a repo the agent is actively editing. Layer + ripgrep → structural search → semantic, and only if a conceptual query + demands it. +- **Elaborate system-prompt frameworks** — the widely-quoted "20–30% + improvement" claims have no published methodology. Keep rulebooks under ~60 + lines. +- **Context windows beyond ~128k** — sweeps plateau around 114k; documented + ceilings sit at 96–112k. Buying more window buys nothing. +- **Learned/RL-trained components** — the hand-written 80% is available for 5% + of the effort. + +## 5. Consequences + +**Positive.** The largest measured deltas land first. Five of seven mechanisms +are days of work. The floor is testable end-to-end without a live provider, +which is how the P0 exit gate is already structured. + +**Negative.** Phase 1's architectural items (event-streaming loop, cache-first +prompt tiers) are deferred behind less glamorous work. This is deliberate: the +evidence does not support them being the biggest lever. + +**Risk.** Effect sizes are largely single-source. Mitigation: §6. + +## 6. Measurement obligation + +Before trusting any A/B of a harness change, **verify test determinism** — run +F2P/P2P repeatedly under gold and base patches. Weak tests inflate resolve +rates by ~6.4 pp, and 1 in 5 "solved" patches on a saturated benchmark are +semantically incorrect. A 3-point improvement sits inside the flaky-test noise +band and means nothing. + +Invest in eval-loop speed. Validation that cannot run in minutes will not get +run, and every mechanism in §3 needs it. + +## 7. Implementation status + +- §3.2 loop detection — **shipped** (`rvagent-core/src/graph.rs`, commit `e709e1a`) +- §3.1 partial — tools write real files; verify-after-write and git diff + extraction outstanding +- §3.4 — holding at 9 builtins +- §3.3, §3.5, §3.6 — outstanding diff --git a/docs/adr/ADR-274-rvagent-context-management.md b/docs/adr/ADR-274-rvagent-context-management.md new file mode 100644 index 0000000000..85bcd98eaf --- /dev/null +++ b/docs/adr/ADR-274-rvagent-context-management.md @@ -0,0 +1,196 @@ +--- +adr: 274 +title: "rvAgent Context Management: Masking over Summarization" +status: accepted +date: 2026-08-01 +authors: [Reuven Cohen] +project: "rvAgent Harness" +related: [ADR-103, ADR-252, ADR-273, ADR-275, ADR-276] +supersedes_parts_of: [ADR-252] +tags: [rvagent, harness, context, compaction, masking, long-horizon, sota] +--- + +# ADR-274 — rvAgent Context Management: Masking over Summarization + +## Status + +**Accepted.** Owner: Reuven Cohen. Project: rvAgent Harness. Date: 2026-08-01. + +Evidence base: `docs/research/rvagent-hermes-harness/04-sota-landscape.md` §4. + +**This ADR reverses a standing design bet.** `03-roadmap.md` Phase 1.3 +(middle-turn summarization), Phase 2.5 (ADR-252 coherence-weighted +compaction), and the shipped default middleware pipeline all rely on LLM +summarization as the primary context strategy. The evidence says that is close +to the worst available option. + +## 1. Decision + +1. **Observation masking is the default** context strategy, not summarization. +2. **Masked observations keep a dereferenceable ID** (addressable recall). +3. **Programmatic tool calling** is added as a first-class capability. +4. **Invariants are re-injected verbatim after every compaction**, never + summarized. +5. **Context features are capability-gated per model tier.** +6. LLM summarization is retained only as a **rubric-guided fallback**. + +## 2. Why the reversal + +**Observation masking matches or beats LLM summarization at roughly half the +cost** (JetBrains, 250-turn SWE-bench trajectories, NeurIPS 2025 workshop: ++2.6% solve rate at 52% lower cost on Qwen3-Coder 480B). + +The mechanism matters more than the number: **LLM summarization extended +trajectories 13–15%** by destroying natural stopping signals. The agent loses +the cue that it already finished something and keeps working. A hybrid cut +cost 7% below pure masking. + +A placeholder reading `[test output, 2,847 tokens, exit 1, elided]` is more +useful than a mediocre summary, because it preserves the *shape* of history +without fabricating its contents. + +**Addressable Recall Compaction** (arXiv 2607.25066) improves on plain masking +by leaving an ID the agent can dereference on demand. It beat full-context, +sliding window, LLM summary, structured state, *and* RAG memory — NIAH 99.40% +vs 88.12% for the best baseline. This removes the main objection to masking +(irreversible loss). + +### 2.1 The largest lever is not compaction at all + +On LOCA-bench, six context strategies were compared across four frontier +models at 128k context. **Programmatic tool calling — the model writes code +that orchestrates tools, so intermediate outputs never enter context — was the +only strategy positive on every model tested** (+6.0 to +13.3 points). + +Summarize-and-continue compaction was the *weakest* of the obvious strategies +(+2.6 points on the model where it helped most). + +This is absent from the current roadmap and is the single highest-value +context item. + +### 2.2 Compaction silently erases invariants + +"Governance Decay" (arXiv 2606.22528) shows safety constraints and +system-prompt instructions **erode through successive compaction cycles with no +failure signal.** This is not jailbreaking — it emerges from the compression +architecture itself. + +Therefore: the task statement, acceptance criteria, safety constraints, and +system prompt are **re-emitted byte-identical** after each compaction. They are +never inputs to a summarizer. Cheap, and directly counters a documented +mechanism. + +### 2.3 Context features can make weaker models worse + +On LOCA-bench, giving DeepSeek-V3.2 a memory tool dropped it from 10.7% to +**8.0%**, and telling it its remaining context budget dropped it to **4.0%** — +while both features *helped* GPT-5.2 and Gemini-3-Flash. + +Per-model-tier feature flags are therefore mandatory, not optional. A feature +that helps the frontier tier must not be enabled by default for weaker models. + +## 3. Design + +### 3.1 Masking + +Replace old tool observations with typed placeholders carrying: tool name, +elided byte/token count, exit status where applicable, and a recall ID. +**Reasoning steps and actions are kept verbatim** — only observations are +masked. Keep the last N observations in full. + +### 3.2 Addressable recall + +Append-only, ID-addressable log of full observations. The placeholder carries +the handle; a `recall` tool dereferences it. No re-execution, no similarity +search, no embedding index. + +### 3.3 Thresholds + +Compact at **70–80% of nominal context**, not 95%. Treat nominal window as +roughly **2× the reliable working budget** — degradation begins well under half +the advertised window, and it is silent. Anthropic's own API default compacts +at 150k on a 200k-class window, which is the most useful real-world datapoint. + +Do not pursue larger context windows as a substitute: sweeps plateau ~114k and +documented ceilings sit at 96–112k (ADR-273 §4). + +### 3.4 Fallback summarization + +When summarization is used, the **rubric is the load-bearing part**, not the +tool. Preserve explicitly: task statement, files touched, failing tests, +decisions made, unresolved issues. Offering a compaction tool without rubric +guidance produces uneven behavior; a paragraph of guidance closes the gap. + +Guard against recursive summarization of summaries — a documented and fixed bug +in Codex's rewrite. + +### 3.5 Prompt-cache interaction + +Compaction invalidates every cached prefix downstream of the break. Keep a +byte-stable system prefix and place cache breakpoints *before* the volatile +region. Design for 10+ compaction cycles per session. + +## 4. Retrieval policy + +Layered escalation, not a choice between grep and semantic: + +1. **ripgrep** — known symbol, exact identifier, error string +2. **Structural search** (ast-grep / LSP / `syn`) — callers of X, impls of Y +3. **Semantic** — only for conceptual cross-cutting queries + +**Never route a short keyword query to a dense retriever.** CoREB (arXiv +2605.04615) shows short keyword queries — "auth flow", "user service", exactly +the shape agents emit — collapse essentially every embedding model to near-zero +nDCG@10. This explains the 2025 industry migration away from vector search in +coding agents mechanistically. + +Summarize search output by default: return paths and match counts, require a +second call to read contents. + +Any index over code the agent is editing is **stale by construction**. If one +is added later it needs invalidation-on-write or explicit staleness bounds. + +## 5. What breaks first in long runs + +Design priority follows the observed failure order: + +1. **Wasted-context accumulation → attention dilution** — earliest, universal, + invisible. Unaided coding agents waste ~1-in-3 file reads. +2. **History error accumulation** — the largest single driver. Process-level + failures are **72.5%** of long-horizon failures (HORIZON, arXiv 2604.11978). + Errors compound *between* steps. Non-linear: sharp collapse past a + domain-specific threshold. +3. **Compaction-induced loss** — self-inflicted by the mitigation for #1. +4. **Goal / identity drift** — *downstream* of 1–3, not independent. +5. **Hallucinated state** — what #3 looks like when the summary is confidently + wrong rather than merely lossy. + +The evidence does **not** support treating goal drift as the primary problem. +It is the observable end-stage of context and error problems, which is why +this ADR targets 1–3 and ADR-273 targets error compounding. + +## 6. Consequences + +**Positive.** Cheaper and faster than summarization (no model call on the hot +path). Deterministic, so it does not itself become a source of nondeterminism +in replay (ADR-277). No fabrication risk. Preserves stopping signals. + +**Negative.** ADR-252 (coherence-weighted compaction) is demoted from the +default path to a fallback. Work already done there is not wasted — it becomes +the rubric-guided fallback of §3.4 — but it is no longer the primary strategy. + +The `summarization` middleware currently in the shipped default pipeline must +be reordered behind masking, or removed from the default set. + +**Risk.** Masking loses information that a good summary would have retained. +Mitigated by §3.2 addressable recall: the information is still there and still +reachable. + +## 7. Implementation order + +1. Observation masking with last-N-in-full (ADR-273 floor item 2) +2. Verbatim invariant re-injection (§2.2) — cheap, high consequence +3. Addressable recall log + `recall` tool (§3.2) +4. Programmatic tool calling (§2.1) — highest value, largest effort +5. Capability gating (§2.3) +6. Demote `summarization` middleware to fallback (§3.4) diff --git a/docs/adr/ADR-275-rvagent-subagent-topology.md b/docs/adr/ADR-275-rvagent-subagent-topology.md new file mode 100644 index 0000000000..59abbfefd2 --- /dev/null +++ b/docs/adr/ADR-275-rvagent-subagent-topology.md @@ -0,0 +1,164 @@ +--- +adr: 275 +title: "rvAgent Subagent Topology: Single Writer with Auxiliary Intelligence" +status: accepted +date: 2026-08-01 +authors: [Reuven Cohen] +project: "rvAgent Harness" +related: [ADR-103, ADR-107, ADR-273, ADR-274, ADR-277] +tags: [rvagent, harness, subagents, multi-agent, concurrency, sota] +--- + +# ADR-275 — rvAgent Subagent Topology: Single Writer with Auxiliary Intelligence + +## Status + +**Accepted.** Owner: Reuven Cohen. Project: rvAgent Harness. Date: 2026-08-01. + +Evidence base: `docs/research/rvagent-hermes-harness/04-sota-landscape.md` §3. + +**This ADR reverses a standing design bet.** `03-roadmap.md` Phase 1.6 +specifies "real subagents (JoinSet spawn, semaphore, CoW fork/merge, CRDT +join)". CoW fork plus CRDT join *is* parallel-writer architecture — the one +multi-agent pattern with strong negative evidence for coding specifically. + +## 1. Decision + +**One writer. Auxiliary intelligence around it. Never parallel writes.** + +The subagent boundary is modelled as **a tool that spawns an isolated context +and returns a String** — not as peer agents with a message bus, shared mutable +state, or a mergeable state type. + +Two subagent roles are adopted; a third is deferred. + +| Role | Status | Shares context? | Writes? | +|---|---|---|---| +| Fresh-context reviewer | Adopt | **No** — deliberately | No | +| Read-only context-gatherer | Adopt | No | No | +| Coordinator / manager | Defer | — | No | +| Parallel writers | **Rejected** | — | — | + +## 2. Why + +### 2.1 The positive multi-agent result does not transfer to coding + +Anthropic's multi-agent research system reports **+90.2%** over single-agent — +and simultaneously that **token usage alone explains 80% of the performance +variance** (95% with tool-call count and model added). Much of the gain is +*buying more compute*, not coordination; the missing control arm is a single +agent at the same 15× budget. + +Anthropic states directly that the architecture suits **breadth-first** +questions with independent paths and is **less effective for tightly +interdependent tasks such as coding.** + +### 2.2 The skeptical position was revised, not refuted + +Cognition's "Don't Build Multi-Agents" (Jun 2025) argued for single-threaded +linear agents because "actions carry implicit decisions" that conflict when +parallelized. + +Their April 2026 revision — after a year of production data — did not reverse +this. It refined it: + +- **One writer, augmented by auxiliary intelligence. Never parallel writes.** +- Code review loop works: Devin Review catches ~**2 bugs per PR, 58% severe**. +- **Reviewers perform better with NO shared context.** Shorter context → less + context rot → deeper analysis. This inverts the usual "share everything" + instinct and is the most actionable finding here. +- Manager delegation ships but "requires extensive context engineering; + managers default to over-prescription without deep codebase knowledge." + +### 2.3 At equal budget, single-agent wins on coding + +The 2026 consensus across sources: at **equal token budget**, single-agent +matches or beats multi-agent on multi-hop reasoning. Multi-agent earns its +overhead only on breadth-first, parallel-decomposable, low-state-sharing tasks. +Coding is the canonical *bad* fit — it is the case where sub-results are +interdependent and merge conflicts are semantic, not textual. + +A CRDT can merge two edits to the same file without conflict. It cannot make +the *result* coherent. That is precisely the failure Cognition describes. + +## 3. Adopted patterns + +### 3.1 Fresh-context reviewer + +Spawns with **no inherited conversation** — only the diff and the task +statement. Returns findings as a string. Does not write. + +The counterintuitive part is load-bearing: do **not** pass the parent's +context. The reviewer's value comes from evaluating the artifact without the +parent's accumulated rationalizations, and from having a short, clean window. + +### 3.2 Read-only context-gatherer + +Explores, reads, greps; returns a summary string. No shared mutable state, no +write tools in its surface. + +Measured (SWE-Edit, Viewer + Editor split): **+2.1 pp resolve, −17.9% cost, +−34.5% main-agent input tokens.** + +**Use a cheap model here.** A specialized Qwen3-8B editor matched GPT-5-nano; +putting GPT-5 in that slot gave **+0.4 pp at 5.8× cost.** Model tiering per +subagent role is part of the design, not an optimization. + +This is also the cleanest lever on ADR-274 §5 failure #1 (wasted-context +accumulation): exploration output never enters the main window. + +### 3.3 Deferred: coordinator + +Only after single-agent is solid. Requires heavy context engineering to avoid +over-prescription. + +## 4. Rejected: parallel writers, CoW fork/merge, CRDT join + +Rejected on evidence for the coding domain. No rigorous positive coding result +exists; the negative evidence is production-scale and from a team that shipped +the architecture and walked it back. + +**What is kept from Phase 1.6:** `JoinSet` spawning and semaphore-bounded +concurrency remain — they are how §3.1 and §3.2 subagents run concurrently with +each other. What is dropped is CoW state forking and CRDT merge, because +nothing writes concurrently and therefore nothing needs merging. + +This is a substantial simplification: it removes a mergeable state type, the +merge-conflict semantics, and the entire class of bugs where two subagents +make locally-valid but jointly-incoherent edits. + +## 5. Concurrency model + +Retained from the existing loop and unchanged by this ADR: + +- **Read-only tools run concurrently; state-mutating tools run sequentially.** + This is the industry-convergent split and matches MCP's `readOnlyHint`. +- Bounded concurrency via semaphore; results returned in call order. +- Each tool runs in its own task so a panicking tool surfaces as a tool error + rather than crashing the loop (already shipped). + +## 6. Consequences + +**Positive.** Removes the most complex unbuilt subsystem in the roadmap. Buys +nearly all demonstrated multi-agent upside — the reviewer and gatherer are the +two patterns with real production numbers — at a fraction of the complexity. +A `Fn(prompt) -> String` boundary is trivially testable and trivially +replayable (ADR-277). + +**Negative.** Forecloses the "swarm of coders on one repo" demo. That demo has +no supporting evidence for coding tasks and would likely produce incoherent +results, so this is a cost worth paying — but it is a visible capability we are +choosing not to build. + +**Interaction with ADR-107** (rvagent native swarm/WASM): swarm topology +remains valid for *independent* tasks across separate workspaces. This ADR +constrains concurrent writers **within a single workspace on a single task**, +which is the case the evidence covers. + +## 7. Implementation + +1. Subagent-as-tool trait: isolated context in, String out +2. Fresh-context reviewer (no inherited history) +3. Read-only gatherer with a write-free tool surface and cheap-tier model +4. Per-role model override +5. Remove CoW fork/merge and CRDT join from the Phase 1.6 scope diff --git a/docs/adr/ADR-276-rvagent-learning-loop-gating.md b/docs/adr/ADR-276-rvagent-learning-loop-gating.md new file mode 100644 index 0000000000..896c80d135 --- /dev/null +++ b/docs/adr/ADR-276-rvagent-learning-loop-gating.md @@ -0,0 +1,237 @@ +--- +adr: 276 +title: "rvAgent Learning Loop: Gating, Trust Tiers and Measurement" +status: accepted +date: 2026-08-01 +authors: [Reuven Cohen] +project: "rvAgent Harness" +related: [ADR-271, ADR-273, ADR-274, ADR-275, ADR-277, ADR-323] +tags: [rvagent, harness, memory, reasoningbank, sona, self-improvement, evaluation, security, sota] +--- + +# ADR-276 — rvAgent Learning Loop: Gating, Trust Tiers and Measurement + +## Status + +**Accepted.** Owner: Reuven Cohen. Project: rvAgent Harness. Date: 2026-08-01. + +Evidence base: `docs/research/rvagent-hermes-harness/04-sota-landscape.md` §5–6. + +**This ADR constrains a standing design bet.** `03-roadmap.md` Phase 2.4 puts +"SONA on the default path". The 2026 literature turned substantially against +this class of system. The component is not cancelled — it is gated behind +measurement it must earn. + +## 1. Decision + +1. **Trajectory-learning memory ships feature-gated OFF by default.** +2. **The measurement apparatus is a precondition for enabling it**, not a + follow-up. +3. **A permanent memory-off control arm** runs for the life of the system. +4. **Episodic storage is immutable**; distilled artifacts are derived. +5. **Consolidation is gated and delta-only** — never end-to-end rewrites. +6. **Trust tiers by verifier**, with untrusted-derived memories barred from + influencing permission or destructive-action decisions. +7. **Promotion uses anytime-valid sequential testing**, never greedy + accept-if-better. + +## 2. Why the constraint + +The case *for* is real and peer-reviewed: ReasoningBank reports +4.6 to +8.3 +points on WebArena across three backbones (ICLR 2026); ACE reports +10.6% on +agents. Distilling from *failures* as well as successes is a genuine +contribution over success-only baselines. + +The case *against* is now stronger: + +- **The gains are confound-sized.** MemDelta (arXiv 2606.29914): swapping the + embedding model alone shifts accuracy **±6.2 pp** — comparable to + ReasoningBank's entire headline gain. Without controlled ablation you cannot + distinguish "our memory design works" from "we picked a better embedder." In + the same work, **agent self-memory (42%) underperformed plain retrieval + (47%)**, and one system reached parity with cloud RAG at **50× the cost**. +- **Memory utility is an inverted U.** "Useful Memories Become Faulty" + (arXiv 2605.12978): utility rises, then degrades *below* the no-memory + baseline. GPT-5.4 failed **54% of previously-solved ARC-AGI problems** when + using consolidated memory. **Episodic-only management doubled accuracy** vs + forced consolidation — the consolidation step is the bug, not the storage. +- **Gains are benchmark-local.** MemoryArena specifically names ReasoningBank's + procedural memory as performing poorly on interdependent multi-session tasks + — the setting closest to real work. +- **No automatic self-evolution method sustains positive gain across settings** + (EvoAgentBench, arXiv 2607.05202). *Curated* ability content transfers across + model families; *automatic extraction* is the failure point. +- **Greedy acceptance is uncontrolled multiple testing.** PACE (arXiv + 2606.08106): "keep it if the score improved" committed **30–42% false + edits**, and made 13–21 spurious modifications when *no true gains existed*, + degrading one agent by 4.9 points. + +Meanwhile **Live-SWE-agent reaches 79.2% on SWE-bench Verified with zero +persistent memory** — on-the-fly tool synthesis from the current trajectory, +discarded after use. It sidesteps every failure mode above. That is not a +coincidence: nothing persistent means nothing to poison, stale, or collapse. + +**Conclusion.** Trajectory-learning memory is a nice-to-have with fragile +upside, not a differentiator (see ADR-277 for what the differentiators are). + +## 3. Design + +### 3.1 Two-tier storage + +**Episodic is immutable.** Append-only raw trajectory store; never rewritten, +never overwritten. Distilled playbook items are *derived artifacts* carrying +pointers back to their source episodes. + +Rationale: consolidation is the documented failure point, and episodic-only +management doubled accuracy against forced consolidation. Raw episodes remain +primary evidence. + +### 3.2 Gated, delta-only consolidation + +Never run consolidation automatically after each task. Never rewrite the +playbook end-to-end. Append or amend individual items with structured deltas. + +Hard per-item length cap (~1,500 chars) and a hard total cap. ACE documents a +single end-to-end rewrite collapsing **18,282 tokens → 122**, dropping +performance *below* the no-adaptation baseline. Production experience +independently shows unconstrained growth past 5,000 chars overfits, and that +length regularization is nearly free (4× compression for −0.8%). + +More data made it worse: 500 samples grew prompt length +75% and *dropped* +performance 2% versus a 20–100 sample sweet spot. + +### 3.3 Trust tiers + +Every candidate memory carries provenance: source episode, verdict source, +verifier type, timestamp. + +| Tier | Backed by | Retrieval | +|---|---|---| +| **A — active** | Programmatic verifier: tests pass, type check, schema validation, invariant assertion | Full weight | +| **B — quarantine** | LLM-judge verdict only | Reduced weight, or withheld until promoted | +| **C — tainted** | Derived from untrusted content: fetched pages, tool output, user-supplied text | Separate namespace. **Never** allowed to influence tool-permission or destructive-action decisions | + +Tier C is a security boundary, not a quality heuristic. See §5. + +### 3.4 Promotion by sequential testing + +Promotion B → A, and any prompt or scaffold edit, requires a **paired +anytime-valid sequential test** (e-process / testing-by-betting) against the +current version on identical held-out instances. Commit only when evidence is +decisive. + +PACE achieved comparable accuracy at **~18% lower evaluation cost** than greedy +acceptance while eliminating the false-commit rate. This is the single +highest-value component of the learning loop. + +### 3.5 Retrieval discipline + +- Inject retrieved memories framed explicitly as **"references, not rules"**. + Nearly free, and drops attack success **20.6% → 13.1%** while raising the + refusal rate 54.4% → 66.9%. +- Relevance-gated top-k with a threshold. **Never concatenate the whole bank** — + ExpeL's documented scaling failure. +- Hard token budget on injected memory; over budget, drop lowest-trust first. +- TTL, decay, and eviction on realized contribution. + +## 4. Measurement — the precondition + +The gate does not open until these run. + +**Primary metric — paired net lift:** +`lift = P(success | memory) − P(success | no memory)` on *the same instances*, +with PACE's e-process providing the stopping rule. Report a confidence +interval, not a point estimate. + +**The regression metric that matters most:** rate of **previously-solved tasks +that now fail with memory on**. This is the 54%-on-ARC-AGI signal and it is the +earliest warning that consolidation has gone bad. Track per consolidation +event; above threshold, roll back and quarantine the items it produced. + +**Permanent control arm.** A fraction of traffic — or a shadow run on a frozen +held-out suite — always executes with memory disabled. Not a one-time +ablation: without a live control the inverted-U crossover is undetectable. + +**Confound controls (run before believing any result):** + +- Hold the embedding model **fixed** across arms; report sensitivity separately. + A ±6.2 pp swing from the embedder alone masquerades as an architecture win. +- Hold the backbone LLM fixed; re-verify on a second backbone. +- **Log refusal rates per arm.** A 63%-refusal arm is not comparable to a 5% one. +- Always include a **plain-retrieval baseline** (BM25 or vanilla embedding RAG + over raw episodes). Failing to beat it is a 50×-cost parity result, not a win. + +**Cost-normalized:** tokens and dollars per *additional* success, not raw +accuracy. + +**Transfer holdout:** a task set from a *different distribution* than the +memories were written from. In-distribution gain is expected and tells you +almost nothing. + +**Per-item attribution:** track retrieval count and conditional lift when +retrieved; evict items with negative or non-significant contribution. This +makes the bank self-pruning and gives an audit trail when something poisons it. + +## 5. Security: the shared brain is the highest-risk surface + +Cross-agent shared memory multiplies the blast radius. Measured +memory-poisoning work reports **~50% attack success and ~41% relapse success**, +with **contextual assimilation** as the primary vector — poisoned entries work +best when they look like ordinary preferences, constraints, or workflow +requirements. Reported >90% of tested agents vulnerable, with **100% relapse** +when teams tried to fix it conversationally. + +Implications: + +- Tier C (§3.3) is mandatory and must be enforced structurally, not by prompt. +- Never write raw credentials, PHI, or secrets to shared memory (already policy). +- Sleeper entries may lie dormant until triggered — per-item attribution (§4) + is the detection mechanism. +- "Misevolution" affects top-tier models; the cheapest known mitigation is the + references-not-rules framing of §3.5. + +## 6. Verifier quality is the binding constraint + +Any learning loop is only as good as its verdict signal, and LLM-as-judge +evidence is poor: a judge surfaced **under 25%** of human-confirmed systematic +problems and **flagged zero** issues in a batch where humans confirmed 23 +distinct defects — implying a **3–6× undercount**. Blind spots are structural: +it catches turn-local problems and is severely blind to cross-turn state. +Calibration drifts — one judge at 0.91 agreement shifted four points after a +model update. + +**Rules:** + +- Programmatic and execution-grounded signals first (tests, type checks, schema + validation, invariants). +- LLM judge is a **secondary, quarantined** signal only (Tier B). +- Treat the judge as a **regression floor, never a promotion authority.** +- Re-anchor against a rolling human-labeled sample after **every** model update. + +## 7. Consequences + +**Positive.** The component ships honestly. If it works we can prove it; if it +degrades we detect the crossover instead of shipping a silent regression. The +measurement apparatus (§4) is reusable for every other harness change and +overlaps with the eval-loop investment ADR-273 §6 already requires. + +**Negative.** Slower to enable than "SONA on the default path". Significant +work lands before any measured benefit. This is the correct trade given §2 — +the alternative is enabling a component whose own literature says it may go +below baseline. + +**Relationship to ADR-271.** ADR-271's Darwin/SONA self-improvement direction +remains valid as *mechanism*. This ADR supplies the gating and acceptance +criteria it lacked, and replaces greedy fitness acceptance with §3.4. + +## 8. Implementation order + +1. Episodic append-only store (immutable, §3.1) +2. Measurement harness: paired lift, previously-solved regression rate, control + arm, plain-retrieval baseline (§4) +3. Trust tiers with structural Tier-C enforcement (§3.3, §5) +4. Retrieval discipline with references-not-rules framing (§3.5) +5. Gated delta-only consolidation with caps (§3.2) +6. PACE-style sequential-test promotion (§3.4) +7. Only then: consider default-on, if and only if §4 shows sustained positive + paired lift on the transfer holdout diff --git a/docs/adr/ADR-277-rvagent-positioning-and-claims.md b/docs/adr/ADR-277-rvagent-positioning-and-claims.md new file mode 100644 index 0000000000..0fe68e681a --- /dev/null +++ b/docs/adr/ADR-277-rvagent-positioning-and-claims.md @@ -0,0 +1,215 @@ +--- +adr: 277 +title: "rvAgent Positioning, Protocols and Benchmark Claims" +status: accepted +date: 2026-08-01 +authors: [Reuven Cohen] +project: "rvAgent Harness" +related: [ADR-159, ADR-267, ADR-273, ADR-274, ADR-275, ADR-276] +tags: [rvagent, harness, positioning, mcp, acp, replay, benchmarks, sota, honesty] +--- + +# ADR-277 — rvAgent Positioning, Protocols and Benchmark Claims + +## Status + +**Accepted.** Owner: Reuven Cohen. Project: rvAgent Harness. Date: 2026-08-01. + +Evidence base: `docs/research/rvagent-hermes-harness/04-sota-landscape.md` §2, §7. + +**This ADR invalidates a premise.** The roadmap's implicit positioning — that a +Rust-native harness is itself differentiating — was true when the research +began and is false as of July 2026. + +## 1. Decision + +1. **Do not position on "Rust is fast."** That ground is taken. +2. **Position on two open gaps:** a stable embeddable **library API with open + governance**, and **deterministic replay** as a core primitive. +3. **Migrate to MCP 2026-07-28** before building further on the tool protocol. +4. **Treat ACP as a first-class target.** +5. **Retire SWE-bench Verified as a claim target.** +6. **No claim ships without harness disclosure.** + +## 2. The field is crowded at the top + +Three of the major 2026 harnesses are already Rust: + +| Harness | Scale | License | Notes | +|---|---|---|---| +| **Codex CLI** (OpenAI) | ~70–80 crates | Apache 2.0 | Rewritten *from* TypeScript | +| **Grok Build** (xAI) | ~844k LOC | Apache 2.0 | Open-sourced **2026-07-15** | +| **Goose** (Block) | — | Apache 2.0 | MCP-native | + +OpenAI's stated reasons for the rewrite are exactly our positioning: +zero-dependency install (Node 22+ blocked enterprise and air-gapped +deployments), no GC pauses in long-running agentic processes, memory-safe +sandbox bindings without FFI shims. + +**Tool-surface novelty is also unavailable.** Grok Build's tools are documented +ports — `apply_patch`, `grep_files`, `list_dir`, `read_file` from Codex; +`bash`, `edit`, `glob`, `grep`, `read`, `skill`, `todowrite`, `write` from +opencode. A frontier lab with a million lines of Rust ported the tool surface +rather than designing one. Convergence is complete. + +## 3. What remains open + +### 3.1 No Rust harness is usable as a library + +- Codex's own `AGENTS.md` **discourages** adding to `codex-core` — it is an + app, not a published SDK. +- **Grok Build has issues and PRs disabled.** Contributions explicitly + rejected; xAI develops internally and syncs a mirror. This is *source + transparency, not open governance.* +- Goose is app-first. +- **Anthropic's Agent SDK is Python and TypeScript only** — the docs instruct + other languages to shell out to the CLI with `-p --output-format json`. + +A stable, semver'd, embeddable harness crate with open governance is +unoccupied. This is the primary position. + +### 3.2 Deterministic replay + +A named gap in the Rust agent ecosystem and thin everywhere. Our existing +append-only witness/segment infrastructure is most of the way there, and +ADR-274's deterministic masking (no model call on the compaction path) and +ADR-275's `Fn(prompt) -> String` subagent boundary both make it tractable. + +**Honesty constraint — this is binding.** Even at temperature 0, hosted +inference is not reproducible: floating-point non-associativity and +batch-size-dependent kernels produce run-to-run variation, with reported +accuracy swings up to 15% across runs. + +Therefore the claim is **replay of the harness, not of the model**, and the +reported metric is **action-match rate**, never "reproducible" without +qualification. Record every LLM call, tool response, and timestamp; replay to +reproduce harness behavior; promote incidents to test fixtures. + +Claiming byte-exact reproducibility would be false and would be caught. + +### 3.3 Why Rust, restated honestly + +Not speed. The defensible argument is ADR-273's: the failure modes that +dominate every harness ablation — patch-apply failures, tool-call loops, +desynced `tool_use`/`tool_result` pairing, unbounded tool output — can be made +**type-unrepresentable** rather than merely rare. Plus structured concurrency +(`JoinSet` + `CancellationToken`) making mid-run interrupt and subagent +lifecycle nearly free, where they are hard in Python. + +**Where Rust is a liability — state these plainly:** + +- **Provider coverage.** LiteLLM's 100+ providers is a moat. Rust's best is + ~20. Permanent maintenance tax. +- **Iteration speed on what matters most.** Prompts, tool descriptions, and + compaction rubrics are where harness performance lives, and they want a + REPL. Mitigation: keep prompts and templates in **hot-reloadable external + files**, never `const &str`. +- **The eval ecosystem is Python.** We will shell out for evaluation. +- **Compile times** on a large workspace are a daily cost. +- **Extension authors don't write Rust.** Mitigation: **the extension language + is MCP, not Rust** — Goose's key insight. Pi has 2,143 third-party extensions + because they are TypeScript. + +## 4. Protocols + +### 4.1 MCP 2026-07-28 — migrate now + +Landed 2026-07-28 and is breaking: + +- **Stateless core.** Protocol-level sessions and `Mcp-Session-Id` **removed**. +- Protocol version, client info, and capabilities now travel in `_meta` on + **every** request. +- New `server/discover` method. +- **Tasks extension:** `tools/call` returns a task handle driven via + `tasks/get` / `tasks/update` / `tasks/cancel`. `tasks/list` removed. +- **Deprecated:** Roots, Sampling, Logging; HTTP+SSE reclassified deprecated. +- Auth aligns with real OAuth 2.0/OIDC; clients must validate `iss` per + RFC 9207 (mix-up attack mitigation). + +Building against the 2025-11-25 shape means a rewrite within months. The +official Rust SDK (`rmcp`) already implements the new spec while remaining +compatible with older ones — start there rather than hand-rolling. + +MCP servers remain gated behind per-session enablement (ADR-273 §3.4): the +tool-count ceiling is not negotiable for protocol convenience. + +### 4.2 ACP — first-class + +ACP went from Zed-only (Jun 2025) to headline feature of Zed 1.0 (2026-04-29), +built into JetBrains since Dec 2025, a public registry (2026-01-28), and 25+ +agents by March 2026. **Its reference implementation is Rust.** + +We already have an `rvagent-acp` crate. Being ACP-native rather than +ACP-bolted-on is closer than it looks and is the natural distribution channel +for a library-shaped harness. + +Division of labor as settled in 2026: **MCP = tools, A2A = agent discovery, +ACP = editor↔agent, AG-UI = agent↔UI.** ADR-159's A2A work sits in the third +slot and remains valid. + +## 5. Benchmark claims + +### 5.1 SWE-bench Verified is retired as a claim target + +- **UTBoost (ACL 2025):** resolve rates inflated **~6.4 pp** by weak tests; + **1 in 5** "solved" patches semantically incorrect; augmenting tests changed + leaderboard ranks in **24.4%** of Verified submissions (40.9% on Lite). +- **OpenAI's own audit:** **59.4%** of the hardest unsolved Verified problems + had flawed test cases. OpenAI stopped reporting Verified in early 2026. + +`03-roadmap.md` Phase 4's exit gate (`≥70% on a 350-instance fixed-model set`) +is stated against a benchmark whose **noise band exceeds the effect sizes we +would be claiming.** It is withdrawn pending replacement. + +Verified may still be used as an internal regression signal — with test +determinism verified per ADR-273 §6 — but not as a published claim. + +### 5.2 Claim conformance + +Two 2026 papers establish that harness choices substantially determine +benchmark results (arXiv 2605.23950) and that harnesses induce systematically +different agent *beliefs* on logically equivalent tasks (arXiv 2607.04528). + +Therefore every published claim must disclose: environment setup, tool +implementations, the full harness configuration, and the evaluation procedure. +**A result without harness disclosure is not a result.** + +Retain the existing honesty apparatus (ADR-267): fixed-model comparison, +conformant packaged submissions, Wilson confidence intervals, and retraction +discipline. Given §5.1, retraction discipline is a feature. + +### 5.3 Differentiator claims + +Claim the axes from §3, not throughput: + +- Stable embeddable library API with open governance (§3.1) +- Harness replay with **action-match rate** reported (§3.2) +- Startup latency, memory footprint, single-binary distribution — real, but + **already claimed by Codex and Grok Build.** Supporting evidence, not the + headline. + +## 6. Consequences + +**Positive.** Positioning now rests on gaps that are actually open and on +claims that survive scrutiny. The library-API framing also improves the +internal architecture — it forces a clean core/app split that ADR-275's +subagent-as-tool boundary and §3.2's replay both need. + +**Negative.** The "first fast Rust harness" story is gone. Phase 4's headline +gate is withdrawn without a replacement in hand (§7). + +**Risk — Grok Build's governance may change.** If xAI opens PRs, the "no +community Rust harness" gap closes quickly. The moat must be the library API +and replay, **not merely that a Rust harness exists.** + +## 7. Open + +Benchmark target selection is unresolved. Required before Phase 4 can restate +its gate: current top scores by model *and* harness, cost-normalized Pareto +position, and conformance criteria for a credible 2026 claim on a +non-saturated benchmark. Tracked in +`04-sota-landscape.md` §9. + +Also outstanding: read `codex-rs/core/src/` directly before adopting its +`Op`/`EventMsg` submit/event design — the survey's account of it is +third-party. From d7e55b1c621079e874ab92dd05d6fb903407af07 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 21:07:58 +0000 Subject: [PATCH 10/25] feat(rvagent): observation masking + tool output caps (ADR-274, ADR-273) Implements the default context strategy from ADR-274. Old tool observations are replaced with compact placeholders before history is sent to the model; reasoning steps and actions pass through verbatim. Masking is a projection, not a mutation: AgentState::messages stays the complete append-only log, and each placeholder carries the tool_call_id as a recall handle, so elided content remains addressable. Placeholders name the tool and the elided byte count so the model can tell what it is missing. Also caps individual tool results at write time (ADR-273 3.3) with an explicit truncation marker, in both the parallel and sequential execution paths. An uncapped tool result can consume the context window in a single call. Writing the end-to-end test for the cap surfaced two defects in read_file's line formatter: - Lines over the length limit were truncated silently, leaving the model believing it had seen the whole line. Now marked with the omitted byte count. - The truncation sliced at a raw byte offset, which panics when a multi-byte character straddles it. A UTF-8 file with a long non-ASCII line would have crashed the tool. Now walks back to a character boundary. The existing formatter test asserted the silent-truncation behaviour, so it was updated rather than left encoding the bug; added coverage for the char-boundary cases. Tests: 12 e2e (2 new: masking reaches the model with the full log preserved, oversized output capped), plus unit coverage for masking, truncation, and multi-byte safety. 38 suites green across the five rvagent crates. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-core/src/graph.rs | 29 +- crates/rvAgent/rvagent-core/src/lib.rs | 1 + crates/rvAgent/rvagent-core/src/masking.rs | 258 ++++++++++++++++++ crates/rvAgent/rvagent-tools/src/lib.rs | 63 ++++- .../rvagent-tools/tests/e2e_tool_calling.rs | 170 ++++++++++-- 5 files changed, 484 insertions(+), 37 deletions(-) create mode 100644 crates/rvAgent/rvagent-core/src/masking.rs diff --git a/crates/rvAgent/rvagent-core/src/graph.rs b/crates/rvAgent/rvagent-core/src/graph.rs index 5f5d93d9ef..3dc5a3e56f 100644 --- a/crates/rvAgent/rvagent-core/src/graph.rs +++ b/crates/rvAgent/rvagent-core/src/graph.rs @@ -9,6 +9,7 @@ use serde::{Deserialize, Serialize}; use tracing::{debug, info, instrument, warn}; use crate::error::{Result, RvAgentError}; +use crate::masking::{mask_observations, truncate_tool_result, MaskConfig}; use crate::messages::{Message, ToolCall}; use crate::models::{ChatModel, ToolDefinition}; use crate::parallel::parallel_execute_limited; @@ -79,6 +80,8 @@ pub struct GraphConfig { /// /// Set to 0 to disable loop detection entirely. pub loop_repeat_threshold: usize, + /// Observation masking and tool-output caps (ADR-274). + pub mask: MaskConfig, } impl Default for GraphConfig { @@ -88,6 +91,7 @@ impl Default for GraphConfig { parallel_tools: true, max_parallel_tools: 8, loop_repeat_threshold: 3, + mask: MaskConfig::default(), } } } @@ -271,10 +275,11 @@ impl AgentGraph { iterations += 1; debug!(iteration = iterations, "graph: invoking model"); - let response = self - .model - .complete(&state.messages, &tool_definitions) - .await?; + // The model sees a masked projection; `state.messages` + // stays the complete log so elided content remains + // addressable by tool_call_id (ADR-274). + let outbound = mask_observations(&state.messages, &self.config.mask); + let response = self.model.complete(&outbound, &tool_definitions).await?; if let Some((input, output)) = usage_from_message(&response) { total_input_tokens += input; total_output_tokens += output; @@ -363,13 +368,25 @@ impl AgentGraph { ) .await; for (id, _name, result) in results { - executed.insert(id, tool_result_content(result)); + executed.insert( + id, + truncate_tool_result( + tool_result_content(result), + self.config.mask.max_tool_result_bytes, + ), + ); } } else { // Sequential execution. for tc in &dispatch { let result = self.tool_executor.execute(tc, &state).await; - executed.insert(tc.id.clone(), tool_result_content(result)); + executed.insert( + tc.id.clone(), + truncate_tool_result( + tool_result_content(result), + self.config.mask.max_tool_result_bytes, + ), + ); } } diff --git a/crates/rvAgent/rvagent-core/src/lib.rs b/crates/rvAgent/rvagent-core/src/lib.rs index 416660e17e..e198e70191 100644 --- a/crates/rvAgent/rvagent-core/src/lib.rs +++ b/crates/rvAgent/rvagent-core/src/lib.rs @@ -24,6 +24,7 @@ pub mod config; pub mod cow_state; pub mod error; pub mod graph; +pub mod masking; pub mod messages; pub mod metrics; pub mod models; diff --git a/crates/rvAgent/rvagent-core/src/masking.rs b/crates/rvAgent/rvagent-core/src/masking.rs new file mode 100644 index 0000000000..4e3b24cb5d --- /dev/null +++ b/crates/rvAgent/rvagent-core/src/masking.rs @@ -0,0 +1,258 @@ +//! Observation masking — the default context strategy (ADR-274). +//! +//! Old tool observations are replaced with compact placeholders before the +//! history is sent to the model. Reasoning steps and actions are kept verbatim; +//! only observations are masked. +//! +//! This is deliberately *not* summarization. Measured comparisons put simple +//! masking at or above LLM summarization on solve rate at roughly half the +//! cost, and show that summarization inflates trajectories 13–15% by destroying +//! the natural stopping signals an agent uses to notice it has finished. A +//! placeholder that says `[read_file output elided: 2847 bytes, recall id tc7]` +//! preserves the *shape* of history without fabricating its contents. +//! +//! Masking is a projection, not a mutation: `AgentState::messages` remains the +//! complete append-only log, which is what makes the elided content +//! addressable — the `tool_call_id` in each placeholder is the recall handle. + +use crate::messages::{Message, ToolMessage}; + +/// Configuration for the observation-masking projection. +#[derive(Debug, Clone)] +pub struct MaskConfig { + /// How many of the most recent observations to keep in full. + /// + /// Set to `usize::MAX` to disable masking. + pub keep_last_observations: usize, + /// Hard cap on a single tool result's size, applied at write time. + /// + /// An uncapped tool result can consume the whole context window in one + /// call. Truncation is explicit and marked so the model knows output was + /// cut rather than silently ending. + pub max_tool_result_bytes: usize, +} + +impl Default for MaskConfig { + fn default() -> Self { + Self { + keep_last_observations: 8, + // ~25k tokens at ~4 bytes/token, matching the industry default. + max_tool_result_bytes: 100_000, + } + } +} + +impl MaskConfig { + /// Whether masking is active at all. + pub fn masking_enabled(&self) -> bool { + self.keep_last_observations != usize::MAX + } +} + +/// Truncate `content` to at most `max_bytes`, on a character boundary, +/// appending an explicit marker when anything was removed. +/// +/// Operates on bytes rather than chars because the cap exists to bound memory +/// and context cost, but never splits a UTF-8 sequence. +pub fn truncate_tool_result(content: String, max_bytes: usize) -> String { + if content.len() <= max_bytes { + return content; + } + // Reserve room for the marker so the result still respects the budget. + let marker = "\n... [output truncated]"; + let budget = max_bytes.saturating_sub(marker.len()); + + // Walk back to a char boundary so we never split a multi-byte sequence. + let mut end = budget.min(content.len()); + while end > 0 && !content.is_char_boundary(end) { + end -= 1; + } + + let mut out = String::with_capacity(end + marker.len()); + out.push_str(&content[..end]); + out.push_str(marker); + out +} + +/// The placeholder substituted for an elided observation. +fn placeholder(msg: &ToolMessage) -> String { + let name = msg.tool_name.as_deref().unwrap_or("tool"); + format!( + "[{} output elided: {} bytes, recall id {}]", + name, + msg.content.len(), + msg.tool_call_id + ) +} + +/// Project `messages` into the view sent to the model, masking all but the +/// most recent `keep_last_observations` tool results. +/// +/// Non-tool messages pass through untouched — masking reasoning or actions +/// would destroy exactly the trail the model needs to stay coherent. +pub fn mask_observations(messages: &[Message], config: &MaskConfig) -> Vec { + if !config.masking_enabled() { + return messages.to_vec(); + } + + let total_observations = messages + .iter() + .filter(|m| matches!(m, Message::Tool(_))) + .count(); + if total_observations <= config.keep_last_observations { + return messages.to_vec(); + } + let mask_before = total_observations - config.keep_last_observations; + + let mut seen = 0usize; + messages + .iter() + .map(|msg| match msg { + Message::Tool(tool) => { + let index = seen; + seen += 1; + if index < mask_before { + Message::Tool(ToolMessage { + tool_call_id: tool.tool_call_id.clone(), + content: placeholder(tool), + tool_name: tool.tool_name.clone(), + metadata: tool.metadata.clone(), + }) + } else { + msg.clone() + } + } + other => other.clone(), + }) + .collect() +} + +#[cfg(test)] +mod tests { + use super::*; + + fn tool(id: &str, name: &str, content: &str) -> Message { + Message::tool_with_name(id, content, name) + } + + #[test] + fn keeps_everything_when_under_the_limit() { + let msgs = vec![ + Message::ai("thinking"), + tool("t1", "read_file", "a"), + tool("t2", "read_file", "b"), + ]; + let out = mask_observations(&msgs, &MaskConfig::default()); + assert_eq!(out, msgs); + } + + #[test] + fn masks_all_but_the_last_n_observations() { + let config = MaskConfig { + keep_last_observations: 2, + ..MaskConfig::default() + }; + let msgs = vec![ + tool("t1", "read_file", "oldest"), + tool("t2", "grep", "older"), + tool("t3", "read_file", "recent"), + tool("t4", "ls", "newest"), + ]; + let out = mask_observations(&msgs, &config); + + let contents: Vec<&str> = out + .iter() + .map(|m| match m { + Message::Tool(t) => t.content.as_str(), + _ => unreachable!(), + }) + .collect(); + + assert!(contents[0].contains("read_file output elided")); + assert!(contents[0].contains("recall id t1")); + assert!(contents[1].contains("grep output elided")); + // The most recent two survive verbatim. + assert_eq!(contents[2], "recent"); + assert_eq!(contents[3], "newest"); + } + + #[test] + fn never_masks_reasoning_or_actions() { + let config = MaskConfig { + keep_last_observations: 0, + ..MaskConfig::default() + }; + let msgs = vec![ + Message::system("rules"), + Message::human("do the thing"), + Message::ai("here is my plan"), + tool("t1", "read_file", "contents"), + ]; + let out = mask_observations(&msgs, &config); + + assert_eq!(out[0], msgs[0]); + assert_eq!(out[1], msgs[1]); + assert_eq!(out[2], msgs[2], "AI reasoning must survive masking"); + match &out[3] { + Message::Tool(t) => assert!(t.content.contains("elided")), + _ => panic!("expected a tool message"), + } + } + + #[test] + fn masking_can_be_disabled() { + let config = MaskConfig { + keep_last_observations: usize::MAX, + ..MaskConfig::default() + }; + let msgs: Vec = (0..50) + .map(|i| tool(&format!("t{i}"), "read_file", "body")) + .collect(); + assert_eq!(mask_observations(&msgs, &config), msgs); + } + + #[test] + fn placeholder_preserves_the_recall_handle() { + let config = MaskConfig { + keep_last_observations: 0, + ..MaskConfig::default() + }; + let msgs = vec![tool("call-42", "grep", "many matches")]; + let out = mask_observations(&msgs, &config); + match &out[0] { + Message::Tool(t) => { + // The id must survive so the full content stays addressable, + // and must still pair with the model's tool_use block. + assert_eq!(t.tool_call_id, "call-42"); + assert!(t.content.contains("recall id call-42")); + assert!(t.content.contains("12 bytes")); + } + _ => panic!("expected a tool message"), + } + } + + #[test] + fn truncation_marks_what_it_removed() { + let out = truncate_tool_result("x".repeat(1000), 100); + assert!(out.len() <= 100); + assert!(out.ends_with("[output truncated]")); + } + + #[test] + fn truncation_leaves_short_content_alone() { + let out = truncate_tool_result("short".into(), 100); + assert_eq!(out, "short"); + } + + #[test] + fn truncation_never_splits_a_multibyte_char() { + // Every char is 4 bytes, so a naive byte cut would split one. + let content = "🙂".repeat(100); + for cap in [10usize, 33, 50, 77, 99] { + let out = truncate_tool_result(content.clone(), cap); + // The real assertion is simply that this did not panic and the + // result is valid UTF-8 by construction. + assert!(out.ends_with("[output truncated]"), "cap {cap}"); + } + } +} diff --git a/crates/rvAgent/rvagent-tools/src/lib.rs b/crates/rvAgent/rvagent-tools/src/lib.rs index 00bfc60375..61c75f61be 100644 --- a/crates/rvAgent/rvagent-tools/src/lib.rs +++ b/crates/rvAgent/rvagent-tools/src/lib.rs @@ -477,16 +477,35 @@ pub fn format_content_with_line_numbers(content: &str, start_line: usize) -> Str if i > 0 { out.push('\n'); } - let truncated = &line[..line.len().min(MAX_LINE_LEN)]; use std::fmt::Write; - write!( - out, - "{:>width$}\t{}", - start_line + i, - truncated, - width = LINE_NUMBER_WIDTH - ) - .unwrap(); + if line.len() <= MAX_LINE_LEN { + write!( + out, + "{:>width$}\t{}", + start_line + i, + line, + width = LINE_NUMBER_WIDTH + ) + .unwrap(); + } else { + // Walk back to a character boundary: slicing at a fixed byte offset + // panics when a multi-byte character straddles it. + let mut end = MAX_LINE_LEN; + while end > 0 && !line.is_char_boundary(end) { + end -= 1; + } + // Mark the cut. Silent truncation leaves the model believing it has + // seen the whole line, which is worse than showing less. + write!( + out, + "{:>width$}\t{}… [line truncated, {} more bytes]", + start_line + i, + &line[..end], + line.len() - end, + width = LINE_NUMBER_WIDTH + ) + .unwrap(); + } } out } @@ -946,9 +965,31 @@ mod tests { let result = format_content_with_line_numbers(&long_line, 1); let lines: Vec<&str> = result.lines().collect(); assert_eq!(lines.len(), 1); - // Extract the content after the line number and tab let content = lines[0].split('\t').nth(1).unwrap(); - assert_eq!(content.len(), MAX_LINE_LEN); + // The cut must be visible: silent truncation leaves the model believing + // it saw the whole line. + assert!(content.starts_with(&"a".repeat(MAX_LINE_LEN))); + assert!(content.contains("[line truncated, 100 more bytes]")); + } + + #[test] + fn test_format_line_truncation_is_char_boundary_safe() { + // A multi-byte character straddling the cut point used to panic on a + // raw byte slice. + let mut line = "a".repeat(MAX_LINE_LEN - 1); + line.push('é'); // 2 bytes, spanning MAX_LINE_LEN + line.push_str(&"b".repeat(50)); + let result = format_content_with_line_numbers(&line, 1); + assert!(result.contains("[line truncated")); + } + + #[test] + fn test_format_multibyte_line_does_not_panic() { + for pad in 0..4 { + let mut line = "a".repeat(MAX_LINE_LEN - pad); + line.push_str(&"🙂".repeat(10)); + let _ = format_content_with_line_numbers(&line, 1); + } } #[test] diff --git a/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs b/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs index 4802b5a168..01ee7c9c63 100644 --- a/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs +++ b/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs @@ -172,26 +172,6 @@ async fn schemas_are_sent_on_every_turn_including_after_tools() { Message::ai("read it"), ])); - // Share the model so the recorded observations survive the graph. - struct SharedModel(Arc); - #[async_trait] - impl ChatModel for SharedModel { - async fn complete( - &self, - messages: &[Message], - tools: &[ToolDefinition], - ) -> Result { - self.0.complete(messages, tools).await - } - async fn stream( - &self, - messages: &[Message], - tools: &[ToolDefinition], - ) -> Result> { - self.0.stream(messages, tools).await - } - } - let graph = AgentGraph::new( SharedModel(Arc::clone(&model)), RealToolExecutor::new(dir.path()), @@ -449,6 +429,156 @@ async fn usage_metadata_is_aggregated() { assert_eq!(totals, (250, 50), "usage metadata was lost or miscounted"); } +// --------------------------------------------------------------------------- +// ADR-274 — observation masking reaches the model +// --------------------------------------------------------------------------- + +/// Shares one `ScriptedModel` so its recorded observations outlive the graph. +struct SharedModel(Arc); + +#[async_trait] +impl ChatModel for SharedModel { + async fn complete(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result { + self.0.complete(messages, tools).await + } + async fn stream(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result> { + self.0.stream(messages, tools).await + } +} + +#[tokio::test] +async fn old_observations_are_masked_before_reaching_the_model() { + let dir = tempfile::tempdir().unwrap(); + for i in 0..5 { + std::fs::write( + dir.path().join(format!("f{i}.txt")), + format!("UNIQUE-BODY-{i}"), + ) + .unwrap(); + } + + // Five sequential read turns, then a final answer. + let mut responses: Vec = (0..5) + .map(|i| { + Message::ai_with_tools( + "", + vec![call( + &format!("t{i}"), + "read_file", + serde_json::json!({ "file_path": format!("f{i}.txt") }), + )], + ) + }) + .collect(); + responses.push(Message::ai("done")); + + let model = Arc::new(ScriptedModel::new(responses)); + let config = GraphConfig { + parallel_tools: false, + mask: rvagent_core::masking::MaskConfig { + keep_last_observations: 2, + ..Default::default() + }, + ..GraphConfig::default() + }; + let graph = AgentGraph::with_config( + SharedModel(Arc::clone(&model)), + RealToolExecutor::new(dir.path()), + config, + ); + let state = graph.run(AgentState::new()).await.unwrap(); + + // What the model saw on the final turn: only the last 2 observations in + // full, the rest elided but still addressable. + let seen = model.last_messages.lock().unwrap().clone(); + let seen_tools: Vec<&Message> = seen + .iter() + .filter(|m| matches!(m, Message::Tool(_))) + .collect(); + assert_eq!(seen_tools.len(), 5, "every call still has a paired result"); + + for (i, msg) in seen_tools.iter().enumerate() { + let Message::Tool(t) = msg else { unreachable!() }; + if i < 3 { + assert!( + t.content.contains("output elided"), + "observation {i} should have been masked: {}", + t.content + ); + assert!( + t.content.contains(&format!("recall id t{i}")), + "masked observation {i} lost its recall handle: {}", + t.content + ); + assert!( + !t.content.contains(&format!("UNIQUE-BODY-{i}")), + "masked observation {i} still carried its full body" + ); + } else { + assert!( + t.content.contains(&format!("UNIQUE-BODY-{i}")), + "recent observation {i} must survive verbatim: {}", + t.content + ); + } + } + + // The stored log keeps everything — masking is a projection, not a + // mutation, which is what makes the elided content recoverable. + let stored = tool_results(&state); + assert_eq!(stored.len(), 5); + for (i, content) in stored.iter().enumerate() { + assert!( + content.contains(&format!("UNIQUE-BODY-{i}")), + "stored observation {i} was destroyed by masking: {content}" + ); + } +} + +#[tokio::test] +async fn oversized_tool_output_is_capped() { + let dir = tempfile::tempdir().unwrap(); + // Many lines, so the total far exceeds the cap. (A single very long line + // would not: read_file truncates individual lines at its own limit.) + let body: String = (0..20_000).map(|i| format!("line {i}\n")).collect(); + std::fs::write(dir.path().join("big.txt"), body).unwrap(); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "", + vec![call( + "t1", + "read_file", + serde_json::json!({"file_path": "big.txt", "limit": 100_000}), + )], + ), + Message::ai("done"), + ]); + let config = GraphConfig { + mask: rvagent_core::masking::MaskConfig { + max_tool_result_bytes: 4_000, + ..Default::default() + }, + ..GraphConfig::default() + }; + let graph = AgentGraph::with_config(model, RealToolExecutor::new(dir.path()), config); + let state = graph.run(AgentState::new()).await.unwrap(); + + let results = tool_results(&state); + assert_eq!(results.len(), 1); + assert!( + results[0].len() <= 4_000, + "tool output was not capped: {} bytes", + results[0].len() + ); + assert!( + results[0].ends_with("[output truncated]"), + "truncation must be explicit so the model knows output was cut; got {} bytes: {:?}", + results[0].len(), + &results[0][..results[0].len().min(200)] + ); +} + // --------------------------------------------------------------------------- // Confinement holds through the full loop // --------------------------------------------------------------------------- From 89d894bfed73b52a02e30aa8730bfd6294e36f09 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 21:43:55 +0000 Subject: [PATCH 11/25] =?UTF-8?q?docs(adr):=20ADR-278=20=E2=80=94=20adopt?= =?UTF-8?q?=20metaharness=20flywheel;=20shift=20self-learning=20memory=20-?= =?UTF-8?q?>=20policy?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Studied ruvnet/metaharness (@metaharness/flywheel@0.1.7, ADR-226/228/236) for reusable self-learning. Three consequences. 1. Do not build a promotion apparatus. The flywheel already implements what ADR-276 3.4 specified, more rigorously: a frozen conjunctive gate with a SHA-256 fingerprint proving it did not move, a holdout PLUS a frozen anchor never optimized against, Ed25519 receipts, independent replay verification, and a lineage DAG that re-bases on the promoted winner. It is deliberately host-agnostic -- Policy = Record with Proposer/Evaluator as the only seams -- so there is no adapter impedance. Offline, so no Rust port. 2. Shift self-learning investment from memory to policy. These are different objects with opposite evidence: policy text (GEPA-style) is the best-evidenced optimizer in the 2026 sweep, while episodic memory accumulation shows an inverted-U and confound-sized gains. RuVector's weight sits on the memory side. ADR-276's gating is not repealed; new effort moves to policy evolution. 3. Adopt noopRate as a score axis. The default gate requires it to strictly improve -- a policy earns promotion by making the executor commit more, not just score higher. Two internal nulls are now binding. ADR-226: a read-only frontier advisor produced zero marginal gold-scored resolves at 5.4x cost while genuinely firing (33 advisories, 3 vetoes) -- independently corroborating the +0.4pp/5.8x figure already cited from the public literature. ADR-236: the flywheel mechanism was proven end-to-end on real SWE-bench and still produced no compounding lift because the base solver was too weak, confirming that a promotion engine cannot rescue an unreliable loop and that ADR-273's reliability-floor-first ordering is correct. ADR-275 3.1 is amended accordingly: the fresh-context reviewer is downgraded from adopted to gated. It was written on Cognition's production data without considering ADR-226. ADR-226 gave its advisor the full transcript where this reviewer sees only the diff, so it does not refute the design -- but it is the null the reviewer must beat before reaching the default path. 3.2 is unaffected and strengthened. Also identifies an upstream contribution: the flywheel gate is single-shot, so many generations against one holdout is uncontrolled multiple testing -- the regime where PACE measured 30-42% false commits. Frozen conjunctive gate AND anytime-valid sequential test is strictly stronger than either. Closes a gap where ADR-271 did not reference the flywheel at all. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- docs/adr/ADR-275-rvagent-subagent-topology.md | 27 ++- docs/adr/ADR-278-rvagent-flywheel-adoption.md | 205 ++++++++++++++++++ .../04-sota-landscape.md | 29 ++- 3 files changed, 258 insertions(+), 3 deletions(-) create mode 100644 docs/adr/ADR-278-rvagent-flywheel-adoption.md diff --git a/docs/adr/ADR-275-rvagent-subagent-topology.md b/docs/adr/ADR-275-rvagent-subagent-topology.md index 59abbfefd2..97077074d2 100644 --- a/docs/adr/ADR-275-rvagent-subagent-topology.md +++ b/docs/adr/ADR-275-rvagent-subagent-topology.md @@ -34,7 +34,7 @@ Two subagent roles are adopted; a third is deferred. | Role | Status | Shares context? | Writes? | |---|---|---|---| -| Fresh-context reviewer | Adopt | **No** — deliberately | No | +| Fresh-context reviewer | **Gated** (amended, see §3.1) | **No** — deliberately | No | | Read-only context-gatherer | Adopt | No | No | | Coordinator / manager | Defer | — | No | | Parallel writers | **Rejected** | — | — | @@ -83,7 +83,7 @@ the *result* coherent. That is precisely the failure Cognition describes. ## 3. Adopted patterns -### 3.1 Fresh-context reviewer +### 3.1 Fresh-context reviewer — **GATED** (amended 2026-08-01, ADR-278 §7) Spawns with **no inherited conversation** — only the diff and the task statement. Returns findings as a string. Does not write. @@ -92,6 +92,24 @@ The counterintuitive part is load-bearing: do **not** pass the parent's context. The reviewer's value comes from evaluating the artifact without the parent's accumulated rationalizations, and from having a short, clean window. +> **Amendment.** This was originally written as *adopted* on the strength of +> Cognition's production data (~2 bugs/PR, 58% severe). That overstated the +> evidence: **metaharness ADR-226 is a gold-scored null on a closely related +> design** — a read-only strong advisor produced **zero marginal resolves at +> 5.4× cost**, while being genuinely active (33 advisories, 3 vetoes). It was +> not considered when this section was written. +> +> The distinction that may preserve this design: ADR-226's advisor received the +> **full transcript**, whereas this reviewer receives **only the diff** — and the +> Cognition finding is precisely that reviewers do better *without* shared +> context. So ADR-226 does not refute §3.1, but it is the strongest nearby +> negative result and cannot be ignored. +> +> **Status is therefore downgraded from adopted to gated.** The reviewer must +> show marginal lift over a no-reviewer control on the same instances before it +> reaches the default path, and ADR-226's configuration is the specific null it +> must beat. §3.2 below is unaffected — ADR-226 independently corroborates it. + ### 3.2 Read-only context-gatherer Explores, reads, greps; returns a summary string. No shared mutable state, no @@ -104,6 +122,11 @@ Measured (SWE-Edit, Viewer + Editor split): **+2.1 pp resolve, −17.9% cost, putting GPT-5 in that slot gave **+0.4 pp at 5.8× cost.** Model tiering per subagent role is part of the design, not an optimization. +This is now corroborated internally: metaharness ADR-226 measured **zero +marginal lift at 5.4× cost** for a frontier model in a read-only slot. Two +independent measurements, near-identical cost multiple — treat "no expensive +model in a read-only slot" as established, not provisional. + This is also the cleanest lever on ADR-274 §5 failure #1 (wasted-context accumulation): exploration output never enters the main window. diff --git a/docs/adr/ADR-278-rvagent-flywheel-adoption.md b/docs/adr/ADR-278-rvagent-flywheel-adoption.md new file mode 100644 index 0000000000..43e1b18bd0 --- /dev/null +++ b/docs/adr/ADR-278-rvagent-flywheel-adoption.md @@ -0,0 +1,205 @@ +--- +adr: 278 +title: "rvAgent Self-Learning: Adopt the metaharness Flywheel; Shift from Memory to Policy" +status: accepted +date: 2026-08-01 +authors: [Reuven Cohen] +project: "rvAgent Harness" +related: [ADR-271, ADR-273, ADR-275, ADR-276, ADR-277] +external: [metaharness ADR-226, metaharness ADR-228, metaharness ADR-236] +tags: [rvagent, self-learning, flywheel, gepa, policy-evolution, metaharness, promotion, sota] +--- + +# ADR-278 — rvAgent Self-Learning: Adopt the metaharness Flywheel; Shift from Memory to Policy + +## Status + +**Accepted.** Owner: Reuven Cohen. Project: rvAgent Harness. Date: 2026-08-01. + +Sources: `@metaharness/flywheel@0.1.7` (`/workspace/metaharness/packages/flywheel`), +metaharness ADR-226 / ADR-228 / ADR-236, and +`docs/research/rvagent-hermes-harness/04-sota-landscape.md`. + +## 1. Decision + +1. **Adopt `@metaharness/flywheel` as rvAgent's promotion engine.** Do not build + one. Do not port it to Rust. +2. **Shift self-learning investment from memory accumulation to policy + evolution.** ReasoningBank/SONA stays gated per ADR-276; GEPA-style policy + evolution becomes the primary self-learning mechanism. +3. **Adopt `noopRate` as a first-class score axis.** +4. **Contribute anytime-valid sequential testing back to the flywheel gate** + rather than only consuming it. +5. **Record the metaharness ADR-226 null as a binding constraint** on rvAgent's + subagent design (see ADR-275 amendment). + +## 2. Why not build our own + +ADR-276 §3.4 specified a promotion apparatus from scratch. `@metaharness/flywheel` +already implements it, and more rigorously. + +It is deliberately host-agnostic — a stated design rule forbids any host, model, +or benchmark from leaking into the package. Its entire vocabulary is +`Policy = Record`, `Score`, `PromotionEvidence`, and a +`PromotionRule`; everything rvAgent-specific enters through an injected +`Evaluator`. There is no adapter impedance to pay. + +What it already provides that ADR-276 was specifying: + +| ADR-276 requirement | Flywheel | +|---|---| +| Promotion gate | `meetsPromotionRule` — **frozen, conjunctive**, every clause load-bearing | +| Proof the gate did not move | `gateFingerprint()` — SHA-256 over the rule source | +| Transfer holdout | `HoldoutSuite` **plus** a frozen `AnchorSuite` never optimized against | +| Audit trail | Ed25519 `PromotionReceipt` + `verifyReceipt()` | +| Independent verification | `verifyReplayBundle()` — reviewer trusts the signature, not us | +| Compounding, not scattering | Lineage DAG re-basing on the promoted winner; `computeLiftCurve()` | + +The anchor deserves emphasis: ADR-276 §4 asked only for a transfer holdout. The +flywheel requires a candidate to clear a holdout **and** a frozen suite it is +never optimized against. That is a strictly stronger anti-Goodhart guard than we +specified. + +`verifyReplayBundle()` is also, in substance, the replay-verification story +ADR-277 §3.2 positions on — working, in JS, today. + +## 3. The `noopRate` clause + +The default gate's second clause requires the no-op rate to **strictly** improve: + +> a policy earns a promotion by making the executor COMMIT more, not just score higher + +This is non-obvious and load-bearing. A policy that raises the primary metric +while leaving the executor more likely to end empty has not improved the agent; +it has found a scoring artifact. + +**Adopt this axis in rvAgent's own scoring.** It pairs naturally with ADR-273's +reliability framing — "never end empty" is a reliability property, and the ++54.3-point patch-application result is the same phenomenon measured a different +way. + +## 4. Memory versus policy — the reframe + +These are different objects with different evidence: + +| | Object | 2026 evidence | +|---|---|---| +| Flywheel / GEPA | **Policy text** — a genome of named string levers | Positive; GEPA is the best-evidenced optimizer in the sweep | +| ReasoningBank / SONA | **Episodic memory** — accumulated trajectories | Negative; inverted-U, confound-sized gains, self-memory underperforming plain retrieval | + +RuVector's self-learning weight currently sits on the memory side, which is the +side the evidence argues against. **Move the weight to policy evolution.** + +metaharness ADR-228 reaches the same conclusion from its own measurements: +redirect strong-model judgment *offline into the executor's standing operating +policy* rather than injecting it as runtime advice. It notes GEPA's candidate is +a `dict[str,str]` of named text components, matching the flywheel's `Policy` +exactly — the same shape rvAgent would supply. + +ADR-276 is not repealed. Its gating, trust tiers, and inverted-U regression +metric remain the conditions under which memory may ever be enabled. This ADR +changes where *new* effort goes. + +## 5. Two nulls we must respect + +### 5.1 ADR-226 — the read-only advisor is dead + +A frontier read-only advisor over a cheap executor produced **zero marginal +gold-scored resolves at 5.4× cost**. The advisor was genuinely active — 33 +advisories and 3 vetoes across the slice — not silently disabled. The track was +killed. + +This **independently corroborates** the figure in ADR-275 §3.2 from the public +literature (a frontier model in the read-only slot: +0.4 pp at 5.8× cost). Two +independent measurements, near-identical cost multiple. Treat the conclusion as +established rather than provisional: **do not put an expensive model in a +read-only slot.** + +It also constrains ADR-275 §3.1 — see §7. + +### 5.2 ADR-236 — a promotion engine cannot rescue a weak loop + +The flywheel mechanism was proven end-to-end on real SWE-bench and still +produced **no compounding lift, because the base solver was too weak.** Recorded +as an honest null rather than buried. + +**Consequence for sequencing:** adopting the flywheel does not shorten ADR-273. +The reliability floor comes first; the flywheel amplifies a loop that already +works and does nothing for one that does not. This validates the ordering +ADR-273 already set. + +## 6. What we contribute back + +The flywheel's gate is a **single-shot** conjunctive comparison. Running many +generations against the same holdout is uncontrolled multiple testing — the +regime where PACE (arXiv 2606.08106) measured **30–42% false commits**, and 13–21 +spurious modifications even when no true gains existed. + +The conjunctive gate plus frozen anchor mitigates this with multiple hurdles, +which is real but is not anytime-valid. **Frozen conjunctive gate ∧ anytime-valid +sequential test is strictly stronger than either**, and PACE reported ~18% lower +evaluation cost as a side effect. + +This is an upstream contribution to `ruvnet/metaharness`, offered as an optional +`PromotionRule` plus a sequential-evidence accumulator — not a change to the +default gate, whose stability is itself the product. + +## 7. Amendment to ADR-275 + +ADR-275 §3.1 adopted a fresh-context reviewer subagent on the strength of +Cognition's production data (~2 bugs/PR, 58% severe). ADR-226 is the closest +*measured null* to that design and was not considered when §3.1 was written. + +The distinction that may preserve it: **ADR-226's advisor received the full +transcript**, whereas the fresh-context reviewer receives only the diff — and the +Cognition finding is specifically that reviewers perform *better* without shared +context. ADR-226 therefore does not refute §3.1, but it is the strongest nearby +negative result. + +**ADR-275 §3.1 is downgraded from adopted to gated.** The reviewer must +demonstrate marginal lift over a no-reviewer control on the same instances +before it goes on the default path, and ADR-226's design is the specific null it +must beat. §3.2 (read-only gatherer on a cheap model) is unaffected and is in +fact strengthened by §5.1. + +## 8. Integration + +**No Rust port.** Promotion is offline; the flywheel is not on the hot path. + +- Run it in CI at the existing ruflo/metaharness seam (roadmap Phase 3). +- rvAgent supplies an `Evaluator` mapping a run onto the four `Score` axes, and + a `Proposer` for the mutation seam. +- rvAgent's policy genome is the natural `Policy`: system-prompt components, + compaction rubric (ADR-274 §3.4), `loop_repeat_threshold`, masking + `keep_last_observations`, tool-surface composition, per-role model tiers. +- metaharness ships `crates/kernel-napi` as the in-process bridge pattern if we + later need it. We do not need it now. + +**Gap this closes:** ADR-271 (`metaharness-darwin-sona-self-improvement`) does +not reference the flywheel at all. This seam was previously unrecorded on our +side. + +## 9. Consequences + +**Positive.** Deletes the largest unbuilt subsystem in ADR-276 — we consume a +maintained engine instead. Inherits an audit and replay story that already +exists. Moves self-learning onto the side of the evidence. Turns a +one-directional dependency into a two-way exchange (§6). + +**Negative.** A cross-repo dependency on a package at `0.1.x`. Mitigated because +the flywheel is thin, runtime-dependency-free (Node `crypto` only), fully typed, +and — being offline — a version pin is low-risk. + +**Risk.** rvAgent's four `Score` axes must be projected honestly. `primary`, +`noopRate`, `costPerWin`, and `regressed` are where all host meaning lands, and +a dishonest projection defeats every downstream guarantee. The Evaluator is the +trust boundary. + +## 10. Implementation order + +1. Record the ADR-275 §3.1 downgrade (§7) — documentation only +2. Add `noopRate` to rvAgent's score axes (§3) +3. Define the rvAgent policy genome (§8) +4. Evaluator mapping a headless run onto the four axes +5. Wire `runFlywheelGenerations` into CI at the ruflo seam +6. Upstream the sequential-testing `PromotionRule` to metaharness (§6) diff --git a/docs/research/rvagent-hermes-harness/04-sota-landscape.md b/docs/research/rvagent-hermes-harness/04-sota-landscape.md index dfe3d8e7a1..f955f7ac20 100644 --- a/docs/research/rvagent-hermes-harness/04-sota-landscape.md +++ b/docs/research/rvagent-hermes-harness/04-sota-landscape.md @@ -358,7 +358,13 @@ subagent boundary as *a tool that spawns an isolated context and returns a String*, not as peers with a message bus. That buys nearly all demonstrated upside at a fraction of the complexity. -### Phase 2 — gate the learning loop +**Amended (ADR-278 §7).** The reviewer is **gated, not adopted**. metaharness +ADR-226 is a gold-scored null on a closely related design. It gave its advisor +the *full transcript* where this reviewer sees *only the diff*, so it does not +refute the pattern — but it is the specific null the reviewer must beat before +reaching the default path. The gatherer is unaffected and corroborated. + +### Phase 2 — gate the learning loop, and shift memory → policy "SONA on the default path" contradicts §5. Move behind a feature gate with the measurement apparatus (paired lift, previously-solved regression rate, control @@ -367,6 +373,27 @@ arm) as the *precondition* for enabling it, not a follow-up. The existing exit gate (≥30% token reduction on a repeated task suite) is well-formed — keep it, and add the control arm. +**Update (ADR-278).** The promotion apparatus does not need building: +`@metaharness/flywheel` already implements a frozen fingerprinted conjunctive +gate, holdout **plus** a never-optimized-against anchor, Ed25519 receipts, +independent replay verification, and a compounding lineage DAG. Adopt it. + +More consequentially, metaharness's own measurements reframe the target. +Self-learning splits into two objects with opposite evidence: **policy text** +(GEPA-style; positive) and **episodic memory** (ReasoningBank/SONA; negative, +per §5). RuVector's weight currently sits on the memory side. Move new effort to +policy evolution. + +Two internal nulls to respect: + +- **ADR-226** — a read-only frontier advisor produced *zero* marginal + gold-scored resolves at **5.4× cost** while genuinely firing (33 advisories, + 3 vetoes). Independently corroborates this document's +0.4 pp / 5.8× figure. +- **ADR-236** — the flywheel mechanism was proven end-to-end on real SWE-bench + and still produced no compounding lift, because the base solver was too weak. + **A promotion engine cannot rescue an unreliable loop**, which confirms the + reliability-floor-first ordering above. + ### Phase 3 — MCP spec migration is now urgent MCP **2026-07-28** landed days ago and is breaking: protocol-level sessions and From 9d11dbce8b2a767ac1290bc9cb95975e26e91377 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 22:26:12 +0000 Subject: [PATCH 12/25] feat(rvagent): environment bootstrap + compaction invariants (ADR-273, ADR-274) Environment bootstrap (ADR-273 3.5). A cheap factual workspace snapshot is injected into the system prompt before the loop starts, so the agent does not spend its first turns discovering cwd, project kind, test command, and branch. Filesystem-only -- no process spawning -- so it is safe to run unconditionally. Two deliberate restraints. build_ok is None unless a caller checks it, and an unchecked build renders nothing at all rather than reading as passing; a guess here would be worse than silence because the agent would trust it. And a truncated directory listing announces how many entries it dropped, since a partial list that looks complete invites the agent to conclude a file is absent. Compaction invariants (ADR-274 2.2). Safety constraints and task statements erode through successive compaction cycles with no failure signal -- a documented mechanism, not a jailbreak. InvariantSet re-emits a small set of rules byte-identical after each compaction; they are never inputs to a summarizer and never masked. Re-inserting an id replaces rather than accumulates, so an updated objective cannot leave a stale copy in force. The prompt composition lives in core as EnvironmentSnapshot::augment_prompt rather than at the CLI call site: rvagent-cli is a binary crate, so anything assembled there is unreachable from a test. This keeps the seam covered. Tests: 160 core unit tests (was 143), including unchecked-build silence, announced truncation, byte-stable invariant round-trips across repeated renders, and no-op augmentation on an empty workspace. 38 suites green, clippy clean. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-cli/src/app.rs | 10 +- crates/rvAgent/rvagent-core/Cargo.toml | 1 + crates/rvAgent/rvagent-core/src/bootstrap.rs | 333 ++++++++++++++++++ crates/rvAgent/rvagent-core/src/invariants.rs | 184 ++++++++++ crates/rvAgent/rvagent-core/src/lib.rs | 2 + 5 files changed, 529 insertions(+), 1 deletion(-) create mode 100644 crates/rvAgent/rvagent-core/src/bootstrap.rs create mode 100644 crates/rvAgent/rvagent-core/src/invariants.rs diff --git a/crates/rvAgent/rvagent-cli/src/app.rs b/crates/rvAgent/rvagent-cli/src/app.rs index 47d513f97b..197d72cdca 100644 --- a/crates/rvAgent/rvagent-cli/src/app.rs +++ b/crates/rvAgent/rvagent-cli/src/app.rs @@ -14,6 +14,7 @@ use rvagent_core::config::{BackendConfig, MiddlewareConfig, RvAgentConfig, Secur use rvagent_core::graph::{AgentGraph, ToolExecutor}; use rvagent_core::messages::{Message, ToolCall as CoreToolCall}; use rvagent_core::models::{resolve_model, ChatModel, ToolDefinition}; +use rvagent_core::bootstrap::EnvironmentSnapshot; use rvagent_core::prompt::BASE_AGENT_PROMPT; use rvagent_core::state::AgentState; @@ -247,11 +248,18 @@ impl App { None => Session::new(model), }; + // Environment bootstrap (ADR-273 §3.5): hand the agent the workspace + // facts up front so it does not spend its first turns discovering + // them. Filesystem-only, so this costs nothing measurable. + let snapshot = + EnvironmentSnapshot::collect(cwd, &rvagent_core::bootstrap::BootstrapConfig::default()); + let system_prompt = snapshot.augment_prompt(BASE_AGENT_PROMPT); + Ok(Self { config, session, cwd: cwd.to_path_buf(), - system_prompt: BASE_AGENT_PROMPT.to_string(), + system_prompt, mcp_registry: McpRegistry::new(), }) } diff --git a/crates/rvAgent/rvagent-core/Cargo.toml b/crates/rvAgent/rvagent-core/Cargo.toml index b30e589600..365334ddcd 100644 --- a/crates/rvAgent/rvagent-core/Cargo.toml +++ b/crates/rvAgent/rvagent-core/Cargo.toml @@ -24,6 +24,7 @@ sha3 = "0.10" rand = "0.8" [dev-dependencies] +tempfile = "3.14" criterion = { workspace = true } tokio = { workspace = true, features = ["test-util"] } proptest = { workspace = true } diff --git a/crates/rvAgent/rvagent-core/src/bootstrap.rs b/crates/rvAgent/rvagent-core/src/bootstrap.rs new file mode 100644 index 0000000000..9281d20303 --- /dev/null +++ b/crates/rvAgent/rvagent-core/src/bootstrap.rs @@ -0,0 +1,333 @@ +//! Environment bootstrap — a workspace snapshot injected before the loop +//! starts (ADR-273 §3.5). +//! +//! Without it the agent spends its first turns discovering what it is looking +//! at: listing the directory, finding the test command, checking whether the +//! build is already broken. Those turns cost tokens, fill context, and produce +//! nothing the harness could not have supplied for free. +//! +//! The snapshot is deliberately small and factual. It is *not* a repo map or a +//! summary — those are lossy and expensive. It reports only what is cheap to +//! observe and expensive for the agent to discover. + +use std::fmt::Write as _; +use std::path::{Path, PathBuf}; + +/// A cheap, factual snapshot of the workspace. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct EnvironmentSnapshot { + /// Absolute working directory. + pub cwd: PathBuf, + /// Top-level entries, directories marked with a trailing `/`. + pub entries: Vec, + /// Detected project kind, e.g. "Rust (cargo)". + pub project_kind: Option, + /// Toolchain versions, as `(tool, version)`. + pub toolchain: Vec<(String, String)>, + /// The command to run tests, when it can be determined. + pub test_command: Option, + /// Whether the tree currently builds, when cheaply checkable. + /// + /// `None` means "not checked" — never guessed. Reporting a guess here + /// would be worse than reporting nothing, since the agent would trust it. + pub build_ok: Option, + /// Current VCS branch, when the workspace is a repository. + pub vcs_branch: Option, +} + +/// How much work the snapshot is allowed to do. +#[derive(Debug, Clone)] +pub struct BootstrapConfig { + /// Maximum top-level entries to list. + pub max_entries: usize, + /// Whether to include hidden entries. + pub include_hidden: bool, +} + +impl Default for BootstrapConfig { + fn default() -> Self { + Self { + max_entries: 50, + include_hidden: false, + } + } +} + +impl EnvironmentSnapshot { + /// Collect a snapshot of `root`. + /// + /// Filesystem-only: nothing here spawns a process, so it is fast and safe + /// to run unconditionally. `build_ok` is left `None` — a build check is a + /// caller decision, since it costs real time. + pub fn collect(root: &Path, config: &BootstrapConfig) -> Self { + let entries = list_entries(root, config); + let project_kind = detect_project_kind(&entries); + let test_command = detect_test_command(&entries); + let vcs_branch = detect_branch(root); + + Self { + cwd: root.to_path_buf(), + entries, + project_kind, + toolchain: Vec::new(), + test_command, + build_ok: None, + vcs_branch, + } + } + + /// Render as a prompt section. + /// + /// Returns `None` when there is nothing worth saying, so an empty or + /// unreadable workspace does not inject a misleading stub. + pub fn to_prompt_section(&self) -> Option { + if self.entries.is_empty() && self.project_kind.is_none() { + return None; + } + + let mut out = String::from("\n"); + let _ = writeln!(out, "cwd: {}", self.cwd.display()); + + if let Some(kind) = &self.project_kind { + let _ = writeln!(out, "project: {kind}"); + } + if let Some(branch) = &self.vcs_branch { + let _ = writeln!(out, "branch: {branch}"); + } + if let Some(cmd) = &self.test_command { + let _ = writeln!(out, "tests: {cmd}"); + } + match self.build_ok { + Some(true) => { + let _ = writeln!(out, "build: passing"); + } + Some(false) => { + let _ = writeln!(out, "build: FAILING before any of your changes"); + } + // Silence is correct here: an unchecked build must not read as passing. + None => {} + } + for (tool, version) in &self.toolchain { + let _ = writeln!(out, "{tool}: {version}"); + } + if !self.entries.is_empty() { + let _ = writeln!(out, "contents: {}", self.entries.join(", ")); + } + out.push_str(""); + Some(out) + } + + /// Append the environment section to a base system prompt. + /// + /// Returns `base` unchanged when there is nothing to report, so callers + /// need no conditional of their own. Kept here rather than at the call site + /// so the composition is covered by tests — the CLI is a binary crate and + /// anything assembled there is unreachable from a test. + pub fn augment_prompt(&self, base: &str) -> String { + match self.to_prompt_section() { + Some(section) => format!("{base}\n\n{section}"), + None => base.to_string(), + } + } +} + +fn list_entries(root: &Path, config: &BootstrapConfig) -> Vec { + let Ok(read) = std::fs::read_dir(root) else { + return Vec::new(); + }; + let mut names: Vec = read + .flatten() + .filter_map(|e| { + let name = e.file_name().to_string_lossy().into_owned(); + if !config.include_hidden && name.starts_with('.') { + return None; + } + let is_dir = e.file_type().map(|t| t.is_dir()).unwrap_or(false); + Some(if is_dir { format!("{name}/") } else { name }) + }) + .collect(); + names.sort(); + // Say so rather than silently showing a partial list: a truncated listing + // that looks complete invites the agent to conclude a file is absent. + if names.len() > config.max_entries { + let hidden = names.len() - config.max_entries; + names.truncate(config.max_entries); + names.push(format!("… and {hidden} more")); + } + names +} + +fn has(entries: &[String], name: &str) -> bool { + entries.iter().any(|e| e == name) +} + +fn detect_project_kind(entries: &[String]) -> Option { + if has(entries, "Cargo.toml") { + Some("Rust (cargo)".into()) + } else if has(entries, "package.json") { + Some("Node (npm)".into()) + } else if has(entries, "pyproject.toml") || has(entries, "setup.py") { + Some("Python".into()) + } else if has(entries, "go.mod") { + Some("Go".into()) + } else { + None + } +} + +fn detect_test_command(entries: &[String]) -> Option { + if has(entries, "Cargo.toml") { + Some("cargo test".into()) + } else if has(entries, "package.json") { + Some("npm test".into()) + } else if has(entries, "pyproject.toml") || has(entries, "setup.py") { + Some("pytest".into()) + } else if has(entries, "go.mod") { + Some("go test ./...".into()) + } else { + None + } +} + +/// Read the current branch from `.git/HEAD` without shelling out. +fn detect_branch(root: &Path) -> Option { + let head = std::fs::read_to_string(root.join(".git").join("HEAD")).ok()?; + let head = head.trim(); + head.strip_prefix("ref: refs/heads/") + .map(str::to_string) + // A detached HEAD is a raw sha; report a short form rather than nothing. + .or_else(|| Some(head.chars().take(12).collect())) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn write(dir: &Path, name: &str, body: &str) { + std::fs::write(dir.join(name), body).unwrap(); + } + + #[test] + fn detects_a_rust_project_and_its_test_command() { + let dir = tempfile::tempdir().unwrap(); + write(dir.path(), "Cargo.toml", "[package]"); + std::fs::create_dir(dir.path().join("src")).unwrap(); + + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + assert_eq!(snap.project_kind.as_deref(), Some("Rust (cargo)")); + assert_eq!(snap.test_command.as_deref(), Some("cargo test")); + assert!(snap.entries.contains(&"src/".to_string())); + assert!(snap.entries.contains(&"Cargo.toml".to_string())); + } + + #[test] + fn renders_a_prompt_section() { + let dir = tempfile::tempdir().unwrap(); + write(dir.path(), "package.json", "{}"); + + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + let section = snap.to_prompt_section().unwrap(); + assert!(section.starts_with("")); + assert!(section.ends_with("")); + assert!(section.contains("project: Node (npm)")); + assert!(section.contains("tests: npm test")); + } + + #[test] + fn unchecked_build_is_silent_not_passing() { + let dir = tempfile::tempdir().unwrap(); + write(dir.path(), "Cargo.toml", "[package]"); + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + assert_eq!(snap.build_ok, None); + let section = snap.to_prompt_section().unwrap(); + assert!( + !section.contains("build:"), + "an unchecked build must not be reported at all: {section}" + ); + } + + #[test] + fn failing_build_is_stated_plainly() { + let dir = tempfile::tempdir().unwrap(); + write(dir.path(), "Cargo.toml", "[package]"); + let mut snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + snap.build_ok = Some(false); + let section = snap.to_prompt_section().unwrap(); + assert!(section.contains("build: FAILING before any of your changes")); + } + + #[test] + fn empty_workspace_yields_no_section() { + let dir = tempfile::tempdir().unwrap(); + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + assert!(snap.to_prompt_section().is_none()); + } + + #[test] + fn truncation_is_announced() { + let dir = tempfile::tempdir().unwrap(); + for i in 0..30 { + write(dir.path(), &format!("f{i:02}.txt"), ""); + } + let config = BootstrapConfig { + max_entries: 10, + ..BootstrapConfig::default() + }; + let snap = EnvironmentSnapshot::collect(dir.path(), &config); + assert_eq!(snap.entries.len(), 11); + assert!(snap.entries.last().unwrap().contains("and 20 more")); + } + + #[test] + fn hidden_entries_are_excluded_by_default() { + let dir = tempfile::tempdir().unwrap(); + write(dir.path(), ".secret", ""); + write(dir.path(), "visible.txt", ""); + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + assert!(!snap.entries.iter().any(|e| e.starts_with('.'))); + assert!(snap.entries.contains(&"visible.txt".to_string())); + } + + #[test] + fn reads_the_branch_from_git_head() { + let dir = tempfile::tempdir().unwrap(); + std::fs::create_dir(dir.path().join(".git")).unwrap(); + write( + &dir.path().join(".git"), + "HEAD", + "ref: refs/heads/feature/x\n", + ); + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + assert_eq!(snap.vcs_branch.as_deref(), Some("feature/x")); + } + + #[test] + fn augment_prompt_appends_the_section() { + let dir = tempfile::tempdir().unwrap(); + write(dir.path(), "Cargo.toml", "[package]"); + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + + let prompt = snap.augment_prompt("BASE PROMPT"); + assert!(prompt.starts_with("BASE PROMPT")); + assert!(prompt.contains("")); + assert!(prompt.contains("tests: cargo test")); + } + + #[test] + fn augment_prompt_is_a_noop_when_there_is_nothing_to_say() { + let dir = tempfile::tempdir().unwrap(); + let snap = EnvironmentSnapshot::collect(dir.path(), &BootstrapConfig::default()); + // No trailing whitespace, no empty stub — byte-identical to the input. + assert_eq!(snap.augment_prompt("BASE PROMPT"), "BASE PROMPT"); + } + + #[test] + fn unreadable_workspace_does_not_panic() { + let snap = EnvironmentSnapshot::collect( + Path::new("/nonexistent/path/xyz"), + &BootstrapConfig::default(), + ); + assert!(snap.entries.is_empty()); + assert!(snap.to_prompt_section().is_none()); + } +} diff --git a/crates/rvAgent/rvagent-core/src/invariants.rs b/crates/rvAgent/rvagent-core/src/invariants.rs new file mode 100644 index 0000000000..d7c39b592b --- /dev/null +++ b/crates/rvAgent/rvagent-core/src/invariants.rs @@ -0,0 +1,184 @@ +//! Invariants that survive compaction verbatim (ADR-274 §2.2). +//! +//! Safety constraints and task statements **erode through successive compaction +//! cycles with no failure signal** — a documented mechanism ("governance +//! decay"), not a jailbreak. Each summarization pass paraphrases a little more +//! away until a rule that was explicit at turn 1 is gone by turn 200, and +//! nothing in the transcript marks the moment it disappeared. +//! +//! The fix is cheap and absolute: a small set of invariants is re-emitted +//! **byte-identical** after every compaction. They are never inputs to a +//! summarizer, never masked (ADR-274 §3.1), and never paraphrased. +//! +//! Keep this set small. Everything here is paid for on every turn after a +//! compaction, and a bloated invariant set recreates the context pressure +//! compaction exists to relieve. + +use crate::messages::Message; + +/// A rule that must never be summarized away. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Invariant { + /// Short stable label, for debugging and dedup. + pub id: String, + /// The exact text to re-emit. Reproduced byte-for-byte. + pub text: String, +} + +impl Invariant { + pub fn new(id: impl Into, text: impl Into) -> Self { + Self { + id: id.into(), + text: text.into(), + } + } +} + +/// The set of invariants carried across compaction boundaries. +#[derive(Debug, Clone, Default)] +pub struct InvariantSet { + invariants: Vec, +} + +impl InvariantSet { + pub fn new() -> Self { + Self::default() + } + + /// Add an invariant. Re-adding an existing `id` replaces it, so a caller + /// cannot accidentally accumulate near-duplicate copies of a rule. + pub fn insert(&mut self, invariant: Invariant) { + match self.invariants.iter_mut().find(|i| i.id == invariant.id) { + Some(existing) => *existing = invariant, + None => self.invariants.push(invariant), + } + } + + pub fn is_empty(&self) -> bool { + self.invariants.is_empty() + } + + pub fn len(&self) -> usize { + self.invariants.len() + } + + pub fn iter(&self) -> impl Iterator { + self.invariants.iter() + } + + /// Render the block re-emitted after a compaction. + /// + /// Returns `None` when empty so callers never inject an empty stub. + pub fn render(&self) -> Option { + if self.invariants.is_empty() { + return None; + } + let mut out = String::from( + "\nThese were established earlier and remain in force. \ + They are reproduced exactly and are not a summary.\n", + ); + for inv in &self.invariants { + out.push('\n'); + out.push_str(&inv.text); + out.push('\n'); + } + out.push_str(""); + Some(out) + } + + /// Append the invariant block to a compacted history. + /// + /// Call this immediately after any operation that drops or rewrites + /// history. Appending at the end — rather than restoring the original + /// position — is deliberate: recency is what survives a long context, and + /// the whole point is that these rules must not be the first thing lost. + pub fn reinject(&self, mut messages: Vec) -> Vec { + if let Some(block) = self.render() { + messages.push(Message::system(block)); + } + messages + } +} + +impl FromIterator for InvariantSet { + fn from_iter>(iter: T) -> Self { + let mut set = Self::new(); + for inv in iter { + set.insert(inv); + } + set + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn set() -> InvariantSet { + [ + Invariant::new("task", "TASK: Fix the failing parser test."), + Invariant::new("safety", "Never force-push to the default branch."), + ] + .into_iter() + .collect() + } + + #[test] + fn renders_text_byte_for_byte() { + let block = set().render().unwrap(); + // The exact strings must appear untransformed — no wrapping, no + // reflowing, no paraphrase. + assert!(block.contains("TASK: Fix the failing parser test.")); + assert!(block.contains("Never force-push to the default branch.")); + } + + #[test] + fn survives_a_round_trip_unchanged() { + let original = set(); + let block = original.render().unwrap(); + // Simulate several compaction cycles: each one re-renders from the + // same source, so the text can never drift. + for _ in 0..10 { + assert_eq!(original.render().unwrap(), block); + } + } + + #[test] + fn reinject_appends_a_system_message() { + let compacted = vec![Message::system("summary of earlier work")]; + let out = set().reinject(compacted); + assert_eq!(out.len(), 2); + match out.last().unwrap() { + Message::System(s) => { + assert!(s.content.contains("TASK: Fix the failing parser test.")); + assert!(s.content.contains("not a summary")); + } + other => panic!("expected a system message, got {other:?}"), + } + } + + #[test] + fn empty_set_injects_nothing() { + let empty = InvariantSet::new(); + assert!(empty.render().is_none()); + let msgs = vec![Message::human("hi")]; + assert_eq!(empty.reinject(msgs.clone()), msgs); + } + + #[test] + fn reinserting_an_id_replaces_rather_than_duplicates() { + let mut s = set(); + s.insert(Invariant::new("task", "TASK: Updated objective.")); + assert_eq!(s.len(), 2, "an updated rule must not accumulate copies"); + let block = s.render().unwrap(); + assert!(block.contains("TASK: Updated objective.")); + assert!(!block.contains("Fix the failing parser test")); + } + + #[test] + fn insertion_order_is_stable() { + let s = set(); + let ids: Vec<&str> = s.iter().map(|i| i.id.as_str()).collect(); + assert_eq!(ids, vec!["task", "safety"]); + } +} diff --git a/crates/rvAgent/rvagent-core/src/lib.rs b/crates/rvAgent/rvagent-core/src/lib.rs index e198e70191..145707b3e8 100644 --- a/crates/rvAgent/rvagent-core/src/lib.rs +++ b/crates/rvAgent/rvagent-core/src/lib.rs @@ -20,10 +20,12 @@ pub mod agi_container; pub mod arena; pub mod budget; +pub mod bootstrap; pub mod config; pub mod cow_state; pub mod error; pub mod graph; +pub mod invariants; pub mod masking; pub mod messages; pub mod metrics; From e00f083d993d8a667bf4dadd7df898968482c917 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 22:26:52 +0000 Subject: [PATCH 13/25] chore(rvagent): record tempfile dev-dependency in Cargo.lock Belongs with 9d11dbc, which added tempfile as a dev-dependency of rvagent-core for the environment-bootstrap tests. The lockfile update was missed because that commit staged only crates/rvAgent. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- Cargo.lock | 1 + 1 file changed, 1 insertion(+) diff --git a/Cargo.lock b/Cargo.lock index 854cd017f6..da0277f162 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -11311,6 +11311,7 @@ dependencies = [ "serde_json", "sha3", "smallvec 1.15.2", + "tempfile", "thiserror 2.0.18", "tokio", "tracing", From 2ca91c56bd15c2dacf1f98b83d6c640a04e3fe71 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 22:32:41 +0000 Subject: [PATCH 14/25] feat(rvagent): verify-after-write + addressable recall (ADR-273, ADR-274) Verify-after-write (ADR-273 3.1). LocalFsBackend now reads back every write and edit, comparing bytes against what was requested. std::fs::write returning Ok means the syscalls succeeded, not that the bytes are on disk and readable -- a full filesystem, a quota, a racing writer, or an unusual mount can all produce a successful-looking write whose content differs. Reporting success there is the worst outcome, because the agent proceeds believing the edit landed and every later step rests on a false premise. That is the failure class that dominates harness ablations. Costs one read per write, negligible against a model round trip. Both failure messages tell the model to re-read the file rather than just reporting that something went wrong. Addressable recall (ADR-274 3.2). Masking is now non-destructive in practice, not just in principle: a `recall` tool dereferences an elided observation by the recall id carried in its placeholder. It is served by the loop from the full message log rather than by a ToolExecutor -- executors do not have the log, and reserving the name in the loop means a workspace tool cannot shadow it. Advertised exactly when masking is active, since a masked observation the model cannot dereference is worse than no masking at all. Tests: 14 e2e (2 new -- recall round-trips content that was elided from the model's view, and an unknown recall id is actionable rather than fatal), plus backend coverage for verified write/edit, content mismatch, unreadable target, and empty/multibyte content. 38 suites green, clippy clean. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-core/src/graph.rs | 37 +++++- crates/rvAgent/rvagent-core/src/masking.rs | 64 +++++++++++ crates/rvAgent/rvagent-tools/src/local.rs | 106 +++++++++++++++++- .../rvagent-tools/tests/e2e_tool_calling.rs | 91 +++++++++++++++ 4 files changed, 286 insertions(+), 12 deletions(-) diff --git a/crates/rvAgent/rvagent-core/src/graph.rs b/crates/rvAgent/rvagent-core/src/graph.rs index 3dc5a3e56f..5f09fd531c 100644 --- a/crates/rvAgent/rvagent-core/src/graph.rs +++ b/crates/rvAgent/rvagent-core/src/graph.rs @@ -9,7 +9,10 @@ use serde::{Deserialize, Serialize}; use tracing::{debug, info, instrument, warn}; use crate::error::{Result, RvAgentError}; -use crate::masking::{mask_observations, truncate_tool_result, MaskConfig}; +use crate::masking::{ + mask_observations, recall, recall_definition, recall_id_from_args, truncate_tool_result, + MaskConfig, RECALL_TOOL, +}; use crate::messages::{Message, ToolCall}; use crate::models::{ChatModel, ToolDefinition}; use crate::parallel::parallel_execute_limited; @@ -246,7 +249,12 @@ impl AgentGraph { let mut iterations: u32 = 0; // Tool schemas advertised to the model on every completion. Without // these the model can never emit a tool call. - let tool_definitions = self.tool_executor.definitions(); + let mut tool_definitions = self.tool_executor.definitions(); + // Masked observations are useless without a way to dereference them, + // so recall is advertised exactly when masking is active (ADR-274 §3.2). + if self.config.mask.masking_enabled() { + tool_definitions.push(recall_definition()); + } // Cumulative token usage across the loop, aggregated from per-message // usage metadata attached by provider backends. let mut total_input_tokens: u64 = 0; @@ -321,19 +329,36 @@ impl AgentGraph { None }); } - // Only calls that cleared loop detection reach the executor. + // Recall is served by the loop from the full log; it never + // reaches the executor, so a workspace tool cannot shadow it. + let mut handled: std::collections::HashMap = + std::collections::HashMap::new(); + for (tc, refused) in tool_calls.iter().zip(&looping) { + if refused.is_none() && tc.name == RECALL_TOOL { + let content = match recall_id_from_args(&tc.args) { + Ok(id) => recall(&state.messages, id), + Err(e) => e, + }; + handled.insert(tc.id.clone(), content); + } + } + + // Only calls that cleared loop detection and were not + // handled in-loop reach the executor. let dispatch: Vec = tool_calls .iter() .zip(&looping) - .filter(|(_, refused)| refused.is_none()) + .filter(|(tc, refused)| { + refused.is_none() && !handled.contains_key(&tc.id) + }) .map(|(tc, _)| tc.clone()) .collect(); // Tool failures are fed back to the model as tool results // rather than aborting the loop — the model must see the // error to recover from it (execution alignment). - let mut executed: std::collections::HashMap = - std::collections::HashMap::with_capacity(dispatch.len()); + let mut executed: std::collections::HashMap = handled; + executed.reserve(dispatch.len()); if self.config.parallel_tools && dispatch.len() > 1 { // True parallel execution (ADR-103 A2): tasks are // spawned onto the runtime with bounded concurrency; diff --git a/crates/rvAgent/rvagent-core/src/masking.rs b/crates/rvAgent/rvagent-core/src/masking.rs index 4e3b24cb5d..5133b7f7cb 100644 --- a/crates/rvAgent/rvagent-core/src/masking.rs +++ b/crates/rvAgent/rvagent-core/src/masking.rs @@ -127,6 +127,70 @@ pub fn mask_observations(messages: &[Message], config: &MaskConfig) -> Vec crate::models::ToolDefinition { + crate::models::ToolDefinition { + name: RECALL_TOOL.to_string(), + description: + "Retrieve the full content of an earlier tool result that was elided from the \ + conversation. Pass the recall id shown in the placeholder, e.g. \ + '[read_file output elided: 2847 bytes, recall id tc7]' -> recall_id \"tc7\"." + .to_string(), + input_schema: serde_json::json!({ + "type": "object", + "properties": { + "recall_id": { + "type": "string", + "description": "The recall id from an elided observation placeholder." + } + }, + "required": ["recall_id"] + }), + } +} + +/// Dereference a recall id against the full message log. +/// +/// `messages` must be the complete log, never a masked projection — recalling +/// from a masked view would return the placeholder rather than the content. +pub fn recall(messages: &[Message], recall_id: &str) -> String { + for msg in messages { + if let Message::Tool(t) = msg { + if t.tool_call_id == recall_id { + return t.content.clone(); + } + } + } + // Actionable rather than opaque: tell the model where valid ids come from. + format!( + "Error: no observation found with recall id '{recall_id}'. Recall ids appear \ + in elided-output placeholders in this conversation; they are not tool names \ + or file paths." + ) +} + +/// Extract the `recall_id` argument from a recall tool call. +pub fn recall_id_from_args(args: &serde_json::Value) -> Result<&str, String> { + args.get("recall_id") + .and_then(|v| v.as_str()) + .ok_or_else(|| { + "Error: recall requires a string 'recall_id' argument, taken from an \ + elided-output placeholder." + .to_string() + }) +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/rvAgent/rvagent-tools/src/local.rs b/crates/rvAgent/rvagent-tools/src/local.rs index cea7b1a29f..9ac8b4b6dd 100644 --- a/crates/rvAgent/rvagent-tools/src/local.rs +++ b/crates/rvAgent/rvagent-tools/src/local.rs @@ -192,11 +192,17 @@ impl Backend for LocalFsBackend { }; } } - match std::fs::write(&target, content) { - Ok(_) => WriteResult::default(), - Err(e) => WriteResult { + if let Err(e) = std::fs::write(&target, content) { + return WriteResult { error: Some(format!("write '{}': {}", target.display(), e)), ..Default::default() + }; + } + match verify_written(&target, content) { + Ok(()) => WriteResult::default(), + Err(e) => WriteResult { + error: Some(e), + ..Default::default() }, } } @@ -246,14 +252,20 @@ impl Backend for LocalFsBackend { } else { content.replacen(old_string, new_string, 1) }; - match std::fs::write(&target, &new_content) { - Ok(_) => WriteResult { + if let Err(e) = std::fs::write(&target, &new_content) { + return WriteResult { + error: Some(format!("write '{}': {}", target.display(), e)), + ..Default::default() + }; + } + match verify_written(&target, &new_content) { + Ok(()) => WriteResult { error: None, occurrences: Some(if replace_all { count } else { 1 }), ..Default::default() }, Err(e) => WriteResult { - error: Some(format!("write '{}': {}", target.display(), e)), + error: Some(e), ..Default::default() }, } @@ -382,6 +394,35 @@ impl Backend for LocalFsBackend { } } +/// Confirm a write actually landed, by reading the file back (ADR-273 §3.1). +/// +/// `std::fs::write` returning `Ok` means the syscalls succeeded, not that the +/// bytes are on disk and readable: a full filesystem, a quota, a racing writer, +/// or an unusual mount can all produce a successful-looking write whose content +/// differs. Reporting success in that case is the worst outcome, because the +/// agent proceeds believing the edit is applied and every later step is built +/// on a false premise — the exact failure mode that dominates harness ablations. +/// +/// Costs one read per write, which is negligible against a model round trip. +fn verify_written(target: &Path, expected: &str) -> Result<(), String> { + match std::fs::read(target) { + Ok(actual) if actual == expected.as_bytes() => Ok(()), + Ok(actual) => Err(format!( + "Error: write to '{}' did not verify — expected {} bytes, file now holds {}. \ + The file may have been modified concurrently or the write was truncated. \ + Re-read the file before making further changes.", + target.display(), + expected.len(), + actual.len() + )), + Err(e) => Err(format!( + "Error: write to '{}' could not be verified: {e}. \ + Treat the file's contents as unknown and re-read it.", + target.display() + )), + } +} + /// Recursively collect files matching a name suffix (simple glob substitute). fn collect_glob_matches(dir: &Path, suffix: &str, results: &mut Vec) { let entries = match std::fs::read_dir(dir) { @@ -514,6 +555,59 @@ mod tests { assert_eq!(backend.read("nested/dir/new.txt", 0, 10).unwrap(), "content"); } + #[test] + fn write_is_verified_by_reading_back() { + let (dir, backend) = backend(); + let result = backend.write("verified.txt", "exact contents"); + assert!(result.error.is_none()); + // The verification path must accept a correct write, not just reject + // bad ones — otherwise it would be a permanent false alarm. + assert_eq!( + std::fs::read_to_string(dir.path().join("verified.txt")).unwrap(), + "exact contents" + ); + } + + #[test] + fn edit_is_verified_by_reading_back() { + let (dir, backend) = backend(); + std::fs::write(dir.path().join("e.txt"), "alpha beta").unwrap(); + let result = backend.edit("e.txt", "alpha", "gamma", false); + assert!(result.error.is_none(), "unexpected: {:?}", result.error); + assert_eq!(result.occurrences, Some(1)); + assert_eq!( + std::fs::read_to_string(dir.path().join("e.txt")).unwrap(), + "gamma beta" + ); + } + + #[test] + fn verification_reports_a_content_mismatch() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("x.txt"); + std::fs::write(&path, "actual").unwrap(); + // Simulate the case the check exists for: what landed differs from + // what was asked for. + let err = verify_written(&path, "expected something longer").unwrap_err(); + assert!(err.contains("did not verify"), "got: {err}"); + assert!(err.contains("Re-read the file"), "must be actionable: {err}"); + } + + #[test] + fn verification_reports_an_unreadable_file() { + let dir = tempfile::tempdir().unwrap(); + let err = verify_written(&dir.path().join("missing.txt"), "anything").unwrap_err(); + assert!(err.contains("could not be verified"), "got: {err}"); + assert!(err.contains("re-read"), "must be actionable: {err}"); + } + + #[test] + fn verification_handles_empty_and_multibyte_content() { + let (_dir, backend) = backend(); + assert!(backend.write("empty.txt", "").error.is_none()); + assert!(backend.write("utf8.txt", "héllo 🙂 wörld").error.is_none()); + } + #[test] fn dotdot_inside_root_still_works() { let (_dir, backend) = backend(); diff --git a/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs b/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs index 01ee7c9c63..d26e51d9ae 100644 --- a/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs +++ b/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs @@ -535,6 +535,97 @@ async fn old_observations_are_masked_before_reaching_the_model() { } } +#[tokio::test] +async fn recall_returns_the_full_content_of_a_masked_observation() { + let dir = tempfile::tempdir().unwrap(); + for i in 0..4 { + std::fs::write( + dir.path().join(format!("f{i}.txt")), + format!("SECRET-BODY-{i}"), + ) + .unwrap(); + } + + let mut responses: Vec = (0..4) + .map(|i| { + Message::ai_with_tools( + "", + vec![call( + &format!("t{i}"), + "read_file", + serde_json::json!({ "file_path": format!("f{i}.txt") }), + )], + ) + }) + .collect(); + // t0 has been masked out of the model's view by now; dereference it. + responses.push(Message::ai_with_tools( + "", + vec![call( + "r1", + "recall", + serde_json::json!({ "recall_id": "t0" }), + )], + )); + responses.push(Message::ai("done")); + + let model = Arc::new(ScriptedModel::new(responses)); + let config = GraphConfig { + parallel_tools: false, + mask: rvagent_core::masking::MaskConfig { + keep_last_observations: 2, + ..Default::default() + }, + ..GraphConfig::default() + }; + let graph = AgentGraph::with_config( + SharedModel(Arc::clone(&model)), + RealToolExecutor::new(dir.path()), + config, + ); + let state = graph.run(AgentState::new()).await.unwrap(); + + // The recall tool must have been advertised, or the model could not call it. + let seen = model.seen_tools.lock().unwrap(); + assert!( + seen.last().unwrap().iter().any(|d| d.name == "recall"), + "recall was not advertised while masking was active" + ); + drop(seen); + + // The recall result carries the content that was elided from the view. + let results = tool_results(&state); + let recalled = results.last().unwrap(); + assert!( + recalled.contains("SECRET-BODY-0"), + "recall did not return the elided content: {recalled}" + ); +} + +#[tokio::test] +async fn recall_with_an_unknown_id_is_actionable_not_fatal() { + let dir = tempfile::tempdir().unwrap(); + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "", + vec![call( + "r1", + "recall", + serde_json::json!({ "recall_id": "nope" }), + )], + ), + Message::ai("done"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + let state = graph.run(AgentState::new()).await.unwrap(); + + let results = tool_results(&state); + assert_eq!(results.len(), 1); + assert!(results[0].contains("no observation found")); + // Must tell the model where valid ids come from, not just that it failed. + assert!(results[0].contains("placeholders")); +} + #[tokio::test] async fn oversized_tool_output_is_capped() { let dir = tempfile::tempdir().unwrap(); From db411ee3d49f5b4bd52f0cdc8ddd52bffe56d044 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 22:40:33 +0000 Subject: [PATCH 15/25] feat(rvagent): demote summarization to opt-in fallback (ADR-274) The shipped default pipeline used LLM summarization as its compaction strategy, which ADR-274 decided against. Until now every run contradicted an accepted decision. Summarization is now gated behind PipelineConfig::enable_summarization, off by default, alongside the existing SONA/HNSW opt-ins. It is removed from the CLI's DEFAULT_MIDDLEWARE (11 -> 10). Observation masking in the agent loop is the default; measured comparisons put simple masking at or above LLM summarization on solve rate at roughly half the cost, and show summarization inflating trajectories 13-15% by destroying the stopping signals an agent uses to notice it has finished. The middleware is kept and still constructible by name, so the fallback is real rather than nominal. Its trigger drops from 0.85 to 0.75 of the token budget when enabled: context degrades well before the nominal limit, and compacting at 85% leaves too little headroom to be selective rather than desperate. Reading the implementation while demoting it confirms the direction. Its summarize() keeps only truncated Human messages and discards all AI reasoning and tool results outright -- considerably lossier than a summarizer that at least attempts to preserve decisions and unresolved issues. Two tests now assert the decision rather than the old shape: summarization is absent from DEFAULT_MIDDLEWARE, and enabling it adds exactly one middleware while remaining resolvable by name. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-cli/src/app.rs | 17 +++++- crates/rvAgent/rvagent-middleware/src/lib.rs | 64 ++++++++++++++++---- 2 files changed, 67 insertions(+), 14 deletions(-) diff --git a/crates/rvAgent/rvagent-cli/src/app.rs b/crates/rvAgent/rvagent-cli/src/app.rs index 197d72cdca..ddefe7f018 100644 --- a/crates/rvAgent/rvagent-cli/src/app.rs +++ b/crates/rvAgent/rvagent-cli/src/app.rs @@ -37,7 +37,9 @@ const DEFAULT_MIDDLEWARE: &[&str] = &[ "skills", "filesystem", "subagent", - "summarization", + // "summarization" removed (ADR-274): observation masking in the agent loop + // is the default compaction strategy. Still available opt-in via + // PipelineConfig::enable_summarization. "prompt_caching", "patch_tool_calls", "witness", @@ -491,7 +493,18 @@ mod tests { #[test] fn test_default_middleware_count() { - assert_eq!(DEFAULT_MIDDLEWARE.len(), 11); + // 10 since ADR-274 demoted summarization to opt-in. + assert_eq!(DEFAULT_MIDDLEWARE.len(), 10); + } + + #[test] + fn test_summarization_is_not_on_the_default_path() { + // The shipped default must match the decided strategy: masking in the + // agent loop, not LLM summarization. + assert!( + !DEFAULT_MIDDLEWARE.contains(&"summarization"), + "summarization is on the default path but ADR-274 decided against it" + ); } #[test] diff --git a/crates/rvAgent/rvagent-middleware/src/lib.rs b/crates/rvAgent/rvagent-middleware/src/lib.rs index c4e0163332..ee1b431852 100644 --- a/crates/rvAgent/rvagent-middleware/src/lib.rs +++ b/crates/rvAgent/rvagent-middleware/src/lib.rs @@ -173,6 +173,16 @@ pub struct PipelineConfig { pub enable_hnsw: bool, /// Enable Unicode security middleware (C7 - CVE mitigation). pub enable_unicode_security: bool, + /// Enable LLM summarization as the compaction strategy (ADR-274). + /// + /// **Off by default.** Observation masking in the agent loop is the default + /// strategy; measured comparisons put simple masking at or above LLM + /// summarization on solve rate at roughly half the cost, and show + /// summarization inflating trajectories 13–15% by destroying the stopping + /// signals an agent uses to notice it has finished. Enable only as a + /// deliberate fallback, and supply a preservation rubric when you do — the + /// rubric is the load-bearing part, not the summarizer. + pub enable_summarization: bool, /// Custom SONA configuration. pub sona_config: Option, /// Custom HNSW configuration. @@ -181,9 +191,10 @@ pub struct PipelineConfig { pub unicode_security_config: Option, } -/// Build the default middleware pipeline per ADR-095 ordering: -/// Todo -> HNSW -> Memory -> Skills -> Filesystem -> SubAgent -> Summarization -/// -> PromptCaching -> PatchToolCalls -> UnicodeSecurityMiddleware -> SONA -> Witness -> ToolSanitizer -> HITL +/// Build the default middleware pipeline (ADR-095 ordering, amended by ADR-274): +/// Todo -> HNSW -> Memory -> Skills -> Filesystem -> SubAgent +/// -> [Summarization, opt-in] -> PromptCaching -> PatchToolCalls +/// -> UnicodeSecurityMiddleware -> SONA -> Witness -> ToolSanitizer -> HITL /// /// HNSW is early in the pipeline to augment context before other middleware. /// UnicodeSecurityMiddleware runs before SONA to sanitize inputs/outputs (C7). @@ -211,9 +222,18 @@ pub fn build_default_pipeline(config: &PipelineConfig) -> MiddlewarePipeline { middlewares.push(Box::new(filesystem::FilesystemMiddleware::new())); middlewares.push(Box::new(subagents::SubAgentMiddleware::new())); - middlewares.push(Box::new(summarization::SummarizationMiddleware::new( - 100_000, 0.85, 0.10, - ))); + + // Summarization is OFF by default (ADR-274). Observation masking in the + // agent loop is the default compaction strategy; this remains available as + // an explicit fallback. Trigger lowered from 0.85 to 0.75 when enabled: + // context degrades well before the nominal limit, so compacting at 85% + // leaves too little headroom to be selective rather than desperate. + if config.enable_summarization { + middlewares.push(Box::new(summarization::SummarizationMiddleware::new( + 100_000, 0.75, 0.10, + ))); + } + middlewares.push(Box::new(prompt_caching::PromptCachingMiddleware::new())); middlewares.push(Box::new(patch_tool_calls::PatchToolCallsMiddleware::new())); @@ -463,9 +483,9 @@ mod tests { fn test_build_default_pipeline_minimal() { let config = PipelineConfig::default(); let pipeline = build_default_pipeline(&config); - // Should have: todo, filesystem, subagent, summarization, prompt_caching, - // patch_tool_calls, tool_sanitizer = 7 - assert!(pipeline.len() >= 7); + // todo, filesystem, subagent, prompt_caching, patch_tool_calls, + // tool_sanitizer = 6. Summarization is opt-in (ADR-274). + assert!(pipeline.len() >= 6); } #[test] @@ -478,14 +498,34 @@ mod tests { enable_sona: false, enable_hnsw: false, enable_unicode_security: false, + enable_summarization: false, sona_config: None, hnsw_config: None, unicode_security_config: None, }; let pipeline = build_default_pipeline(&config); - // todo + memory + skills + filesystem + subagent + summarization + prompt_caching - // + patch_tool_calls + witness + tool_sanitizer + hitl = 11 - assert_eq!(pipeline.len(), 11); + // todo + memory + skills + filesystem + subagent + prompt_caching + // + patch_tool_calls + witness + tool_sanitizer + hitl = 10. + // Summarization is absent by default (ADR-274). + assert_eq!(pipeline.len(), 10); + } + + #[test] + fn test_summarization_is_absent_by_default_and_available_opt_in() { + // The default path must not carry summarization: masking in the agent + // loop is the decided strategy (ADR-274). + let default_len = build_default_pipeline(&PipelineConfig::default()).len(); + let opted_in = build_default_pipeline(&PipelineConfig { + enable_summarization: true, + ..PipelineConfig::default() + }); + assert_eq!( + opted_in.len(), + default_len + 1, + "enabling summarization must add exactly one middleware" + ); + // And it must still be constructible by name, so the fallback is real. + assert!(middleware_by_name("summarization").is_some()); } #[test] From bf841866ef4484831a9148d7029f9d317a93acff Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 22:46:29 +0000 Subject: [PATCH 16/25] fix(rvagent): add enable_summarization to bench PipelineConfig initializers db411ee added a field to PipelineConfig but only ran `cargo test`, which does not build bench targets. `cargo clippy --all-targets` caught it; the commit went out anyway because the command chained the commit after an echo rather than gating on the lint result. Verified with `cargo clippy --workspace --all-targets`. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs | 2 ++ 1 file changed, 2 insertions(+) diff --git a/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs b/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs index ce7a5a9501..417add7919 100644 --- a/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs +++ b/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs @@ -36,6 +36,7 @@ fn bench_full_pipeline(c: &mut Criterion) { enable_sona: false, enable_hnsw: false, enable_unicode_security: false, + enable_summarization: false, sona_config: None, hnsw_config: None, unicode_security_config: None, @@ -126,6 +127,7 @@ fn bench_pipeline_modify_request(c: &mut Criterion) { enable_sona: false, enable_hnsw: false, enable_unicode_security: false, + enable_summarization: false, sona_config: None, hnsw_config: None, unicode_security_config: None, From 7c148f3200069cbf8479f62175710baa52d63752 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 22:47:20 +0000 Subject: [PATCH 17/25] =?UTF-8?q?docs(adr):=20resolve=20the=20Phase=204=20?= =?UTF-8?q?exit=20gate=20(ADR-277=20=C2=A77)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ADR-277 §5.1 withdrew the SWE-bench-Verified gate without a replacement. This supplies one, and the research that produced it killed the obvious candidate: SWE-bench Pro was retracted by OpenAI on 2026-07-08 after an audit flagged 27.4% of its 731 public tasks broken automatically and 34.1% by five independent human reviewers. Two benchmark retractions in six months is the context every claim now lands in. Also ruled out: SWE-Lancer (archived), Aider polyglot (frozen since Nov 2025), LiveCodeBench (model benchmark, not harness-sensitive), OSWorld (self-reported, meeting-gated verification), and bare GAIA (30-50 point spread on identical tasks from scaffolding alone). Four conjunctive gates: 1. Terminal-Bench 2.1 at >=78.0% on a mid-tier model, 5 trials, CI half-width <=1.5pp, team-verified. Deliberately mid-board: 78% on a mid-tier model is a stronger result than 84% on a frontier model, and claiming the top would be overreach given CI widths. 2. The actual harness claim -- >=+4.0pp over Terminus 2 on an identical model, CI-disjoint, across >=3 models and >=2 vendors, with variance decomposition separating harness- from model-induced effects. Absolute pass rate is a joint model x harness measurement; only the fixed-model delta is ours. Requires publishing the falsification point where the delta vanishes. 3. Cost-normalized Pareto. Terminal-Bench publishes no cost column and HAL has paused submissions, so no operating cost-normalized agentic-coding board exists -- uncontested ground, and the natural claim for a Rust harness. 4. SWE-rebench on the rolling window, where memorization is structurally impossible, to prove the result is not terminal-specific or contaminated. Adds required caveat language for every published number and a data-hygiene note: SEO aggregators are publishing figures that do not appear on primary boards, using real model names. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- .../ADR-277-rvagent-positioning-and-claims.md | 114 ++++++++++++++++-- .../04-sota-landscape.md | 9 +- 2 files changed, 109 insertions(+), 14 deletions(-) diff --git a/docs/adr/ADR-277-rvagent-positioning-and-claims.md b/docs/adr/ADR-277-rvagent-positioning-and-claims.md index 0fe68e681a..dd2e969460 100644 --- a/docs/adr/ADR-277-rvagent-positioning-and-claims.md +++ b/docs/adr/ADR-277-rvagent-positioning-and-claims.md @@ -202,14 +202,106 @@ gate is withdrawn without a replacement in hand (§7). community Rust harness" gap closes quickly. The moat must be the library API and replay, **not merely that a Rust harness exists.** -## 7. Open - -Benchmark target selection is unresolved. Required before Phase 4 can restate -its gate: current top scores by model *and* harness, cost-normalized Pareto -position, and conformance criteria for a credible 2026 claim on a -non-saturated benchmark. Tracked in -`04-sota-landscape.md` §9. - -Also outstanding: read `codex-rs/core/src/` directly before adopting its -`Op`/`EventMsg` submit/event design — the survey's account of it is -third-party. +## 7. The Phase 4 exit gate (resolved 2026-08-01) + +§5.1 withdrew the SWE-bench-Verified gate without a replacement. This section +supplies one. + +**The obvious successor is also gone.** SWE-bench **Pro** was **retracted by +OpenAI on 2026-07-08**: an audit of its 731 public tasks flagged **27.4% +broken** automatically and **34.1%** by five independent human reviewers, the +dominant failure being over-strict hidden tests enforcing unspecified +implementation details. Two benchmark retractions in six months is the context +every claim we publish now lands in. + +Also dead or unusable: SWE-Lancer (archived 2025-07-18), Aider polyglot (frozen +2025-11-20, no 2026 models), LiveCodeBench (a *model* benchmark — a harness +contributes nothing), OSWorld (self-reported rows, meeting-gated verification), +bare GAIA (a documented **30–50 point** spread on identical tasks purely from +scaffolding). + +### The gate — all four must pass + +**Gate 1 — Terminal-Bench 2.1 absolute.** +**≥ 78.0% pass rate, 5 trials, bootstrap 95% CI half-width ≤ 1.5 pp, on a +mid-tier model, team-verified.** + +Terminal-Bench 2.1 is the only board that is simultaneously team-verified, +CI-reporting, adversarially audited during construction, and *structurally a +harness comparison* (Claude Code, Codex, Terminus 2, Cursor CLI and +mini-SWE-agent all appear on shared models). + +78.0% is deliberate. It sits mid-board and CI-disjoint above the weakest +entries while explicitly **not** claiming to beat the leader at 83.8% ± 1.2. +**78% on a mid-tier model is a stronger result than 84% on a frontier model**, +and it is the honest version of our story. Claiming ≥84% would be overreach +given the CI widths and the reward-hackability base rate. + +**Gate 2 — Fixed-model cross-harness delta.** *This is the actual harness claim.* +**≥ +4.0 pp over Terminus 2 on the identical model, CI-disjoint, replicated on +≥3 models spanning ≥2 vendors.** + +Absolute pass rate is a joint model×harness measurement; only the fixed-model +delta is attributable to us. Cross-vendor is required — a single-vendor result +is indistinguishable from prompt-fitting to one tokenizer. + +Baselines: **Terminus 2** (the reference scaffold) and **mini-SWE-agent** (the +minimal-scaffold control). Both are on the current board, so the delta is +directly auditable. + +Report per-task pass rates, bootstrap CIs, and a **variance decomposition +separating harness-induced from model-induced variance** (arXiv 2605.23950, +which documents model-ranking *reversals* under different harnesses). + +**Publish the falsification:** the model tier where our delta vanishes or +inverts. A harness result with no stated inversion point reads as +cherry-picked, and per §2 of ADR-273 the effect should shrink as model +capability rises — if it does not, that is evidence something is wrong. + +**Gate 3 — Cost-normalized Pareto.** *Our differentiator.* +**Match or beat the best open-scaffold entry's pass rate at ≤ 40% of its +$/task; publish $/task, input+output tokens/task, and wall-clock/task for every +cell of the fixed-model matrix.** + +Terminal-Bench publishes **no cost column at all**, and HAL — the only board +that treats cost as a first-class axis — has **paused submissions**. There is +currently *no* operating cost-normalized agentic-coding leaderboard. Publishing +one in HAL's format is uncontested ground and is the natural claim for a Rust +harness. + +The framing anchor is HAL's own finding: *agents can be 100× more expensive +while being 1% better.* On SWE-bench-Verified-Mini its frontier runs from a +$65.31 cost-efficient knee to a $1,351 point that scores **11 points lower**. +A good harness on a cheap model dominating a mediocre harness on an expensive +one is an existence proof, not a hope. + +**Gate 4 — Contamination-resistant corroboration.** +**A SWE-rebench run on the current rolling window (not a frozen split), +reporting resolved% ± CI and pass@5 on ≥2 of the fixed models.** + +The rolling window makes memorization structurally impossible. This proves the +Terminal-Bench result is neither terminal-specific nor contaminated. + +### Required caveat language + +Every headline number ships with: model, harness, trial count, CI, verification +status, and the fixed-model delta vs Terminus 2 — plus an explicit statement +that absolute pass rate is a joint model×harness measurement and only the delta +is attributable to the harness. Cite arXiv 2605.23950. + +This is cheap, and it is the single thing separating a defensible claim from +the pattern that got two benchmarks retracted inside six months. + +### Data hygiene + +SEO aggregators are publishing leaderboard numbers that do not appear on +primary boards (e.g. inflated Terminal-Bench and SWE-bench figures). Some use +real model names, which makes them more dangerous rather than less. **Cite only +primary leaderboards.** + +## 8. Still open + +Read `codex-rs/core/src/` directly before adopting its `Op`/`EventMsg` +submit/event design — the survey's account is third-party. Note this session +cannot attach `openai/codex` (cross-owner adds unsupported); fetch the files +directly or read them in a session rooted on that repo. diff --git a/docs/research/rvagent-hermes-harness/04-sota-landscape.md b/docs/research/rvagent-hermes-harness/04-sota-landscape.md index f955f7ac20..1198020098 100644 --- a/docs/research/rvagent-hermes-harness/04-sota-landscape.md +++ b/docs/research/rvagent-hermes-harness/04-sota-landscape.md @@ -420,9 +420,12 @@ apparatus — given §7, retraction discipline is the feature. ## 9. Open items -- Benchmark/leaderboard sweep (current top scores by model+harness, - cost-normalized Pareto, credible-claim conformance criteria) is still - outstanding; §7 covers hygiene but not target selection. +- ~~Benchmark/leaderboard sweep~~ **Resolved 2026-08-01.** Headline: **SWE-bench + Pro was retracted by OpenAI on 2026-07-08** (27.4% of public tasks flagged + broken automatically, 34.1% by human reviewers) — the obvious successor to + Verified is also gone. Terminal-Bench 2.1 is the only credible harness board; + no operating leaderboard reports cost, which makes a cost-normalized Pareto + uncontested ground. Full gate in ADR-277 §7. - `codex-rs` internals in §2 come from a third-party architecture writeup, not the repo. Worth reading `codex-rs/core/src/` directly before copying the `Op`/`EventMsg` design. From d64155b17afad077a186176cc1eaa0c903672575 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 1 Aug 2026 22:50:39 +0000 Subject: [PATCH 18/25] feat(rvagent): diagnose failed edits instead of reporting "not found" (ADR-273) "Error: old_string not found" is the highest-frequency tool failure in an editing agent and is nearly useless on its own: the model already believed the string was there, so restating that it isn't gives it nothing to change. It then retries a near-identical call, which is precisely the input condition for loop detection. The edit path now works out WHY the match failed and says so, checking causes in order of real-world frequency: CRLF-vs-LF line endings, whitespace (indentation, trailing spaces, tabs-vs-spaces), case, and finally the nearest line in the file echoed back verbatim so the model can copy the real text. When nothing is close it says that plainly and tells the agent to read the file rather than assume its contents. No new tool. edit_file already implements str_replace semantics, and ADR-273 3.4 makes the 8-15 tool budget something to defend rather than spend -- the published reproductions show the ergonomics are what move the number, not the tool's existence. Writing the end-to-end test corrected a wrong assumption: omitting leading indentation does NOT fail, because matching is substring-based. The failure that actually happens is supplying the WRONG indentation. Both the e2e test and a misleading unit-test comment were fixed rather than left describing a scenario that cannot occur. Tests: 111 unit + 15 e2e in rvagent-tools, including CRLF, tabs-vs-spaces, case, nearest-line, multibyte safety, and a guard that a trivially short needle does not produce a confident wrong guess. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-tools/src/edit_diag.rs | 188 ++++++++++++++++++ crates/rvAgent/rvagent-tools/src/lib.rs | 2 + crates/rvAgent/rvagent-tools/src/local.rs | 8 +- .../rvagent-tools/tests/e2e_tool_calling.rs | 41 ++++ 4 files changed, 236 insertions(+), 3 deletions(-) create mode 100644 crates/rvAgent/rvagent-tools/src/edit_diag.rs diff --git a/crates/rvAgent/rvagent-tools/src/edit_diag.rs b/crates/rvAgent/rvagent-tools/src/edit_diag.rs new file mode 100644 index 0000000000..2bfa5160a6 --- /dev/null +++ b/crates/rvAgent/rvagent-tools/src/edit_diag.rs @@ -0,0 +1,188 @@ +//! Diagnostics for failed string edits (ADR-273 §3.1, §3.3). +//! +//! "Error: old_string not found" is the single highest-frequency tool failure +//! in an editing agent, and it is nearly useless on its own: the model already +//! believed the string was there, so restating that it isn't gives it nothing +//! to change. It then retries a near-identical call, which is the input +//! condition for loop detection. +//! +//! Edit-tool ergonomics are load-bearing rather than incidental — in the +//! published reproductions, only the flavor with real failure diagnostics moved +//! the benchmark number; a plain `edit`/`write_file` pair gave no improvement +//! at all. So this module works out *why* a match failed and says so. + +/// Maximum characters of a candidate line echoed back in a diagnostic. +const SNIPPET_LEN: usize = 160; + +/// Explain why `old_string` did not match anything in `content`. +/// +/// Returns an actionable message naming the likely cause. Ordered by how +/// common the cause is in practice, so the first plausible explanation is the +/// most likely one. +pub fn diagnose_edit_failure(content: &str, old_string: &str, path: &str) -> String { + let base = format!("Error: old_string not found in {path}."); + + // 1. Line endings. Invisible, and it defeats an otherwise exact match. + if content.contains("\r\n") && !old_string.contains("\r\n") { + let normalized = old_string.replace('\n', "\r\n"); + if content.contains(&normalized) { + return format!( + "{base} The file uses CRLF line endings but old_string uses LF. \ + The text is present — re-read the file and copy the exact bytes." + ); + } + } + + // 2. Whitespace. Indentation drift is the classic cause: the model + // reconstructs the line from memory and gets the leading spaces wrong. + let squeeze = |s: &str| -> String { s.split_whitespace().collect::>().join(" ") }; + let squeezed_old = squeeze(old_string); + if !squeezed_old.is_empty() && squeeze(content).contains(&squeezed_old) { + return format!( + "{base} A match exists when whitespace is ignored, so the difference is \ + indentation, trailing spaces, or tabs-vs-spaces. Re-read the file and \ + copy the exact leading whitespace." + ); + } + + // 3. Case. + if content.to_lowercase().contains(&old_string.to_lowercase()) { + return format!( + "{base} A match exists ignoring case — the difference is capitalization only." + ); + } + + // 4. Partial match: locate the anchor line and show what is actually there. + // This is the most useful case, because it hands the model the real text. + if let Some(hint) = nearest_line_hint(content, old_string) { + return format!("{base} {hint}"); + } + + format!( + "{base} No similar text was found. The file may not contain this code at all — \ + read the file before editing it, rather than assuming its contents." + ) +} + +/// Find the line in `content` most similar to the first line of `old_string`, +/// and describe the mismatch. +fn nearest_line_hint(content: &str, old_string: &str) -> Option { + let needle = old_string.lines().next()?.trim(); + if needle.len() < 4 { + // Too short to attribute a near-match to anything meaningful. + return None; + } + + let mut best: Option<(usize, usize, &str)> = None; // (score, line_no, text) + for (i, line) in content.lines().enumerate() { + let score = shared_prefix_len(line.trim(), needle); + if score >= 4 && best.map(|(b, _, _)| score > b).unwrap_or(true) { + best = Some((score, i + 1, line)); + } + } + + let (_, line_no, text) = best?; + Some(format!( + "The closest line in the file is line {line_no}: {:?}. \ + Copy it exactly, including whitespace.", + truncate(text.trim_end(), SNIPPET_LEN) + )) +} + +/// Length of the common prefix of two strings, in characters. +fn shared_prefix_len(a: &str, b: &str) -> usize { + a.chars().zip(b.chars()).take_while(|(x, y)| x == y).count() +} + +/// Truncate on a character boundary, marking the cut. +fn truncate(s: &str, max: usize) -> String { + if s.chars().count() <= max { + return s.to_string(); + } + let cut: String = s.chars().take(max).collect(); + format!("{cut}…") +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn detects_indentation_mismatch() { + let content = "fn main() {\n let x = 1;\n}\n"; + // Wrong indentation, not missing: matching is substring-based, so an + // omitted indent still matches and never reaches this diagnostic. + let msg = diagnose_edit_failure(content, " let x = 1;", "a.rs"); + assert!(msg.contains("whitespace is ignored"), "got: {msg}"); + assert!(msg.contains("exact leading whitespace")); + } + + #[test] + fn detects_tabs_versus_spaces() { + let content = "fn main() {\n\tlet x = 1;\n}\n"; + let msg = diagnose_edit_failure(content, " let x = 1;", "a.rs"); + assert!(msg.contains("tabs-vs-spaces"), "got: {msg}"); + } + + #[test] + fn detects_crlf_mismatch() { + let content = "line one\r\nline two\r\n"; + let msg = diagnose_edit_failure(content, "line one\nline two", "a.txt"); + assert!(msg.contains("CRLF"), "got: {msg}"); + assert!(msg.contains("The text is present")); + } + + #[test] + fn detects_case_mismatch() { + let content = "let Value = 1;\n"; + let msg = diagnose_edit_failure(content, "let value = 1;", "a.rs"); + assert!(msg.contains("capitalization"), "got: {msg}"); + } + + #[test] + fn shows_the_nearest_line_when_content_drifted() { + let content = "fn compute(a: u32, b: u32) -> u32 {\n a + b\n}\n"; + // Same opening but a different signature — the most common real case. + let msg = diagnose_edit_failure(content, "fn compute(a: u32) -> u32 {", "a.rs"); + assert!(msg.contains("closest line in the file is line 1"), "got: {msg}"); + assert!(msg.contains("fn compute(a: u32, b: u32)"), "got: {msg}"); + } + + #[test] + fn says_so_plainly_when_nothing_is_close() { + let content = "completely unrelated file contents\n"; + let msg = diagnose_edit_failure(content, "fn transmogrify() {", "a.rs"); + assert!(msg.contains("No similar text was found"), "got: {msg}"); + assert!(msg.contains("read the file before editing")); + } + + #[test] + fn always_names_the_path() { + let msg = diagnose_edit_failure("x", "y", "src/lib.rs"); + assert!(msg.contains("src/lib.rs")); + } + + #[test] + fn handles_multibyte_content_without_panicking() { + let content = "let s = \"héllo 🙂 wörld\";\n"; + let msg = diagnose_edit_failure(content, "let s = \"hello world\";", "a.rs"); + assert!(!msg.is_empty()); + } + + #[test] + fn does_not_guess_from_a_trivially_short_needle() { + let content = "aaaa bbbb\ncccc dddd\n"; + // A 2-char needle would "nearly match" almost any line; it must not + // produce a confident and wrong nearest-line claim. + let msg = diagnose_edit_failure(content, "xy", "a.txt"); + assert!(msg.contains("No similar text was found"), "got: {msg}"); + } + + #[test] + fn truncates_a_very_long_candidate_line() { + let long = "x".repeat(500); + let content = format!("prefix_{long}\n"); + let msg = diagnose_edit_failure(&content, &format!("prefix_{}", "y".repeat(20)), "a.txt"); + assert!(msg.len() < 400, "diagnostic should stay compact: {}", msg.len()); + } +} diff --git a/crates/rvAgent/rvagent-tools/src/lib.rs b/crates/rvAgent/rvagent-tools/src/lib.rs index 61c75f61be..e76d748b55 100644 --- a/crates/rvAgent/rvagent-tools/src/lib.rs +++ b/crates/rvAgent/rvagent-tools/src/lib.rs @@ -3,6 +3,7 @@ //! Provides the `Tool` trait, `BuiltinTool`/`AnyTool` enum dispatch, //! `ToolRuntime` context, and parallel execution (ADR-103 A2). +pub mod edit_diag; pub mod edit_file; pub mod execute; pub mod glob; @@ -20,6 +21,7 @@ use std::collections::HashMap; use std::fmt; use std::sync::Arc; +pub use edit_diag::diagnose_edit_failure; pub use edit_file::EditFileTool; pub use execute::ExecuteTool; pub use glob::GlobTool; diff --git a/crates/rvAgent/rvagent-tools/src/local.rs b/crates/rvAgent/rvagent-tools/src/local.rs index 9ac8b4b6dd..5b07196c97 100644 --- a/crates/rvAgent/rvagent-tools/src/local.rs +++ b/crates/rvAgent/rvagent-tools/src/local.rs @@ -229,10 +229,12 @@ impl Backend for LocalFsBackend { }; let count = content.matches(old_string).count(); if count == 0 { + // Say why it failed, not just that it did (ADR-273 §3.1/§3.3). return WriteResult { - error: Some(format!( - "Error: old_string not found in {}", - target.display() + error: Some(crate::diagnose_edit_failure( + &content, + old_string, + &target.display().to_string(), )), ..Default::default() }; diff --git a/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs b/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs index d26e51d9ae..fdfd6b3a16 100644 --- a/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs +++ b/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs @@ -602,6 +602,47 @@ async fn recall_returns_the_full_content_of_a_masked_observation() { ); } +#[tokio::test] +async fn failed_edit_explains_why_it_failed() { + let dir = tempfile::tempdir().unwrap(); + std::fs::write( + dir.path().join("code.rs"), + "fn main() {\n let total = 1;\n}\n", + ) + .unwrap(); + + let model = ScriptedModel::new(vec![ + Message::ai_with_tools( + "", + vec![call( + "t1", + "edit_file", + // Wrong indentation (8 spaces, file has 4). Omitting the + // indent entirely would still match as a substring; supplying + // the wrong amount is the failure that actually happens. + serde_json::json!({ + "file_path": "code.rs", + "old_string": " let total = 1;", + "new_string": " let total = 2;" + }), + )], + ), + Message::ai("done"), + ]); + let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); + let state = graph.run(AgentState::new()).await.unwrap(); + + let results = tool_results(&state); + assert_eq!(results.len(), 1); + // "not found" alone is useless — the model already believed it was there. + assert!( + results[0].contains("whitespace is ignored"), + "edit failure was not diagnosed: {}", + results[0] + ); + assert!(results[0].contains("exact leading whitespace")); +} + #[tokio::test] async fn recall_with_an_unknown_id_is_actionable_not_fatal() { let dir = tempfile::tempdir().unwrap(); From 3c5ed9ca6513de9d812d244c1a600308aea8c02d Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 2 Aug 2026 00:36:57 +0000 Subject: [PATCH 19/25] feat(rvagent): policy genome and flywheel score axes (ADR-278) The Rust half of the flywheel seam: the thing being evolved (PolicyGenome) and the thing being measured (Score). rvAgent does not implement a promotion engine -- @metaharness/flywheel already provides the frozen fingerprinted gate, holdout plus never-optimized-against anchor, Ed25519 receipts, replay verification, and lineage DAG. PolicyGenome is shaped as the flywheel's Policy = Record so there is no adapter impedance, and is backed by a BTreeMap so serialization is deterministic -- iteration order feeds gate fingerprints and replay, and a nondeterministic order would make identical genomes hash differently. It maps onto real config: max_iterations, parallel_tools, max_parallel_tools, loop_repeat_threshold, keep_last_observations, max_tool_result_bytes, plus two text levers for the prompt layer. Unknown levers are rejected rather than silently skipped. A mutation to a lever the harness does not apply produces a run identical to baseline, which the optimizer would score as "no effect" and burn generations on. Score projects runs onto the gate's four axes, including noopRate -- the non-obvious clause the default gate requires to strictly improve. A run that reports success while committing nothing counts as a no-op, so a policy cannot earn promotion by making the agent talk rather than act. Fixes a real bug at the JS seam found while testing: cost-per-win with zero wins was f64::INFINITY, which serde serializes as JSON null, and the gate's `candidate.costPerWin > baseline.costPerWin` evaluates `null > n` as false in JavaScript -- so a policy that won nothing would silently PASS the cost clause. Verified by running the gate's own logic against both encodings: null does not fire the clause, the finite sentinel does. Now uses f64::MAX as an explicit COST_PER_WIN_NO_WINS constant, with a test asserting no axis ever serializes non-finite. Tests: 17 policy tests including deterministic serialization, unknown-lever rejection, a guard that KNOWN_LEVERS cannot drift out of sync with apply_to, and the non-finite serialization guard. 174 core tests green. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-core/src/lib.rs | 1 + crates/rvAgent/rvagent-core/src/policy.rs | 426 ++++++++++++++++++++++ 2 files changed, 427 insertions(+) create mode 100644 crates/rvAgent/rvagent-core/src/policy.rs diff --git a/crates/rvAgent/rvagent-core/src/lib.rs b/crates/rvAgent/rvagent-core/src/lib.rs index 145707b3e8..7ebb68f988 100644 --- a/crates/rvAgent/rvagent-core/src/lib.rs +++ b/crates/rvAgent/rvagent-core/src/lib.rs @@ -31,6 +31,7 @@ pub mod messages; pub mod metrics; pub mod models; pub mod parallel; +pub mod policy; pub mod prompt; pub mod rvf_bridge; pub mod session_crypto; diff --git a/crates/rvAgent/rvagent-core/src/policy.rs b/crates/rvAgent/rvagent-core/src/policy.rs new file mode 100644 index 0000000000..00bd74da4b --- /dev/null +++ b/crates/rvAgent/rvagent-core/src/policy.rs @@ -0,0 +1,426 @@ +//! Policy genome and score axes for the promotion flywheel (ADR-278). +//! +//! rvAgent does not implement a promotion engine. `@metaharness/flywheel` +//! already provides a frozen fingerprinted conjunctive gate, a holdout plus a +//! never-optimized-against anchor, Ed25519 receipts, independent replay +//! verification, and a compounding lineage DAG. This module is the Rust half of +//! that seam: the thing being evolved (a [`PolicyGenome`]) and the thing being +//! measured (a [`Score`]). +//! +//! The genome is deliberately shaped as the flywheel's `Policy` — +//! `Record` — so there is no adapter impedance. It is also what +//! GEPA-style optimizers consume, whose candidate is likewise a named set of +//! text components. +//! +//! # Why policy and not memory +//! +//! Self-learning splits into two objects with opposite evidence: policy text +//! (positive) and accumulated episodic memory (negative — an inverted-U where +//! utility eventually falls below no-memory). ADR-278 moves new effort here. + +use std::collections::BTreeMap; + +use serde::{Deserialize, Serialize}; + +use crate::graph::GraphConfig; +use crate::masking::MaskConfig; + +/// The levers this harness knows how to apply. +/// +/// A genome naming anything outside this set is rejected rather than ignored — +/// see [`PolicyGenome::apply_to`]. +pub const KNOWN_LEVERS: &[&str] = &[ + "max_iterations", + "parallel_tools", + "max_parallel_tools", + "loop_repeat_threshold", + "keep_last_observations", + "max_tool_result_bytes", + "system_prompt_suffix", + "compaction_rubric", +]; + +/// An operating policy: named string levers, the unit the flywheel evolves. +/// +/// `BTreeMap` rather than `HashMap` so serialization is deterministic — +/// iteration order feeds gate fingerprints and replay, and a nondeterministic +/// ordering would make identical genomes hash differently. +#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)] +#[serde(transparent)] +pub struct PolicyGenome { + levers: BTreeMap, +} + +/// Why a genome could not be applied. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum PolicyError { + /// The genome names a lever this harness does not implement. + /// + /// This must be an error, not a silent skip. A mutation to an unapplied + /// lever produces a run identical to baseline, which the flywheel would + /// score as "no effect" and burn generations on — while a scoring artifact + /// could even promote it. Failing loudly keeps the search honest. + UnknownLever(String), + /// The value could not be parsed for that lever's type. + BadValue { lever: String, value: String }, +} + +impl std::fmt::Display for PolicyError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + PolicyError::UnknownLever(name) => write!( + f, + "unknown policy lever '{name}' — this harness would ignore it, so the \ + evaluation would be meaningless. Known levers: {}", + KNOWN_LEVERS.join(", ") + ), + PolicyError::BadValue { lever, value } => { + write!(f, "lever '{lever}' cannot take value {value:?}") + } + } + } +} + +impl std::error::Error for PolicyError {} + +impl PolicyGenome { + pub fn new() -> Self { + Self::default() + } + + /// Set a lever. Chainable. + pub fn with(mut self, lever: impl Into, value: impl Into) -> Self { + self.levers.insert(lever.into(), value.into()); + self + } + + pub fn get(&self, lever: &str) -> Option<&str> { + self.levers.get(lever).map(String::as_str) + } + + pub fn is_empty(&self) -> bool { + self.levers.is_empty() + } + + pub fn len(&self) -> usize { + self.levers.len() + } + + pub fn iter(&self) -> impl Iterator { + self.levers.iter() + } + + /// Free-text levers, which shape prompts rather than numeric config. + pub fn system_prompt_suffix(&self) -> Option<&str> { + self.get("system_prompt_suffix") + } + + pub fn compaction_rubric(&self) -> Option<&str> { + self.get("compaction_rubric") + } + + /// Apply the numeric and boolean levers onto a config. + /// + /// Rejects unknown levers (see [`PolicyError::UnknownLever`]). Text levers + /// are validated as known but applied by the caller, since they belong to + /// the prompt layer rather than the loop config. + pub fn apply_to(&self, base: GraphConfig) -> Result { + let mut config = base; + let mut mask: MaskConfig = config.mask.clone(); + + for (lever, value) in &self.levers { + match lever.as_str() { + "max_iterations" => config.max_iterations = parse(lever, value)?, + "parallel_tools" => config.parallel_tools = parse(lever, value)?, + "max_parallel_tools" => config.max_parallel_tools = parse(lever, value)?, + "loop_repeat_threshold" => config.loop_repeat_threshold = parse(lever, value)?, + "keep_last_observations" => mask.keep_last_observations = parse(lever, value)?, + "max_tool_result_bytes" => mask.max_tool_result_bytes = parse(lever, value)?, + // Known, but applied at the prompt layer. + "system_prompt_suffix" | "compaction_rubric" => {} + other => return Err(PolicyError::UnknownLever(other.to_string())), + } + } + + config.mask = mask; + Ok(config) + } +} + +fn parse(lever: &str, value: &str) -> Result { + value.trim().parse::().map_err(|_| PolicyError::BadValue { + lever: lever.to_string(), + value: value.to_string(), + }) +} + +// --------------------------------------------------------------------------- +// Scoring +// --------------------------------------------------------------------------- + +/// Cost-per-win when a policy won nothing. +/// +/// **Not `f64::INFINITY`.** JSON has no infinity, so serde emits `null`, and the +/// gate's comparison `candidate.costPerWin > baseline.costPerWin` evaluates +/// `null > n` as `false` in JavaScript — meaning a policy that won nothing +/// would silently *pass* the cost clause. The largest finite double is the +/// honest encoding of "unboundedly bad" and compares correctly on both sides. +pub const COST_PER_WIN_NO_WINS: f64 = f64::MAX; + +/// The outcome of one evaluated run. +#[derive(Debug, Clone, PartialEq)] +pub struct RunOutcome { + /// Did the run achieve the task (tests pass, issue resolved)? + pub succeeded: bool, + /// Did the run actually change anything? + /// + /// A run that ends without committing any change is a **no-op** even when + /// it reports success — the agent talked itself to a stop. This is the + /// signal `noop_rate` exists to catch. + pub made_changes: bool, + /// Total cost in USD. + pub cost_usd: f64, + /// Hard safety or security regression. Any `true` blocks promotion. + pub regressed: bool, +} + +/// The four axes the flywheel's gate decides over. +/// +/// Named generically on purpose — the gate is host- and benchmark-agnostic, and +/// projecting rvAgent's meaning onto these axes honestly is the trust boundary +/// for every downstream guarantee. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +pub struct Score { + /// Main quality signal — higher is better. + pub primary: f64, + /// Fraction of runs that committed nothing — lower is better. + /// + /// The default gate requires this to **strictly** improve: a policy earns + /// promotion by making the executor commit more, not merely score higher. + /// A policy that raises `primary` while leaving the agent more likely to + /// end empty has found a scoring artifact, not an improvement. + #[serde(rename = "noopRate")] + pub noop_rate: f64, + /// Resource cost per success — lower is better. + #[serde(rename = "costPerWin")] + pub cost_per_win: f64, + /// Hard safety/security stop. + pub regressed: bool, +} + +impl Score { + /// Aggregate run outcomes into the four axes. + /// + /// An empty set scores as maximally bad rather than perfect: zero runs must + /// never look like a clean sweep to the gate. + pub fn from_runs(runs: &[RunOutcome]) -> Self { + if runs.is_empty() { + return Self { + primary: 0.0, + noop_rate: 1.0, + cost_per_win: COST_PER_WIN_NO_WINS, + regressed: false, + }; + } + + let total = runs.len() as f64; + let wins = runs.iter().filter(|r| r.succeeded).count(); + let noops = runs.iter().filter(|r| !r.made_changes).count(); + let cost: f64 = runs.iter().map(|r| r.cost_usd).sum(); + + Self { + primary: wins as f64 / total, + noop_rate: noops as f64 / total, + cost_per_win: if wins == 0 { + COST_PER_WIN_NO_WINS + } else { + cost / wins as f64 + }, + regressed: runs.iter().any(|r| r.regressed), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn run(succeeded: bool, made_changes: bool, cost_usd: f64) -> RunOutcome { + RunOutcome { + succeeded, + made_changes, + cost_usd, + regressed: false, + } + } + + #[test] + fn genome_serializes_as_a_flat_string_map() { + let g = PolicyGenome::new() + .with("max_iterations", "50") + .with("system_prompt_suffix", "Be concise."); + let json = serde_json::to_string(&g).unwrap(); + // Must match the flywheel's Policy = Record exactly. + assert_eq!( + json, + r#"{"max_iterations":"50","system_prompt_suffix":"Be concise."}"# + ); + let back: PolicyGenome = serde_json::from_str(&json).unwrap(); + assert_eq!(back, g); + } + + #[test] + fn serialization_order_is_deterministic() { + let a = PolicyGenome::new().with("zebra", "1").with("alpha", "2"); + let b = PolicyGenome::new().with("alpha", "2").with("zebra", "1"); + // Identical genomes must serialize identically, or fingerprints and + // replay diverge for no reason. + assert_eq!( + serde_json::to_string(&a).unwrap(), + serde_json::to_string(&b).unwrap() + ); + } + + #[test] + fn applies_numeric_and_boolean_levers() { + let g = PolicyGenome::new() + .with("max_iterations", "42") + .with("parallel_tools", "false") + .with("loop_repeat_threshold", "5") + .with("keep_last_observations", "3"); + let config = g.apply_to(GraphConfig::default()).unwrap(); + + assert_eq!(config.max_iterations, 42); + assert!(!config.parallel_tools); + assert_eq!(config.loop_repeat_threshold, 5); + assert_eq!(config.mask.keep_last_observations, 3); + } + + #[test] + fn unknown_lever_is_rejected_not_ignored() { + let g = PolicyGenome::new().with("nonexistent_knob", "7"); + let err = g.apply_to(GraphConfig::default()).unwrap_err(); + assert_eq!(err, PolicyError::UnknownLever("nonexistent_knob".into())); + // The message must name the valid set, or the optimizer cannot recover. + assert!(err.to_string().contains("max_iterations")); + } + + #[test] + fn bad_value_is_reported_with_the_lever_name() { + let g = PolicyGenome::new().with("max_iterations", "not-a-number"); + let err = g.apply_to(GraphConfig::default()).unwrap_err(); + assert!(matches!(err, PolicyError::BadValue { .. })); + assert!(err.to_string().contains("max_iterations")); + } + + #[test] + fn text_levers_are_accepted_and_readable() { + let g = PolicyGenome::new() + .with("system_prompt_suffix", "Prefer small diffs.") + .with("compaction_rubric", "Preserve failing tests."); + assert!(g.apply_to(GraphConfig::default()).is_ok()); + assert_eq!(g.system_prompt_suffix(), Some("Prefer small diffs.")); + assert_eq!(g.compaction_rubric(), Some("Preserve failing tests.")); + } + + #[test] + fn empty_genome_leaves_config_untouched() { + let base = GraphConfig::default(); + let applied = PolicyGenome::new().apply_to(base.clone()).unwrap(); + assert_eq!(applied.max_iterations, base.max_iterations); + assert_eq!( + applied.mask.keep_last_observations, + base.mask.keep_last_observations + ); + } + + #[test] + fn score_computes_the_four_axes() { + let runs = vec![ + run(true, true, 1.0), + run(true, true, 3.0), + run(false, true, 2.0), + run(false, false, 0.5), + ]; + let s = Score::from_runs(&runs); + assert_eq!(s.primary, 0.5); + assert_eq!(s.noop_rate, 0.25); + assert_eq!(s.cost_per_win, 6.5 / 2.0); + assert!(!s.regressed); + } + + #[test] + fn a_successful_run_that_changed_nothing_still_counts_as_a_noop() { + // The case the axis exists for: the agent reports success but committed + // nothing. Scoring that as a win would reward talking over doing. + let s = Score::from_runs(&[run(true, false, 1.0)]); + assert_eq!(s.primary, 1.0); + assert_eq!(s.noop_rate, 1.0); + } + + #[test] + fn zero_wins_gives_maximally_bad_cost_per_win_not_zero() { + let s = Score::from_runs(&[run(false, true, 5.0)]); + assert_eq!(s.cost_per_win, COST_PER_WIN_NO_WINS); + } + + #[test] + fn score_never_serializes_a_non_finite_number() { + // JSON has no infinity: serde emits null, and the gate reads `null > n` + // as false in JS, so a zero-win policy would pass the cost clause. + for runs in [vec![], vec![run(false, true, 5.0)], vec![run(true, true, 1.0)]] { + let json = serde_json::to_value(Score::from_runs(&runs)).unwrap(); + let cost = json.get("costPerWin").unwrap(); + assert!( + cost.is_number(), + "costPerWin must serialize as a number, got {cost}" + ); + assert!(cost.as_f64().unwrap().is_finite()); + } + } + + #[test] + fn empty_run_set_scores_as_bad_not_perfect() { + let s = Score::from_runs(&[]); + assert_eq!(s.primary, 0.0); + assert_eq!(s.noop_rate, 1.0); + assert_eq!(s.cost_per_win, COST_PER_WIN_NO_WINS); + } + + #[test] + fn any_regression_sets_the_hard_stop() { + let mut bad = run(true, true, 1.0); + bad.regressed = true; + let s = Score::from_runs(&[run(true, true, 1.0), bad]); + assert!(s.regressed); + } + + #[test] + fn score_serializes_with_the_gate_field_names() { + let s = Score::from_runs(&[run(true, true, 2.0)]); + let json = serde_json::to_value(&s).unwrap(); + // These names are the gate's contract; renaming them silently breaks it. + assert!(json.get("primary").is_some()); + assert!(json.get("noopRate").is_some()); + assert!(json.get("costPerWin").is_some()); + assert!(json.get("regressed").is_some()); + } + + + #[test] + fn every_known_lever_is_actually_applicable() { + // Guards against KNOWN_LEVERS drifting out of sync with apply_to. + for lever in KNOWN_LEVERS { + let value = match *lever { + "parallel_tools" => "true", + "system_prompt_suffix" | "compaction_rubric" => "text", + _ => "4", + }; + let g = PolicyGenome::new().with(*lever, value); + assert!( + g.apply_to(GraphConfig::default()).is_ok(), + "declared lever '{lever}' is not handled by apply_to" + ); + } + } +} From 18d10a1fc9cc4f7bfb31c9fe616143c76c4aa56e Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 2 Aug 2026 01:14:02 +0000 Subject: [PATCH 20/25] =?UTF-8?q?feat(rvagent):=20subagent=20boundary=20?= =?UTF-8?q?=E2=80=94=20isolated=20context=20in,=20String=20out=20(ADR-275)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Implements the subagent boundary as a tool that spawns an isolated context and returns a String. Not peer agents with a message bus, shared mutable state, or a mergeable state type. The signature is the architecture. A subagent returns text and nothing else -- no state update, no file handle, no mergeable value -- so a caller cannot wire one up as a concurrent writer even by accident. SubagentRequest carries only a role and a prompt, with no parent-state field by construction, so a subagent cannot reach the parent's conversation, files, or todos. This is what replaces Phase 1.6's CoW fork/merge + CRDT join. A CRDT can merge two edits to the same file without textual conflict; it cannot make the result coherent. Making the writer singular removes that entire failure class rather than managing it. Two roles, both read-only. may_write() and inherits_parent_context() are encoded as methods returning false rather than assumed, so adding a role that violates either is a visible decision rather than an oversight -- with tests asserting the invariant directly. Gatherer runs on the cheap tier by design: a frontier model in a read-only slot measured +0.4pp at 5.8x cost, corroborated by metaharness ADR-226's 5.4x null. Reviewer is documented as GATED rather than adopted, per ADR-278 §7, with ADR-226 named as the null it must beat. Summary returns are capped at the boundary rather than trusted: a subagent exists to reduce what reaches the parent's context, so one returning its whole transcript has inverted its own purpose. Tests: 182 core tests (was 174), including the single-writer invariant across all roles and the summary budget. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-core/src/lib.rs | 1 + crates/rvAgent/rvagent-core/src/subagent.rs | 203 ++++++++++++++++++++ 2 files changed, 204 insertions(+) create mode 100644 crates/rvAgent/rvagent-core/src/subagent.rs diff --git a/crates/rvAgent/rvagent-core/src/lib.rs b/crates/rvAgent/rvagent-core/src/lib.rs index 7ebb68f988..62dd52d89a 100644 --- a/crates/rvAgent/rvagent-core/src/lib.rs +++ b/crates/rvAgent/rvagent-core/src/lib.rs @@ -36,6 +36,7 @@ pub mod prompt; pub mod rvf_bridge; pub mod session_crypto; pub mod state; +pub mod subagent; pub mod string_pool; // Re-export key types at crate root for convenience. diff --git a/crates/rvAgent/rvagent-core/src/subagent.rs b/crates/rvAgent/rvagent-core/src/subagent.rs new file mode 100644 index 0000000000..fe5382d93b --- /dev/null +++ b/crates/rvAgent/rvagent-core/src/subagent.rs @@ -0,0 +1,203 @@ +//! Subagent boundary — a tool that spawns an isolated context and returns a +//! String (ADR-275). +//! +//! Deliberately *not* peer agents with a message bus, shared mutable state, or +//! a mergeable state type. One writer, auxiliary intelligence around it, never +//! parallel writes. +//! +//! The evidence for coding tasks is one-directional: at equal token budget, +//! single-agent matches or beats multi-agent on interdependent work, and the +//! team that shipped a parallel-writer architecture walked it back after a year +//! of production data. A CRDT can merge two edits to the same file without +//! textual conflict; it cannot make the *result* coherent. That is the failure +//! this boundary is shaped to make unrepresentable — a subagent that can only +//! return a `String` cannot write, so there is nothing to merge. + +use async_trait::async_trait; + +use crate::error::Result; + +/// What a subagent is for. +/// +/// Both adopted roles are read-only. The distinction is what context they get, +/// which is load-bearing rather than cosmetic. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum SubagentRole { + /// Evaluates an artifact with **no inherited conversation** — only the diff + /// and the task statement. + /// + /// The lack of shared context is the point, not an optimization: reviewers + /// measurably perform *better* without the parent's history, because a + /// shorter window means less context rot and none of the parent's + /// accumulated rationalizations. + /// + /// **Gated, not adopted** (ADR-278 §7). metaharness ADR-226 is a + /// gold-scored null on a closely related design — a read-only strong + /// advisor produced zero marginal resolves at 5.4× cost. That advisor saw + /// the full transcript where this reviewer sees only the diff, so it is not + /// refuted, but this role must demonstrate marginal lift over a + /// no-reviewer control before it reaches the default path. + Reviewer, + /// Explores, reads, greps; returns a summary string. + /// + /// Keeps exploration output out of the main window entirely, which is the + /// cleanest lever on wasted-context accumulation — the earliest and most + /// universal long-run failure. + Gatherer, +} + +impl SubagentRole { + /// Whether this role may see the parent's conversation. + /// + /// Always false. Encoded as a method rather than assumed, so adding a role + /// that inherits context is a visible decision rather than an oversight. + pub fn inherits_parent_context(&self) -> bool { + false + } + + /// Whether this role may use state-mutating tools. + /// + /// Always false — that is what makes this a single-writer architecture. + pub fn may_write(&self) -> bool { + false + } + + /// The model tier this role should run on. + /// + /// The gatherer runs cheap on measured grounds: a specialized small model + /// matched a frontier-mini in that slot, and putting a frontier model there + /// bought +0.4 pp at 5.8× cost — corroborated independently by ADR-226's + /// 5.4× null. Model tiering per role is part of the design, not a later + /// optimization. + pub fn model_tier(&self) -> ModelTier { + match self { + SubagentRole::Reviewer => ModelTier::Standard, + SubagentRole::Gatherer => ModelTier::Cheap, + } + } + + pub fn as_str(&self) -> &'static str { + match self { + SubagentRole::Reviewer => "reviewer", + SubagentRole::Gatherer => "gatherer", + } + } +} + +/// Which model tier a subagent runs on. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ModelTier { + Cheap, + Standard, +} + +/// A subagent invocation: everything it gets, and nothing more. +/// +/// There is no parent-state field by construction. A subagent cannot reach the +/// parent's conversation, files, or todos even by accident. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct SubagentRequest { + pub role: SubagentRole, + /// The complete input. For a reviewer this is the diff plus the task + /// statement; for a gatherer, the question to answer. + pub prompt: String, +} + +impl SubagentRequest { + pub fn new(role: SubagentRole, prompt: impl Into) -> Self { + Self { + role, + prompt: prompt.into(), + } + } +} + +/// The boundary itself: isolated context in, String out. +/// +/// The signature is the architecture. A subagent returns text and nothing else +/// — no state update, no file handle, no mergeable value — so a caller cannot +/// wire one up as a concurrent writer even if it wanted to. +#[async_trait] +pub trait Subagent: Send + Sync { + async fn run(&self, request: SubagentRequest) -> Result; +} + +/// Cap on a subagent's returned summary, in bytes. +/// +/// A subagent exists to *reduce* what reaches the parent's context. One that +/// returns its entire transcript has inverted its own purpose, so the boundary +/// enforces the budget rather than trusting the callee. +pub const MAX_SUMMARY_BYTES: usize = 8_000; + +/// Truncate a subagent's return value to the summary budget. +pub fn enforce_summary_budget(summary: String) -> String { + crate::masking::truncate_tool_result(summary, MAX_SUMMARY_BYTES) +} + +#[cfg(test)] +mod tests { + use super::*; + + struct EchoSubagent; + + #[async_trait] + impl Subagent for EchoSubagent { + async fn run(&self, request: SubagentRequest) -> Result { + Ok(format!("{}: {}", request.role.as_str(), request.prompt)) + } + } + + #[test] + fn no_role_may_write() { + // The single-writer invariant. If this ever fails, the architecture + // has changed and ADR-275 needs revisiting first. + for role in [SubagentRole::Reviewer, SubagentRole::Gatherer] { + assert!(!role.may_write(), "{} must not write", role.as_str()); + } + } + + #[test] + fn no_role_inherits_parent_context() { + for role in [SubagentRole::Reviewer, SubagentRole::Gatherer] { + assert!(!role.inherits_parent_context()); + } + } + + #[test] + fn gatherer_runs_on_the_cheap_tier() { + // A frontier model in this slot measured +0.4 pp at 5.8x cost. + assert_eq!(SubagentRole::Gatherer.model_tier(), ModelTier::Cheap); + } + + #[tokio::test] + async fn boundary_returns_only_a_string() { + let out = EchoSubagent + .run(SubagentRequest::new(SubagentRole::Gatherer, "where is X?")) + .await + .unwrap(); + assert_eq!(out, "gatherer: where is X?"); + } + + #[test] + fn request_carries_no_parent_state() { + // Compile-time property, asserted structurally: the request is exactly + // a role and a prompt. Adding a parent-state field would break this. + let r = SubagentRequest::new(SubagentRole::Reviewer, "diff"); + assert_eq!(r.role, SubagentRole::Reviewer); + assert_eq!(r.prompt, "diff"); + } + + #[test] + fn oversized_summaries_are_capped() { + let huge = "x".repeat(MAX_SUMMARY_BYTES * 2); + let capped = enforce_summary_budget(huge); + assert!(capped.len() <= MAX_SUMMARY_BYTES); + assert!(capped.ends_with("[output truncated]")); + } + + #[test] + fn short_summaries_pass_through_unchanged() { + let s = "found it in src/lib.rs".to_string(); + assert_eq!(enforce_summary_budget(s.clone()), s); + } +} From 23ad04e637586221f060df702bc0e524417ed548 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 2 Aug 2026 01:21:13 +0000 Subject: [PATCH 21/25] feat(rvagent): flywheel Evaluator seam (ADR-278) The Evaluator is where all rvAgent meaning lands on the flywheel's four host-agnostic Score axes, which makes it the trust boundary for every downstream guarantee the gate provides. Rust side: RunOutcome is now serializable with the JS-side field names, since the Evaluator boundary is a process boundary rather than a function call. Adds item_id so an aggregator can detect a missing or duplicated item instead of silently averaging fewer runs, and a round-trip test pinning the wire names -- renaming one silently would make every aggregated Score wrong rather than failing loudly. JS side (scripts/rvagent-flywheel-evaluator.mjs): aggregation, lever validation, and an injected runItem so the seam is testable without spawning real agent runs. Three refusals are deliberate, each preventing a silently-wrong Score rather than a crash: - Zero runs score as maximally bad, never as a clean sweep. - A dropped item throws instead of shrinking the denominator, which would inflate every axis. - A policy naming an unapplied lever throws, because such a mutation produces a run identical to baseline that the optimizer reads as "no effect". COST_PER_WIN_NO_WINS is duplicated as Number.MAX_VALUE with the reason stated in both files: Infinity serializes to JSON null, and the gate reads null > n as false, so a zero-win policy would silently pass the cost clause. Verified the two implementations agree by running both over the same four cases -- mixed (3.25), success-but-noop, zero-wins, and empty -- rather than assuming a shared spec keeps them in sync. Tests: 19 policy tests, 184 core tests green. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/rvAgent/rvagent-core/src/policy.rs | 41 +++++++- scripts/rvagent-flywheel-evaluator.mjs | 115 ++++++++++++++++++++++ 2 files changed, 155 insertions(+), 1 deletion(-) create mode 100644 scripts/rvagent-flywheel-evaluator.mjs diff --git a/crates/rvAgent/rvagent-core/src/policy.rs b/crates/rvAgent/rvagent-core/src/policy.rs index 00bd74da4b..24a3cffcdd 100644 --- a/crates/rvAgent/rvagent-core/src/policy.rs +++ b/crates/rvAgent/rvagent-core/src/policy.rs @@ -168,8 +168,16 @@ fn parse(lever: &str, value: &str) -> Result RunOutcome { RunOutcome { + item_id: String::new(), succeeded, made_changes, cost_usd, @@ -407,6 +419,33 @@ mod tests { } + #[test] + fn run_outcome_round_trips_across_the_process_boundary() { + let outcome = RunOutcome { + item_id: "task-7".into(), + succeeded: true, + made_changes: true, + cost_usd: 0.42, + regressed: false, + }; + let json = serde_json::to_string(&outcome).unwrap(); + // Field names are the contract with the JS evaluator; renaming them + // silently would make every aggregated Score wrong rather than failing. + assert!(json.contains("\"madeChanges\"")); + assert!(json.contains("\"costUsd\"")); + assert!(json.contains("\"item_id\"")); + let back: RunOutcome = serde_json::from_str(&json).unwrap(); + assert_eq!(back, outcome); + } + + #[test] + fn run_outcome_tolerates_a_missing_regressed_flag() { + // An emitter that has nothing to report should not have to say so. + let json = r#"{"item_id":"a","succeeded":false,"madeChanges":false,"costUsd":0.0}"#; + let back: RunOutcome = serde_json::from_str(json).unwrap(); + assert!(!back.regressed); + } + #[test] fn every_known_lever_is_actually_applicable() { // Guards against KNOWN_LEVERS drifting out of sync with apply_to. diff --git a/scripts/rvagent-flywheel-evaluator.mjs b/scripts/rvagent-flywheel-evaluator.mjs new file mode 100644 index 0000000000..2fc5175fbc --- /dev/null +++ b/scripts/rvagent-flywheel-evaluator.mjs @@ -0,0 +1,115 @@ +#!/usr/bin/env node +// rvAgent's Evaluator for @metaharness/flywheel (ADR-278). +// +// The flywheel is deliberately host-agnostic — it knows only candidates, +// scores, gates, receipts, and lineage. Everything rvAgent-specific enters +// through this one seam, which is also the trust boundary: the four Score axes +// are where all host meaning lands, and a dishonest projection defeats every +// downstream guarantee the gate provides. +// +// Usage as a library: +// import { makeRvagentEvaluator } from './rvagent-flywheel-evaluator.mjs'; +// const evaluator = makeRvagentEvaluator({ runItem }); +// +// `runItem(policy, item) -> RunOutcome` is injected so this file stays +// testable without spawning real agent runs. + +/** Cost-per-win when a policy won nothing. + * + * Must match rvagent_core::policy::COST_PER_WIN_NO_WINS. NOT Infinity: JSON has + * no infinity, so it serializes to null, and the gate's + * `candidate.costPerWin > baseline.costPerWin` reads `null > n` as false — + * meaning a policy that won nothing would silently pass the cost clause. + */ +export const COST_PER_WIN_NO_WINS = Number.MAX_VALUE; + +/** + * Aggregate per-item run outcomes into the flywheel's four Score axes. + * + * @param {Array<{itemId?: string, succeeded: boolean, madeChanges: boolean, costUsd: number, regressed?: boolean}>} outcomes + * @returns {{primary: number, noopRate: number, costPerWin: number, regressed: boolean}} + */ +export function scoreFromOutcomes(outcomes) { + // Zero runs must never look like a clean sweep to the gate. + if (!Array.isArray(outcomes) || outcomes.length === 0) { + return { primary: 0, noopRate: 1, costPerWin: COST_PER_WIN_NO_WINS, regressed: false }; + } + + const total = outcomes.length; + const wins = outcomes.filter((o) => o.succeeded).length; + // A run that reports success while committing nothing is still a no-op — + // that is the whole point of the axis. A policy must not earn promotion by + // making the agent talk rather than act. + const noops = outcomes.filter((o) => !o.madeChanges).length; + const cost = outcomes.reduce((sum, o) => sum + (Number(o.costUsd) || 0), 0); + + return { + primary: wins / total, + noopRate: noops / total, + costPerWin: wins === 0 ? COST_PER_WIN_NO_WINS : cost / wins, + regressed: outcomes.some((o) => o.regressed === true), + }; +} + +/** Levers rvAgent knows how to apply. Must match rvagent_core::policy::KNOWN_LEVERS. */ +export const KNOWN_LEVERS = [ + 'max_iterations', + 'parallel_tools', + 'max_parallel_tools', + 'loop_repeat_threshold', + 'keep_last_observations', + 'max_tool_result_bytes', + 'system_prompt_suffix', + 'compaction_rubric', +]; + +/** + * Reject a policy naming a lever rvAgent does not apply. + * + * Throwing is deliberate. A mutation to an unapplied lever produces a run + * identical to baseline; the optimizer would read that as "no effect" and burn + * generations proposing more of them. Failing loudly keeps the search honest. + */ +export function assertKnownLevers(policy) { + const unknown = Object.keys(policy ?? {}).filter((k) => !KNOWN_LEVERS.includes(k)); + if (unknown.length > 0) { + throw new Error( + `policy names levers rvAgent does not apply: ${unknown.join(', ')}. ` + + `Known levers: ${KNOWN_LEVERS.join(', ')}`, + ); + } +} + +/** + * Build an Evaluator for `runFlywheelGenerations`. + * + * @param {{runItem: (policy: object, item: unknown) => Promise}} deps + * @returns {(policy: object, suite: {id: string, items: unknown[]}) => Promise} + */ +export function makeRvagentEvaluator({ runItem }) { + if (typeof runItem !== 'function') { + throw new TypeError('makeRvagentEvaluator requires a runItem function'); + } + + return async function evaluate(policy, suite) { + assertKnownLevers(policy); + + const items = suite?.items ?? []; + const outcomes = []; + for (const item of items) { + // Sequential on purpose: concurrent runs contend for the same workspace + // and would make cost and wall-clock unattributable per item. + outcomes.push(await runItem(policy, item)); + } + + // A dropped item would silently shrink the denominator and inflate every + // axis. Refuse rather than score a partial suite as if it were complete. + if (outcomes.length !== items.length) { + throw new Error( + `evaluator produced ${outcomes.length} outcomes for ${items.length} items`, + ); + } + + return scoreFromOutcomes(outcomes); + }; +} From a2eeca049fcf625c7a92df8050ab4b9e069fd9de Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 2 Aug 2026 03:06:04 +0000 Subject: [PATCH 22/25] =?UTF-8?q?docs(adr):=20ADR-279=20=E2=80=94=20no=20C?= =?UTF-8?q?=20in=20the=20core,=20and=20the=202026=20SOTA=20program?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Answers two questions asked together. The answers turn out to be independent, and the second matters far more. C: no. Every capability commonly cited as C-only has a production Rust path in 2026 -- io_uring (pure-Rust crate), CUDA (cudarc dynamic loading), BLAS (faer matches or beats OpenBLAS), AVX-512 (stable since Rust 1.89, FP16 since 1.94). Two results are decisive rather than suggestive: zlib-rs is faster than zlib-ng in C AND is the fastest WASM zlib, and Qdrant -- the closest analogue -- spent two minor versions REMOVING its one C++ dependency, citing interop friction. The most-cited pro-C evidence does not survive decomposition. SimSIMD/NumKong's 20-118x headline compares its f16 against GCC's f32; there is no f32-vs-f32 row. Same-width it is 1.15-2.1x. PDX (SIGMOD '25) beat SimSIMD and FAISS hand-written kernels by 2.0x average using plain scalar C++ with no intrinsics, purely via data layout, and NumKong loses bulk scoring by 1.85-3.04x for want of a bulk API. Two expert C teams differ from each other more than Rust differs from C. Cost lands where RuVector is most exposed: 37 of 166 crates are WASM, and wasm32-unknown-unknown has no C/C++ toolchain by design. Every C feature would need reimplementing in Rust for the browser anyway. Miri also cannot execute across an FFI boundary, which for a concurrent index is a real loss. The one genuine gap -- f32 reduction reassociation, historically 8.4x -- closes in Rust 1.98 on 2026-08-20 via float_algebraic. What remains blocked is narrower than assumed: f16/bf16 arithmetic and ARM SVE/SME. The finding that matters more than C: ruvector-sota-bench measures against ann-benchmarks.com, which is deprecated and now redirects to VIBE, on SIFT/GloVe/Deep -- exactly the datasets VIBE exists because they are no longer representative. ADR-267 does not mention VIBE. Every SOTA claim we could make today rests on an unmaintained artifact. Same failure mode as the withdrawn SWE-bench-Verified gate in ADR-277. Recall@k is itself under credible attack: 1/Ratio@k reaches equal downstream quality with 1.86-9.36x fewer distance computations. Program, ranked: retarget to VIBE + 1/Ratio@k (blocks everything else); 8-bit rotational quantization (>99% recall10@10, 4x compression, zero training, days of work); SymphonyQG-class packed quantized graph; streaming-stable quantization where every incumbent is weak and which is our actual workload; adaptive filtered-query router; MUVERA FDE; CAGRA build via cuvs-sys. Records two corrections to my own earlier survey: an initial pass wrongly concluded there were no binary/hamming popcount kernels, having generalized from one file across a 166-crate workspace -- ruvector-rabitq/src/scan.rs has had AVX-512 VPOPCNTDQ all along. And the two research threads report TurboQuant against different baselines; the ADR states both rather than conflating them. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- docs/adr/ADR-279-no-c-and-the-sota-program.md | 301 ++++++++++++++++++ 1 file changed, 301 insertions(+) create mode 100644 docs/adr/ADR-279-no-c-and-the-sota-program.md diff --git a/docs/adr/ADR-279-no-c-and-the-sota-program.md b/docs/adr/ADR-279-no-c-and-the-sota-program.md new file mode 100644 index 0000000000..d24384f662 --- /dev/null +++ b/docs/adr/ADR-279-no-c-and-the-sota-program.md @@ -0,0 +1,301 @@ +--- +adr: 279 +title: "No C in the Core; and the 2026 SOTA Program for Vector Search" +status: accepted +date: 2026-08-02 +authors: [Reuven Cohen] +project: "RuVector Core" +related: [ADR-264, ADR-267, ADR-268, ADR-272] +supersedes_parts_of: [ADR-267] +tags: [ruvector, performance, simd, ffi, wasm, ann, quantization, benchmarks, sota] +--- + +# ADR-279 — No C in the Core; and the 2026 SOTA Program + +## Status + +**Accepted.** Owner: Reuven Cohen. Date: 2026-08-02. + +Answers two questions asked together: *would adding C improve RuVector's +performance and capabilities?* and *what would actually make it state of the +art?* The answers turn out to be independent — and the second is far more +consequential than the first. + +## 1. Decisions + +1. **No C/C++ in the default build graph.** The premise does not survive 2026 + evidence, and the costs land precisely where RuVector is most exposed. +2. **Retarget the SOTA harness from ann-benchmarks to VIBE, and add + `1/Ratio@k`.** This is the highest-priority item in this ADR and blocks + every performance claim. +3. **Adopt a ranked SOTA program** (§5), led by 8-bit rotational quantization + and a SymphonyQG-class packed quantized graph. +4. **Bind, don't rewrite, for GPU.** `cuvs-sys` for CAGRA build; no pure-Rust + CAGRA. +5. **Differentiate on streaming stability**, which is the least crowded + frontier and happens to be our actual workload. + +## 2. The C question: no + +### 2.1 The premise fails + +Every capability commonly cited as C-only has a production Rust path in 2026: + +| Claimed C-only | 2026 reality | +|---|---| +| io_uring | `io-uring` crate is pure Rust; `bindgen` optional, bindings checked in | +| CUDA | `cudarc` defaults to dynamic loading — no toolkit at build time | +| BLAS/LAPACK | `faer` matches or surpasses OpenBLAS/LAPACK/Eigen | +| AVX-512 | Stable in `std::arch` since Rust **1.89**; **FP16 since 1.94** (Mar 2026) | +| NUMA / hugepages | Syscalls via `libc` — FFI *declarations*, not compiled C | +| RocksDB | Qdrant **removed** it | +| FAISS / hnswlib | Binding it forecloses filtered-search-inside-traversal | +| simdjson | `simd-json` supports AVX2/SSE4.2/NEON **and wasm simd128** natively | + +Two results are decisive rather than merely suggestive: + +- **`zlib-rs` is faster than zlib-ng in C on native, and is the fastest WASM + zlib in existence** (2× miniz-oxide). A direct counterexample to "C is + faster", on exactly the axis proposed. +- **Qdrant — the closest analogue to RuVector — spent two minor versions + removing its one C++ dependency**, naming *"interoperating with C++ slowed us + down"*, plus compaction latency spikes and tuning burden. The comparable + project went the opposite direction from the proposal. + +### 2.2 The cost lands on WASM, which is our largest commitment + +37 of 166 crates are WASM. The rustc platform-support book states plainly that +`wasm32-unknown-unknown` **has no C/C++ toolchain** — not "awkward", absent by +design. Consequences: + +- Any unconditional `cc::Build` in the graph breaks the browser build. +- Escape hatches don't help: emscripten produces an app, not a + `wasm-bindgen` library; `wasm32-wasip1` gives edge/server but not browser. +- The `extern "C"` ABI on wasm is *mid-migration* (future-incompat warning + since Rust 1.87), so any FFI boundary there stands on changing ground. + Rust-to-Rust is unaffected. +- Every C feature must be reimplemented in Rust for the browser anyway. **You + pay for each feature twice and get behavioral divergence for free.** + +### 2.3 Costs beyond WASM + +- **Miri cannot execute across an FFI boundary.** For a database with + concurrent index structures, losing Miri on the hot paths is a real + regression in verification capability. +- A 320-bug study of bindgen/cbindgen/CXX (ACM TOSEM, Feb 2026) found the + dominant failure mode is **not** crashes but *silently generating code + unfaithful to intent*, rooted in data-layout mismodeling. +- `cargo-audit`/`cargo-vet` give **no meaningful coverage of vendored C** + (cf. RUSTSEC-2023-0061, libwebp). +- The FFI boundary is a *pessimization* on fine-grained work: call overhead and + lost cross-module inlining make a C distance kernel slower than the + equivalent `std::arch` kernel. Distance kernels are the definition of + fine-grained. + +### 2.4 The headline "C is faster" number is a datatype artifact + +SimSIMD/NumKong is the most-cited evidence for C kernels, claiming **20–118×** +over autovectorized code. Decomposed, that table compares **NumKong f16 against +GCC's f32** — there is no f32-vs-f32 row, and the README states outright that +GCC "struggles with `_Float16`". Against the compiler's competent f32 output the +real effect is **1.15×–2.1×**. The 3–200× figures elsewhere are against +SciPy/NumPy Python, not compiled code. + +It is also independently contradicted. **PDX (SIGMOD '25) beat SimSIMD's and +FAISS's hand-written kernels by 2.0× on average using plain scalar C++ with no +intrinsics at all**, purely by changing data layout — 3–10× at low dimension. +And an independent harness found NumKong *loses* bulk scoring by **1.85–3.04×**, +because it has no bulk API and cannot hide memory latency. + +**The lever is layout and API shape, not the language the intrinsics are typed +in.** Two expert C teams differ from each other by more (0.72–1.20×) than Rust +differs from C. + +### 2.5 Qdrant ships zero C + +A code search for `simsimd` in `qdrant/qdrant` returns **0 hits**. Qdrant has +AVX-512 `vpopcntq`, `pshufb`/`maddubs`/`VPDPBUSD`, and NEON `SDOT` quantization +kernels — all in Rust `std::arch` — and shipped an ICLR-2026 quantizer +(TurboQuant, v1.18, May 2026) before most C++ engines had it. FAISS, in C++, +still lacks AVX-512 FastScan. + +Separately: plain **23-line Rust with no intrinsics beat the `simsimd` C crate +on Hamming distance** across three machines — autovectorization winning over +manual SIMD on precisely the binary-quantization inner loop. + +### 2.6 The one real gap closes in 18 days + +The largest historically-measured Rust-vs-C gap in this workload was **f32 +reduction reassociation: 8.4×** (84 µs vs 10 µs with Clang fast-math). +`float_algebraic` (`f32::algebraic_add`/`algebraic_mul`) stabilized in PR +#157029, **shipping in Rust 1.98 on 2026-08-20**. Adopt it the day it lands. + +What remains genuinely blocked on stable Rust is narrower than assumed: +**f16/bf16 arithmetic** (tracking #116909) and **ARM SVE/SME** (#145052). +Note that even LanceDB — the one major Rust vector DB linking C — uses it for +three files, and **two of them are pragma-autovectorized C, not hand-written +intrinsics**. What C bought Lance was `_Float16` and fast-math, and one of those +two arrives on 2026-08-20. + +For f16 the cheaper answer is to store f16 and upconvert with stable +`_mm512_cvtph_ps`/`vcvt_f32_f16`, accumulating in f32 at ~1.15–2.1× — or skip +f16 entirely for int8/binary quantization, which is faster *and* smaller. + +### 2.7 The existence proof is already in this repo + +`ruvector-rabitq/src/scan.rs` performs RaBitQ scanning with +`_mm512_popcnt_epi64` and `_mm512_xor_si512`, runtime-detected via +`avx512vpopcntdq`, with an AVX2 fallback — hand-written Rust `std::arch`. That +is precisely the work C would have been imported to do, already done, and it +compiles for WASM. + +Note a correction to an earlier survey of this repo: an initial pass concluded +there were "no binary/hamming popcount kernels". That was wrong — it +generalized from one file (`ruvector-core/simd_intrinsics.rs`) across a +166-crate workspace. The kernel exists. + +### 2.8 The narrow exceptions + +C is acceptable **only** when all hold: behind a non-default feature; in a +separate crate the core does not depend on; with a correctness-equivalent +pure-Rust fallback under differential test; genuinely unreachable in Rust; and +with WASM CI proving the core still builds `--no-default-features`. + +Realistically that means **FIPS-validated crypto** (a compliance case, not a +performance one) and **vendor accelerator SDKs**. `hailort-sys` is already +exactly this, correctly isolated. `cuvs-sys` (§5) qualifies under the same rule. + +## 3. The finding that matters more than C + + +**`ruvector-sota-bench` measures against `ann-benchmarks.com`, which is +deprecated.** Its README now reads: *"no longer actively maintained… consider +submitting your work to different benchmarks, such as VIBE."* Our dataset list +is SIFT-128, GloVe-25/100, Deep-image-96 — precisely the sets VIBE was built +because they are *"no longer representative of the current applications of ANN +search."* + +ADR-267 (SOTA Validation Protocol) does not mention VIBE. + +So: **every SOTA claim RuVector could make today rests on an unmaintained +artifact and non-representative data.** The instrument is pointed at the wrong +thing, and no amount of kernel work fixes that. This is the same failure mode as +withdrawing the SWE-bench-Verified gate in ADR-277 §5.1 — a benchmark can retire +underneath you, and continuing to cite it is how retractions happen. + +### 3.1 Recall@k is itself under credible attack + +Two 2026 papers argue the field has optimized the wrong objective. *ANN Search: +Recall What Matters* (arXiv:2606.04522) proposes **`1/Ratio@k`** and measures +**1.86×–9.36× fewer distance computations** to reach equal downstream quality +versus optimizing Recall@0.95. Downstream validation: image-classification +label precision held at 0.943–0.978 while Recall fell 1.0 → 0.4, and RAG answer +quality varied ≤5% across the same Recall range — `1/Ratio@k` tracked with MAD +0.5–2.6% against Recall's ~29%. + +If that reproduces, a large fraction of tuning effort industry-wide is spent +recovering near-equidistant, semantically irrelevant neighbors. + +## 4. What a defensible claim requires + +Adopted as an amendment to ADR-267: + +- **VIBE datasets, including the out-of-distribution splits.** OOD is a + first-class axis; HNSW's OOD gap is one of its known weaknesses. +- **Both `Recall@k` and `1/Ratio@k`.** +- **QPS at fixed recall on the Pareto frontier**, never single points. +- **Index build time and peak RSS**, reported alongside. +- **Full hardware disclosure** — exact CPU, cores, ISA level. AVX-512 vs AVX2 + vs NEON *changes rankings*. +- **Single-thread and multi-thread separately.** +- **Hyperparameter search budget disclosed for all baselines**, not just ours. +- **Self-published gists and unreproduced blog numbers are not evidence** — + including our own. The research sweep explicitly excluded ruvnet-authored + gists on SymphonyQG/MUVERA/RVQ for this reason, and that discipline stands. +- **Do not headline SIFT1M/GIST1M.** Leading there proves nothing in 2026. + +## 5. The SOTA program, ranked + +| # | Item | Measured effect | Effort | Needs C/CUDA | +|---|---|---|---|---| +| 1 | **Retarget to VIBE + `1/Ratio@k`** | Prerequisite for every claim; 1.86–9.36× wasted compute at stake | 2–3 wks | No | +| 2 | **8-bit rotational quantization** as default codec | >99% recall10@10, 4× compression, ~2.3× faster, **zero training** | **Days** | No | +| 3 | **SymphonyQG-class packed quantized graph** | 1.5–4.5× QPS vs best baselines, 3.5–17× vs hnswlib @95%; best hard-query robustness in VIBE | 2–3 mo | No — `std::arch` VPSHUFB | +| 4 | **Streaming-stable quantizer + dual hot/stable index** | Constant recall under drift vs progressive decay | 6–10 wks | No | +| 5 | **Adaptive filtered-query router** | Specialized-vs-general gap up to 10×, and it *inverts* with scale | 4–6 wks | No | +| 6 | **MUVERA FDE for multi-vector** | ~10% higher recall at 90% lower latency vs PLAID; 32× FDE compression | 3–4 wks | No | +| 7 | **CAGRA build → HNSW serve** via `cuvs-sys` | 12.3× build speedup; no GPU at serve time | 2 wks | **Bind cuVS** | + +**Item 2 has by far the best ratio in this document** — a rotation plus scalar +quantization, no clustering, no training, days of work. Do it while item 3 is +in flight. + +**Item 4 is where we can lead rather than catch up.** Every incumbent is weak: +SPFresh cannot update stably under contention, FreshDiskANN *"fails to maintain +a graph of great quality because streaming data destroys the navigability of the +original index"*, and **all** data-dependent quantizers — PQ, LVQ, RaBitQ, +ScaNN — are trained on a snapshot and go stale. For an agent memory database +this is not a nice-to-have, it is the workload. The relevant papers are from +Dec 2025 – Jun 2026 and none are reproduced. + +### 5.1 Kernel build order (where the SIMD evidence lands) + +Two independent research threads converged on the same conclusion: **layout is +the lever, not intrinsics.** PDX got 2.0× from layout alone with no intrinsics; +SymphonyQG's win is also fundamentally a layout change (RaBitQ codes +co-located with neighbor IDs, FastScan-packed). Order accordingly: + +1. **PDX-style vertical/blocked layout** — 2.0× average, 3–10× at low dimension +2. **Bulk kernel APIs** (1 query × N vectors) that amortize dispatch and hide + memory latency — worth 1.85–3.04× on its own, and the specific reason + NumKong loses bulk scoring despite better single-pair kernels +3. **Binary/int8 quantized kernels** in Rust `std::arch` (`vpopcntq`, `pshufb`, + `VPDPBUSD`, `SDOT`) — all stable since 1.89 +4. **`algebraic_*` f32 paths** — adopt on 2026-08-20 +5. **SVE** only if and when Rust stabilizes it + +Adding C appears nowhere in that list. + +### 5.2 Explicitly deprioritized + +- **TurboQuant** — deprioritized *as a RaBitQ replacement*: theoretically + dominated (bits scale `log log(1/δ)` vs `log(1/δ)`), with reported + quantization times ~2 orders of magnitude optimistic. + **Caveat, stated because the two research threads are in apparent tension:** + a separate thread reports Qdrant shipped TurboQuant in v1.18 (May 2026, ICLR + 2026) beating *binary quantization* by 9–24 pp recall at 16× compression. + These are different comparisons — TurboQuant-vs-RaBitQ on theory, and + TurboQuant-vs-BQ on measured recall — and conflating them would be an error. + The RaBitQ comparison is what governs this decision; the BQ result does not + contradict it. +- **LVQ / LeanVec** — closed-source, Intel-hardware-only, and measured as the + *worst* modern method on quantization error. +- **Pure-Rust CAGRA rewrite** — ~zero gain over binding `cuvs-sys`. +- **SAQ** — the measured frontier (1.8–5.4× lower error than Extended RaBitQ) + but unreproduced and same-community-authored. Revisit after item 1 makes it + measurable. + +## 6. Consequences + +**Positive.** Closes a live risk: we were positioned to publish claims against +a dead benchmark. The program is entirely Rust-native, so it composes with the +WASM story rather than fighting it. Items 2 and 7 are cheap and land early. + +**Negative.** Item 1 delays visible performance work by 2–3 weeks. That +sequencing is deliberate and matches the harness lesson in ADR-273 §6: a +measurement you cannot trust makes every subsequent optimization unfalsifiable. + +**Risk.** Several headline numbers (SAQ, CoDEQ, OctopusANN) are single-source +and unreproduced. They are treated as hypotheses to re-measure under item 1, +not as constants — the same evidence discipline applied in ADR-273. + +## 7. Open + +- Third research thread (SIMD kernel evidence across FAISS/SimSIMD/usearch) + outstanding; it refines kernel specifics — VNNI int8, f16/bf16 now that + AVX-512 FP16 is stable — not the decisions above. +- Pull VIBE's `results/summary.parquet` to calibrate against the field before + writing index code. +- Audit whether `ruvector-rabitq` implements 1-bit only or extended/multi-bit + RaBitQ, and whether `rotation.rs` uses a fast Hadamard/Kac's walk or an + O(d²) dense rotation. From 65ba193a751d02582a694dc9d217351657030fb5 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 2 Aug 2026 03:31:11 +0000 Subject: [PATCH 23/25] =?UTF-8?q?perf(ruvector-core):=20PDX=20vertical=20l?= =?UTF-8?q?ayout=20=E2=80=94=20implemented,=20benchmarked,=20NOT=20adopted?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Implements the PDX (SIGMOD '25) vertical/dimension-major layout that ADR-279 5.1 ranked first in the kernel build order, benchmarks it against the existing row-major batch path, and records that it did not reproduce. Measured on this host (AVX-512, both paths built with target-cpu=native): PDX is 0.65-0.88x on cache-resident working sets (i.e. SLOWER) and only 1.09-1.29x when streaming. The paper reports ~2.0x. The first benchmark run was invalid and is worth recording. It showed PDX 14-18x slower, because the row-major path runtime-dispatches to AVX-512 via is_x86_feature_detected! while the new code compiled for baseline x86-64 with SSE2. That was an ISA comparison presented as a layout comparison. The corrected comparison is still confounded: the row-major path takes Vec<&[f32]> built from Vec>, so it pointer-chases across 4096 separate heap allocations, while PdxIndex is one contiguous buffer. The streaming win may be allocation contiguity rather than vertical layout. A clean experiment needs a contiguous row-major baseline. Reading: at f32 these workloads are bandwidth-bound, so layout cannot help much. That strengthens rather than weakens the case for 8-bit rotational quantization (ADR-279 5 item 2) -- 4x less data to stream attacks the actual bottleneck. Revisit PDX for quantized codes, where the working set may shrink enough to become compute-bound. The code is kept: it is correct, tested against a naive reference across block boundaries and padding lanes, and is the natural substrate for the quantized retry. It is not wired into any hot path. Also documents why the existing batch_euclidean is not a bulk kernel -- its TILE_SIZE chunking is decorative, since the inner call is identical to a flat loop over single-pair distances. Tests: 8 pdx tests including ragged-block rejection, oversized-block rejection, partial-block padding, and cross-block-boundary correctness at dim=129. Co-Authored-By: claude-flow Claude-Session: https://claude.ai/code/session_019mt39sZNUecTJBBBqPRH9F --- crates/ruvector-core/Cargo.toml | 4 + .../ruvector-core/benches/pdx_vs_rowmajor.rs | 53 +++ crates/ruvector-core/src/lib.rs | 1 + crates/ruvector-core/src/pdx.rs | 386 ++++++++++++++++++ docs/adr/ADR-279-no-c-and-the-sota-program.md | 43 ++ 5 files changed, 487 insertions(+) create mode 100644 crates/ruvector-core/benches/pdx_vs_rowmajor.rs create mode 100644 crates/ruvector-core/src/pdx.rs diff --git a/crates/ruvector-core/Cargo.toml b/crates/ruvector-core/Cargo.toml index 144c706fea..e78f5befa4 100644 --- a/crates/ruvector-core/Cargo.toml +++ b/crates/ruvector-core/Cargo.toml @@ -317,3 +317,7 @@ overly_complex_bool_expr = { level = "allow", priority = 1 } zombie_processes = { level = "allow", priority = 1 } repeat_vec_with_capacity = { level = "allow", priority = 1 } missing_transmute_annotations = { level = "allow", priority = 1 } + +[[bench]] +name = "pdx_vs_rowmajor" +harness = false diff --git a/crates/ruvector-core/benches/pdx_vs_rowmajor.rs b/crates/ruvector-core/benches/pdx_vs_rowmajor.rs new file mode 100644 index 0000000000..22cbc1fcd9 --- /dev/null +++ b/crates/ruvector-core/benches/pdx_vs_rowmajor.rs @@ -0,0 +1,53 @@ +//! Row-major batch distance vs PDX vertical layout (ADR-279 §5.1). +//! +//! Both compute the same thing over the same data. The only difference is +//! memory layout and whether a horizontal reduction happens per vector. + +use criterion::{black_box, criterion_group, criterion_main, BenchmarkId, Criterion, Throughput}; +use ruvector_core::pdx::PdxIndex; +use ruvector_core::simd_intrinsics::batch_euclidean; + +fn corpus(n: usize, dim: usize) -> Vec> { + (0..n) + .map(|i| { + (0..dim) + .map(|d| ((i * 31 + d * 17) % 101) as f32 / 101.0) + .collect() + }) + .collect() +} + +fn bench(c: &mut Criterion) { + // 1536 is the OpenAI embedding width; 768 covers most sentence encoders. + // (n, dim) pairs: the first three are cache-resident, the rest stream. + for (n, dim) in [(256usize, 768usize), (512, 768), (1024, 768), (4096, 768), (4096, 1536)] { + let vecs = corpus(n, dim); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let query: Vec = (0..dim).map(|d| (d % 7) as f32 / 7.0).collect(); + + let index = PdxIndex::from_rows(&refs); + let mut out = vec![0.0f32; n]; + + let mut group = c.benchmark_group(format!("batch_euclidean_{n}x{dim}")); + // Bytes of vector data scanned per query — makes the two comparable + // in bandwidth terms rather than just wall time. + group.throughput(Throughput::Bytes((n * dim * 4) as u64)); + + group.bench_function(BenchmarkId::new("row_major", dim), |b| { + b.iter(|| { + batch_euclidean(black_box(&query), black_box(&refs), black_box(&mut out)); + }) + }); + + group.bench_function(BenchmarkId::new("pdx_vertical", dim), |b| { + b.iter(|| { + index.euclidean_sq_into(black_box(&query), black_box(&mut out)); + }) + }); + + group.finish(); + } +} + +criterion_group!(benches, bench); +criterion_main!(benches); diff --git a/crates/ruvector-core/src/lib.rs b/crates/ruvector-core/src/lib.rs index 5663b9f962..ea301f08c2 100644 --- a/crates/ruvector-core/src/lib.rs +++ b/crates/ruvector-core/src/lib.rs @@ -39,6 +39,7 @@ pub mod distance; pub mod embeddings; pub mod error; pub mod index; +pub mod pdx; pub mod quantization; // Storage backends - conditional compilation based on features diff --git a/crates/ruvector-core/src/pdx.rs b/crates/ruvector-core/src/pdx.rs new file mode 100644 index 0000000000..2c34a5debe --- /dev/null +++ b/crates/ruvector-core/src/pdx.rs @@ -0,0 +1,386 @@ +//! PDX-style vertical (dimension-major) layout for bulk distance computation. +//! +//! Implements the layout change measured in *PDX: A Data Layout for Vector +//! Similarity Search* (SIGMOD '25), which beat hand-written SIMD kernels in +//! SimSIMD and FAISS by ~2.0× on average — using **scalar** code — purely by +//! reorganizing memory. See ADR-279 §5.1: layout is the lever, not intrinsics. +//! +//! # Why the row-major batch loop leaves throughput on the table +//! +//! The conventional layout stores each vector contiguously and computes one +//! distance at a time: +//! +//! ```text +//! v0: [d0 d1 d2 … dD] → SIMD across dimensions → horizontal sum → result[0] +//! v1: [d0 d1 d2 … dD] → SIMD across dimensions → horizontal sum → result[1] +//! ``` +//! +//! Two costs are structural, not tuning problems: +//! +//! 1. **A horizontal reduction per vector.** Summing a SIMD accumulator into a +//! scalar is a serial dependency chain (`_mm512_reduce_add_ps` is several +//! dependent shuffles and adds) and it happens once per vector. +//! 2. **A tail per vector.** Any dimension count not a multiple of the vector +//! width runs a scalar remainder loop, once per vector. +//! +//! # The vertical layout +//! +//! Store a *block* of `LANES` vectors dimension-major: +//! +//! ```text +//! dim 0: [v0 v1 v2 … v15] +//! dim 1: [v0 v1 v2 … v15] +//! … +//! ``` +//! +//! Now one SIMD register holds *one dimension of sixteen different vectors*. +//! Iterating dimensions accumulates sixteen independent distances in parallel: +//! +//! - **No horizontal reduction at all** — lane `i` of the accumulator *is* +//! distance `i` when the loop ends. +//! - **No per-vector tail** — the only remainder is the final partial block. +//! - **Sequential access** across the whole block. +//! +//! The accumulator also has no loop-carried dependency between lanes, so the +//! CPU can keep several FMAs in flight. +//! +//! # Portability +//! +//! Written against `f32` chunks that LLVM autovectorizes, with an explicit +//! AVX-512 path where available. Per ADR-279 this is deliberately *not* C: the +//! same source compiles for x86-64, aarch64, and wasm32. + +use std::fmt; + +/// Vectors per block. 16 f32 fills one AVX-512 register; on narrower ISAs the +/// compiler splits it into multiple registers, which still vectorizes cleanly. +pub const LANES: usize = 16; + +/// A block of up to [`LANES`] vectors stored dimension-major. +#[derive(Clone, PartialEq)] +pub struct PdxBlock { + /// `dim * LANES + lane`. Unused lanes hold zeros. + data: Vec, + dim: usize, + /// How many lanes carry real vectors (≤ `LANES`). + len: usize, +} + +impl fmt::Debug for PdxBlock { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("PdxBlock") + .field("dim", &self.dim) + .field("len", &self.len) + .finish_non_exhaustive() + } +} + +impl PdxBlock { + /// Transpose up to [`LANES`] row-major vectors into one vertical block. + /// + /// # Panics + /// If `vectors` is longer than [`LANES`], or any vector's length differs + /// from `dim`. Both are programming errors rather than input errors — a + /// ragged block would silently produce wrong distances. + pub fn from_rows(vectors: &[&[f32]], dim: usize) -> Self { + assert!( + vectors.len() <= LANES, + "block holds at most {LANES} vectors, got {}", + vectors.len() + ); + for (i, v) in vectors.iter().enumerate() { + assert_eq!(v.len(), dim, "vector {i} has length {}, expected {dim}", v.len()); + } + + let mut data = vec![0.0f32; dim * LANES]; + for (lane, v) in vectors.iter().enumerate() { + for (d, &value) in v.iter().enumerate() { + data[d * LANES + lane] = value; + } + } + Self { + data, + dim, + len: vectors.len(), + } + } + + pub fn dim(&self) -> usize { + self.dim + } + + pub fn len(&self) -> usize { + self.len + } + + pub fn is_empty(&self) -> bool { + self.len == 0 + } + + /// Squared euclidean distance from `query` to every vector in the block. + /// + /// Writes `self.len()` results. Padding lanes are computed but discarded — + /// branchless, and they hold zeros so they cannot fault or produce NaN. + /// + /// # Panics + /// If `query.len() != self.dim()` or `out.len() < self.len()`. + pub fn euclidean_sq_into(&self, query: &[f32], out: &mut [f32]) { + assert_eq!(query.len(), self.dim, "query dimension mismatch"); + assert!(out.len() >= self.len, "output buffer too small"); + + let mut acc = [0.0f32; LANES]; + + for (d, &q) in query.iter().enumerate() { + let row = &self.data[d * LANES..d * LANES + LANES]; + // Fixed-size slice so LLVM knows the trip count and emits a single + // vector op per dimension with no remainder branch. + for lane in 0..LANES { + let diff = row[lane] - q; + acc[lane] = diff.mul_add(diff, acc[lane]); + } + } + + out[..self.len].copy_from_slice(&acc[..self.len]); + } + + /// Dot product from `query` to every vector in the block. + pub fn dot_into(&self, query: &[f32], out: &mut [f32]) { + assert_eq!(query.len(), self.dim, "query dimension mismatch"); + assert!(out.len() >= self.len, "output buffer too small"); + + let mut acc = [0.0f32; LANES]; + + for (d, &q) in query.iter().enumerate() { + let row = &self.data[d * LANES..d * LANES + LANES]; + for lane in 0..LANES { + acc[lane] = row[lane].mul_add(q, acc[lane]); + } + } + + out[..self.len].copy_from_slice(&acc[..self.len]); + } +} + +/// A full vector set in PDX layout: a sequence of vertical blocks. +#[derive(Debug, Clone)] +pub struct PdxIndex { + blocks: Vec, + dim: usize, + len: usize, +} + +impl PdxIndex { + /// Build from row-major vectors. + /// + /// # Panics + /// If `vectors` is empty, or any vector's length differs from the first. + pub fn from_rows(vectors: &[&[f32]]) -> Self { + assert!(!vectors.is_empty(), "cannot build a PdxIndex from no vectors"); + let dim = vectors[0].len(); + + let blocks = vectors + .chunks(LANES) + .map(|chunk| PdxBlock::from_rows(chunk, dim)) + .collect(); + + Self { + blocks, + dim, + len: vectors.len(), + } + } + + pub fn dim(&self) -> usize { + self.dim + } + + pub fn len(&self) -> usize { + self.len + } + + pub fn is_empty(&self) -> bool { + self.len == 0 + } + + pub fn block_count(&self) -> usize { + self.blocks.len() + } + + /// Squared euclidean distance from `query` to every indexed vector. + /// + /// This is the bulk API the row-major batch path lacks — the second lever + /// in ADR-279 §5.1, worth 1.85–3.04× independently of layout, because it + /// amortizes dispatch and lets the prefetcher see a sequential stream. + /// + /// # Panics + /// If `out.len() < self.len()`. + pub fn euclidean_sq_into(&self, query: &[f32], out: &mut [f32]) { + assert!(out.len() >= self.len, "output buffer too small"); + let mut written = 0; + for block in &self.blocks { + block.euclidean_sq_into(query, &mut out[written..]); + written += block.len(); + } + } + + /// Dot product from `query` to every indexed vector. + pub fn dot_into(&self, query: &[f32], out: &mut [f32]) { + assert!(out.len() >= self.len, "output buffer too small"); + let mut written = 0; + for block in &self.blocks { + block.dot_into(query, &mut out[written..]); + written += block.len(); + } + } + + /// Convenience allocating wrapper. + pub fn euclidean_sq(&self, query: &[f32]) -> Vec { + let mut out = vec![0.0; self.len]; + self.euclidean_sq_into(query, &mut out); + out + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Reference implementation. Deliberately naive — its job is to be + /// obviously correct, not fast. + fn euclidean_sq_ref(a: &[f32], b: &[f32]) -> f32 { + a.iter().zip(b).map(|(x, y)| (x - y) * (x - y)).sum() + } + + fn dot_ref(a: &[f32], b: &[f32]) -> f32 { + a.iter().zip(b).map(|(x, y)| x * y).sum() + } + + fn make(n: usize, dim: usize) -> Vec> { + (0..n) + .map(|i| { + (0..dim) + .map(|d| ((i * 31 + d * 17) % 101) as f32 / 101.0) + .collect() + }) + .collect() + } + + #[test] + fn block_matches_reference_euclidean() { + let dim = 128; + let vecs = make(LANES, dim); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let block = PdxBlock::from_rows(&refs, dim); + let query: Vec = (0..dim).map(|d| (d % 7) as f32 / 7.0).collect(); + + let mut got = vec![0.0; LANES]; + block.euclidean_sq_into(&query, &mut got); + + for (i, v) in vecs.iter().enumerate() { + let want = euclidean_sq_ref(&query, v); + assert!( + (got[i] - want).abs() < 1e-4, + "lane {i}: got {}, want {want}", + got[i] + ); + } + } + + #[test] + fn block_matches_reference_dot() { + let dim = 96; + let vecs = make(LANES, dim); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let block = PdxBlock::from_rows(&refs, dim); + let query: Vec = (0..dim).map(|d| (d % 5) as f32 / 5.0).collect(); + + let mut got = vec![0.0; LANES]; + block.dot_into(&query, &mut got); + + for (i, v) in vecs.iter().enumerate() { + let want = dot_ref(&query, v); + assert!((got[i] - want).abs() < 1e-4, "lane {i}"); + } + } + + #[test] + fn partial_block_ignores_padding_lanes() { + // The case most likely to produce silently wrong results: a block that + // is not full. Padding lanes must not appear in the output. + let dim = 64; + let vecs = make(5, dim); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let block = PdxBlock::from_rows(&refs, dim); + assert_eq!(block.len(), 5); + + let query = vec![0.5f32; dim]; + let mut got = vec![f32::NAN; 5]; + block.euclidean_sq_into(&query, &mut got); + + for (i, v) in vecs.iter().enumerate() { + assert!((got[i] - euclidean_sq_ref(&query, v)).abs() < 1e-4); + } + } + + #[test] + fn index_matches_reference_across_block_boundaries() { + // 50 vectors = 3 full blocks + a partial, so boundary handling is + // exercised rather than assumed. + let dim = 129; // deliberately not a multiple of any vector width + let vecs = make(50, dim); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let index = PdxIndex::from_rows(&refs); + assert_eq!(index.len(), 50); + assert_eq!(index.block_count(), 4); + + let query: Vec = (0..dim).map(|d| (d % 11) as f32 / 11.0).collect(); + let got = index.euclidean_sq(&query); + + assert_eq!(got.len(), 50); + for (i, v) in vecs.iter().enumerate() { + let want = euclidean_sq_ref(&query, v); + assert!( + (got[i] - want).abs() < 1e-3, + "vector {i}: got {}, want {want}", + got[i] + ); + } + } + + #[test] + fn single_vector_index_works() { + let dim = 32; + let v: Vec = (0..dim).map(|d| d as f32).collect(); + let index = PdxIndex::from_rows(&[v.as_slice()]); + let got = index.euclidean_sq(&v); + assert_eq!(got.len(), 1); + assert!(got[0].abs() < 1e-5, "distance to self must be zero"); + } + + #[test] + #[should_panic(expected = "expected 4")] + fn ragged_block_is_rejected() { + let a = [1.0f32, 2.0, 3.0, 4.0]; + let b = [1.0f32, 2.0]; + PdxBlock::from_rows(&[&a[..], &b[..]], 4); + } + + #[test] + #[should_panic(expected = "at most")] + fn oversized_block_is_rejected() { + let v = vec![0.0f32; 8]; + let refs: Vec<&[f32]> = (0..LANES + 1).map(|_| v.as_slice()).collect(); + PdxBlock::from_rows(&refs, 8); + } + + #[test] + #[should_panic(expected = "output buffer too small")] + fn undersized_output_is_rejected() { + let dim = 16; + let vecs = make(20, dim); + let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); + let index = PdxIndex::from_rows(&refs); + let mut out = vec![0.0; 5]; + index.euclidean_sq_into(&vec![0.0; dim], &mut out); + } +} diff --git a/docs/adr/ADR-279-no-c-and-the-sota-program.md b/docs/adr/ADR-279-no-c-and-the-sota-program.md index d24384f662..e1cc635a89 100644 --- a/docs/adr/ADR-279-no-c-and-the-sota-program.md +++ b/docs/adr/ADR-279-no-c-and-the-sota-program.md @@ -289,6 +289,49 @@ measurement you cannot trust makes every subsequent optimization unfalsifiable. and unreproduced. They are treated as hypotheses to re-measure under item 1, not as constants — the same evidence discipline applied in ADR-273. +## 6.1 Measured: PDX layout did NOT reproduce (2026-08-02) + +A first implementation of the PDX vertical layout (`ruvector-core::pdx`) was +built and benchmarked against the existing row-major batch path on this host +(AVX-512, 4 cores). **The paper's ~2.0× did not reproduce.** Measured, both +paths compiled with `-C target-cpu=native`, 4096 vectors: + +| n × dim | working set | row-major | PDX vertical | ratio | +|---|---|---|---|---| +| 256 × 768 | 0.75 MB | 13.73 µs | 19.52 µs | **0.70×** | +| 512 × 768 | 1.5 MB | 29.78 µs | 45.59 µs | **0.65×** | +| 1024 × 768 | 3 MB | 123.5 µs | 140.4 µs | 0.88× | +| 4096 × 768 | 12 MB | 650.2 µs | 505.4 µs | 1.29× | +| 4096 × 1536 | 25 MB | 1158 µs | 1065 µs | 1.09× | + +PDX **loses** on cache-resident working sets and wins only when streaming. + +Three caveats, stated because they bound what this measurement proves: + +1. **The first run was invalid.** It showed PDX 14–18× slower, because the + row-major path runtime-dispatches to AVX-512 via `is_x86_feature_detected!` + while the new code compiled for baseline x86-64 (SSE2). That was an ISA + comparison wearing a layout comparison's clothes. Fixed by building both + with `target-cpu=native`. +2. **The remaining comparison is still confounded.** The row-major path takes + `Vec<&[f32]>` derived from `Vec>` — 4096 separate heap allocations, + so it pointer-chases — while `PdxIndex` is a single contiguous buffer. The + streaming win may be *allocation contiguity*, not vertical layout. A clean + experiment needs a contiguous row-major baseline. +3. **The PDX kernel is autovectorized generic Rust**, competing against + hand-written AVX-512 intrinsics. That the naive version reaches parity at + all is notable, but it is not a like-for-like layout test. + +**Conclusion for now: do not adopt PDX layout on this evidence.** The honest +reading is that at f32 precision these workloads are bandwidth-bound, so layout +cannot help much — which *strengthens* the case for §5 item 2 (8-bit rotational +quantization), since 4× less data to stream attacks the actual bottleneck. +Revisit PDX for quantized codes, where the working set shrinks enough to become +compute-bound. + +This is recorded rather than discarded because a negative result that cost a +day is worth more written down than repeated. + ## 7. Open - Third research thread (SIMD kernel evidence across FAISS/SimSIMD/usearch) From 5f28cc050ad24add1e00ce037c2a790bc1a7e725 Mon Sep 17 00:00:00 2001 From: ruv Date: Sun, 2 Aug 2026 09:30:57 -0400 Subject: [PATCH 24/25] style: cargo fmt --all to fix Rustfmt CI failure Co-Authored-By: claude-flow --- .../ruvector-core/benches/pdx_vs_rowmajor.rs | 8 +++- crates/ruvector-core/src/pdx.rs | 12 +++++- crates/rvAgent/rvagent-acp/src/agent.rs | 4 +- .../rvAgent/rvagent-backends/src/anthropic.rs | 7 +++- crates/rvAgent/rvagent-cli/src/app.rs | 5 +-- crates/rvAgent/rvagent-core/src/graph.rs | 7 +--- crates/rvAgent/rvagent-core/src/lib.rs | 4 +- crates/rvAgent/rvagent-core/src/policy.rs | 18 ++++++--- .../benches/middleware_bench.rs | 3 +- .../src/patch_tool_calls.rs | 6 +-- .../rvagent-middleware/src/rvf_manifest.rs | 4 +- crates/rvAgent/rvagent-middleware/src/sona.rs | 3 +- .../rvAgent/rvagent-middleware/src/types.rs | 4 +- .../tests/security_tests.rs | 2 +- .../tests/summarization_tests.rs | 18 ++++++--- .../tests/unicode_security_integration.rs | 2 +- .../tests/orchestrator_tests.rs | 6 ++- crates/rvAgent/rvagent-tools/src/edit_diag.rs | 11 +++++- crates/rvAgent/rvagent-tools/src/local.rs | 10 ++++- .../rvagent-tools/tests/e2e_tool_calling.rs | 38 ++++++++++++------- 20 files changed, 115 insertions(+), 57 deletions(-) diff --git a/crates/ruvector-core/benches/pdx_vs_rowmajor.rs b/crates/ruvector-core/benches/pdx_vs_rowmajor.rs index 22cbc1fcd9..77f2ce6a30 100644 --- a/crates/ruvector-core/benches/pdx_vs_rowmajor.rs +++ b/crates/ruvector-core/benches/pdx_vs_rowmajor.rs @@ -20,7 +20,13 @@ fn corpus(n: usize, dim: usize) -> Vec> { fn bench(c: &mut Criterion) { // 1536 is the OpenAI embedding width; 768 covers most sentence encoders. // (n, dim) pairs: the first three are cache-resident, the rest stream. - for (n, dim) in [(256usize, 768usize), (512, 768), (1024, 768), (4096, 768), (4096, 1536)] { + for (n, dim) in [ + (256usize, 768usize), + (512, 768), + (1024, 768), + (4096, 768), + (4096, 1536), + ] { let vecs = corpus(n, dim); let refs: Vec<&[f32]> = vecs.iter().map(|v| v.as_slice()).collect(); let query: Vec = (0..dim).map(|d| (d % 7) as f32 / 7.0).collect(); diff --git a/crates/ruvector-core/src/pdx.rs b/crates/ruvector-core/src/pdx.rs index 2c34a5debe..9d7d5334e5 100644 --- a/crates/ruvector-core/src/pdx.rs +++ b/crates/ruvector-core/src/pdx.rs @@ -89,7 +89,12 @@ impl PdxBlock { vectors.len() ); for (i, v) in vectors.iter().enumerate() { - assert_eq!(v.len(), dim, "vector {i} has length {}, expected {dim}", v.len()); + assert_eq!( + v.len(), + dim, + "vector {i} has length {}, expected {dim}", + v.len() + ); } let mut data = vec![0.0f32; dim * LANES]; @@ -175,7 +180,10 @@ impl PdxIndex { /// # Panics /// If `vectors` is empty, or any vector's length differs from the first. pub fn from_rows(vectors: &[&[f32]]) -> Self { - assert!(!vectors.is_empty(), "cannot build a PdxIndex from no vectors"); + assert!( + !vectors.is_empty(), + "cannot build a PdxIndex from no vectors" + ); let dim = vectors[0].len(); let blocks = vectors diff --git a/crates/rvAgent/rvagent-acp/src/agent.rs b/crates/rvAgent/rvagent-acp/src/agent.rs index 2f31a60c50..268fafcd81 100644 --- a/crates/rvAgent/rvagent-acp/src/agent.rs +++ b/crates/rvAgent/rvagent-acp/src/agent.rs @@ -209,7 +209,9 @@ impl AcpAgent { // Resolve the configured middleware names (unknown names warn and // are skipped); an empty configuration gets the default pipeline. let pipeline = if self.config.middleware.is_empty() { - rvagent_middleware::build_default_pipeline(&rvagent_middleware::PipelineConfig::default()) + rvagent_middleware::build_default_pipeline( + &rvagent_middleware::PipelineConfig::default(), + ) } else { let names: Vec<&str> = self .config diff --git a/crates/rvAgent/rvagent-backends/src/anthropic.rs b/crates/rvAgent/rvagent-backends/src/anthropic.rs index 01e1fc6f98..f58c2466fe 100644 --- a/crates/rvAgent/rvagent-backends/src/anthropic.rs +++ b/crates/rvAgent/rvagent-backends/src/anthropic.rs @@ -185,7 +185,12 @@ impl AnthropicClient { } /// Build the API request body from rvAgent messages and tool definitions. - fn build_request(&self, messages: &[Message], tools: &[ToolDefinition], stream: bool) -> ApiRequest { + fn build_request( + &self, + messages: &[Message], + tools: &[ToolDefinition], + stream: bool, + ) -> ApiRequest { let mut system_text: Option = None; let mut api_messages: Vec = Vec::new(); diff --git a/crates/rvAgent/rvagent-cli/src/app.rs b/crates/rvAgent/rvagent-cli/src/app.rs index ddefe7f018..68de1301da 100644 --- a/crates/rvAgent/rvagent-cli/src/app.rs +++ b/crates/rvAgent/rvagent-cli/src/app.rs @@ -10,11 +10,11 @@ use anyhow::{Context, Result}; use async_trait::async_trait; use tracing::{info, warn}; +use rvagent_core::bootstrap::EnvironmentSnapshot; use rvagent_core::config::{BackendConfig, MiddlewareConfig, RvAgentConfig, SecurityPolicy}; use rvagent_core::graph::{AgentGraph, ToolExecutor}; use rvagent_core::messages::{Message, ToolCall as CoreToolCall}; use rvagent_core::models::{resolve_model, ChatModel, ToolDefinition}; -use rvagent_core::bootstrap::EnvironmentSnapshot; use rvagent_core::prompt::BASE_AGENT_PROMPT; use rvagent_core::state::AgentState; @@ -145,8 +145,7 @@ struct CliToolExecutor { impl CliToolExecutor { fn new(cwd: &Path) -> Self { // Confined to `cwd`: tool-supplied paths cannot escape the workspace. - let backend: rvagent_tools::BackendRef = - Arc::new(rvagent_tools::LocalFsBackend::new(cwd)); + let backend: rvagent_tools::BackendRef = Arc::new(rvagent_tools::LocalFsBackend::new(cwd)); Self { tools: rvagent_tools::builtin_tools(), backend, diff --git a/crates/rvAgent/rvagent-core/src/graph.rs b/crates/rvAgent/rvagent-core/src/graph.rs index 5f09fd531c..973264201e 100644 --- a/crates/rvAgent/rvagent-core/src/graph.rs +++ b/crates/rvAgent/rvagent-core/src/graph.rs @@ -348,9 +348,7 @@ impl AgentGraph { let dispatch: Vec = tool_calls .iter() .zip(&looping) - .filter(|(tc, refused)| { - refused.is_none() && !handled.contains_key(&tc.id) - }) + .filter(|(tc, refused)| refused.is_none() && !handled.contains_key(&tc.id)) .map(|(tc, _)| tc.clone()) .collect(); @@ -667,8 +665,7 @@ mod tests { #[async_trait] impl ToolExecutor for CountingExecutor { async fn execute(&self, call: &ToolCall, _state: &AgentState) -> Result { - self.calls - .fetch_add(1, std::sync::atomic::Ordering::SeqCst); + self.calls.fetch_add(1, std::sync::atomic::Ordering::SeqCst); Ok(format!("result of {}", call.name)) } } diff --git a/crates/rvAgent/rvagent-core/src/lib.rs b/crates/rvAgent/rvagent-core/src/lib.rs index 62dd52d89a..aba0196c6d 100644 --- a/crates/rvAgent/rvagent-core/src/lib.rs +++ b/crates/rvAgent/rvagent-core/src/lib.rs @@ -19,8 +19,8 @@ pub mod agi_container; pub mod arena; -pub mod budget; pub mod bootstrap; +pub mod budget; pub mod config; pub mod cow_state; pub mod error; @@ -36,8 +36,8 @@ pub mod prompt; pub mod rvf_bridge; pub mod session_crypto; pub mod state; -pub mod subagent; pub mod string_pool; +pub mod subagent; // Re-export key types at crate root for convenience. pub use agi_container::{ diff --git a/crates/rvAgent/rvagent-core/src/policy.rs b/crates/rvAgent/rvagent-core/src/policy.rs index 24a3cffcdd..d3152c5479 100644 --- a/crates/rvAgent/rvagent-core/src/policy.rs +++ b/crates/rvAgent/rvagent-core/src/policy.rs @@ -148,10 +148,13 @@ impl PolicyGenome { } fn parse(lever: &str, value: &str) -> Result { - value.trim().parse::().map_err(|_| PolicyError::BadValue { - lever: lever.to_string(), - value: value.to_string(), - }) + value + .trim() + .parse::() + .map_err(|_| PolicyError::BadValue { + lever: lever.to_string(), + value: value.to_string(), + }) } // --------------------------------------------------------------------------- @@ -380,7 +383,11 @@ mod tests { fn score_never_serializes_a_non_finite_number() { // JSON has no infinity: serde emits null, and the gate reads `null > n` // as false in JS, so a zero-win policy would pass the cost clause. - for runs in [vec![], vec![run(false, true, 5.0)], vec![run(true, true, 1.0)]] { + for runs in [ + vec![], + vec![run(false, true, 5.0)], + vec![run(true, true, 1.0)], + ] { let json = serde_json::to_value(Score::from_runs(&runs)).unwrap(); let cost = json.get("costPerWin").unwrap(); assert!( @@ -418,7 +425,6 @@ mod tests { assert!(json.get("regressed").is_some()); } - #[test] fn run_outcome_round_trips_across_the_process_boundary() { let outcome = RunOutcome { diff --git a/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs b/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs index 417add7919..5cd9aee86e 100644 --- a/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs +++ b/crates/rvAgent/rvagent-middleware/benches/middleware_bench.rs @@ -55,8 +55,7 @@ fn bench_full_pipeline(c: &mut Criterion) { Message::ai("Hi there"), Message::human("Write some code"), ]); - let response = - rt.block_on(pipeline.run_wrap_model_call(black_box(request), &handler)); + let response = rt.block_on(pipeline.run_wrap_model_call(black_box(request), &handler)); black_box(response); }); }); diff --git a/crates/rvAgent/rvagent-middleware/src/patch_tool_calls.rs b/crates/rvAgent/rvagent-middleware/src/patch_tool_calls.rs index ab6b8b4de5..89b4ce7fdf 100644 --- a/crates/rvAgent/rvagent-middleware/src/patch_tool_calls.rs +++ b/crates/rvAgent/rvagent-middleware/src/patch_tool_calls.rs @@ -74,9 +74,9 @@ impl Middleware for PatchToolCallsMiddleware { continue; } - let has_response = state.messages[i + 1..].iter().any(|m| { - matches!(m, Message::Tool(t) if t.tool_call_id == tc.id) - }); + let has_response = state.messages[i + 1..] + .iter() + .any(|m| matches!(m, Message::Tool(t) if t.tool_call_id == tc.id)); if !has_response { patched.push(Message::tool_with_name( diff --git a/crates/rvAgent/rvagent-middleware/src/rvf_manifest.rs b/crates/rvAgent/rvagent-middleware/src/rvf_manifest.rs index 4b60cb8244..850799bc9d 100644 --- a/crates/rvAgent/rvagent-middleware/src/rvf_manifest.rs +++ b/crates/rvAgent/rvagent-middleware/src/rvf_manifest.rs @@ -280,7 +280,9 @@ mod tests { let tools = mw.tools(); let lint = tools.iter().find(|t| t.name() == "rvf:lint").unwrap(); - let result = lint.invoke(serde_json::json!({"path": "src/main.rs"})).await; + let result = lint + .invoke(serde_json::json!({"path": "src/main.rs"})) + .await; assert!(result.is_ok()); assert!(result.unwrap().contains("rvf:lint")); } diff --git a/crates/rvAgent/rvagent-middleware/src/sona.rs b/crates/rvAgent/rvagent-middleware/src/sona.rs index e90b5b1e37..fefd351d77 100644 --- a/crates/rvAgent/rvagent-middleware/src/sona.rs +++ b/crates/rvAgent/rvagent-middleware/src/sona.rs @@ -321,8 +321,7 @@ impl SonaState { let mut builder = TrajectoryBuilder::new(id, query_embedding); // Add response as a step - let response_embedding = - generate_embedding(response.content(), self.config.embedding_dim); + let response_embedding = generate_embedding(response.content(), self.config.embedding_dim); let quality = estimate_quality(request, response); builder.add_step(response_embedding, vec![], quality); diff --git a/crates/rvAgent/rvagent-middleware/src/types.rs b/crates/rvAgent/rvagent-middleware/src/types.rs index 797d517900..2c48d35a29 100644 --- a/crates/rvAgent/rvagent-middleware/src/types.rs +++ b/crates/rvAgent/rvagent-middleware/src/types.rs @@ -178,9 +178,7 @@ mod tests { let mut state = AgentState::default(); let mut update = AgentStateUpdate::default(); update.messages = Some(vec![Message::human("hi")]); - update - .extensions - .insert("k".into(), serde_json::json!("v")); + update.extensions.insert("k".into(), serde_json::json!("v")); update.apply_to(&mut state); assert_eq!(state.message_count(), 1); assert_eq!(json_extension(&state, "k"), Some(&serde_json::json!("v"))); diff --git a/crates/rvAgent/rvagent-middleware/tests/security_tests.rs b/crates/rvAgent/rvagent-middleware/tests/security_tests.rs index de499e0e6f..aaaea662c2 100644 --- a/crates/rvAgent/rvagent-middleware/tests/security_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/security_tests.rs @@ -11,6 +11,7 @@ use std::collections::HashMap; use std::sync::{Arc, Mutex}; +use async_trait::async_trait; use rvagent_middleware::memory::{ compute_sha3_256, MemoryMiddleware, SecurityPolicy, TrustManifest, TrustVerification, MAX_MEMORY_FILE_SIZE, @@ -19,7 +20,6 @@ use rvagent_middleware::patch_tool_calls::PatchToolCallsMiddleware; use rvagent_middleware::skills::{parse_skill_metadata, validate_skill_name, MAX_SKILL_FILE_SIZE}; use rvagent_middleware::tool_sanitizer::ToolResultSanitizerMiddleware; use rvagent_middleware::witness::{WitnessBuilder, WitnessMiddleware}; -use async_trait::async_trait; use rvagent_middleware::{ AgentState, Message, Middleware, ModelHandler, ModelRequest, ModelResponse, RunnableConfig, Runtime, ToolCall, diff --git a/crates/rvAgent/rvagent-middleware/tests/summarization_tests.rs b/crates/rvAgent/rvagent-middleware/tests/summarization_tests.rs index fb7345a790..765442980a 100644 --- a/crates/rvAgent/rvagent-middleware/tests/summarization_tests.rs +++ b/crates/rvAgent/rvagent-middleware/tests/summarization_tests.rs @@ -72,18 +72,24 @@ async fn test_auto_compact_triggers() { // With a single short message below threshold, no compaction let mw_high = SummarizationMiddleware::new(100_000, 0.85, 0.10); let short_request = ModelRequest::new(vec![Message::human("hello")]); - let short_response = mw_high.wrap_model_call(short_request, &MessageCountHandler).await; + let short_response = mw_high + .wrap_model_call(short_request, &MessageCountHandler) + .await; assert_eq!( - short_response.content(), "count=1", + short_response.content(), + "count=1", "Short conversation must not be compacted" ); // Edge case: single message above threshold should not compact (need >1 messages) let mw_tiny = SummarizationMiddleware::new(1, 0.1, 0.5); let single_request = ModelRequest::new(vec![Message::human("a long message that exceeds")]); - let single_response = mw_tiny.wrap_model_call(single_request, &MessageCountHandler).await; + let single_response = mw_tiny + .wrap_model_call(single_request, &MessageCountHandler) + .await; assert_eq!( - single_response.content(), "count=1", + single_response.content(), + "count=1", "Single message should not be compacted even above threshold" ); } @@ -226,7 +232,9 @@ async fn test_file_permissions() { } let request = ModelRequest::new(messages); - let response = mw_compact.wrap_model_call(request, &FirstMessageHandler).await; + let response = mw_compact + .wrap_model_call(request, &FirstMessageHandler) + .await; // When compaction triggers, the first message should be the summary (System role) assert!( diff --git a/crates/rvAgent/rvagent-middleware/tests/unicode_security_integration.rs b/crates/rvAgent/rvagent-middleware/tests/unicode_security_integration.rs index 4051bc1705..06bc995c29 100644 --- a/crates/rvAgent/rvagent-middleware/tests/unicode_security_integration.rs +++ b/crates/rvAgent/rvagent-middleware/tests/unicode_security_integration.rs @@ -201,7 +201,7 @@ async fn test_comprehensive_attack_scenario() { let new_msgs = update.unwrap().messages.unwrap(); // User message: zero-width stripped assert_eq!(new_msgs[0].content(), "Visit pаypal.comnow!"); // Confusable remains - // Tool message: BiDi stripped + // Tool message: BiDi stripped assert_eq!(new_msgs[1].content(), "Downloaded: eviltxt.exe"); } diff --git a/crates/rvAgent/rvagent-subagents/tests/orchestrator_tests.rs b/crates/rvAgent/rvagent-subagents/tests/orchestrator_tests.rs index ec919abe18..0d40c5a420 100644 --- a/crates/rvAgent/rvagent-subagents/tests/orchestrator_tests.rs +++ b/crates/rvAgent/rvagent-subagents/tests/orchestrator_tests.rs @@ -157,7 +157,11 @@ fn test_state_isolation() { ); // Messages must be replaced with task description - assert_eq!(child.message_count(), 1, "Child must have exactly 1 message"); + assert_eq!( + child.message_count(), + 1, + "Child must have exactly 1 message" + ); assert!(matches!( child.messages[0], rvagent_core::messages::Message::Human(_) diff --git a/crates/rvAgent/rvagent-tools/src/edit_diag.rs b/crates/rvAgent/rvagent-tools/src/edit_diag.rs index 2bfa5160a6..9accdd0d2a 100644 --- a/crates/rvAgent/rvagent-tools/src/edit_diag.rs +++ b/crates/rvAgent/rvagent-tools/src/edit_diag.rs @@ -144,7 +144,10 @@ mod tests { let content = "fn compute(a: u32, b: u32) -> u32 {\n a + b\n}\n"; // Same opening but a different signature — the most common real case. let msg = diagnose_edit_failure(content, "fn compute(a: u32) -> u32 {", "a.rs"); - assert!(msg.contains("closest line in the file is line 1"), "got: {msg}"); + assert!( + msg.contains("closest line in the file is line 1"), + "got: {msg}" + ); assert!(msg.contains("fn compute(a: u32, b: u32)"), "got: {msg}"); } @@ -183,6 +186,10 @@ mod tests { let long = "x".repeat(500); let content = format!("prefix_{long}\n"); let msg = diagnose_edit_failure(&content, &format!("prefix_{}", "y".repeat(20)), "a.txt"); - assert!(msg.len() < 400, "diagnostic should stay compact: {}", msg.len()); + assert!( + msg.len() < 400, + "diagnostic should stay compact: {}", + msg.len() + ); } } diff --git a/crates/rvAgent/rvagent-tools/src/local.rs b/crates/rvAgent/rvagent-tools/src/local.rs index 5b07196c97..6b81183772 100644 --- a/crates/rvAgent/rvagent-tools/src/local.rs +++ b/crates/rvAgent/rvagent-tools/src/local.rs @@ -554,7 +554,10 @@ mod tests { let (_dir, backend) = backend(); let result = backend.write("nested/dir/new.txt", "content"); assert!(result.error.is_none(), "unexpected: {:?}", result.error); - assert_eq!(backend.read("nested/dir/new.txt", 0, 10).unwrap(), "content"); + assert_eq!( + backend.read("nested/dir/new.txt", 0, 10).unwrap(), + "content" + ); } #[test] @@ -592,7 +595,10 @@ mod tests { // what was asked for. let err = verify_written(&path, "expected something longer").unwrap_err(); assert!(err.contains("did not verify"), "got: {err}"); - assert!(err.contains("Re-read the file"), "must be actionable: {err}"); + assert!( + err.contains("Re-read the file"), + "must be actionable: {err}" + ); } #[test] diff --git a/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs b/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs index fdfd6b3a16..f7f6752f2d 100644 --- a/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs +++ b/crates/rvAgent/rvagent-tools/tests/e2e_tool_calling.rs @@ -143,7 +143,11 @@ async fn schemas_reach_the_model() { assert!(defs.iter().any(|d| d.name == "read_file")); assert!(defs.iter().any(|d| d.name == "write_file")); for def in &defs { - assert!(!def.description.is_empty(), "{} has no description", def.name); + assert!( + !def.description.is_empty(), + "{} has no description", + def.name + ); assert_eq!( def.input_schema.get("type").and_then(|v| v.as_str()), Some("object"), @@ -167,7 +171,11 @@ async fn schemas_are_sent_on_every_turn_including_after_tools() { let model = Arc::new(ScriptedModel::new(vec![ Message::ai_with_tools( "reading", - vec![call("t1", "read_file", serde_json::json!({"file_path": "f.txt"}))], + vec![call( + "t1", + "read_file", + serde_json::json!({"file_path": "f.txt"}), + )], ), Message::ai("read it"), ])); @@ -286,10 +294,7 @@ async fn tool_error_feeds_back_and_loop_continues() { async fn unknown_tool_is_reported_not_fatal() { let dir = tempfile::tempdir().unwrap(); let model = ScriptedModel::new(vec![ - Message::ai_with_tools( - "", - vec![call("t1", "no_such_tool", serde_json::json!({}))], - ), + Message::ai_with_tools("", vec![call("t1", "no_such_tool", serde_json::json!({}))]), Message::ai("ok"), ]); let graph = AgentGraph::new(model, RealToolExecutor::new(dir.path())); @@ -391,10 +396,8 @@ async fn usage_metadata_is_aggregated() { let dir = tempfile::tempdir().unwrap(); // Two turns carrying provider usage metadata, as the backends attach it. - let mut first = Message::ai_with_tools( - "", - vec![call("t1", "ls", serde_json::json!({"path": "."}))], - ); + let mut first = + Message::ai_with_tools("", vec![call("t1", "ls", serde_json::json!({"path": "."}))]); let mut second = Message::ai("done"); for (msg, input, output) in [(&mut first, 100u64, 20u64), (&mut second, 150u64, 30u64)] { if let Message::Ai(ai) = msg { @@ -419,7 +422,10 @@ async fn usage_metadata_is_aggregated() { }) .fold((0, 0), |(i, o), usage| { ( - i + usage.get("input_tokens").and_then(|v| v.as_u64()).unwrap_or(0), + i + usage + .get("input_tokens") + .and_then(|v| v.as_u64()) + .unwrap_or(0), o + usage .get("output_tokens") .and_then(|v| v.as_u64()) @@ -498,7 +504,9 @@ async fn old_observations_are_masked_before_reaching_the_model() { assert_eq!(seen_tools.len(), 5, "every call still has a paired result"); for (i, msg) in seen_tools.iter().enumerate() { - let Message::Tool(t) = msg else { unreachable!() }; + let Message::Tool(t) = msg else { + unreachable!() + }; if i < 3 { assert!( t.content.contains("output elided"), @@ -723,7 +731,11 @@ async fn path_escape_is_refused_through_the_loop() { Message::ai_with_tools( "exfiltrating", vec![ - call("t1", "read_file", serde_json::json!({"file_path": "/etc/passwd"})), + call( + "t1", + "read_file", + serde_json::json!({"file_path": "/etc/passwd"}), + ), call( "t2", "read_file", From c2062913646365c0e6546bcca60a69d1a71cd9cd Mon Sep 17 00:00:00 2001 From: ruv Date: Sun, 2 Aug 2026 10:16:32 -0400 Subject: [PATCH 25/25] =?UTF-8?q?fix(rvagent):=20close=20review=20blockers?= =?UTF-8?q?=20=E2=80=94=20symlink/hard-link=20write=20escapes,=20no-op=20H?= =?UTF-8?q?ITL=20gate,=20Gemini=20request=20bugs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Code-review fixes on top of the harness work: rvagent-tools (local.rs): - Reject dangling-symlink path components and writes through any symlink; refuse writes through hard links (nlink > 1, unix) — closes workspace write-escape; module doc now states exact confinement guarantees - Saturating offset+limit in read (u64::MAX no longer panics) - Char-boundary-safe output truncation - execute(): own process group, concurrent pipe draining (no >64KB deadlock), SIGKILL of the whole tree on timeout, child reaped - glob/grep no longer follow symlinked dirs (loop + read-escape fix) rvagent-middleware: - middleware_by_name takes PipelineConfig; HITL built from interrupt_on with a conservative default gate (shell/file mutation) shared by build_default_pipeline — both construction paths now match - HITL block message states reality (call dropped, how to configure) - Unknown middleware names are hard errors instead of silent skips - unicode_security/memory/skills/summarization by-name arms honor config - Retry backoff capped at 60s; summarization preview char-boundary-safe - masking truncate_tool_result respects caps smaller than the marker rvagent-backends: - Gemini: consecutive tool results grouped into one Content (parallel tool calls no longer 400); Candidate deserializes finishReason; missing candidates/content/parts produce descriptive errors instead of silent empty success; tool schemas sanitized to Gemini's OpenAPI subset (nullable arrays, format filtering, type-less fallbacks); parameterless tools omit parameters - Anthropic: consecutive tool results grouped into one user message rvagent-cli: - Pipeline built via validated by-name path; RVAGENT_AUTO_APPROVE=1 opt-out (stderr-visible) for unattended runs, fail-closed by default Co-Authored-By: claude-flow --- Cargo.lock | 1 + crates/rvAgent/rvagent-acp/src/agent.rs | 12 +- .../rvAgent/rvagent-backends/src/anthropic.rs | 119 ++- crates/rvAgent/rvagent-backends/src/gemini.rs | 724 +++++++++++++++++- crates/rvAgent/rvagent-cli/src/app.rs | 34 +- crates/rvAgent/rvagent-core/src/graph.rs | 99 ++- crates/rvAgent/rvagent-core/src/masking.rs | 37 +- crates/rvAgent/rvagent-middleware/src/hitl.rs | 49 +- crates/rvAgent/rvagent-middleware/src/lib.rs | 322 ++++++-- .../rvAgent/rvagent-middleware/src/retry.rs | 41 +- .../rvagent-middleware/src/summarization.rs | 45 +- crates/rvAgent/rvagent-tools/Cargo.toml | 5 + crates/rvAgent/rvagent-tools/src/lib.rs | 2 +- crates/rvAgent/rvagent-tools/src/local.rs | 534 +++++++++++-- 14 files changed, 1815 insertions(+), 209 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index da0277f162..03d9322bc8 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -11401,6 +11401,7 @@ dependencies = [ "async-trait", "criterion 0.5.1", "glob", + "libc", "mockall", "rvagent-backends", "rvagent-core", diff --git a/crates/rvAgent/rvagent-acp/src/agent.rs b/crates/rvAgent/rvagent-acp/src/agent.rs index 268fafcd81..a96ffa55cb 100644 --- a/crates/rvAgent/rvagent-acp/src/agent.rs +++ b/crates/rvAgent/rvagent-acp/src/agent.rs @@ -206,12 +206,11 @@ impl AcpAgent { ..GraphConfig::default() }; - // Resolve the configured middleware names (unknown names warn and - // are skipped); an empty configuration gets the default pipeline. + // Resolve the configured middleware names (an unknown name is an + // error); an empty configuration gets the default pipeline. + let pipeline_config = rvagent_middleware::PipelineConfig::default(); let pipeline = if self.config.middleware.is_empty() { - rvagent_middleware::build_default_pipeline( - &rvagent_middleware::PipelineConfig::default(), - ) + rvagent_middleware::build_default_pipeline(&pipeline_config) } else { let names: Vec<&str> = self .config @@ -219,7 +218,8 @@ impl AcpAgent { .iter() .map(|m| m.name.as_str()) .collect(); - rvagent_middleware::build_pipeline_from_names(&names) + rvagent_middleware::build_pipeline_from_names(&names, &pipeline_config) + .map_err(|e| e.to_string())? }; let pipeline = Arc::new(pipeline); diff --git a/crates/rvAgent/rvagent-backends/src/anthropic.rs b/crates/rvAgent/rvagent-backends/src/anthropic.rs index f58c2466fe..0740da6f02 100644 --- a/crates/rvAgent/rvagent-backends/src/anthropic.rs +++ b/crates/rvAgent/rvagent-backends/src/anthropic.rs @@ -193,8 +193,19 @@ impl AnthropicClient { ) -> ApiRequest { let mut system_text: Option = None; let mut api_messages: Vec = Vec::new(); + // Anthropic expects every tool_result answering one assistant turn in + // a single user message; splitting them suppresses parallel tool + // calling. Buffer consecutive tool results and flush them together. + let mut pending_tool_results: Vec = Vec::new(); for msg in messages { + if !matches!(msg, Message::Tool(_)) && !pending_tool_results.is_empty() { + api_messages.push(ApiMessage { + role: "user".to_string(), + content: ApiContent::Blocks(std::mem::take(&mut pending_tool_results)), + }); + } + match msg { Message::System(s) => { // Anthropic uses a top-level `system` field; merge multiple system messages. @@ -240,17 +251,21 @@ impl AnthropicClient { } } Message::Tool(t) => { - api_messages.push(ApiMessage { - role: "user".to_string(), - content: ApiContent::Blocks(vec![ContentBlock::ToolResult { - tool_use_id: t.tool_call_id.clone(), - content: t.content.clone(), - }]), + pending_tool_results.push(ContentBlock::ToolResult { + tool_use_id: t.tool_call_id.clone(), + content: t.content.clone(), }); } } } + if !pending_tool_results.is_empty() { + api_messages.push(ApiMessage { + role: "user".to_string(), + content: ApiContent::Blocks(pending_tool_results), + }); + } + ApiRequest { model: self.config.model_id.clone(), max_tokens: self.config.max_tokens, @@ -549,6 +564,98 @@ mod tests { } } + #[test] + fn test_build_request_groups_parallel_tool_results() { + let client = + AnthropicClient::with_http(test_config(), reqwest::Client::new(), "key".to_string()); + let messages = vec![ + Message::human("Read both files."), + Message::ai_with_tools( + "Reading them.", + vec![ + ToolCall { + id: "tc_1".to_string(), + name: "read_file".to_string(), + args: json!({"path": "/tmp/a.txt"}), + }, + ToolCall { + id: "tc_2".to_string(), + name: "read_file".to_string(), + args: json!({"path": "/tmp/b.txt"}), + }, + ], + ), + Message::tool("tc_1", "a contents"), + Message::tool("tc_2", "b contents"), + ]; + let req = client.build_request(&messages, &[], false); + + // user + assistant + ONE user message holding both tool_results. + assert_eq!(req.messages.len(), 3); + assert_eq!(req.messages[2].role, "user"); + match &req.messages[2].content { + ApiContent::Blocks(blocks) => { + assert_eq!(blocks.len(), 2); + match (&blocks[0], &blocks[1]) { + ( + ContentBlock::ToolResult { + tool_use_id: first, .. + }, + ContentBlock::ToolResult { + tool_use_id: second, + .. + }, + ) => { + assert_eq!(first, "tc_1"); + assert_eq!(second, "tc_2"); + } + _ => panic!("expected two tool_result blocks"), + } + } + _ => panic!("expected Blocks content for grouped tool results"), + } + } + + #[test] + fn test_build_request_tool_results_flush_before_next_turn() { + let client = + AnthropicClient::with_http(test_config(), reqwest::Client::new(), "key".to_string()); + // Results answering distinct assistant turns must stay separate. + let messages = vec![ + Message::ai_with_tools( + "", + vec![ToolCall { + id: "tc_1".to_string(), + name: "first".to_string(), + args: json!({}), + }], + ), + Message::tool("tc_1", "one"), + Message::ai_with_tools( + "", + vec![ToolCall { + id: "tc_2".to_string(), + name: "second".to_string(), + args: json!({}), + }], + ), + Message::tool("tc_2", "two"), + ]; + let req = client.build_request(&messages, &[], false); + + assert_eq!(req.messages.len(), 4); + assert_eq!(req.messages[0].role, "assistant"); + assert_eq!(req.messages[1].role, "user"); + assert_eq!(req.messages[2].role, "assistant"); + assert_eq!(req.messages[3].role, "user"); + for idx in [1usize, 3] { + match &req.messages[idx].content { + ApiContent::Blocks(blocks) => assert_eq!(blocks.len(), 1), + _ => panic!("expected Blocks content at index {idx}"), + } + } + } + #[test] fn test_build_request_with_tool_definitions() { let client = diff --git a/crates/rvAgent/rvagent-backends/src/gemini.rs b/crates/rvAgent/rvagent-backends/src/gemini.rs index 28959ad922..d82f4b2082 100644 --- a/crates/rvAgent/rvagent-backends/src/gemini.rs +++ b/crates/rvAgent/rvagent-backends/src/gemini.rs @@ -71,7 +71,10 @@ impl Part { struct FunctionDeclaration { name: String, description: String, - parameters: serde_json::Value, + /// Omitted entirely for parameterless tools: Gemini rejects an OBJECT + /// schema with an empty property map, and `parameters` is optional. + #[serde(default, skip_serializing_if = "Option::is_none")] + parameters: Option, } /// A tool group in the Gemini API format. @@ -82,9 +85,14 @@ struct GeminiTool { } /// A single message in the Gemini API format. -#[derive(Debug, Clone, Serialize, Deserialize)] +/// +/// Both fields default on deserialization: Gemini omits `parts` entirely on a +/// truncated or filtered candidate, and omits `role` on some content blocks. +#[derive(Debug, Clone, Serialize, Deserialize, Default)] struct GeminiContent { + #[serde(default)] role: String, + #[serde(default)] parts: Vec, } @@ -111,20 +119,34 @@ struct GeminiRequest { /// A candidate response from Gemini. #[derive(Debug, Deserialize)] +#[serde(rename_all = "camelCase")] struct Candidate { - content: GeminiContent, - #[allow(dead_code)] + /// Absent when Gemini blocks the candidate outright (e.g. a SAFETY stop + /// yields `{"finishReason": "SAFETY"}` with no content at all). + #[serde(default)] + content: Option, #[serde(default)] finish_reason: Option, } +/// Feedback about the prompt itself, present when Gemini refuses to answer. +#[derive(Debug, Deserialize)] +#[serde(rename_all = "camelCase")] +struct PromptFeedback { + #[serde(default)] + block_reason: Option, +} + /// The response body from the Gemini API. #[derive(Debug, Deserialize)] #[serde(rename_all = "camelCase")] struct GeminiResponse { + #[serde(default)] candidates: Vec, #[serde(default)] usage_metadata: Option, + #[serde(default)] + prompt_feedback: Option, } /// Token usage information. @@ -198,8 +220,19 @@ impl GeminiClient { // back to function names so tool results can be sent as // functionResponse parts. let mut call_id_to_name: HashMap = HashMap::new(); + // Gemini requires every functionResponse answering one model turn to + // live in a single user Content. Buffer consecutive tool results and + // flush them together. + let mut pending_tool_parts: Vec = Vec::new(); for msg in messages { + if !matches!(msg, Message::Tool(_)) && !pending_tool_parts.is_empty() { + contents.push(GeminiContent { + role: "user".to_string(), + parts: std::mem::take(&mut pending_tool_parts), + }); + } + match msg { Message::System(s) => { system_instruction = Some(GeminiContent { @@ -246,20 +279,24 @@ impl GeminiClient { .clone() .or_else(|| call_id_to_name.get(&t.tool_call_id).cloned()) .unwrap_or_else(|| t.tool_call_id.clone()); - contents.push(GeminiContent { - role: "user".to_string(), - parts: vec![Part { - function_response: Some(FunctionResponse { - name, - response: serde_json::json!({ "result": t.content }), - }), - ..Part::default() - }], + pending_tool_parts.push(Part { + function_response: Some(FunctionResponse { + name, + response: serde_json::json!({ "result": t.content }), + }), + ..Part::default() }); } } } + if !pending_tool_parts.is_empty() { + contents.push(GeminiContent { + role: "user".to_string(), + parts: pending_tool_parts, + }); + } + let gemini_tools = if tools.is_empty() { Vec::new() } else { @@ -269,7 +306,7 @@ impl GeminiClient { .map(|t| FunctionDeclaration { name: t.name.clone(), description: t.description.clone(), - parameters: t.input_schema.clone(), + parameters: tool_parameters(&t.input_schema), }) .collect(), }] @@ -373,14 +410,9 @@ impl GeminiClient { Err(last_err .unwrap_or_else(|| RvAgentError::model("Gemini API request failed after all retries"))) } -} - -#[async_trait] -impl ChatModel for GeminiClient { - async fn complete(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result { - let request = self.build_request(messages, tools); - let response = self.send_with_retry(&request).await?; + /// Convert an API response into an rvAgent [`Message`]. + fn parse_response(response: GeminiResponse) -> Result { // Collect text and function-call parts from the first candidate. // Gemini does not assign tool-call IDs, so synthesize unique ones: // a process-wide counter prevents id collisions across turns (the @@ -389,19 +421,52 @@ impl ChatModel for GeminiClient { static CALL_COUNTER: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(0); let mut text_parts: Vec = Vec::new(); let mut tool_calls: Vec = Vec::new(); - if let Some(candidate) = response.candidates.first() { - for part in &candidate.content.parts { - if let Some(text) = &part.text { - text_parts.push(text.clone()); - } - if let Some(fc) = &part.function_call { - let n = CALL_COUNTER.fetch_add(1, std::sync::atomic::Ordering::Relaxed); - tool_calls.push(ToolCall { - id: format!("gemini_call_{n}_{}", fc.name), - name: fc.name.clone(), - args: fc.args.clone(), - }); - } + // An empty candidate list means Gemini produced nothing — typically a + // safety block, which carries a promptFeedback.blockReason. Surfacing + // it as an error keeps the agent loop from treating a refusal as an + // ordinary empty reply. + let candidate = response.candidates.first().ok_or_else(|| { + match response + .prompt_feedback + .as_ref() + .and_then(|f| f.block_reason.as_deref()) + { + Some(reason) => RvAgentError::model(format!( + "Gemini returned no candidates: prompt blocked (blockReason: {reason})" + )), + None => RvAgentError::model( + "Gemini returned no candidates (response may have been blocked)", + ), + } + })?; + + // A candidate can arrive with no content (SAFETY block) or with + // content but no parts (MAX_TOKENS truncation on the 2.5 series). + // Both are failures, not empty replies — report the finishReason so + // the caller can tell truncation from filtering. + let finish_reason = candidate.finish_reason.as_deref().unwrap_or("unspecified"); + let content = candidate.content.as_ref().ok_or_else(|| { + RvAgentError::model(format!( + "Gemini candidate had no content (finishReason: {finish_reason})" + )) + })?; + if content.parts.is_empty() { + return Err(RvAgentError::model(format!( + "Gemini candidate had no content parts (finishReason: {finish_reason})" + ))); + } + + for part in &content.parts { + if let Some(text) = &part.text { + text_parts.push(text.clone()); + } + if let Some(fc) = &part.function_call { + let n = CALL_COUNTER.fetch_add(1, std::sync::atomic::Ordering::Relaxed); + tool_calls.push(ToolCall { + id: format!("gemini_call_{n}_{}", fc.name), + name: fc.name.clone(), + args: fc.args.clone(), + }); } } @@ -422,6 +487,15 @@ impl ChatModel for GeminiClient { metadata, })) } +} + +#[async_trait] +impl ChatModel for GeminiClient { + async fn complete(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result { + let request = self.build_request(messages, tools); + let response = self.send_with_retry(&request).await?; + Self::parse_response(response) + } async fn stream(&self, messages: &[Message], tools: &[ToolDefinition]) -> Result> { // For now, use non-streaming completion @@ -434,6 +508,160 @@ impl ChatModel for GeminiClient { // Helpers // --------------------------------------------------------------------------- +/// JSON Schema keywords Gemini's restricted OpenAPI subset accepts. +const GEMINI_SCHEMA_KEYS: &[&str] = &[ + "type", + "format", + "description", + "nullable", + "enum", + "items", + "properties", + "required", + "minimum", + "maximum", + "minItems", + "maxItems", + "minLength", + "maxLength", + "pattern", + "anyOf", +]; + +/// Sanitize a tool's input schema into the `parameters` field of a +/// [`FunctionDeclaration`]. +/// +/// A parameterless tool sanitizes down to a bare `{"type": "object"}` (the +/// in-tree rvagent-mcp registry `ping` tool declares `{"type": "object", +/// "properties": {}}`). `parameters` is optional in the Gemini API, so the +/// field is omitted outright rather than sent as a property-less object. +fn tool_parameters(input_schema: &serde_json::Value) -> Option { + let sanitized = sanitize_schema(input_schema); + let is_bare_object = sanitized.get("type").and_then(serde_json::Value::as_str) + == Some("object") + && sanitized.get("properties").is_none(); + if is_bare_object { + None + } else { + Some(sanitized) + } +} + +/// `format` values Gemini accepts, keyed by the schema's resolved type. +/// Anything else (uri, email, uuid, date, binary, …) is a 400. +const GEMINI_STRING_FORMATS: &[&str] = &["enum", "date-time"]; +const GEMINI_INTEGER_FORMATS: &[&str] = &["int32", "int64"]; +const GEMINI_NUMBER_FORMATS: &[&str] = &["float", "double"]; + +/// Rewrite a JSON Schema into Gemini's restricted OpenAPI subset. +/// +/// Beyond stripping keywords Gemini rejects outright (`default`, `$ref`, +/// `oneOf`, `additionalProperties`, `$schema`, `examples`, …), this repairs +/// four shapes that are legal JSON Schema but 400 on Gemini: +/// +/// - an OBJECT with an empty (or emptied) property map — the `properties` key +/// is dropped, leaving a bare `{"type": "object"}`; +/// - the OpenAPI 3.1 / pydantic array type form `{"type": ["string", "null"]}` +/// — collapsed to the first non-null entry plus `"nullable": true`; +/// - a `format` the resolved type does not support — dropped; +/// - a subschema left type-less by stripping (it carried only `$ref`/`oneOf`) +/// or an `enum` with no type — defaulted to `"string"`, since a best-effort +/// type beats a guaranteed 400. +/// +/// Applied recursively so nested `properties`, `items`, and `anyOf` +/// subschemas are repaired too. +fn sanitize_schema(schema: &serde_json::Value) -> serde_json::Value { + use serde_json::Value; + + let Value::Object(map) = schema else { + return schema.clone(); + }; + + let mut out = serde_json::Map::new(); + for (key, value) in map { + if !GEMINI_SCHEMA_KEYS.contains(&key.as_str()) { + continue; + } + let cleaned = match key.as_str() { + "properties" => match value { + Value::Object(props) => Value::Object( + props + .iter() + .map(|(name, sub)| (name.clone(), sanitize_schema(sub))) + .collect(), + ), + other => other.clone(), + }, + // `items` is a subschema, or an array of subschemas in tuple form. + "items" => match value { + Value::Array(subs) => Value::Array(subs.iter().map(sanitize_schema).collect()), + sub => sanitize_schema(sub), + }, + "anyOf" => match value { + Value::Array(subs) => Value::Array(subs.iter().map(sanitize_schema).collect()), + other => other.clone(), + }, + _ => value.clone(), + }; + out.insert(key.clone(), cleaned); + } + + // Collapse the array type form to a single type + nullable flag. + if let Some(Value::Array(variants)) = out.get("type").cloned() { + let has_null = variants.iter().any(|v| v.as_str() == Some("null")); + let primary = variants + .iter() + .find(|v| matches!(v.as_str(), Some(s) if s != "null")) + .cloned() + .unwrap_or_else(|| Value::String("string".to_string())); + out.insert("type".to_string(), primary); + if has_null { + out.insert("nullable".to_string(), Value::Bool(true)); + } + } + + // An enum is a string enum unless the schema says otherwise. + if out.contains_key("enum") && !out.contains_key("type") { + out.insert("type".to_string(), Value::String("string".to_string())); + } + + // Gemini rejects OBJECT with an empty property map; a bare + // `{"type": "object"}` is accepted. + let properties_empty = out + .get("properties") + .is_some_and(|p| p.as_object().is_none_or(|m| m.is_empty())); + if properties_empty { + out.remove("properties"); + } + + // `format` is only valid for the types listed above. + if let Some(format) = out + .get("format") + .and_then(Value::as_str) + .map(str::to_string) + { + let supported = match out.get("type").and_then(Value::as_str) { + Some("string") => GEMINI_STRING_FORMATS.contains(&format.as_str()), + Some("integer") => GEMINI_INTEGER_FORMATS.contains(&format.as_str()), + Some("number") => GEMINI_NUMBER_FORMATS.contains(&format.as_str()), + _ => false, + }; + if !supported { + out.remove("format"); + } + } + + // Nothing left to describe the shape (e.g. the schema was only a `$ref`). + if !["type", "anyOf", "enum", "properties", "items"] + .iter() + .any(|k| out.contains_key(*k)) + { + out.insert("type".to_string(), Value::String("string".to_string())); + } + + Value::Object(out) +} + fn resolve_api_key(source: &ApiKeySource) -> Result { match source { ApiKeySource::Env(var) => std::env::var(var).map_err(|_| { @@ -491,7 +719,9 @@ mod tests { let response: GeminiResponse = serde_json::from_str(json).unwrap(); assert_eq!(response.candidates.len(), 1); assert_eq!( - response.candidates[0].content.parts[0].text.as_deref(), + response.candidates[0].content.as_ref().unwrap().parts[0] + .text + .as_deref(), Some("Hello there!") ); } @@ -546,7 +776,7 @@ mod tests { }"#; let response: GeminiResponse = serde_json::from_str(json).unwrap(); - let parts = &response.candidates[0].content.parts; + let parts = &response.candidates[0].content.as_ref().unwrap().parts; assert_eq!(parts.len(), 2); assert!(parts[0].text.is_some()); let fc = parts[1].function_call.as_ref().unwrap(); @@ -624,4 +854,426 @@ mod tests { let fr = req.contents[1].parts[0].function_response.as_ref().unwrap(); assert_eq!(fr.name, "read_file"); } + + #[test] + fn test_gemini_parallel_tool_results_share_one_content() { + let config = ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }; + let client = GeminiClient { + config, + http: reqwest::Client::new(), + api_key: "test".to_string(), + }; + + let messages = vec![ + Message::human("read both"), + Message::ai_with_tools( + "", + vec![ + ToolCall { + id: "gemini_call_0_read_file".to_string(), + name: "read_file".to_string(), + args: serde_json::json!({"path": "/tmp/a"}), + }, + ToolCall { + id: "gemini_call_1_list_dir".to_string(), + name: "list_dir".to_string(), + args: serde_json::json!({"path": "/tmp"}), + }, + ], + ), + Message::tool("gemini_call_0_read_file", "a contents"), + Message::tool("gemini_call_1_list_dir", "a\nb"), + ]; + let req = client.build_request(&messages, &[]); + + // human + model + ONE grouped tool-result turn. + assert_eq!(req.contents.len(), 3); + let tool_turn = &req.contents[2]; + assert_eq!(tool_turn.role, "user"); + assert_eq!(tool_turn.parts.len(), 2); + assert_eq!( + tool_turn.parts[0] + .function_response + .as_ref() + .unwrap() + .name + .as_str(), + "read_file" + ); + assert_eq!( + tool_turn.parts[1] + .function_response + .as_ref() + .unwrap() + .name + .as_str(), + "list_dir" + ); + } + + #[test] + fn test_gemini_tool_results_flush_before_following_turn() { + let config = ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }; + let client = GeminiClient { + config, + http: reqwest::Client::new(), + api_key: "test".to_string(), + }; + + // Two separate model turns, each answered by its own tool result: + // grouping must not merge across the intervening model turn. + let messages = vec![ + Message::ai_with_tools( + "", + vec![ToolCall { + id: "c0".to_string(), + name: "first".to_string(), + args: serde_json::json!({}), + }], + ), + Message::tool("c0", "one"), + Message::ai_with_tools( + "", + vec![ToolCall { + id: "c1".to_string(), + name: "second".to_string(), + args: serde_json::json!({}), + }], + ), + Message::tool("c1", "two"), + ]; + let req = client.build_request(&messages, &[]); + + assert_eq!(req.contents.len(), 4); + assert_eq!(req.contents[0].role, "model"); + assert_eq!(req.contents[1].parts.len(), 1); + assert_eq!(req.contents[2].role, "model"); + assert_eq!(req.contents[3].parts.len(), 1); + } + + #[test] + fn test_gemini_candidate_finish_reason_deserializes() { + let json = r#"{ + "candidates": [{ + "content": { + "role": "model", + "parts": [{"text": "truncated..."}] + }, + "finishReason": "MAX_TOKENS", + "index": 0 + }], + "usageMetadata": {"promptTokenCount": 8, "candidatesTokenCount": 1024} + }"#; + + let response: GeminiResponse = serde_json::from_str(json).unwrap(); + assert_eq!( + response.candidates[0].finish_reason.as_deref(), + Some("MAX_TOKENS") + ); + } + + #[test] + fn test_gemini_empty_candidates_is_error() { + let json = r#"{ + "candidates": [], + "promptFeedback": {"blockReason": "SAFETY"} + }"#; + let response: GeminiResponse = serde_json::from_str(json).unwrap(); + let err = GeminiClient::parse_response(response).unwrap_err(); + let msg = err.to_string(); + assert!(msg.contains("no candidates"), "unexpected error: {msg}"); + assert!(msg.contains("SAFETY"), "block reason missing: {msg}"); + } + + #[test] + fn test_gemini_missing_candidates_is_error() { + // No candidates key at all and no promptFeedback. + let response: GeminiResponse = serde_json::from_str("{}").unwrap(); + let err = GeminiClient::parse_response(response).unwrap_err(); + assert!(err.to_string().contains("no candidates")); + } + + #[test] + fn test_sanitize_schema_strips_unsupported_keys() { + let schema = serde_json::json!({ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "additionalProperties": false, + "properties": { + "path": { + "type": "string", + "description": "file path", + "default": "/tmp", + "examples": ["/tmp/a"] + }, + "mode": { + "$ref": "#/definitions/Mode", + "oneOf": [{"type": "string"}], + "enum": ["read", "write"] + }, + "tags": { + "type": "array", + "items": {"type": "string", "default": "x"} + } + }, + "required": ["path"] + }); + + let cleaned = sanitize_schema(&schema); + + assert!(cleaned.get("$schema").is_none()); + assert!(cleaned.get("additionalProperties").is_none()); + assert_eq!(cleaned["type"], "object"); + assert_eq!(cleaned["required"], serde_json::json!(["path"])); + + let path = &cleaned["properties"]["path"]; + assert!(path.get("default").is_none()); + assert!(path.get("examples").is_none()); + assert_eq!(path["description"], "file path"); + + let mode = &cleaned["properties"]["mode"]; + assert!(mode.get("$ref").is_none()); + assert!(mode.get("oneOf").is_none()); + assert_eq!(mode["enum"], serde_json::json!(["read", "write"])); + // An enum stripped down to no type must not stay type-less. + assert_eq!(mode["type"], "string"); + + // Nested subschemas under `items` are sanitized too. + assert!(cleaned["properties"]["tags"]["items"] + .get("default") + .is_none()); + + // And nothing unsupported survives serialization into the request. + let config = ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }; + let client = GeminiClient { + config, + http: reqwest::Client::new(), + api_key: "test".to_string(), + }; + let tools = vec![ToolDefinition { + name: "read_file".to_string(), + description: "Read a file".to_string(), + input_schema: schema, + }]; + let req = client.build_request(&[Message::human("go")], &tools); + let json = serde_json::to_string(&req).unwrap(); + assert!(!json.contains("\"default\"")); + assert!(!json.contains("$ref")); + assert!(!json.contains("additionalProperties")); + } + + fn test_client() -> GeminiClient { + GeminiClient { + config: ModelConfig { + provider: rvagent_core::models::Provider::Google, + model_id: "gemini-2.5-pro".to_string(), + api_key_source: ApiKeySource::None, + max_tokens: 1024, + temperature: 0.0, + }, + http: reqwest::Client::new(), + api_key: "test".to_string(), + } + } + + #[test] + fn test_sanitize_schema_drops_empty_properties() { + // The rvagent-mcp registry `ping` tool declares exactly this. + let cleaned = sanitize_schema(&serde_json::json!({ + "type": "object", + "properties": {} + })); + assert_eq!(cleaned, serde_json::json!({"type": "object"})); + + // Same when the map is only emptied by sanitizing (map schemas whose + // sole key was additionalProperties). + let cleaned = sanitize_schema(&serde_json::json!({ + "type": "object", + "additionalProperties": {"type": "string"} + })); + assert_eq!(cleaned, serde_json::json!({"type": "object"})); + } + + #[test] + fn test_parameterless_tool_omits_parameters() { + let tools = vec![ToolDefinition { + name: "ping".to_string(), + description: "Ping the server".to_string(), + input_schema: serde_json::json!({"type": "object", "properties": {}}), + }]; + let req = test_client().build_request(&[Message::human("ping")], &tools); + assert!(req.tools[0].function_declarations[0].parameters.is_none()); + + // `parameters` must not appear on the wire at all. + let json = serde_json::to_value(&req).unwrap(); + assert!(json["tools"][0]["functionDeclarations"][0] + .get("parameters") + .is_none()); + // A tool that does take parameters still carries them. + let tools = vec![ToolDefinition { + name: "read_file".to_string(), + description: "Read a file".to_string(), + input_schema: serde_json::json!({ + "type": "object", + "properties": {"path": {"type": "string"}} + }), + }]; + let req = test_client().build_request(&[Message::human("go")], &tools); + assert!(req.tools[0].function_declarations[0].parameters.is_some()); + } + + #[test] + fn test_sanitize_schema_normalizes_array_type_form() { + // OpenAPI 3.1 / pydantic nullable form. + let cleaned = sanitize_schema(&serde_json::json!({ + "type": ["string", "null"], + "description": "maybe a name" + })); + assert_eq!(cleaned["type"], "string"); + assert_eq!(cleaned["nullable"], true); + assert_eq!(cleaned["description"], "maybe a name"); + + // No null present: collapse to the first entry, no nullable flag. + let cleaned = sanitize_schema(&serde_json::json!({"type": ["integer", "string"]})); + assert_eq!(cleaned["type"], "integer"); + assert!(cleaned.get("nullable").is_none()); + + // Degenerate null-only form still yields a usable type. + let cleaned = sanitize_schema(&serde_json::json!({"type": ["null"]})); + assert_eq!(cleaned["type"], "string"); + assert_eq!(cleaned["nullable"], true); + + // Nested under properties too. + let cleaned = sanitize_schema(&serde_json::json!({ + "type": "object", + "properties": {"nick": {"type": ["string", "null"]}} + })); + assert_eq!(cleaned["properties"]["nick"]["type"], "string"); + assert_eq!(cleaned["properties"]["nick"]["nullable"], true); + } + + #[test] + fn test_sanitize_schema_filters_unsupported_formats() { + // Unsupported string formats are dropped, the type survives. + for format in ["uri", "email", "uuid", "date", "binary", "hostname"] { + let cleaned = sanitize_schema(&serde_json::json!({"type": "string", "format": format})); + assert_eq!( + cleaned, + serde_json::json!({"type": "string"}), + "format {format} should have been dropped" + ); + } + + // Supported ones are kept. + for format in ["enum", "date-time"] { + let cleaned = sanitize_schema(&serde_json::json!({"type": "string", "format": format})); + assert_eq!(cleaned["format"], format); + } + for (ty, format) in [ + ("integer", "int32"), + ("integer", "int64"), + ("number", "float"), + ("number", "double"), + ] { + let cleaned = sanitize_schema(&serde_json::json!({"type": ty, "format": format})); + assert_eq!(cleaned["format"], format); + } + + // Right format, wrong type => dropped. + let cleaned = sanitize_schema(&serde_json::json!({"type": "string", "format": "int32"})); + assert!(cleaned.get("format").is_none()); + let cleaned = sanitize_schema(&serde_json::json!({"type": "integer", "format": "double"})); + assert!(cleaned.get("format").is_none()); + } + + #[test] + fn test_sanitize_schema_defaults_typeless_subschemas() { + // A schema carrying only unsupported keys must not end up type-less. + let cleaned = sanitize_schema(&serde_json::json!({"$ref": "#/definitions/Mode"})); + assert_eq!(cleaned, serde_json::json!({"type": "string"})); + + let cleaned = sanitize_schema(&serde_json::json!({ + "allOf": [{"type": "string"}], + "oneOf": [{"type": "integer"}] + })); + assert_eq!(cleaned, serde_json::json!({"type": "string"})); + + // Enum without a type gains one. + let cleaned = sanitize_schema(&serde_json::json!({"enum": ["a", "b"]})); + assert_eq!(cleaned["type"], "string"); + assert_eq!(cleaned["enum"], serde_json::json!(["a", "b"])); + + // Schemas that already describe a shape are left alone. + let cleaned = sanitize_schema(&serde_json::json!({"anyOf": [{"type": "string"}]})); + assert!(cleaned.get("type").is_none()); + let cleaned = sanitize_schema(&serde_json::json!({ + "items": {"type": "string"}, "type": "array" + })); + assert_eq!(cleaned["type"], "array"); + } + + #[test] + fn test_gemini_candidate_without_content_deserializes_and_errors() { + // SAFETY stop: candidate present, content absent entirely. + let json = r#"{"candidates": [{"finishReason": "SAFETY", "index": 0}]}"#; + let response: GeminiResponse = serde_json::from_str(json).unwrap(); + assert_eq!(response.candidates.len(), 1); + assert!(response.candidates[0].content.is_none()); + + let err = GeminiClient::parse_response(response) + .unwrap_err() + .to_string(); + assert!(err.contains("no content"), "unexpected error: {err}"); + assert!(err.contains("SAFETY"), "finishReason missing: {err}"); + } + + #[test] + fn test_gemini_candidate_with_empty_parts_deserializes_and_errors() { + // MAX_TOKENS truncation on the 2.5 series: content with no parts. + let json = r#"{ + "candidates": [{"content": {"role": "model"}, "finishReason": "MAX_TOKENS"}], + "usageMetadata": {"promptTokenCount": 9, "candidatesTokenCount": 1024} + }"#; + let response: GeminiResponse = serde_json::from_str(json).unwrap(); + assert!(response.candidates[0] + .content + .as_ref() + .unwrap() + .parts + .is_empty()); + + let err = GeminiClient::parse_response(response) + .unwrap_err() + .to_string(); + assert!(err.contains("no content parts"), "unexpected error: {err}"); + assert!(err.contains("MAX_TOKENS"), "finishReason missing: {err}"); + } + + #[test] + fn test_gemini_candidate_missing_finish_reason_reports_unspecified() { + let json = r#"{"candidates": [{}]}"#; + let response: GeminiResponse = serde_json::from_str(json).unwrap(); + let err = GeminiClient::parse_response(response) + .unwrap_err() + .to_string(); + assert!(err.contains("unspecified"), "unexpected error: {err}"); + } } diff --git a/crates/rvAgent/rvagent-cli/src/app.rs b/crates/rvAgent/rvagent-cli/src/app.rs index 68de1301da..2238b40773 100644 --- a/crates/rvAgent/rvagent-cli/src/app.rs +++ b/crates/rvAgent/rvagent-cli/src/app.rs @@ -409,17 +409,41 @@ impl App { }; // Wire the middleware pipeline (P0.3): resolve the configured - // middleware names (DEFAULT_MIDDLEWARE) into instances — unknown - // names warn and are skipped — and run all model calls through it. + // middleware names (DEFAULT_MIDDLEWARE) into instances — an unknown + // name is fatal — and run all model calls through it. + // + // The pipeline config carries the settings the middleware need to be + // built correctly; leaving `interrupt_on` unset gives HITL its + // conservative built-in gate rather than an empty (approve-everything) + // pattern list. The CLI has no interactive approval prompt yet, so + // gated calls fail closed; RVAGENT_AUTO_APPROVE=1 is the explicit, + // logged opt-out for unattended use. let middleware_names: Vec<&str> = self .config .middleware .iter() .map(|m| m.name.as_str()) .collect(); - let pipeline = Arc::new(rvagent_middleware::build_pipeline_from_names( - &middleware_names, - )); + let mut pipeline_config = rvagent_middleware::PipelineConfig::default(); + if matches!( + std::env::var("RVAGENT_AUTO_APPROVE").as_deref(), + Ok("1") | Ok("true") | Ok("yes") + ) { + // Straight to stderr, not just tracing: the TUI installs no + // subscriber and the non-TUI default is ERROR-only, so a `warn!` + // here is invisible in exactly the modes people run. A security + // downgrade the operator cannot see is one they cannot revoke. + eprintln!( + "warning: RVAGENT_AUTO_APPROVE set — HITL approval gate disabled; \ + all tool calls (including shell execution and file writes) run unattended" + ); + warn!("RVAGENT_AUTO_APPROVE set: HITL approval gate disabled; all tool calls run unattended"); + pipeline_config.interrupt_on = Some(Vec::new()); + } + let pipeline = Arc::new( + rvagent_middleware::build_pipeline_from_names(&middleware_names, &pipeline_config) + .context("failed to build middleware pipeline")?, + ); info!(middlewares = ?pipeline.names(), "middleware pipeline wired"); // Run before_agent hooks (state patching, context injection). diff --git a/crates/rvAgent/rvagent-core/src/graph.rs b/crates/rvAgent/rvagent-core/src/graph.rs index 973264201e..66e1fa9abe 100644 --- a/crates/rvAgent/rvagent-core/src/graph.rs +++ b/crates/rvAgent/rvagent-core/src/graph.rs @@ -265,14 +265,6 @@ impl AgentGraph { info!(node = ?current_node, tools = tool_definitions.len(), "graph: starting agent loop"); loop { - if iterations >= self.config.max_iterations { - warn!(iterations, "graph: max iterations reached"); - return Err(RvAgentError::timeout(format!( - "agent loop exceeded {} iterations", - self.config.max_iterations - ))); - } - match current_node { AgentNode::Start => { debug!("graph: Start → Agent"); @@ -280,6 +272,19 @@ impl AgentGraph { } AgentNode::Agent => { + // The budget is checked here rather than at the top of the + // loop so that reaching End *within* the budget completes + // the run. Checking before every node discarded a finished + // run whose last allowed iteration produced the answer — + // with `max_iterations: 1`, every successful single-turn + // run failed. + if iterations >= self.config.max_iterations { + warn!(iterations, "graph: max iterations reached"); + return Err(RvAgentError::timeout(format!( + "agent loop exceeded {} iterations", + self.config.max_iterations + ))); + } iterations += 1; debug!(iteration = iterations, "graph: invoking model"); @@ -400,9 +405,26 @@ impl AgentGraph { ); } } else { - // Sequential execution. + // Sequential execution. Each call still runs in its own + // task, so a panicking tool becomes a tool error here + // exactly as it does on the parallel path — a panic on + // the single-call path would otherwise take down the + // whole agent loop. for tc in &dispatch { - let result = self.tool_executor.execute(tc, &state).await; + let executor = Arc::clone(&self.tool_executor); + let exec_state = state.clone(); + let call = tc.clone(); + let name = tc.name.clone(); + let result = match tokio::spawn(async move { + executor.execute(&call, &exec_state).await + }) + .await + { + Ok(res) => res, + Err(join_err) => Err(RvAgentError::tool(format!( + "tool '{name}' execution task failed: {join_err}" + ))), + }; executed.insert( tc.id.clone(), truncate_tool_result( @@ -628,6 +650,63 @@ mod tests { assert!(contents[1].contains("result of ok_tool")); } + #[tokio::test] + async fn test_single_tool_panic_is_contained() { + // The single-call path runs sequentially even with parallel_tools on, + // so it needs its own containment — a panic here used to abort the run. + let model = MockModel::new(vec![ + Message::ai_with_tools( + "", + vec![ToolCall { + id: "tc1".into(), + name: "boom".into(), + args: serde_json::json!({}), + }], + ), + Message::ai("done"), + ]); + let graph = AgentGraph::with_config( + model, + PanickyExecutor, + GraphConfig { + parallel_tools: false, + ..GraphConfig::default() + }, + ); + + let result = graph.run(AgentState::new()).await.unwrap(); + + let contents: Vec<&str> = result + .messages + .iter() + .filter_map(|m| match m { + Message::Tool(t) => Some(t.content.as_str()), + _ => None, + }) + .collect(); + assert_eq!(contents.len(), 1); + assert!(contents[0].contains("Tool execution error")); + } + + #[tokio::test] + async fn test_max_iterations_of_one_allows_a_single_turn() { + // A run that answers on its last allowed iteration has not exceeded the + // budget; discarding it made max_iterations: 1 unusable. + let model = MockModel::new(vec![Message::ai("Hello!")]); + let graph = AgentGraph::with_config( + model, + MockToolExecutor, + GraphConfig { + max_iterations: 1, + ..GraphConfig::default() + }, + ); + + let result = graph.run(AgentState::with_system_message("sys")).await; + let state = result.expect("single-turn run within budget must succeed"); + assert!(matches!(state.messages.last(), Some(Message::Ai(_)))); + } + #[tokio::test] async fn test_max_iterations() { // Model always returns tool calls → should hit max iterations. diff --git a/crates/rvAgent/rvagent-core/src/masking.rs b/crates/rvAgent/rvagent-core/src/masking.rs index 5133b7f7cb..acea64a481 100644 --- a/crates/rvAgent/rvagent-core/src/masking.rs +++ b/crates/rvAgent/rvagent-core/src/masking.rs @@ -60,20 +60,31 @@ pub fn truncate_tool_result(content: String, max_bytes: usize) -> String { } // Reserve room for the marker so the result still respects the budget. let marker = "\n... [output truncated]"; - let budget = max_bytes.saturating_sub(marker.len()); - // Walk back to a char boundary so we never split a multi-byte sequence. - let mut end = budget.min(content.len()); - while end > 0 && !content.is_char_boundary(end) { - end -= 1; + // A cap smaller than the marker itself cannot carry the marker and stay + // within budget. The cap wins: it is what bounds context cost, and + // announcing the truncation is the part that can be given up. + if max_bytes < marker.len() { + return content[..floor_char_boundary(&content, max_bytes)].to_string(); } + let end = floor_char_boundary(&content, max_bytes - marker.len()); let mut out = String::with_capacity(end + marker.len()); out.push_str(&content[..end]); out.push_str(marker); out } +/// Largest index `<= max` that starts a character, so slicing there never +/// splits a multi-byte sequence. +fn floor_char_boundary(s: &str, max: usize) -> usize { + let mut n = max.min(s.len()); + while n > 0 && !s.is_char_boundary(n) { + n -= 1; + } + n +} + /// The placeholder substituted for an elided observation. fn placeholder(msg: &ToolMessage) -> String { let name = msg.tool_name.as_deref().unwrap_or("tool"); @@ -312,11 +323,21 @@ mod tests { fn truncation_never_splits_a_multibyte_char() { // Every char is 4 bytes, so a naive byte cut would split one. let content = "🙂".repeat(100); - for cap in [10usize, 33, 50, 77, 99] { + for cap in [33usize, 50, 77, 99] { let out = truncate_tool_result(content.clone(), cap); - // The real assertion is simply that this did not panic and the - // result is valid UTF-8 by construction. + // The result is valid UTF-8 by construction; the assertion is that + // this did not panic and stayed inside the cap. assert!(out.ends_with("[output truncated]"), "cap {cap}"); + assert!(out.len() <= cap, "cap {cap} exceeded: {} bytes", out.len()); + } + } + + #[test] + fn truncation_respects_caps_too_small_for_the_marker() { + // The cap bounds context cost, so it wins over announcing the cut. + for cap in [0usize, 1, 5, 10, 22] { + let out = truncate_tool_result("🙂".repeat(100), cap); + assert!(out.len() <= cap, "cap {cap} exceeded: {} bytes", out.len()); } } } diff --git a/crates/rvAgent/rvagent-middleware/src/hitl.rs b/crates/rvAgent/rvagent-middleware/src/hitl.rs index 425ee9f683..65835eb11d 100644 --- a/crates/rvAgent/rvagent-middleware/src/hitl.rs +++ b/crates/rvAgent/rvagent-middleware/src/hitl.rs @@ -1,5 +1,10 @@ -//! HumanInTheLoopMiddleware — intercepts tool calls matching interrupt patterns, -//! pausing execution awaiting human approval. +//! HumanInTheLoopMiddleware — intercepts tool calls matching interrupt +//! patterns and drops them, reporting the block to the model. +//! +//! There is no approval channel yet: a blocked call is not queued for a human +//! and never resumes. The middleware is a gate, not a pause, and the message it +//! injects says so — telling the model to wait for an approval that cannot +//! arrive is what turns a gate into a hang. use async_trait::async_trait; @@ -15,8 +20,9 @@ pub enum ApprovalDecision { /// Middleware that intercepts tool calls matching configurable interrupt patterns. /// -/// - `wrap_model_call`: after the model returns, checks if any tool calls match -/// the interrupt patterns. If so, pauses execution awaiting human approval. +/// - `wrap_model_call`: after the model returns, drops any tool calls matching +/// the interrupt patterns and appends a note naming them, so the model can +/// adapt rather than wait. pub struct HumanInTheLoopMiddleware { /// Tool name patterns that trigger human approval. interrupt_patterns: Vec, @@ -78,8 +84,16 @@ impl Middleware for HumanInTheLoopMiddleware { if !content.is_empty() { content.push_str("\n\n"); } + // State what actually happened: the calls were dropped, not queued. + // "Awaiting approval" implied something would come back for them and + // nothing does, which leaves the model waiting on a resolution that + // never arrives instead of adapting. content.push_str(&format!( - "[HITL] Awaiting approval for: {}", + "[HITL] Blocked tool call(s) requiring approval: {}. No approval \ + mechanism is wired in this runtime, so these calls were not \ + executed and will not be retried. Configure `interrupt_on` to \ + change which tools are gated (or set RVAGENT_AUTO_APPROVE=1 in \ + the CLI to run unattended).", pending_names.join(", ") )); } @@ -156,6 +170,31 @@ mod tests { assert!(response.content().contains("execute")); } + #[tokio::test] + async fn test_block_message_states_reality_and_next_step() { + // The message is the model's only signal about what happened; if it + // says "awaiting" when nothing is coming, the model stalls. + let mw = HumanInTheLoopMiddleware::new(vec!["execute".into()]); + let response = mw + .wrap_model_call(ModelRequest::new(vec![Message::human("x")]), &EchoHandler) + .await; + let content = response.content().to_string(); + + assert!(content.contains("execute"), "must name the blocked call"); + assert!( + !content.contains("Awaiting"), + "must not imply a pending approval that never resolves: {content}" + ); + assert!( + content.contains("not be retried"), + "must tell the model the call is gone for good: {content}" + ); + assert!( + content.contains("interrupt_on"), + "must name the knob that changes gating: {content}" + ); + } + #[tokio::test] async fn test_wrap_model_call_no_interrupt() { let mw = HumanInTheLoopMiddleware::new(vec!["dangerous_tool".into()]); diff --git a/crates/rvAgent/rvagent-middleware/src/lib.rs b/crates/rvAgent/rvagent-middleware/src/lib.rs index ee1b431852..c706f5d3d7 100644 --- a/crates/rvAgent/rvagent-middleware/src/lib.rs +++ b/crates/rvAgent/rvagent-middleware/src/lib.rs @@ -267,11 +267,13 @@ pub fn build_default_pipeline(config: &PipelineConfig) -> MiddlewarePipeline { tool_sanitizer::ToolResultSanitizerMiddleware::new(), )); - if let Some(patterns) = &config.interrupt_on { - middlewares.push(Box::new(hitl::HumanInTheLoopMiddleware::new( - patterns.clone(), - ))); - } + // Unconditional, and via the same fallback the by-name path uses: an + // unconfigured caller (the ACP agent reaches this with a default config) + // must not silently get an agent with no approval gate at all. An explicit + // `Some(vec![])` still opts out — it matches nothing. + middlewares.push(Box::new(hitl::HumanInTheLoopMiddleware::new( + interrupt_patterns(config), + ))); MiddlewarePipeline::new(middlewares) } @@ -280,55 +282,151 @@ pub fn build_default_pipeline(config: &PipelineConfig) -> MiddlewarePipeline { // Name-based middleware resolution (CLI wiring) // --------------------------------------------------------------------------- +/// Tool-name patterns gated by HITL when no `interrupt_on` is configured. +/// +/// Every pipeline builder falls back to this set: an approval gate that +/// approves everything — or a pipeline with no gate at all — is worse than a +/// real one, because it reads as protection that isn't there. Shell execution +/// and file mutation are the irreversible operations, so those are what a +/// caller who never configured `interrupt_on` gets gated on. +/// +/// `write_todos` is deliberately absent — gating the agent's own scratchpad +/// would interrupt every turn without protecting anything. +pub const DEFAULT_INTERRUPT_PATTERNS: &[&str] = &[ + // Shell / arbitrary command execution. + "execute", + "execute_command", + "shell*", + "bash*", + "run_command*", + // File mutation. + "write_file", + "edit_file", + "apply_patch", + "delete_file", + "move_file", +]; + +fn default_interrupt_patterns() -> Vec { + DEFAULT_INTERRUPT_PATTERNS + .iter() + .map(|s| (*s).to_string()) + .collect() +} + +/// The HITL patterns `config` asks for, or the conservative built-in set. +/// +/// Both pipeline builders resolve gating through here. The two paths having +/// separate fallbacks is what produced an ACP agent with no gate at all while +/// the CLI had one. +fn interrupt_patterns(config: &PipelineConfig) -> Vec { + config + .interrupt_on + .clone() + .unwrap_or_else(default_interrupt_patterns) +} + +/// A middleware name that could not be resolved to an implementation. +/// +/// Unknown names are fatal rather than skipped: a typo in the middleware list +/// silently drops whatever that entry was supposed to do, and the entries most +/// worth typo-ing are the security ones (`hitl`, `unicode_security`, +/// `tool_result_sanitizer`). +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct UnknownMiddlewareError { + pub name: String, +} + +impl fmt::Display for UnknownMiddlewareError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!( + f, + "unknown middleware '{}' (see rvagent_middleware::middleware_by_name for valid names)", + self.name + ) + } +} + +impl std::error::Error for UnknownMiddlewareError {} + /// Resolve a middleware name (as used in `RvAgentConfig::middleware` / /// the CLI `DEFAULT_MIDDLEWARE` list) into a middleware instance. /// -/// Returns `None` for unknown names — callers should warn and skip. -pub fn middleware_by_name(name: &str) -> Option> { - match name { - "todo" | "todos" | "todolist" => Some(Box::new(todolist::TodoListMiddleware::new())), - "memory" => Some(Box::new(memory::MemoryMiddleware::new(vec![ - "AGENTS.md".into() - ]))), - "skills" => Some(Box::new(skills::SkillsMiddleware::new(vec![ - ".skills".into() - ]))), - "filesystem" => Some(Box::new(filesystem::FilesystemMiddleware::new())), - "subagent" | "subagents" => Some(Box::new(subagents::SubAgentMiddleware::new())), - "summarization" => Some(Box::new(summarization::SummarizationMiddleware::new( - 100_000, 0.85, 0.10, +/// `config` supplies the same settings [`build_default_pipeline`] uses, so the +/// two construction paths produce equivalently configured middleware. +/// +/// Returns `Err` for unknown names. +pub fn middleware_by_name( + name: &str, + config: &PipelineConfig, +) -> std::result::Result, UnknownMiddlewareError> { + let mw: Box = match name { + "todo" | "todos" | "todolist" => Box::new(todolist::TodoListMiddleware::new()), + "memory" => Box::new(memory::MemoryMiddleware::new( + config + .memory_sources + .clone() + .unwrap_or_else(|| vec!["AGENTS.md".into()]), + )), + "skills" => Box::new(skills::SkillsMiddleware::new( + config + .skill_sources + .clone() + .unwrap_or_else(|| vec![".skills".into()]), + )), + "filesystem" => Box::new(filesystem::FilesystemMiddleware::new()), + "subagent" | "subagents" => Box::new(subagents::SubAgentMiddleware::new()), + "summarization" => Box::new(summarization::SummarizationMiddleware::new( + 100_000, 0.75, 0.10, + )), + "prompt_caching" => Box::new(prompt_caching::PromptCachingMiddleware::new()), + "patch_tool_calls" => Box::new(patch_tool_calls::PatchToolCallsMiddleware::new()), + "witness" => Box::new(witness::WitnessMiddleware::new()), + "tool_result_sanitizer" | "tool_sanitizer" => { + Box::new(tool_sanitizer::ToolResultSanitizerMiddleware::new()) + } + "hitl" => Box::new(hitl::HumanInTheLoopMiddleware::new(interrupt_patterns( + config, ))), - "prompt_caching" => Some(Box::new(prompt_caching::PromptCachingMiddleware::new())), - "patch_tool_calls" => Some(Box::new(patch_tool_calls::PatchToolCallsMiddleware::new())), - "witness" => Some(Box::new(witness::WitnessMiddleware::new())), - "tool_result_sanitizer" | "tool_sanitizer" => Some(Box::new( - tool_sanitizer::ToolResultSanitizerMiddleware::new(), + "retry" => Box::new(retry::RetryMiddleware::default()), + "hnsw" => Box::new(hnsw::HnswMiddleware::new( + config.hnsw_config.clone().unwrap_or_default(), )), - // HITL with no interrupt patterns configured never interrupts. - "hitl" => Some(Box::new(hitl::HumanInTheLoopMiddleware::new(Vec::new()))), - "retry" => Some(Box::new(retry::RetryMiddleware::default())), - "hnsw" => Some(Box::new(hnsw::HnswMiddleware::default_config())), - "sona" => Some(Box::new(sona::SonaMiddleware::default_config())), - "unicode_security" => Some(Box::new(UnicodeSecurityMiddleware::strict())), - "mcp_bridge" => Some(Box::new(mcp_bridge::McpBridgeMiddleware::new())), - _ => None, - } + "sona" => Box::new(sona::SonaMiddleware::new( + config.sona_config.clone().unwrap_or_default(), + )), + "unicode_security" => Box::new( + UnicodeSecurityMiddleware::new( + config + .unicode_security_config + .clone() + .unwrap_or_else(UnicodeSecurityConfig::strict), + ) + .with_input_sanitization(true) + .with_output_sanitization(false), // Log only by default + ), + "mcp_bridge" => Box::new(mcp_bridge::McpBridgeMiddleware::new()), + _ => { + return Err(UnknownMiddlewareError { + name: name.to_string(), + }) + } + }; + Ok(mw) } /// Build a pipeline from an ordered list of middleware names. -/// Unknown names are logged (warn) and skipped. -pub fn build_pipeline_from_names>(names: &[S]) -> MiddlewarePipeline { +/// +/// An unknown name is an error, not a skip — see [`UnknownMiddlewareError`]. +pub fn build_pipeline_from_names>( + names: &[S], + config: &PipelineConfig, +) -> std::result::Result { let mut pipeline = MiddlewarePipeline::empty(); for name in names { - match middleware_by_name(name.as_ref()) { - Some(mw) => pipeline.push(mw), - None => tracing::warn!( - "unknown middleware '{}' — skipping (see rvagent_middleware::middleware_by_name)", - name.as_ref() - ), - } + pipeline.push(middleware_by_name(name.as_ref(), config)?); } - pipeline + Ok(pipeline) } // --------------------------------------------------------------------------- @@ -484,8 +582,48 @@ mod tests { let config = PipelineConfig::default(); let pipeline = build_default_pipeline(&config); // todo, filesystem, subagent, prompt_caching, patch_tool_calls, - // tool_sanitizer = 6. Summarization is opt-in (ADR-274). - assert!(pipeline.len() >= 6); + // tool_sanitizer, hitl = 7. Summarization is opt-in (ADR-274). + assert!(pipeline.len() >= 7); + assert!(pipeline.names().contains(&"hitl")); + } + + #[tokio::test] + async fn test_build_default_pipeline_gates_execute_without_config() { + // rvagent-acp builds this pipeline with a default config whenever the + // agent config lists no middleware; it must not come out ungated. + let pipeline = build_default_pipeline(&PipelineConfig::default()); + let response = pipeline + .run_wrap_model_call( + ModelRequest::new(vec![Message::human("go")]), + &DangerousCallHandler, + ) + .await; + + let survived: Vec<&str> = response + .tool_calls + .iter() + .map(|c| c.name.as_str()) + .collect(); + assert_eq!(survived, vec!["read_file"]); + assert!(response.content().contains("[HITL]")); + } + + #[tokio::test] + async fn test_build_default_pipeline_explicit_empty_interrupt_on_disables_gating() { + // The documented opt-out for unattended runs must keep working. + let pipeline = build_default_pipeline(&PipelineConfig { + interrupt_on: Some(Vec::new()), + ..PipelineConfig::default() + }); + let response = pipeline + .run_wrap_model_call( + ModelRequest::new(vec![Message::human("go")]), + &DangerousCallHandler, + ) + .await; + + assert_eq!(response.tool_calls.len(), 3); + assert!(!response.content().contains("[HITL]")); } #[test] @@ -525,7 +663,7 @@ mod tests { "enabling summarization must add exactly one middleware" ); // And it must still be constructible by name, so the fallback is real. - assert!(middleware_by_name("summarization").is_some()); + assert!(middleware_by_name("summarization", &PipelineConfig::default()).is_ok()); } #[test] @@ -537,6 +675,7 @@ mod tests { #[test] fn test_middleware_by_name_known() { + let config = PipelineConfig::default(); for name in [ "todo", "memory", @@ -551,18 +690,99 @@ mod tests { "hitl", "retry", ] { - assert!(middleware_by_name(name).is_some(), "should resolve {name}"); + assert!( + middleware_by_name(name, &config).is_ok(), + "should resolve {name}" + ); } } #[test] fn test_middleware_by_name_unknown() { - assert!(middleware_by_name("does_not_exist").is_none()); + let err = middleware_by_name("does_not_exist", &PipelineConfig::default()).unwrap_err(); + assert_eq!(err.name, "does_not_exist"); } #[test] - fn test_build_pipeline_from_names_skips_unknown() { - let pipeline = build_pipeline_from_names(&["todo", "bogus", "filesystem"]); - assert_eq!(pipeline.len(), 2); + fn test_build_pipeline_from_names_rejects_unknown() { + // A typo must not silently drop a middleware — the ones most worth + // typo-ing are the security ones. + let err = + build_pipeline_from_names(&["todo", "bogus", "filesystem"], &PipelineConfig::default()) + .err() + .expect("an unknown middleware name must fail the build"); + assert_eq!(err.name, "bogus"); + } + + /// Returns one shell call and one read-only call, so an approval gate is + /// observable by which calls survive. + struct DangerousCallHandler; + + #[async_trait] + impl ModelHandler for DangerousCallHandler { + async fn call(&self, _request: ModelRequest) -> ModelResponse { + let mut response = ModelResponse::text(""); + response.tool_calls = vec![ + ToolCall { + id: "c1".into(), + name: "execute".into(), + args: serde_json::json!({"command": "rm -rf /"}), + }, + ToolCall { + id: "c2".into(), + name: "write_file".into(), + args: serde_json::json!({"path": "a.txt"}), + }, + ToolCall { + id: "c3".into(), + name: "read_file".into(), + args: serde_json::json!({"path": "a.txt"}), + }, + ]; + response + } + } + + #[tokio::test] + async fn test_by_name_hitl_gates_shell_and_writes_without_config() { + // The unconfigured by-name path is what the CLI runs; an approval gate + // that approves everything would be worse than no gate at all. + let pipeline = build_pipeline_from_names(&["hitl"], &PipelineConfig::default()).unwrap(); + let response = pipeline + .run_wrap_model_call( + ModelRequest::new(vec![Message::human("go")]), + &DangerousCallHandler, + ) + .await; + + let survived: Vec<&str> = response + .tool_calls + .iter() + .map(|c| c.name.as_str()) + .collect(); + assert_eq!(survived, vec!["read_file"]); + assert!(response.content().contains("[HITL]")); + } + + #[tokio::test] + async fn test_by_name_hitl_honours_configured_interrupt_on() { + let config = PipelineConfig { + interrupt_on: Some(vec!["read_file".into()]), + ..PipelineConfig::default() + }; + let pipeline = build_pipeline_from_names(&["hitl"], &config).unwrap(); + let response = pipeline + .run_wrap_model_call( + ModelRequest::new(vec![Message::human("go")]), + &DangerousCallHandler, + ) + .await; + + let survived: Vec<&str> = response + .tool_calls + .iter() + .map(|c| c.name.as_str()) + .collect(); + assert_eq!(survived, vec!["execute", "write_file"]); } } diff --git a/crates/rvAgent/rvagent-middleware/src/retry.rs b/crates/rvAgent/rvagent-middleware/src/retry.rs index 298d1b9d70..8bc6fb941b 100644 --- a/crates/rvAgent/rvagent-middleware/src/retry.rs +++ b/crates/rvAgent/rvagent-middleware/src/retry.rs @@ -22,6 +22,13 @@ fn is_transient_error(response: &ModelResponse) -> bool { || content.to_ascii_lowercase().starts_with("error:") } +/// Upper bound on a single backoff delay (1 minute). +/// +/// Doubling is unbounded by nature, so a caller-supplied `max_retries` of 63 +/// yields a sleep of ~584 million years — indistinguishable from a hang. The +/// cap is what makes the backoff recoverable rather than terminal. +pub const MAX_BACKOFF_MS: u64 = 60_000; + /// Retry middleware that wraps model calls with exponential backoff. /// /// # Configuration @@ -31,7 +38,8 @@ fn is_transient_error(response: &ModelResponse) -> bool { /// | `max_retries` | 3 | Maximum number of retry attempts | /// | `initial_delay_ms` | 100 | Delay before the first retry (ms) | /// -/// The delay doubles after each attempt: `initial_delay_ms * 2^attempt`. +/// The delay doubles after each attempt: `initial_delay_ms * 2^attempt`, +/// capped at [`MAX_BACKOFF_MS`]. /// /// # Metrics /// @@ -67,6 +75,16 @@ impl RetryMiddleware { self.total_retries.load(Ordering::Relaxed) } + /// Backoff delay before retry `attempt` (0-based), capped and overflow-safe. + /// + /// Saturating arithmetic keeps `2^attempt` from panicking in debug builds; + /// the cap is what keeps the resulting sleep finite in practice. + fn backoff_ms(&self, attempt: u32) -> u64 { + self.initial_delay_ms + .saturating_mul(2u64.saturating_pow(attempt)) + .min(MAX_BACKOFF_MS) + } + /// Reset all counters to zero. pub fn reset_metrics(&self) { self.retry_count.store(0, Ordering::Relaxed); @@ -101,7 +119,7 @@ impl Middleware for RetryMiddleware { self.retry_count.fetch_add(1, Ordering::Relaxed); for attempt in 0..self.max_retries { - let delay_ms = self.initial_delay_ms * 2u64.pow(attempt); + let delay_ms = self.backoff_ms(attempt); tokio::time::sleep(Duration::from_millis(delay_ms)).await; self.total_retries.fetch_add(1, Ordering::Relaxed); @@ -128,6 +146,25 @@ mod tests { use crate::{Message, ModelRequest, ModelResponse}; use std::sync::atomic::AtomicU32; + #[test] + fn test_backoff_doubles_then_caps() { + let mw = RetryMiddleware::new(3, 100); + assert_eq!(mw.backoff_ms(0), 100); + assert_eq!(mw.backoff_ms(1), 200); + assert_eq!(mw.backoff_ms(2), 400); + // Doubling past the cap flattens instead of running away. + assert_eq!(mw.backoff_ms(20), MAX_BACKOFF_MS); + } + + #[test] + fn test_backoff_survives_absurd_attempt_counts() { + // attempt 63 overflows `2^attempt` and, uncapped, sleeps ~584M years. + let mw = RetryMiddleware::new(u32::MAX, u64::MAX); + for attempt in [31u32, 63, 64, u32::MAX] { + assert_eq!(mw.backoff_ms(attempt), MAX_BACKOFF_MS, "attempt {attempt}"); + } + } + /// A handler that fails `n` times then succeeds. struct FailNHandler { remaining_failures: AtomicU32, diff --git a/crates/rvAgent/rvagent-middleware/src/summarization.rs b/crates/rvAgent/rvagent-middleware/src/summarization.rs index 82e60c2b8f..dbaaf7ae0c 100644 --- a/crates/rvAgent/rvagent-middleware/src/summarization.rs +++ b/crates/rvAgent/rvagent-middleware/src/summarization.rs @@ -6,6 +6,20 @@ use uuid::Uuid; use crate::{Message, Middleware, ModelHandler, ModelRequest, ModelResponse}; +/// Bytes of each user message kept in a compaction summary preview. +const PREVIEW_BYTES: usize = 100; + +/// Largest index `<= max` that starts a character, so slicing there can never +/// split a multi-byte sequence. Conversation content is arbitrary user text, +/// so a byte-index slice is a panic waiting for the first non-ASCII message. +fn floor_char_boundary(s: &str, max: usize) -> usize { + let mut n = max.min(s.len()); + while n > 0 && !s.is_char_boundary(n) { + n -= 1; + } + n +} + /// Trigger configuration for auto-compaction. pub enum TriggerConfig { /// Fraction of context window that triggers compaction. @@ -70,8 +84,11 @@ impl SummarizationMiddleware { for msg in messages { if let Message::Human(h) = msg { - let preview = if h.content.len() > 100 { - format!("{}...", &h.content[..100]) + let preview = if h.content.len() > PREVIEW_BYTES { + format!( + "{}...", + &h.content[..floor_char_boundary(&h.content, PREVIEW_BYTES)] + ) } else { h.content.clone() }; @@ -240,6 +257,30 @@ mod tests { assert!(summary.content().contains("What is Rust?")); } + #[test] + fn test_summarize_does_not_split_multibyte_chars() { + // Byte 100 lands mid-character for a 3-byte-per-char message, which a + // plain `&content[..100]` would panic on. + let content = "日".repeat(200); + let messages = vec![Message::human(content.clone())]; + let summary = SummarizationMiddleware::summarize(&messages); + let text = summary.content().to_string(); + assert!(text.contains("...")); + // 100 / 3 = 33 whole characters fit. + assert!(text.contains(&"日".repeat(33))); + assert!(!text.contains(&"日".repeat(34))); + } + + #[test] + fn test_summarize_preview_boundary_cases() { + for len in [98usize, 99, 100, 101, 150] { + let messages = vec![Message::human("é".repeat(len))]; + // The assertion is that this does not panic and stays valid UTF-8. + let summary = SummarizationMiddleware::summarize(&messages); + assert!(summary.content().contains("User:")); + } + } + #[test] fn test_format_for_offload() { let messages = vec![Message::human("test content")]; diff --git a/crates/rvAgent/rvagent-tools/Cargo.toml b/crates/rvAgent/rvagent-tools/Cargo.toml index 8e9acb67c9..4becdf4c14 100644 --- a/crates/rvAgent/rvagent-tools/Cargo.toml +++ b/crates/rvAgent/rvagent-tools/Cargo.toml @@ -20,6 +20,11 @@ async-trait = "0.1" glob = "0.3" walkdir = "2.5" +# Killing a timed-out command's whole process group needs kill(2) on a +# negative pid, which std does not expose. +[target.'cfg(unix)'.dependencies] +libc = "0.2" + [dev-dependencies] criterion = { workspace = true } tokio = { workspace = true, features = ["test-util"] } diff --git a/crates/rvAgent/rvagent-tools/src/lib.rs b/crates/rvAgent/rvagent-tools/src/lib.rs index e76d748b55..8d53ba5a32 100644 --- a/crates/rvAgent/rvagent-tools/src/lib.rs +++ b/crates/rvAgent/rvagent-tools/src/lib.rs @@ -572,7 +572,7 @@ pub(crate) mod tests_common { if offset >= lines.len() { return Ok(String::new()); } - let end = (offset + limit).min(lines.len()); + let end = offset.saturating_add(limit).min(lines.len()); Ok(lines[offset..end].join("\n")) } None => Err(format!("File not found: {}", path)), diff --git a/crates/rvAgent/rvagent-tools/src/local.rs b/crates/rvAgent/rvagent-tools/src/local.rs index 6b81183772..1b4db2b440 100644 --- a/crates/rvAgent/rvagent-tools/src/local.rs +++ b/crates/rvAgent/rvagent-tools/src/local.rs @@ -4,11 +4,33 @@ //! # Path confinement //! //! Every path a tool supplies is model-controlled input, so it is resolved -//! against a fixed root and rejected if it escapes. Resolution is symlink-aware: -//! the deepest existing ancestor is canonicalized before the containment check, -//! so a symlink pointing outside the root cannot be used as a bridge. Without -//! this, `read_file {"path": "/etc/passwd"}` — or a `write_file` anywhere on -//! disk — is a single tool call away. +//! against a fixed root and rejected if it escapes. Without this, +//! `read_file {"path": "/etc/passwd"}` — or a `write_file` anywhere on disk — +//! is a single tool call away. +//! +//! What confinement guarantees: +//! +//! - **Lexical escapes** (`..`, absolute paths) are normalized and then +//! rejected by a containment check against the canonical root. +//! - **Symlinks** cannot bridge out. The deepest existing ancestor is +//! canonicalized before the containment check, a dangling link is refused +//! outright rather than treated as a not-yet-existing path, and writes +//! refuse to travel through a link at all — including one pointing back +//! inside the root, which would otherwise pass containment. +//! - **Hard links** cannot bridge out on unix: a write to a file with +//! `nlink > 1` is refused, because the same inode may have another name +//! outside the root and nothing in the path tells us it does not. +//! +//! What it does not guarantee: +//! +//! - **TOCTOU races.** Resolution and the write are separate syscalls, so a +//! concurrent process that swaps a path component between them can still +//! redirect a write. Defending that needs `openat`-based traversal with +//! `O_NOFOLLOW`, which is out of scope here; the threat model is a +//! misbehaving model, not a local attacker racing the agent. +//! - **Non-unix hard links.** The `nlink` check is unix-only. +//! - **Anything below the root.** Confinement is a boundary, not a +//! permission system: every file under the root is writable. use std::path::{Component, Path, PathBuf}; @@ -45,17 +67,9 @@ impl LocalFsBackend { /// climb out, then the deepest existing ancestor is canonicalized to defeat /// symlink escapes. fn resolve(&self, path: &str) -> Result { - let raw = Path::new(path); - let joined = if raw.is_absolute() { - raw.to_path_buf() - } else if path.is_empty() || path == "." { - self.root.clone() - } else { - self.root.join(raw) - }; - - let normalized = lexical_normalize(&joined); - let resolved = canonicalize_existing_prefix(&normalized); + let normalized = self.literal_path(path); + let resolved = canonicalize_existing_prefix(&normalized) + .map_err(|e| format!("Error: path '{path}' is not resolvable: {e}"))?; if resolved.starts_with(&self.root) { Ok(resolved) @@ -66,13 +80,70 @@ impl LocalFsBackend { } } + /// The lexically normalized path, before any symlink resolution. + /// + /// This is what the caller literally named. `resolve` canonicalizes it for + /// the containment check, which erases the fact that a component *was* a + /// symlink — so checks that care about that must run against this. + fn literal_path(&self, path: &str) -> PathBuf { + let raw = Path::new(path); + let joined = if raw.is_absolute() { + raw.to_path_buf() + } else if path.is_empty() || path == "." { + self.root.clone() + } else { + self.root.join(raw) + }; + lexical_normalize(&joined) + } + /// Resolve for write-style operations, returning the error as a /// `WriteResult` rather than a bare string. + /// + /// Beyond containment, a write must never travel *through* a link. + /// + /// `std::fs::write` follows a symlink, so a link inside the root is a + /// write primitive for wherever it points — and `resolve` cannot catch + /// it, because canonicalizing a link to an in-root file yields an in-root + /// path. The link is checked pre-resolution instead. + /// + /// A hard link is invisible to both checks: there is no link to follow and + /// `symlink_metadata` reports an ordinary file. The only signal is the + /// inode's link count, so a file with more than one name is refused — + /// another of those names may well be outside the root. fn resolve_for_write(&self, path: &str) -> Result { - self.resolve(path).map_err(|e| WriteResult { + let fail = |e: String| WriteResult { error: Some(e), ..Default::default() - }) + }; + let resolved = self.resolve(path).map_err(fail)?; + let literal = self.literal_path(path); + if let Ok(meta) = std::fs::symlink_metadata(&literal) { + if meta.file_type().is_symlink() { + return Err(fail(format!( + "Error: refusing to write through symlink '{}'. \ + Write to the link's target directly.", + literal.display() + ))); + } + } + #[cfg(unix)] + { + use std::os::unix::fs::MetadataExt; + if let Ok(meta) = std::fs::metadata(&resolved) { + // Directories always have several names (`.`, `..`, each + // child), so the count only means anything for a file. + if meta.is_file() && meta.nlink() > 1 { + return Err(fail(format!( + "Error: refusing to write through a hard link (nlink > 1): '{}'. \ + The file has another name, possibly outside the workspace, \ + which this write would also change.", + resolved.display() + ))); + } + } + } + Ok(resolved) } } @@ -106,17 +177,30 @@ fn lexical_normalize(path: &Path) -> PathBuf { /// component is a symlink out of the root, canonicalization exposes it before /// the `starts_with` test. `path` must already be lexically normalized, so /// re-appending the tail cannot reintroduce `..`. -fn canonicalize_existing_prefix(path: &Path) -> PathBuf { +/// +/// A *dangling* symlink is rejected outright. It cannot be canonicalized, so +/// treating it as part of the non-existent tail would leave it looking like a +/// fresh path under the canonical root — while an actual write through it +/// lands wherever the link points, anywhere on disk. +fn canonicalize_existing_prefix(path: &Path) -> Result { let mut existing = path; let mut tail: Vec<&std::ffi::OsStr> = Vec::new(); loop { + if let Ok(meta) = std::fs::symlink_metadata(existing) { + if meta.file_type().is_symlink() && std::fs::canonicalize(existing).is_err() { + return Err(format!( + "'{}' is a broken symlink; refusing to resolve through it", + existing.display() + )); + } + } if let Ok(canonical) = std::fs::canonicalize(existing) { let mut out = canonical; for seg in tail.iter().rev() { out.push(seg); } - return out; + return Ok(out); } match (existing.file_name(), existing.parent()) { (Some(name), Some(parent)) => { @@ -124,7 +208,7 @@ fn canonicalize_existing_prefix(path: &Path) -> PathBuf { existing = parent; } // Nothing along the chain exists; fall back to the lexical path. - _ => return path.to_path_buf(), + _ => return Ok(path.to_path_buf()), } } } @@ -166,7 +250,11 @@ impl Backend for LocalFsBackend { if offset >= lines.len() { return Ok(String::new()); } - let end = (offset + limit).min(lines.len()); + // `limit` is model-supplied and arrives as an unbounded integer, so a + // plain `offset + limit` overflows and panics on values like u64::MAX. + // Saturating means "the rest of the file", which is what such a limit + // asks for anyway. + let end = offset.saturating_add(limit).min(lines.len()); Ok(lines[offset..end].join("\n")) } @@ -308,94 +396,166 @@ impl Backend for LocalFsBackend { use std::time::Duration; // Security: environment sanitization — strip sensitive variables - // (SEC-005 / ADR-103 C2). Only a safe allowlist reaches the child. + // (SEC-005 / ADR-103 C2). The allowlist *is* the control: nothing + // outside it reaches the child, so no name-pattern denylist is needed. const SAFE_ENV_VARS: &[&str] = &[ "PATH", "HOME", "USER", "SHELL", "LANG", "LC_ALL", "LC_CTYPE", "TERM", "TMPDIR", "TZ", ]; - // Patterns identifying vars that must never reach child processes. - const SENSITIVE_PATTERNS: &[&str] = &[ - "SECRET", - "KEY", - "TOKEN", - "PASSWORD", - "CREDENTIAL", - "AWS_", - "AZURE_", - "GCP_", - "DATABASE_URL", - "PRIVATE", - "API_KEY", - "AUTH", - "BEARER", - "JWT", - "SESSION", - ]; let mut cmd = Command::new("sh"); cmd.arg("-c").arg(command).current_dir(&self.root); cmd.env_clear(); for var in SAFE_ENV_VARS { if let Ok(val) = std::env::var(var) { - let upper = var.to_uppercase(); - let sensitive = SENSITIVE_PATTERNS.iter().any(|pat| upper.contains(pat)); - if !sensitive { - cmd.env(var, val); - } + cmd.env(var, val); } } - cmd.stdout(Stdio::piped()).stderr(Stdio::piped()); + cmd.stdin(Stdio::null()) + .stdout(Stdio::piped()) + .stderr(Stdio::piped()); + + // Put the child in its own process group so a timeout can kill the + // whole tree. `child.kill()` signals only `sh`, leaving any grandchild + // running — and holding the pipe open — after we have given up on it. + #[cfg(unix)] + { + use std::os::unix::process::CommandExt; + cmd.process_group(0); + } let timeout = if timeout_secs == 0 { 30 } else { timeout_secs }; let deadline = std::time::Instant::now() + Duration::from_secs(timeout as u64); let mut child = cmd.spawn().map_err(|e| format!("execute failed: {}", e))?; - // Poll for completion with a deadline to enforce the timeout. - loop { + // Drain both pipes concurrently. A pipe holds only ~64 KB; if nobody + // reads while we poll, a chatty command blocks on write until the + // deadline and is then reported as a timeout it never actually hit. + // + // Each stream keeps one byte past the cap so that an oversized stream + // still trips the `> cap` test below and gets the truncation marker, + // rather than arriving as a silently shortened but cap-sized string. + let keep = MAX_OUTPUT_BYTES.saturating_add(1); + let out_pipe = child.stdout.take(); + let err_pipe = child.stderr.take(); + let out_reader = std::thread::spawn(move || match out_pipe { + Some(p) => drain_capped(p, keep), + None => Vec::new(), + }); + let err_reader = std::thread::spawn(move || match err_pipe { + Some(p) => drain_capped(p, keep), + None => Vec::new(), + }); + + let mut timed_out = false; + let status = loop { match child .try_wait() .map_err(|e| format!("wait failed: {}", e))? { - Some(_) => break, + Some(status) => break Some(status), None => { if std::time::Instant::now() >= deadline { - let _ = child.kill(); - return Ok(ExecuteResponse { - output: format!("Command timed out after {} seconds", timeout), - exit_code: -1, - }); + kill_process_tree(&mut child); + // Reap, or the child lingers as a zombie for the life + // of the process. + let _ = child.wait(); + timed_out = true; + break None; } - std::thread::sleep(Duration::from_millis(50)); + std::thread::sleep(Duration::from_millis(10)); } } - } + }; - let output = child - .wait_with_output() - .map_err(|e| format!("output collection failed: {}", e))?; + // The pipes are closed once the tree is gone, so these always finish. + let stdout_bytes = out_reader.join().unwrap_or_default(); + let stderr_bytes = err_reader.join().unwrap_or_default(); + + if timed_out { + return Ok(ExecuteResponse { + output: format!("Command timed out after {} seconds", timeout), + exit_code: -1, + }); + } - let stdout = String::from_utf8_lossy(&output.stdout); - let stderr = String::from_utf8_lossy(&output.stderr); - let mut combined = if stderr.is_empty() { + let stdout = String::from_utf8_lossy(&stdout_bytes); + let stderr = String::from_utf8_lossy(&stderr_bytes); + let combined = if stderr.is_empty() { stdout.into_owned() } else { format!("{}\n{}", stdout, stderr) }; - // Security: cap output size to prevent memory exhaustion. - const MAX_OUTPUT_BYTES: usize = 1024 * 1024; - if combined.len() > MAX_OUTPUT_BYTES { - combined.truncate(MAX_OUTPUT_BYTES); - combined.push_str("\n... [output truncated at 1 MB]"); - } - Ok(ExecuteResponse { - output: combined, - exit_code: output.status.code().unwrap_or(-1), + output: truncate_output(combined, MAX_OUTPUT_BYTES), + exit_code: status.and_then(|s| s.code()).unwrap_or(-1), }) } } +/// Security: cap on captured command output, to prevent memory exhaustion. +const MAX_OUTPUT_BYTES: usize = 1024 * 1024; + +/// Read a pipe to EOF, keeping at most `cap` bytes. +/// +/// Reading past the cap and discarding is deliberate: stopping early would +/// leave the writer blocked on a full pipe, which is the deadlock this exists +/// to avoid. +fn drain_capped(mut reader: R, cap: usize) -> Vec { + let mut kept = Vec::new(); + let mut buf = [0u8; 8192]; + loop { + match reader.read(&mut buf) { + Ok(0) => break, + Ok(n) => { + if kept.len() < cap { + let take = n.min(cap - kept.len()); + kept.extend_from_slice(&buf[..take]); + } + } + Err(ref e) if e.kind() == std::io::ErrorKind::Interrupted => {} + Err(_) => break, + } + } + kept +} + +/// Kill the child and everything it spawned. +#[cfg(unix)] +fn kill_process_tree(child: &mut std::process::Child) { + // The child leads its own group (see `process_group(0)`), so its pid is + // the pgid and a negative pid signals the whole group. + let pid = child.id() as i32; + unsafe { + libc::kill(-pid, libc::SIGKILL); + } + let _ = child.kill(); +} + +/// Kill the child. Without process groups only the direct child is reachable. +#[cfg(not(unix))] +fn kill_process_tree(child: &mut std::process::Child) { + let _ = child.kill(); +} + +/// Cap `s` at `cap` bytes, cutting on a character boundary. +/// +/// `String::truncate` panics when the cut lands inside a multi-byte character, +/// which is reachable whenever a command's output happens to straddle the cap. +fn truncate_output(mut s: String, cap: usize) -> String { + if s.len() <= cap { + return s; + } + let mut end = cap; + while end > 0 && !s.is_char_boundary(end) { + end -= 1; + } + s.truncate(end); + s.push_str(&format!("\n... [output truncated at {} bytes]", cap)); + s +} + /// Confirm a write actually landed, by reading the file back (ADR-273 §3.1). /// /// `std::fs::write` returning `Ok` means the syscalls succeeded, not that the @@ -426,20 +586,30 @@ fn verify_written(target: &Path, expected: &str) -> Result<(), String> { } /// Recursively collect files matching a name suffix (simple glob substitute). +/// +/// Symlinks are skipped rather than followed. `Path::is_dir` follows them, and +/// a single `ln -s . loop` inside the tree then recurses until the stack +/// overflows — an unauthenticated crash from an ordinary directory listing. fn collect_glob_matches(dir: &Path, suffix: &str, results: &mut Vec) { let entries = match std::fs::read_dir(dir) { Ok(e) => e, Err(_) => return, }; for entry in entries.flatten() { + // `DirEntry::file_type` does not follow symlinks, so a link is neither + // a file nor a directory here and falls through untouched. + let file_type = match entry.file_type() { + Ok(t) => t, + Err(_) => continue, + }; let path = entry.path(); let name = path .file_name() .map(|n| n.to_string_lossy().to_string()) .unwrap_or_default(); - if path.is_file() && name.ends_with(suffix) { + if file_type.is_file() && name.ends_with(suffix) { results.push(path.to_string_lossy().into_owned()); - } else if path.is_dir() && !name.starts_with('.') { + } else if file_type.is_dir() && !name.starts_with('.') { collect_glob_matches(&path, suffix, results); } } @@ -463,15 +633,22 @@ fn grep_file(path: &Path, pattern: &str, matches: &mut Vec) -> Result Ok(()) } -/// Recursively grep a directory, skipping hidden directories. +/// Recursively grep a directory, skipping hidden directories and symlinks. +/// +/// Symlinks are skipped for the same reason as in `collect_glob_matches`: a +/// self-referencing link makes the descent unbounded. fn grep_dir(dir: &Path, pattern: &str, matches: &mut Vec) -> Result<(), String> { let entries = std::fs::read_dir(dir).map_err(|e| format!("read_dir: {}", e))?; for entry in entries { let entry = entry.map_err(|e| format!("entry: {}", e))?; + let file_type = match entry.file_type() { + Ok(t) => t, + Err(_) => continue, + }; let path = entry.path(); - if path.is_file() { + if file_type.is_file() { grep_file(&path, pattern, matches)?; - } else if path.is_dir() { + } else if file_type.is_dir() { let name = path .file_name() .map(|n| n.to_string_lossy().to_string()) @@ -549,6 +726,209 @@ mod tests { assert!(err.contains("outside the workspace root"), "got: {err}"); } + #[cfg(unix)] + #[test] + fn write_to_a_dangling_symlink_is_rejected() { + let (dir, backend) = backend(); + let outside = tempfile::tempdir().unwrap(); + let escaped = outside.path().join("escaped.txt"); + // The link target does not exist, so `canonicalize` fails and the path + // looks like a fresh file under the root — but `fs::write` would + // follow the link and create the file out there. + std::os::unix::fs::symlink(&escaped, dir.path().join("link")).unwrap(); + + let result = backend.write("link", "pwned"); + assert!( + result.error.is_some(), + "write through a broken link must fail" + ); + assert!( + !escaped.exists(), + "write must not have escaped to {}", + escaped.display() + ); + } + + #[cfg(unix)] + #[test] + fn write_through_a_dangling_symlink_directory_is_rejected() { + let (dir, backend) = backend(); + let outside = tempfile::tempdir().unwrap(); + let escaped_dir = outside.path().join("nope"); + std::os::unix::fs::symlink(&escaped_dir, dir.path().join("linkdir")).unwrap(); + + let result = backend.write("linkdir/f.txt", "pwned"); + assert!( + result.error.is_some(), + "write through a broken link must fail" + ); + assert!( + !escaped_dir.exists(), + "write must not have created the target" + ); + } + + #[cfg(unix)] + #[test] + fn write_through_a_symlink_to_an_inside_file_is_rejected() { + let (dir, backend) = backend(); + std::fs::write(dir.path().join("real.txt"), "original").unwrap(); + std::os::unix::fs::symlink(dir.path().join("real.txt"), dir.path().join("alias")).unwrap(); + + let result = backend.write("alias", "replaced"); + assert!( + result + .error + .as_deref() + .is_some_and(|e| e.contains("symlink")), + "got: {:?}", + result.error + ); + assert_eq!( + std::fs::read_to_string(dir.path().join("real.txt")).unwrap(), + "original" + ); + } + + #[cfg(unix)] + #[test] + fn write_or_edit_through_a_hard_link_is_rejected() { + let (dir, backend) = backend(); + let outside = tempfile::tempdir().unwrap(); + let outside_file = outside.path().join("secret.txt"); + std::fs::write(&outside_file, "original").unwrap(); + // A hard link has no target to canonicalize and looks like an ordinary + // file to `symlink_metadata`, so only the link count gives it away. + // Git cannot store one, but a prepared workspace can plant one. + std::fs::hard_link(&outside_file, dir.path().join("alias.txt")).unwrap(); + + let written = backend.write("alias.txt", "pwned"); + assert!( + written + .error + .as_deref() + .is_some_and(|e| e.contains("hard link")), + "got: {:?}", + written.error + ); + + let edited = backend.edit("alias.txt", "original", "pwned", false); + assert!( + edited + .error + .as_deref() + .is_some_and(|e| e.contains("hard link")), + "got: {:?}", + edited.error + ); + + assert_eq!( + std::fs::read_to_string(&outside_file).unwrap(), + "original", + "the file outside the root must be untouched" + ); + } + + #[cfg(unix)] + #[test] + fn an_ordinary_file_is_not_mistaken_for_a_hard_link() { + // The nlink check must not fire on the normal case, or every edit in + // the workspace breaks. + let (dir, backend) = backend(); + std::fs::write(dir.path().join("plain.txt"), "alpha").unwrap(); + let edited = backend.edit("plain.txt", "alpha", "beta", false); + assert!(edited.error.is_none(), "unexpected: {:?}", edited.error); + assert!(backend.write("fresh.txt", "new").error.is_none()); + } + + #[cfg(unix)] + #[test] + fn traversal_does_not_follow_a_self_referencing_symlink() { + let (dir, backend) = backend(); + std::fs::write(dir.path().join("a.rs"), "needle").unwrap(); + // Without a symlink check this recurses until the stack overflows. + std::os::unix::fs::symlink(dir.path(), dir.path().join("loop")).unwrap(); + + assert_eq!(backend.glob_info("*.rs", ".").unwrap().len(), 1); + assert_eq!( + backend.grep_raw("needle", Some("."), None).unwrap().len(), + 1 + ); + } + + #[test] + fn read_with_an_enormous_limit_does_not_overflow() { + let (dir, backend) = backend(); + std::fs::write(dir.path().join("big.txt"), "one\ntwo\nthree").unwrap(); + // A model can send any integer; `offset + limit` used to wrap and panic. + assert_eq!( + backend.read("big.txt", 0, u64::MAX as usize).unwrap(), + "one\ntwo\nthree" + ); + assert_eq!( + backend.read("big.txt", 1, usize::MAX).unwrap(), + "two\nthree" + ); + // Past EOF is an empty read, not a panic. + assert_eq!(backend.read("big.txt", 99, usize::MAX).unwrap(), ""); + } + + #[test] + fn output_truncation_cuts_on_a_char_boundary() { + // 'é' is two bytes, so a cap of 5 lands inside the third one — the + // case where `String::truncate` panics. + let out = truncate_output("ééé".to_string(), 5); + assert!(out.starts_with("éé"), "got: {out}"); + assert!(out.contains("[output truncated"), "got: {out}"); + let untouched = truncate_output("ééé".to_string(), 64); + assert_eq!(untouched, "ééé"); + } + + #[cfg(unix)] + #[test] + fn execute_captures_oversized_output_without_hanging() { + use std::time::{Duration, Instant}; + let (_dir, backend) = backend(); + let started = Instant::now(); + // 2 MB — far past a pipe buffer, so an undrained child would block + // until the deadline and be misreported as a timeout. + let result = backend + .execute("head -c 2000000 /dev/zero | tr '\\0' 'a'", 60) + .unwrap(); + assert_eq!(result.exit_code, 0, "got: {}", result.output); + assert!( + result.output.contains("[output truncated"), + "expected truncation, got {} bytes", + result.output.len() + ); + assert!( + started.elapsed() < Duration::from_secs(30), + "took {:?} — the pipe was not drained", + started.elapsed() + ); + } + + #[cfg(unix)] + #[test] + fn execute_timeout_kills_the_whole_process_tree() { + use std::time::{Duration, Instant}; + let (_dir, backend) = backend(); + let started = Instant::now(); + // `sh` waits on a grandchild; killing only `sh` would leave it running. + let result = backend.execute("sleep 30 & wait", 1).unwrap(); + assert_eq!(result.exit_code, -1); + assert!( + result.output.contains("timed out"), + "got: {}", + result.output + ); + assert!( + started.elapsed() < Duration::from_secs(15), + "timeout path hung for {:?}", + started.elapsed() + ); + } + #[test] fn write_then_read_roundtrip_inside_root() { let (_dir, backend) = backend();