Skip to main content

switchyard_libsy/algorithms/util/
tool_signals.rs

1// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
2// SPDX-License-Identifier: Apache-2.0
3
4//! Tool-result context signals extracted from the conversation history.
5//!
6//! The extractor walks normalized messages, finds tool calls and results,
7//! reads explicit failure flags, matches text against an error table, and aggregates
8//! conversation-history metrics used by [`crate::StageRouter`] and the
9//! advisor gate's request-side guards.
10//!
11//! All logic is pure and deterministic — no I/O, no shared state.
12
13#![allow(dead_code)]
14
15use std::collections::HashSet;
16use std::path::Path;
17
18use async_trait::async_trait;
19use serde::Deserialize;
20use serde_json::Value;
21use switchyard_protocol::codex_namespaces::{split_qualified_name, tool_namespaces};
22use switchyard_protocol::{ContentBlock, Request, Role, WireFormat};
23
24use crate::{LibsyError, Result};
25
26use crate::core::processor::{Event, Processor};
27use crate::core::state::State;
28
29// ─── severity constants ───────────────────────────────────────────────────────
30
31const SOFT: f32 = 0.3;
32const HARD: f32 = 0.7;
33const CRITICAL: f32 = 1.0;
34
35// ─── pattern table ────────────────────────────────────────────────────────────
36
37/// (name, severity, lower-cased substrings — any hit fires the pattern)
38static ERROR_PATTERNS: &[(&str, f32, &[&str])] = &[
39    (
40        "oom",
41        CRITICAL,
42        &["out of memory", "memoryerror", "cannot allocate memory"],
43    ),
44    (
45        "connection_refused",
46        HARD,
47        &[
48            "connection refused",
49            "connectionrefusederror",
50            "econnrefused",
51        ],
52    ),
53    ("traceback", HARD, &["traceback (most recent call last)"]),
54    (
55        "import_error",
56        HARD,
57        &["modulenotfounderror:", "importerror:", "no module named "],
58    ),
59    (
60        "cmd_not_found",
61        HARD,
62        &["command not found", "not found\n", "/usr/bin/env: "],
63    ),
64    ("assertion", HARD, &["assertionerror"]),
65    ("value_error", HARD, &["valueerror:"]),
66    ("syntax_error", HARD, &["syntaxerror:"]),
67    (
68        "timeout",
69        HARD,
70        &[
71            "timed out",
72            "timeouterror",
73            "timeout expired",
74            "deadline exceeded",
75        ],
76    ),
77    (
78        "no_such_file",
79        HARD,
80        &[
81            "filenotfounderror:",
82            "no such file or directory",
83            // Claude Code's missing-file message. Reads are exempt (`is_missing_file`).
84            // Anchored as "file does not exist" (not a bare "does not exist", which
85            // fires on `ls` output and prose) — trace-mined across 1006 local
86            // trajectories at 22 true / 2 false positives.
87            "file does not exist",
88        ],
89    ),
90    // SOFT: plain non-zero exit without a recognisable exception traceback.
91    ("exit_nonzero", SOFT, &["returned non-zero"]),
92];
93
94static NONZERO_EXIT_PHRASES: &[&str] = &[
95    "exit code",
96    "exit status",
97    "exited with code",
98    "exited with status",
99];
100
101static EDIT_TOOL_NAMES: &[&str] = &[
102    "edit",
103    "multiedit",
104    "notebookedit",
105    "str_replace",
106    "str_replace_based_edit_tool",
107    "apply_patch", // codex's edit tool
108    "text_editor",
109    "patch", // hermes's str_replace-style edit tool
110];
111
112/// Editor tools whose `command` argument picks the action. `view` only reads.
113static EDITOR_TOOL_NAMES: &[&str] = &["str_replace_based_edit_tool", "text_editor"];
114
115static WRITE_TOOL_NAMES: &[&str] = &["write", "create_file", "new_file", "write_file"];
116
117// Bash subcommand patterns. Lowercased; callers must lowercase the command
118// before matching. Bucketed into write_count / edit_count alongside the
119// dedicated `Write` / `Edit` tools.
120static BASH_WRITE_PATTERNS: &[&str] = &[
121    "cat >",
122    "cat >>",
123    "echo >",
124    "echo >>",
125    "tee ",
126    "printf >",
127    "printf >>",
128    "> /",
129    ">> /",
130    "<< 'eof'",
131    "<<eof",
132    "<<'eof'",
133    "<< eof",
134];
135
136/// Python file-write expressions, which only indicate a write when an
137/// interpreter is running them rather than a search looking for them.
138static PYTHON_WRITE_PATTERNS: &[&str] = &["write_text(", "writelines(", ".write("];
139
140static JAVASCRIPT_WRITE_PATTERNS: &[&str] = &[
141    "writefilesync(",
142    "writefile(",
143    "appendfilesync(",
144    "appendfile(",
145];
146
147static BASH_EDIT_PATTERNS: &[&str] = &[
148    "sed -i",
149    "sed --in-place",
150    "awk -i inplace",
151    "awk 'inplace=1'",
152    "patch ",
153    "patch -p",
154    "perl -i",
155    "perl -p -i",
156    "perl -pi",
157];
158
159// Read-like Bash inspections. Match only when none of the write/edit patterns
160// fire (redirection / in-place edit trumps the read intent of the command).
161static BASH_READ_PATTERNS: &[&str] = &[
162    "cat /", "cat ./", "cat ../", "grep ", "ls ", "ls -", "find ", "head ", "tail ", "wc ",
163    "diff ", "which ", "ps ", "df ", "du ", "stat ", "file ", "less ", "more ",
164];
165
166/// Read-only shell programs seen in Codex trajectories. Matching is limited to
167/// command-segment starts so prose and arguments do not masquerade as actions.
168static BASH_READ_COMMANDS: &[&str] = &[
169    "cat", "rg", "nl", "jq", "pwd", "tree", "sed", "grep", "ls", "find", "head", "tail", "wc",
170    "diff", "which", "ps", "df", "du", "stat", "file", "less", "more", "readlink", "realpath",
171    "basename", "dirname", "printenv",
172];
173
174static GIT_READ_SUBCOMMANDS: &[&str] = &[
175    "status",
176    "diff",
177    "log",
178    "show",
179    "show-ref",
180    "rev-parse",
181    "ls-files",
182    "ls-remote",
183    "ls-tree",
184    "grep",
185    "blame",
186    "merge-base",
187    "check-ignore",
188    "tag",
189];
190
191static READ_TOOL_NAMES: &[&str] = &[
192    "read",
193    "view",
194    "read_file",
195    "search_files",
196    "glob",
197    "grep",
198    "find",
199    "ls",
200];
201
202// Planning / scratchpad tool calls — investigative (non-producing) activity.
203// `update_plan` is codex's equivalent of `todowrite`.
204static PLAN_TOOL_NAMES: &[&str] = &[
205    "todowrite",
206    "todo_write",
207    "todo",
208    "update_plan",
209    "todo_list",
210];
211
212// Tool names that route through Bash-command pattern matching. `bash` is
213// claude-code's name; `shell_command` is codex's; `shell` / `local_shell_call`
214// are seen on some OpenAI-derived harnesses; `terminal` is hermes's (it carries
215// a `command` arg like the others, so its intent comes from the pattern match).
216static BASH_TOOL_NAMES: &[&str] = &[
217    "bash",
218    "shell_command",
219    "shell",
220    "local_shell_call",
221    "terminal",
222    "exec_command", // codex
223    "exec",         // openclaw
224    "powershell",   // pi on Windows
225];
226
227// Prefer false negatives: tests_passed clears a capable hold, so a false positive
228// could hand an unfinished task back too early.
229static TEST_PASS_PHRASES: &[&str] = &[
230    " passed",
231    "passed in",
232    "tests passed",
233    "all tests passed",
234    "test ok",
235    "test result: ok",
236    "passed.\n",
237    "tests pass",
238    "\nok ", // go test; newline-anchored to avoid "...lookup..." mid-text
239    "✓ ",
240];
241
242// Literal failure phrases that cannot appear inside a clean run. Substring
243// matched as-is. Patterns that pair with a count (e.g. "failed", "errors")
244// are handled separately by `has_nonzero_failure_count` so "0 failed" /
245// "0 errors" do not trigger a false negative.
246static TEST_FAILURE_LITERAL: &[&str] = &["✗ ", "fatal:", "assertionerror", "error:"];
247
248// Count-prefixed failure keywords. Trip only when a nonzero integer precedes
249// the keyword (modulo whitespace), so cargo's "0 failed" and go's
250// "0 errors" summaries on a clean run are not misread as failures.
251static NUMERIC_FAILURE_KEYWORDS: &[&str] = &["failed", "failure", "failures", "errors", "error"];
252
253/// Default sliding-window size for `recent_*` counts and windowed severity.
254///
255/// A short horizon captures "what is the agent doing right now" while keeping
256/// signals sticky — an error or stall persists a few recovery turns instead of
257/// flickering off the moment one clean result lands. Override per request by
258/// passing a window to [`ToolSignals::from_request`].
259pub const DEFAULT_RECENT_WINDOW: usize = 3;
260
261/// Exact tool-name semantics added to the built-in vocabulary.
262///
263/// Matching is ASCII case-insensitive. An MCP or Codex namespaced tool also
264/// matches by its bare tool name. These lists are additive: built-in tool names
265/// cannot be reclassified.
266#[derive(Clone, Debug, Default, Deserialize, PartialEq, Eq)]
267#[serde(default, deny_unknown_fields)]
268pub struct ToolSemantics {
269    /// Read-only lookup or inspection tools.
270    pub observe: Vec<String>,
271    /// Tools that change task or external state.
272    pub mutate: Vec<String>,
273    /// Explicit planning or task-decomposition tools.
274    pub plan: Vec<String>,
275    /// Tools that demonstrate new forward activity without favoring either tier.
276    pub new: Vec<String>,
277}
278
279impl ToolSemantics {
280    /// Rejects ambiguous mappings and attempts to reclassify built-in tools.
281    pub fn validate(&self) -> Result<()> {
282        let mut seen: Vec<(String, &'static str)> = Vec::new();
283        for (category, names) in [
284            ("observe", &self.observe),
285            ("mutate", &self.mutate),
286            ("plan", &self.plan),
287            ("new", &self.new),
288        ] {
289            for name in names {
290                if name.trim().is_empty() {
291                    return Err(tool_semantics_error(format!(
292                        "tool_semantics.{category} contains an empty tool name"
293                    )));
294                }
295                let normalized = name.to_ascii_lowercase();
296                if is_builtin_tool_name(&name.to_lowercase()) {
297                    return Err(tool_semantics_error(format!(
298                        "tool {name:?} already has built-in semantics and cannot be reclassified"
299                    )));
300                }
301                if let Some((_, previous)) = seen.iter().find(|(seen, _)| seen == &normalized) {
302                    return Err(tool_semantics_error(format!(
303                        "tool {name:?} appears in both tool_semantics.{previous} and tool_semantics.{category}"
304                    )));
305                }
306                seen.push((normalized, category));
307            }
308        }
309        Ok(())
310    }
311
312    fn classify(&self, name: &str) -> Option<ToolSemantic> {
313        if contains_name(&self.observe, name) {
314            Some(ToolSemantic::Observe)
315        } else if contains_name(&self.mutate, name) {
316            // The stage scorer treats writes and edits identically. Custom
317            // mutations use the write counter to preserve the public signal shape.
318            Some(ToolSemantic::Mutate(MutationKind::Write))
319        } else if contains_name(&self.plan, name) {
320            Some(ToolSemantic::Plan)
321        } else if contains_name(&self.new, name) {
322            Some(ToolSemantic::New)
323        } else {
324            None
325        }
326    }
327}
328
329fn contains_name(names: &[String], candidate: &str) -> bool {
330    names
331        .iter()
332        .any(|name| name.eq_ignore_ascii_case(candidate))
333}
334
335fn tool_semantics_error(message: String) -> LibsyError {
336    LibsyError::AlgorithmError { message }
337}
338
339// ─── output type ─────────────────────────────────────────────────────────────
340
341/// Tool-execution signals extracted from a normalized [`Request`].
342///
343/// A request-side processor stores these signals in [`State`](crate::State) for
344/// [`crate::StageRouter`] and its classifier to consume. The advisor gate's
345/// request-side guards read the conversation-shape counts directly via
346/// [`ToolSignals::from_request`].
347#[derive(Clone, Debug, Default)]
348pub struct ToolSignals {
349    /// Max severity across the recent window (last `recent_window` tool results):
350    /// `0.0` clean · `0.3` soft (exit_nonzero) · `0.7` hard · `1.0` critical.
351    /// Windowed so an error persists through the recovery turns instead of clearing
352    /// the instant the next result is clean.
353    pub severity: f32,
354    /// The same hard-or-critical failure appeared at least twice in the recent
355    /// tool-result window.
356    pub repeated_failure: bool,
357    /// Consecutive clean tool results back from the most recent. `0` if the last failed.
358    pub no_error_streak: u32,
359    /// Total edit-style tool calls in the request.
360    pub edit_count: u32,
361    /// Total write-style tool calls in the request.
362    pub write_count: u32,
363    /// Read-type calls (Read tool + read-like Bash). Used by the build-pit gate.
364    pub read_count: u32,
365    /// TodoWrite / planning tool calls. Investigative (non-producing) activity —
366    /// recent todowrites distinguish `exploring` from `spinning` in the scorer.
367    pub todowrite_count: u32,
368    /// Edit-type calls within the configured recent window (default: [`DEFAULT_RECENT_WINDOW`]).
369    pub recent_edit_count: u32,
370    /// Write-type calls within the configured recent window (default: [`DEFAULT_RECENT_WINDOW`]).
371    pub recent_write_count: u32,
372    /// Read-type calls within the configured recent window (default: [`DEFAULT_RECENT_WINDOW`]).
373    pub recent_read_count: u32,
374    /// TodoWrite calls within the configured recent window (default: [`DEFAULT_RECENT_WINDOW`]).
375    pub recent_todowrite_count: u32,
376    /// Configured `new` tool calls across the full request history.
377    pub new_count: u32,
378    /// Configured `new` tool calls within the recent window.
379    pub recent_new_count: u32,
380    /// Consecutive trailing tool calls in the `Unknown` category (no Write/Edit/Read/
381    /// Plan match). Surfaced in the classifier state summary; not scored directly.
382    pub pure_bash_streak: u32,
383    /// A tool result after the latest recent failure matched a test-pass pattern.
384    pub tests_passed: bool,
385    /// Total `ToolResult` blocks, counted per block (a message batching N
386    /// results contributes N) and including empty-content results.
387    pub tool_result_count: u32,
388    /// Messages with `Role::Assistant`, unlike [`ToolSignals::turn_depth`],
389    /// which counts every message regardless of role.
390    pub assistant_turn_count: u32,
391    /// Message-count proxy for turn depth. Wire-format dependent (Anthropic batches
392    /// tool results into fewer messages than OpenAI-chat), so gates keyed on it are
393    /// approximate across request origins.
394    pub turn_depth: u32,
395    /// The request carries a context-compaction summary (the agent's context was
396    /// summarised after overflowing). Compaction resets the router's accumulated
397    /// signals, so a task that was on the strong tier de-escalates back to weak — the
398    /// picker uses this to force + hold the strong tier. Self-latching: the summary
399    /// stays in the context prefix on every subsequent turn.
400    pub compacted: bool,
401}
402
403impl ToolSignals {
404    /// Extracts tool and activity signals from `request`.
405    ///
406    /// `window_size` limits recent counters to the newest tool results. `None`
407    /// uses [`DEFAULT_RECENT_WINDOW`].
408    pub fn from_request(request: &Request, window_size: Option<usize>) -> Self {
409        Self::from_request_with_semantics(request, window_size, &ToolSemantics::default())
410    }
411
412    /// Extracts signals using the built-in vocabulary plus additive semantics.
413    pub fn from_request_with_semantics(
414        request: &Request,
415        window_size: Option<usize>,
416        semantics: &ToolSemantics,
417    ) -> Self {
418        extract_tool_signals_with_window_and_semantics(
419            request,
420            window_size.unwrap_or(DEFAULT_RECENT_WINDOW),
421            semantics,
422        )
423    }
424}
425
426// `command` is the lowercased Bash command line; None for non-Bash tools.
427// `bare_name` is the tool's own name when `name` joins it to a namespace or MCP server.
428// `is_retrieval` marks a call that only reads; it counts as an observation.
429#[derive(Debug, Clone)]
430struct ObservedToolCall<'a> {
431    name: String,
432    bare_name: Option<&'a str>,
433    command: Option<String>,
434    is_retrieval: bool,
435}
436
437#[derive(Debug, Clone, Copy, PartialEq, Eq)]
438enum MutationKind {
439    Write,
440    Edit,
441}
442
443/// Domain-neutral meaning assigned to an observed tool call.
444#[derive(Debug, Clone, Copy, PartialEq, Eq)]
445enum ToolSemantic {
446    Mutate(MutationKind),
447    Observe,
448    Plan,
449    New,
450    Unknown,
451}
452
453/// Request-side processor that extracts tool-result signals from each request
454/// and stores them on the request `State` for downstream routing.
455#[derive(Debug, Clone)]
456pub struct ToolSignalProcessor {
457    /// Number of trailing tool results the `recent_*` counts and windowed
458    /// severity are computed over.
459    pub recent_window: usize,
460    /// Route-scoped additions to the built-in tool vocabulary.
461    pub tool_semantics: ToolSemantics,
462}
463
464impl Default for ToolSignalProcessor {
465    fn default() -> Self {
466        Self {
467            recent_window: DEFAULT_RECENT_WINDOW,
468            tool_semantics: ToolSemantics::default(),
469        }
470    }
471}
472
473#[async_trait]
474impl Processor<State> for ToolSignalProcessor {
475    async fn process(&self, state: &mut State, event: Event<'_>) -> Result<()> {
476        if let Event::Request { request: req, .. } = event {
477            let tool_signal = ToolSignals::from_request_with_semantics(
478                req,
479                Some(self.recent_window),
480                &self.tool_semantics,
481            );
482            state.tool_signals = Some(tool_signal);
483        }
484        Ok(())
485    }
486}
487
488fn classify_tool_call(name: &str, command: Option<&str>) -> ToolSemantic {
489    classify_tool_call_with_semantics(name, command, &ToolSemantics::default())
490}
491
492fn classify_tool_call_with_semantics(
493    name: &str,
494    command: Option<&str>,
495    semantics: &ToolSemantics,
496) -> ToolSemantic {
497    // Built-in names and Bash command inference take precedence over route-scoped mappings.
498    let lower = name.to_lowercase();
499    if WRITE_TOOL_NAMES.contains(&lower.as_str()) {
500        return ToolSemantic::Mutate(MutationKind::Write);
501    }
502    if EDITOR_TOOL_NAMES.contains(&lower.as_str()) && command == Some("view") {
503        return ToolSemantic::Observe;
504    }
505    if EDIT_TOOL_NAMES.contains(&lower.as_str()) {
506        return ToolSemantic::Mutate(MutationKind::Edit);
507    }
508    if READ_TOOL_NAMES.contains(&lower.as_str()) {
509        return ToolSemantic::Observe;
510    }
511    if PLAN_TOOL_NAMES.contains(&lower.as_str()) {
512        return ToolSemantic::Plan;
513    }
514    if BASH_TOOL_NAMES.contains(&lower.as_str())
515        && let Some(cmd) = command
516    {
517        // Write/edit redirection trumps read-like operands.
518        if BASH_WRITE_PATTERNS.iter().any(|p| cmd.contains(p)) || shell_command_is_write(cmd) {
519            return ToolSemantic::Mutate(MutationKind::Write);
520        }
521        if cmd.contains("python") && PYTHON_WRITE_PATTERNS.iter().any(|p| cmd.contains(p)) {
522            return ToolSemantic::Mutate(MutationKind::Write);
523        }
524        if shell_invokes_program(cmd, "node")
525            && JAVASCRIPT_WRITE_PATTERNS
526                .iter()
527                .any(|pattern| cmd.contains(pattern))
528        {
529            return ToolSemantic::Mutate(MutationKind::Write);
530        }
531        if BASH_EDIT_PATTERNS.iter().any(|p| cmd.contains(p)) || shell_command_is_edit(cmd) {
532            return ToolSemantic::Mutate(MutationKind::Edit);
533        }
534        if BASH_READ_PATTERNS.iter().any(|p| cmd.contains(p)) || shell_command_is_read(cmd) {
535            return ToolSemantic::Observe;
536        }
537    }
538    semantics.classify(name).unwrap_or(ToolSemantic::Unknown)
539}
540
541/// Tools whose successful output is retrieved content or inspection data.
542fn is_retrieval_tool(name: &str, command: Option<&Value>) -> bool {
543    let lower = name.to_lowercase();
544    READ_TOOL_NAMES.contains(&lower.as_str())
545        || (EDITOR_TOOL_NAMES.contains(&lower.as_str())
546            && command
547                .and_then(Value::as_str)
548                .is_some_and(|command| command.eq_ignore_ascii_case("view")))
549        // PowerShell quoting and escapes do not follow the POSIX rules below.
550        || (lower != "powershell"
551            && BASH_TOOL_NAMES.contains(&lower.as_str())
552            && command.is_some_and(command_is_retrieval))
553}
554
555/// A shell command field given as one line or as argv words.
556fn command_is_retrieval(command: &Value) -> bool {
557    match command {
558        Value::String(line) => shell_is_retrieval(line, 0),
559        Value::Array(argv) => {
560            // Charge each word a separator byte, as in the one-line form, so the
561            // budget also bounds the argument count.
562            argv.iter()
563                .try_fold(0, |size, word| {
564                    let size = size + word.as_str()?.len() + 1;
565                    (size <= MAX_RETRIEVAL_COMMAND_BYTES).then_some(size)
566                })
567                .is_some()
568                && retrieval_command(
569                    &argv.iter().filter_map(Value::as_str).collect::<Vec<_>>(),
570                    0,
571                )
572        }
573        _ => false,
574    }
575}
576
577// Keep shell parsing bounded on the request thread. Larger commands retain
578// ordinary error scanning, just like unrecognized shell syntax.
579const MAX_RETRIEVAL_COMMAND_BYTES: usize = 16 * 1024;
580
581/// Only suppress output when every shell segment is a recognized inspection.
582/// Observation counters can count mixed commands; suppressing errors needs all
583/// commands to qualify. Substitutions and redirects keep their error signals.
584fn shell_is_retrieval(command: &str, depth: usize) -> bool {
585    if command.len() > MAX_RETRIEVAL_COMMAND_BYTES {
586        return false;
587    }
588    if quoted_chars(command).any(|(_, c, quote)| {
589        (quote != Some('\'') && matches!(c, '$' | '`'))
590            || (quote.is_none() && matches!(c, '<' | '>' | '(' | ')' | '{' | '}' | '#'))
591    }) {
592        return false;
593    }
594    let mut segments = shell_segments(command).peekable();
595    segments.peek().is_some()
596        && segments.all(|segment| {
597            // `split` also rejects an unclosed quote or a trailing backslash.
598            shlex::split(segment).is_some_and(|words| {
599                retrieval_command(&words.iter().map(String::as_str).collect::<Vec<_>>(), depth)
600            })
601        })
602}
603
604fn skip_shell_options<'a>(mut args: &'a [&'a str], takes_value: &[&str]) -> &'a [&'a str] {
605    while let Some((option, rest)) = args.split_first() {
606        if !option.starts_with('-') || *option == "-" {
607            break;
608        }
609        args = rest;
610        if *option == "--" {
611            break;
612        }
613        if takes_value.contains(option) {
614            args = args.get(1..).unwrap_or_default();
615        }
616    }
617    args
618}
619
620/// A short-option word such as `-Hx` that sets any of `flags`.
621fn has_short_flag(word: &str, flags: &[char]) -> bool {
622    word.starts_with('-') && !word.starts_with("--") && word.contains(flags)
623}
624
625// Bounds recursion through nested wrappers and `sh -c` scripts.
626const MAX_COMMAND_DEPTH: usize = 20;
627
628fn retrieval_command(mut words: &[&str], depth: usize) -> bool {
629    if depth > MAX_COMMAND_DEPTH {
630        return false;
631    }
632    while words.first().is_some_and(|word| {
633        word.split_once('=').is_some_and(|(name, _)| {
634            !name.is_empty()
635                && name.bytes().enumerate().all(|(i, c)| {
636                    c == b'_' || c.is_ascii_alphabetic() || (i > 0 && c.is_ascii_digit())
637                })
638        })
639    }) {
640        words = &words[1..];
641    }
642    let Some((&program, args)) = words.split_first() else {
643        return false;
644    };
645    // The standard bin dirs hold the real programs. Any other path, such as
646    // `./test`, names a local script.
647    let program = [
648        "/bin/",
649        "/sbin/",
650        "/usr/bin/",
651        "/usr/sbin/",
652        "/usr/local/bin/",
653        "/opt/homebrew/bin/",
654    ]
655    .iter()
656    .find_map(|dir| program.strip_prefix(dir))
657    .unwrap_or(program);
658    // Unwrap launchers before inspecting the actual command and its flags.
659    let wrapper_options: Option<&[&str]> = match program {
660        "sudo" => Some(&[
661            "-u", "-g", "-h", "-p", "-C", "-T", "--user", "--group", "--host",
662        ]),
663        "env" => Some(&["-u", "--unset", "-C", "--chdir"]),
664        "command" | "nohup" => Some(&[]),
665        "exec" => Some(&["-a"]),
666        "time" => Some(&["-f", "--format", "-o", "--output"]),
667        "timeout" => Some(&["-s", "--signal", "-k", "--kill-after"]),
668        "xargs" => Some(&[
669            "-n",
670            "--max-args",
671            "-P",
672            "--max-procs",
673            "-I",
674            "--replace",
675            "-d",
676            "--delimiter",
677            "-L",
678            "--max-lines",
679            "-s",
680            "--max-chars",
681        ]),
682        _ => None,
683    };
684    if let Some(options) = wrapper_options {
685        // `env -S` splits its argument into a new command line.
686        if program == "env"
687            && args
688                .iter()
689                .any(|arg| arg.starts_with("--split-string") || has_short_flag(arg, &['S']))
690        {
691            return false;
692        }
693        let mut rest = skip_shell_options(args, options);
694        if program == "timeout" {
695            rest = rest.get(1..).unwrap_or_default();
696        }
697        return (program == "env" && rest.is_empty()) || retrieval_command(rest, depth + 1);
698    }
699    if matches!(program, "bash" | "sh" | "dash" | "zsh" | "ksh") {
700        return args
701            .first()
702            .is_some_and(|option| has_short_flag(option, &['c']))
703            && args
704                .get(1)
705                .is_some_and(|command| shell_is_retrieval(command, depth + 1));
706    }
707    if program == "git" {
708        return git_is_retrieval(args);
709    }
710    if matches!(args, ["--help" | "--version"]) {
711        return true;
712    }
713    match program {
714        "find" => !args.iter().any(|arg| {
715            matches!(
716                *arg,
717                "-delete"
718                    | "-exec"
719                    | "-execdir"
720                    | "-ok"
721                    | "-okdir"
722                    | "-fprint"
723                    | "-fprint0"
724                    | "-fprintf"
725            )
726        }),
727        // fd joins short flags, so `-Hx` runs a command too.
728        "fd" => !args
729            .iter()
730            .any(|arg| arg.starts_with("--exec") || has_short_flag(arg, &['x', 'X'])),
731        // Only the common line-range printing form; sed scripts can execute commands.
732        "sed" => {
733            args.len() >= 2
734                && args[0] == "-n"
735                && args[2..].iter().all(|arg| !arg.starts_with('-'))
736                && args[1].strip_suffix('p').is_some_and(|range| {
737                    !range.is_empty()
738                        && range
739                            .bytes()
740                            .all(|c| c.is_ascii_digit() || matches!(c, b',' | b'$'))
741                })
742        }
743        "sort" => !args.iter().any(|arg| {
744            arg.starts_with("--compress-program")
745                || arg.starts_with("--output")
746                || arg.starts_with("-o")
747        }),
748        "rg" => !args.iter().any(|arg| arg.starts_with("--pre")),
749        "xxd" => !args.contains(&"-r") && !args.contains(&"-revert"),
750        "go" => {
751            matches!(args.first(), Some(&"list" | &"doc" | &"env" | &"version"))
752                && !args.iter().any(|arg| matches!(*arg, "-w" | "-u"))
753        }
754        "docker" | "podman" => matches!(args.first(), Some(&"ps" | &"version")),
755        "cat" | "grep" | "ls" | "nl" | "head" | "tail" | "wc" | "pwd" | "stat" | "file" | "du"
756        | "df" | "which" | "type" | "diff" | "cmp" | "jq" | "uniq" | "cut" | "readlink"
757        | "realpath" | "tree" | "basename" | "dirname" | "printenv" | "echo" | "printf"
758        | "less" | "more" | "test" | "[" | "ps" | "pgrep" | "pkg-config" | "strings" | "uname"
759        | "od" | "true" | ":" | "cd" | "pstree" | "sha256sum" | "sha1sum" | "md5sum" | "lsof"
760        | "tr" | "free" | "id" | "namei" | "whoami" | "paste" | "ss" | "getent" => true,
761        _ => false,
762    }
763}
764
765fn git_is_retrieval(args: &[&str]) -> bool {
766    let args = skip_shell_options(
767        args,
768        &["-C", "-c", "--git-dir", "--work-tree", "--namespace"],
769    );
770    let Some((subcommand, rest)) = args.split_first() else {
771        return false;
772    };
773    match *subcommand {
774        "diff" => !rest.contains(&"--check"),
775        "branch" => {
776            rest.is_empty()
777                || rest.iter().all(|arg| {
778                    matches!(
779                        *arg,
780                        "-a" | "-r"
781                            | "--all"
782                            | "--list"
783                            | "--show-current"
784                            | "-v"
785                            | "-vv"
786                            | "--verbose"
787                    )
788                })
789        }
790        "remote" => {
791            rest.is_empty()
792                || matches!(rest, ["-v" | "--verbose"])
793                || rest.first() == Some(&"get-url")
794        }
795        "tag" => rest.is_empty() || matches!(rest[0], "-l" | "--list"),
796        "config" => matches!(
797            rest.first(),
798            Some(&"--get" | &"--get-all" | &"--list" | &"-l")
799        ),
800        "worktree" => rest.first() == Some(&"list"),
801        "submodule" => rest.first() == Some(&"status"),
802        "status" | "log" | "show" | "blame" | "ls-files" | "ls-remote" | "rev-parse"
803        | "merge-base" | "grep" | "describe" | "show-ref" | "check-ignore" | "rev-list"
804        | "ls-tree" | "diff-tree" | "version" => true,
805        _ => false,
806    }
807}
808
809fn is_builtin_tool_name(lower: &str) -> bool {
810    WRITE_TOOL_NAMES.contains(&lower)
811        || EDIT_TOOL_NAMES.contains(&lower)
812        || READ_TOOL_NAMES.contains(&lower)
813        || PLAN_TOOL_NAMES.contains(&lower)
814        || BASH_TOOL_NAMES.contains(&lower)
815}
816
817/// Split a shell line at unquoted command separators. This intentionally avoids
818/// pretending to be a full shell parser; only the leading program and flags of
819/// each segment are inspected below.
820fn shell_segments(command: &str) -> impl Iterator<Item = &str> {
821    let mut start = 0;
822    quoted_chars(command)
823        .filter(|&(_, c, quote)| quote.is_none() && matches!(c, '\n' | ';' | '|' | '&'))
824        .map(|(index, _, _)| index)
825        .chain([command.len()])
826        .filter_map(move |end| {
827            let segment = command[start..end].trim();
828            // Separators are one byte.
829            start = end + 1;
830            (!segment.is_empty()).then_some(segment)
831        })
832}
833
834/// Unescaped characters with their byte index and the quote around them.
835fn quoted_chars(command: &str) -> impl Iterator<Item = (usize, char, Option<char>)> {
836    let mut quote = None;
837    let mut escaped = false;
838    command.char_indices().filter_map(move |(index, c)| {
839        if escaped {
840            escaped = false;
841            return None;
842        }
843        if c == '\\' && quote != Some('\'') {
844            escaped = true;
845            return None;
846        }
847        let around = quote;
848        if quote == Some(c) {
849            quote = None;
850        } else if quote.is_none() && matches!(c, '\'' | '"') {
851            quote = Some(c);
852        }
853        Some((index, c, around))
854    })
855}
856
857fn shell_words(segment: &str) -> std::iter::Peekable<std::str::SplitAsciiWhitespace<'_>> {
858    let mut words = segment.split_ascii_whitespace().peekable();
859
860    if words.peek().copied() == Some("env") {
861        words.next();
862        while words.peek().is_some_and(|word| word.starts_with('-')) {
863            words.next();
864        }
865    }
866    while words
867        .peek()
868        .is_some_and(|word| word.contains('=') && !word.starts_with('='))
869    {
870        words.next();
871    }
872
873    words
874}
875
876fn program_name(word: &str) -> &str {
877    Path::new(word)
878        .file_name()
879        .and_then(|name| name.to_str())
880        .unwrap_or(word)
881}
882
883fn shell_invokes_program(command: &str, expected: &str) -> bool {
884    shell_segments(command).any(|segment| {
885        shell_words(segment)
886            .next()
887            .is_some_and(|word| program_name(word) == expected)
888    })
889}
890
891fn shell_command_is_write(command: &str) -> bool {
892    shell_segments(command).any(|segment| {
893        let mut words = shell_words(segment);
894        let Some(program) = words.next().map(program_name) else {
895            return false;
896        };
897        if matches!(program, "cp" | "mkdir" | "touch" | "install") {
898            return true;
899        }
900
901        let redirects_output = words.any(|word| matches!(word, ">" | ">>"));
902        redirects_output
903            && (matches!(program, "echo" | "printf" | "git")
904                || BASH_READ_COMMANDS.contains(&program))
905    })
906}
907
908fn shell_command_is_edit(command: &str) -> bool {
909    shell_segments(command).any(|segment| {
910        let mut words = shell_words(segment);
911        let Some(program) = words.next().map(program_name) else {
912            return false;
913        };
914        let has_arg = |arg: &str| words.clone().any(|word| word == arg);
915
916        match program {
917            "mv" | "rm" => true,
918            "perl" => words
919                .take_while(|word| word.starts_with('-'))
920                .any(|option| {
921                    option
922                        .trim_start_matches('-')
923                        .chars()
924                        .any(|flag| flag == 'i')
925                }),
926            "git" => words
927                .next()
928                .is_some_and(|subcommand| matches!(subcommand, "apply" | "am" | "restore")),
929            "gofmt" => has_arg("-w"),
930            "cargo" => words.clone().next() == Some("fmt") && !has_arg("--check"),
931            "ruff" => {
932                let subcommand = words.clone().next();
933                (subcommand == Some("format") && !has_arg("--check"))
934                    || (subcommand == Some("check") && has_arg("--fix"))
935            }
936            "prettier" => has_arg("--write"),
937            "black" => !has_arg("--check"),
938            _ => {
939                (words.clone().any(|word| program_name(word) == "prettier") && has_arg("--write"))
940                    || (words.clone().any(|word| program_name(word) == "ruff")
941                        && ((has_arg("format") && !has_arg("--check"))
942                            || (has_arg("check") && has_arg("--fix"))))
943            }
944        }
945    })
946}
947
948fn shell_command_is_read(command: &str) -> bool {
949    shell_segments(command).any(|segment| {
950        if segment == "env" {
951            return true;
952        }
953        let mut words = shell_words(segment);
954        let Some(program) = words.next().map(program_name) else {
955            return false;
956        };
957
958        if BASH_READ_COMMANDS.contains(&program) {
959            return true;
960        }
961        if program == "command" && words.next() == Some("-v") {
962            return true;
963        }
964        if program == "type" {
965            return true;
966        }
967        if program != "git" {
968            return false;
969        }
970
971        match words.next() {
972            Some("branch") => words.next().is_none_or(|arg| arg.starts_with('-')),
973            Some("remote") => words
974                .next()
975                .is_none_or(|arg| arg.starts_with('-') || arg == "get-url"),
976            Some("config") => words
977                .next()
978                .is_some_and(|arg| matches!(arg, "--get" | "--get-all" | "--list" | "-l")),
979            Some(subcommand) => GIT_READ_SUBCOMMANDS.contains(&subcommand),
980            None => false,
981        }
982    })
983}
984
985// ─── extraction entry point ───────────────────────────────────────────────────
986
987/// Extract all tool-execution signals from a normalized [`Request`].
988///
989/// Returns [`ToolSignals::default()`] when the message history contains no tool
990/// activity, so callers can always inspect the signal fields.
991fn extract_tool_signals_with_window(request: &Request, recent_window: usize) -> ToolSignals {
992    extract_tool_signals_with_window_and_semantics(
993        request,
994        recent_window,
995        &ToolSemantics::default(),
996    )
997}
998
999fn extract_tool_signals_with_window_and_semantics(
1000    request: &Request,
1001    recent_window: usize,
1002    semantics: &ToolSemantics,
1003) -> ToolSignals {
1004    // Read the decoded conversation, including preserved built-in tool outputs.
1005    let messages = &request.llm_request.messages;
1006    let namespaces = tool_namespaces(&request.llm_request.extensions);
1007    let mut tool_texts: Vec<(String, bool)> = Vec::new();
1008    let mut tool_calls: Vec<ObservedToolCall> = Vec::new();
1009    // IDs whose latest call is a retrieval tool.
1010    let mut retrieval_calls: HashSet<&str> = HashSet::new();
1011    let mut compacted = false;
1012    let mut tool_result_count = 0usize;
1013    let mut assistant_turn_count = 0usize;
1014
1015    for message in messages {
1016        if message.role == Role::Assistant {
1017            assistant_turn_count += 1;
1018        }
1019        for block in &message.content {
1020            match block {
1021                ContentBlock::ToolCall(call) => {
1022                    // Responses namespaced tools arrive as `<namespace>__<tool>`.
1023                    let bare_name = namespaces
1024                        .and_then(|namespaces| split_qualified_name(namespaces, &call.name))
1025                        .map(|(tool, _)| tool)
1026                        .or_else(|| mcp_tool_name(&call.name));
1027                    // The Responses wire format sends arguments as a JSON string.
1028                    let decoded = call
1029                        .arguments
1030                        .as_str()
1031                        .and_then(|raw| serde_json::from_str::<Value>(raw).ok());
1032                    let command_field = command_of(decoded.as_ref().unwrap_or(&call.arguments));
1033                    let command = command_field.and_then(command_text);
1034                    // The joined name wins, as in `build_signal`. A joined name
1035                    // configured as observe still counts when its bare name is a
1036                    // retrieval tool, such as `mcp__files__read`.
1037                    let full = classify_tool_call_with_semantics(
1038                        &call.name,
1039                        command.as_deref(),
1040                        semantics,
1041                    );
1042                    let name = match (full, bare_name) {
1043                        (ToolSemantic::Unknown | ToolSemantic::Observe, Some(bare_name)) => {
1044                            bare_name
1045                        }
1046                        _ => call.name.as_str(),
1047                    };
1048                    let is_retrieval = is_retrieval_tool(name, command_field);
1049                    if !call.id.is_empty() {
1050                        // A reused ID links to its latest call.
1051                        if is_retrieval {
1052                            retrieval_calls.insert(call.id.as_str());
1053                        } else {
1054                            retrieval_calls.remove(call.id.as_str());
1055                        }
1056                    }
1057                    tool_calls.push(ObservedToolCall {
1058                        name: call.name.clone(),
1059                        bare_name,
1060                        command,
1061                        is_retrieval,
1062                    });
1063                }
1064                ContentBlock::ToolResult(result) => {
1065                    // Before the empty-text filter: empty results still count.
1066                    tool_result_count += 1;
1067                    let is_error = result.is_error == Some(true);
1068                    let text = result
1069                        .content
1070                        .iter()
1071                        .filter_map(text_of)
1072                        .collect::<Vec<_>>()
1073                        .join("\n");
1074                    let is_retrieval_result =
1075                        retrieval_calls.contains(result.tool_call_id.as_str());
1076                    if is_retrieval_result {
1077                        // Shell reads can return bare JSON from a file. Hermes wraps
1078                        // terminal output with output and exit_code fields.
1079                        let shell_result = serde_json::Deserializer::from_str(&text)
1080                            .into_iter::<Value>()
1081                            .next()
1082                            .and_then(|result| result.ok())
1083                            .filter(|value| {
1084                                value["output"].is_string() && value.get("exit_code").is_some()
1085                            });
1086                        let shell_failed = shell_result.as_ref().is_some_and(|value| {
1087                            value["exit_code"].as_i64().is_some_and(|code| code != 0)
1088                                || value["success"].as_bool() == Some(false)
1089                                || value["error"]
1090                                    .as_str()
1091                                    .is_some_and(|error| !error.trim().is_empty())
1092                        }) || has_nonzero_exit_status(&text.to_lowercase());
1093                        if is_missing_file(&text) || (!is_error && !shell_failed) {
1094                            // Keep the window slot after dropping retrieved content.
1095                            if !text.is_empty() {
1096                                tool_texts.push((String::new(), false));
1097                            }
1098                            continue;
1099                        }
1100                    }
1101                    // An explicit failure remains a signal even without text.
1102                    if !text.is_empty() || is_error {
1103                        tool_texts.push((text, is_error));
1104                    }
1105                }
1106                // Built-in tool history stays opaque so it can be replayed unchanged.
1107                ContentBlock::Unknown { provider, raw }
1108                    if provider.as_str() == WireFormat::OpenAiResponses.as_str()
1109                        && raw.get("type").and_then(Value::as_str)
1110                            == Some("apply_patch_call_output") =>
1111                {
1112                    tool_result_count += 1;
1113                    let text = raw
1114                        .get("output")
1115                        .and_then(Value::as_str)
1116                        .unwrap_or_default();
1117                    let is_error = raw.get("status").and_then(Value::as_str) == Some("failed");
1118                    tool_texts.push((text.to_owned(), is_error));
1119                }
1120                ContentBlock::Unknown { provider, raw }
1121                    if provider.as_str() == WireFormat::OpenAiResponses.as_str()
1122                        && raw.get("type").and_then(Value::as_str) == Some("shell_call_output") =>
1123                {
1124                    tool_result_count += 1;
1125                    let mut texts = Vec::new();
1126                    let mut is_error = false;
1127                    if let Some(outputs) = raw.get("output").and_then(Value::as_array) {
1128                        for output in outputs {
1129                            for field in ["stdout", "stderr"] {
1130                                if let Some(text) = output.get(field).and_then(Value::as_str)
1131                                    && !text.is_empty()
1132                                {
1133                                    texts.push(text);
1134                                }
1135                            }
1136                            if let Some(outcome) = output.get("outcome") {
1137                                is_error |= match outcome.get("type").and_then(Value::as_str) {
1138                                    Some("timeout") => true,
1139                                    Some("exit")
1140                                        if outcome
1141                                            .get("exit_code")
1142                                            .and_then(Value::as_i64)
1143                                            .is_some_and(|code| code != 0) =>
1144                                    {
1145                                        // Keep plain nonzero exits SOFT, including empty output.
1146                                        texts.push("returned non-zero");
1147                                        output
1148                                            .get("stderr")
1149                                            .and_then(Value::as_str)
1150                                            .is_some_and(|text| !text.trim().is_empty())
1151                                    }
1152                                    _ => false,
1153                                };
1154                            }
1155                        }
1156                    }
1157                    let text = texts.join("\n");
1158                    tool_texts.push((text, is_error));
1159                }
1160                // Compaction is detected anywhere in the conversation: the summary
1161                // stays in the prefix on every later turn, so this self-latches
1162                // once it fires.
1163                ContentBlock::Text { text } => {
1164                    let text = text.to_lowercase();
1165                    compacted |= text.contains(COMPACTION_MARKER)
1166                        || text.lines().any(|line| {
1167                            HERMES_COMPACTION_MARKERS
1168                                .iter()
1169                                .any(|marker| line.trim_start().starts_with(marker))
1170                        });
1171                }
1172                _ => {}
1173            }
1174        }
1175    }
1176
1177    let mut signal = build_signal(
1178        tool_texts,
1179        tool_calls,
1180        messages.len() as u32,
1181        recent_window,
1182        semantics,
1183    );
1184    signal.compacted = compacted;
1185    signal.tool_result_count = u32::try_from(tool_result_count).unwrap_or(u32::MAX);
1186    signal.assistant_turn_count = u32::try_from(assistant_turn_count).unwrap_or(u32::MAX);
1187    signal
1188}
1189
1190/// Distinctive preamble Claude Code injects as a user message when it compacts an
1191/// overflowed context. Matched case-insensitively; normal task text never contains it.
1192const COMPACTION_MARKER: &str = "session is being continued";
1193
1194// Hermes can merge the summary into an existing message or restate an active task.
1195const HERMES_COMPACTION_MARKERS: &[&str] = &[
1196    "[context compaction \u{2014} reference only]",
1197    "[still in progress \u{2014} this is the active request, restated after the compaction boundary",
1198];
1199
1200/// The tool part of an `mcp__<server>__<tool>` name, the form Claude Code uses
1201/// for MCP tools. The server name is assumed not to contain `__`; the tool name
1202/// may.
1203fn mcp_tool_name(name: &str) -> Option<&str> {
1204    let (_server, tool) = name.strip_prefix("mcp__")?.split_once("__")?;
1205    (!tool.is_empty()).then_some(tool)
1206}
1207
1208/// The command field a tool call carries, when it has one. Harnesses name it
1209/// `command`, `cmd` or `input`; anything else is a tool whose category comes
1210/// from its name.
1211fn command_of(arguments: &Value) -> Option<&Value> {
1212    ["command", "cmd", "input"]
1213        .iter()
1214        .find_map(|key| arguments.get(*key))
1215}
1216
1217/// A command field as lowercase text, from a string or an argv array.
1218fn command_text(value: &Value) -> Option<String> {
1219    match value {
1220        Value::String(text) => Some(text.to_lowercase()),
1221        Value::Array(parts) => {
1222            let joined = parts
1223                .iter()
1224                .filter_map(Value::as_str)
1225                .collect::<Vec<_>>()
1226                .join(" ");
1227            (!joined.is_empty()).then(|| joined.to_lowercase())
1228        }
1229        _ => None,
1230    }
1231}
1232
1233/// Text carried by a content block, ignoring the non-textual kinds.
1234fn text_of(block: &ContentBlock) -> Option<&str> {
1235    match block {
1236        ContentBlock::Text { text } | ContentBlock::Refusal { text } => Some(text.as_str()),
1237        _ => None,
1238    }
1239}
1240
1241/// Reads of missing paths happen as often on tasks that succeed as on tasks
1242/// that fail, so they are not a failure signal.
1243fn is_missing_file(text: &str) -> bool {
1244    let lower = text.to_lowercase();
1245    // The OS error from shell tools, and Claude Code's Read tool message.
1246    lower.contains("no such file or directory") || lower.contains("file does not exist")
1247}
1248
1249fn build_signal(
1250    tool_texts: Vec<(String, bool)>,
1251    tool_calls: Vec<ObservedToolCall>,
1252    turn_depth: u32,
1253    recent_window: usize,
1254    semantics: &ToolSemantics,
1255) -> ToolSignals {
1256    // Windowed severity: take the MAX severity across the last `recent_window` tool
1257    // results rather than only the last one. An error's severity then persists for
1258    // the recent window and decays out of it — parallel to the windowed `recent_*`
1259    // counts — so a fix written a couple of turns after an error still routes on the
1260    // error signal instead of the router flapping straight back to the weak tier.
1261    let sev_start = tool_texts.len().saturating_sub(recent_window.max(1));
1262    let mut severity = 0.0f32;
1263    let mut failure_fingerprints = Vec::new();
1264    let mut repeated_failure = false;
1265    for (text, is_error) in &tool_texts[sev_start..] {
1266        let (sev, _patterns) = classify_text(text);
1267        // Explicit failure is at least hard; retain stronger text diagnostics.
1268        let sev = if *is_error { sev.max(HARD) } else { sev };
1269        if sev > severity {
1270            severity = sev;
1271        }
1272        if let Some(fingerprint) = failure_fingerprint(text, *is_error) {
1273            repeated_failure |= failure_fingerprints.contains(&fingerprint);
1274            failure_fingerprints.push(fingerprint);
1275        }
1276    }
1277
1278    let no_error_streak = compute_no_error_streak(&tool_texts);
1279
1280    // Single pass: cumulative + sliding-window counters together. Also tracks
1281    // the trailing pure-bash streak (consecutive `Unknown` calls back
1282    // from the end) — the build-pit proxy.
1283    let recent_start = tool_calls.len().saturating_sub(recent_window);
1284    let mut write_count = 0u32;
1285    let mut edit_count = 0u32;
1286    let mut read_count = 0u32;
1287    let mut todowrite_count = 0u32;
1288    let mut recent_write_count = 0u32;
1289    let mut recent_edit_count = 0u32;
1290    let mut recent_read_count = 0u32;
1291    let mut recent_todowrite_count = 0u32;
1292    let mut new_count = 0u32;
1293    let mut recent_new_count = 0u32;
1294    let mut pure_bash_streak = 0u32;
1295    let mut streak_open = true;
1296    for (i, tc) in tool_calls.iter().enumerate().rev() {
1297        // Every read, found or missed, is an observation. The read check parses
1298        // the command, so it beats the substring patterns. Otherwise the joined
1299        // name wins, so configs that list it keep working.
1300        let mut cat = if tc.is_retrieval {
1301            ToolSemantic::Observe
1302        } else {
1303            classify_tool_call_with_semantics(&tc.name, tc.command.as_deref(), semantics)
1304        };
1305        if matches!(cat, ToolSemantic::Unknown)
1306            && let Some(bare_name) = tc.bare_name
1307        {
1308            cat = classify_tool_call_with_semantics(bare_name, tc.command.as_deref(), semantics);
1309        }
1310        if streak_open {
1311            if matches!(cat, ToolSemantic::Unknown) {
1312                pure_bash_streak += 1;
1313            } else {
1314                streak_open = false;
1315            }
1316        }
1317        match cat {
1318            ToolSemantic::Mutate(MutationKind::Write) => {
1319                write_count += 1;
1320                if i >= recent_start {
1321                    recent_write_count += 1;
1322                }
1323            }
1324            ToolSemantic::Mutate(MutationKind::Edit) => {
1325                edit_count += 1;
1326                if i >= recent_start {
1327                    recent_edit_count += 1;
1328                }
1329            }
1330            ToolSemantic::Observe => {
1331                read_count += 1;
1332                if i >= recent_start {
1333                    recent_read_count += 1;
1334                }
1335            }
1336            ToolSemantic::Plan => {
1337                todowrite_count += 1;
1338                if i >= recent_start {
1339                    recent_todowrite_count += 1;
1340                }
1341            }
1342            ToolSemantic::New => {
1343                new_count += 1;
1344                if i >= recent_start {
1345                    recent_new_count += 1;
1346                }
1347            }
1348            ToolSemantic::Unknown => {}
1349        }
1350    }
1351
1352    let tests_passed = detect_tests_passed(&tool_texts, recent_window);
1353
1354    ToolSignals {
1355        severity,
1356        repeated_failure,
1357        no_error_streak,
1358        edit_count,
1359        write_count,
1360        read_count,
1361        todowrite_count,
1362        recent_edit_count,
1363        recent_write_count,
1364        recent_read_count,
1365        recent_todowrite_count,
1366        new_count,
1367        recent_new_count,
1368        pure_bash_streak,
1369        tests_passed,
1370        turn_depth,
1371        // Set by extract_tool_signals_with_window after the format-specific extract,
1372        // which scans all message contents for the compaction marker and tallies
1373        // the raw conversation-shape counts.
1374        tool_result_count: 0,
1375        assistant_turn_count: 0,
1376        compacted: false,
1377    }
1378}
1379
1380// ─── pure helpers ─────────────────────────────────────────────────────────────
1381
1382/// Normalise a JSON tool-result content value to a plain string.
1383fn content_to_text(content: Option<&Value>) -> Option<String> {
1384    match content? {
1385        Value::String(s) => Some(s.clone()),
1386        Value::Array(blocks) => {
1387            let parts: Vec<&str> = blocks
1388                .iter()
1389                .filter_map(|b| {
1390                    b.as_object()
1391                        .filter(|o| o.get("type").and_then(Value::as_str) == Some("text"))
1392                        .and_then(|o| o.get("text"))
1393                        .and_then(Value::as_str)
1394                })
1395                .collect();
1396            if parts.is_empty() {
1397                None
1398            } else {
1399                Some(parts.join("\n"))
1400            }
1401        }
1402        _ => None,
1403    }
1404}
1405
1406/// Match tool text and structured result fields against error patterns.
1407///
1408/// Returns `(max_severity, matched_pattern_names)`.
1409pub(crate) fn classify_text(text: &str) -> (f32, Vec<String>) {
1410    let lower = text.to_lowercase();
1411    let mut patterns = Vec::new();
1412    let mut severity: f32 = 0.0;
1413    for (name, sev, substrings) in ERROR_PATTERNS {
1414        if substrings.iter().any(|sub| lower.contains(sub)) {
1415            patterns.push(name.to_string());
1416            severity = severity.max(*sev);
1417        }
1418    }
1419    // Hermes can append a loop warning after the JSON result.
1420    let result = serde_json::Deserializer::from_str(text)
1421        .into_iter::<Value>()
1422        .next()
1423        .and_then(|result| result.ok())
1424        .unwrap_or_default();
1425    let nonzero_exit = result["exit_code"].as_i64().is_some_and(|code| code != 0);
1426    let tool_error = result["success"].as_bool() == Some(false)
1427        || result["error"]
1428            .as_str()
1429            .is_some_and(|error| !error.trim().is_empty());
1430    if (nonzero_exit || has_nonzero_exit_status(&lower))
1431        && !patterns.iter().any(|p| p == "exit_nonzero")
1432    {
1433        patterns.push("exit_nonzero".to_string());
1434        severity = severity.max(SOFT);
1435    }
1436    for (name, matched) in [
1437        ("tool_error", tool_error),
1438        ("compile_error", has_compiler_diagnostic(&lower)),
1439        ("runtime_exception", has_runtime_exception(&lower)),
1440        ("runtime_panic", has_runtime_panic(&lower)),
1441        ("patch_error", has_patch_failure(&lower)),
1442    ] {
1443        if matched && !patterns.iter().any(|pattern| pattern == name) {
1444            patterns.push(name.to_string());
1445            severity = severity.max(HARD);
1446        }
1447    }
1448    (severity, patterns)
1449}
1450
1451/// Stable identity for a material failure. Soft non-zero exits need an explicit
1452/// failure flag to count as a repeated mistake.
1453fn failure_fingerprint(text: &str, is_error: bool) -> Option<String> {
1454    let (severity, patterns) = classify_text(text);
1455    if severity < HARD && !is_error {
1456        return None;
1457    }
1458
1459    let lower = text.to_lowercase();
1460    let diagnostic = lower
1461        .lines()
1462        .find(|line| is_failure_diagnostic(line))
1463        .or_else(|| lower.lines().find(|line| !line.trim().is_empty()))
1464        .unwrap_or_default();
1465    let normalized = normalize_failure_text(diagnostic);
1466    Some(format!("{}|{normalized}", patterns.join(",")))
1467}
1468
1469fn is_failure_diagnostic(line: &str) -> bool {
1470    let line = line.trim();
1471    [
1472        "error",
1473        "exception",
1474        "panic",
1475        "failed",
1476        "timed out",
1477        "timeout",
1478        "connection refused",
1479        "cannot allocate memory",
1480        "out of memory",
1481        "not found",
1482    ]
1483    .iter()
1484    .any(|marker| line.contains(marker))
1485}
1486
1487/// Removes values that normally change between retries while retaining the
1488/// diagnostic wording that distinguishes one failure from another.
1489fn normalize_failure_text(text: &str) -> String {
1490    let mut normalized = String::new();
1491    for word in text.split_whitespace() {
1492        if !normalized.is_empty() {
1493            normalized.push(' ');
1494        }
1495        let mut in_digits = false;
1496        if word.starts_with('/') || word.contains("/src/") || word.contains("/tmp/") {
1497            normalized.push_str("<path>");
1498            continue;
1499        }
1500        for character in word.chars() {
1501            if character.is_ascii_digit() {
1502                if !in_digits {
1503                    normalized.push('#');
1504                    in_digits = true;
1505                }
1506            } else {
1507                normalized.push(character);
1508                in_digits = false;
1509            }
1510        }
1511    }
1512    normalized.chars().take(240).collect()
1513}
1514
1515fn has_compiler_diagnostic(lower: &str) -> bool {
1516    lower.lines().any(|line| {
1517        let line = line.trim_start();
1518        if matches!(
1519            line,
1520            "compilation failed" | "error: compilation failed" | "error: could not compile"
1521        ) || line.starts_with("error: could not compile ")
1522        {
1523            return true;
1524        }
1525
1526        let Some(rest) = line.strip_prefix("error[e") else {
1527            return false;
1528        };
1529        let Some((code, _)) = rest.split_once("]:") else {
1530            return false;
1531        };
1532        !code.is_empty() && code.chars().all(|character| character.is_ascii_digit())
1533    })
1534}
1535
1536fn has_runtime_exception(lower: &str) -> bool {
1537    let has_exception_line = lower.lines().any(|line| {
1538        let line = line.trim_start();
1539        [
1540            "typeerror:",
1541            "referenceerror:",
1542            "rangeerror:",
1543            "runtimeerror:",
1544            "keyerror:",
1545            "attributeerror:",
1546        ]
1547        .iter()
1548        .any(|prefix| line.starts_with(prefix))
1549    });
1550    has_exception_line && (lower.contains("\n    at ") || lower.contains("\n  at "))
1551}
1552
1553fn has_runtime_panic(lower: &str) -> bool {
1554    lower
1555        .lines()
1556        .any(|line| line.trim_start().starts_with("panic: runtime error:"))
1557        && (lower.contains("\ngoroutine ") || lower.contains("[signal sig"))
1558}
1559
1560fn has_patch_failure(lower: &str) -> bool {
1561    lower.lines().any(|line| {
1562        let line = line.trim_start();
1563        line.starts_with("error: patch failed:")
1564            || line.starts_with("patch failed:")
1565            || line.contains(": patch does not apply")
1566            || line.starts_with("invalid context")
1567    })
1568}
1569
1570/// Detects `exit_nonzero` only when a supported exit phrase is followed by a
1571/// nonzero decimal status.
1572///
1573/// Codex includes "Process exited with code 0" on clean tool results, so exit
1574/// phrases must parse their numeric status instead of matching the phrase alone.
1575fn has_nonzero_exit_status(lower: &str) -> bool {
1576    NONZERO_EXIT_PHRASES
1577        .iter()
1578        .any(|phrase| phrase_followed_by_nonzero_integer(lower, phrase))
1579}
1580
1581/// Matches common "exit code/status N" spellings after optional separators.
1582fn phrase_followed_by_nonzero_integer(lower: &str, phrase: &str) -> bool {
1583    let mut cursor = 0usize;
1584    while let Some(rel) = lower[cursor..].find(phrase) {
1585        let value_start = cursor + rel + phrase.len();
1586        let rest = lower[value_start..].trim_start_matches(|c: char| {
1587            c.is_ascii_whitespace() || matches!(c, ':' | '=' | '\'' | '"' | '`')
1588        });
1589        let digits: String = rest.chars().take_while(|c| c.is_ascii_digit()).collect();
1590        if !digits.is_empty() && digits.chars().any(|d| d != '0') {
1591            return true;
1592        }
1593        cursor = value_start;
1594    }
1595    false
1596}
1597
1598fn compute_no_error_streak(tool_texts: &[(String, bool)]) -> u32 {
1599    let mut streak = 0u32;
1600    for (text, is_error) in tool_texts.iter().rev() {
1601        let (sev, _) = classify_text(text);
1602        if *is_error || sev > 0.0 {
1603            break;
1604        }
1605        streak += 1;
1606    }
1607    streak
1608}
1609
1610fn detect_tests_passed(tool_texts: &[(String, bool)], recent_window: usize) -> bool {
1611    let start = tool_texts.len().saturating_sub(recent_window.max(1));
1612    let recent = &tool_texts[start..];
1613    let after_latest_failure = recent
1614        .iter()
1615        .rposition(|(text, is_error)| *is_error || classify_text(text).0 > 0.0)
1616        .map_or(recent, |index| &recent[index + 1..]);
1617    after_latest_failure.iter().any(|(text, _)| {
1618        let lower = text.to_lowercase();
1619        TEST_PASS_PHRASES.iter().any(|p| lower.contains(p))
1620            && !TEST_FAILURE_LITERAL.iter().any(|p| lower.contains(p))
1621            && !has_nonzero_failure_count(&lower)
1622    })
1623}
1624
1625// True iff `lower` contains a `NUMERIC_FAILURE_KEYWORDS` token preceded
1626// (modulo whitespace) by a nonzero integer. The "modulo whitespace" lets
1627// "1 failed", "1\nfailed", and "1  failed" all trip; the nonzero guard
1628// keeps cargo's "0 failed" / go's "0 errors" / pytest's "0 errors in"
1629// summaries from being misread as failures on a clean run.
1630fn has_nonzero_failure_count(lower: &str) -> bool {
1631    for kw in NUMERIC_FAILURE_KEYWORDS {
1632        let mut cursor = 0usize;
1633        while let Some(rel) = lower[cursor..].find(kw) {
1634            let kw_start = cursor + rel;
1635            let kw_end = kw_start + kw.len();
1636            // Word boundary AFTER the keyword — "errors" mid-word (e.g.
1637            // "errored") shouldn't count as a failure-count site.
1638            let boundary_after = lower[kw_end..]
1639                .chars()
1640                .next()
1641                .is_none_or(|c| !c.is_ascii_alphanumeric());
1642            if boundary_after {
1643                let prefix = &lower[..kw_start];
1644                let trimmed = prefix.trim_end_matches(|c: char| c.is_whitespace());
1645                let digits_rev: String = trimmed
1646                    .chars()
1647                    .rev()
1648                    .take_while(|c| c.is_ascii_digit())
1649                    .collect();
1650                if !digits_rev.is_empty() && digits_rev.chars().any(|d| d != '0') {
1651                    return true;
1652                }
1653            }
1654            cursor = kw_start + kw.len();
1655        }
1656    }
1657    false
1658}
1659
1660// ─── tests ───────────────────────────────────────────────────────────────────
1661
1662#[cfg(test)]
1663mod tests {
1664    use super::*;
1665    use crate::algorithms::util::stage::score_signal;
1666    use serde_json::json;
1667    use switchyard_protocol::codex_namespaces::TOOL_NAMESPACES_KEY;
1668    use switchyard_protocol::{
1669        ContentBlock, LlmRequest, Message, Metadata, Role, ToolCall, ToolResult,
1670    };
1671
1672    fn with_messages(messages: Vec<Message>) -> Request {
1673        Request {
1674            llm_request: LlmRequest {
1675                messages,
1676                ..LlmRequest::default()
1677            },
1678            raw_request: None,
1679            metadata: None,
1680        }
1681    }
1682
1683    // assistant message with a single named tool call
1684    fn tc(name: &str) -> Message {
1685        Message {
1686            role: Role::Assistant,
1687            content: vec![ContentBlock::ToolCall(ToolCall {
1688                id: String::new(),
1689                name: name.to_string(),
1690                arguments: json!({}),
1691            })],
1692        }
1693    }
1694
1695    // assistant Bash message carrying `command`
1696    fn bash(command: &str) -> Message {
1697        Message {
1698            role: Role::Assistant,
1699            content: vec![ContentBlock::ToolCall(ToolCall {
1700                id: String::new(),
1701                name: "Bash".to_string(),
1702                arguments: json!({"command": command}),
1703            })],
1704        }
1705    }
1706
1707    // a tool result message (goes in a user-role message, as in Anthropic's normalised form)
1708    fn tr(text: &str) -> Message {
1709        Message {
1710            role: Role::User,
1711            content: vec![ContentBlock::ToolResult(ToolResult {
1712                tool_call_id: String::new(),
1713                content: vec![ContentBlock::Text {
1714                    text: text.to_string(),
1715                }],
1716                is_error: None,
1717            })],
1718        }
1719    }
1720
1721    #[test]
1722    fn clean_text_has_zero_severity() {
1723        let (sev, patterns) = classify_text("everything went fine");
1724        assert_eq!(sev, 0.0);
1725        assert!(patterns.is_empty());
1726    }
1727
1728    #[test]
1729    fn structured_tool_failures_affect_recovery_signals() {
1730        for (text, severity) in [
1731            (r#"{"output":"","exit_code":7,"error":null}"#, SOFT),
1732            (r#"{"output":"","exit_code":-1,"error":null}"#, SOFT),
1733            (r#"{"success":false,"error":"No matching text"}"#, HARD),
1734            (r#"{"success":false,"error":null}"#, HARD),
1735            (
1736                r#"{"error":"Overwrite refused","stale_write_blocked":true}"#,
1737                HARD,
1738            ),
1739            (r#"{"output":"out of memory","exit_code":1}"#, CRITICAL),
1740            (r#"{"output":"done","exit_code":0,"error":null}"#, 0.0),
1741            (r#"{"success":true,"error":"  "}"#, 0.0),
1742            (r#"{"output":"running","exit_code":null,"error":null}"#, 0.0),
1743            (r#"{"output":{"success":false},"exit_code":0}"#, 0.0),
1744        ] {
1745            let warned = format!("{text}\n\n[Tool loop warning: repeated identical call]");
1746            let request = with_messages(vec![tr("5 passed in 0.12s"), tr(text), tr(&warned)]);
1747            let signal = ToolSignals::from_request(&request, None);
1748            let clean = severity == 0.0;
1749            assert_eq!(signal.severity, severity, "{text}");
1750            assert_eq!(signal.no_error_streak, if clean { 3 } else { 0 }, "{text}");
1751            assert_eq!(signal.repeated_failure, severity >= HARD, "{text}");
1752            assert_eq!(signal.tests_passed, clean, "{text}");
1753        }
1754    }
1755
1756    #[test]
1757    fn traceback_is_hard() {
1758        let (sev, patterns) = classify_text("Traceback (most recent call last):\n  ValueError");
1759        assert_eq!(sev, HARD);
1760        assert!(patterns.contains(&"traceback".to_string()));
1761    }
1762
1763    #[test]
1764    fn oom_is_critical() {
1765        let (sev, _) = classify_text("Out of memory: kill process 1234");
1766        assert_eq!(sev, CRITICAL);
1767    }
1768
1769    #[test]
1770    fn connection_refused_is_hard() {
1771        let (severity, _) = classify_text("Connection refused on port 8000");
1772        assert_eq!(severity, HARD);
1773    }
1774
1775    #[test]
1776    fn repeated_failure_ignores_volatile_paths_and_numbers() {
1777        let request = with_messages(vec![
1778            tr("error[E0308]: mismatched types at /tmp/a/src/lib.rs:12"),
1779            tr("error[E0308]: mismatched types at /tmp/b/src/lib.rs:47"),
1780        ]);
1781        assert!(ToolSignals::from_request(&request, None).repeated_failure);
1782    }
1783
1784    #[test]
1785    fn different_failures_are_not_repeated() {
1786        let request = with_messages(vec![
1787            tr("error[E0308]: mismatched types"),
1788            tr("error[E0509]: cannot move out"),
1789        ]);
1790        assert!(!ToolSignals::from_request(&request, None).repeated_failure);
1791    }
1792
1793    #[test]
1794    fn one_material_failure_is_not_repeated() {
1795        let request = with_messages(vec![tr("Connection refused on port 8000")]);
1796        assert!(!ToolSignals::from_request(&request, None).repeated_failure);
1797    }
1798
1799    /// Explicit failures count even without diagnostic text and cannot signal recovery.
1800    #[test]
1801    fn structured_tool_failures_feed_error_and_recovery_signals() {
1802        for text in ["Dependency unavailable", "", "5 passed in 0.12s"] {
1803            let mut failed = tr(text);
1804            let ContentBlock::ToolResult(result) = &mut failed.content[0] else {
1805                panic!("expected tool result");
1806            };
1807            result.is_error = Some(true);
1808            let mut request = with_messages(vec![tr("5 passed in 0.12s"), failed.clone()]);
1809            let signals = ToolSignals::from_request(&request, Some(3));
1810            assert_eq!(signals.severity, HARD);
1811            assert!(!signals.repeated_failure);
1812            assert_eq!(signals.no_error_streak, 0);
1813            assert!(!signals.tests_passed);
1814
1815            request.llm_request.messages.push(failed);
1816            assert!(ToolSignals::from_request(&request, Some(3)).repeated_failure);
1817            request.llm_request.messages.push(tr("5 passed in 0.12s"));
1818            let recovered = ToolSignals::from_request(&request, Some(1));
1819            assert_eq!(recovered.severity, 0.0);
1820            assert!(!recovered.repeated_failure);
1821            assert_eq!(recovered.no_error_streak, 1);
1822            assert!(recovered.tests_passed);
1823        }
1824    }
1825
1826    #[test]
1827    fn severity_is_max_across_patterns() {
1828        // exit_nonzero (SOFT) + traceback (HARD) → HARD.
1829        let (sev, _) = classify_text("exit code 1\nTraceback (most recent call last):");
1830        assert_eq!(sev, HARD);
1831    }
1832
1833    #[test]
1834    fn codex_process_exit_zero_stays_clean() {
1835        let (sev, patterns) =
1836            classify_text("Chunk ID: abc\nProcess exited with code 0\nOutput:\nok");
1837        assert_eq!(sev, 0.0);
1838        assert!(!patterns.contains(&"exit_nonzero".to_string()));
1839    }
1840
1841    #[test]
1842    fn nonzero_exit_codes_are_soft_errors() {
1843        let cases = [
1844            "Process exited with code 1",
1845            "Process exited with code 127",
1846            "exit code: 2",
1847            "exit status 3",
1848            "exited with status 9",
1849        ];
1850        for case in cases {
1851            let (sev, patterns) = classify_text(case);
1852            assert_eq!(sev, SOFT, "expected soft severity for {case}");
1853            assert!(patterns.contains(&"exit_nonzero".to_string()));
1854        }
1855    }
1856
1857    #[test]
1858    fn partial_process_failures_are_hard_errors() {
1859        let cases = [
1860            (
1861                "Process running with session ID 12\nOutput:\nerror[E0509]: cannot move out",
1862                "compile_error",
1863            ),
1864            (
1865                "Process exited with code 0\nOutput:\nTypeError: value is undefined\n    at main.js:1:2",
1866                "runtime_exception",
1867            ),
1868            (
1869                "Process running with session ID 13\nOutput:\npanic: runtime error: index out of range\n\ngoroutine 6 [running]:",
1870                "runtime_panic",
1871            ),
1872            (
1873                "Process exited with code 0\nOutput:\nerror: patch failed: src/lib.rs:4\nerror: src/lib.rs: patch does not apply",
1874                "patch_error",
1875            ),
1876        ];
1877        for (text, expected_pattern) in cases {
1878            let (severity, patterns) = classify_text(text);
1879            assert_eq!(severity, HARD, "expected hard severity for {text}");
1880            assert!(patterns.iter().any(|pattern| pattern == expected_pattern));
1881        }
1882    }
1883
1884    #[test]
1885    fn source_text_that_names_exceptions_stays_clean() {
1886        let text =
1887            "pub enum TypeError: this is documentation\nlet sample = 'panic: runtime error:';";
1888        assert_eq!(classify_text(text).0, 0.0);
1889    }
1890
1891    #[test]
1892    fn file_does_not_exist_is_hard() {
1893        // Claude Code Read-tool miss. Trace-mined addition (22 true / 2 false positives).
1894        let (sev, patterns) =
1895            classify_text("Error: File does not exist. Note: current working directory is /app.");
1896        assert_eq!(sev, HARD);
1897        assert!(patterns.contains(&"no_such_file".to_string()));
1898    }
1899
1900    #[test]
1901    fn bare_does_not_exist_stays_clean() {
1902        // Precision guard: only the anchored "file does not exist" fires, so a bare
1903        // "does not exist" in prose or directory output must not trip a false error.
1904        let (sev, _) = classify_text("The directory does not exist yet, creating it now.");
1905        assert_eq!(sev, 0.0);
1906    }
1907
1908    #[test]
1909    fn no_error_streak_all_clean() {
1910        let texts = vec![("ok".to_string(), false), ("all good".to_string(), false)];
1911        assert_eq!(compute_no_error_streak(&texts), 2);
1912    }
1913
1914    #[test]
1915    fn no_error_streak_stops_at_error() {
1916        let texts = vec![
1917            ("Traceback (most recent call last):".to_string(), false),
1918            ("ok".to_string(), false),
1919            ("ok".to_string(), false),
1920        ];
1921        assert_eq!(compute_no_error_streak(&texts), 2);
1922    }
1923
1924    #[test]
1925    fn tests_passed_detects_pytest_output() {
1926        assert!(detect_tests_passed(
1927            &[("====== 5 passed in 0.12s ======".to_string(), false)],
1928            DEFAULT_RECENT_WINDOW
1929        ));
1930    }
1931
1932    #[test]
1933    fn tests_passed_ignores_partial_failures() {
1934        assert!(!detect_tests_passed(
1935            &[("2 failed, 5 passed in 0.56s".to_string(), false)],
1936            DEFAULT_RECENT_WINDOW
1937        ));
1938    }
1939
1940    #[test]
1941    fn tests_passed_must_follow_the_latest_failure() {
1942        assert!(!detect_tests_passed(
1943            &[
1944                ("5 passed in 0.12s".to_string(), false),
1945                (
1946                    "Traceback (most recent call last):\nValueError".to_string(),
1947                    false
1948                ),
1949                ("edit applied".to_string(), false),
1950            ],
1951            DEFAULT_RECENT_WINDOW
1952        ));
1953        assert!(detect_tests_passed(
1954            &[
1955                (
1956                    "Traceback (most recent call last):\nValueError".to_string(),
1957                    false
1958                ),
1959                ("5 passed in 0.12s".to_string(), false),
1960            ],
1961            DEFAULT_RECENT_WINDOW
1962        ));
1963    }
1964
1965    #[test]
1966    fn retrieved_file_contents_are_ignored() {
1967        let call = |id: &str, name: &str, arguments: Value| Message {
1968            role: Role::Assistant,
1969            content: vec![ContentBlock::ToolCall(ToolCall {
1970                id: id.to_string(),
1971                name: name.to_string(),
1972                arguments,
1973            })],
1974        };
1975        let result = |id: &str, text: &str| Message {
1976            role: Role::User,
1977            content: vec![ContentBlock::ToolResult(ToolResult {
1978                tool_call_id: id.to_string(),
1979                content: vec![ContentBlock::Text {
1980                    text: text.to_string(),
1981                }],
1982                is_error: None,
1983            })],
1984        };
1985        let signal = extract_tool_signals_with_window(
1986            &with_messages(vec![
1987                call("a", "Bash", json!({"command": "pytest"})),
1988                result("a", "Traceback (most recent call last):\nValueError"),
1989                call("b", "Read", json!({"file_path": "notes.md"})),
1990                result(
1991                    "b",
1992                    r#"{"success":false,"error":"out of memory","exit_code":7}"#,
1993                ),
1994                call("c", "Grep", json!({"pattern": "passed"})),
1995                result("c", "CHANGELOG.md: all tests passed"),
1996            ]),
1997            DEFAULT_RECENT_WINDOW,
1998        );
1999        // Only the real pytest run counts.
2000        assert_eq!(signal.severity, HARD);
2001        assert!(!signal.tests_passed);
2002        assert_eq!(signal.tool_result_count, 3);
2003
2004        for (command, text, severity) in [
2005            ("cat config.json", r#"{"error":"fixture data"}"#, 0.0),
2006            (
2007                "cat missing.json",
2008                r#"{"output":"","exit_code":7,"error":null}"#,
2009                SOFT,
2010            ),
2011            (
2012                "cat config.json && python check.py",
2013                r#"{"error":"check rejected"}"#,
2014                HARD,
2015            ),
2016        ] {
2017            let request = with_messages(vec![
2018                call("shell", "Bash", json!({"command": command})),
2019                result("shell", text),
2020            ]);
2021            assert_eq!(
2022                ToolSignals::from_request(&request, None).severity,
2023                severity,
2024                "{text}"
2025            );
2026        }
2027    }
2028
2029    // A request whose tool result answers the tool call.
2030    fn call_with_result(mut call: Message, mut result: Message) -> Request {
2031        if let ContentBlock::ToolCall(call) = &mut call.content[0] {
2032            call.id = "call-1".into();
2033        }
2034        if let ContentBlock::ToolResult(result) = &mut result.content[0] {
2035            result.tool_call_id = "call-1".into();
2036        }
2037        with_messages(vec![call, result])
2038    }
2039
2040    #[test]
2041    fn shell_reads_are_ignored_but_keep_their_window_slot() {
2042        for command in [
2043            "/usr/bin/cat logfile.txt",
2044            "cd /repo && tail -n 100 logfile.txt | grep MemoryError",
2045            "env MODE=debug timeout 5s rg 'MemoryError' .",
2046        ] {
2047            let mut request = call_with_result(bash(command), tr("MemoryError\nall tests passed"));
2048            request.llm_request.messages.splice(
2049                0..0,
2050                [bash("pytest"), tr("Traceback (most recent call last):")],
2051            );
2052            // The read adds no error and no test pass, so the earlier failure stands.
2053            let signal = ToolSignals::from_request(&request, Some(2));
2054            assert_eq!(signal.severity, HARD, "{command}");
2055            assert!(!signal.tests_passed, "{command}");
2056            // The read still fills a window slot, so the failure can age out.
2057            let signal = ToolSignals::from_request(&request, Some(1));
2058            assert_eq!(signal.severity, 0.0, "{command}");
2059        }
2060    }
2061
2062    #[test]
2063    fn shell_commands_that_run_code_keep_signals() {
2064        let oversized = format!("cat {}", "a".repeat(MAX_RETRIEVAL_COMMAND_BYTES));
2065        for command in [
2066            "cat logfile.txt; pytest",
2067            "echo $(pytest)",
2068            "find . -name '*.py' -exec pytest ';'",
2069            oversized.as_str(),
2070        ] {
2071            let request = call_with_result(bash(command), tr("MemoryError"));
2072            let signal = ToolSignals::from_request(&request, None);
2073            assert_eq!(signal.severity, CRITICAL, "{command}");
2074        }
2075    }
2076
2077    #[test]
2078    fn failed_reads_signal_unless_the_file_is_missing() {
2079        let failed = |text: &str| {
2080            let mut result = tr(text);
2081            if let ContentBlock::ToolResult(result) = &mut result.content[0] {
2082                result.is_error = Some(true);
2083            }
2084            result
2085        };
2086        let missing = "missing.txt: No such file or directory";
2087        let severities = [
2088            call_with_result(bash("cat missing.txt"), failed(missing)),
2089            call_with_result(tc("Read"), failed("File does not exist.")),
2090            // A read that fails for another reason still counts.
2091            call_with_result(bash("cat logfile.txt"), failed("MemoryError")),
2092            // Only reads are exempt.
2093            call_with_result(bash("python missing.txt"), failed(missing)),
2094        ]
2095        .map(|request| ToolSignals::from_request(&request, None).severity);
2096        assert_eq!(severities, [0.0, 0.0, CRITICAL, HARD]);
2097
2098        let request = call_with_result(
2099            bash("cat missing.json"),
2100            tr(r#"{"output":"No such file or directory","exit_code":1}"#),
2101        );
2102        assert_eq!(ToolSignals::from_request(&request, None).severity, 0.0);
2103    }
2104
2105    #[test]
2106    fn argv_commands_keep_word_boundaries() {
2107        let signal = |argv: Value| {
2108            let call = exec_command(json!({"cmd": argv}));
2109            ToolSignals::from_request(&call_with_result(call, tr("MemoryError")), None)
2110        };
2111        let mut many_args = vec![""; MAX_RETRIEVAL_COMMAND_BYTES];
2112        many_args[0] = "cat";
2113        let severities = [
2114            json!(["bash", "-lc", "cd repo && rg foo"]),
2115            json!(["bash", "-lc", "cat log; pytest"]),
2116            // One argv word is one file name, even with a separator in it.
2117            json!(["cat", "log; pytest"]),
2118            // Over the byte budget, by size and by argument count.
2119            json!(["cat", "a".repeat(MAX_RETRIEVAL_COMMAND_BYTES)]),
2120            json!(many_args),
2121        ]
2122        .map(|argv| signal(argv).severity);
2123        assert_eq!(severities, [0.0, CRITICAL, 0.0, CRITICAL, CRITICAL]);
2124        // A read is an observation, even as a whole `-lc` script.
2125        let read = signal(json!(["bash", "-lc", "sed -n 1,200p x"]));
2126        assert_eq!(read.read_count, 1);
2127        // Other commands are counted inside the script too.
2128        let write = signal(json!(["bash", "-lc", "cd repo && mkdir x"]));
2129        assert_eq!(write.write_count, 1);
2130    }
2131
2132    #[test]
2133    fn severity_is_windowed_over_recent_results() {
2134        // An error two results back, then two clean results.
2135        let request = with_messages(vec![
2136            tr("Traceback (most recent call last):\n  ValueError"),
2137            tr("ok"),
2138            tr("ok"),
2139        ]);
2140        // window covers the error → severity persists (max over the window)
2141        assert_eq!(extract_tool_signals_with_window(&request, 3).severity, HARD);
2142        // window of 1 sees only the last (clean) result → severity has decayed out
2143        assert_eq!(extract_tool_signals_with_window(&request, 1).severity, 0.0);
2144    }
2145
2146    #[test]
2147    fn extract_openai_chat_tool_results() {
2148        let request = with_messages(vec![
2149            Message::text(Role::User, "do something"),
2150            tc("Edit"),
2151            tr("Traceback (most recent call last):\n  ValueError"),
2152        ]);
2153        let sig = ToolSignals::from_request(&request, None);
2154        assert_eq!(sig.severity, HARD);
2155        assert_eq!(sig.edit_count, 1);
2156        assert_eq!(sig.turn_depth, 3);
2157    }
2158
2159    #[test]
2160    fn extract_anthropic_tool_results() {
2161        let request = with_messages(vec![tr("Traceback (most recent call last):\n  ValueError")]);
2162        let sig = ToolSignals::from_request(&request, None);
2163        assert_eq!(sig.severity, HARD);
2164    }
2165
2166    #[test]
2167    fn extract_responses_api_tool_results() {
2168        let request = with_messages(vec![tc("Write"), tr("file written successfully")]);
2169        let sig = ToolSignals::from_request(&request, None);
2170        assert_eq!(sig.severity, 0.0);
2171        assert_eq!(sig.write_count, 1);
2172    }
2173
2174    #[test]
2175    fn responses_builtin_tool_failures_escalate() {
2176        use crate::algorithms::util::stage::{PickOutcome, PickerMode, Tier, pick_tier};
2177
2178        let mut cases = Vec::new();
2179        for (status, output) in [
2180            (
2181                "failed",
2182                "Synthetic dependency unavailable; retry with the recovery path.",
2183            ),
2184            (
2185                "completed",
2186                "Synthetic dependency unavailable; retry with the recovery path.",
2187            ),
2188            ("failed", ""),
2189        ] {
2190            cases.push((
2191                json!({
2192                    "type": "apply_patch_call_output",
2193                    "status": status,
2194                    "output": output,
2195                }),
2196                if status == "failed" { HARD } else { 0.0 },
2197            ));
2198        }
2199        for (outcome, stdout, stderr, severity) in [
2200            (json!({"type": "exit", "exit_code": 1}), "", "", SOFT),
2201            (json!({"type": "exit", "exit_code": 1}), "", " \n", SOFT),
2202            (
2203                json!({"type": "exit", "exit_code": 1}),
2204                "",
2205                "command failed",
2206                HARD,
2207            ),
2208            (json!({"type": "timeout"}), "", "", HARD),
2209            (json!({"type": "exit", "exit_code": 0}), "done", "", 0.0),
2210            (
2211                json!({"type": "exit", "exit_code": 0}),
2212                "Traceback (most recent call last):",
2213                "",
2214                HARD,
2215            ),
2216            (
2217                json!({"type": "exit", "exit_code": 0}),
2218                "",
2219                "Traceback (most recent call last):",
2220                HARD,
2221            ),
2222        ] {
2223            cases.push((
2224                json!({
2225                    "type": "shell_call_output",
2226                    "output": [
2227                        {"stdout": stdout, "stderr": stderr, "outcome": outcome},
2228                        {"stdout": "", "stderr": "", "outcome": {"type": "exit", "exit_code": 0}}
2229                    ],
2230                }),
2231                severity,
2232            ));
2233        }
2234        for (raw, severity) in cases {
2235            let is_error = severity >= HARD;
2236            let mut request = with_messages(
2237                ["call_1", "call_2"]
2238                    .into_iter()
2239                    .map(|call_id| {
2240                        let mut raw = raw.clone();
2241                        raw["call_id"] = json!(call_id);
2242                        Message {
2243                            role: Role::User,
2244                            content: vec![ContentBlock::Unknown {
2245                                provider: WireFormat::OpenAiResponses.into(),
2246                                raw,
2247                            }],
2248                        }
2249                    })
2250                    .collect(),
2251            );
2252            let signal = ToolSignals::from_request(&request, Some(3));
2253            assert_eq!(signal.severity, severity, "{raw}");
2254            assert_eq!(signal.repeated_failure, is_error, "{raw}");
2255            assert_eq!(signal.tool_result_count, 2);
2256            assert_eq!(
2257                matches!(
2258                    pick_tier(&signal, PickerMode::EfficientFirst, 0.5),
2259                    PickOutcome::Resolved {
2260                        tier: Tier::Capable,
2261                        ..
2262                    }
2263                ),
2264                is_error,
2265                "{raw}"
2266            );
2267
2268            let mut success = match raw["type"].as_str() {
2269                Some("apply_patch_call_output") => json!({
2270                    "type": "apply_patch_call_output", "status": "completed", "output": ""
2271                }),
2272                Some("shell_call_output") => json!({
2273                    "type": "shell_call_output",
2274                    "output": [{"stdout": "", "stderr": "", "outcome": {"type": "exit", "exit_code": 0}}]
2275                }),
2276                _ => unreachable!(),
2277            };
2278            for index in 0..3 {
2279                success["call_id"] = json!(format!("success_{index}"));
2280                request.llm_request.messages.push(Message {
2281                    role: Role::User,
2282                    content: vec![ContentBlock::Unknown {
2283                        provider: WireFormat::OpenAiResponses.into(),
2284                        raw: success.clone(),
2285                    }],
2286                });
2287            }
2288            let recovered = ToolSignals::from_request(&request, Some(3));
2289            assert_eq!(recovered.severity, 0.0, "{raw}");
2290            assert!(!recovered.repeated_failure, "{raw}");
2291            assert!(recovered.no_error_streak >= 3, "{raw}");
2292            assert_eq!(recovered.tool_result_count, 5);
2293        }
2294    }
2295
2296    #[test]
2297    fn conversation_counts_are_per_block_and_role_aware() {
2298        // A batched user message (Anthropic shape) contributes one count per
2299        // ToolResult block, empty-content results included. assistant_turn_count
2300        // tracks Role::Assistant only, while turn_depth counts every message.
2301        let result = |content: Vec<ContentBlock>| {
2302            ContentBlock::ToolResult(ToolResult {
2303                tool_call_id: String::new(),
2304                content,
2305                is_error: None,
2306            })
2307        };
2308        let request = with_messages(vec![
2309            Message::text(Role::User, "do something"),
2310            Message::text(Role::Assistant, "working"),
2311            Message {
2312                role: Role::User,
2313                content: vec![
2314                    result(vec![ContentBlock::Text {
2315                        text: "ok".to_string(),
2316                    }]),
2317                    result(Vec::new()),
2318                ],
2319            },
2320            tc("Bash"),
2321        ]);
2322        let sig = ToolSignals::from_request(&request, None);
2323        assert_eq!(sig.tool_result_count, 2);
2324        assert_eq!(sig.assistant_turn_count, 2);
2325        assert_eq!(sig.turn_depth, 4);
2326    }
2327
2328    #[test]
2329    fn recent_window_counts_only_last_default_window_tool_calls() {
2330        // 5 writes + 1 edit at the end → the default window (3) should see
2331        // the last 3 calls: 1 edit + 2 writes (not all 6 calls).
2332        let request = with_messages(vec![
2333            tc("Write"),
2334            tr("ok"),
2335            tc("Write"),
2336            tr("ok"),
2337            tc("Write"),
2338            tr("ok"),
2339            tc("Write"),
2340            tr("ok"),
2341            tc("Write"),
2342            tr("ok"),
2343            tc("Edit"),
2344            tr("ok"),
2345        ]);
2346        let sig = ToolSignals::from_request(&request, None);
2347        assert_eq!(sig.write_count, 5);
2348        assert_eq!(sig.edit_count, 1);
2349        assert_eq!(sig.recent_write_count, 2);
2350        assert_eq!(sig.recent_edit_count, 1);
2351    }
2352
2353    #[test]
2354    fn codex_apply_patch_counts_as_an_edit() {
2355        let request = with_messages(vec![tc("apply_patch"), tr("Success. Updated the file")]);
2356        let sig = ToolSignals::from_request(&request, None);
2357        assert_eq!(sig.edit_count, 1);
2358        assert_eq!(sig.recent_edit_count, 1);
2359    }
2360
2361    fn exec_command(cmd: Value) -> Message {
2362        Message {
2363            role: Role::Assistant,
2364            content: vec![ContentBlock::ToolCall(ToolCall {
2365                id: String::new(),
2366                name: "exec_command".to_string(),
2367                arguments: cmd,
2368            })],
2369        }
2370    }
2371
2372    #[test]
2373    fn codex_exec_command_is_classified() {
2374        // arguments arrive as a JSON string, with the command under `cmd`
2375        let args = json!(r#"{"cmd":"sed -i s/a/b/ src/lib.rs","workdir":"/x"}"#);
2376        let request = with_messages(vec![exec_command(args), tr("ok")]);
2377        assert_eq!(
2378            ToolSignals::from_request(&request, None).recent_edit_count,
2379            1
2380        );
2381    }
2382
2383    #[test]
2384    fn python_write_expressions_need_a_python_command() {
2385        let write = with_messages(vec![
2386            exec_command(json!({"cmd": "python3 - <<'PY'\np.write_text(s)\nPY"})),
2387            tr("ok"),
2388        ]);
2389        assert_eq!(
2390            ToolSignals::from_request(&write, None).recent_write_count,
2391            1
2392        );
2393
2394        let search = with_messages(vec![
2395            exec_command(json!({"cmd": "grep -R '.write(' src"})),
2396            tr("ok"),
2397        ]);
2398        assert_eq!(
2399            ToolSignals::from_request(&search, None).recent_write_count,
2400            0
2401        );
2402    }
2403
2404    #[test]
2405    fn recent_window_size_is_caller_overridable() {
2406        // Same six tool calls (1 edit at the end, 5 writes before).
2407        // With recent_window=3 → recent_writes=2, recent_edits=1.
2408        // With recent_window=6 → recent_writes=5, recent_edits=1 (all calls).
2409        let request = with_messages(vec![
2410            tc("Write"),
2411            tr("ok"),
2412            tc("Write"),
2413            tr("ok"),
2414            tc("Write"),
2415            tr("ok"),
2416            tc("Write"),
2417            tr("ok"),
2418            tc("Write"),
2419            tr("ok"),
2420            tc("Edit"),
2421            tr("ok"),
2422        ]);
2423        let narrow = extract_tool_signals_with_window(&request, 3);
2424        assert_eq!(narrow.recent_write_count, 2);
2425        assert_eq!(narrow.recent_edit_count, 1);
2426
2427        let wide = extract_tool_signals_with_window(&request, 6);
2428        assert_eq!(wide.recent_write_count, 5);
2429        assert_eq!(wide.recent_edit_count, 1);
2430    }
2431
2432    #[test]
2433    fn compaction_marker_sets_compacted() {
2434        // The compaction summary is a user message carrying Claude Code's preamble.
2435        let request = with_messages(vec![
2436            Message::text(
2437                Role::User,
2438                "This session is being continued from a previous conversation that ran out of context.",
2439            ),
2440            bash("ls"),
2441        ]);
2442        assert!(ToolSignals::from_request(&request, None).compacted);
2443    }
2444
2445    #[test]
2446    fn hermes_compaction_headers_set_compacted() {
2447        for header in [
2448            "[CONTEXT COMPACTION \u{2014} REFERENCE ONLY] Earlier turns were compacted into the summary below.",
2449            "[STILL IN PROGRESS \u{2014} this is the active request, restated after the compaction boundary because it was not finished yet. Continue it\u{3b} do not start over.]",
2450        ] {
2451            for role in [Role::User, Role::Assistant] {
2452                let request = with_messages(vec![
2453                    Message::text(role, format!("{header}\nContinue the task.")),
2454                    bash("ls"),
2455                ]);
2456                assert!(ToolSignals::from_request(&request, None).compacted);
2457            }
2458            let request = with_messages(vec![Message::text(
2459                Role::User,
2460                format!("Prior context.\n\n  {header}\nContinue the task."),
2461            )]);
2462            assert!(ToolSignals::from_request(&request, None).compacted);
2463            let request = with_messages(vec![tr(header)]);
2464            assert!(!ToolSignals::from_request(&request, None).compacted);
2465        }
2466    }
2467
2468    #[test]
2469    fn ordinary_compaction_text_stays_uncompacted() {
2470        for text in [
2471            "The task is still in progress after the compaction boundary.",
2472            "Explain [CONTEXT COMPACTION \u{2014} REFERENCE ONLY] in the docs.",
2473            "[STILL IN PROGRESS] Continue the task.",
2474        ] {
2475            let request = with_messages(vec![Message::text(Role::User, text)]);
2476            assert!(!ToolSignals::from_request(&request, None).compacted);
2477        }
2478    }
2479
2480    #[test]
2481    fn codex_compaction_metadata_stays_on_parent_route() {
2482        let mut request = with_messages(vec![bash("ls")]);
2483        request.metadata = Some(Metadata {
2484            is_subagent: true,
2485            agent_kind: Some("compact".to_string()),
2486            ..Default::default()
2487        });
2488        assert!(!ToolSignals::from_request(&request, None).compacted);
2489    }
2490
2491    #[test]
2492    fn no_compaction_marker_stays_uncompacted() {
2493        let request = with_messages(vec![
2494            Message::text(Role::User, "Write a script that parses the log file."),
2495            bash("ls"),
2496        ]);
2497        assert!(!ToolSignals::from_request(&request, None).compacted);
2498    }
2499
2500    #[test]
2501    fn bash_heredoc_counts_as_write() {
2502        // Claude Code's pattern on TB 2.0 — write a scratch file via heredoc.
2503        let request = with_messages(vec![bash("cat > /tmp/test.py <<'EOF'\nprint(1)\nEOF")]);
2504        let sig = ToolSignals::from_request(&request, None);
2505        assert_eq!(
2506            sig.write_count, 1,
2507            "Bash heredoc should bucket into write_count"
2508        );
2509        assert_eq!(sig.edit_count, 0);
2510    }
2511
2512    #[test]
2513    fn bash_sed_inplace_counts_as_edit() {
2514        let request = with_messages(vec![bash("sed -i 's/foo/bar/g' /app/file.py")]);
2515        let sig = ToolSignals::from_request(&request, None);
2516        assert_eq!(
2517            sig.edit_count, 1,
2518            "Bash sed -i should bucket into edit_count"
2519        );
2520        assert_eq!(sig.write_count, 0);
2521    }
2522
2523    #[test]
2524    fn bash_non_mutating_does_not_count() {
2525        // ls, cat, grep — should not increment either counter.
2526        let request = with_messages(vec![bash("ls -la /app"), bash("cat /app/main.py")]);
2527        let sig = ToolSignals::from_request(&request, None);
2528        assert_eq!(sig.write_count, 0);
2529        assert_eq!(sig.edit_count, 0);
2530    }
2531
2532    #[test]
2533    fn tests_passed_detects_pytest_with_failure_block() {
2534        // Mixed pytest run: 2 failed + 5 passed → NOT considered tests_passed.
2535        assert!(!detect_tests_passed(
2536            &[("2 failed, 5 passed in 0.56s".to_string(), false)],
2537            DEFAULT_RECENT_WINDOW
2538        ));
2539    }
2540
2541    #[test]
2542    fn tests_passed_accepts_cargo_clean_summary() {
2543        // Cargo's clean-run summary contains "0 failed" — must not trip the
2544        // failure list (regression: previously substring-matched "failed").
2545        assert!(detect_tests_passed(
2546            &[(
2547                "running 3 tests\ntest result: ok. 3 passed; 0 failed; 0 ignored".to_string(),
2548                false
2549            )],
2550            DEFAULT_RECENT_WINDOW
2551        ));
2552    }
2553
2554    #[test]
2555    fn tests_passed_rejects_cargo_real_failure() {
2556        // Cargo's actual-failure summary: nonzero count before "failed".
2557        assert!(!detect_tests_passed(
2558            &[(
2559                "running 3 tests\ntest result: FAILED. 2 passed; 1 failed; 0 ignored".to_string(),
2560                false
2561            )],
2562            DEFAULT_RECENT_WINDOW
2563        ));
2564    }
2565
2566    #[test]
2567    fn tests_passed_accepts_go_clean_summary() {
2568        // Go test's clean-run "0 errors" must not trip (regression).
2569        assert!(detect_tests_passed(
2570            &[(
2571                "ok  github.com/foo/bar\t0.012s (5 passed, 0 errors)".to_string(),
2572                false
2573            )],
2574            DEFAULT_RECENT_WINDOW
2575        ));
2576    }
2577
2578    #[test]
2579    fn tests_passed_accepts_pytest_zero_errors() {
2580        // Pytest long-form: "0 errors in 0.3s" on a clean run.
2581        assert!(detect_tests_passed(
2582            &[("5 passed, 0 errors in 0.30s".to_string(), false)],
2583            DEFAULT_RECENT_WINDOW
2584        ));
2585    }
2586
2587    #[test]
2588    fn tests_passed_detects_diy_checkmark() {
2589        assert!(detect_tests_passed(
2590            &[("✓ all checks passed".to_string(), false)],
2591            DEFAULT_RECENT_WINDOW
2592        ));
2593    }
2594
2595    #[test]
2596    fn anthropic_bash_heredoc_extracts_command() {
2597        // Anthropic format: tool_use.input is an object, not a JSON string.
2598        let request = with_messages(vec![bash("cat > /tmp/foo.txt << 'EOF'\nhi\nEOF")]);
2599        let sig = ToolSignals::from_request(&request, None);
2600        assert_eq!(
2601            sig.write_count, 1,
2602            "Anthropic Bash heredoc must also be detected"
2603        );
2604    }
2605
2606    #[test]
2607    fn recent_window_falls_back_to_full_history_when_short() {
2608        let request = with_messages(vec![tc("Write")]);
2609        let sig = ToolSignals::from_request(&request, None);
2610        assert_eq!(sig.recent_write_count, 1);
2611        assert_eq!(sig.recent_edit_count, 0);
2612    }
2613
2614    #[test]
2615    fn clean_tool_result_has_zero_severity_and_non_empty_streak() {
2616        let request = with_messages(vec![tr("output ok"), tr("another ok")]);
2617        let sig = ToolSignals::from_request(&request, None);
2618        assert_eq!(sig.severity, 0.0);
2619        assert_eq!(sig.no_error_streak, 2);
2620    }
2621
2622    // ─── asymmetric-signal extensions ────────────────────────────────────
2623
2624    #[test]
2625    fn todowrite_classifies_as_plan() {
2626        assert_eq!(classify_tool_call("TodoWrite", None), ToolSemantic::Plan);
2627        assert_eq!(classify_tool_call("todo_write", None), ToolSemantic::Plan);
2628    }
2629
2630    #[test]
2631    fn codex_update_plan_classifies_as_plan() {
2632        assert_eq!(classify_tool_call("update_plan", None), ToolSemantic::Plan);
2633    }
2634
2635    #[test]
2636    fn codex_shell_command_runs_bash_pattern_match() {
2637        // shell_command + heredoc -> Write.
2638        assert_eq!(
2639            classify_tool_call("shell_command", Some("cat > /app/foo.py <<'eof'\nx=1\neof")),
2640            ToolSemantic::Mutate(MutationKind::Write),
2641        );
2642        // shell_command + read-like inspection -> Read.
2643        assert_eq!(
2644            classify_tool_call("shell_command", Some("ls /app")),
2645            ToolSemantic::Observe,
2646        );
2647        // shell_command without matching patterns -> Unknown.
2648        assert_eq!(
2649            classify_tool_call("shell_command", Some("./run_tests.sh")),
2650            ToolSemantic::Unknown,
2651        );
2652    }
2653
2654    #[test]
2655    fn text_editor_view_is_a_read() {
2656        for name in ["str_replace_based_edit_tool", "text_editor"] {
2657            assert_eq!(
2658                classify_tool_call(name, Some("view")),
2659                ToolSemantic::Observe
2660            );
2661            for command in [
2662                Some("create"),
2663                Some("insert"),
2664                Some("str_replace"),
2665                Some("undo_edit"),
2666                None,
2667            ] {
2668                assert_eq!(
2669                    classify_tool_call(name, command),
2670                    ToolSemantic::Mutate(MutationKind::Edit),
2671                );
2672            }
2673        }
2674
2675        let arguments = [
2676            json!({"command": "view", "path": "/app/main.py"}),
2677            // the Responses wire format sends arguments as a JSON string
2678            json!(r#"{"command":"view","path":"/app/main.py"}"#),
2679        ];
2680        for arguments in arguments {
2681            let call = Message {
2682                role: Role::Assistant,
2683                content: vec![ContentBlock::ToolCall(ToolCall {
2684                    id: String::new(),
2685                    name: "str_replace_based_edit_tool".to_string(),
2686                    arguments,
2687                })],
2688            };
2689            let request = with_messages(vec![call, tr("print('hi')")]);
2690            let sig = ToolSignals::from_request(&request, None);
2691            assert_eq!(sig.read_count, 1);
2692            assert_eq!(sig.recent_read_count, 1);
2693            assert_eq!(sig.edit_count, 0);
2694        }
2695    }
2696
2697    #[test]
2698    fn read_tool_classifies_as_read() {
2699        assert_eq!(classify_tool_call("Read", None), ToolSemantic::Observe);
2700        assert_eq!(classify_tool_call("View", None), ToolSemantic::Observe);
2701    }
2702
2703    #[test]
2704    fn hermes_tool_names_classify() {
2705        // Hermes (NousResearch) file tools route by name.
2706        assert_eq!(
2707            classify_tool_call("write_file", None),
2708            ToolSemantic::Mutate(MutationKind::Write)
2709        );
2710        assert_eq!(
2711            classify_tool_call("patch", None),
2712            ToolSemantic::Mutate(MutationKind::Edit)
2713        );
2714        assert_eq!(classify_tool_call("read_file", None), ToolSemantic::Observe);
2715        assert_eq!(
2716            classify_tool_call("search_files", None),
2717            ToolSemantic::Observe
2718        );
2719        // Hermes runs shell through `terminal`, which carries a `command` arg,
2720        // so its intent comes from the Bash-pattern match like codex's shell_command.
2721        assert_eq!(
2722            classify_tool_call("terminal", Some("sed -i 's/a/b/' /app/x.py")),
2723            ToolSemantic::Mutate(MutationKind::Edit),
2724        );
2725        assert_eq!(
2726            classify_tool_call("terminal", Some("grep foo /app")),
2727            ToolSemantic::Observe,
2728        );
2729        assert_eq!(
2730            classify_tool_call("terminal", Some("./run_tests.sh")),
2731            ToolSemantic::Unknown,
2732        );
2733    }
2734
2735    #[test]
2736    fn bash_read_patterns_classify_as_read() {
2737        let cases = [
2738            "cat /etc/passwd",
2739            "grep foo bar.txt",
2740            "ls /app",
2741            "find . -name '*.py'",
2742        ];
2743        for cmd in cases {
2744            assert_eq!(
2745                classify_tool_call("Bash", Some(cmd)),
2746                ToolSemantic::Observe,
2747                "expected Read for {cmd}"
2748            );
2749        }
2750    }
2751
2752    #[test]
2753    fn codex_inspection_commands_classify_as_read() {
2754        let cases = [
2755            "sed -n '1,80p' src/lib.rs",
2756            "rg -n 'needle' src",
2757            "nl -ba src/lib.rs",
2758            "cat package.json",
2759            "jq '.scripts' package.json",
2760            "git status --short",
2761            "git log --oneline -5",
2762            "git show HEAD:src/lib.rs",
2763            "git branch --show-current",
2764            "git remote -v",
2765            "git config --get remote.origin.url",
2766        ];
2767        for command in cases {
2768            assert_eq!(
2769                classify_tool_call("exec_command", Some(command)),
2770                ToolSemantic::Observe,
2771                "expected Read for {command}"
2772            );
2773        }
2774    }
2775
2776    #[test]
2777    fn quoted_shell_separators_do_not_create_commands() {
2778        for command in ["rg 'foo|rm obsolete.rs'", "rg \"foo; rm obsolete.rs\""] {
2779            assert_eq!(
2780                classify_tool_call("exec_command", Some(command)),
2781                ToolSemantic::Observe,
2782                "quoted text must not be parsed as a command: {command}"
2783            );
2784        }
2785    }
2786
2787    #[test]
2788    fn codex_shell_mutations_classify_as_production() {
2789        let writes = [
2790            "cp source.rs destination.rs",
2791            "mkdir -p src/generated",
2792            "touch src/generated/mod.rs",
2793            "git show HEAD:file.rs > file.rs",
2794            "node <<'node'\nfs.writefilesync('file.js', text)\nnode",
2795        ];
2796        for command in writes {
2797            assert_eq!(
2798                classify_tool_call("exec_command", Some(command)),
2799                ToolSemantic::Mutate(MutationKind::Write),
2800                "expected Write for {command}"
2801            );
2802        }
2803
2804        let edits = [
2805            "mv old.rs new.rs",
2806            "rm obsolete.rs",
2807            "gofmt -w main.go",
2808            "cargo fmt",
2809            "ruff check --fix src",
2810            "perl -0pi -e 's/old/new/' src/lib.rs",
2811            "npx prettier --write src/lib.ts",
2812            "uv run ruff format src",
2813            "git apply fix.patch",
2814        ];
2815        for command in edits {
2816            assert_eq!(
2817                classify_tool_call("exec_command", Some(command)),
2818                ToolSemantic::Mutate(MutationKind::Edit),
2819                "expected Edit for {command}"
2820            );
2821        }
2822    }
2823
2824    #[test]
2825    fn formatter_checks_are_not_edits() {
2826        for command in [
2827            "cargo fmt --check",
2828            "ruff format --check src",
2829            "black --check src",
2830        ] {
2831            assert_ne!(
2832                classify_tool_call("exec_command", Some(command)),
2833                ToolSemantic::Mutate(MutationKind::Edit),
2834                "read-only formatter check must not be Edit: {command}"
2835            );
2836        }
2837    }
2838
2839    #[test]
2840    fn embedded_comparison_is_not_a_shell_write() {
2841        let command = "node <<'node'\nif (index > 0) console.log(index)\nnode";
2842        assert_eq!(
2843            classify_tool_call("exec_command", Some(command)),
2844            ToolSemantic::Unknown
2845        );
2846    }
2847
2848    #[test]
2849    fn bash_write_precedence_over_read() {
2850        // `cat /file > out` contains both `cat /` (read) and ` > ` (write);
2851        // write redirection must win.
2852        assert_eq!(
2853            classify_tool_call("Bash", Some("cat /etc/hosts > /tmp/out")),
2854            ToolSemantic::Mutate(MutationKind::Write),
2855        );
2856    }
2857
2858    #[test]
2859    fn pure_bash_streak_counts_trailing_other() {
2860        // 5 trailing non-classified Bash calls → streak == 5.
2861        let request = with_messages(vec![
2862            bash("make"),
2863            tr("ok"),
2864            bash("./configure"),
2865            tr("ok"),
2866            bash("make install"),
2867            tr("ok"),
2868            bash("./run.sh"),
2869            tr("ok"),
2870            bash("./test"),
2871            tr("ok"),
2872        ]);
2873        let sig = ToolSignals::from_request(&request, None);
2874        assert_eq!(sig.pure_bash_streak, 5);
2875        assert_eq!(sig.write_count, 0);
2876        assert_eq!(sig.read_count, 0);
2877    }
2878
2879    #[test]
2880    fn pure_bash_streak_resets_on_write() {
2881        let request = with_messages(vec![bash("make"), tr("ok"), tc("Write"), tr("ok")]);
2882        let sig = ToolSignals::from_request(&request, None);
2883        assert_eq!(sig.pure_bash_streak, 0);
2884        assert_eq!(sig.write_count, 1);
2885    }
2886
2887    #[test]
2888    fn recent_window_tracks_todowrite_and_read() {
2889        // Final 3 tool calls: TodoWrite, Read, TodoWrite.
2890        let request = with_messages(vec![
2891            bash("make"),
2892            tr("ok"),
2893            tc("TodoWrite"),
2894            tr("ok"),
2895            tc("Read"),
2896            tr("ok"),
2897            tc("TodoWrite"),
2898            tr("ok"),
2899        ]);
2900        let sig = ToolSignals::from_request(&request, None);
2901        assert_eq!(sig.todowrite_count, 2);
2902        assert_eq!(sig.recent_todowrite_count, 2);
2903        assert_eq!(sig.read_count, 1);
2904        assert_eq!(sig.recent_read_count, 1);
2905    }
2906
2907    #[test]
2908    fn configured_tool_semantics_extend_the_builtin_vocabulary() {
2909        let semantics = ToolSemantics {
2910            observe: vec!["KB_search".to_string()],
2911            mutate: vec!["send_payment_request".to_string()],
2912            plan: vec!["create_research_plan".to_string()],
2913            new: vec!["send_message_to_user".to_string()],
2914        };
2915        semantics.validate().expect("valid additive semantics");
2916        let request = with_messages(vec![
2917            tc("Read"),
2918            tc("Write"),
2919            tc("TodoWrite"),
2920            tc("kb_SEARCH"),
2921            tc("send_payment_request"),
2922            tc("create_research_plan"),
2923            tc("send_message_to_user"),
2924            tc("unlisted_tool"),
2925        ]);
2926
2927        let signal = ToolSignals::from_request_with_semantics(&request, None, &semantics);
2928
2929        assert_eq!(signal.read_count, 2);
2930        assert_eq!(signal.write_count, 2);
2931        assert_eq!(signal.todowrite_count, 2);
2932        assert_eq!(signal.new_count, 1);
2933        assert_eq!(signal.recent_new_count, 1);
2934        assert_eq!(signal.pure_bash_streak, 1);
2935    }
2936
2937    #[test]
2938    fn configured_tool_semantics_match_namespaced_and_mcp_tools() {
2939        // The Responses decoder flattens namespaced tools and records the mapping.
2940        let mut request = with_messages(vec![tc("mcp__billing__send_payment_request")]);
2941        request.llm_request.extensions.fields.insert(
2942            TOOL_NAMESPACES_KEY.to_string(),
2943            json!({"mcp__billing__send_payment_request": "mcp__billing"}),
2944        );
2945
2946        // Claude Code sends MCP tools flat, with no namespace mapping.
2947        let claude_request = with_messages(vec![tc("mcp__billing__send_payment_request")]);
2948
2949        for request in [&request, &claude_request] {
2950            for name in ["send_payment_request", "mcp__billing__send_payment_request"] {
2951                let semantics = ToolSemantics {
2952                    mutate: vec![name.to_string()],
2953                    ..Default::default()
2954                };
2955                let signal = ToolSignals::from_request_with_semantics(request, None, &semantics);
2956                assert_eq!(signal.write_count, 1, "{name}");
2957            }
2958        }
2959    }
2960
2961    #[test]
2962    fn configured_tool_semantics_only_fold_ascii_case() {
2963        let semantics = ToolSemantics {
2964            observe: vec!["kb_search".to_string()],
2965            ..Default::default()
2966        };
2967
2968        assert_eq!(
2969            classify_tool_call_with_semantics("KB_SEARCH", None, &semantics),
2970            ToolSemantic::Observe
2971        );
2972        // U+212A lowercases to ASCII `k` under Unicode rules, but custom names
2973        // intentionally ignore only ASCII case.
2974        assert_eq!(
2975            classify_tool_call_with_semantics("KB_SEARCH", None, &semantics),
2976            ToolSemantic::Unknown
2977        );
2978    }
2979
2980    #[test]
2981    fn custom_semantics_preserve_builtin_unicode_lowercasing() {
2982        let semantics = ToolSemantics {
2983            observe: vec!["lookup_customer".to_string()],
2984            ..Default::default()
2985        };
2986
2987        // This matched the built-in `notebookedit` before custom semantics existed.
2988        assert_eq!(
2989            classify_tool_call_with_semantics("notebooKedit", None, &semantics),
2990            ToolSemantic::Mutate(MutationKind::Edit)
2991        );
2992    }
2993
2994    #[test]
2995    fn configured_semantics_never_replace_builtin_classifications() {
2996        let semantics = ToolSemantics {
2997            observe: vec!["lookup_customer".to_string()],
2998            mutate: vec!["send_payment".to_string()],
2999            plan: vec!["create_workflow".to_string()],
3000            new: vec!["send_message".to_string()],
3001        };
3002
3003        for name in WRITE_TOOL_NAMES {
3004            assert_eq!(
3005                classify_tool_call_with_semantics(name, None, &semantics),
3006                ToolSemantic::Mutate(MutationKind::Write),
3007                "write tool {name:?} changed classification"
3008            );
3009        }
3010        for name in EDIT_TOOL_NAMES {
3011            assert_eq!(
3012                classify_tool_call_with_semantics(name, None, &semantics),
3013                ToolSemantic::Mutate(MutationKind::Edit),
3014                "edit tool {name:?} changed classification"
3015            );
3016        }
3017        for name in READ_TOOL_NAMES {
3018            assert_eq!(
3019                classify_tool_call_with_semantics(name, None, &semantics),
3020                ToolSemantic::Observe,
3021                "read tool {name:?} changed classification"
3022            );
3023        }
3024        for name in PLAN_TOOL_NAMES {
3025            assert_eq!(
3026                classify_tool_call_with_semantics(name, None, &semantics),
3027                ToolSemantic::Plan,
3028                "plan tool {name:?} changed classification"
3029            );
3030        }
3031
3032        for (command, expected) in [
3033            ("cat /tmp/input", ToolSemantic::Observe),
3034            (
3035                "cat /tmp/input > /tmp/output",
3036                ToolSemantic::Mutate(MutationKind::Write),
3037            ),
3038            (
3039                "sed -i 's/a/b/' /tmp/file",
3040                ToolSemantic::Mutate(MutationKind::Edit),
3041            ),
3042            ("./run_tests.sh", ToolSemantic::Unknown),
3043        ] {
3044            assert_eq!(
3045                classify_tool_call_with_semantics("BASH", Some(command), &semantics),
3046                expected,
3047                "bash command {command:?} changed classification"
3048            );
3049        }
3050    }
3051
3052    #[test]
3053    fn configured_semantics_score_like_their_builtin_equivalents() {
3054        let semantics = ToolSemantics {
3055            observe: vec!["lookup_customer".to_string()],
3056            mutate: vec!["send_payment".to_string()],
3057            plan: vec!["create_workflow".to_string()],
3058            ..Default::default()
3059        };
3060
3061        for (builtin, configured) in [
3062            ("Read", "lookup_customer"),
3063            ("Write", "send_payment"),
3064            ("TodoWrite", "create_workflow"),
3065        ] {
3066            let messages_before_tool = || {
3067                vec![
3068                    Message::text(Role::User, "start"),
3069                    Message::text(Role::Assistant, "working"),
3070                    Message::text(Role::User, "continue"),
3071                    Message::text(Role::Assistant, "working"),
3072                    Message::text(Role::User, "continue"),
3073                    Message::text(Role::Assistant, "working"),
3074                    Message::text(Role::User, "continue"),
3075                ]
3076            };
3077            let mut builtin_messages = messages_before_tool();
3078            builtin_messages.push(tc(builtin));
3079            let mut configured_messages = messages_before_tool();
3080            configured_messages.push(tc(configured));
3081
3082            let builtin_score = score_signal(&ToolSignals::from_request(
3083                &with_messages(builtin_messages),
3084                None,
3085            ));
3086            let configured_score = score_signal(&ToolSignals::from_request_with_semantics(
3087                &with_messages(configured_messages),
3088                None,
3089                &semantics,
3090            ));
3091
3092            assert_ne!(
3093                builtin_score.score, 0.0,
3094                "the {builtin:?} control must exercise a scoring dimension"
3095            );
3096            assert_eq!(
3097                configured_score, builtin_score,
3098                "configured tool {configured:?} must score exactly like {builtin:?}"
3099            );
3100        }
3101    }
3102
3103    #[test]
3104    fn tool_semantics_reject_duplicates_and_builtin_reclassification() {
3105        let duplicate = ToolSemantics {
3106            observe: vec!["lookup".to_string()],
3107            mutate: vec!["LOOKUP".to_string()],
3108            ..Default::default()
3109        };
3110        assert!(
3111            duplicate
3112                .validate()
3113                .expect_err("duplicate should fail")
3114                .to_string()
3115                .contains("appears in both")
3116        );
3117
3118        let builtin = ToolSemantics {
3119            new: vec!["write_file".to_string()],
3120            ..Default::default()
3121        };
3122        assert!(
3123            builtin
3124                .validate()
3125                .expect_err("built-in should fail")
3126                .to_string()
3127                .contains("built-in semantics")
3128        );
3129
3130        let empty = ToolSemantics {
3131            observe: vec![" \t".to_string()],
3132            ..Default::default()
3133        };
3134        assert!(
3135            empty
3136                .validate()
3137                .expect_err("empty name should fail")
3138                .to_string()
3139                .contains("empty tool name")
3140        );
3141    }
3142}