"""Recorded failures, kept as data so they can never happen twice. Every scenario in ``dee/data/golden/`` is a real incident. Not a hypothetical — something Turing actually did, to a real user, that was wrong. The mining rollup (:mod:`dee.core.learn_signal`) finds these; this is where they go so that finding them once is enough. The format is deliberately small. A scenario is a user message, a scripted sequence of model turns, canned tool results, and a handful of assertions. Adding a newly-observed failure should mean dropping in a JSON file, not writing a bespoke test — otherwise the loop stops at "we noticed" and never reaches "it cannot recur". WHAT THESE CAN AND CANNOT TEST ------------------------------ The model is scripted, so these do NOT test whether the real model picks the right tool — that needs a live call, costs money, and is non-deterministic. What they DO test is everything downstream of the choice, which is where the harm actually landed in every recorded incident: * a fabricated sequence reaching the user (provenance guard), * a tool that should have been reachable not being called, * the SEQUENCING of calls, where order changes the meaning — verifying a construct before the edit lands reports a clean bill of health for a change that never entered the record (see `order_violation`), * the engine treating "described it" as "did it", * a tool description that fails to disambiguate two tools users confuse. A scripted run can also prove the engine REQUIRES something, not merely that it permits it: script the misbehaviour and assert the runtime intervened. The fabrication scenarios do this (the script writes a recalled sequence; the test asserts the guard fired and sent the run back). Where no such assertion exists, the scenario proves only that good behaviour is possible — which is why `04-edit-claimed-without-verifying` says so in its own note rather than reading as a guarantee it does not provide. That last one is static, and it is the assertion that actually moves the model in production: the pCAMBIA mis-pick was fixed by what `fetch_sequence`'s description says about vectors, not by anything at runtime. """ from __future__ import annotations import json from pathlib import Path from typing import Any, Dict, List GOLDEN_DIR = Path(__file__).resolve().parent.parent / "data" / "golden" # Every key a scenario may declare. Unknown keys are an error rather than a # silent no-op: a typo'd assertion that quietly passes is worse than no test, # because it reads as coverage. _REQUIRED = {"name", "incident", "first_seen", "user", "model_script"} # `approve`: answer the confirm gate with a yes and let the run continue. # Needed by any scenario touching a gated tool (log_outcome, edit_sequence, # blast_sequence) — without it the run correctly parks and the scenario would # be asserting the gate rather than the behaviour after it. _OPTIONAL = {"tool_results", "expect", "workspace", "note", "approve"} _EXPECT_KEYS = { "no_unsourced_sequence", # bool — nothing recalled from weights survives "must_call", # [str] — tools the run has to actually execute "must_call_in_order", # [str] — ...and in this relative order "must_not_call", # [str] — tools it must not "must_confirm", # [str] — tools that must STOP and ask first "tool_descriptions_say", # {tool: [substrings]} — the disambiguating text "final_text_contains", # [str] "final_text_lacks", # [str] "status", # str } class GoldenError(ValueError): """A malformed scenario. Raised loudly — a broken fixture is not a skip.""" def load_all() -> List[Dict[str, Any]]: """Every scenario on disk, validated, sorted by name for stable ordering.""" if not GOLDEN_DIR.is_dir(): return [] out = [] for path in sorted(GOLDEN_DIR.glob("*.json")): try: scenario = json.loads(path.read_text(encoding="utf-8")) except json.JSONDecodeError as exc: raise GoldenError(f"{path.name}: not valid JSON ({exc})") from exc validate(scenario, source=path.name) out.append(scenario) return out def validate(scenario: Dict[str, Any], source: str = "") -> None: keys = set(scenario) missing = _REQUIRED - keys if missing: raise GoldenError(f"{source}: missing {sorted(missing)}") unknown = keys - _REQUIRED - _OPTIONAL if unknown: raise GoldenError(f"{source}: unknown key(s) {sorted(unknown)}") bad_expect = set(scenario.get("expect") or {}) - _EXPECT_KEYS if bad_expect: raise GoldenError(f"{source}: unknown expect key(s) {sorted(bad_expect)}") if not isinstance(scenario.get("model_script"), list) or not scenario["model_script"]: raise GoldenError(f"{source}: model_script must be a non-empty list") for i, turn in enumerate(scenario["model_script"]): if not isinstance(turn, dict) or not ({"text", "tool"} & set(turn)): raise GoldenError(f"{source}: turn {i} needs 'text' or 'tool'") order = (scenario.get("expect") or {}).get("must_call_in_order") if order is not None: if not isinstance(order, list) or not all(isinstance(n, str) for n in order): raise GoldenError(f"{source}: must_call_in_order must be a list of names") # One name is not an order, it is membership — and `must_call` already # says that, more clearly. Rejecting it stops a scenario from looking # like it checks sequencing when it checks nothing of the kind. if len(order) < 2: raise GoldenError( f"{source}: must_call_in_order needs 2+ names; use must_call") # An incident with no assertions is a note, not a regression test. if not (scenario.get("expect") or {}): raise GoldenError(f"{source}: no expectations — this asserts nothing") def order_violation(expected: List[str], actual: List[str]) -> str: """Why `actual` breaks the required order, or "" if it doesn't. WHY ORDER NEEDS ITS OWN ASSERTION. `must_call` is membership, so it cannot tell "edited, then verified" from "verified, then edited" — and in this domain those are opposite outcomes. Re-mapping a construct BEFORE the edit lands reports a clean bill of health for a change that never entered the record, which is exactly the failure `_bind_target`'s edit_sequence branch exists to prevent. A scenario asserting only that both tools ran passes either way. SUBSEQUENCE, not adjacency: other calls may interleave. "fetch, then edit" must still hold when the agent folds a structure in between, because that is ordinary good behaviour and a stricter rule would punish it. A pure function so the check itself is testable. An assertion helper that can never fail is the failure mode this corpus is most prone to — see `validate`'s rejection of a scenario that asserts nothing. """ missing = [n for n in expected if n not in actual] if missing: return "never ran: " + ", ".join(missing) remaining = list(expected) for name in actual: if remaining and name == remaining[0]: remaining.pop(0) if remaining: return (f"ran out of order — expected {' → '.join(expected)}, " f"got {' → '.join(actual)}") return "" def to_messages(scenario: Dict[str, Any]) -> List[Dict[str, Any]]: """Turn a scenario's script into the message shapes `llm.call` returns.""" msgs = [] for i, turn in enumerate(scenario["model_script"]): if turn.get("tool"): msgs.append({ "content": turn.get("text"), "tool_calls": [{ "id": turn.get("id") or f"g{i}", "function": {"name": turn["tool"], "arguments": json.dumps(turn.get("args") or {})}, }], }) else: msgs.append({"content": turn.get("text") or "", "tool_calls": []}) return msgs