syntheogenesis / dee /core /golden.py
github-actions[bot]
Deploy 0ab3126
ff6973d
Raw History Blame Contribute Delete
8.14 kB
"""Recorded failures, kept as data so they can never happen twice.
Every scenario in ``dee/data/golden/`` is a real incident. Not a hypothetical
β€” something Turing actually did, to a real user, that was wrong. The mining
rollup (:mod:`dee.core.learn_signal`) finds these; this is where they go so
that finding them once is enough.
The format is deliberately small. A scenario is a user message, a scripted
sequence of model turns, canned tool results, and a handful of assertions.
Adding a newly-observed failure should mean dropping in a JSON file, not
writing a bespoke test β€” otherwise the loop stops at "we noticed" and never
reaches "it cannot recur".
WHAT THESE CAN AND CANNOT TEST
------------------------------
The model is scripted, so these do NOT test whether the real model picks the
right tool β€” that needs a live call, costs money, and is non-deterministic.
What they DO test is everything downstream of the choice, which is where the
harm actually landed in every recorded incident:
* a fabricated sequence reaching the user (provenance guard),
* a tool that should have been reachable not being called,
* the SEQUENCING of calls, where order changes the meaning β€” verifying a
construct before the edit lands reports a clean bill of health for a
change that never entered the record (see `order_violation`),
* the engine treating "described it" as "did it",
* a tool description that fails to disambiguate two tools users confuse.
A scripted run can also prove the engine REQUIRES something, not merely that it
permits it: script the misbehaviour and assert the runtime intervened. The
fabrication scenarios do this (the script writes a recalled sequence; the test
asserts the guard fired and sent the run back). Where no such assertion exists,
the scenario proves only that good behaviour is possible β€” which is why
`04-edit-claimed-without-verifying` says so in its own note rather than reading
as a guarantee it does not provide.
That last one is static, and it is the assertion that actually moves the model
in production: the pCAMBIA mis-pick was fixed by what `fetch_sequence`'s
description says about vectors, not by anything at runtime.
"""
from __future__ import annotations
import json
from pathlib import Path
from typing import Any, Dict, List
GOLDEN_DIR = Path(__file__).resolve().parent.parent / "data" / "golden"
# Every key a scenario may declare. Unknown keys are an error rather than a
# silent no-op: a typo'd assertion that quietly passes is worse than no test,
# because it reads as coverage.
_REQUIRED = {"name", "incident", "first_seen", "user", "model_script"}
# `approve`: answer the confirm gate with a yes and let the run continue.
# Needed by any scenario touching a gated tool (log_outcome, edit_sequence,
# blast_sequence) β€” without it the run correctly parks and the scenario would
# be asserting the gate rather than the behaviour after it.
_OPTIONAL = {"tool_results", "expect", "workspace", "note", "approve"}
_EXPECT_KEYS = {
"no_unsourced_sequence", # bool β€” nothing recalled from weights survives
"must_call", # [str] β€” tools the run has to actually execute
"must_call_in_order", # [str] β€” ...and in this relative order
"must_not_call", # [str] β€” tools it must not
"must_confirm", # [str] β€” tools that must STOP and ask first
"tool_descriptions_say", # {tool: [substrings]} β€” the disambiguating text
"final_text_contains", # [str]
"final_text_lacks", # [str]
"status", # str
}
class GoldenError(ValueError):
"""A malformed scenario. Raised loudly β€” a broken fixture is not a skip."""
def load_all() -> List[Dict[str, Any]]:
"""Every scenario on disk, validated, sorted by name for stable ordering."""
if not GOLDEN_DIR.is_dir():
return []
out = []
for path in sorted(GOLDEN_DIR.glob("*.json")):
try:
scenario = json.loads(path.read_text(encoding="utf-8"))
except json.JSONDecodeError as exc:
raise GoldenError(f"{path.name}: not valid JSON ({exc})") from exc
validate(scenario, source=path.name)
out.append(scenario)
return out
def validate(scenario: Dict[str, Any], source: str = "<memory>") -> None:
keys = set(scenario)
missing = _REQUIRED - keys
if missing:
raise GoldenError(f"{source}: missing {sorted(missing)}")
unknown = keys - _REQUIRED - _OPTIONAL
if unknown:
raise GoldenError(f"{source}: unknown key(s) {sorted(unknown)}")
bad_expect = set(scenario.get("expect") or {}) - _EXPECT_KEYS
if bad_expect:
raise GoldenError(f"{source}: unknown expect key(s) {sorted(bad_expect)}")
if not isinstance(scenario.get("model_script"), list) or not scenario["model_script"]:
raise GoldenError(f"{source}: model_script must be a non-empty list")
for i, turn in enumerate(scenario["model_script"]):
if not isinstance(turn, dict) or not ({"text", "tool"} & set(turn)):
raise GoldenError(f"{source}: turn {i} needs 'text' or 'tool'")
order = (scenario.get("expect") or {}).get("must_call_in_order")
if order is not None:
if not isinstance(order, list) or not all(isinstance(n, str) for n in order):
raise GoldenError(f"{source}: must_call_in_order must be a list of names")
# One name is not an order, it is membership β€” and `must_call` already
# says that, more clearly. Rejecting it stops a scenario from looking
# like it checks sequencing when it checks nothing of the kind.
if len(order) < 2:
raise GoldenError(
f"{source}: must_call_in_order needs 2+ names; use must_call")
# An incident with no assertions is a note, not a regression test.
if not (scenario.get("expect") or {}):
raise GoldenError(f"{source}: no expectations β€” this asserts nothing")
def order_violation(expected: List[str], actual: List[str]) -> str:
"""Why `actual` breaks the required order, or "" if it doesn't.
WHY ORDER NEEDS ITS OWN ASSERTION. `must_call` is membership, so it cannot
tell "edited, then verified" from "verified, then edited" β€” and in this
domain those are opposite outcomes. Re-mapping a construct BEFORE the edit
lands reports a clean bill of health for a change that never entered the
record, which is exactly the failure `_bind_target`'s edit_sequence branch
exists to prevent. A scenario asserting only that both tools ran passes
either way.
SUBSEQUENCE, not adjacency: other calls may interleave. "fetch, then edit"
must still hold when the agent folds a structure in between, because that
is ordinary good behaviour and a stricter rule would punish it.
A pure function so the check itself is testable. An assertion helper that
can never fail is the failure mode this corpus is most prone to β€” see
`validate`'s rejection of a scenario that asserts nothing.
"""
missing = [n for n in expected if n not in actual]
if missing:
return "never ran: " + ", ".join(missing)
remaining = list(expected)
for name in actual:
if remaining and name == remaining[0]:
remaining.pop(0)
if remaining:
return (f"ran out of order β€” expected {' β†’ '.join(expected)}, "
f"got {' β†’ '.join(actual)}")
return ""
def to_messages(scenario: Dict[str, Any]) -> List[Dict[str, Any]]:
"""Turn a scenario's script into the message shapes `llm.call` returns."""
msgs = []
for i, turn in enumerate(scenario["model_script"]):
if turn.get("tool"):
msgs.append({
"content": turn.get("text"),
"tool_calls": [{
"id": turn.get("id") or f"g{i}",
"function": {"name": turn["tool"],
"arguments": json.dumps(turn.get("args") or {})},
}],
})
else:
msgs.append({"content": turn.get("text") or "", "tool_calls": []})
return msgs