Spaces:
Running
Running
Download dee/core/golden.py from WINTER4000/syntheogenesis: direct link, hf CLI and curl.
- Browser
- Download file 8.14 kB
-
https://huggingface.co/spaces/WINTER4000/syntheogenesis/resolve/main/dee/core/golden.py
- Command line
-
hf download hf://spaces/WINTER4000/syntheogenesis/dee/core/golden.py
-
curl -L -o golden.py https://huggingface.co/spaces/WINTER4000/syntheogenesis/resolve/main/dee/core/golden.py
8.14 kB
| """Recorded failures, kept as data so they can never happen twice. | |
| Every scenario in ``dee/data/golden/`` is a real incident. Not a hypothetical | |
| β something Turing actually did, to a real user, that was wrong. The mining | |
| rollup (:mod:`dee.core.learn_signal`) finds these; this is where they go so | |
| that finding them once is enough. | |
| The format is deliberately small. A scenario is a user message, a scripted | |
| sequence of model turns, canned tool results, and a handful of assertions. | |
| Adding a newly-observed failure should mean dropping in a JSON file, not | |
| writing a bespoke test β otherwise the loop stops at "we noticed" and never | |
| reaches "it cannot recur". | |
| WHAT THESE CAN AND CANNOT TEST | |
| ------------------------------ | |
| The model is scripted, so these do NOT test whether the real model picks the | |
| right tool β that needs a live call, costs money, and is non-deterministic. | |
| What they DO test is everything downstream of the choice, which is where the | |
| harm actually landed in every recorded incident: | |
| * a fabricated sequence reaching the user (provenance guard), | |
| * a tool that should have been reachable not being called, | |
| * the SEQUENCING of calls, where order changes the meaning β verifying a | |
| construct before the edit lands reports a clean bill of health for a | |
| change that never entered the record (see `order_violation`), | |
| * the engine treating "described it" as "did it", | |
| * a tool description that fails to disambiguate two tools users confuse. | |
| A scripted run can also prove the engine REQUIRES something, not merely that it | |
| permits it: script the misbehaviour and assert the runtime intervened. The | |
| fabrication scenarios do this (the script writes a recalled sequence; the test | |
| asserts the guard fired and sent the run back). Where no such assertion exists, | |
| the scenario proves only that good behaviour is possible β which is why | |
| `04-edit-claimed-without-verifying` says so in its own note rather than reading | |
| as a guarantee it does not provide. | |
| That last one is static, and it is the assertion that actually moves the model | |
| in production: the pCAMBIA mis-pick was fixed by what `fetch_sequence`'s | |
| description says about vectors, not by anything at runtime. | |
| """ | |
| from __future__ import annotations | |
| import json | |
| from pathlib import Path | |
| from typing import Any, Dict, List | |
| GOLDEN_DIR = Path(__file__).resolve().parent.parent / "data" / "golden" | |
| # Every key a scenario may declare. Unknown keys are an error rather than a | |
| # silent no-op: a typo'd assertion that quietly passes is worse than no test, | |
| # because it reads as coverage. | |
| _REQUIRED = {"name", "incident", "first_seen", "user", "model_script"} | |
| # `approve`: answer the confirm gate with a yes and let the run continue. | |
| # Needed by any scenario touching a gated tool (log_outcome, edit_sequence, | |
| # blast_sequence) β without it the run correctly parks and the scenario would | |
| # be asserting the gate rather than the behaviour after it. | |
| _OPTIONAL = {"tool_results", "expect", "workspace", "note", "approve"} | |
| _EXPECT_KEYS = { | |
| "no_unsourced_sequence", # bool β nothing recalled from weights survives | |
| "must_call", # [str] β tools the run has to actually execute | |
| "must_call_in_order", # [str] β ...and in this relative order | |
| "must_not_call", # [str] β tools it must not | |
| "must_confirm", # [str] β tools that must STOP and ask first | |
| "tool_descriptions_say", # {tool: [substrings]} β the disambiguating text | |
| "final_text_contains", # [str] | |
| "final_text_lacks", # [str] | |
| "status", # str | |
| } | |
| class GoldenError(ValueError): | |
| """A malformed scenario. Raised loudly β a broken fixture is not a skip.""" | |
| def load_all() -> List[Dict[str, Any]]: | |
| """Every scenario on disk, validated, sorted by name for stable ordering.""" | |
| if not GOLDEN_DIR.is_dir(): | |
| return [] | |
| out = [] | |
| for path in sorted(GOLDEN_DIR.glob("*.json")): | |
| try: | |
| scenario = json.loads(path.read_text(encoding="utf-8")) | |
| except json.JSONDecodeError as exc: | |
| raise GoldenError(f"{path.name}: not valid JSON ({exc})") from exc | |
| validate(scenario, source=path.name) | |
| out.append(scenario) | |
| return out | |
| def validate(scenario: Dict[str, Any], source: str = "<memory>") -> None: | |
| keys = set(scenario) | |
| missing = _REQUIRED - keys | |
| if missing: | |
| raise GoldenError(f"{source}: missing {sorted(missing)}") | |
| unknown = keys - _REQUIRED - _OPTIONAL | |
| if unknown: | |
| raise GoldenError(f"{source}: unknown key(s) {sorted(unknown)}") | |
| bad_expect = set(scenario.get("expect") or {}) - _EXPECT_KEYS | |
| if bad_expect: | |
| raise GoldenError(f"{source}: unknown expect key(s) {sorted(bad_expect)}") | |
| if not isinstance(scenario.get("model_script"), list) or not scenario["model_script"]: | |
| raise GoldenError(f"{source}: model_script must be a non-empty list") | |
| for i, turn in enumerate(scenario["model_script"]): | |
| if not isinstance(turn, dict) or not ({"text", "tool"} & set(turn)): | |
| raise GoldenError(f"{source}: turn {i} needs 'text' or 'tool'") | |
| order = (scenario.get("expect") or {}).get("must_call_in_order") | |
| if order is not None: | |
| if not isinstance(order, list) or not all(isinstance(n, str) for n in order): | |
| raise GoldenError(f"{source}: must_call_in_order must be a list of names") | |
| # One name is not an order, it is membership β and `must_call` already | |
| # says that, more clearly. Rejecting it stops a scenario from looking | |
| # like it checks sequencing when it checks nothing of the kind. | |
| if len(order) < 2: | |
| raise GoldenError( | |
| f"{source}: must_call_in_order needs 2+ names; use must_call") | |
| # An incident with no assertions is a note, not a regression test. | |
| if not (scenario.get("expect") or {}): | |
| raise GoldenError(f"{source}: no expectations β this asserts nothing") | |
| def order_violation(expected: List[str], actual: List[str]) -> str: | |
| """Why `actual` breaks the required order, or "" if it doesn't. | |
| WHY ORDER NEEDS ITS OWN ASSERTION. `must_call` is membership, so it cannot | |
| tell "edited, then verified" from "verified, then edited" β and in this | |
| domain those are opposite outcomes. Re-mapping a construct BEFORE the edit | |
| lands reports a clean bill of health for a change that never entered the | |
| record, which is exactly the failure `_bind_target`'s edit_sequence branch | |
| exists to prevent. A scenario asserting only that both tools ran passes | |
| either way. | |
| SUBSEQUENCE, not adjacency: other calls may interleave. "fetch, then edit" | |
| must still hold when the agent folds a structure in between, because that | |
| is ordinary good behaviour and a stricter rule would punish it. | |
| A pure function so the check itself is testable. An assertion helper that | |
| can never fail is the failure mode this corpus is most prone to β see | |
| `validate`'s rejection of a scenario that asserts nothing. | |
| """ | |
| missing = [n for n in expected if n not in actual] | |
| if missing: | |
| return "never ran: " + ", ".join(missing) | |
| remaining = list(expected) | |
| for name in actual: | |
| if remaining and name == remaining[0]: | |
| remaining.pop(0) | |
| if remaining: | |
| return (f"ran out of order β expected {' β '.join(expected)}, " | |
| f"got {' β '.join(actual)}") | |
| return "" | |
| def to_messages(scenario: Dict[str, Any]) -> List[Dict[str, Any]]: | |
| """Turn a scenario's script into the message shapes `llm.call` returns.""" | |
| msgs = [] | |
| for i, turn in enumerate(scenario["model_script"]): | |
| if turn.get("tool"): | |
| msgs.append({ | |
| "content": turn.get("text"), | |
| "tool_calls": [{ | |
| "id": turn.get("id") or f"g{i}", | |
| "function": {"name": turn["tool"], | |
| "arguments": json.dumps(turn.get("args") or {})}, | |
| }], | |
| }) | |
| else: | |
| msgs.append({"content": turn.get("text") or "", "tool_calls": []}) | |
| return msgs | |