395 lines
19 KiB
Python
Executable file
395 lines
19 KiB
Python
Executable file
#!/usr/bin/env python3
|
|
"""
|
|
test_run.py -- regression tests for the decision log in run.py.
|
|
|
|
The log is the only record of what the model was asked and what it answered, so
|
|
two things must hold or every later analysis is wrong:
|
|
|
|
* a model decision's row carries that decision's answers;
|
|
* a decision that asked NOTHING carries none -- no inheritance from the
|
|
previous step, which is what `client.last = None` before each decide() is
|
|
for, and which fails silently if it is ever dropped.
|
|
|
|
Runs offline: a stub client stands in for JevClient and a fake sts2 module
|
|
stands in for the game, so no network and no running game are needed.
|
|
|
|
Run: python3 test_run.py
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import contextlib
|
|
import io
|
|
import itertools
|
|
import json
|
|
import pathlib
|
|
import sys
|
|
import tempfile
|
|
import types
|
|
from unittest.mock import patch
|
|
|
|
import brain
|
|
import jev
|
|
import run
|
|
from jev import ChoiceAnswer, JevResponse, NoulAnswer, ScoreAnswer
|
|
|
|
PASS = 0
|
|
FAIL = 0
|
|
|
|
|
|
def check(label: str, got, want) -> None:
|
|
global PASS, FAIL
|
|
if got == want:
|
|
PASS += 1
|
|
print(f" ok {label}: {got!r}")
|
|
else:
|
|
FAIL += 1
|
|
print(f" FAIL {label}: got {got!r}, want {want!r}")
|
|
|
|
|
|
class StubClient:
|
|
"""Answers every question locally, so the log can be tested without a model."""
|
|
|
|
model = "stub"
|
|
|
|
def __init__(self, noul: float = 0.9):
|
|
self.noul = noul
|
|
self.calls = 0
|
|
|
|
def __repr__(self) -> str:
|
|
return "StubClient()"
|
|
|
|
def ask(self, state, questions, model=None):
|
|
self.calls += 1
|
|
answers = {}
|
|
for qid, q in questions.items():
|
|
if q.get("type") == "choice":
|
|
options = list((q.get("criteria") or {}).keys())
|
|
answers[qid] = ChoiceAnswer(
|
|
choice=options[0] if options else "",
|
|
probabilities={o: (0.9 if o == options[0] else 0.05)
|
|
for o in options},
|
|
confidence=0.9,
|
|
)
|
|
elif q.get("type") == "noul":
|
|
answers[qid] = NoulAnswer(noul=self.noul)
|
|
else:
|
|
answers[qid] = ScoreAnswer(score=1.0, confidence=0.9)
|
|
return JevResponse(answers=answers, model="stub", input_tokens=0,
|
|
output_tokens=0, latency_s=0.0)
|
|
|
|
|
|
class FakeSts2:
|
|
"""A fixed sequence of states, then the last one forever."""
|
|
|
|
class Sts2Error(RuntimeError):
|
|
pass
|
|
|
|
BASE = "offline://game"
|
|
|
|
def __init__(self, states, *, action_ok=True, action_error=None):
|
|
self.states = states
|
|
self.i = 0
|
|
self.actions = []
|
|
self.action_ok = action_ok
|
|
self.action_error = action_error
|
|
|
|
def is_up(self) -> bool:
|
|
return True
|
|
|
|
def state(self) -> dict:
|
|
state = self.states[min(self.i, len(self.states) - 1)]
|
|
self.i += 1
|
|
if isinstance(state, Exception):
|
|
raise state
|
|
return state
|
|
|
|
def act(self, *a, **k):
|
|
self.actions.append((a, k))
|
|
if self.action_error:
|
|
raise self.action_error
|
|
ok = (self.action_ok[min(len(self.actions) - 1, len(self.action_ok) - 1)]
|
|
if isinstance(self.action_ok, list) else self.action_ok)
|
|
return types.SimpleNamespace(ok=ok, message="rejected" if not ok else "")
|
|
|
|
|
|
def invoke(fake, *flags, client=None, clock=None, decide=None):
|
|
"""Run the real loop with isolated files, no delays, and no live services."""
|
|
callbacks = []
|
|
with tempfile.TemporaryDirectory() as directory, contextlib.ExitStack() as stack:
|
|
capdir = pathlib.Path(directory) / "capture"
|
|
stack.enter_context(patch.object(run, "sts2", fake))
|
|
stack.enter_context(patch.object(run, "JevClient", return_value=client or StubClient()))
|
|
stack.enter_context(patch.object(run, "history_snapshot", return_value=set()))
|
|
stack.enter_context(patch.object(run, "load_deck", return_value=None))
|
|
save = stack.enter_context(patch.object(run, "save_deck"))
|
|
stack.enter_context(patch.object(run.atexit, "register", side_effect=callbacks.append))
|
|
stack.enter_context(patch.object(run.time, "sleep"))
|
|
stack.enter_context(patch.object(sys, "argv", ["run.py", "--steps", "10", "--pause", "0",
|
|
"--capture-dir", str(capdir), *flags]))
|
|
if clock is not None:
|
|
stack.enter_context(patch.object(run.time, "monotonic", side_effect=clock))
|
|
if decide is not None:
|
|
stack.enter_context(patch.object(brain, "decide", side_effect=decide))
|
|
output = stack.enter_context(contextlib.redirect_stdout(io.StringIO()))
|
|
rc = run.main()
|
|
for callback in callbacks:
|
|
callback()
|
|
def rows(name):
|
|
path = capdir / name
|
|
return [json.loads(line) for line in path.read_text().splitlines()] if path.exists() else []
|
|
return types.SimpleNamespace(rc=rc, rows=rows("decisions.jsonl"),
|
|
sessions=rows("sessions.jsonl"), saved=save.call_count,
|
|
output=output.getvalue())
|
|
|
|
|
|
print("=== 1. answer_record shapes (what the log stores) ===")
|
|
check("noul", jev.answer_record(NoulAnswer(0.72)),
|
|
{"kind": "noul", "noul": 0.72, "yes": True, "gated": True})
|
|
check("noul gate outcome is recorded", jev.answer_record(NoulAnswer(0.52))["gated"], False)
|
|
choice = jev.answer_record(ChoiceAnswer("a", {"a": 0.7, "b": 0.2}, 0.5))
|
|
check("choice margin", choice["margin"], 0.5)
|
|
check("choice gated", choice["gated"], True)
|
|
check("choice probabilities survive", choice["probabilities"], {"a": 0.7, "b": 0.2})
|
|
check("score", jev.answer_record(ScoreAnswer(1.4, {}, {}, 0.9))["kind"], "score")
|
|
|
|
print()
|
|
print("=== 2. a decision that asked nothing logs jev: null ===")
|
|
stub = run.RecordingClient(StubClient())
|
|
code_decision = brain.Decision("end_turn", {}, "no playable cards", "code")
|
|
row = run.decision_record(1, "monster", {"act": 1, "floor": 1}, code_decision,
|
|
None, stub)
|
|
check("jev is null", row["jev"], None)
|
|
check("the session id travels with the row", row["session"], run.SESSION_ID)
|
|
check("the action is still recorded", row["action"], "end_turn")
|
|
|
|
print()
|
|
print("=== 3. end to end: a model row keeps its answers, the next row does not ===")
|
|
# Step 1 is a card_reward, which always asks the model. Step 2 is the main menu,
|
|
# which never does. If the reset is dropped, row 2 inherits row 1's answers --
|
|
# the failure this test exists to catch.
|
|
card_reward = {
|
|
"state_type": "card_reward",
|
|
"card_reward": {"can_skip": True, "cards": [
|
|
{"id": "A", "name": "Stomp", "type": "Attack", "cost": "3",
|
|
"description": "Deal 12 damage.", "rarity": "Uncommon",
|
|
"is_upgraded": False, "index": 0},
|
|
{"id": "B", "name": "Inflame", "type": "Power", "cost": "1",
|
|
"description": "Gain 2 Strength.", "rarity": "Uncommon",
|
|
"is_upgraded": False, "index": 1},
|
|
]},
|
|
"run": {"act": 1, "floor": 2, "ascension": 0},
|
|
"player": {"character": "The Ironclad", "hp": 70, "max_hp": 80, "gold": 99},
|
|
}
|
|
menu = {"state_type": "menu", "menu_screen": "main",
|
|
"options": ["singleplayer", "quit"], "run": None}
|
|
|
|
# Preflight consumes the menu; the two loop steps see the reward and menu.
|
|
result = invoke(FakeSts2([menu, card_reward, menu]), "--steps", "2", "--dry-run")
|
|
check("main() completed", result.rc, 0)
|
|
rows = [r for r in result.rows if r["event"] == "decide"]
|
|
check("one row per decided action", len(rows), 2)
|
|
check("row 1 came from the model", rows[0]["source"], "jev")
|
|
check("...and carries its answers", sorted(rows[0]["jev"]["answers"]),
|
|
["good_card0", "good_card1", "skip_all"])
|
|
check("...with its value and gate outcome, not just a yes/no",
|
|
(rows[0]["jev"]["answers"]["good_card0"]["noul"],
|
|
rows[0]["jev"]["answers"]["good_card0"]["gated"]), (0.9, True))
|
|
check("row 2 asked nothing", rows[1]["jev"], None)
|
|
check("...and is a code decision", rows[1]["source"], "code")
|
|
check("every row carries the session", {r["session"] for r in rows}, {run.SESSION_ID})
|
|
|
|
print()
|
|
print("=== 4. the session row maps to an outcome (join half) ===")
|
|
with tempfile.TemporaryDirectory() as directory, patch.object(run, "HISTORY_DIR", pathlib.Path(directory)):
|
|
record = {"win": False, "killed_by_encounter": "ENCOUNTER.TEST", "seed": "fixture",
|
|
"players": [{"deck": ["Strike"]}], "map_point_history": [[{}, {}]], "run_time": 42}
|
|
(run.HISTORY_DIR / "fixture.run").write_text(json.dumps(record))
|
|
check("history records visible", run.history_snapshot(), {"fixture.run"})
|
|
outcome = run.run_outcome("fixture.run")
|
|
check("outcome names the file", outcome["file"], "fixture.run")
|
|
check("outcome has the killer", outcome["killed_by"], "TEST")
|
|
check("outcome has the deck size", outcome["deck_size"], 1)
|
|
|
|
print()
|
|
print("=== 5. dry-run never sends game actions or updates the deck cache ===")
|
|
game_over = {"state_type": "game_over"}
|
|
combat = {
|
|
"state_type": "monster", "run": {"act": 1, "floor": 1},
|
|
"battle": {"round": 1, "turn": "player", "is_play_phase": True,
|
|
"enemies": [{"entity_id": "E_0", "name": "E", "hp": 100, "max_hp": 100,
|
|
"block": 0, "status": [], "intents": []}]},
|
|
"player": {"hp": 80, "max_hp": 80, "energy": 1, "block": 0, "status": [], "potions": [],
|
|
"hand": [{"index": 0, "name": "Strike", "type": "Attack", "cost": "1",
|
|
"description": "Deal 6 damage.", "target_type": "AnyEnemy", "can_play": True}]},
|
|
}
|
|
for states, extra, label in [
|
|
([game_over, menu], [], "parked game-over preflight"),
|
|
([menu, game_over], ["--stop-on-run-end"], "run-end dismissal"),
|
|
([menu, game_over], [], "ordinary game-over decision"),
|
|
([menu, combat], [], "combat action and deck snapshot"),
|
|
]:
|
|
fake = FakeSts2(states)
|
|
result = invoke(fake, "--dry-run", "--no-jev", "--steps", "1", *extra)
|
|
check(label + ": no POST", fake.actions, [])
|
|
check(label + ": completed preview", result.rc, 0)
|
|
check(label + ": no deck write", result.saved, 0)
|
|
|
|
print()
|
|
print("=== 6. failed sessions do not return success ===")
|
|
for fake, flags, expected, label in [
|
|
(FakeSts2([game_over], action_ok=False), [], 2, "preflight rejection"),
|
|
(FakeSts2([menu, game_over], action_ok=False), ["--stop-on-run-end"], 1, "run-end rejection"),
|
|
(FakeSts2([menu, game_over], action_error=FakeSts2.Sts2Error("timeout")), ["--stop-on-run-end"], 1, "run-end timeout"),
|
|
(FakeSts2([menu, {"state_type": "not-supported"}]), [], 1, "no decision"),
|
|
(FakeSts2([menu, menu], action_error=FakeSts2.Sts2Error("timeout")), [], 1, "action transport failure"),
|
|
(FakeSts2([menu, menu], action_ok=False), ["--steps", "6"], 1, "rejection budget"),
|
|
(FakeSts2([menu, menu], action_ok=False), ["--steps", "1"], 1, "last action rejected at step limit"),
|
|
(FakeSts2([menu, FakeSts2.Sts2Error("read timeout")]), [], 1, "state transport failure"),
|
|
(FakeSts2([menu, combat]), ["--stop-on-run-end", "--steps", "1"], 4, "unfinished run at step limit"),
|
|
]:
|
|
result = invoke(fake, "--no-jev", *flags)
|
|
check(label, result.rc, expected)
|
|
if result.sessions:
|
|
check(label + ": session carries status", result.sessions[0].get("exit_code"), expected)
|
|
check(label + ": session has a reason", bool(result.sessions[0].get("stop_reason")), True)
|
|
|
|
result = invoke(FakeSts2([menu, {"state_type": "overlay"}]), "--no-jev", "--stuck-seconds", "1.5",
|
|
clock=itertools.count().__next__)
|
|
check("unchanged-state timeout is a failure", result.rc, 1)
|
|
check("unchanged-state timeout is recorded", result.sessions[0].get("stop_reason"), "stuck")
|
|
result = invoke(FakeSts2([menu, menu]), "--no-jev", decide=ValueError("bad observation"))
|
|
check("unexpected policy error stops instead of hiding the bug", result.rc, 1)
|
|
|
|
print()
|
|
print("=== 7. combat outages use the runner's fallback and failure budget ===")
|
|
class OutageClient(StubClient):
|
|
def ask(self, *args, **kwargs):
|
|
self.calls += 1
|
|
raise jev.JevError("offline injected outage")
|
|
|
|
client = OutageClient()
|
|
fake = FakeSts2([menu, combat, {"state_type": "overlay"}, combat])
|
|
result = invoke(fake, "--max-jev-errors", "2", client=client)
|
|
check("abort after two model failures, even with a code-only wait between", result.rc, 3)
|
|
check("the model was tried twice", client.calls, 2)
|
|
check("first failed call gets one heuristic combat action", len(fake.actions), 1)
|
|
decisions = [r for r in result.rows if r["event"] == "decide"]
|
|
check("combat fallback is recorded", decisions[0]["source"] if decisions else None, "fallback")
|
|
check("combat fallback preserves its error", "JevError" in (decisions[0]["error"] or "") if decisions else False, True)
|
|
check("terminal model failure is recorded", len([r for r in result.rows if r["event"] == "model_error"]), 2)
|
|
check("model abort carries a session stop reason", result.sessions[0].get("stop_reason"), "model_failures")
|
|
|
|
class RecoveringClient(StubClient):
|
|
def ask(self, *args, **kwargs):
|
|
if self.calls == 1:
|
|
return super().ask(*args, **kwargs)
|
|
self.calls += 1
|
|
raise jev.JevError("offline injected outage")
|
|
|
|
client = RecoveringClient()
|
|
result = invoke(FakeSts2([menu, combat]), "--max-jev-errors", "2", client=client)
|
|
check("successful model call resets the failure budget", client.calls, 4)
|
|
check("two failures after recovery abort", result.rc, 3)
|
|
|
|
print()
|
|
print("=== 8. policy memory follows action results and fresh observations ===")
|
|
# Exercise the real policy and runner together. No global reset between sessions.
|
|
grid = {"state_type": "card_select", "card_select": {
|
|
"screen_type": "select", "prompt": "Choose 2 cards to Remove.",
|
|
"cards": [{"index": 0, "name": "Strike"}, {"index": 1, "name": "Defend"}],
|
|
"can_confirm": False, "preview_showing": False}}
|
|
confirmable = dict(grid, card_select=dict(grid["card_select"], can_confirm=True))
|
|
# One rejected toggle, two accepted toggles, a delayed confirm, then transition.
|
|
fake = FakeSts2([menu, grid, grid, grid, confirmable, confirmable,
|
|
{"state_type": "overlay"}, confirmable, menu],
|
|
action_ok=[False, True])
|
|
result = invoke(fake, "--no-jev", "--steps", "8", "--max-duplicate-waits", "0")
|
|
check("selection flow completes", result.rc, 0)
|
|
check("rejected selection is retried; accepted indices are not toggled twice",
|
|
fake.actions, [(("select_card",), {"index": 0}),
|
|
(("select_card",), {"index": 0}),
|
|
(("select_card",), {"index": 1}),
|
|
(("confirm_selection",), {}),
|
|
(("menu_select",), {"option": "singleplayer"})])
|
|
changed_grid = dict(grid, card_select=dict(grid["card_select"],
|
|
cards=[{"index": 0, "name": "Defend"}, {"index": 1, "name": "Strike"}]))
|
|
fake = FakeSts2([menu, grid, changed_grid])
|
|
result = invoke(fake, "--no-jev", "--steps", "2", "--max-duplicate-waits", "0")
|
|
check("changed grid does not reuse the old selection indices", len(fake.actions), 1)
|
|
for session in range(2):
|
|
fake = FakeSts2([menu, grid])
|
|
result = invoke(fake, "--no-jev", "--steps", "1")
|
|
check(f"session {session + 1} starts with fresh selection memory",
|
|
fake.actions, [(("select_card",), {"index": 0})])
|
|
fake = FakeSts2([menu, grid])
|
|
result = invoke(fake, "--no-jev", "--steps", "3", "--dry-run", "--max-duplicate-waits", "0")
|
|
check("dry-run sends no selection actions", fake.actions, [])
|
|
check("dry-run proposals never advance selection memory",
|
|
[r["params"] for r in result.rows if r["event"] == "decide"], [{"index": 0}] * 3)
|
|
|
|
shop = {"state_type": "shop", "shop": {"items": [
|
|
{"index": 0, "category": "card", "card_name": "Inflame",
|
|
"card_description": "Gain 2 Strength.", "price": 50,
|
|
"is_stocked": True, "can_afford": True}]}}
|
|
sold = {"state_type": "shop", "shop": {"items": []}}
|
|
affordability_only = dict(shop, shop={"items": [dict(shop["shop"]["items"][0], can_afford=False)]})
|
|
fake = FakeSts2([menu, shop, shop, affordability_only, sold])
|
|
result = invoke(fake, "--steps", "4", "--max-duplicate-waits", "0")
|
|
check("purchase waits through unchanged inventory and unrelated state changes",
|
|
fake.actions, [(("shop_purchase",), {"index": 0}), (("proceed",), {})])
|
|
|
|
bundle = {"state_type": "bundle_select", "bundle_select": {
|
|
"bundles": [{"index": 2, "cards": []}], "preview_showing": False}}
|
|
preview = dict(bundle, bundle_select=dict(bundle["bundle_select"],
|
|
preview_showing=True, can_confirm=True))
|
|
fake = FakeSts2([menu, bundle, bundle, preview, preview, menu])
|
|
result = invoke(fake, "--no-jev", "--steps", "5", "--max-duplicate-waits", "0")
|
|
check("bundle waits for preview and never cancels a delayed confirmation",
|
|
[a[0][0] for a in fake.actions], ["select_bundle", "confirm_bundle_selection", "menu_select"])
|
|
|
|
crystal = {"state_type": "crystal_sphere", "crystal_sphere": {
|
|
"grid_width": 3, "grid_height": 3, "clickable_cells": [{"x": 1, "y": 1}],
|
|
"can_proceed": False}}
|
|
revealed = dict(crystal, crystal_sphere=dict(crystal["crystal_sphere"], can_proceed=True))
|
|
fake = FakeSts2([menu, crystal, crystal, crystal, revealed], action_ok=[False, True])
|
|
result = invoke(fake, "--no-jev", "--steps", "4", "--max-duplicate-waits", "0")
|
|
check("rejected crystal click stays available; accepted click waits for evidence",
|
|
[a[0][0] for a in fake.actions],
|
|
["crystal_sphere_click_cell", "crystal_sphere_click_cell", "crystal_sphere_proceed"])
|
|
|
|
characters = {"state_type": "menu", "menu_screen": "character_select", "options": ["IRONCLAD", "embark"]}
|
|
fake = FakeSts2([menu, characters, characters, characters, characters, characters,
|
|
{"state_type": "unknown"}, characters, menu],
|
|
action_ok=[True, False, True])
|
|
result = invoke(fake, "--no-jev", "--steps", "8", "--max-duplicate-waits", "0")
|
|
check("rejected embark re-selects; accepted embark waits through transitions",
|
|
[a[1].get("option") for a in fake.actions],
|
|
["IRONCLAD", "embark", "IRONCLAD", "embark", "singleplayer"])
|
|
|
|
rewards = {"state_type": "rewards", "rewards": {"items": [{"index": 0, "type": "card"}]}}
|
|
fake = FakeSts2([menu, card_reward, card_reward, rewards, menu, rewards], action_ok=[False, True])
|
|
result = invoke(fake, "--steps", "5", "--max-duplicate-waits", "0")
|
|
check("skip memory follows acceptance and clears outside the reward flow",
|
|
[a[0][0] for a in fake.actions],
|
|
["skip_card_reward", "skip_card_reward", "proceed", "menu_select", "claim_reward"])
|
|
|
|
hand = {"state_type": "hand_select", "hand_select": {
|
|
"mode": "simple_select", "prompt": "Choose a card to Exhaust.",
|
|
"cards": [{"index": 2, "name": "Strike"}], "can_confirm": False}}
|
|
selected_hand = dict(hand, hand_select=dict(hand["hand_select"], can_confirm=True,
|
|
selected_cards=[{"index": 0, "name": "Strike"}]))
|
|
fake = FakeSts2([menu, hand, hand, selected_hand, selected_hand, menu])
|
|
result = invoke(fake, "--no-jev", "--steps", "5", "--max-duplicate-waits", "0")
|
|
check("hand selection waits for selected count, not the separate selected-card index",
|
|
[a[0][0] for a in fake.actions], ["combat_select_card", "combat_confirm_selection", "menu_select"])
|
|
|
|
fake = FakeSts2([menu] + [dict(shop, animation_tick=i) for i in range(100)])
|
|
result = invoke(fake, "--steps", "100", "--stuck-seconds", "10", clock=itertools.count())
|
|
check("pending purchase times out despite unrelated observation changes",
|
|
(result.rc, result.sessions[0]["stop_reason"], len(fake.actions)),
|
|
(1, "pending_action_timeout", 1))
|
|
|
|
print()
|
|
print(f"=== {PASS} passed, {FAIL} failed ===")
|
|
sys.exit(1 if FAIL else 0)
|