"""Tests for scripts/loop-runner.py (task add-loop-runner). All harness subprocess calls and the gate/context-floor subprocess calls are stubbed via monkeypatch. No live LLM calls in CI. """ import json import os import sys import time import subprocess import importlib.util from pathlib import Path from typing import Optional import pytest _RUNNER_PATH = Path.home() / ".automaton" / "scripts" / "loop-runner.py" _spec = importlib.util.spec_from_file_location("loop_runner", _RUNNER_PATH) lr = importlib.util.module_from_spec(_spec) _spec.loader.exec_module(lr) STATUS = Path.home() / ".automaton" / "scripts" / "status.py" VRAM = Path.home() / ".automaton" / "scripts" / "vram_detect.py" RUNNER = _RUNNER_PATH def _state(loop_path: Path) -> dict: return json.loads((loop_path / ".state.loop").read_text()) def _write_state(loop_path: Path, state: dict) -> None: (loop_path / ".state.loop").write_text(json.dumps(state, indent=2, sort_keys=True) + "\n") def _make_loop(project: Path, name: str = "ci-loop", cfg_overrides: Optional[dict] = None) -> Path: """Bootstrap a loop dir + .state.loop + loop.json directly (no subprocess), so tests that monkeypatch subprocess.run can still build fixtures.""" lp = project / ".automaton" / "loops" / name lp.mkdir(parents=True, exist_ok=True) # Default cfg mirrors templates/loops/ci-triage/loop.json. cfg = { "name": name, "description": "test loop", "schedule": {"interval_seconds": 3600}, "brakes": {"max_iterations": 25, "max_budget_usd": None, "score_plateau_window": 5}, "blast_radius": {"file_scope": [], "use_worktree": False}, "roles": {"implement": None, "verify": None, "orchestrate": None}, } if cfg_overrides: cfg.update(cfg_overrides) (lp / "loop.json").write_text(json.dumps(cfg, indent=2) + "\n") (lp / ".state.loop").write_text(json.dumps({ "schema_version": 1, "name": name, "status": "running", "halt_reason": None, "iteration_count": 0, "resumed_count": 0, "last_tick_at": None, "last_verdict": None, "score_history": [], "current_task": None, "worktree_branch": None, "worktree_path": None}, indent=2, sort_keys=True) + "\n") (lp / ".state.log").write_text("") roles = cfg.get("roles") or {} if isinstance(roles, dict): for role_cfg in roles.values(): if isinstance(role_cfg, dict) and role_cfg.get("prompt"): prompt_ref = role_cfg["prompt"] try: (lp / prompt_ref).write_text(f"prompt: {prompt_ref}\n") except OSError: pass return lp def _ci_loop_cfg(roles=None, harness=None, brakes=None): """Compose a loop.json dict with given roles dict {implement, verify, orchestrate}.""" return roles or { "implement": {"prompt": "test-impl.md"}, "verify": {"prompt": "test-verify.md"}, "orchestrate": {"prompt": "test-orch.md"}, } class _FakeSubprocess: """A configurable fake subprocess.run dispatcher keyed on argv patterns. Calls register(pattern -> callable(given_argv) -> CompletedProcess-like). """ def __init__(self): self.rules: list[tuple[str, callable]] = [] self.invocations: list[list[str]] = [] def add(self, needle: str, handler: callable) -> None: self.rules.append((needle, handler)) def add_simple(self, needle: str, stdout: str = "", rc: int = 0) -> None: def handler(argv): class R: pass r = R() r.stdout = stdout r.returncode = rc return r self.rules.append((needle, handler)) def run(self, argv, *args, **kwargs): self.invocations.append(list(argv)) for needle, handler in self.rules: if any(needle in a for a in argv): return handler(argv) # Default: empty stdout, rc 0. class R: pass r = R() r.stdout = "" r.returncode = 0 return r @pytest.fixture def tmp_project(tmp_path): (tmp_path / ".automaton" / "tasks").mkdir(parents=True) return tmp_path @pytest.fixture def fake_run(monkeypatch): fake = _FakeSubprocess() monkeypatch.setattr(subprocess, "run", fake.run) return fake @pytest.fixture def ctx_ok(fake_run): """Stub vram_detect --loop-mode --json to say we are eligible.""" fake_run.add_simple("--loop-mode", json.dumps({"loop_mode_eligible": True})) # --------------------------------------------------------------------------- # R1 -- entrypoint / unknown mode # --------------------------------------------------------------------------- class TestEntrypoint: def test_unknown_loop_exits_zero(self, tmp_project, fake_run): # No .state.loop; runner must exit 0 (clean scheduler exit). out = subprocess.run( [sys.executable, str(RUNNER), "--mode", "tick", "--loop", "ghost", "--project", str(tmp_project)], capture_output=True, text=True) assert out.returncode == 0 def test_unknown_mode_rejected(self, tmp_project): out = subprocess.run( [sys.executable, str(RUNNER), "--mode", "bogus", "--loop", "x", "--project", str(tmp_project)], capture_output=True, text=True) assert out.returncode == 2 # --------------------------------------------------------------------------- # R2 -- tick flow happy path + skips # --------------------------------------------------------------------------- class TestTickFlow: def test_tick_pass(self, tmp_project, fake_run, ctx_ok): lp = _make_loop(tmp_project, cfg_overrides={"roles": { "implement": {"prompt": "test-impl.md"}, "verify": {"prompt": "test-verify.md"}, "orchestrate": {"prompt": "test-orch.md"}}}) s = _state(lp) s["current_task"] = "demo" _write_state(lp, s) # Gate returns ok. fake_run.add_simple("--check-gate", json.dumps({"ok": True, "reason": "running"})) # Verifier output must parse as JSON. The verify-role invocation is # identified by the "test-verify" substring in its {prompt_content} # argv element (the loop-local test-verify.md prompt file content). fake_run.rules.insert(0, ("test-verify", lambda argv: _make_completed( json.dumps({"pass": True, "score": 0.9})))) args = _ns(loop="ci-loop", project=str(tmp_project)) summary = lr.cmd_tick(args) assert summary["skipped"] is False assert summary["halted"] is False assert summary["iter"] == 1 assert summary["verdict"]["pass"] is True assert summary["verdict"]["score"] == 0.9 s2 = _state(lp) assert s2["iteration_count"] == 1 assert s2["last_verdict"]["pass"] is True log = (lp / ".state.log").read_text() assert "TICK pass=True score=0.9 iter=1" in log def test_tick_skip_when_halted(self, tmp_project, fake_run, ctx_ok): lp = _make_loop(tmp_project) s = _state(lp); s["status"] = "halted"; s["halt_reason"] = "drift_detected" s["current_task"] = "demo" _write_state(lp, s) fake_run.add_simple("--check-gate", json.dumps( {"ok": False, "reason": "halted:drift_detected", "halt_reason": "drift_detected"})) args = _ns(loop="ci-loop", project=str(tmp_project)) summary = lr.cmd_tick(args) assert summary["skipped"] is True assert "halted:drift_detected" in summary["reason"] # State unchanged. assert _state(lp)["iteration_count"] == 0 def test_tick_skip_when_untracked(self, tmp_project, fake_run, ctx_ok): # Create dir but no .state.loop. (tmp_project / ".automaton" / "loops" / "stray").mkdir(parents=True) args = _ns(loop="stray", project=str(tmp_project)) summary = lr.cmd_tick(args) assert summary["skipped"] is True assert summary["reason"] == "untracked" def test_tick_skip_no_current_task(self, tmp_project, fake_run, ctx_ok): lp = _make_loop(tmp_project, cfg_overrides={"roles": { "implement": {"prompt": "test-impl.md"}, "verify": {"prompt": "test-verify.md"}, "orchestrate": {"prompt": "test-orch.md"}}}) fake_run.add_simple("--check-gate", json.dumps({"ok": True})) args = _ns(loop="ci-loop", project=str(tmp_project)) summary = lr.cmd_tick(args) assert summary["skipped"] is True assert summary["reason"] == "no_current_task" def test_test_skip_when_gate_subprocess_fails(self, tmp_project, fake_run, ctx_ok): lp = _make_loop(tmp_project) s = _state(lp); s["current_task"] = "demo" _write_state(lp, s) # Don't register any --check-gate rule -> _run_json returns None. args = _ns(loop="ci-loop", project=str(tmp_project)) summary = lr.cmd_tick(args) assert summary["skipped"] is True assert summary["reason"] == "gate_subprocess_failed" # --------------------------------------------------------------------------- # R5 / context-floor guard # --------------------------------------------------------------------------- class TestContextFloor: def test_context_floor_refuses(self, tmp_project, fake_run): lp = _make_loop(tmp_project) s = _state(lp); s["current_task"] = "demo" _write_state(lp, s) fake_run.add_simple("--check-gate", json.dumps({"ok": True})) fake_run.add_simple("--loop-mode", json.dumps({"loop_mode_eligible": False})) args = _ns(loop="ci-loop", project=str(tmp_project)) summary = lr.cmd_tick(args) assert summary["halted"] is True assert summary["reason"] == "context_below_floor" s2 = _state(lp) assert s2["status"] == "halted" assert s2["halt_reason"] == "human_intervention" # Implement subprocess never started -- no harness invocation should be recorded. assert not any("opencode" in " ".join(inv) for inv in fake_run.invocations) # --------------------------------------------------------------------------- # R6 / idempotence: verifier parse failure # --------------------------------------------------------------------------- class TestVerifierParseFailure: def test_parse_failure_halts_no_state_advance(self, tmp_project, fake_run, ctx_ok): lp = _make_loop(tmp_project) s = _state(lp); s["current_task"] = "demo"; s["iteration_count"] = 5 _write_state(lp, s) fake_run.add_simple("--check-gate", json.dumps({"ok": True})) fake_run.rules.insert(0, ("test-verify", lambda argv: _make_completed("not valid json"))) args = _ns(loop="ci-loop", project=str(tmp_project)) summary = lr.cmd_tick(args) assert summary["halted"] is True assert "verifier_failed" in summary["reason"] s2 = _state(lp) assert s2["iteration_count"] == 5 # unchanged assert s2["status"] == "halted" assert s2["halt_reason"] == "verifier_failed" def test_parse_fenced_json(self): text = "```json\n{\"pass\": true, \"score\": 0.8}\n```" v = lr.parse_verdict(text) assert v is not None assert v["pass"] is True assert v["score"] == 0.8 def test_parse_with_line_comments(self): text = "# verdict from verifier\n// signed: GLM-5\n{\"pass\": false, \"score\": 0.2, \"reasons\": [\"x\"]}" v = lr.parse_verdict(text) assert v is not None assert v["pass"] is False assert v["score"] == 0.2 assert v["reasons"] == ["x"] def test_parse_missing_pass_key_returns_none(self): text = "{\"score\": 0.5}" # missing 'pass' v = lr.parse_verdict(text) assert v is None def test_parse_empty(self): assert lr.parse_verdict("") is None assert lr.parse_verdict(" ") is None # --------------------------------------------------------------------------- # R6 / score history capping # --------------------------------------------------------------------------- class TestScoreHistory: def test_score_history_capped(self, tmp_project, fake_run, ctx_ok): lp = _make_loop(tmp_project, cfg_overrides={ "brakes": {"score_plateau_window": 3, "max_iterations": 25}, "roles": { "implement": {"prompt": "test-impl.md"}, "verify": {"prompt": "test-verify.md"}, "orchestrate": {"prompt": "test-orch.md"}}}) s = _state(lp); s["current_task"] = "demo" _write_state(lp, s) fake_run.add_simple("--check-gate", json.dumps({"ok": True})) scores = [0.5, 0.4, 0.4, 0.4, 0.4] seen_iter_counts = [] for sc in scores: fake_run.rules = [r for r in fake_run.rules if r[0] != "test-verify"] fake_run.rules.insert(0, ("test-verify", lambda argv, _sc=sc: _make_completed( json.dumps({"pass": False, "score": _sc})))) args = _ns(loop="ci-loop", project=str(tmp_project)) summary = lr.cmd_tick(args) seen_iter_counts.append((int(sc * 10), summary.get("skipped"), summary.get("halted"), summary.get("reason"), summary.get("iter"))) s2 = _state(lp) assert s2["iteration_count"] == 5, f"trace: {seen_iter_counts}" assert len(s2["score_history"]) == 3 assert s2["score_history"] == [0.4, 0.4, 0.4] # --------------------------------------------------------------------------- # R4 / harness command substitution # --------------------------------------------------------------------------- class TestHarnessSubstitution: def test_custom_command_with_output_token(self, tmp_project, fake_run, ctx_ok): lp = _make_loop(tmp_project, cfg_overrides={ "harness": {"command": ["my-harness", "--prompt", "{prompt}", "--cwd", "{cwd}", "--out", "{output}", "--artifact", "{artifact}"]}, "roles": { "implement": {"prompt": "test-impl.md"}, "verify": {"prompt": "test-verify.md"}, "orchestrate": {"prompt": "test-orch.md"}}}) s = _state(lp); s["current_task"] = "demo" _write_state(lp, s) fake_run.add_simple("--check-gate", json.dumps({"ok": True})) def harness_handler(argv): class R: pass r = R() r.stdout = "" r.returncode = 0 return r fake_run.rules.append(("my-harness", harness_handler)) # Verifier handler takes precedence over the generic my-harness rule. # The verify role is identified by "verify-prompt" in the resolved # prompt file path (e.g. /outputs/tick1-verify-prompt.md). fake_run.rules.insert(0, ("verify-prompt", lambda argv: _make_completed( json.dumps({"pass": True, "score": 0.5})))) args = _ns(loop="ci-loop", project=str(tmp_project)) lr.cmd_tick(args) # Each role invocation must include the closed-over cwd and prompt tokens. seen_artifacts: list[str] = [] for inv in fake_run.invocations: if "my-harness" in inv: assert "--cwd" in inv assert "--prompt" in inv if "--artifact" in inv: a_idx = inv.index("--artifact") + 1 if a_idx < len(inv) and inv[a_idx] != "{artifact}": seen_artifacts.append(inv[a_idx]) # The verify role's invocation must carry the Implement output path via --artifact . verify_invocations = [inv for inv in fake_run.invocations if "my-harness" in inv and "verify-prompt" in " ".join(inv)] assert any(a.endswith("-implement.json") for a in seen_artifacts), ( f"verify role must receive the implement artifact path via --artifact; " f"saw: {seen_artifacts}") assert verify_invocations, "no verify-role invocation captured" # --------------------------------------------------------------------------- # R3 / daemon mode # --------------------------------------------------------------------------- class TestDaemonMode: def test_daemon_runs_n_iterations(self, tmp_project, fake_run, ctx_ok, monkeypatch): lp = _make_loop(tmp_project, cfg_overrides={"roles": { "implement": {"prompt": "test-impl.md"}, "verify": {"prompt": "test-verify.md"}, "orchestrate": {"prompt": "test-orch.md"}}}) s = _state(lp); s["current_task"] = "demo" _write_state(lp, s) fake_run.add_simple("--check-gate", json.dumps({"ok": True})) fake_run.rules.insert(0, ("test-verify", lambda argv: _make_completed( json.dumps({"pass": True, "score": 0.5})))) # Speed up sleeps. monkeypatch.setattr(time, "sleep", lambda s: None) args = _ns(loop="ci-loop", project=str(tmp_project), mode="daemon", max_iterations=3, interval=1) rc = lr.cmd_daemon(args) assert rc == 0 assert _state(lp)["iteration_count"] == 3 # --------------------------------------------------------------------------- # R7 / orchestrator-ordering # --------------------------------------------------------------------------- class TestOrchestratorOrdering: def test_roles_invoked_in_order(self, tmp_project, fake_run, ctx_ok): lp = _make_loop(tmp_project, cfg_overrides={"roles": { "implement": {"prompt": "test-impl.md"}, "verify": {"prompt": "test-verify.md"}, "orchestrate": {"prompt": "test-orch.md"}}}) s = _state(lp); s["current_task"] = "demo" _write_state(lp, s) fake_run.add_simple("--check-gate", json.dumps({"ok": True})) order: list[str] = [] def make_handler(tag): def handler(argv): order.append(tag) class R: pass r = R(); r.stdout = ""; r.returncode = 0 return r return handler fake_run.rules.append(("test-impl", make_handler("implement"))) fake_run.rules.append(("test-verify", lambda argv: ( order.append("verify"), _make_completed(json.dumps({"pass": True, "score": 0.5})))[1])) fake_run.rules.append(("test-orch", make_handler("orchestrate"))) args = _ns(loop="ci-loop", project=str(tmp_project)) lr.cmd_tick(args) assert order == ["check-gate-implied", "implement", "verify", "orchestrate"] or \ order == ["implement", "verify", "orchestrate"] # --------------------------------------------------------------------------- # R7 / JSON output # --------------------------------------------------------------------------- class TestJsonOutput: def test_json_output_emits_summary(self, tmp_project, fake_run, ctx_ok, capsys, monkeypatch): lp = _make_loop(tmp_project, cfg_overrides={"roles": { "implement": {"prompt": "test-impl.md"}, "verify": {"prompt": "test-verify.md"}, "orchestrate": {"prompt": "test-orch.md"}}}) s = _state(lp); s["current_task"] = "demo" _write_state(lp, s) fake_run.add_simple("--check-gate", json.dumps({"ok": True})) fake_run.rules.insert(0, ("test-verify", lambda argv: _make_completed( json.dumps({"pass": True, "score": 0.7})))) monkeypatch.setattr(sys, "argv", [ "loop-runner.py", "--mode", "tick", "--loop", "ci-loop", "--project", str(tmp_project), "--json"]) rc = lr.main() assert rc == 0 out = capsys.readouterr().out payload = json.loads(out.splitlines()[-1]) assert payload["loop"] == "ci-loop" assert payload["skipped"] is False assert payload["iter"] == 1 assert payload["verdict"]["pass"] is True # --------------------------------------------------------------------------- # helpers # --------------------------------------------------------------------------- def _make_completed(stdout: str): class R: pass r = R() r.stdout = stdout r.returncode = 0 return r class _Args: def __init__(self, **kw): self.__dict__.update(kw) def _ns(loop: str, project: str, mode: str = "tick", interval: Optional[int] = None, max_iterations: Optional[int] = None, json_output: bool = False) -> _Args: return _Args(loop=loop, project=project, mode=mode, interval=interval, max_iterations=max_iterations, json_output=json_output)