Files

502 lines
20 KiB
Python
Raw Permalink Normal View History

"""Tests for scripts/loop-runner.py (task add-loop-runner).
All harness subprocess calls and the gate/context-floor subprocess calls are
stubbed via monkeypatch. No live LLM calls in CI.
"""
import json
import os
import sys
import time
import subprocess
import importlib.util
from pathlib import Path
from typing import Optional
import pytest
_RUNNER_PATH = Path.home() / ".automaton" / "scripts" / "loop-runner.py"
_spec = importlib.util.spec_from_file_location("loop_runner", _RUNNER_PATH)
lr = importlib.util.module_from_spec(_spec)
_spec.loader.exec_module(lr)
STATUS = Path.home() / ".automaton" / "scripts" / "status.py"
VRAM = Path.home() / ".automaton" / "scripts" / "vram_detect.py"
RUNNER = _RUNNER_PATH
def _state(loop_path: Path) -> dict:
return json.loads((loop_path / ".state.loop").read_text())
def _write_state(loop_path: Path, state: dict) -> None:
(loop_path / ".state.loop").write_text(json.dumps(state, indent=2, sort_keys=True) + "\n")
def _make_loop(project: Path, name: str = "ci-loop", cfg_overrides: Optional[dict] = None) -> Path:
"""Bootstrap a loop dir + .state.loop + loop.json directly (no subprocess),
so tests that monkeypatch subprocess.run can still build fixtures."""
lp = project / ".automaton" / "loops" / name
lp.mkdir(parents=True, exist_ok=True)
# Default cfg mirrors templates/loops/ci-triage/loop.json.
cfg = {
"name": name,
"description": "test loop",
"schedule": {"interval_seconds": 3600},
"brakes": {"max_iterations": 25, "max_budget_usd": None, "score_plateau_window": 5},
"blast_radius": {"file_scope": [], "use_worktree": False},
"roles": {"implement": None, "verify": None, "orchestrate": None},
}
if cfg_overrides:
cfg.update(cfg_overrides)
(lp / "loop.json").write_text(json.dumps(cfg, indent=2) + "\n")
(lp / ".state.loop").write_text(json.dumps({
"schema_version": 1, "name": name, "status": "running", "halt_reason": None,
"iteration_count": 0, "resumed_count": 0, "last_tick_at": None,
"last_verdict": None, "score_history": [], "current_task": None,
"worktree_branch": None, "worktree_path": None}, indent=2, sort_keys=True) + "\n")
(lp / ".state.log").write_text("")
roles = cfg.get("roles") or {}
if isinstance(roles, dict):
for role_cfg in roles.values():
if isinstance(role_cfg, dict) and role_cfg.get("prompt"):
prompt_ref = role_cfg["prompt"]
try:
(lp / prompt_ref).write_text(f"prompt: {prompt_ref}\n")
except OSError:
pass
return lp
def _ci_loop_cfg(roles=None, harness=None, brakes=None):
"""Compose a loop.json dict with given roles dict {implement, verify, orchestrate}."""
return roles or {
"implement": {"prompt": "test-impl.md"},
"verify": {"prompt": "test-verify.md"},
"orchestrate": {"prompt": "test-orch.md"},
}
class _FakeSubprocess:
"""A configurable fake subprocess.run dispatcher keyed on argv patterns.
Calls register(pattern -> callable(given_argv) -> CompletedProcess-like).
"""
def __init__(self):
self.rules: list[tuple[str, callable]] = []
self.invocations: list[list[str]] = []
def add(self, needle: str, handler: callable) -> None:
self.rules.append((needle, handler))
def add_simple(self, needle: str, stdout: str = "", rc: int = 0) -> None:
def handler(argv):
class R:
pass
r = R()
r.stdout = stdout
r.returncode = rc
return r
self.rules.append((needle, handler))
def run(self, argv, *args, **kwargs):
self.invocations.append(list(argv))
for needle, handler in self.rules:
if any(needle in a for a in argv):
return handler(argv)
# Default: empty stdout, rc 0.
class R:
pass
r = R()
r.stdout = ""
r.returncode = 0
return r
@pytest.fixture
def tmp_project(tmp_path):
(tmp_path / ".automaton" / "tasks").mkdir(parents=True)
return tmp_path
@pytest.fixture
def fake_run(monkeypatch):
fake = _FakeSubprocess()
monkeypatch.setattr(subprocess, "run", fake.run)
return fake
@pytest.fixture
def ctx_ok(fake_run):
"""Stub vram_detect --loop-mode --json to say we are eligible."""
fake_run.add_simple("--loop-mode", json.dumps({"loop_mode_eligible": True}))
# ---------------------------------------------------------------------------
# R1 -- entrypoint / unknown mode
# ---------------------------------------------------------------------------
class TestEntrypoint:
def test_unknown_loop_exits_zero(self, tmp_project, fake_run):
# No .state.loop; runner must exit 0 (clean scheduler exit).
out = subprocess.run(
[sys.executable, str(RUNNER), "--mode", "tick", "--loop", "ghost",
"--project", str(tmp_project)],
capture_output=True, text=True)
assert out.returncode == 0
def test_unknown_mode_rejected(self, tmp_project):
out = subprocess.run(
[sys.executable, str(RUNNER), "--mode", "bogus", "--loop", "x",
"--project", str(tmp_project)],
capture_output=True, text=True)
assert out.returncode == 2
# ---------------------------------------------------------------------------
# R2 -- tick flow happy path + skips
# ---------------------------------------------------------------------------
class TestTickFlow:
def test_tick_pass(self, tmp_project, fake_run, ctx_ok):
lp = _make_loop(tmp_project, cfg_overrides={"roles": {
"implement": {"prompt": "test-impl.md"},
"verify": {"prompt": "test-verify.md"},
"orchestrate": {"prompt": "test-orch.md"}}})
s = _state(lp)
s["current_task"] = "demo"
_write_state(lp, s)
# Gate returns ok.
fake_run.add_simple("--check-gate", json.dumps({"ok": True, "reason": "running"}))
# Verifier output must parse as JSON. The verify-role invocation is
# identified by the "test-verify" substring in its {prompt_content}
# argv element (the loop-local test-verify.md prompt file content).
fake_run.rules.insert(0, ("test-verify", lambda argv: _make_completed(
json.dumps({"pass": True, "score": 0.9}))))
args = _ns(loop="ci-loop", project=str(tmp_project))
summary = lr.cmd_tick(args)
assert summary["skipped"] is False
assert summary["halted"] is False
assert summary["iter"] == 1
assert summary["verdict"]["pass"] is True
assert summary["verdict"]["score"] == 0.9
s2 = _state(lp)
assert s2["iteration_count"] == 1
assert s2["last_verdict"]["pass"] is True
log = (lp / ".state.log").read_text()
assert "TICK pass=True score=0.9 iter=1" in log
def test_tick_skip_when_halted(self, tmp_project, fake_run, ctx_ok):
lp = _make_loop(tmp_project)
s = _state(lp); s["status"] = "halted"; s["halt_reason"] = "drift_detected"
s["current_task"] = "demo"
_write_state(lp, s)
fake_run.add_simple("--check-gate", json.dumps(
{"ok": False, "reason": "halted:drift_detected", "halt_reason": "drift_detected"}))
args = _ns(loop="ci-loop", project=str(tmp_project))
summary = lr.cmd_tick(args)
assert summary["skipped"] is True
assert "halted:drift_detected" in summary["reason"]
# State unchanged.
assert _state(lp)["iteration_count"] == 0
def test_tick_skip_when_untracked(self, tmp_project, fake_run, ctx_ok):
# Create dir but no .state.loop.
(tmp_project / ".automaton" / "loops" / "stray").mkdir(parents=True)
args = _ns(loop="stray", project=str(tmp_project))
summary = lr.cmd_tick(args)
assert summary["skipped"] is True
assert summary["reason"] == "untracked"
def test_tick_skip_no_current_task(self, tmp_project, fake_run, ctx_ok):
lp = _make_loop(tmp_project, cfg_overrides={"roles": {
"implement": {"prompt": "test-impl.md"},
"verify": {"prompt": "test-verify.md"},
"orchestrate": {"prompt": "test-orch.md"}}})
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
args = _ns(loop="ci-loop", project=str(tmp_project))
summary = lr.cmd_tick(args)
assert summary["skipped"] is True
assert summary["reason"] == "no_current_task"
def test_test_skip_when_gate_subprocess_fails(self, tmp_project, fake_run, ctx_ok):
lp = _make_loop(tmp_project)
s = _state(lp); s["current_task"] = "demo"
_write_state(lp, s)
# Don't register any --check-gate rule -> _run_json returns None.
args = _ns(loop="ci-loop", project=str(tmp_project))
summary = lr.cmd_tick(args)
assert summary["skipped"] is True
assert summary["reason"] == "gate_subprocess_failed"
# ---------------------------------------------------------------------------
# R5 / context-floor guard
# ---------------------------------------------------------------------------
class TestContextFloor:
def test_context_floor_refuses(self, tmp_project, fake_run):
lp = _make_loop(tmp_project)
s = _state(lp); s["current_task"] = "demo"
_write_state(lp, s)
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
fake_run.add_simple("--loop-mode", json.dumps({"loop_mode_eligible": False}))
args = _ns(loop="ci-loop", project=str(tmp_project))
summary = lr.cmd_tick(args)
assert summary["halted"] is True
assert summary["reason"] == "context_below_floor"
s2 = _state(lp)
assert s2["status"] == "halted"
assert s2["halt_reason"] == "human_intervention"
# Implement subprocess never started -- no harness invocation should be recorded.
assert not any("opencode" in " ".join(inv) for inv in fake_run.invocations)
# ---------------------------------------------------------------------------
# R6 / idempotence: verifier parse failure
# ---------------------------------------------------------------------------
class TestVerifierParseFailure:
def test_parse_failure_halts_no_state_advance(self, tmp_project, fake_run, ctx_ok):
lp = _make_loop(tmp_project)
s = _state(lp); s["current_task"] = "demo"; s["iteration_count"] = 5
_write_state(lp, s)
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
fake_run.rules.insert(0, ("test-verify", lambda argv: _make_completed("not valid json")))
args = _ns(loop="ci-loop", project=str(tmp_project))
summary = lr.cmd_tick(args)
assert summary["halted"] is True
assert "verifier_failed" in summary["reason"]
s2 = _state(lp)
assert s2["iteration_count"] == 5 # unchanged
assert s2["status"] == "halted"
assert s2["halt_reason"] == "verifier_failed"
def test_parse_fenced_json(self):
text = "```json\n{\"pass\": true, \"score\": 0.8}\n```"
v = lr.parse_verdict(text)
assert v is not None
assert v["pass"] is True
assert v["score"] == 0.8
def test_parse_with_line_comments(self):
text = "# verdict from verifier\n// signed: GLM-5\n{\"pass\": false, \"score\": 0.2, \"reasons\": [\"x\"]}"
v = lr.parse_verdict(text)
assert v is not None
assert v["pass"] is False
assert v["score"] == 0.2
assert v["reasons"] == ["x"]
def test_parse_missing_pass_key_returns_none(self):
text = "{\"score\": 0.5}" # missing 'pass'
v = lr.parse_verdict(text)
assert v is None
def test_parse_empty(self):
assert lr.parse_verdict("") is None
assert lr.parse_verdict(" ") is None
# ---------------------------------------------------------------------------
# R6 / score history capping
# ---------------------------------------------------------------------------
class TestScoreHistory:
def test_score_history_capped(self, tmp_project, fake_run, ctx_ok):
lp = _make_loop(tmp_project, cfg_overrides={
"brakes": {"score_plateau_window": 3, "max_iterations": 25},
"roles": {
"implement": {"prompt": "test-impl.md"},
"verify": {"prompt": "test-verify.md"},
"orchestrate": {"prompt": "test-orch.md"}}})
s = _state(lp); s["current_task"] = "demo"
_write_state(lp, s)
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
scores = [0.5, 0.4, 0.4, 0.4, 0.4]
seen_iter_counts = []
for sc in scores:
fake_run.rules = [r for r in fake_run.rules if r[0] != "test-verify"]
fake_run.rules.insert(0, ("test-verify", lambda argv, _sc=sc: _make_completed(
json.dumps({"pass": False, "score": _sc}))))
args = _ns(loop="ci-loop", project=str(tmp_project))
summary = lr.cmd_tick(args)
seen_iter_counts.append((int(sc * 10), summary.get("skipped"), summary.get("halted"),
summary.get("reason"), summary.get("iter")))
s2 = _state(lp)
assert s2["iteration_count"] == 5, f"trace: {seen_iter_counts}"
assert len(s2["score_history"]) == 3
assert s2["score_history"] == [0.4, 0.4, 0.4]
# ---------------------------------------------------------------------------
# R4 / harness command substitution
# ---------------------------------------------------------------------------
class TestHarnessSubstitution:
def test_custom_command_with_output_token(self, tmp_project, fake_run, ctx_ok):
lp = _make_loop(tmp_project, cfg_overrides={
"harness": {"command": ["my-harness", "--prompt", "{prompt}",
"--cwd", "{cwd}", "--out", "{output}",
"--artifact", "{artifact}"]},
"roles": {
"implement": {"prompt": "test-impl.md"},
"verify": {"prompt": "test-verify.md"},
"orchestrate": {"prompt": "test-orch.md"}}})
s = _state(lp); s["current_task"] = "demo"
_write_state(lp, s)
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
def harness_handler(argv):
class R:
pass
r = R()
r.stdout = ""
r.returncode = 0
return r
fake_run.rules.append(("my-harness", harness_handler))
# Verifier handler takes precedence over the generic my-harness rule.
# The verify role is identified by "verify-prompt" in the resolved
# prompt file path (e.g. <loop>/outputs/tick1-verify-prompt.md).
fake_run.rules.insert(0, ("verify-prompt", lambda argv: _make_completed(
json.dumps({"pass": True, "score": 0.5}))))
args = _ns(loop="ci-loop", project=str(tmp_project))
lr.cmd_tick(args)
# Each role invocation must include the closed-over cwd and prompt tokens.
seen_artifacts: list[str] = []
for inv in fake_run.invocations:
if "my-harness" in inv:
assert "--cwd" in inv
assert "--prompt" in inv
if "--artifact" in inv:
a_idx = inv.index("--artifact") + 1
if a_idx < len(inv) and inv[a_idx] != "{artifact}":
seen_artifacts.append(inv[a_idx])
# The verify role's invocation must carry the Implement output path via --artifact <path>.
verify_invocations = [inv for inv in fake_run.invocations
if "my-harness" in inv and "verify-prompt" in " ".join(inv)]
assert any(a.endswith("-implement.json") for a in seen_artifacts), (
f"verify role must receive the implement artifact path via --artifact; "
f"saw: {seen_artifacts}")
assert verify_invocations, "no verify-role invocation captured"
# ---------------------------------------------------------------------------
# R3 / daemon mode
# ---------------------------------------------------------------------------
class TestDaemonMode:
def test_daemon_runs_n_iterations(self, tmp_project, fake_run, ctx_ok, monkeypatch):
lp = _make_loop(tmp_project, cfg_overrides={"roles": {
"implement": {"prompt": "test-impl.md"},
"verify": {"prompt": "test-verify.md"},
"orchestrate": {"prompt": "test-orch.md"}}})
s = _state(lp); s["current_task"] = "demo"
_write_state(lp, s)
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
fake_run.rules.insert(0, ("test-verify", lambda argv: _make_completed(
json.dumps({"pass": True, "score": 0.5}))))
# Speed up sleeps.
monkeypatch.setattr(time, "sleep", lambda s: None)
args = _ns(loop="ci-loop", project=str(tmp_project),
mode="daemon", max_iterations=3, interval=1)
rc = lr.cmd_daemon(args)
assert rc == 0
assert _state(lp)["iteration_count"] == 3
# ---------------------------------------------------------------------------
# R7 / orchestrator-ordering
# ---------------------------------------------------------------------------
class TestOrchestratorOrdering:
def test_roles_invoked_in_order(self, tmp_project, fake_run, ctx_ok):
lp = _make_loop(tmp_project, cfg_overrides={"roles": {
"implement": {"prompt": "test-impl.md"},
"verify": {"prompt": "test-verify.md"},
"orchestrate": {"prompt": "test-orch.md"}}})
s = _state(lp); s["current_task"] = "demo"
_write_state(lp, s)
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
order: list[str] = []
def make_handler(tag):
def handler(argv):
order.append(tag)
class R:
pass
r = R(); r.stdout = ""; r.returncode = 0
return r
return handler
fake_run.rules.append(("test-impl", make_handler("implement")))
fake_run.rules.append(("test-verify", lambda argv: (
order.append("verify"),
_make_completed(json.dumps({"pass": True, "score": 0.5})))[1]))
fake_run.rules.append(("test-orch", make_handler("orchestrate")))
args = _ns(loop="ci-loop", project=str(tmp_project))
lr.cmd_tick(args)
assert order == ["check-gate-implied", "implement", "verify", "orchestrate"] or \
order == ["implement", "verify", "orchestrate"]
# ---------------------------------------------------------------------------
# R7 / JSON output
# ---------------------------------------------------------------------------
class TestJsonOutput:
def test_json_output_emits_summary(self, tmp_project, fake_run, ctx_ok, capsys, monkeypatch):
lp = _make_loop(tmp_project, cfg_overrides={"roles": {
"implement": {"prompt": "test-impl.md"},
"verify": {"prompt": "test-verify.md"},
"orchestrate": {"prompt": "test-orch.md"}}})
s = _state(lp); s["current_task"] = "demo"
_write_state(lp, s)
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
fake_run.rules.insert(0, ("test-verify", lambda argv: _make_completed(
json.dumps({"pass": True, "score": 0.7}))))
monkeypatch.setattr(sys, "argv", [
"loop-runner.py", "--mode", "tick", "--loop", "ci-loop",
"--project", str(tmp_project), "--json"])
rc = lr.main()
assert rc == 0
out = capsys.readouterr().out
payload = json.loads(out.splitlines()[-1])
assert payload["loop"] == "ci-loop"
assert payload["skipped"] is False
assert payload["iter"] == 1
assert payload["verdict"]["pass"] is True
# ---------------------------------------------------------------------------
# helpers
# ---------------------------------------------------------------------------
def _make_completed(stdout: str):
class R:
pass
r = R()
r.stdout = stdout
r.returncode = 0
return r
class _Args:
def __init__(self, **kw):
self.__dict__.update(kw)
def _ns(loop: str, project: str, mode: str = "tick", interval: Optional[int] = None,
max_iterations: Optional[int] = None, json_output: bool = False) -> _Args:
return _Args(loop=loop, project=project, mode=mode, interval=interval,
max_iterations=max_iterations, json_output=json_output)