502 lines
20 KiB
Python
502 lines
20 KiB
Python
"""Tests for scripts/loop-runner.py (task add-loop-runner).
|
|
|
|
All harness subprocess calls and the gate/context-floor subprocess calls are
|
|
stubbed via monkeypatch. No live LLM calls in CI.
|
|
"""
|
|
|
|
import json
|
|
import os
|
|
import sys
|
|
import time
|
|
import subprocess
|
|
import importlib.util
|
|
from pathlib import Path
|
|
from typing import Optional
|
|
|
|
import pytest
|
|
|
|
_RUNNER_PATH = Path.home() / ".automaton" / "scripts" / "loop-runner.py"
|
|
_spec = importlib.util.spec_from_file_location("loop_runner", _RUNNER_PATH)
|
|
lr = importlib.util.module_from_spec(_spec)
|
|
_spec.loader.exec_module(lr)
|
|
|
|
STATUS = Path.home() / ".automaton" / "scripts" / "status.py"
|
|
VRAM = Path.home() / ".automaton" / "scripts" / "vram_detect.py"
|
|
RUNNER = _RUNNER_PATH
|
|
|
|
|
|
def _state(loop_path: Path) -> dict:
|
|
return json.loads((loop_path / ".state.loop").read_text())
|
|
|
|
|
|
def _write_state(loop_path: Path, state: dict) -> None:
|
|
(loop_path / ".state.loop").write_text(json.dumps(state, indent=2, sort_keys=True) + "\n")
|
|
|
|
|
|
def _make_loop(project: Path, name: str = "ci-loop", cfg_overrides: Optional[dict] = None) -> Path:
|
|
"""Bootstrap a loop dir + .state.loop + loop.json directly (no subprocess),
|
|
so tests that monkeypatch subprocess.run can still build fixtures."""
|
|
lp = project / ".automaton" / "loops" / name
|
|
lp.mkdir(parents=True, exist_ok=True)
|
|
# Default cfg mirrors templates/loops/ci-triage/loop.json.
|
|
cfg = {
|
|
"name": name,
|
|
"description": "test loop",
|
|
"schedule": {"interval_seconds": 3600},
|
|
"brakes": {"max_iterations": 25, "max_budget_usd": None, "score_plateau_window": 5},
|
|
"blast_radius": {"file_scope": [], "use_worktree": False},
|
|
"roles": {"implement": None, "verify": None, "orchestrate": None},
|
|
}
|
|
if cfg_overrides:
|
|
cfg.update(cfg_overrides)
|
|
(lp / "loop.json").write_text(json.dumps(cfg, indent=2) + "\n")
|
|
(lp / ".state.loop").write_text(json.dumps({
|
|
"schema_version": 1, "name": name, "status": "running", "halt_reason": None,
|
|
"iteration_count": 0, "resumed_count": 0, "last_tick_at": None,
|
|
"last_verdict": None, "score_history": [], "current_task": None,
|
|
"worktree_branch": None, "worktree_path": None}, indent=2, sort_keys=True) + "\n")
|
|
(lp / ".state.log").write_text("")
|
|
roles = cfg.get("roles") or {}
|
|
if isinstance(roles, dict):
|
|
for role_cfg in roles.values():
|
|
if isinstance(role_cfg, dict) and role_cfg.get("prompt"):
|
|
prompt_ref = role_cfg["prompt"]
|
|
try:
|
|
(lp / prompt_ref).write_text(f"prompt: {prompt_ref}\n")
|
|
except OSError:
|
|
pass
|
|
return lp
|
|
|
|
|
|
def _ci_loop_cfg(roles=None, harness=None, brakes=None):
|
|
"""Compose a loop.json dict with given roles dict {implement, verify, orchestrate}."""
|
|
return roles or {
|
|
"implement": {"prompt": "test-impl.md"},
|
|
"verify": {"prompt": "test-verify.md"},
|
|
"orchestrate": {"prompt": "test-orch.md"},
|
|
}
|
|
|
|
|
|
class _FakeSubprocess:
|
|
"""A configurable fake subprocess.run dispatcher keyed on argv patterns.
|
|
|
|
Calls register(pattern -> callable(given_argv) -> CompletedProcess-like).
|
|
"""
|
|
|
|
def __init__(self):
|
|
self.rules: list[tuple[str, callable]] = []
|
|
self.invocations: list[list[str]] = []
|
|
|
|
def add(self, needle: str, handler: callable) -> None:
|
|
self.rules.append((needle, handler))
|
|
|
|
def add_simple(self, needle: str, stdout: str = "", rc: int = 0) -> None:
|
|
def handler(argv):
|
|
class R:
|
|
pass
|
|
r = R()
|
|
r.stdout = stdout
|
|
r.returncode = rc
|
|
return r
|
|
self.rules.append((needle, handler))
|
|
|
|
def run(self, argv, *args, **kwargs):
|
|
self.invocations.append(list(argv))
|
|
for needle, handler in self.rules:
|
|
if any(needle in a for a in argv):
|
|
return handler(argv)
|
|
# Default: empty stdout, rc 0.
|
|
class R:
|
|
pass
|
|
r = R()
|
|
r.stdout = ""
|
|
r.returncode = 0
|
|
return r
|
|
|
|
|
|
@pytest.fixture
|
|
def tmp_project(tmp_path):
|
|
(tmp_path / ".automaton" / "tasks").mkdir(parents=True)
|
|
return tmp_path
|
|
|
|
|
|
@pytest.fixture
|
|
def fake_run(monkeypatch):
|
|
fake = _FakeSubprocess()
|
|
monkeypatch.setattr(subprocess, "run", fake.run)
|
|
return fake
|
|
|
|
|
|
@pytest.fixture
|
|
def ctx_ok(fake_run):
|
|
"""Stub vram_detect --loop-mode --json to say we are eligible."""
|
|
fake_run.add_simple("--loop-mode", json.dumps({"loop_mode_eligible": True}))
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R1 -- entrypoint / unknown mode
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestEntrypoint:
|
|
def test_unknown_loop_exits_zero(self, tmp_project, fake_run):
|
|
# No .state.loop; runner must exit 0 (clean scheduler exit).
|
|
out = subprocess.run(
|
|
[sys.executable, str(RUNNER), "--mode", "tick", "--loop", "ghost",
|
|
"--project", str(tmp_project)],
|
|
capture_output=True, text=True)
|
|
assert out.returncode == 0
|
|
|
|
def test_unknown_mode_rejected(self, tmp_project):
|
|
out = subprocess.run(
|
|
[sys.executable, str(RUNNER), "--mode", "bogus", "--loop", "x",
|
|
"--project", str(tmp_project)],
|
|
capture_output=True, text=True)
|
|
assert out.returncode == 2
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R2 -- tick flow happy path + skips
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestTickFlow:
|
|
def test_tick_pass(self, tmp_project, fake_run, ctx_ok):
|
|
lp = _make_loop(tmp_project, cfg_overrides={"roles": {
|
|
"implement": {"prompt": "test-impl.md"},
|
|
"verify": {"prompt": "test-verify.md"},
|
|
"orchestrate": {"prompt": "test-orch.md"}}})
|
|
s = _state(lp)
|
|
s["current_task"] = "demo"
|
|
_write_state(lp, s)
|
|
# Gate returns ok.
|
|
fake_run.add_simple("--check-gate", json.dumps({"ok": True, "reason": "running"}))
|
|
# Verifier output must parse as JSON. The verify-role invocation is
|
|
# identified by the "test-verify" substring in its {prompt_content}
|
|
# argv element (the loop-local test-verify.md prompt file content).
|
|
fake_run.rules.insert(0, ("test-verify", lambda argv: _make_completed(
|
|
json.dumps({"pass": True, "score": 0.9}))))
|
|
args = _ns(loop="ci-loop", project=str(tmp_project))
|
|
summary = lr.cmd_tick(args)
|
|
assert summary["skipped"] is False
|
|
assert summary["halted"] is False
|
|
assert summary["iter"] == 1
|
|
assert summary["verdict"]["pass"] is True
|
|
assert summary["verdict"]["score"] == 0.9
|
|
s2 = _state(lp)
|
|
assert s2["iteration_count"] == 1
|
|
assert s2["last_verdict"]["pass"] is True
|
|
log = (lp / ".state.log").read_text()
|
|
assert "TICK pass=True score=0.9 iter=1" in log
|
|
|
|
def test_tick_skip_when_halted(self, tmp_project, fake_run, ctx_ok):
|
|
lp = _make_loop(tmp_project)
|
|
s = _state(lp); s["status"] = "halted"; s["halt_reason"] = "drift_detected"
|
|
s["current_task"] = "demo"
|
|
_write_state(lp, s)
|
|
fake_run.add_simple("--check-gate", json.dumps(
|
|
{"ok": False, "reason": "halted:drift_detected", "halt_reason": "drift_detected"}))
|
|
args = _ns(loop="ci-loop", project=str(tmp_project))
|
|
summary = lr.cmd_tick(args)
|
|
assert summary["skipped"] is True
|
|
assert "halted:drift_detected" in summary["reason"]
|
|
# State unchanged.
|
|
assert _state(lp)["iteration_count"] == 0
|
|
|
|
def test_tick_skip_when_untracked(self, tmp_project, fake_run, ctx_ok):
|
|
# Create dir but no .state.loop.
|
|
(tmp_project / ".automaton" / "loops" / "stray").mkdir(parents=True)
|
|
args = _ns(loop="stray", project=str(tmp_project))
|
|
summary = lr.cmd_tick(args)
|
|
assert summary["skipped"] is True
|
|
assert summary["reason"] == "untracked"
|
|
|
|
def test_tick_skip_no_current_task(self, tmp_project, fake_run, ctx_ok):
|
|
lp = _make_loop(tmp_project, cfg_overrides={"roles": {
|
|
"implement": {"prompt": "test-impl.md"},
|
|
"verify": {"prompt": "test-verify.md"},
|
|
"orchestrate": {"prompt": "test-orch.md"}}})
|
|
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
|
|
args = _ns(loop="ci-loop", project=str(tmp_project))
|
|
summary = lr.cmd_tick(args)
|
|
assert summary["skipped"] is True
|
|
assert summary["reason"] == "no_current_task"
|
|
|
|
def test_test_skip_when_gate_subprocess_fails(self, tmp_project, fake_run, ctx_ok):
|
|
lp = _make_loop(tmp_project)
|
|
s = _state(lp); s["current_task"] = "demo"
|
|
_write_state(lp, s)
|
|
# Don't register any --check-gate rule -> _run_json returns None.
|
|
args = _ns(loop="ci-loop", project=str(tmp_project))
|
|
summary = lr.cmd_tick(args)
|
|
assert summary["skipped"] is True
|
|
assert summary["reason"] == "gate_subprocess_failed"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R5 / context-floor guard
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestContextFloor:
|
|
def test_context_floor_refuses(self, tmp_project, fake_run):
|
|
lp = _make_loop(tmp_project)
|
|
s = _state(lp); s["current_task"] = "demo"
|
|
_write_state(lp, s)
|
|
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
|
|
fake_run.add_simple("--loop-mode", json.dumps({"loop_mode_eligible": False}))
|
|
args = _ns(loop="ci-loop", project=str(tmp_project))
|
|
summary = lr.cmd_tick(args)
|
|
assert summary["halted"] is True
|
|
assert summary["reason"] == "context_below_floor"
|
|
s2 = _state(lp)
|
|
assert s2["status"] == "halted"
|
|
assert s2["halt_reason"] == "human_intervention"
|
|
# Implement subprocess never started -- no harness invocation should be recorded.
|
|
assert not any("opencode" in " ".join(inv) for inv in fake_run.invocations)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R6 / idempotence: verifier parse failure
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestVerifierParseFailure:
|
|
def test_parse_failure_halts_no_state_advance(self, tmp_project, fake_run, ctx_ok):
|
|
lp = _make_loop(tmp_project)
|
|
s = _state(lp); s["current_task"] = "demo"; s["iteration_count"] = 5
|
|
_write_state(lp, s)
|
|
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
|
|
fake_run.rules.insert(0, ("test-verify", lambda argv: _make_completed("not valid json")))
|
|
args = _ns(loop="ci-loop", project=str(tmp_project))
|
|
summary = lr.cmd_tick(args)
|
|
assert summary["halted"] is True
|
|
assert "verifier_failed" in summary["reason"]
|
|
s2 = _state(lp)
|
|
assert s2["iteration_count"] == 5 # unchanged
|
|
assert s2["status"] == "halted"
|
|
assert s2["halt_reason"] == "verifier_failed"
|
|
|
|
def test_parse_fenced_json(self):
|
|
text = "```json\n{\"pass\": true, \"score\": 0.8}\n```"
|
|
v = lr.parse_verdict(text)
|
|
assert v is not None
|
|
assert v["pass"] is True
|
|
assert v["score"] == 0.8
|
|
|
|
def test_parse_with_line_comments(self):
|
|
text = "# verdict from verifier\n// signed: GLM-5\n{\"pass\": false, \"score\": 0.2, \"reasons\": [\"x\"]}"
|
|
v = lr.parse_verdict(text)
|
|
assert v is not None
|
|
assert v["pass"] is False
|
|
assert v["score"] == 0.2
|
|
assert v["reasons"] == ["x"]
|
|
|
|
def test_parse_missing_pass_key_returns_none(self):
|
|
text = "{\"score\": 0.5}" # missing 'pass'
|
|
v = lr.parse_verdict(text)
|
|
assert v is None
|
|
|
|
def test_parse_empty(self):
|
|
assert lr.parse_verdict("") is None
|
|
assert lr.parse_verdict(" ") is None
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R6 / score history capping
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestScoreHistory:
|
|
def test_score_history_capped(self, tmp_project, fake_run, ctx_ok):
|
|
lp = _make_loop(tmp_project, cfg_overrides={
|
|
"brakes": {"score_plateau_window": 3, "max_iterations": 25},
|
|
"roles": {
|
|
"implement": {"prompt": "test-impl.md"},
|
|
"verify": {"prompt": "test-verify.md"},
|
|
"orchestrate": {"prompt": "test-orch.md"}}})
|
|
s = _state(lp); s["current_task"] = "demo"
|
|
_write_state(lp, s)
|
|
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
|
|
scores = [0.5, 0.4, 0.4, 0.4, 0.4]
|
|
seen_iter_counts = []
|
|
for sc in scores:
|
|
fake_run.rules = [r for r in fake_run.rules if r[0] != "test-verify"]
|
|
fake_run.rules.insert(0, ("test-verify", lambda argv, _sc=sc: _make_completed(
|
|
json.dumps({"pass": False, "score": _sc}))))
|
|
args = _ns(loop="ci-loop", project=str(tmp_project))
|
|
summary = lr.cmd_tick(args)
|
|
seen_iter_counts.append((int(sc * 10), summary.get("skipped"), summary.get("halted"),
|
|
summary.get("reason"), summary.get("iter")))
|
|
s2 = _state(lp)
|
|
assert s2["iteration_count"] == 5, f"trace: {seen_iter_counts}"
|
|
assert len(s2["score_history"]) == 3
|
|
assert s2["score_history"] == [0.4, 0.4, 0.4]
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R4 / harness command substitution
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestHarnessSubstitution:
|
|
def test_custom_command_with_output_token(self, tmp_project, fake_run, ctx_ok):
|
|
lp = _make_loop(tmp_project, cfg_overrides={
|
|
"harness": {"command": ["my-harness", "--prompt", "{prompt}",
|
|
"--cwd", "{cwd}", "--out", "{output}",
|
|
"--artifact", "{artifact}"]},
|
|
"roles": {
|
|
"implement": {"prompt": "test-impl.md"},
|
|
"verify": {"prompt": "test-verify.md"},
|
|
"orchestrate": {"prompt": "test-orch.md"}}})
|
|
s = _state(lp); s["current_task"] = "demo"
|
|
_write_state(lp, s)
|
|
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
|
|
|
|
def harness_handler(argv):
|
|
class R:
|
|
pass
|
|
r = R()
|
|
r.stdout = ""
|
|
r.returncode = 0
|
|
return r
|
|
|
|
fake_run.rules.append(("my-harness", harness_handler))
|
|
# Verifier handler takes precedence over the generic my-harness rule.
|
|
# The verify role is identified by "verify-prompt" in the resolved
|
|
# prompt file path (e.g. <loop>/outputs/tick1-verify-prompt.md).
|
|
fake_run.rules.insert(0, ("verify-prompt", lambda argv: _make_completed(
|
|
json.dumps({"pass": True, "score": 0.5}))))
|
|
args = _ns(loop="ci-loop", project=str(tmp_project))
|
|
lr.cmd_tick(args)
|
|
# Each role invocation must include the closed-over cwd and prompt tokens.
|
|
seen_artifacts: list[str] = []
|
|
for inv in fake_run.invocations:
|
|
if "my-harness" in inv:
|
|
assert "--cwd" in inv
|
|
assert "--prompt" in inv
|
|
if "--artifact" in inv:
|
|
a_idx = inv.index("--artifact") + 1
|
|
if a_idx < len(inv) and inv[a_idx] != "{artifact}":
|
|
seen_artifacts.append(inv[a_idx])
|
|
# The verify role's invocation must carry the Implement output path via --artifact <path>.
|
|
verify_invocations = [inv for inv in fake_run.invocations
|
|
if "my-harness" in inv and "verify-prompt" in " ".join(inv)]
|
|
assert any(a.endswith("-implement.json") for a in seen_artifacts), (
|
|
f"verify role must receive the implement artifact path via --artifact; "
|
|
f"saw: {seen_artifacts}")
|
|
assert verify_invocations, "no verify-role invocation captured"
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R3 / daemon mode
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestDaemonMode:
|
|
def test_daemon_runs_n_iterations(self, tmp_project, fake_run, ctx_ok, monkeypatch):
|
|
lp = _make_loop(tmp_project, cfg_overrides={"roles": {
|
|
"implement": {"prompt": "test-impl.md"},
|
|
"verify": {"prompt": "test-verify.md"},
|
|
"orchestrate": {"prompt": "test-orch.md"}}})
|
|
s = _state(lp); s["current_task"] = "demo"
|
|
_write_state(lp, s)
|
|
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
|
|
fake_run.rules.insert(0, ("test-verify", lambda argv: _make_completed(
|
|
json.dumps({"pass": True, "score": 0.5}))))
|
|
# Speed up sleeps.
|
|
monkeypatch.setattr(time, "sleep", lambda s: None)
|
|
args = _ns(loop="ci-loop", project=str(tmp_project),
|
|
mode="daemon", max_iterations=3, interval=1)
|
|
rc = lr.cmd_daemon(args)
|
|
assert rc == 0
|
|
assert _state(lp)["iteration_count"] == 3
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R7 / orchestrator-ordering
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestOrchestratorOrdering:
|
|
def test_roles_invoked_in_order(self, tmp_project, fake_run, ctx_ok):
|
|
lp = _make_loop(tmp_project, cfg_overrides={"roles": {
|
|
"implement": {"prompt": "test-impl.md"},
|
|
"verify": {"prompt": "test-verify.md"},
|
|
"orchestrate": {"prompt": "test-orch.md"}}})
|
|
s = _state(lp); s["current_task"] = "demo"
|
|
_write_state(lp, s)
|
|
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
|
|
order: list[str] = []
|
|
|
|
def make_handler(tag):
|
|
def handler(argv):
|
|
order.append(tag)
|
|
class R:
|
|
pass
|
|
r = R(); r.stdout = ""; r.returncode = 0
|
|
return r
|
|
return handler
|
|
fake_run.rules.append(("test-impl", make_handler("implement")))
|
|
fake_run.rules.append(("test-verify", lambda argv: (
|
|
order.append("verify"),
|
|
_make_completed(json.dumps({"pass": True, "score": 0.5})))[1]))
|
|
fake_run.rules.append(("test-orch", make_handler("orchestrate")))
|
|
args = _ns(loop="ci-loop", project=str(tmp_project))
|
|
lr.cmd_tick(args)
|
|
assert order == ["check-gate-implied", "implement", "verify", "orchestrate"] or \
|
|
order == ["implement", "verify", "orchestrate"]
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R7 / JSON output
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestJsonOutput:
|
|
def test_json_output_emits_summary(self, tmp_project, fake_run, ctx_ok, capsys, monkeypatch):
|
|
lp = _make_loop(tmp_project, cfg_overrides={"roles": {
|
|
"implement": {"prompt": "test-impl.md"},
|
|
"verify": {"prompt": "test-verify.md"},
|
|
"orchestrate": {"prompt": "test-orch.md"}}})
|
|
s = _state(lp); s["current_task"] = "demo"
|
|
_write_state(lp, s)
|
|
fake_run.add_simple("--check-gate", json.dumps({"ok": True}))
|
|
fake_run.rules.insert(0, ("test-verify", lambda argv: _make_completed(
|
|
json.dumps({"pass": True, "score": 0.7}))))
|
|
monkeypatch.setattr(sys, "argv", [
|
|
"loop-runner.py", "--mode", "tick", "--loop", "ci-loop",
|
|
"--project", str(tmp_project), "--json"])
|
|
rc = lr.main()
|
|
assert rc == 0
|
|
out = capsys.readouterr().out
|
|
payload = json.loads(out.splitlines()[-1])
|
|
assert payload["loop"] == "ci-loop"
|
|
assert payload["skipped"] is False
|
|
assert payload["iter"] == 1
|
|
assert payload["verdict"]["pass"] is True
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# helpers
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _make_completed(stdout: str):
|
|
class R:
|
|
pass
|
|
r = R()
|
|
r.stdout = stdout
|
|
r.returncode = 0
|
|
return r
|
|
|
|
|
|
class _Args:
|
|
def __init__(self, **kw):
|
|
self.__dict__.update(kw)
|
|
|
|
|
|
def _ns(loop: str, project: str, mode: str = "tick", interval: Optional[int] = None,
|
|
max_iterations: Optional[int] = None, json_output: bool = False) -> _Args:
|
|
return _Args(loop=loop, project=project, mode=mode, interval=interval,
|
|
max_iterations=max_iterations, json_output=json_output) |