"""Tests for task add-goal-mode. Covers R1-R8 from tasks/add-goal-mode/SPEC.md plus one regression test for backward compat with task-3 fixtures. All subprocess calls stubbed via monkeypatch; no live LLM in CI. """ import json import subprocess import sys import importlib.util from pathlib import Path from typing import Optional import pytest _RUNNER_PATH = Path.home() / ".automaton" / "scripts" / "loop-runner.py" _STATUS_PATH = Path.home() / ".automaton" / "scripts" / "status.py" _spec = importlib.util.spec_from_file_location("loop_runner_gm", _RUNNER_PATH) lr = importlib.util.module_from_spec(_spec) _spec.loader.exec_module(lr) _TEMPLATE_PATH = Path.home() / ".automaton" / "templates" / "loops" / "ci-triage" / "loop.json" def _state(loop_path: Path) -> dict: return json.loads((loop_path / ".state.loop").read_text()) def _write_state(loop_path: Path, state: dict) -> None: (loop_path / ".state.loop").write_text(json.dumps(state, indent=2, sort_keys=True) + "\n") def _make_loop(project: Path, name: str = "gm-loop", cfg_overrides: Optional[dict] = None, state_overrides: Optional[dict] = None) -> Path: lp = project / ".automaton" / "loops" / name lp.mkdir(parents=True, exist_ok=True) cfg = { "name": name, "description": "test loop", "schedule": {"interval_seconds": 3600}, "brakes": {"max_iterations": 25, "max_budget_usd": None, "score_plateau_window": 5}, "blast_radius": {"file_scope": [], "use_worktree": False}, "work_source": {"kind": "single"}, "acceptance_criteria": ["spec implemented", "tests pass"], "roles": { "implement": {"prompt": "test-impl.md"}, "verify": {"prompt": "test-verify.md"}, "orchestrate": {"prompt": "test-orch.md"}, }, } if cfg_overrides: cfg.update(cfg_overrides) (lp / "loop.json").write_text(json.dumps(cfg, indent=2) + "\n") state = { "schema_version": 1, "name": name, "status": "running", "halt_reason": None, "iteration_count": 0, "resumed_count": 0, "last_tick_at": None, "last_verdict": None, "score_history": [], "current_task": None, "worktree_branch": None, "worktree_path": None, } if state_overrides: state.update(state_overrides) (lp / ".state.loop").write_text(json.dumps(state, indent=2, sort_keys=True) + "\n") (lp / ".state.log").write_text("") roles = cfg.get("roles") or {} if isinstance(roles, dict): for role_cfg in roles.values(): if isinstance(role_cfg, dict) and role_cfg.get("prompt"): prompt_ref = role_cfg["prompt"] if not (Path.home() / ".automaton" / "prompts" / prompt_ref).exists(): try: (lp / prompt_ref).write_text(f"prompt: {prompt_ref}\n") except OSError: pass return lp def _make_task(project: Path, name: str, brief: str = "") -> Path: """Directly create a task dir + .state (no subprocess; works under monkeypatch).""" tp = project / ".automaton" / "tasks" / name tp.mkdir(parents=True, exist_ok=True) (tp / ".state").write_text("implement\n") if brief: (tp / "RESEARCH.md").write_text(brief) return tp class _FakeSubprocess: def __init__(self): self.rules: list[tuple[str, callable]] = [] self.invocations: list[list[str]] = [] def add(self, needle: str, handler: callable) -> None: self.rules.append((needle, handler)) def add_simple(self, needle: str, stdout: str = "", rc: int = 0) -> None: def handler(argv): class R: pass r = R() r.stdout = stdout r.returncode = rc return r self.rules.append((needle, handler)) def run(self, argv, *args, **kwargs): self.invocations.append(list(argv)) for needle, handler in self.rules: if any(needle in a for a in argv): return handler(argv) class R: pass r = R() r.stdout = "" r.returncode = 0 return r @pytest.fixture def tmp_project(tmp_path): (tmp_path / ".automaton" / "tasks").mkdir(parents=True) return tmp_path @pytest.fixture def fake_run(monkeypatch): fake = _FakeSubprocess() monkeypatch.setattr(subprocess, "run", fake.run) return fake def _gate_ok(fake: _FakeSubprocess): fake.add_simple("--check-gate", json.dumps({"ok": True})) def _ctx_ok(fake: _FakeSubprocess): fake.add_simple("--loop-mode", json.dumps({"loop_mode_eligible": True})) def _verdict_stdout(p: bool = True, score: float = 0.9, hint: str = "") -> str: body = {"pass": p, "score": score} if hint: body["next_hint"] = hint return json.dumps(body) def _tick_args(loop_name: str, project: Path): class A: mode = "tick" a = A() a.mode = "tick" a.loop = loop_name a.project = str(project) a.json_output = False return a def _run_tick(loop_name: str, project: Path) -> dict: return lr.cmd_tick(_tick_args(loop_name, project)) # --------------------------------------------------------------------------- # R1 -- find_work dispatch # --------------------------------------------------------------------------- class TestFindWorkDispatch: def test_find_work_single(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) _make_task(tmp_project, "task-a") lp = _make_loop(tmp_project, state_overrides={"current_task": "task-a"}) fake_run.add_simple("test-impl", _verdict_stdout(True, 0.9, "fix r1")) fake_run.add_simple("test-verify", _verdict_stdout(True, 0.9, "fix r1")) fake_run.add_simple("test-orch", "") summary = _run_tick("gm-loop", tmp_project) assert summary["skipped"] is False assert summary["iter"] == 1 assert _state(lp)["current_task"] == "task-a" def test_find_work_missing_work_source_falls_back_to_single(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) _make_task(tmp_project, "task-b") lp = _make_loop(tmp_project, cfg_overrides={"work_source": None}, state_overrides={"current_task": "task-b"}) fake_run.add_simple("test-impl", _verdict_stdout()) fake_run.add_simple("test-verify", _verdict_stdout()) fake_run.add_simple("test-orch", "") summary = _run_tick("gm-loop", tmp_project) assert summary["skipped"] is False def test_find_work_unknown_kind_warns_and_falls_back(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) _make_task(tmp_project, "task-c") lp = _make_loop(tmp_project, cfg_overrides={"work_source": {"kind": "bogus"}}, state_overrides={"current_task": "task-c"}) fake_run.add_simple("test-impl", _verdict_stdout()) fake_run.add_simple("test-verify", _verdict_stdout()) fake_run.add_simple("test-orch", "") summary = _run_tick("gm-loop", tmp_project) assert summary["skipped"] is False log = (lp / ".state.log").read_text() assert "unknown work_source.kind" in log # --------------------------------------------------------------------------- # R2 -- audit work_source # --------------------------------------------------------------------------- class TestAuditWorkSource: def test_audit_picks_highest_severity_violation(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) _make_task(tmp_project, "alpha") _make_task(tmp_project, "beta") audit_data = {"violations": [ {"category": 1, "severity": "low", "task": "alpha", "message": "low", "resolved": False}, {"category": 1, "severity": "high", "task": "beta", "message": "high", "resolved": False}, ], "loops": [], "total_tasks": 2, "untracked_tasks": 0} fake_run.add_simple("--audit", json.dumps(audit_data)) fake_run.add_simple("test-impl", _verdict_stdout()) fake_run.add_simple("test-verify", _verdict_stdout()) fake_run.add_simple("test-orch", "") lp = _make_loop(tmp_project, cfg_overrides={"work_source": {"kind": "audit"}}) summary = _run_tick("gm-loop", tmp_project) assert summary["skipped"] is False assert _state(lp)["current_task"] == "beta" def test_audit_creates_task_when_violation_has_no_task(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) audit_data = {"violations": [ {"category": 4, "severity": "high", "task": None, "message": "Some Broken Thing", "resolved": False}, ], "loops": [], "total_tasks": 0, "untracked_tasks": 1} fake_run.add_simple("--audit", json.dumps(audit_data)) fake_run.add_simple("--create-task", "") fake_run.add_simple("test-impl", _verdict_stdout()) fake_run.add_simple("test-verify", _verdict_stdout()) fake_run.add_simple("test-orch", "") lp = _make_loop(tmp_project, cfg_overrides={"work_source": {"kind": "audit"}}) summary = _run_tick("gm-loop", tmp_project) assert summary["skipped"] is False ct = _state(lp)["current_task"] assert ct and ct.startswith("some") def test_audit_skip_when_no_violations(self, tmp_project, fake_run): _gate_ok(fake_run) audit_data = {"violations": [], "loops": [], "total_tasks": 0, "untracked_tasks": 0} fake_run.add_simple("--audit", json.dumps(audit_data)) lp = _make_loop(tmp_project, cfg_overrides={"work_source": {"kind": "audit"}}) summary = _run_tick("gm-loop", tmp_project) assert summary["skipped"] is True assert summary["reason"] == "no_work" def test_audit_uses_work_source_project(self, tmp_project, fake_run, tmp_path_factory): _gate_ok(fake_run) other_project = tmp_path_factory.mktemp("other-proj") (other_project / ".automaton" / "tasks").mkdir(parents=True) _make_task(other_project, "remote-task") audit_data = {"violations": [ {"category": 1, "severity": "high", "task": "remote-task", "message": "boom", "resolved": False}, ], "loops": [], "total_tasks": 1, "untracked_tasks": 0} fake_run.add_simple("--audit", json.dumps(audit_data)) fake_run.add_simple("test-impl", _verdict_stdout()) fake_run.add_simple("test-verify", _verdict_stdout()) fake_run.add_simple("test-orch", "") lp = _make_loop(tmp_project, cfg_overrides={ "work_source": {"kind": "audit", "project": str(other_project)}}) summary = _run_tick("gm-loop", tmp_project) assert summary["skipped"] is False audit_call = next(inv for inv in fake_run.invocations if "--audit" in inv) assert str(other_project) in audit_call # --------------------------------------------------------------------------- # R3 -- backlog work_source # --------------------------------------------------------------------------- class TestBacklogWorkSource: def test_backlog_picks_top_unchecked_item(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) design_dir = tmp_project / "design" / "loops" design_dir.mkdir(parents=True) (design_dir / "BACKLOG.md").write_text( "# Backlog\n\n- [x] done-item\n- [ ] **design-fix-x** some work\n- [ ] **design-fix-y** more work\n") _make_task(tmp_project, "design-fix-x") lp = _make_loop(tmp_project, cfg_overrides={"work_source": {"kind": "backlog", "area": "loops"}}) fake_run.add_simple("test-impl", _verdict_stdout()) fake_run.add_simple("test-verify", _verdict_stdout()) fake_run.add_simple("test-orch", "") summary = _run_tick("gm-loop", tmp_project) assert summary["skipped"] is False assert _state(lp)["current_task"] == "design-fix-x" def test_backlog_skip_when_empty(self, tmp_project, fake_run): _gate_ok(fake_run) design_dir = tmp_project / "design" / "loops" design_dir.mkdir(parents=True) (design_dir / "BACKLOG.md").write_text("# Backlog\n\n- [x] all done\n") lp = _make_loop(tmp_project, cfg_overrides={"work_source": {"kind": "backlog", "area": "loops"}}) summary = _run_tick("gm-loop", tmp_project) assert summary["skipped"] is True assert summary["reason"] == "no_work" def test_backlog_uses_area_path(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) design_dir = tmp_project / "design" / "context-sizing" design_dir.mkdir(parents=True) (design_dir / "BACKLOG.md").write_text( "- [ ] **context-fix-q** next item\n") _make_task(tmp_project, "context-fix-q") lp = _make_loop(tmp_project, cfg_overrides={ "work_source": {"kind": "backlog", "area": "context-sizing"}}) fake_run.add_simple("test-impl", _verdict_stdout()) fake_run.add_simple("test-verify", _verdict_stdout()) fake_run.add_simple("test-orch", "") summary = _run_tick("gm-loop", tmp_project) assert summary["skipped"] is False assert _state(lp)["current_task"] == "context-fix-q" # --------------------------------------------------------------------------- # R4 -- verifier-prompt tokens # --------------------------------------------------------------------------- class TestVerifierTokens: def test_task_brief_substituted_from_research(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) _make_task(tmp_project, "dt", brief="THE-BRIEF-MARKER") _make_loop(tmp_project, state_overrides={"current_task": "dt"}, cfg_overrides={"harness": {"command": ["echo", "{prompt}", "{cwd}", "{task_brief}"]}}) captured = {} def capture_harness(role_marker): def handler(argv): prompt_path = None for tok in argv: if tok and tok.endswith((".md", ".txt")) or "/" in tok or "\\" in tok: if role_marker in str(tok) or True: pass for i, t in enumerate(argv): if t == "--prompt-file" and i + 1 < len(argv): prompt_path = argv[i + 1] class R: pass r = R() r.stdout = _verdict_stdout() if role_marker == "test-verify" else "" r.returncode = 0 captured.setdefault(role_marker, []).append({"prompt": prompt_path, "argv": list(argv)}) return r return handler fake_run.add("implement-prompt", capture_harness("test-impl")) fake_run.add("verify-prompt", capture_harness("test-verify")) fake_run.add("orchestrate-prompt", capture_harness("test-orch")) _run_tick("gm-loop", tmp_project) impl_argv = captured["test-impl"][0]["argv"] verify_argv = captured["test-verify"][0]["argv"] assert "THE-BRIEF-MARKER" in " ".join(impl_argv) assert "THE-BRIEF-MARKER" in " ".join(verify_argv) def test_acceptance_criteria_substituted_from_loop_json_list(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) _make_task(tmp_project, "ac") _make_loop(tmp_project, cfg_overrides={"acceptance_criteria": ["CRIT-A", "CRIT-B"], "harness": {"command": ["echo", "{prompt}", "{cwd}", "{acceptance_criteria}"]}}, state_overrides={"current_task": "ac"}) captured = {} def capture(role_marker): def handler(argv): captured.setdefault(role_marker, []).append(list(argv)) class R: pass r = R() r.stdout = _verdict_stdout() if role_marker == "test-verify" else "" r.returncode = 0 return r return handler fake_run.add("implement-prompt", capture("test-impl")) fake_run.add("verify-prompt", capture("test-verify")) fake_run.add("orchestrate-prompt", capture("test-orch")) _run_tick("gm-loop", tmp_project) impl_text = " ".join(captured["test-impl"][0]) verify_text = " ".join(captured["test-verify"][0]) assert "CRIT-A" in impl_text assert "CRIT-B" in verify_text def test_next_hint_substituted_from_last_verdict(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) _make_task(tmp_project, "nh") _make_loop(tmp_project, state_overrides={ "current_task": "nh", "last_verdict": {"pass": True, "score": 0.5, "next_hint": "PRIOR-HINT-MARKER"}, }, cfg_overrides={"harness": {"command": ["echo", "{prompt}", "{cwd}", "{next_hint}"]}}) captured = {} def capture(role_marker): def handler(argv): captured.setdefault(role_marker, []).append(list(argv)) class R: pass r = R() r.stdout = _verdict_stdout() if role_marker == "test-verify" else "" r.returncode = 0 return r return handler fake_run.add("implement-prompt", capture("test-impl")) fake_run.add("verify-prompt", capture("test-verify")) fake_run.add("orchestrate-prompt", capture("test-orch")) _run_tick("gm-loop", tmp_project) impl_text = " ".join(captured["test-impl"][0]) assert "PRIOR-HINT-MARKER" in impl_text def test_missing_tokens_leave_prompt_intact(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) _make_task(tmp_project, "mt") _make_loop(tmp_project, cfg_overrides={"acceptance_criteria": None}, state_overrides={"current_task": "mt", "last_verdict": None}) captured = {} def capture(role_marker): def handler(argv): captured.setdefault(role_marker, []).append(list(argv)) class R: pass r = R() r.stdout = _verdict_stdout() if role_marker == "test-verify" else "" r.returncode = 0 return r return handler fake_run.add("test-impl", capture("test-impl")) fake_run.add("test-verify", capture("test-verify")) fake_run.add("test-orch", capture("test-orch")) summary = _run_tick("gm-loop", tmp_project) assert summary["skipped"] is False impl_text = " ".join(captured["test-impl"][0]) assert "{task_brief}" not in impl_text assert "{acceptance_criteria}" not in impl_text assert "{next_hint}" not in impl_text # --------------------------------------------------------------------------- # R5 -- _truncate_tokens # --------------------------------------------------------------------------- class TestTruncateTokens: def test_truncate_short_text_unchanged(self): assert lr._truncate_tokens("hello world", 100) == "hello world" def test_truncate_long_text_capped_with_marker(self): long_text = "x" * 1000 out = lr._truncate_tokens(long_text, 10) assert out.endswith("…[truncated]") assert len(out) <= 40 + len(" …[truncated]") def test_truncate_returns_empty_for_empty_input(self): assert lr._truncate_tokens("", 100) == "" # --------------------------------------------------------------------------- # R6 -- next_hint feedback loop # --------------------------------------------------------------------------- class TestNextHintFeedback: def test_next_hint_fed_into_next_tick_implement(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) _make_task(tmp_project, "fb") lp = _make_loop(tmp_project, state_overrides={"current_task": "fb"}) fake_run.add_simple("test-impl", _verdict_stdout(True, 0.9, "carry-this-hint")) fake_run.add_simple("test-verify", _verdict_stdout(True, 0.95, "carry-this-hint")) fake_run.add_simple("test-orch", "") _run_tick("gm-loop", tmp_project) st = _state(lp) assert st["last_verdict"]["next_hint"] == "carry-this-hint" def test_first_tick_has_empty_next_hint(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) _make_task(tmp_project, "ft") _make_loop(tmp_project, state_overrides={"current_task": "ft", "last_verdict": None}) captured = {} def capture(role_marker): def handler(argv): captured.setdefault(role_marker, []).append(list(argv)) class R: pass r = R() r.stdout = _verdict_stdout() if role_marker == "test-verify" else "" r.returncode = 0 return r return handler fake_run.add("test-impl", capture("test-impl")) fake_run.add("test-verify", capture("test-verify")) fake_run.add("test-orch", capture("test-orch")) _run_tick("gm-loop", tmp_project) impl_text = " ".join(captured["test-impl"][0]) marker_count = impl_text.count("{next_hint}") assert marker_count == 0 # --------------------------------------------------------------------------- # R7 -- loop.json schema additions # --------------------------------------------------------------------------- class TestLoopJsonSchemaAdditions: def test_ci_triage_template_has_work_source(self): cfg = json.loads(_TEMPLATE_PATH.read_text()) assert cfg.get("work_source", {}).get("kind") == "single" def test_ci_triage_template_has_acceptance_criteria(self): cfg = json.loads(_TEMPLATE_PATH.read_text()) ac = cfg.get("acceptance_criteria") assert isinstance(ac, list) and len(ac) >= 1 def test_create_loop_preserves_acceptance_criteria(self, tmp_project, fake_run): src = Path.home() / ".automaton" / "templates" / "loops" / "ci-triage" from_path = src loop_dir = tmp_project / ".automaton" / "loops" / "tla" loop_dir.mkdir(parents=True) loop_cfg = json.loads((from_path / "loop.json").read_text()) loop_cfg["name"] = "tla" (loop_dir / "loop.json").write_text(json.dumps(loop_cfg, indent=2) + "\n") (loop_dir / ".state.loop").write_text(json.dumps({ "schema_version": 1, "name": "tla", "status": "running", "halt_reason": None, "iteration_count": 0, "resumed_count": 0, "last_tick_at": None, "last_verdict": None, "score_history": [], "current_task": None, "worktree_branch": None, "worktree_path": None, }, indent=2, sort_keys=True) + "\n") (loop_dir / ".state.log").write_text("") assert json.loads((loop_dir / "loop.json").read_text()).get("acceptance_criteria") # --------------------------------------------------------------------------- # R8 -- status.py --audit --json # --------------------------------------------------------------------------- def _status_module(): spec = importlib.util.spec_from_file_location("status_gm", _STATUS_PATH) mod = importlib.util.module_from_spec(spec) spec.loader.exec_module(mod) return mod class TestAuditJson: def test_audit_json_emits_violations_array(self, tmp_project, monkeypatch): st = _status_module() task = tmp_project / ".automaton" / "tasks" / "z" task.mkdir(parents=True) (task / ".state").write_text("implement\n") (task / "IMPLEMENTATION.md").write_text("") class A: json_output = True audit = True project = str(tmp_project) import io from contextlib import redirect_stdout buf = io.StringIO() with redirect_stdout(buf): rc = st.cmd_audit(A()) out = buf.getvalue().strip().splitlines()[-1] data = json.loads(out) assert "violations" in data assert isinstance(data["violations"], list) assert data["total_tasks"] == 1 assert any(v["category"] == 2 for v in data["violations"]) def test_audit_json_includes_loops_block(self, tmp_project): st = _status_module() lp = tmp_project / ".automaton" / "loops" / "zloop" lp.mkdir(parents=True) (lp / "loop.json").write_text(json.dumps({"name": "zloop"})) (lp / ".state.loop").write_text(json.dumps({ "schema_version": 1, "name": "zloop", "status": "running", "halt_reason": None, "iteration_count": 0, "resumed_count": 0, "last_tick_at": None, "last_verdict": None, "score_history": [], "current_task": None, "worktree_branch": None, "worktree_path": None, }, indent=2, sort_keys=True) + "\n") class A: json_output = True audit = True project = str(tmp_project) import io from contextlib import redirect_stdout buf = io.StringIO() with redirect_stdout(buf): st.cmd_audit(A()) data = json.loads(buf.getvalue().strip().splitlines()[-1]) assert any(loop["name"] == "zloop" for loop in data["loops"]) def test_audit_json_pickable_by_runner_run_json(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) _make_task(tmp_project, "pk") audit_data = {"violations": [ {"category": 1, "severity": "high", "task": "pk", "message": "x", "resolved": False}, ], "loops": [], "total_tasks": 1, "untracked_tasks": 0} fake_run.add_simple("--audit", json.dumps(audit_data)) fake_run.add_simple("test-impl", _verdict_stdout()) fake_run.add_simple("test-verify", _verdict_stdout()) fake_run.add_simple("test-orch", "") _make_loop(tmp_project, cfg_overrides={"work_source": {"kind": "audit"}}) summary = _run_tick("gm-loop", tmp_project) assert summary["skipped"] is False assert summary["iter"] == 1 # --------------------------------------------------------------------------- # Regression -- existing single loop ticks unchanged # --------------------------------------------------------------------------- class TestRegressionBackwardCompat: def test_existing_single_work_source_loop_ticks_unchanged(self, tmp_project, fake_run): _gate_ok(fake_run) _ctx_ok(fake_run) _make_task(tmp_project, "reg") _make_loop(tmp_project, cfg_overrides={ "work_source": None, "acceptance_criteria": None, }, state_overrides={"current_task": "reg"}) fake_run.add_simple("test-impl", _verdict_stdout(True, 0.9, "n")) fake_run.add_simple("test-verify", _verdict_stdout(True, 0.9, "n")) fake_run.add_simple("test-orch", "") summary = _run_tick("gm-loop", tmp_project) assert summary["skipped"] is False assert summary["iter"] == 1 assert summary["verdict"]["pass"] is True