673 lines
27 KiB
Python
673 lines
27 KiB
Python
"""Tests for task add-goal-mode.
|
|||
|
|
|
||
|
|
Covers R1-R8 from tasks/add-goal-mode/SPEC.md plus one regression test for
|
||
|
|
backward compat with task-3 fixtures. All subprocess calls stubbed via
|
||
|
|
monkeypatch; no live LLM in CI.
|
||
|
|
"""
|
||
|
|
|
||
|
|
import json
|
||
|
|
import subprocess
|
||
|
|
import sys
|
||
|
|
import importlib.util
|
||
|
|
from pathlib import Path
|
||
|
|
from typing import Optional
|
||
|
|
|
||
|
|
import pytest
|
||
|
|
|
||
|
|
_RUNNER_PATH = Path.home() / ".automaton" / "scripts" / "loop-runner.py"
|
||
|
|
_STATUS_PATH = Path.home() / ".automaton" / "scripts" / "status.py"
|
||
|
|
_spec = importlib.util.spec_from_file_location("loop_runner_gm", _RUNNER_PATH)
|
||
|
|
lr = importlib.util.module_from_spec(_spec)
|
||
|
|
_spec.loader.exec_module(lr)
|
||
|
|
|
||
|
|
_TEMPLATE_PATH = Path.home() / ".automaton" / "templates" / "loops" / "ci-triage" / "loop.json"
|
||
|
|
|
||
|
|
|
||
|
|
def _state(loop_path: Path) -> dict:
|
||
|
|
return json.loads((loop_path / ".state.loop").read_text())
|
||
|
|
|
||
|
|
|
||
|
|
def _write_state(loop_path: Path, state: dict) -> None:
|
||
|
|
(loop_path / ".state.loop").write_text(json.dumps(state, indent=2, sort_keys=True) + "\n")
|
||
|
|
|
||
|
|
|
||
|
|
def _make_loop(project: Path, name: str = "gm-loop",
|
||
|
|
cfg_overrides: Optional[dict] = None,
|
||
|
|
state_overrides: Optional[dict] = None) -> Path:
|
||
|
|
lp = project / ".automaton" / "loops" / name
|
||
|
|
lp.mkdir(parents=True, exist_ok=True)
|
||
|
|
|
||
|
|
cfg = {
|
||
|
|
"name": name,
|
||
|
|
"description": "test loop",
|
||
|
|
"schedule": {"interval_seconds": 3600},
|
||
|
|
"brakes": {"max_iterations": 25, "max_budget_usd": None,
|
||
|
|
"score_plateau_window": 5},
|
||
|
|
"blast_radius": {"file_scope": [], "use_worktree": False},
|
||
|
|
"work_source": {"kind": "single"},
|
||
|
|
"acceptance_criteria": ["spec implemented", "tests pass"],
|
||
|
|
"roles": {
|
||
|
|
"implement": {"prompt": "test-impl.md"},
|
||
|
|
"verify": {"prompt": "test-verify.md"},
|
||
|
|
"orchestrate": {"prompt": "test-orch.md"},
|
||
|
|
},
|
||
|
|
}
|
||
|
|
if cfg_overrides:
|
||
|
|
cfg.update(cfg_overrides)
|
||
|
|
(lp / "loop.json").write_text(json.dumps(cfg, indent=2) + "\n")
|
||
|
|
|
||
|
|
state = {
|
||
|
|
"schema_version": 1, "name": name, "status": "running",
|
||
|
|
"halt_reason": None, "iteration_count": 0, "resumed_count": 0,
|
||
|
|
"last_tick_at": None, "last_verdict": None, "score_history": [],
|
||
|
|
"current_task": None, "worktree_branch": None, "worktree_path": None,
|
||
|
|
}
|
||
|
|
if state_overrides:
|
||
|
|
state.update(state_overrides)
|
||
|
|
(lp / ".state.loop").write_text(json.dumps(state, indent=2, sort_keys=True) + "\n")
|
||
|
|
(lp / ".state.log").write_text("")
|
||
|
|
roles = cfg.get("roles") or {}
|
||
|
|
if isinstance(roles, dict):
|
||
|
|
for role_cfg in roles.values():
|
||
|
|
if isinstance(role_cfg, dict) and role_cfg.get("prompt"):
|
||
|
|
prompt_ref = role_cfg["prompt"]
|
||
|
|
if not (Path.home() / ".automaton" / "prompts" / prompt_ref).exists():
|
||
|
|
try:
|
||
|
|
(lp / prompt_ref).write_text(f"prompt: {prompt_ref}\n")
|
||
|
|
except OSError:
|
||
|
|
pass
|
||
|
|
return lp
|
||
|
|
|
||
|
|
|
||
|
|
def _make_task(project: Path, name: str, brief: str = "") -> Path:
|
||
|
|
"""Directly create a task dir + .state (no subprocess; works under monkeypatch)."""
|
||
|
|
tp = project / ".automaton" / "tasks" / name
|
||
|
|
tp.mkdir(parents=True, exist_ok=True)
|
||
|
|
(tp / ".state").write_text("implement\n")
|
||
|
|
if brief:
|
||
|
|
(tp / "RESEARCH.md").write_text(brief)
|
||
|
|
return tp
|
||
|
|
|
||
|
|
|
||
|
|
class _FakeSubprocess:
|
||
|
|
def __init__(self):
|
||
|
|
self.rules: list[tuple[str, callable]] = []
|
||
|
|
self.invocations: list[list[str]] = []
|
||
|
|
|
||
|
|
def add(self, needle: str, handler: callable) -> None:
|
||
|
|
self.rules.append((needle, handler))
|
||
|
|
|
||
|
|
def add_simple(self, needle: str, stdout: str = "", rc: int = 0) -> None:
|
||
|
|
def handler(argv):
|
||
|
|
class R:
|
||
|
|
pass
|
||
|
|
r = R()
|
||
|
|
r.stdout = stdout
|
||
|
|
r.returncode = rc
|
||
|
|
return r
|
||
|
|
self.rules.append((needle, handler))
|
||
|
|
|
||
|
|
def run(self, argv, *args, **kwargs):
|
||
|
|
self.invocations.append(list(argv))
|
||
|
|
for needle, handler in self.rules:
|
||
|
|
if any(needle in a for a in argv):
|
||
|
|
return handler(argv)
|
||
|
|
class R:
|
||
|
|
pass
|
||
|
|
r = R()
|
||
|
|
r.stdout = ""
|
||
|
|
r.returncode = 0
|
||
|
|
return r
|
||
|
|
|
||
|
|
|
||
|
|
@pytest.fixture
|
||
|
|
def tmp_project(tmp_path):
|
||
|
|
(tmp_path / ".automaton" / "tasks").mkdir(parents=True)
|
||
|
|
return tmp_path
|
||
|
|
|
||
|
|
|
||
|
|
@pytest.fixture
|
||
|
|
def fake_run(monkeypatch):
|
||
|
|
fake = _FakeSubprocess()
|
||
|
|
monkeypatch.setattr(subprocess, "run", fake.run)
|
||
|
|
return fake
|
||
|
|
|
||
|
|
|
||
|
|
def _gate_ok(fake: _FakeSubprocess):
|
||
|
|
fake.add_simple("--check-gate", json.dumps({"ok": True}))
|
||
|
|
|
||
|
|
|
||
|
|
def _ctx_ok(fake: _FakeSubprocess):
|
||
|
|
fake.add_simple("--loop-mode", json.dumps({"loop_mode_eligible": True}))
|
||
|
|
|
||
|
|
|
||
|
|
def _verdict_stdout(p: bool = True, score: float = 0.9, hint: str = "") -> str:
|
||
|
|
body = {"pass": p, "score": score}
|
||
|
|
if hint:
|
||
|
|
body["next_hint"] = hint
|
||
|
|
return json.dumps(body)
|
||
|
|
|
||
|
|
|
||
|
|
def _tick_args(loop_name: str, project: Path):
|
||
|
|
class A:
|
||
|
|
mode = "tick"
|
||
|
|
a = A()
|
||
|
|
a.mode = "tick"
|
||
|
|
a.loop = loop_name
|
||
|
|
a.project = str(project)
|
||
|
|
a.json_output = False
|
||
|
|
return a
|
||
|
|
|
||
|
|
|
||
|
|
def _run_tick(loop_name: str, project: Path) -> dict:
|
||
|
|
return lr.cmd_tick(_tick_args(loop_name, project))
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# R1 -- find_work dispatch
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
class TestFindWorkDispatch:
|
||
|
|
def test_find_work_single(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
_make_task(tmp_project, "task-a")
|
||
|
|
lp = _make_loop(tmp_project, state_overrides={"current_task": "task-a"})
|
||
|
|
fake_run.add_simple("test-impl", _verdict_stdout(True, 0.9, "fix r1"))
|
||
|
|
fake_run.add_simple("test-verify", _verdict_stdout(True, 0.9, "fix r1"))
|
||
|
|
fake_run.add_simple("test-orch", "")
|
||
|
|
summary = _run_tick("gm-loop", tmp_project)
|
||
|
|
assert summary["skipped"] is False
|
||
|
|
assert summary["iter"] == 1
|
||
|
|
assert _state(lp)["current_task"] == "task-a"
|
||
|
|
|
||
|
|
def test_find_work_missing_work_source_falls_back_to_single(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
_make_task(tmp_project, "task-b")
|
||
|
|
lp = _make_loop(tmp_project, cfg_overrides={"work_source": None},
|
||
|
|
state_overrides={"current_task": "task-b"})
|
||
|
|
fake_run.add_simple("test-impl", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-verify", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-orch", "")
|
||
|
|
summary = _run_tick("gm-loop", tmp_project)
|
||
|
|
assert summary["skipped"] is False
|
||
|
|
|
||
|
|
def test_find_work_unknown_kind_warns_and_falls_back(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
_make_task(tmp_project, "task-c")
|
||
|
|
lp = _make_loop(tmp_project,
|
||
|
|
cfg_overrides={"work_source": {"kind": "bogus"}},
|
||
|
|
state_overrides={"current_task": "task-c"})
|
||
|
|
fake_run.add_simple("test-impl", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-verify", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-orch", "")
|
||
|
|
summary = _run_tick("gm-loop", tmp_project)
|
||
|
|
assert summary["skipped"] is False
|
||
|
|
log = (lp / ".state.log").read_text()
|
||
|
|
assert "unknown work_source.kind" in log
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# R2 -- audit work_source
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
class TestAuditWorkSource:
|
||
|
|
def test_audit_picks_highest_severity_violation(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
_make_task(tmp_project, "alpha")
|
||
|
|
_make_task(tmp_project, "beta")
|
||
|
|
audit_data = {"violations": [
|
||
|
|
{"category": 1, "severity": "low", "task": "alpha", "message": "low", "resolved": False},
|
||
|
|
{"category": 1, "severity": "high", "task": "beta", "message": "high", "resolved": False},
|
||
|
|
], "loops": [], "total_tasks": 2, "untracked_tasks": 0}
|
||
|
|
fake_run.add_simple("--audit", json.dumps(audit_data))
|
||
|
|
fake_run.add_simple("test-impl", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-verify", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-orch", "")
|
||
|
|
lp = _make_loop(tmp_project,
|
||
|
|
cfg_overrides={"work_source": {"kind": "audit"}})
|
||
|
|
summary = _run_tick("gm-loop", tmp_project)
|
||
|
|
assert summary["skipped"] is False
|
||
|
|
assert _state(lp)["current_task"] == "beta"
|
||
|
|
|
||
|
|
def test_audit_creates_task_when_violation_has_no_task(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
audit_data = {"violations": [
|
||
|
|
{"category": 4, "severity": "high", "task": None,
|
||
|
|
"message": "Some Broken Thing", "resolved": False},
|
||
|
|
], "loops": [], "total_tasks": 0, "untracked_tasks": 1}
|
||
|
|
fake_run.add_simple("--audit", json.dumps(audit_data))
|
||
|
|
fake_run.add_simple("--create-task", "")
|
||
|
|
fake_run.add_simple("test-impl", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-verify", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-orch", "")
|
||
|
|
lp = _make_loop(tmp_project,
|
||
|
|
cfg_overrides={"work_source": {"kind": "audit"}})
|
||
|
|
summary = _run_tick("gm-loop", tmp_project)
|
||
|
|
assert summary["skipped"] is False
|
||
|
|
ct = _state(lp)["current_task"]
|
||
|
|
assert ct and ct.startswith("some")
|
||
|
|
|
||
|
|
def test_audit_skip_when_no_violations(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
audit_data = {"violations": [], "loops": [],
|
||
|
|
"total_tasks": 0, "untracked_tasks": 0}
|
||
|
|
fake_run.add_simple("--audit", json.dumps(audit_data))
|
||
|
|
lp = _make_loop(tmp_project,
|
||
|
|
cfg_overrides={"work_source": {"kind": "audit"}})
|
||
|
|
summary = _run_tick("gm-loop", tmp_project)
|
||
|
|
assert summary["skipped"] is True
|
||
|
|
assert summary["reason"] == "no_work"
|
||
|
|
|
||
|
|
def test_audit_uses_work_source_project(self, tmp_project, fake_run, tmp_path_factory):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
other_project = tmp_path_factory.mktemp("other-proj")
|
||
|
|
(other_project / ".automaton" / "tasks").mkdir(parents=True)
|
||
|
|
_make_task(other_project, "remote-task")
|
||
|
|
audit_data = {"violations": [
|
||
|
|
{"category": 1, "severity": "high", "task": "remote-task",
|
||
|
|
"message": "boom", "resolved": False},
|
||
|
|
], "loops": [], "total_tasks": 1, "untracked_tasks": 0}
|
||
|
|
fake_run.add_simple("--audit", json.dumps(audit_data))
|
||
|
|
fake_run.add_simple("test-impl", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-verify", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-orch", "")
|
||
|
|
lp = _make_loop(tmp_project, cfg_overrides={
|
||
|
|
"work_source": {"kind": "audit", "project": str(other_project)}})
|
||
|
|
summary = _run_tick("gm-loop", tmp_project)
|
||
|
|
assert summary["skipped"] is False
|
||
|
|
audit_call = next(inv for inv in fake_run.invocations if "--audit" in inv)
|
||
|
|
assert str(other_project) in audit_call
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# R3 -- backlog work_source
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
class TestBacklogWorkSource:
|
||
|
|
def test_backlog_picks_top_unchecked_item(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
design_dir = tmp_project / "design" / "loops"
|
||
|
|
design_dir.mkdir(parents=True)
|
||
|
|
(design_dir / "BACKLOG.md").write_text(
|
||
|
|
"# Backlog\n\n- [x] done-item\n- [ ] **design-fix-x** some work\n- [ ] **design-fix-y** more work\n")
|
||
|
|
_make_task(tmp_project, "design-fix-x")
|
||
|
|
lp = _make_loop(tmp_project,
|
||
|
|
cfg_overrides={"work_source": {"kind": "backlog", "area": "loops"}})
|
||
|
|
fake_run.add_simple("test-impl", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-verify", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-orch", "")
|
||
|
|
summary = _run_tick("gm-loop", tmp_project)
|
||
|
|
assert summary["skipped"] is False
|
||
|
|
assert _state(lp)["current_task"] == "design-fix-x"
|
||
|
|
|
||
|
|
def test_backlog_skip_when_empty(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
design_dir = tmp_project / "design" / "loops"
|
||
|
|
design_dir.mkdir(parents=True)
|
||
|
|
(design_dir / "BACKLOG.md").write_text("# Backlog\n\n- [x] all done\n")
|
||
|
|
lp = _make_loop(tmp_project,
|
||
|
|
cfg_overrides={"work_source": {"kind": "backlog", "area": "loops"}})
|
||
|
|
summary = _run_tick("gm-loop", tmp_project)
|
||
|
|
assert summary["skipped"] is True
|
||
|
|
assert summary["reason"] == "no_work"
|
||
|
|
|
||
|
|
def test_backlog_uses_area_path(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
design_dir = tmp_project / "design" / "context-sizing"
|
||
|
|
design_dir.mkdir(parents=True)
|
||
|
|
(design_dir / "BACKLOG.md").write_text(
|
||
|
|
"- [ ] **context-fix-q** next item\n")
|
||
|
|
_make_task(tmp_project, "context-fix-q")
|
||
|
|
lp = _make_loop(tmp_project, cfg_overrides={
|
||
|
|
"work_source": {"kind": "backlog", "area": "context-sizing"}})
|
||
|
|
fake_run.add_simple("test-impl", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-verify", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-orch", "")
|
||
|
|
summary = _run_tick("gm-loop", tmp_project)
|
||
|
|
assert summary["skipped"] is False
|
||
|
|
assert _state(lp)["current_task"] == "context-fix-q"
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# R4 -- verifier-prompt tokens
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
class TestVerifierTokens:
|
||
|
|
def test_task_brief_substituted_from_research(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
_make_task(tmp_project, "dt", brief="THE-BRIEF-MARKER")
|
||
|
|
_make_loop(tmp_project, state_overrides={"current_task": "dt"},
|
||
|
|
cfg_overrides={"harness": {"command": ["echo", "{prompt}", "{cwd}", "{task_brief}"]}})
|
||
|
|
captured = {}
|
||
|
|
|
||
|
|
def capture_harness(role_marker):
|
||
|
|
def handler(argv):
|
||
|
|
prompt_path = None
|
||
|
|
for tok in argv:
|
||
|
|
if tok and tok.endswith((".md", ".txt")) or "/" in tok or "\\" in tok:
|
||
|
|
if role_marker in str(tok) or True:
|
||
|
|
pass
|
||
|
|
for i, t in enumerate(argv):
|
||
|
|
if t == "--prompt-file" and i + 1 < len(argv):
|
||
|
|
prompt_path = argv[i + 1]
|
||
|
|
class R:
|
||
|
|
pass
|
||
|
|
r = R()
|
||
|
|
r.stdout = _verdict_stdout() if role_marker == "test-verify" else ""
|
||
|
|
r.returncode = 0
|
||
|
|
captured.setdefault(role_marker, []).append({"prompt": prompt_path, "argv": list(argv)})
|
||
|
|
return r
|
||
|
|
return handler
|
||
|
|
|
||
|
|
fake_run.add("implement-prompt", capture_harness("test-impl"))
|
||
|
|
fake_run.add("verify-prompt", capture_harness("test-verify"))
|
||
|
|
fake_run.add("orchestrate-prompt", capture_harness("test-orch"))
|
||
|
|
|
||
|
|
_run_tick("gm-loop", tmp_project)
|
||
|
|
impl_argv = captured["test-impl"][0]["argv"]
|
||
|
|
verify_argv = captured["test-verify"][0]["argv"]
|
||
|
|
assert "THE-BRIEF-MARKER" in " ".join(impl_argv)
|
||
|
|
assert "THE-BRIEF-MARKER" in " ".join(verify_argv)
|
||
|
|
|
||
|
|
def test_acceptance_criteria_substituted_from_loop_json_list(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
_make_task(tmp_project, "ac")
|
||
|
|
_make_loop(tmp_project,
|
||
|
|
cfg_overrides={"acceptance_criteria": ["CRIT-A", "CRIT-B"],
|
||
|
|
"harness": {"command": ["echo", "{prompt}", "{cwd}", "{acceptance_criteria}"]}},
|
||
|
|
state_overrides={"current_task": "ac"})
|
||
|
|
captured = {}
|
||
|
|
|
||
|
|
def capture(role_marker):
|
||
|
|
def handler(argv):
|
||
|
|
captured.setdefault(role_marker, []).append(list(argv))
|
||
|
|
class R:
|
||
|
|
pass
|
||
|
|
r = R()
|
||
|
|
r.stdout = _verdict_stdout() if role_marker == "test-verify" else ""
|
||
|
|
r.returncode = 0
|
||
|
|
return r
|
||
|
|
return handler
|
||
|
|
|
||
|
|
fake_run.add("implement-prompt", capture("test-impl"))
|
||
|
|
fake_run.add("verify-prompt", capture("test-verify"))
|
||
|
|
fake_run.add("orchestrate-prompt", capture("test-orch"))
|
||
|
|
|
||
|
|
_run_tick("gm-loop", tmp_project)
|
||
|
|
impl_text = " ".join(captured["test-impl"][0])
|
||
|
|
verify_text = " ".join(captured["test-verify"][0])
|
||
|
|
assert "CRIT-A" in impl_text
|
||
|
|
assert "CRIT-B" in verify_text
|
||
|
|
|
||
|
|
def test_next_hint_substituted_from_last_verdict(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
_make_task(tmp_project, "nh")
|
||
|
|
_make_loop(tmp_project, state_overrides={
|
||
|
|
"current_task": "nh",
|
||
|
|
"last_verdict": {"pass": True, "score": 0.5, "next_hint": "PRIOR-HINT-MARKER"},
|
||
|
|
},
|
||
|
|
cfg_overrides={"harness": {"command": ["echo", "{prompt}", "{cwd}", "{next_hint}"]}})
|
||
|
|
captured = {}
|
||
|
|
|
||
|
|
def capture(role_marker):
|
||
|
|
def handler(argv):
|
||
|
|
captured.setdefault(role_marker, []).append(list(argv))
|
||
|
|
class R:
|
||
|
|
pass
|
||
|
|
r = R()
|
||
|
|
r.stdout = _verdict_stdout() if role_marker == "test-verify" else ""
|
||
|
|
r.returncode = 0
|
||
|
|
return r
|
||
|
|
return handler
|
||
|
|
|
||
|
|
fake_run.add("implement-prompt", capture("test-impl"))
|
||
|
|
fake_run.add("verify-prompt", capture("test-verify"))
|
||
|
|
fake_run.add("orchestrate-prompt", capture("test-orch"))
|
||
|
|
|
||
|
|
_run_tick("gm-loop", tmp_project)
|
||
|
|
impl_text = " ".join(captured["test-impl"][0])
|
||
|
|
assert "PRIOR-HINT-MARKER" in impl_text
|
||
|
|
|
||
|
|
def test_missing_tokens_leave_prompt_intact(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
_make_task(tmp_project, "mt")
|
||
|
|
_make_loop(tmp_project, cfg_overrides={"acceptance_criteria": None},
|
||
|
|
state_overrides={"current_task": "mt",
|
||
|
|
"last_verdict": None})
|
||
|
|
captured = {}
|
||
|
|
|
||
|
|
def capture(role_marker):
|
||
|
|
def handler(argv):
|
||
|
|
captured.setdefault(role_marker, []).append(list(argv))
|
||
|
|
class R:
|
||
|
|
pass
|
||
|
|
r = R()
|
||
|
|
r.stdout = _verdict_stdout() if role_marker == "test-verify" else ""
|
||
|
|
r.returncode = 0
|
||
|
|
return r
|
||
|
|
return handler
|
||
|
|
|
||
|
|
fake_run.add("test-impl", capture("test-impl"))
|
||
|
|
fake_run.add("test-verify", capture("test-verify"))
|
||
|
|
fake_run.add("test-orch", capture("test-orch"))
|
||
|
|
|
||
|
|
summary = _run_tick("gm-loop", tmp_project)
|
||
|
|
assert summary["skipped"] is False
|
||
|
|
impl_text = " ".join(captured["test-impl"][0])
|
||
|
|
assert "{task_brief}" not in impl_text
|
||
|
|
assert "{acceptance_criteria}" not in impl_text
|
||
|
|
assert "{next_hint}" not in impl_text
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# R5 -- _truncate_tokens
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
class TestTruncateTokens:
|
||
|
|
def test_truncate_short_text_unchanged(self):
|
||
|
|
assert lr._truncate_tokens("hello world", 100) == "hello world"
|
||
|
|
|
||
|
|
def test_truncate_long_text_capped_with_marker(self):
|
||
|
|
long_text = "x" * 1000
|
||
|
|
out = lr._truncate_tokens(long_text, 10)
|
||
|
|
assert out.endswith("…[truncated]")
|
||
|
|
assert len(out) <= 40 + len(" …[truncated]")
|
||
|
|
|
||
|
|
def test_truncate_returns_empty_for_empty_input(self):
|
||
|
|
assert lr._truncate_tokens("", 100) == ""
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# R6 -- next_hint feedback loop
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
class TestNextHintFeedback:
|
||
|
|
def test_next_hint_fed_into_next_tick_implement(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
_make_task(tmp_project, "fb")
|
||
|
|
lp = _make_loop(tmp_project, state_overrides={"current_task": "fb"})
|
||
|
|
fake_run.add_simple("test-impl", _verdict_stdout(True, 0.9, "carry-this-hint"))
|
||
|
|
fake_run.add_simple("test-verify", _verdict_stdout(True, 0.95, "carry-this-hint"))
|
||
|
|
fake_run.add_simple("test-orch", "")
|
||
|
|
_run_tick("gm-loop", tmp_project)
|
||
|
|
st = _state(lp)
|
||
|
|
assert st["last_verdict"]["next_hint"] == "carry-this-hint"
|
||
|
|
|
||
|
|
def test_first_tick_has_empty_next_hint(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
_make_task(tmp_project, "ft")
|
||
|
|
_make_loop(tmp_project, state_overrides={"current_task": "ft",
|
||
|
|
"last_verdict": None})
|
||
|
|
captured = {}
|
||
|
|
|
||
|
|
def capture(role_marker):
|
||
|
|
def handler(argv):
|
||
|
|
captured.setdefault(role_marker, []).append(list(argv))
|
||
|
|
class R:
|
||
|
|
pass
|
||
|
|
r = R()
|
||
|
|
r.stdout = _verdict_stdout() if role_marker == "test-verify" else ""
|
||
|
|
r.returncode = 0
|
||
|
|
return r
|
||
|
|
return handler
|
||
|
|
|
||
|
|
fake_run.add("test-impl", capture("test-impl"))
|
||
|
|
fake_run.add("test-verify", capture("test-verify"))
|
||
|
|
fake_run.add("test-orch", capture("test-orch"))
|
||
|
|
|
||
|
|
_run_tick("gm-loop", tmp_project)
|
||
|
|
impl_text = " ".join(captured["test-impl"][0])
|
||
|
|
marker_count = impl_text.count("{next_hint}")
|
||
|
|
assert marker_count == 0
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# R7 -- loop.json schema additions
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
class TestLoopJsonSchemaAdditions:
|
||
|
|
def test_ci_triage_template_has_work_source(self):
|
||
|
|
cfg = json.loads(_TEMPLATE_PATH.read_text())
|
||
|
|
assert cfg.get("work_source", {}).get("kind") == "single"
|
||
|
|
|
||
|
|
def test_ci_triage_template_has_acceptance_criteria(self):
|
||
|
|
cfg = json.loads(_TEMPLATE_PATH.read_text())
|
||
|
|
ac = cfg.get("acceptance_criteria")
|
||
|
|
assert isinstance(ac, list) and len(ac) >= 1
|
||
|
|
|
||
|
|
def test_create_loop_preserves_acceptance_criteria(self, tmp_project, fake_run):
|
||
|
|
src = Path.home() / ".automaton" / "templates" / "loops" / "ci-triage"
|
||
|
|
from_path = src
|
||
|
|
loop_dir = tmp_project / ".automaton" / "loops" / "tla"
|
||
|
|
loop_dir.mkdir(parents=True)
|
||
|
|
loop_cfg = json.loads((from_path / "loop.json").read_text())
|
||
|
|
loop_cfg["name"] = "tla"
|
||
|
|
(loop_dir / "loop.json").write_text(json.dumps(loop_cfg, indent=2) + "\n")
|
||
|
|
(loop_dir / ".state.loop").write_text(json.dumps({
|
||
|
|
"schema_version": 1, "name": "tla", "status": "running",
|
||
|
|
"halt_reason": None, "iteration_count": 0, "resumed_count": 0,
|
||
|
|
"last_tick_at": None, "last_verdict": None, "score_history": [],
|
||
|
|
"current_task": None, "worktree_branch": None, "worktree_path": None,
|
||
|
|
}, indent=2, sort_keys=True) + "\n")
|
||
|
|
(loop_dir / ".state.log").write_text("")
|
||
|
|
assert json.loads((loop_dir / "loop.json").read_text()).get("acceptance_criteria")
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# R8 -- status.py --audit --json
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
def _status_module():
|
||
|
|
spec = importlib.util.spec_from_file_location("status_gm", _STATUS_PATH)
|
||
|
|
mod = importlib.util.module_from_spec(spec)
|
||
|
|
spec.loader.exec_module(mod)
|
||
|
|
return mod
|
||
|
|
|
||
|
|
|
||
|
|
class TestAuditJson:
|
||
|
|
def test_audit_json_emits_violations_array(self, tmp_project, monkeypatch):
|
||
|
|
st = _status_module()
|
||
|
|
task = tmp_project / ".automaton" / "tasks" / "z"
|
||
|
|
task.mkdir(parents=True)
|
||
|
|
(task / ".state").write_text("implement\n")
|
||
|
|
(task / "IMPLEMENTATION.md").write_text("")
|
||
|
|
class A:
|
||
|
|
json_output = True
|
||
|
|
audit = True
|
||
|
|
project = str(tmp_project)
|
||
|
|
import io
|
||
|
|
from contextlib import redirect_stdout
|
||
|
|
buf = io.StringIO()
|
||
|
|
with redirect_stdout(buf):
|
||
|
|
rc = st.cmd_audit(A())
|
||
|
|
out = buf.getvalue().strip().splitlines()[-1]
|
||
|
|
data = json.loads(out)
|
||
|
|
assert "violations" in data
|
||
|
|
assert isinstance(data["violations"], list)
|
||
|
|
assert data["total_tasks"] == 1
|
||
|
|
assert any(v["category"] == 2 for v in data["violations"])
|
||
|
|
|
||
|
|
def test_audit_json_includes_loops_block(self, tmp_project):
|
||
|
|
st = _status_module()
|
||
|
|
lp = tmp_project / ".automaton" / "loops" / "zloop"
|
||
|
|
lp.mkdir(parents=True)
|
||
|
|
(lp / "loop.json").write_text(json.dumps({"name": "zloop"}))
|
||
|
|
(lp / ".state.loop").write_text(json.dumps({
|
||
|
|
"schema_version": 1, "name": "zloop", "status": "running",
|
||
|
|
"halt_reason": None, "iteration_count": 0, "resumed_count": 0,
|
||
|
|
"last_tick_at": None, "last_verdict": None, "score_history": [],
|
||
|
|
"current_task": None, "worktree_branch": None, "worktree_path": None,
|
||
|
|
}, indent=2, sort_keys=True) + "\n")
|
||
|
|
class A:
|
||
|
|
json_output = True
|
||
|
|
audit = True
|
||
|
|
project = str(tmp_project)
|
||
|
|
import io
|
||
|
|
from contextlib import redirect_stdout
|
||
|
|
buf = io.StringIO()
|
||
|
|
with redirect_stdout(buf):
|
||
|
|
st.cmd_audit(A())
|
||
|
|
data = json.loads(buf.getvalue().strip().splitlines()[-1])
|
||
|
|
assert any(loop["name"] == "zloop" for loop in data["loops"])
|
||
|
|
|
||
|
|
def test_audit_json_pickable_by_runner_run_json(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
_make_task(tmp_project, "pk")
|
||
|
|
audit_data = {"violations": [
|
||
|
|
{"category": 1, "severity": "high", "task": "pk",
|
||
|
|
"message": "x", "resolved": False},
|
||
|
|
], "loops": [], "total_tasks": 1, "untracked_tasks": 0}
|
||
|
|
fake_run.add_simple("--audit", json.dumps(audit_data))
|
||
|
|
fake_run.add_simple("test-impl", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-verify", _verdict_stdout())
|
||
|
|
fake_run.add_simple("test-orch", "")
|
||
|
|
_make_loop(tmp_project,
|
||
|
|
cfg_overrides={"work_source": {"kind": "audit"}})
|
||
|
|
summary = _run_tick("gm-loop", tmp_project)
|
||
|
|
assert summary["skipped"] is False
|
||
|
|
assert summary["iter"] == 1
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# Regression -- existing single loop ticks unchanged
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
class TestRegressionBackwardCompat:
|
||
|
|
def test_existing_single_work_source_loop_ticks_unchanged(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
_make_task(tmp_project, "reg")
|
||
|
|
_make_loop(tmp_project, cfg_overrides={
|
||
|
|
"work_source": None,
|
||
|
|
"acceptance_criteria": None,
|
||
|
|
}, state_overrides={"current_task": "reg"})
|
||
|
|
fake_run.add_simple("test-impl", _verdict_stdout(True, 0.9, "n"))
|
||
|
|
fake_run.add_simple("test-verify", _verdict_stdout(True, 0.9, "n"))
|
||
|
|
fake_run.add_simple("test-orch", "")
|
||
|
|
summary = _run_tick("gm-loop", tmp_project)
|
||
|
|
assert summary["skipped"] is False
|
||
|
|
assert summary["iter"] == 1
|
||
|
|
assert summary["verdict"]["pass"] is True
|