345 lines
12 KiB
Python
345 lines
12 KiB
Python
"""Tests for task add-loop-templates-onboarding.
|
|||
|
|
|
||
|
|
Covers R1-R6 from tasks/add-loop-templates-onboarding/SPEC.md: prompt-file
|
||
|
|
token substitution, prompt file content, template updates, and self-improvement
|
||
|
|
template creation.
|
||
|
|
"""
|
||
|
|
|
||
|
|
import json
|
||
|
|
import subprocess
|
||
|
|
import sys
|
||
|
|
import importlib.util
|
||
|
|
from pathlib import Path
|
||
|
|
from typing import Optional
|
||
|
|
|
||
|
|
import pytest
|
||
|
|
|
||
|
|
_RUNNER_PATH = Path.home() / ".automaton" / "scripts" / "loop-runner.py"
|
||
|
|
_spec = importlib.util.spec_from_file_location("loop_runner_tmpl", _RUNNER_PATH)
|
||
|
|
lr = importlib.util.module_from_spec(_spec)
|
||
|
|
_spec.loader.exec_module(lr)
|
||
|
|
|
||
|
|
_PROMPTS_DIR = Path.home() / ".automaton" / "prompts"
|
||
|
|
_CI_TRIAGE = Path.home() / ".automaton" / "templates" / "loops" / "ci-triage" / "loop.json"
|
||
|
|
_SELF_IMP = Path.home() / ".automaton" / "templates" / "loops" / "self-improvement" / "loop.json"
|
||
|
|
|
||
|
|
|
||
|
|
def _state(loop_path: Path) -> dict:
|
||
|
|
return json.loads((loop_path / ".state.loop").read_text())
|
||
|
|
|
||
|
|
|
||
|
|
def _make_loop(project: Path, name: str = "tmpl-loop",
|
||
|
|
cfg_overrides: Optional[dict] = None,
|
||
|
|
state_overrides: Optional[dict] = None) -> Path:
|
||
|
|
lp = project / ".automaton" / "loops" / name
|
||
|
|
lp.mkdir(parents=True, exist_ok=True)
|
||
|
|
cfg = {
|
||
|
|
"name": name, "description": "test",
|
||
|
|
"schedule": {"interval_seconds": 3600},
|
||
|
|
"brakes": {"max_iterations": 25, "max_budget_usd": None, "score_plateau_window": 5},
|
||
|
|
"blast_radius": {"file_scope": [], "use_worktree": False},
|
||
|
|
"work_source": {"kind": "single"},
|
||
|
|
"roles": {
|
||
|
|
"implement": {"prompt": "loop-implement.md"},
|
||
|
|
"verify": {"prompt": "loop-verifier.md"},
|
||
|
|
"orchestrate": {"prompt": "loop-orchestrate.md"},
|
||
|
|
},
|
||
|
|
}
|
||
|
|
if cfg_overrides:
|
||
|
|
cfg.update(cfg_overrides)
|
||
|
|
(lp / "loop.json").write_text(json.dumps(cfg, indent=2) + "\n")
|
||
|
|
state = {
|
||
|
|
"schema_version": 1, "name": name, "status": "running",
|
||
|
|
"halt_reason": None, "iteration_count": 0, "resumed_count": 0,
|
||
|
|
"last_tick_at": None, "last_verdict": None, "score_history": [],
|
||
|
|
"current_task": None, "worktree_branch": None, "worktree_path": None,
|
||
|
|
}
|
||
|
|
if state_overrides:
|
||
|
|
state.update(state_overrides)
|
||
|
|
(lp / ".state.loop").write_text(json.dumps(state, indent=2, sort_keys=True) + "\n")
|
||
|
|
(lp / ".state.log").write_text("")
|
||
|
|
roles = cfg.get("roles") or {}
|
||
|
|
if isinstance(roles, dict):
|
||
|
|
for role_cfg in roles.values():
|
||
|
|
if isinstance(role_cfg, dict) and role_cfg.get("prompt"):
|
||
|
|
prompt_ref = role_cfg["prompt"]
|
||
|
|
if not (Path.home() / ".automaton" / "prompts" / prompt_ref).exists():
|
||
|
|
try:
|
||
|
|
(lp / prompt_ref).write_text(f"prompt: {prompt_ref}\n")
|
||
|
|
except OSError:
|
||
|
|
pass
|
||
|
|
return lp
|
||
|
|
|
||
|
|
|
||
|
|
def _make_task(project: Path, name: str) -> Path:
|
||
|
|
tp = project / ".automaton" / "tasks" / name
|
||
|
|
tp.mkdir(parents=True, exist_ok=True)
|
||
|
|
(tp / ".state").write_text("implement\n")
|
||
|
|
return tp
|
||
|
|
|
||
|
|
|
||
|
|
class _FakeSubprocess:
|
||
|
|
def __init__(self):
|
||
|
|
self.rules: list = []
|
||
|
|
self.invocations: list = []
|
||
|
|
|
||
|
|
def add(self, needle, handler):
|
||
|
|
self.rules.append((needle, handler))
|
||
|
|
|
||
|
|
def add_simple(self, needle, stdout="", rc=0):
|
||
|
|
def handler(argv):
|
||
|
|
class R:
|
||
|
|
pass
|
||
|
|
r = R()
|
||
|
|
r.stdout = stdout
|
||
|
|
r.stderr = ""
|
||
|
|
r.returncode = rc
|
||
|
|
return r
|
||
|
|
self.rules.append((needle, handler))
|
||
|
|
|
||
|
|
def run(self, argv, *args, **kwargs):
|
||
|
|
self.invocations.append(list(argv))
|
||
|
|
for needle, handler in self.rules:
|
||
|
|
if any(needle in str(a) for a in argv):
|
||
|
|
return handler(argv)
|
||
|
|
class R:
|
||
|
|
pass
|
||
|
|
r = R()
|
||
|
|
r.stdout = ""
|
||
|
|
r.stderr = ""
|
||
|
|
r.returncode = 0
|
||
|
|
return r
|
||
|
|
|
||
|
|
|
||
|
|
@pytest.fixture
|
||
|
|
def tmp_project(tmp_path):
|
||
|
|
(tmp_path / ".automaton" / "tasks").mkdir(parents=True)
|
||
|
|
return tmp_path
|
||
|
|
|
||
|
|
|
||
|
|
@pytest.fixture
|
||
|
|
def fake_run(monkeypatch):
|
||
|
|
fake = _FakeSubprocess()
|
||
|
|
monkeypatch.setattr(subprocess, "run", fake.run)
|
||
|
|
return fake
|
||
|
|
|
||
|
|
|
||
|
|
def _gate_ok(fake):
|
||
|
|
fake.add_simple("--check-gate", json.dumps({"ok": True}))
|
||
|
|
|
||
|
|
|
||
|
|
def _ctx_ok(fake):
|
||
|
|
fake.add_simple("--loop-mode", json.dumps({"loop_mode_eligible": True}))
|
||
|
|
|
||
|
|
|
||
|
|
def _verdict_stdout(p=True, score=0.9, hint=""):
|
||
|
|
body = {"pass": p, "score": score}
|
||
|
|
if hint:
|
||
|
|
body["next_hint"] = hint
|
||
|
|
return json.dumps(body)
|
||
|
|
|
||
|
|
|
||
|
|
def _tick_args(loop_name, project):
|
||
|
|
class A:
|
||
|
|
pass
|
||
|
|
a = A()
|
||
|
|
a.mode = "tick"
|
||
|
|
a.loop = loop_name
|
||
|
|
a.project = str(project)
|
||
|
|
a.json_output = False
|
||
|
|
return a
|
||
|
|
|
||
|
|
|
||
|
|
def _run_tick(loop_name, project):
|
||
|
|
return lr.cmd_tick(_tick_args(loop_name, project))
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# R1 -- _resolve_prompt
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
class TestResolvePrompt:
|
||
|
|
def test_resolve_prompt_substitutes_tokens(self, tmp_path):
|
||
|
|
lp = tmp_path / "loop"
|
||
|
|
lp.mkdir()
|
||
|
|
(lp / "outputs").mkdir()
|
||
|
|
prompt_file = lp / "test-prompt.md"
|
||
|
|
prompt_file.write_text("Task: {task_brief}\nCriteria: {acceptance_criteria}\nHint: {next_hint}")
|
||
|
|
resolved = lr._resolve_prompt(
|
||
|
|
"test-prompt.md",
|
||
|
|
{"task_brief": "BRIEF", "acceptance_criteria": "CRIT", "next_hint": "HINT"},
|
||
|
|
lp, 1, "implement")
|
||
|
|
content = Path(resolved).read_text()
|
||
|
|
assert "BRIEF" in content
|
||
|
|
assert "CRIT" in content
|
||
|
|
assert "HINT" in content
|
||
|
|
|
||
|
|
def test_resolve_prompt_reads_artifact_content(self, tmp_path):
|
||
|
|
lp = tmp_path / "loop"
|
||
|
|
lp.mkdir()
|
||
|
|
(lp / "outputs").mkdir()
|
||
|
|
prompt_file = lp / "verify-prompt.md"
|
||
|
|
prompt_file.write_text("Artifact:\n{artifact_content}\n")
|
||
|
|
artifact = lp / "outputs" / "tick1-implement.json"
|
||
|
|
artifact.write_text("IMPLEMENTED CODE")
|
||
|
|
resolved = lr._resolve_prompt(
|
||
|
|
"verify-prompt.md",
|
||
|
|
{"artifact": str(artifact)},
|
||
|
|
lp, 1, "verify")
|
||
|
|
content = Path(resolved).read_text()
|
||
|
|
assert "IMPLEMENTED CODE" in content
|
||
|
|
|
||
|
|
def test_resolve_prompt_fallback_when_file_missing(self, tmp_path):
|
||
|
|
lp = tmp_path / "loop"
|
||
|
|
lp.mkdir()
|
||
|
|
result = lr._resolve_prompt("nonexistent.md", None, lp, 1, "implement")
|
||
|
|
assert result == "nonexistent.md"
|
||
|
|
|
||
|
|
def test_resolve_prompt_searches_loop_dir_then_framework(self, tmp_path):
|
||
|
|
lp = tmp_path / "loop"
|
||
|
|
lp.mkdir()
|
||
|
|
(lp / "outputs").mkdir()
|
||
|
|
local_prompt = lp / "local-prompt.md"
|
||
|
|
local_prompt.write_text("LOCAL")
|
||
|
|
resolved = lr._resolve_prompt("local-prompt.md", None, lp, 1, "implement")
|
||
|
|
assert "LOCAL" in Path(resolved).read_text()
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# R2-R4 -- Prompt file content
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
class TestPromptFiles:
|
||
|
|
def test_loop_implement_prompt_has_tokens(self):
|
||
|
|
content = (_PROMPTS_DIR / "loop-implement.md").read_text()
|
||
|
|
assert "{task_brief}" in content
|
||
|
|
assert "{acceptance_criteria}" in content
|
||
|
|
assert "{next_hint}" in content
|
||
|
|
assert "{current_task}" in content
|
||
|
|
|
||
|
|
def test_loop_implement_prompt_has_forbidden_section(self):
|
||
|
|
content = (_PROMPTS_DIR / "loop-implement.md").read_text()
|
||
|
|
assert "FORBIDDEN" in content
|
||
|
|
assert "--transition" in content
|
||
|
|
assert "--approve" in content
|
||
|
|
|
||
|
|
def test_loop_verifier_prompt_has_json_instruction(self):
|
||
|
|
content = (_PROMPTS_DIR / "loop-verifier.md").read_text()
|
||
|
|
assert "json" in content.lower()
|
||
|
|
assert "pass" in content
|
||
|
|
assert "score" in content
|
||
|
|
assert "next_hint" in content
|
||
|
|
|
||
|
|
def test_loop_verifier_prompt_has_score_rubric(self):
|
||
|
|
content = (_PROMPTS_DIR / "loop-verifier.md").read_text()
|
||
|
|
assert "1.0" in content
|
||
|
|
assert "0.7" in content
|
||
|
|
assert "0.4" in content
|
||
|
|
assert "0.0" in content
|
||
|
|
|
||
|
|
def test_loop_verifier_prompt_has_tokens(self):
|
||
|
|
content = (_PROMPTS_DIR / "loop-verifier.md").read_text()
|
||
|
|
assert "{artifact_content}" in content
|
||
|
|
assert "{task_brief}" in content
|
||
|
|
assert "{acceptance_criteria}" in content
|
||
|
|
|
||
|
|
def test_loop_orchestrate_prompt_has_verdict_token(self):
|
||
|
|
content = (_PROMPTS_DIR / "loop-orchestrate.md").read_text()
|
||
|
|
assert "{verdict}" in content
|
||
|
|
assert "{current_task}" in content
|
||
|
|
|
||
|
|
def test_loop_orchestrate_prompt_has_no_edit_rule(self):
|
||
|
|
content = (_PROMPTS_DIR / "loop-orchestrate.md").read_text()
|
||
|
|
assert "FORBIDDEN" in content
|
||
|
|
assert "NOT edit" in content
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# R5 -- ci-triage template
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
class TestCiTriageTemplate:
|
||
|
|
def test_ci_triage_template_has_prompt_refs(self):
|
||
|
|
cfg = json.loads(_CI_TRIAGE.read_text())
|
||
|
|
roles = cfg.get("roles", {})
|
||
|
|
assert roles.get("implement", {}).get("prompt") == "loop-implement.md"
|
||
|
|
assert roles.get("verify", {}).get("prompt") == "loop-verifier.md"
|
||
|
|
assert roles.get("orchestrate", {}).get("prompt") == "loop-orchestrate.md"
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# R6 -- self-improvement template
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
class TestSelfImprovementTemplate:
|
||
|
|
def test_self_improvement_template_exists(self):
|
||
|
|
assert _SELF_IMP.exists()
|
||
|
|
|
||
|
|
def test_self_improvement_template_has_audit_work_source(self):
|
||
|
|
cfg = json.loads(_SELF_IMP.read_text())
|
||
|
|
ws = cfg.get("work_source", {})
|
||
|
|
assert ws.get("kind") == "audit"
|
||
|
|
|
||
|
|
def test_self_improvement_template_has_file_scope(self):
|
||
|
|
cfg = json.loads(_SELF_IMP.read_text())
|
||
|
|
scope = cfg.get("blast_radius", {}).get("file_scope", [])
|
||
|
|
assert "scripts/" in scope
|
||
|
|
assert "prompts/" in scope
|
||
|
|
assert "tests/" in scope
|
||
|
|
assert "design/" in scope
|
||
|
|
|
||
|
|
def test_self_improvement_template_has_prompt_refs(self):
|
||
|
|
cfg = json.loads(_SELF_IMP.read_text())
|
||
|
|
roles = cfg.get("roles", {})
|
||
|
|
assert roles.get("implement", {}).get("prompt") == "loop-implement.md"
|
||
|
|
assert roles.get("verify", {}).get("prompt") == "loop-verifier.md"
|
||
|
|
|
||
|
|
def test_self_improvement_template_has_brakes(self):
|
||
|
|
cfg = json.loads(_SELF_IMP.read_text())
|
||
|
|
brakes = cfg.get("brakes", {})
|
||
|
|
assert brakes.get("max_iterations") == 10
|
||
|
|
assert brakes.get("score_plateau_window") == 3
|
||
|
|
|
||
|
|
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
# R1 integration -- tick with prompt substitution
|
||
|
|
# ---------------------------------------------------------------------------
|
||
|
|
|
||
|
|
|
||
|
|
class TestTickPromptSubstitution:
|
||
|
|
def test_tick_substitutes_prompt_tokens(self, tmp_project, fake_run):
|
||
|
|
_gate_ok(fake_run)
|
||
|
|
_ctx_ok(fake_run)
|
||
|
|
_make_task(tmp_project, "pt-task")
|
||
|
|
lp = _make_loop(tmp_project,
|
||
|
|
cfg_overrides={"harness": {"command": ["test-bin", "--prompt-file", "{prompt}"]}},
|
||
|
|
state_overrides={"current_task": "pt-task"})
|
||
|
|
captured = {}
|
||
|
|
|
||
|
|
def capture_harness(role_marker):
|
||
|
|
def handler(argv):
|
||
|
|
prompt_path = None
|
||
|
|
for i, t in enumerate(argv):
|
||
|
|
if t == "--prompt-file" and i + 1 < len(argv):
|
||
|
|
prompt_path = argv[i + 1]
|
||
|
|
if prompt_path:
|
||
|
|
captured[role_marker] = Path(prompt_path).read_text()
|
||
|
|
class R:
|
||
|
|
pass
|
||
|
|
r = R()
|
||
|
|
r.stdout = _verdict_stdout() if role_marker == "loop-verifier" else ""
|
||
|
|
r.returncode = 0
|
||
|
|
return r
|
||
|
|
return handler
|
||
|
|
|
||
|
|
fake_run.add("implement-prompt", capture_harness("loop-implement"))
|
||
|
|
fake_run.add("verify-prompt", capture_harness("loop-verifier"))
|
||
|
|
fake_run.add("orchestrate-prompt", capture_harness("loop-orchestrate"))
|
||
|
|
_run_tick("tmpl-loop", tmp_project)
|
||
|
|
assert "pt-task" in captured["loop-implement"]
|
||
|
|
assert "pt-task" in captured["loop-verifier"]
|