v2.0: state enforcement, project scoping, harness integration
CI / build (push) Has been cancelled

State Enforcement (v2.0):
- .state file as single source of truth for task phase
- Approval gates for research, decomposition, design, test_design
- status.py --transition refuses illegal phase transitions
- status.py --validate-folder detects out-of-order artifacts
- status.py --audit checks all tasks for violations
- status.py --create-task is the only valid way to create tasks
- Pre-v2.0 tasks without .state are UNTRACKED -- all commands refuse them
- New --upgrade command bootstraps .state files for existing tasks

Project Scoping:
- --project flag added to all status.py commands across 16+ files
- _find_project_dir errors instead of silently falling back to ~/.automaton/
- --scope-check marks framework files OUT_OF_SCOPE when working on a project
- Dashboard handlers use stored project_root instead of re-detecting from CWD
- Prompts reference ~/.automaton/scripts/vram_detect.py (not {project}/.automaton/)

Harness Integration:
- status.py --can-edit now supports project-level checks (no --task required)
- --can-edit --file checks file scope without --task
- --json output for machine-readable harness integration
- opencode plugin (plugins/automaton-guard/plugin.ts) intercepts edit/write
- Git pre-commit hook (scripts/git-hooks/pre-commit) blocks commits without task
- Formal integration contract (contracts/harness-integration.md)

Other:
- upgrade.sh delegates to status.py --upgrade instead of manual heuristics
- Phase prompts reference --project {project} for multi-project scoping
- 200 tests passing (14 new)
This commit is contained in:
2026-06-15 14:16:46 -04:00
parent 79b783864e
commit 05c76852a2
151 changed files with 7295 additions and 632 deletions
+156
View File
@@ -86,3 +86,159 @@ def test_static_valid_file(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> N
assert response_status == [200]
assert any(h[0] == "Content-Type" and h[1] == "text/html" for h in response_headers)
assert handler.wfile.getvalue() == b"<html></html>"
class TestCORSAndSecurityHeaders:
"""Tests for CORS and security headers on API responses."""
def test_send_json_includes_cors(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
html_dir = tmp_path / "html"
html_dir.mkdir()
monkeypatch.setattr(DashboardHandler, "dashboard_path", html_dir)
handler = DashboardHandler.__new__(DashboardHandler)
response_headers: list[tuple[str, str]] = []
handler.send_response = lambda code: None
handler.send_header = lambda k, v: response_headers.append((k, v))
handler.end_headers = lambda: None
handler.wfile = io.BytesIO()
handler._send_json({"test": True})
header_dict = dict(response_headers)
assert header_dict.get("Access-Control-Allow-Origin") == "*"
assert header_dict.get("X-Content-Type-Options") == "nosniff"
def test_send_error_includes_cors(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
html_dir = tmp_path / "html"
html_dir.mkdir()
monkeypatch.setattr(DashboardHandler, "dashboard_path", html_dir)
handler = DashboardHandler.__new__(DashboardHandler)
response_headers: list[tuple[str, str]] = []
handler.send_response = lambda code: None
handler.send_header = lambda k, v: response_headers.append((k, v))
handler.end_headers = lambda: None
handler.wfile = io.BytesIO()
handler._send_error(404, "Not found")
header_dict = dict(response_headers)
assert header_dict.get("Access-Control-Allow-Origin") == "*"
assert header_dict.get("X-Content-Type-Options") == "nosniff"
class TestContentLengthBound:
"""Tests for POST content-length limits."""
def test_max_post_body_constant(self) -> None:
from automaton.dashboard.ui.app import MAX_POST_BODY, MAX_REVIEW_COMMENT_LENGTH
assert MAX_POST_BODY == 65536
assert MAX_REVIEW_COMMENT_LENGTH == 4096
class TestTaskCache:
"""Tests for server-side task caching."""
def test_cache_returns_tasks(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
from automaton.dashboard.ui.app import _get_cached_tasks, _task_cache
tasks_dir = tmp_path / ".automaton" / "tasks"
task_dir = tasks_dir / "my-task"
task_dir.mkdir(parents=True)
(task_dir / "SPEC.md").write_text("# Spec")
_task_cache["timestamp"] = 0.0
_task_cache["tasks"] = []
result = _get_cached_tasks(tmp_path)
assert len(result) == 1
assert result[0].name == "my-task"
def test_cache_uses_ttl(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
import time
from automaton.dashboard.ui.app import _get_cached_tasks, _task_cache, CACHE_TTL
tasks_dir = tmp_path / ".automaton" / "tasks"
task_dir = tasks_dir / "cached-task"
task_dir.mkdir(parents=True)
(task_dir / "SPEC.md").write_text("# Spec")
_task_cache["timestamp"] = 0.0
_task_cache["tasks"] = []
result1 = _get_cached_tasks(tmp_path)
assert len(result1) == 1
_task_cache["timestamp"] = time.time() + CACHE_TTL + 10
new_task = tasks_dir / "new-task"
new_task.mkdir()
(new_task / "SPEC.md").write_text("# New")
result2 = _get_cached_tasks(tmp_path)
assert len(result2) == 1
_task_cache["timestamp"] = 0.0
result3 = _get_cached_tasks(tmp_path)
assert len(result3) == 2
def test_invalidate_cache(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
import time
from automaton.dashboard.ui.app import _invalidate_task_cache, _get_cached_tasks, _task_cache
tasks_dir = tmp_path / ".automaton" / "tasks"
task_dir = tasks_dir / "inv-task"
task_dir.mkdir(parents=True)
(task_dir / "SPEC.md").write_text("# Spec")
_task_cache["timestamp"] = time.time() + 9999
_task_cache["tasks"] = []
_invalidate_task_cache()
assert _task_cache["timestamp"] == 0.0
class TestConfigEndpoint:
"""Tests for GET/PUT /api/config."""
def _make_handler(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, config=None):
from automaton.dashboard.ui.app import DashboardHandler
from automaton.dashboard.config import DashboardConfig
html_dir = tmp_path / "html"
html_dir.mkdir()
monkeypatch.setattr(DashboardHandler, "dashboard_path", html_dir)
handler = DashboardHandler.__new__(DashboardHandler)
handler.config = config or DashboardConfig()
handler.path = "/api/config"
return handler
def test_serve_config(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
from automaton.dashboard.config import DashboardConfig
handler = self._make_handler(tmp_path, monkeypatch)
response_data = {}
handler._send_json = lambda d: response_data.update(d)
handler._serve_config()
assert response_data["theme"] == "default"
assert response_data["auto_refresh_interval"] == 2
assert response_data["default_view"] == "board"
def test_serve_config_not_available(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
handler = self._make_handler(tmp_path, monkeypatch)
handler.config = None
errors = []
handler._send_error = lambda c, m: errors.append((c, m))
handler._serve_config()
assert errors[0][0] == 503
def test_handle_config_update(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
from automaton.dashboard.config import DashboardConfig
from automaton.dashboard.ui.app import DashboardHandler
tasks_dir = tmp_path / ".automaton" / "tasks"
tasks_dir.mkdir(parents=True)
monkeypatch.setattr("automaton.dashboard.ui.app.get_config_path",
lambda pr: tmp_path / ".automaton" / "dashboard-config.json")
handler = self._make_handler(tmp_path, monkeypatch)
handler.project_root = tmp_path
handler.headers = {"Content-Length": "19"}
handler.rfile = io.BytesIO(b'{"theme": "dark"}')
response_data = {}
handler._send_json = lambda d: response_data.update(d)
handler._handle_config_update()
assert response_data["theme"] == "dark"
assert handler.config.theme == "dark"
def test_handle_config_update_invalid(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
from automaton.dashboard.config import DashboardConfig
handler = self._make_handler(tmp_path, monkeypatch)
handler.headers = {"Content-Length": "39"}
handler.rfile = io.BytesIO(b'{"auto_refresh_interval": 999}')
errors = []
handler._send_error = lambda c, m: errors.append((c, m))
handler._handle_config_update()
assert errors[0][0] == 400
+254
View File
@@ -0,0 +1,254 @@
"""Framework self-consistency tests.
These tests enforce structural rules on the framework itself — prompt
consistency, canonical paths, state machine integrity, and configuration
validity. They catch regressions that unit tests alone cannot.
"""
import re
from pathlib import Path
import pytest
ROOT = Path(__file__).resolve().parent.parent
PROMPTS_DIR = ROOT / "prompts"
RULES_FILE = ROOT / ".rules.md"
PYPROJECT_FILE = ROOT / "pyproject.toml"
CI_FILE = ROOT / ".gitea" / "workflows" / "ci.yml"
CSS_FILE = ROOT / "automaton" / "dashboard" / "html" / "styles.css"
JS_FILE = ROOT / "automaton" / "dashboard" / "html" / "dashboard.js"
class TestDeliveryPromptsHaveStopConditions:
"""R1.A: Every delivery prompt must contain a stop condition block."""
EXCLUDED = {"orchestrate.md", "compaction.md", "workflow.md", "subtask_management.md", "onboarding.md"}
@pytest.fixture()
def delivery_prompts(self):
if not PROMPTS_DIR.exists():
pytest.skip("prompts/ directory not found")
return [
f
for f in sorted(PROMPTS_DIR.iterdir())
if f.is_file() and f.suffix == ".md" and f.name not in self.EXCLUDED
]
def test_each_prompt_has_stop_condition(self, delivery_prompts):
for prompt in delivery_prompts:
content = prompt.read_text()
has_stop = "## Stop Condition" in content or "STOP CONDITION" in content or "CONTRACT_MET" in content
assert has_stop, f"{prompt.name} is missing a stop condition block"
class TestNoHardcodedURLs:
"""R1.B: Prompts, contracts, and templates must not contain hardcoded repo URLs."""
ALLOWED_FILES = {"install.sh"}
def _check_dir(self, directory: Path, pattern: re.Pattern):
violations = []
if not directory.exists():
return violations
for f in directory.rglob("*.md"):
if f.name in self.ALLOWED_FILES:
continue
content = f.read_text()
for line_no, line in enumerate(content.splitlines(), 1):
if pattern.search(line):
violations.append(f"{f.relative_to(ROOT)}:{line_no}")
return violations
def test_no_hardcoded_ip_urls(self):
pattern = re.compile(r"\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}[:/]")
dirs = [PROMPTS_DIR, ROOT / "contracts", ROOT / "templates"]
violations = []
for d in dirs:
violations.extend(self._check_dir(d, pattern))
assert not violations, f"Hardcoded IP URLs found in: {violations}"
def test_no_hardcoded_localhost_ports(self):
pattern = re.compile(r"localhost:\d{4,5}")
dirs = [PROMPTS_DIR, ROOT / "contracts", ROOT / "templates"]
violations = []
for d in dirs:
violations.extend(self._check_dir(d, pattern))
assert not violations, f"Hardcoded localhost URLs found in: {violations}"
class TestRulesMdSections:
"""R1.C & R1.D: .rules.md must contain mandatory sections."""
MANDATORY_SECTIONS = [
"Task-Driven Development",
"VRAM",
"Changelog",
"Session Discipline",
"Scope Confinement",
"Artifact Integrity",
]
def test_mandatory_sections_present(self):
if not RULES_FILE.exists():
pytest.skip(".rules.md not found")
content = RULES_FILE.read_text()
for section in self.MANDATORY_SECTIONS:
assert section in content, f".rules.md missing mandatory section: {section}"
def test_self_improvement_has_example(self):
if not RULES_FILE.exists():
pytest.skip(".rules.md not found")
content = RULES_FILE.read_text()
assert "Past failure" in content, ".rules.md Self-Improvement section must reference at least one real failure mode"
class TestCanonicalTaskPaths:
"""R1.E: All task path references in prompts must use the canonical format."""
CANONICAL_PATTERN = re.compile(r"\{project\}/\.automaton/tasks/\w")
DEPRECATED_PATTERN = re.compile(r"\{project\}/tasks/[\w-]+/")
def test_no_deprecated_task_paths(self):
if not PROMPTS_DIR.exists():
pytest.skip("prompts/ directory not found")
violations = []
for f in sorted(PROMPTS_DIR.iterdir()):
if not f.is_file() or f.suffix != ".md":
continue
content = f.read_text()
for line_no, line in enumerate(content.splitlines(), 1):
if self.DEPRECATED_PATTERN.search(line):
violations.append(f"{f.name}:{line_no}: {line.strip()}")
assert not violations, f"Deprecated task paths found: {violations}"
class TestPyprojectNoStaleExtras:
"""R1.F: pyproject.toml must not reference inotify."""
def test_no_inotify_dependency(self):
if not PYPROJECT_FILE.exists():
pytest.skip("pyproject.toml not found")
content = PYPROJECT_FILE.read_text()
assert "inotify" not in content, "pyproject.toml still references inotify"
class TestDashboardCSSThemes:
"""R1.G: Light and dark themes must define the same variable set."""
def _extract_vars(self, section: str) -> set:
pattern = re.compile(r"--([\w-]+)\s*:", re.MULTILINE)
return set(pattern.findall(section))
def test_theme_variable_parity(self):
if not CSS_FILE.exists():
pytest.skip("styles.css not found")
content = CSS_FILE.read_text()
root_match = re.search(r":root\s*\{([^}]+)\}", content, re.DOTALL)
light_match = re.search(r'\[data-theme="light"\]\s*\{([^}]+)\}', content, re.DOTALL)
dark_match = re.search(r'\[data-theme="dark"\]\s*\{([^}]+)\}', content, re.DOTALL)
if not root_match:
pytest.skip(":root CSS variables not found")
root_vars = self._extract_vars(root_match.group(1))
theme_vars = set()
if light_match:
theme_vars |= self._extract_vars(light_match.group(1))
if dark_match:
theme_vars |= self._extract_vars(dark_match.group(1))
if not theme_vars:
pytest.skip("No theme sections found in CSS")
structural_vars = {"radius-sm", "radius-md", "radius-lg"}
themable_root_vars = root_vars - structural_vars
missing_from_themes = themable_root_vars - theme_vars
assert not missing_from_themes, f"CSS variables defined in :root but missing from theme overrides: {missing_from_themes}"
class TestVerdictParsingRegression:
"""R3: Verdict parsing regression tests for the critical false-BLOCKED bug."""
def test_pass_verdict_mentioning_fail_is_done(self):
from automaton.dashboard.core.task import determine_task_state, TaskState
from pathlib import Path
import tempfile
with tempfile.TemporaryDirectory() as tmp:
task_dir = Path(tmp) / "test-task"
task_dir.mkdir()
(task_dir / "VERDICT.md").write_text("## Status: PASS\n\nThe bug in the FAIL case is now fixed.\n")
state, _ = determine_task_state(task_dir)
assert state == TaskState.DONE, f"Expected DONE, got {state}"
def test_pass_verdict_with_needs_review_mention_is_done(self):
from automaton.dashboard.core.task import determine_task_state, TaskState
from pathlib import Path
import tempfile
with tempfile.TemporaryDirectory() as tmp:
task_dir = Path(tmp) / "test-task"
task_dir.mkdir()
(task_dir / "VERDICT.md").write_text("## Status: PASS\n\nPreviously flagged as NEEDS_REVIEW but resolved.\n")
state, _ = determine_task_state(task_dir)
assert state == TaskState.DONE, f"Expected DONE, got {state}"
def test_structured_fail_verdict_is_blocked(self):
from automaton.dashboard.core.task import determine_task_state, TaskState
from pathlib import Path
import tempfile
with tempfile.TemporaryDirectory() as tmp:
task_dir = Path(tmp) / "test-task"
task_dir.mkdir()
(task_dir / "VERDICT.md").write_text("## Status: FAIL\n\nThe implementation has a critical bug.\n")
state, _ = determine_task_state(task_dir)
assert state == TaskState.BLOCKED
def test_implementation_alone_is_bug_find(self):
from automaton.dashboard.core.task import determine_task_state, TaskState
from pathlib import Path
import tempfile
with tempfile.TemporaryDirectory() as tmp:
task_dir = Path(tmp) / "test-task"
task_dir.mkdir()
(task_dir / "IMPLEMENTATION.md").write_text("# Implementation\nDone.\n")
state, _ = determine_task_state(task_dir)
assert state == TaskState.BUG_FIND
def test_empty_verdict_is_blocked(self):
from automaton.dashboard.core.task import determine_task_state, TaskState
from pathlib import Path
import tempfile
with tempfile.TemporaryDirectory() as tmp:
task_dir = Path(tmp) / "test-task"
task_dir.mkdir()
(task_dir / "VERDICT.md").write_text("")
state, _ = determine_task_state(task_dir)
assert state == TaskState.BLOCKED
def test_adv_bug_report_alone_is_bug_find(self):
from automaton.dashboard.core.task import determine_task_state, TaskState
from pathlib import Path
import tempfile
with tempfile.TemporaryDirectory() as tmp:
task_dir = Path(tmp) / "test-task"
task_dir.mkdir()
(task_dir / "ADVERSARIAL_BUG_REPORT.md").write_text("# Bug Report\nFound issue.\n")
state, _ = determine_task_state(task_dir)
assert state == TaskState.BUG_FIND
class TestCIWorkflowValidation:
"""R4: CI workflow must compile, test, and check shell scripts."""
def test_ci_runs_py_compile(self):
if not CI_FILE.exists():
pytest.skip("CI workflow not found")
content = CI_FILE.read_text()
assert "py_compile" in content, "CI must run py_compile"
def test_ci_runs_pytest(self):
if not CI_FILE.exists():
pytest.skip("CI workflow not found")
content = CI_FILE.read_text()
assert "pytest" in content, "CI must run pytest"
def test_ci_checks_shell_scripts(self):
if not CI_FILE.exists():
pytest.skip("CI workflow not found")
content = CI_FILE.read_text()
assert "bash -n" in content, "CI must syntax-check shell scripts"
+30 -1
View File
@@ -9,10 +9,13 @@ ROOT = Path(__file__).resolve().parent.parent
PROMPTS_DIR = ROOT / "prompts"
TEMPLATES_DIR = ROOT / "templates"
# Legacy path pattern that should no longer appear.
# Legacy path pattern that should no longer appear (with placeholder).
LEGACY_PATH = re.compile(r"\{project\}/tasks/\{task-name\}/")
# Canonical path pattern that should be used instead.
CANONICAL_PATH = re.compile(r"\{project\}/\.automaton/tasks/\{task-name\}/")
# Concrete legacy path pattern: {project}/tasks/ followed by any name.
# Catches paths like {project}/tasks/onboarding/ that use concrete names.
CONCRETE_LEGACY_PATH = re.compile(r"\{project\}/tasks/[\w-]+/")
def _markdown_files(*directories: Path) -> list[Path]:
@@ -34,6 +37,32 @@ def test_no_legacy_task_paths(path: Path) -> None:
)
@pytest.mark.parametrize("path", _markdown_files(PROMPTS_DIR, TEMPLATES_DIR))
def test_no_concrete_legacy_task_paths(path: Path) -> None:
"""Every prompt/template must not use {project}/tasks/ with concrete task names."""
text = path.read_text(encoding="utf-8")
# Exclude the migration detection reference, which legitimately mentions the deprecated path.
concrete_matches = CONCRETE_LEGACY_PATH.findall(text)
# Filter out the onboarding migration detection section that references the legacy path.
if concrete_matches and "Migration Check" in text:
lines = text.splitlines()
filtered = []
in_migration = False
for line in lines:
if "Migration Check" in line or "Migration Detection" in line:
in_migration = True
if in_migration and line.startswith("#") and "Migration" not in line:
in_migration = False
if not in_migration:
for m in CONCRETE_LEGACY_PATH.findall(line):
filtered.append(m)
concrete_matches = filtered
assert not concrete_matches, (
f"Found concrete legacy task path in {path.relative_to(ROOT)}: {concrete_matches}\n"
"Use {project}/.automaton/tasks/ instead of {project}/tasks/."
)
def test_canonical_path_present_in_prompts() -> None:
"""At least one prompt uses the canonical path (sanity check)."""
found = False
+365
View File
@@ -0,0 +1,365 @@
"""Tests for status.py enforcement script."""
import os
import sys
import tempfile
import pytest
from pathlib import Path
sys.path.insert(0, str(Path(__file__).parent.parent / "scripts"))
@pytest.fixture
def tmp_project(tmp_path):
"""Create a temporary project with .automaton/tasks structure."""
auto_dir = tmp_path / ".automaton"
tasks_dir = auto_dir / "tasks"
tasks_dir.mkdir(parents=True)
return tmp_path
@pytest.fixture
def status_script():
"""Return path to status.py."""
return Path.home() / ".automaton" / "scripts" / "status.py"
def _run_status(args, project=None):
"""Run status.py with given args and return (stdout, exit_code)."""
import subprocess
cmd = [sys.executable, str(Path.home() / ".automaton" / "scripts" / "status.py")]
if project:
cmd.extend(["--project", str(project)])
cmd.extend(args)
result = subprocess.run(cmd, capture_output=True, text=True)
return result.stdout.strip(), result.returncode
def _create_task(project, task_name, phase=None):
"""Create a task and optionally set its phase."""
out, code = _run_status(["--create-task", task_name], project)
assert code == 0, f"Failed to create task: {out}"
if phase:
task_dir = project / ".automaton" / "tasks" / task_name
(task_dir / ".state").write_text(f"{phase}\n")
return project / ".automaton" / "tasks" / task_name
class TestCreateTask:
def test_create_valid_task(self, tmp_project):
out, code = _run_status(["--create-task", "my-test-task"], tmp_project)
assert code == 0
assert "Created task 'my-test-task'" in out
task_dir = tmp_project / ".automaton" / "tasks" / "my-test-task"
assert task_dir.exists()
assert (task_dir / ".state").read_text().strip() == "new"
assert (task_dir / ".state.approvals").exists()
def test_create_task_rejects_spaces(self, tmp_project):
out, code = _run_status(["--create-task", "my test task"], tmp_project)
assert code == 2
assert "kebab-case" in out
def test_create_task_rejects_uppercase(self, tmp_project):
out, code = _run_status(["--create-task", "My-Task"], tmp_project)
assert code == 2
def test_create_task_rejects_duplicate(self, tmp_project):
_run_status(["--create-task", "my-task"], tmp_project)
out, code = _run_status(["--create-task", "my-task"], tmp_project)
assert code == 2
assert "already exists" in out
class TestStateTransitions:
def test_transition_new_to_research(self, tmp_project):
task_dir = _create_task(tmp_project, "trans-test", "new")
out, code = _run_status(["--task", "trans-test", "--transition", "research"], tmp_project)
assert code == 0
assert "Transitioned" in out
assert (task_dir / ".state").read_text().strip() == "research"
def test_illegal_transition_refused(self, tmp_project):
task_dir = _create_task(tmp_project, "illegal-test", "new")
out, code = _run_status(["--task", "illegal-test", "--transition", "implement"], tmp_project)
assert code == 1
assert "Cannot transition" in out
def test_approval_gate_blocks_transition(self, tmp_project):
task_dir = _create_task(tmp_project, "approval-test", "research:awaiting_approval")
out, code = _run_status(["--task", "approval-test", "--transition", "design"], tmp_project)
assert code == 1
assert "research:approved" in out
class TestApprovalGates:
def test_approve_transitions_to_approved(self, tmp_project):
task_dir = _create_task(tmp_project, "approve-test", "research:awaiting_approval")
out, code = _run_status(["--approve", "--task", "approve-test"], tmp_project)
assert code == 0
assert "Approved" in out
assert (task_dir / ".state").read_text().strip() == "research:approved"
def test_approve_records_in_log(self, tmp_project):
task_dir = _create_task(tmp_project, "log-test", "research:awaiting_approval")
_run_status(["--approve", "--task", "log-test"], tmp_project)
approvals = (task_dir / ".state.approvals").read_text()
assert "research:approved" in approvals
assert "user" in approvals
def test_approve_refused_for_wrong_phase(self, tmp_project):
_create_task(tmp_project, "wrong-phase", "implement")
out, code = _run_status(["--approve", "--task", "wrong-phase"], tmp_project)
assert "does not require approval" in out
def test_approve_refused_without_awaiting(self, tmp_project):
_create_task(tmp_project, "no-await", "research")
out, code = _run_status(["--approve", "--task", "no-await"], tmp_project)
assert code == 1
assert "not awaiting approval" in out
class TestValidateFolder:
def test_valid_folder_passes(self, tmp_project):
task_dir = _create_task(tmp_project, "valid-test", "research")
# Write SPEC.md (expected artifact for research)
(task_dir / "SPEC.md").write_text("# Spec")
out, code = _run_status(["--validate-folder", "--task", "valid-test"], tmp_project)
assert "PASS" in out
def test_folder_without_state_flagged(self, tmp_project):
tasks_dir = tmp_project / ".automaton" / "tasks"
task_dir = tasks_dir / "manual-task"
task_dir.mkdir()
(task_dir / "SPEC.md").write_text("# Spec")
out, code = _run_status(["--validate-folder", "--task", "manual-task"], tmp_project)
assert "no .state file" in out or "manually" in out
def test_out_of_order_artifacts_flagged(self, tmp_project):
task_dir = _create_task(tmp_project, "skip-test", "research")
(task_dir / "SPEC.md").write_text("# Spec")
(task_dir / "IMPLEMENTATION.md").write_text("# Impl")
out, code = _run_status(["--validate-folder", "--task", "skip-test"], tmp_project)
assert "FAIL" in out
assert "IMPLEMENTATION.md" in out
class TestArtifactValidation:
def test_missing_required_artifact_blocks_transition(self, tmp_project):
_create_task(tmp_project, "no-spec", "research")
out, code = _run_status(["--task", "no-spec", "--transition", "design"], tmp_project)
assert code == 1
assert "SPEC.md" in out
def test_forbidden_artifact_blocks_transition(self, tmp_project):
task_dir = _create_task(tmp_project, "skip-artifact", "research")
(task_dir / "SPEC.md").write_text("# Spec")
(task_dir / "IMPLEMENTATION.md").write_text("# Impl")
out, code = _run_status(["--task", "skip-artifact", "--transition", "design"], tmp_project)
assert code == 1
assert "out-of-order" in out.lower() or "forbidden" in out.lower()
class TestShowTask:
def test_show_task_displays_phase(self, tmp_project):
_create_task(tmp_project, "show-test", "implement")
out, code = _run_status(["--task", "show-test"], tmp_project)
assert code == 0
assert "Phase: implement" in out
def test_show_task_displays_allowed_forbidden(self, tmp_project):
_create_task(tmp_project, "show-impl", "implement")
out, code = _run_status(["--task", "show-impl"], tmp_project)
assert "Allowed actions" in out
assert "Edit code" in out
assert "Forbidden actions" in out
class TestList:
def test_list_shows_tasks(self, tmp_project):
_create_task(tmp_project, "list-a", "research")
_create_task(tmp_project, "list-b", "implement")
out, code = _run_status(["--list"], tmp_project)
assert code == 0
assert "list-a" in out
assert "list-b" in out
class TestToolHooks:
def test_can_edit_denied_in_research(self, tmp_project):
_create_task(tmp_project, "edit-no", "research")
out, code = _run_status(["--can-edit", "--task", "edit-no"], tmp_project)
assert code == 1
assert "DENIED" in out
def test_can_edit_allowed_in_implement(self, tmp_project):
_create_task(tmp_project, "edit-yes", "implement")
out, code = _run_status(["--can-edit", "--task", "edit-yes"], tmp_project)
assert code == 0
assert "ALLOWED" in out
def test_scope_check_in_scope(self, tmp_project):
test_file = tmp_project / "src" / "main.py"
test_file.parent.mkdir(parents=True, exist_ok=True)
test_file.write_text("# test")
out, code = _run_status(["--scope-check", "--task", "scope-test", "--file", str(test_file)], tmp_project)
assert "IN_SCOPE" in out
def test_scope_check_out_of_scope(self, tmp_project):
_create_task(tmp_project, "scope-test", "research")
out, code = _run_status(["--scope-check", "--task", "scope-test", "--file", "/tmp/some_random_file.py"], tmp_project)
assert code == 1
assert "OUT_OF_SCOPE" in out
def test_scope_check_framework_out_of_scope_for_project(self, tmp_project):
_create_task(tmp_project, "scope-proj", "research")
framework_file = Path.home() / ".automaton" / "scripts" / "status.py"
if framework_file.exists():
out, code = _run_status(["--scope-check", "--task", "scope-proj", "--file", str(framework_file)], tmp_project)
assert code == 1
assert "OUT_OF_SCOPE" in out
class TestProjectScoping:
def test_no_project_errors_without_flag(self):
import subprocess
cmd = [sys.executable, str(Path.home() / ".automaton" / "scripts" / "status.py"), "--list"]
result = subprocess.run(cmd, capture_output=True, text=True, cwd="/tmp")
assert result.returncode != 0
def test_project_flag_targets_correct_tasks(self, tmp_project):
_create_task(tmp_project, "scoped-task", "research")
out, code = _run_status(["--list"], tmp_project)
assert code == 0
assert "scoped-task" in out
class TestUntrackedTasks:
def test_transition_refuses_untracked_task(self, tmp_project):
tasks_dir = tmp_project / ".automaton" / "tasks"
task_dir = tasks_dir / "untracked-task"
task_dir.mkdir()
(task_dir / "SPEC.md").write_text("# Spec")
out, code = _run_status(["--transition", "design", "--task", "untracked-task"], tmp_project)
assert code == 1
assert "no .state file" in out
def test_can_edit_refuses_untracked_task(self, tmp_project):
tasks_dir = tmp_project / ".automaton" / "tasks"
task_dir = tasks_dir / "untracked-edit"
task_dir.mkdir()
(task_dir / "SPEC.md").write_text("# Spec")
out, code = _run_status(["--can-edit", "--task", "untracked-edit"], tmp_project)
assert code == 1
assert "no .state file" in out
def test_show_task_refuses_untracked_task(self, tmp_project):
tasks_dir = tmp_project / ".automaton" / "tasks"
task_dir = tasks_dir / "untracked-show"
task_dir.mkdir()
(task_dir / "SPEC.md").write_text("# Spec")
out, code = _run_status(["--task", "untracked-show"], tmp_project)
assert code == 1
assert "no .state file" in out
def test_list_shows_untracked_task(self, tmp_project):
tasks_dir = tmp_project / ".automaton" / "tasks"
task_dir = tasks_dir / "untracked-list"
task_dir.mkdir()
(task_dir / "SPEC.md").write_text("# Spec")
out, code = _run_status(["--list"], tmp_project)
assert code == 0
assert "UNTRACKED" in out
def test_upgrade_bootstraps_state_file(self, tmp_project):
tasks_dir = tmp_project / ".automaton" / "tasks"
task_dir = tasks_dir / "upgrade-test"
task_dir.mkdir()
(task_dir / "SPEC.md").write_text("# Spec")
out, code = _run_status(["--upgrade", "--task", "upgrade-test"], tmp_project)
assert code == 0
assert "Bootstrapped" in out
state_file = task_dir / ".state"
assert state_file.exists()
def test_upgrade_all_tasks(self, tmp_project):
tasks_dir = tmp_project / ".automaton" / "tasks"
t1 = tasks_dir / "task-a"
t1.mkdir()
(t1 / "SPEC.md").write_text("# Spec")
t2 = tasks_dir / "task-b"
t2.mkdir()
out, code = _run_status(["--upgrade"], tmp_project)
assert code == 0
assert (t1 / ".state").exists()
assert (t2 / ".state").exists()
class TestCanEditProject:
def test_can_edit_no_task_allowed_when_implement(self, tmp_project):
_create_task(tmp_project, "edit-impl", "implement")
out, code = _run_status(["--can-edit"], tmp_project)
assert code == 0
assert "ALLOWED" in out
assert "edit-impl" in out
def test_can_edit_no_task_allowed_when_doc_review(self, tmp_project):
_create_task(tmp_project, "edit-doc", "doc_review")
out, code = _run_status(["--can-edit"], tmp_project)
assert code == 0
assert "ALLOWED" in out
assert "edit-doc" in out
def test_can_edit_no_task_denied_when_no_edit_phase(self, tmp_project):
_create_task(tmp_project, "edit-res", "research")
out, code = _run_status(["--can-edit"], tmp_project)
assert code == 1
assert "DENIED" in out
def test_can_edit_no_task_denied_when_no_tasks(self, tmp_project):
out, code = _run_status(["--can-edit"], tmp_project)
assert code == 1
assert "DENIED" in out
def test_can_edit_with_file_scope(self, tmp_project):
_create_task(tmp_project, "edit-scope", "implement")
test_file = tmp_project / "src" / "main.py"
test_file.parent.mkdir(parents=True, exist_ok=True)
test_file.write_text("# test")
out, code = _run_status(["--can-edit", "--file", str(test_file)], tmp_project)
assert code == 0
assert "ALLOWED" in out
def test_can_edit_with_file_out_of_scope(self, tmp_project):
_create_task(tmp_project, "edit-scope-out", "implement")
out, code = _run_status(["--can-edit", "--file", "/tmp/some_random_file.py"], tmp_project)
assert code == 1
assert "outside project" in out
def test_can_edit_json_output(self, tmp_project):
_create_task(tmp_project, "edit-json", "implement")
out, code = _run_status(["--can-edit", "--json"], tmp_project)
assert code == 0
assert '"allowed": true' in out.lower() or '"allowed": True' in out
def test_can_edit_json_denied(self, tmp_project):
_create_task(tmp_project, "edit-json-denied", "research")
out, code = _run_status(["--can-edit", "--json"], tmp_project)
assert code == 1
assert '"allowed": false' in out.lower() or '"allowed": False' in out
class TestAudit:
def test_audit_reports_manual_task(self, tmp_project):
tasks_dir = tmp_project / ".automaton" / "tasks"
task_dir = tasks_dir / "manual-only"
task_dir.mkdir()
(task_dir / "SPEC.md").write_text("# Spec")
out, code = _run_status(["--audit"], tmp_project)
assert "manually created" in out
def test_audit_clean_task(self, tmp_project):
task_dir = _create_task(tmp_project, "clean-audit", "research")
(task_dir / "SPEC.md").write_text("# Spec")
out, code = _run_status(["--audit"], tmp_project)
assert "PASS" in out
+264 -1
View File
@@ -8,6 +8,9 @@ from automaton.dashboard.core.task import (
determine_task_state,
discover_tasks,
parse_sub_tasks,
parse_waves,
parse_vram_config,
WaveGroup,
TaskState,
)
@@ -39,7 +42,17 @@ def test_implementation_state(tmp_path: Path) -> None:
task_dir = _make_task(
tmp_path,
"impl-task",
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl"},
{"SPEC.md": "# Spec", "TEST_PLAN.md": "# Tests", "IMPLEMENTATION.md": "# Impl"},
)
state, _ = determine_task_state(task_dir)
assert state == TaskState.BUG_FIND
def test_implementation_from_test_plan(tmp_path: Path) -> None:
task_dir = _make_task(
tmp_path,
"impl-test-plan",
{"SPEC.md": "# Spec", "TEST_PLAN.md": "# Tests"},
)
state, _ = determine_task_state(task_dir)
assert state == TaskState.IMPLEMENT
@@ -132,3 +145,253 @@ def test_discover_tasks_skips_subtasks_root(tmp_path: Path) -> None:
assert len(tasks) == 1
assert tasks[0].name == "parent"
assert len(tasks[0].sub_tasks) == 1
class TestVerdictParsing:
"""Tests for structured verdict status parsing (R1-R3 of fix-verdict-parsing SPEC)."""
def test_pass_verdict_with_fail_in_findings(self, tmp_path: Path) -> None:
ver = "## Status: PASS\n\nThe previous FAIL finding was resolved."
task_dir = _make_task(tmp_path, "pass-with-fail", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
assert state == TaskState.DONE
def test_pass_verdict_with_needs_review_in_body(self, tmp_path: Path) -> None:
ver = "## Status: PASS\n\nNote: NEEDS_REVIEW was discussed but resolved."
task_dir = _make_task(tmp_path, "pass-with-nr", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
assert state == TaskState.DONE
def test_fail_verdict_structured(self, tmp_path: Path) -> None:
ver = "## Status: FAIL\n\n2 tests PASS, 1 test FAIL."
task_dir = _make_task(tmp_path, "fail-mentions-pass", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
assert state == TaskState.BLOCKED
def test_needs_review_verdict_structured(self, tmp_path: Path) -> None:
ver = "## Status: NEEDS_REVIEW\n\nSome items PASS but need review."
task_dir = _make_task(tmp_path, "nr-mentions-pass", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
assert state == TaskState.BLOCKED
def test_verdict_with_bold_status(self, tmp_path: Path) -> None:
ver = "# Verdict\n\n- **Status**: PASS\n- **Timestamp**: 2025-01-01"
task_dir = _make_task(tmp_path, "bold-status", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
assert state == TaskState.DONE
def test_verdict_no_status_line(self, tmp_path: Path) -> None:
ver = "# Verdict\nEverything looks good, PASS!"
task_dir = _make_task(tmp_path, "no-status-line", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
assert state == TaskState.DONE
def test_verdict_no_status_no_keywords(self, tmp_path: Path) -> None:
ver = "# Verdict\n\nNeeds further discussion."
task_dir = _make_task(tmp_path, "no-status-no-keywords", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
assert state == TaskState.REFEREE
class TestStateMachineAlignment:
"""Tests for state machine alignment with orchestrate.md (R4)."""
def test_implementation_alone_shows_bug_find(self, tmp_path: Path) -> None:
task_dir = _make_task(tmp_path, "impl-only", {"IMPLEMENTATION.md": "# Impl"})
state, _ = determine_task_state(task_dir)
assert state == TaskState.BUG_FIND
def test_bug_report_without_adversarial(self, tmp_path: Path) -> None:
task_dir = _make_task(
tmp_path, "bug-only",
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl", "BUG_REPORT.md": "# Bugs"},
)
state, _ = determine_task_state(task_dir)
assert state == TaskState.BUG_FIND
def test_both_bug_reports_shows_adv_bug_find(self, tmp_path: Path) -> None:
task_dir = _make_task(
tmp_path, "both-bugs",
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl",
"BUG_REPORT.md": "# Bugs", "ADVERSARIAL_BUG_REPORT.md": "# Adv"},
)
state, _ = determine_task_state(task_dir)
assert state == TaskState.ADV_BUG_FIND
def test_adv_bug_report_alone_shows_bug_find(self, tmp_path: Path) -> None:
task_dir = _make_task(
tmp_path, "adv-only",
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl",
"ADVERSARIAL_BUG_REPORT.md": "# Adv bugs only"},
)
state, _ = determine_task_state(task_dir)
assert state == TaskState.BUG_FIND
def test_spec_alone_shows_research(self, tmp_path: Path) -> None:
task_dir = _make_task(tmp_path, "spec-only", {"SPEC.md": "# Spec"})
state, _ = determine_task_state(task_dir)
assert state == TaskState.RESEARCH
def test_test_plan_shows_implement(self, tmp_path: Path) -> None:
task_dir = _make_task(tmp_path, "testplan", {"SPEC.md": "# Spec", "TEST_PLAN.md": "# Tests"})
state, _ = determine_task_state(task_dir)
assert state == TaskState.IMPLEMENT
def test_design_with_spec_shows_design(self, tmp_path: Path) -> None:
task_dir = _make_task(tmp_path, "design-spec", {"SPEC.md": "# Spec", "DESIGN.md": "# Design"})
state, _ = determine_task_state(task_dir)
assert state == TaskState.DESIGN
def test_doc_review_shows_doc_review(self, tmp_path: Path) -> None:
task_dir = _make_task(
tmp_path, "doc-review",
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl",
"BUG_REPORT.md": "# Bugs", "ADVERSARIAL_BUG_REPORT.md": "# Adv",
"DOC_REVIEW.md": "# Docs"},
)
state, _ = determine_task_state(task_dir)
assert state == TaskState.DOC_REVIEW
class TestTaskNameValidation:
"""Tests for filesystem-sourced task name validation."""
def test_valid_task_names(self, tmp_path: Path) -> None:
for name in ["my-task", "task_1", "Task-Name-123"]:
d = tmp_path / name
d.mkdir()
(d / "SPEC.md").write_text("# Spec")
tasks = discover_tasks(tmp_path)
assert len(tasks) == 3
def test_invalid_task_name_skipped(self, tmp_path: Path) -> None:
valid = tmp_path / "good-task"
valid.mkdir()
(valid / "SPEC.md").write_text("# Spec")
bad = tmp_path / "task with spaces"
bad.mkdir()
(bad / "SPEC.md").write_text("# Bad")
tasks = discover_tasks(tmp_path)
assert len(tasks) == 1
assert tasks[0].name == "good-task"
class TestParseWaves:
def test_parse_waves_from_template(self) -> None:
content = "# Decomposition: Test\n\n## Sub-Tasks\n\n### Wave 1 (Parallel)\n- subtask-a: Implement core\n- subtask-b: Implement model\n\n### Wave 2 (Dependent)\n- subtask-c: Implement UI\n- subtask-d: Implement tests\n"
waves = parse_waves(content)
assert len(waves) == 2
assert waves[0].wave_number == 1
assert waves[0].label == "Parallel"
assert waves[0].sub_task_names == ["subtask-a", "subtask-b"]
assert waves[1].wave_number == 2
assert waves[1].label == "Dependent"
assert waves[1].sub_task_names == ["subtask-c", "subtask-d"]
def test_parse_waves_empty(self) -> None:
assert parse_waves("") == []
assert parse_waves(None) == []
def test_parse_waves_no_waves(self) -> None:
content = "# No waves here\nJust text\n"
assert parse_waves(content) == []
def test_parse_waves_colon_format(self) -> None:
content = "### Wave 1: Setup\n- task-alpha: Do setup\n### Wave 2: Execution\n- task-beta: Do execution\n"
waves = parse_waves(content)
assert len(waves) == 2
assert waves[0].label == "Setup"
assert waves[0].sub_task_names == ["task-alpha"]
assert waves[1].label == "Execution"
assert waves[1].sub_task_names == ["task-beta"]
class TestDecompositionContent:
def test_task_has_decomposition_content(self, tmp_path: Path) -> None:
task_dir = tmp_path / "my-task"
task_dir.mkdir()
(task_dir / "SPEC.md").write_text("# Spec")
(task_dir / "DECOMPOSITION.md").write_text("### Wave 1 (Build)\n- sub-a: Build core\n")
tasks = discover_tasks(tmp_path)
assert len(tasks) == 1
assert tasks[0].decomposition_content is not None
assert "Wave 1" in tasks[0].decomposition_content
assert len(tasks[0].waves) == 1
assert tasks[0].waves[0].sub_task_names == ["sub-a"]
class TestParentSpecAndVramConfig:
def test_parent_spec_content(self, tmp_path: Path) -> None:
task_dir = tmp_path / "my-task"
task_dir.mkdir()
(task_dir / "SPEC.md").write_text("# Spec")
(task_dir / "PARENT_SPEC.md").write_text("# Parent Context\nDetails here")
tasks = discover_tasks(tmp_path)
assert tasks[0].parent_spec_content is not None
assert "Parent Context" in tasks[0].parent_spec_content
def test_vram_config_content(self, tmp_path: Path) -> None:
task_dir = tmp_path / "my-task"
task_dir.mkdir()
(task_dir / "SPEC.md").write_text("# Spec")
(task_dir / "VRAM_CONFIG.md").write_text("# VRAM\nmodel: llama-3")
tasks = discover_tasks(tmp_path)
assert tasks[0].vram_config_content is not None
assert "llama-3" in tasks[0].vram_config_content
def test_subtask_has_parent_spec(self, tmp_path: Path) -> None:
tasks_dir = tmp_path
parent = tasks_dir / "parent-task"
parent.mkdir()
(parent / "SPEC.md").write_text("# Parent")
subtasks_dir = parent / "subtasks"
subtasks_dir.mkdir()
sub = subtasks_dir / "child-a"
sub.mkdir()
(sub / "SPEC.md").write_text("# Child")
(sub / "PARENT_SPEC.md").write_text("# Parent Spec for child")
(sub / "VRAM_CONFIG.md").write_text("# VRAM config")
sub_tasks = parse_sub_tasks(parent)
assert len(sub_tasks) == 1
# discover_tasks doesn't recurse into subtasks for content, but parse_sub_tasks returns SubTask objects
# Verify the files exist
assert (sub / "PARENT_SPEC.md").exists()
class TestStatusReason:
def test_done_task_reason(self, tmp_path: Path) -> None:
task_dir = tmp_path / "done-task"
task_dir.mkdir()
(task_dir / "VERDICT.md").write_text("## Status: PASS\nAll good.\n")
(task_dir / "SPEC.md").write_text("# Spec\n")
tasks = discover_tasks(tmp_path)
assert tasks[0].status_reason == "Verdict: PASS"
def test_blocked_fail_reason(self, tmp_path: Path) -> None:
task_dir = tmp_path / "blocked-task"
task_dir.mkdir()
(task_dir / "VERDICT.md").write_text("## Status: FAIL\nBroken.\n")
(task_dir / "SPEC.md").write_text("# Spec\n")
tasks = discover_tasks(tmp_path)
assert "FAIL" in tasks[0].status_reason
def test_blocked_needs_review_reason(self, tmp_path: Path) -> None:
task_dir = tmp_path / "review-task"
task_dir.mkdir()
(task_dir / "VERDICT.md").write_text("## Status: NEEDS_REVIEW\nUnclear.\n")
(task_dir / "SPEC.md").write_text("# Spec\n")
tasks = discover_tasks(tmp_path)
assert "NEEDS_REVIEW" in tasks[0].status_reason
def test_bug_find_implementation_reason(self, tmp_path: Path) -> None:
task_dir = tmp_path / "impl-task"
task_dir.mkdir()
(task_dir / "IMPLEMENTATION.md").write_text("# Implementation\n")
tasks = discover_tasks(tmp_path)
assert "Implementation complete" in tasks[0].status_reason
def test_backlog_reason(self, tmp_path: Path) -> None:
task_dir = tmp_path / "empty-task"
task_dir.mkdir()
tasks = discover_tasks(tmp_path)
assert "No artifacts" in tasks[0].status_reason