State Enforcement (v2.0):
- .state file as single source of truth for task phase
- Approval gates for research, decomposition, design, test_design
- status.py --transition refuses illegal phase transitions
- status.py --validate-folder detects out-of-order artifacts
- status.py --audit checks all tasks for violations
- status.py --create-task is the only valid way to create tasks
- Pre-v2.0 tasks without .state are UNTRACKED -- all commands refuse them
- New --upgrade command bootstraps .state files for existing tasks
Project Scoping:
- --project flag added to all status.py commands across 16+ files
- _find_project_dir errors instead of silently falling back to ~/.automaton/
- --scope-check marks framework files OUT_OF_SCOPE when working on a project
- Dashboard handlers use stored project_root instead of re-detecting from CWD
- Prompts reference ~/.automaton/scripts/vram_detect.py (not {project}/.automaton/)
Harness Integration:
- status.py --can-edit now supports project-level checks (no --task required)
- --can-edit --file checks file scope without --task
- --json output for machine-readable harness integration
- opencode plugin (plugins/automaton-guard/plugin.ts) intercepts edit/write
- Git pre-commit hook (scripts/git-hooks/pre-commit) blocks commits without task
- Formal integration contract (contracts/harness-integration.md)
Other:
- upgrade.sh delegates to status.py --upgrade instead of manual heuristics
- Phase prompts reference --project {project} for multi-project scoping
- 200 tests passing (14 new)
This commit is contained in:
@@ -86,3 +86,159 @@ def test_static_valid_file(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> N
|
||||
assert response_status == [200]
|
||||
assert any(h[0] == "Content-Type" and h[1] == "text/html" for h in response_headers)
|
||||
assert handler.wfile.getvalue() == b"<html></html>"
|
||||
|
||||
|
||||
class TestCORSAndSecurityHeaders:
|
||||
"""Tests for CORS and security headers on API responses."""
|
||||
|
||||
def test_send_json_includes_cors(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
html_dir = tmp_path / "html"
|
||||
html_dir.mkdir()
|
||||
monkeypatch.setattr(DashboardHandler, "dashboard_path", html_dir)
|
||||
|
||||
handler = DashboardHandler.__new__(DashboardHandler)
|
||||
response_headers: list[tuple[str, str]] = []
|
||||
handler.send_response = lambda code: None
|
||||
handler.send_header = lambda k, v: response_headers.append((k, v))
|
||||
handler.end_headers = lambda: None
|
||||
handler.wfile = io.BytesIO()
|
||||
|
||||
handler._send_json({"test": True})
|
||||
header_dict = dict(response_headers)
|
||||
assert header_dict.get("Access-Control-Allow-Origin") == "*"
|
||||
assert header_dict.get("X-Content-Type-Options") == "nosniff"
|
||||
|
||||
def test_send_error_includes_cors(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
html_dir = tmp_path / "html"
|
||||
html_dir.mkdir()
|
||||
monkeypatch.setattr(DashboardHandler, "dashboard_path", html_dir)
|
||||
|
||||
handler = DashboardHandler.__new__(DashboardHandler)
|
||||
response_headers: list[tuple[str, str]] = []
|
||||
handler.send_response = lambda code: None
|
||||
handler.send_header = lambda k, v: response_headers.append((k, v))
|
||||
handler.end_headers = lambda: None
|
||||
handler.wfile = io.BytesIO()
|
||||
|
||||
handler._send_error(404, "Not found")
|
||||
header_dict = dict(response_headers)
|
||||
assert header_dict.get("Access-Control-Allow-Origin") == "*"
|
||||
assert header_dict.get("X-Content-Type-Options") == "nosniff"
|
||||
|
||||
|
||||
class TestContentLengthBound:
|
||||
"""Tests for POST content-length limits."""
|
||||
|
||||
def test_max_post_body_constant(self) -> None:
|
||||
from automaton.dashboard.ui.app import MAX_POST_BODY, MAX_REVIEW_COMMENT_LENGTH
|
||||
assert MAX_POST_BODY == 65536
|
||||
assert MAX_REVIEW_COMMENT_LENGTH == 4096
|
||||
|
||||
|
||||
class TestTaskCache:
|
||||
"""Tests for server-side task caching."""
|
||||
|
||||
def test_cache_returns_tasks(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
from automaton.dashboard.ui.app import _get_cached_tasks, _task_cache
|
||||
tasks_dir = tmp_path / ".automaton" / "tasks"
|
||||
task_dir = tasks_dir / "my-task"
|
||||
task_dir.mkdir(parents=True)
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
_task_cache["timestamp"] = 0.0
|
||||
_task_cache["tasks"] = []
|
||||
result = _get_cached_tasks(tmp_path)
|
||||
assert len(result) == 1
|
||||
assert result[0].name == "my-task"
|
||||
|
||||
def test_cache_uses_ttl(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
import time
|
||||
from automaton.dashboard.ui.app import _get_cached_tasks, _task_cache, CACHE_TTL
|
||||
tasks_dir = tmp_path / ".automaton" / "tasks"
|
||||
task_dir = tasks_dir / "cached-task"
|
||||
task_dir.mkdir(parents=True)
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
_task_cache["timestamp"] = 0.0
|
||||
_task_cache["tasks"] = []
|
||||
result1 = _get_cached_tasks(tmp_path)
|
||||
assert len(result1) == 1
|
||||
_task_cache["timestamp"] = time.time() + CACHE_TTL + 10
|
||||
new_task = tasks_dir / "new-task"
|
||||
new_task.mkdir()
|
||||
(new_task / "SPEC.md").write_text("# New")
|
||||
result2 = _get_cached_tasks(tmp_path)
|
||||
assert len(result2) == 1
|
||||
_task_cache["timestamp"] = 0.0
|
||||
result3 = _get_cached_tasks(tmp_path)
|
||||
assert len(result3) == 2
|
||||
|
||||
def test_invalidate_cache(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
import time
|
||||
from automaton.dashboard.ui.app import _invalidate_task_cache, _get_cached_tasks, _task_cache
|
||||
tasks_dir = tmp_path / ".automaton" / "tasks"
|
||||
task_dir = tasks_dir / "inv-task"
|
||||
task_dir.mkdir(parents=True)
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
_task_cache["timestamp"] = time.time() + 9999
|
||||
_task_cache["tasks"] = []
|
||||
_invalidate_task_cache()
|
||||
assert _task_cache["timestamp"] == 0.0
|
||||
|
||||
|
||||
class TestConfigEndpoint:
|
||||
"""Tests for GET/PUT /api/config."""
|
||||
|
||||
def _make_handler(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch, config=None):
|
||||
from automaton.dashboard.ui.app import DashboardHandler
|
||||
from automaton.dashboard.config import DashboardConfig
|
||||
html_dir = tmp_path / "html"
|
||||
html_dir.mkdir()
|
||||
monkeypatch.setattr(DashboardHandler, "dashboard_path", html_dir)
|
||||
handler = DashboardHandler.__new__(DashboardHandler)
|
||||
handler.config = config or DashboardConfig()
|
||||
handler.path = "/api/config"
|
||||
return handler
|
||||
|
||||
def test_serve_config(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
from automaton.dashboard.config import DashboardConfig
|
||||
handler = self._make_handler(tmp_path, monkeypatch)
|
||||
response_data = {}
|
||||
handler._send_json = lambda d: response_data.update(d)
|
||||
handler._serve_config()
|
||||
assert response_data["theme"] == "default"
|
||||
assert response_data["auto_refresh_interval"] == 2
|
||||
assert response_data["default_view"] == "board"
|
||||
|
||||
def test_serve_config_not_available(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
handler = self._make_handler(tmp_path, monkeypatch)
|
||||
handler.config = None
|
||||
errors = []
|
||||
handler._send_error = lambda c, m: errors.append((c, m))
|
||||
handler._serve_config()
|
||||
assert errors[0][0] == 503
|
||||
|
||||
def test_handle_config_update(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
from automaton.dashboard.config import DashboardConfig
|
||||
from automaton.dashboard.ui.app import DashboardHandler
|
||||
tasks_dir = tmp_path / ".automaton" / "tasks"
|
||||
tasks_dir.mkdir(parents=True)
|
||||
monkeypatch.setattr("automaton.dashboard.ui.app.get_config_path",
|
||||
lambda pr: tmp_path / ".automaton" / "dashboard-config.json")
|
||||
handler = self._make_handler(tmp_path, monkeypatch)
|
||||
handler.project_root = tmp_path
|
||||
handler.headers = {"Content-Length": "19"}
|
||||
handler.rfile = io.BytesIO(b'{"theme": "dark"}')
|
||||
response_data = {}
|
||||
handler._send_json = lambda d: response_data.update(d)
|
||||
handler._handle_config_update()
|
||||
assert response_data["theme"] == "dark"
|
||||
assert handler.config.theme == "dark"
|
||||
|
||||
def test_handle_config_update_invalid(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
from automaton.dashboard.config import DashboardConfig
|
||||
handler = self._make_handler(tmp_path, monkeypatch)
|
||||
handler.headers = {"Content-Length": "39"}
|
||||
handler.rfile = io.BytesIO(b'{"auto_refresh_interval": 999}')
|
||||
errors = []
|
||||
handler._send_error = lambda c, m: errors.append((c, m))
|
||||
handler._handle_config_update()
|
||||
assert errors[0][0] == 400
|
||||
|
||||
@@ -0,0 +1,254 @@
|
||||
"""Framework self-consistency tests.
|
||||
|
||||
These tests enforce structural rules on the framework itself — prompt
|
||||
consistency, canonical paths, state machine integrity, and configuration
|
||||
validity. They catch regressions that unit tests alone cannot.
|
||||
"""
|
||||
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
PROMPTS_DIR = ROOT / "prompts"
|
||||
RULES_FILE = ROOT / ".rules.md"
|
||||
PYPROJECT_FILE = ROOT / "pyproject.toml"
|
||||
CI_FILE = ROOT / ".gitea" / "workflows" / "ci.yml"
|
||||
CSS_FILE = ROOT / "automaton" / "dashboard" / "html" / "styles.css"
|
||||
JS_FILE = ROOT / "automaton" / "dashboard" / "html" / "dashboard.js"
|
||||
|
||||
|
||||
class TestDeliveryPromptsHaveStopConditions:
|
||||
"""R1.A: Every delivery prompt must contain a stop condition block."""
|
||||
|
||||
EXCLUDED = {"orchestrate.md", "compaction.md", "workflow.md", "subtask_management.md", "onboarding.md"}
|
||||
|
||||
@pytest.fixture()
|
||||
def delivery_prompts(self):
|
||||
if not PROMPTS_DIR.exists():
|
||||
pytest.skip("prompts/ directory not found")
|
||||
return [
|
||||
f
|
||||
for f in sorted(PROMPTS_DIR.iterdir())
|
||||
if f.is_file() and f.suffix == ".md" and f.name not in self.EXCLUDED
|
||||
]
|
||||
|
||||
def test_each_prompt_has_stop_condition(self, delivery_prompts):
|
||||
for prompt in delivery_prompts:
|
||||
content = prompt.read_text()
|
||||
has_stop = "## Stop Condition" in content or "STOP CONDITION" in content or "CONTRACT_MET" in content
|
||||
assert has_stop, f"{prompt.name} is missing a stop condition block"
|
||||
|
||||
|
||||
class TestNoHardcodedURLs:
|
||||
"""R1.B: Prompts, contracts, and templates must not contain hardcoded repo URLs."""
|
||||
|
||||
ALLOWED_FILES = {"install.sh"}
|
||||
|
||||
def _check_dir(self, directory: Path, pattern: re.Pattern):
|
||||
violations = []
|
||||
if not directory.exists():
|
||||
return violations
|
||||
for f in directory.rglob("*.md"):
|
||||
if f.name in self.ALLOWED_FILES:
|
||||
continue
|
||||
content = f.read_text()
|
||||
for line_no, line in enumerate(content.splitlines(), 1):
|
||||
if pattern.search(line):
|
||||
violations.append(f"{f.relative_to(ROOT)}:{line_no}")
|
||||
return violations
|
||||
|
||||
def test_no_hardcoded_ip_urls(self):
|
||||
pattern = re.compile(r"\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}[:/]")
|
||||
dirs = [PROMPTS_DIR, ROOT / "contracts", ROOT / "templates"]
|
||||
violations = []
|
||||
for d in dirs:
|
||||
violations.extend(self._check_dir(d, pattern))
|
||||
assert not violations, f"Hardcoded IP URLs found in: {violations}"
|
||||
|
||||
def test_no_hardcoded_localhost_ports(self):
|
||||
pattern = re.compile(r"localhost:\d{4,5}")
|
||||
dirs = [PROMPTS_DIR, ROOT / "contracts", ROOT / "templates"]
|
||||
violations = []
|
||||
for d in dirs:
|
||||
violations.extend(self._check_dir(d, pattern))
|
||||
assert not violations, f"Hardcoded localhost URLs found in: {violations}"
|
||||
|
||||
|
||||
class TestRulesMdSections:
|
||||
"""R1.C & R1.D: .rules.md must contain mandatory sections."""
|
||||
|
||||
MANDATORY_SECTIONS = [
|
||||
"Task-Driven Development",
|
||||
"VRAM",
|
||||
"Changelog",
|
||||
"Session Discipline",
|
||||
"Scope Confinement",
|
||||
"Artifact Integrity",
|
||||
]
|
||||
|
||||
def test_mandatory_sections_present(self):
|
||||
if not RULES_FILE.exists():
|
||||
pytest.skip(".rules.md not found")
|
||||
content = RULES_FILE.read_text()
|
||||
for section in self.MANDATORY_SECTIONS:
|
||||
assert section in content, f".rules.md missing mandatory section: {section}"
|
||||
|
||||
def test_self_improvement_has_example(self):
|
||||
if not RULES_FILE.exists():
|
||||
pytest.skip(".rules.md not found")
|
||||
content = RULES_FILE.read_text()
|
||||
assert "Past failure" in content, ".rules.md Self-Improvement section must reference at least one real failure mode"
|
||||
|
||||
|
||||
class TestCanonicalTaskPaths:
|
||||
"""R1.E: All task path references in prompts must use the canonical format."""
|
||||
|
||||
CANONICAL_PATTERN = re.compile(r"\{project\}/\.automaton/tasks/\w")
|
||||
DEPRECATED_PATTERN = re.compile(r"\{project\}/tasks/[\w-]+/")
|
||||
|
||||
def test_no_deprecated_task_paths(self):
|
||||
if not PROMPTS_DIR.exists():
|
||||
pytest.skip("prompts/ directory not found")
|
||||
violations = []
|
||||
for f in sorted(PROMPTS_DIR.iterdir()):
|
||||
if not f.is_file() or f.suffix != ".md":
|
||||
continue
|
||||
content = f.read_text()
|
||||
for line_no, line in enumerate(content.splitlines(), 1):
|
||||
if self.DEPRECATED_PATTERN.search(line):
|
||||
violations.append(f"{f.name}:{line_no}: {line.strip()}")
|
||||
assert not violations, f"Deprecated task paths found: {violations}"
|
||||
|
||||
|
||||
class TestPyprojectNoStaleExtras:
|
||||
"""R1.F: pyproject.toml must not reference inotify."""
|
||||
|
||||
def test_no_inotify_dependency(self):
|
||||
if not PYPROJECT_FILE.exists():
|
||||
pytest.skip("pyproject.toml not found")
|
||||
content = PYPROJECT_FILE.read_text()
|
||||
assert "inotify" not in content, "pyproject.toml still references inotify"
|
||||
|
||||
|
||||
class TestDashboardCSSThemes:
|
||||
"""R1.G: Light and dark themes must define the same variable set."""
|
||||
|
||||
def _extract_vars(self, section: str) -> set:
|
||||
pattern = re.compile(r"--([\w-]+)\s*:", re.MULTILINE)
|
||||
return set(pattern.findall(section))
|
||||
|
||||
def test_theme_variable_parity(self):
|
||||
if not CSS_FILE.exists():
|
||||
pytest.skip("styles.css not found")
|
||||
content = CSS_FILE.read_text()
|
||||
root_match = re.search(r":root\s*\{([^}]+)\}", content, re.DOTALL)
|
||||
light_match = re.search(r'\[data-theme="light"\]\s*\{([^}]+)\}', content, re.DOTALL)
|
||||
dark_match = re.search(r'\[data-theme="dark"\]\s*\{([^}]+)\}', content, re.DOTALL)
|
||||
if not root_match:
|
||||
pytest.skip(":root CSS variables not found")
|
||||
root_vars = self._extract_vars(root_match.group(1))
|
||||
theme_vars = set()
|
||||
if light_match:
|
||||
theme_vars |= self._extract_vars(light_match.group(1))
|
||||
if dark_match:
|
||||
theme_vars |= self._extract_vars(dark_match.group(1))
|
||||
if not theme_vars:
|
||||
pytest.skip("No theme sections found in CSS")
|
||||
structural_vars = {"radius-sm", "radius-md", "radius-lg"}
|
||||
themable_root_vars = root_vars - structural_vars
|
||||
missing_from_themes = themable_root_vars - theme_vars
|
||||
assert not missing_from_themes, f"CSS variables defined in :root but missing from theme overrides: {missing_from_themes}"
|
||||
|
||||
|
||||
class TestVerdictParsingRegression:
|
||||
"""R3: Verdict parsing regression tests for the critical false-BLOCKED bug."""
|
||||
|
||||
def test_pass_verdict_mentioning_fail_is_done(self):
|
||||
from automaton.dashboard.core.task import determine_task_state, TaskState
|
||||
from pathlib import Path
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
task_dir = Path(tmp) / "test-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "VERDICT.md").write_text("## Status: PASS\n\nThe bug in the FAIL case is now fixed.\n")
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.DONE, f"Expected DONE, got {state}"
|
||||
|
||||
def test_pass_verdict_with_needs_review_mention_is_done(self):
|
||||
from automaton.dashboard.core.task import determine_task_state, TaskState
|
||||
from pathlib import Path
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
task_dir = Path(tmp) / "test-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "VERDICT.md").write_text("## Status: PASS\n\nPreviously flagged as NEEDS_REVIEW but resolved.\n")
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.DONE, f"Expected DONE, got {state}"
|
||||
|
||||
def test_structured_fail_verdict_is_blocked(self):
|
||||
from automaton.dashboard.core.task import determine_task_state, TaskState
|
||||
from pathlib import Path
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
task_dir = Path(tmp) / "test-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "VERDICT.md").write_text("## Status: FAIL\n\nThe implementation has a critical bug.\n")
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.BLOCKED
|
||||
|
||||
def test_implementation_alone_is_bug_find(self):
|
||||
from automaton.dashboard.core.task import determine_task_state, TaskState
|
||||
from pathlib import Path
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
task_dir = Path(tmp) / "test-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "IMPLEMENTATION.md").write_text("# Implementation\nDone.\n")
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.BUG_FIND
|
||||
|
||||
def test_empty_verdict_is_blocked(self):
|
||||
from automaton.dashboard.core.task import determine_task_state, TaskState
|
||||
from pathlib import Path
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
task_dir = Path(tmp) / "test-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "VERDICT.md").write_text("")
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.BLOCKED
|
||||
|
||||
def test_adv_bug_report_alone_is_bug_find(self):
|
||||
from automaton.dashboard.core.task import determine_task_state, TaskState
|
||||
from pathlib import Path
|
||||
import tempfile
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
task_dir = Path(tmp) / "test-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "ADVERSARIAL_BUG_REPORT.md").write_text("# Bug Report\nFound issue.\n")
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.BUG_FIND
|
||||
|
||||
|
||||
class TestCIWorkflowValidation:
|
||||
"""R4: CI workflow must compile, test, and check shell scripts."""
|
||||
|
||||
def test_ci_runs_py_compile(self):
|
||||
if not CI_FILE.exists():
|
||||
pytest.skip("CI workflow not found")
|
||||
content = CI_FILE.read_text()
|
||||
assert "py_compile" in content, "CI must run py_compile"
|
||||
|
||||
def test_ci_runs_pytest(self):
|
||||
if not CI_FILE.exists():
|
||||
pytest.skip("CI workflow not found")
|
||||
content = CI_FILE.read_text()
|
||||
assert "pytest" in content, "CI must run pytest"
|
||||
|
||||
def test_ci_checks_shell_scripts(self):
|
||||
if not CI_FILE.exists():
|
||||
pytest.skip("CI workflow not found")
|
||||
content = CI_FILE.read_text()
|
||||
assert "bash -n" in content, "CI must syntax-check shell scripts"
|
||||
@@ -9,10 +9,13 @@ ROOT = Path(__file__).resolve().parent.parent
|
||||
PROMPTS_DIR = ROOT / "prompts"
|
||||
TEMPLATES_DIR = ROOT / "templates"
|
||||
|
||||
# Legacy path pattern that should no longer appear.
|
||||
# Legacy path pattern that should no longer appear (with placeholder).
|
||||
LEGACY_PATH = re.compile(r"\{project\}/tasks/\{task-name\}/")
|
||||
# Canonical path pattern that should be used instead.
|
||||
CANONICAL_PATH = re.compile(r"\{project\}/\.automaton/tasks/\{task-name\}/")
|
||||
# Concrete legacy path pattern: {project}/tasks/ followed by any name.
|
||||
# Catches paths like {project}/tasks/onboarding/ that use concrete names.
|
||||
CONCRETE_LEGACY_PATH = re.compile(r"\{project\}/tasks/[\w-]+/")
|
||||
|
||||
|
||||
def _markdown_files(*directories: Path) -> list[Path]:
|
||||
@@ -34,6 +37,32 @@ def test_no_legacy_task_paths(path: Path) -> None:
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("path", _markdown_files(PROMPTS_DIR, TEMPLATES_DIR))
|
||||
def test_no_concrete_legacy_task_paths(path: Path) -> None:
|
||||
"""Every prompt/template must not use {project}/tasks/ with concrete task names."""
|
||||
text = path.read_text(encoding="utf-8")
|
||||
# Exclude the migration detection reference, which legitimately mentions the deprecated path.
|
||||
concrete_matches = CONCRETE_LEGACY_PATH.findall(text)
|
||||
# Filter out the onboarding migration detection section that references the legacy path.
|
||||
if concrete_matches and "Migration Check" in text:
|
||||
lines = text.splitlines()
|
||||
filtered = []
|
||||
in_migration = False
|
||||
for line in lines:
|
||||
if "Migration Check" in line or "Migration Detection" in line:
|
||||
in_migration = True
|
||||
if in_migration and line.startswith("#") and "Migration" not in line:
|
||||
in_migration = False
|
||||
if not in_migration:
|
||||
for m in CONCRETE_LEGACY_PATH.findall(line):
|
||||
filtered.append(m)
|
||||
concrete_matches = filtered
|
||||
assert not concrete_matches, (
|
||||
f"Found concrete legacy task path in {path.relative_to(ROOT)}: {concrete_matches}\n"
|
||||
"Use {project}/.automaton/tasks/ instead of {project}/tasks/."
|
||||
)
|
||||
|
||||
|
||||
def test_canonical_path_present_in_prompts() -> None:
|
||||
"""At least one prompt uses the canonical path (sanity check)."""
|
||||
found = False
|
||||
|
||||
@@ -0,0 +1,365 @@
|
||||
"""Tests for status.py enforcement script."""
|
||||
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent / "scripts"))
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def tmp_project(tmp_path):
|
||||
"""Create a temporary project with .automaton/tasks structure."""
|
||||
auto_dir = tmp_path / ".automaton"
|
||||
tasks_dir = auto_dir / "tasks"
|
||||
tasks_dir.mkdir(parents=True)
|
||||
return tmp_path
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def status_script():
|
||||
"""Return path to status.py."""
|
||||
return Path.home() / ".automaton" / "scripts" / "status.py"
|
||||
|
||||
|
||||
def _run_status(args, project=None):
|
||||
"""Run status.py with given args and return (stdout, exit_code)."""
|
||||
import subprocess
|
||||
cmd = [sys.executable, str(Path.home() / ".automaton" / "scripts" / "status.py")]
|
||||
if project:
|
||||
cmd.extend(["--project", str(project)])
|
||||
cmd.extend(args)
|
||||
result = subprocess.run(cmd, capture_output=True, text=True)
|
||||
return result.stdout.strip(), result.returncode
|
||||
|
||||
|
||||
def _create_task(project, task_name, phase=None):
|
||||
"""Create a task and optionally set its phase."""
|
||||
out, code = _run_status(["--create-task", task_name], project)
|
||||
assert code == 0, f"Failed to create task: {out}"
|
||||
if phase:
|
||||
task_dir = project / ".automaton" / "tasks" / task_name
|
||||
(task_dir / ".state").write_text(f"{phase}\n")
|
||||
return project / ".automaton" / "tasks" / task_name
|
||||
|
||||
|
||||
class TestCreateTask:
|
||||
def test_create_valid_task(self, tmp_project):
|
||||
out, code = _run_status(["--create-task", "my-test-task"], tmp_project)
|
||||
assert code == 0
|
||||
assert "Created task 'my-test-task'" in out
|
||||
task_dir = tmp_project / ".automaton" / "tasks" / "my-test-task"
|
||||
assert task_dir.exists()
|
||||
assert (task_dir / ".state").read_text().strip() == "new"
|
||||
assert (task_dir / ".state.approvals").exists()
|
||||
|
||||
def test_create_task_rejects_spaces(self, tmp_project):
|
||||
out, code = _run_status(["--create-task", "my test task"], tmp_project)
|
||||
assert code == 2
|
||||
assert "kebab-case" in out
|
||||
|
||||
def test_create_task_rejects_uppercase(self, tmp_project):
|
||||
out, code = _run_status(["--create-task", "My-Task"], tmp_project)
|
||||
assert code == 2
|
||||
|
||||
def test_create_task_rejects_duplicate(self, tmp_project):
|
||||
_run_status(["--create-task", "my-task"], tmp_project)
|
||||
out, code = _run_status(["--create-task", "my-task"], tmp_project)
|
||||
assert code == 2
|
||||
assert "already exists" in out
|
||||
|
||||
|
||||
class TestStateTransitions:
|
||||
def test_transition_new_to_research(self, tmp_project):
|
||||
task_dir = _create_task(tmp_project, "trans-test", "new")
|
||||
out, code = _run_status(["--task", "trans-test", "--transition", "research"], tmp_project)
|
||||
assert code == 0
|
||||
assert "Transitioned" in out
|
||||
assert (task_dir / ".state").read_text().strip() == "research"
|
||||
|
||||
def test_illegal_transition_refused(self, tmp_project):
|
||||
task_dir = _create_task(tmp_project, "illegal-test", "new")
|
||||
out, code = _run_status(["--task", "illegal-test", "--transition", "implement"], tmp_project)
|
||||
assert code == 1
|
||||
assert "Cannot transition" in out
|
||||
|
||||
def test_approval_gate_blocks_transition(self, tmp_project):
|
||||
task_dir = _create_task(tmp_project, "approval-test", "research:awaiting_approval")
|
||||
out, code = _run_status(["--task", "approval-test", "--transition", "design"], tmp_project)
|
||||
assert code == 1
|
||||
assert "research:approved" in out
|
||||
|
||||
|
||||
class TestApprovalGates:
|
||||
def test_approve_transitions_to_approved(self, tmp_project):
|
||||
task_dir = _create_task(tmp_project, "approve-test", "research:awaiting_approval")
|
||||
out, code = _run_status(["--approve", "--task", "approve-test"], tmp_project)
|
||||
assert code == 0
|
||||
assert "Approved" in out
|
||||
assert (task_dir / ".state").read_text().strip() == "research:approved"
|
||||
|
||||
def test_approve_records_in_log(self, tmp_project):
|
||||
task_dir = _create_task(tmp_project, "log-test", "research:awaiting_approval")
|
||||
_run_status(["--approve", "--task", "log-test"], tmp_project)
|
||||
approvals = (task_dir / ".state.approvals").read_text()
|
||||
assert "research:approved" in approvals
|
||||
assert "user" in approvals
|
||||
|
||||
def test_approve_refused_for_wrong_phase(self, tmp_project):
|
||||
_create_task(tmp_project, "wrong-phase", "implement")
|
||||
out, code = _run_status(["--approve", "--task", "wrong-phase"], tmp_project)
|
||||
assert "does not require approval" in out
|
||||
|
||||
def test_approve_refused_without_awaiting(self, tmp_project):
|
||||
_create_task(tmp_project, "no-await", "research")
|
||||
out, code = _run_status(["--approve", "--task", "no-await"], tmp_project)
|
||||
assert code == 1
|
||||
assert "not awaiting approval" in out
|
||||
|
||||
|
||||
class TestValidateFolder:
|
||||
def test_valid_folder_passes(self, tmp_project):
|
||||
task_dir = _create_task(tmp_project, "valid-test", "research")
|
||||
# Write SPEC.md (expected artifact for research)
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
out, code = _run_status(["--validate-folder", "--task", "valid-test"], tmp_project)
|
||||
assert "PASS" in out
|
||||
|
||||
def test_folder_without_state_flagged(self, tmp_project):
|
||||
tasks_dir = tmp_project / ".automaton" / "tasks"
|
||||
task_dir = tasks_dir / "manual-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
out, code = _run_status(["--validate-folder", "--task", "manual-task"], tmp_project)
|
||||
assert "no .state file" in out or "manually" in out
|
||||
|
||||
def test_out_of_order_artifacts_flagged(self, tmp_project):
|
||||
task_dir = _create_task(tmp_project, "skip-test", "research")
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
(task_dir / "IMPLEMENTATION.md").write_text("# Impl")
|
||||
out, code = _run_status(["--validate-folder", "--task", "skip-test"], tmp_project)
|
||||
assert "FAIL" in out
|
||||
assert "IMPLEMENTATION.md" in out
|
||||
|
||||
|
||||
class TestArtifactValidation:
|
||||
def test_missing_required_artifact_blocks_transition(self, tmp_project):
|
||||
_create_task(tmp_project, "no-spec", "research")
|
||||
out, code = _run_status(["--task", "no-spec", "--transition", "design"], tmp_project)
|
||||
assert code == 1
|
||||
assert "SPEC.md" in out
|
||||
|
||||
def test_forbidden_artifact_blocks_transition(self, tmp_project):
|
||||
task_dir = _create_task(tmp_project, "skip-artifact", "research")
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
(task_dir / "IMPLEMENTATION.md").write_text("# Impl")
|
||||
out, code = _run_status(["--task", "skip-artifact", "--transition", "design"], tmp_project)
|
||||
assert code == 1
|
||||
assert "out-of-order" in out.lower() or "forbidden" in out.lower()
|
||||
|
||||
|
||||
class TestShowTask:
|
||||
def test_show_task_displays_phase(self, tmp_project):
|
||||
_create_task(tmp_project, "show-test", "implement")
|
||||
out, code = _run_status(["--task", "show-test"], tmp_project)
|
||||
assert code == 0
|
||||
assert "Phase: implement" in out
|
||||
|
||||
def test_show_task_displays_allowed_forbidden(self, tmp_project):
|
||||
_create_task(tmp_project, "show-impl", "implement")
|
||||
out, code = _run_status(["--task", "show-impl"], tmp_project)
|
||||
assert "Allowed actions" in out
|
||||
assert "Edit code" in out
|
||||
assert "Forbidden actions" in out
|
||||
|
||||
|
||||
class TestList:
|
||||
def test_list_shows_tasks(self, tmp_project):
|
||||
_create_task(tmp_project, "list-a", "research")
|
||||
_create_task(tmp_project, "list-b", "implement")
|
||||
out, code = _run_status(["--list"], tmp_project)
|
||||
assert code == 0
|
||||
assert "list-a" in out
|
||||
assert "list-b" in out
|
||||
|
||||
|
||||
class TestToolHooks:
|
||||
def test_can_edit_denied_in_research(self, tmp_project):
|
||||
_create_task(tmp_project, "edit-no", "research")
|
||||
out, code = _run_status(["--can-edit", "--task", "edit-no"], tmp_project)
|
||||
assert code == 1
|
||||
assert "DENIED" in out
|
||||
|
||||
def test_can_edit_allowed_in_implement(self, tmp_project):
|
||||
_create_task(tmp_project, "edit-yes", "implement")
|
||||
out, code = _run_status(["--can-edit", "--task", "edit-yes"], tmp_project)
|
||||
assert code == 0
|
||||
assert "ALLOWED" in out
|
||||
|
||||
def test_scope_check_in_scope(self, tmp_project):
|
||||
test_file = tmp_project / "src" / "main.py"
|
||||
test_file.parent.mkdir(parents=True, exist_ok=True)
|
||||
test_file.write_text("# test")
|
||||
out, code = _run_status(["--scope-check", "--task", "scope-test", "--file", str(test_file)], tmp_project)
|
||||
assert "IN_SCOPE" in out
|
||||
|
||||
def test_scope_check_out_of_scope(self, tmp_project):
|
||||
_create_task(tmp_project, "scope-test", "research")
|
||||
out, code = _run_status(["--scope-check", "--task", "scope-test", "--file", "/tmp/some_random_file.py"], tmp_project)
|
||||
assert code == 1
|
||||
assert "OUT_OF_SCOPE" in out
|
||||
|
||||
def test_scope_check_framework_out_of_scope_for_project(self, tmp_project):
|
||||
_create_task(tmp_project, "scope-proj", "research")
|
||||
framework_file = Path.home() / ".automaton" / "scripts" / "status.py"
|
||||
if framework_file.exists():
|
||||
out, code = _run_status(["--scope-check", "--task", "scope-proj", "--file", str(framework_file)], tmp_project)
|
||||
assert code == 1
|
||||
assert "OUT_OF_SCOPE" in out
|
||||
|
||||
|
||||
class TestProjectScoping:
|
||||
def test_no_project_errors_without_flag(self):
|
||||
import subprocess
|
||||
cmd = [sys.executable, str(Path.home() / ".automaton" / "scripts" / "status.py"), "--list"]
|
||||
result = subprocess.run(cmd, capture_output=True, text=True, cwd="/tmp")
|
||||
assert result.returncode != 0
|
||||
|
||||
def test_project_flag_targets_correct_tasks(self, tmp_project):
|
||||
_create_task(tmp_project, "scoped-task", "research")
|
||||
out, code = _run_status(["--list"], tmp_project)
|
||||
assert code == 0
|
||||
assert "scoped-task" in out
|
||||
|
||||
|
||||
class TestUntrackedTasks:
|
||||
def test_transition_refuses_untracked_task(self, tmp_project):
|
||||
tasks_dir = tmp_project / ".automaton" / "tasks"
|
||||
task_dir = tasks_dir / "untracked-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
out, code = _run_status(["--transition", "design", "--task", "untracked-task"], tmp_project)
|
||||
assert code == 1
|
||||
assert "no .state file" in out
|
||||
|
||||
def test_can_edit_refuses_untracked_task(self, tmp_project):
|
||||
tasks_dir = tmp_project / ".automaton" / "tasks"
|
||||
task_dir = tasks_dir / "untracked-edit"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
out, code = _run_status(["--can-edit", "--task", "untracked-edit"], tmp_project)
|
||||
assert code == 1
|
||||
assert "no .state file" in out
|
||||
|
||||
def test_show_task_refuses_untracked_task(self, tmp_project):
|
||||
tasks_dir = tmp_project / ".automaton" / "tasks"
|
||||
task_dir = tasks_dir / "untracked-show"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
out, code = _run_status(["--task", "untracked-show"], tmp_project)
|
||||
assert code == 1
|
||||
assert "no .state file" in out
|
||||
|
||||
def test_list_shows_untracked_task(self, tmp_project):
|
||||
tasks_dir = tmp_project / ".automaton" / "tasks"
|
||||
task_dir = tasks_dir / "untracked-list"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
out, code = _run_status(["--list"], tmp_project)
|
||||
assert code == 0
|
||||
assert "UNTRACKED" in out
|
||||
|
||||
def test_upgrade_bootstraps_state_file(self, tmp_project):
|
||||
tasks_dir = tmp_project / ".automaton" / "tasks"
|
||||
task_dir = tasks_dir / "upgrade-test"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
out, code = _run_status(["--upgrade", "--task", "upgrade-test"], tmp_project)
|
||||
assert code == 0
|
||||
assert "Bootstrapped" in out
|
||||
state_file = task_dir / ".state"
|
||||
assert state_file.exists()
|
||||
|
||||
def test_upgrade_all_tasks(self, tmp_project):
|
||||
tasks_dir = tmp_project / ".automaton" / "tasks"
|
||||
t1 = tasks_dir / "task-a"
|
||||
t1.mkdir()
|
||||
(t1 / "SPEC.md").write_text("# Spec")
|
||||
t2 = tasks_dir / "task-b"
|
||||
t2.mkdir()
|
||||
out, code = _run_status(["--upgrade"], tmp_project)
|
||||
assert code == 0
|
||||
assert (t1 / ".state").exists()
|
||||
assert (t2 / ".state").exists()
|
||||
|
||||
|
||||
class TestCanEditProject:
|
||||
def test_can_edit_no_task_allowed_when_implement(self, tmp_project):
|
||||
_create_task(tmp_project, "edit-impl", "implement")
|
||||
out, code = _run_status(["--can-edit"], tmp_project)
|
||||
assert code == 0
|
||||
assert "ALLOWED" in out
|
||||
assert "edit-impl" in out
|
||||
|
||||
def test_can_edit_no_task_allowed_when_doc_review(self, tmp_project):
|
||||
_create_task(tmp_project, "edit-doc", "doc_review")
|
||||
out, code = _run_status(["--can-edit"], tmp_project)
|
||||
assert code == 0
|
||||
assert "ALLOWED" in out
|
||||
assert "edit-doc" in out
|
||||
|
||||
def test_can_edit_no_task_denied_when_no_edit_phase(self, tmp_project):
|
||||
_create_task(tmp_project, "edit-res", "research")
|
||||
out, code = _run_status(["--can-edit"], tmp_project)
|
||||
assert code == 1
|
||||
assert "DENIED" in out
|
||||
|
||||
def test_can_edit_no_task_denied_when_no_tasks(self, tmp_project):
|
||||
out, code = _run_status(["--can-edit"], tmp_project)
|
||||
assert code == 1
|
||||
assert "DENIED" in out
|
||||
|
||||
def test_can_edit_with_file_scope(self, tmp_project):
|
||||
_create_task(tmp_project, "edit-scope", "implement")
|
||||
test_file = tmp_project / "src" / "main.py"
|
||||
test_file.parent.mkdir(parents=True, exist_ok=True)
|
||||
test_file.write_text("# test")
|
||||
out, code = _run_status(["--can-edit", "--file", str(test_file)], tmp_project)
|
||||
assert code == 0
|
||||
assert "ALLOWED" in out
|
||||
|
||||
def test_can_edit_with_file_out_of_scope(self, tmp_project):
|
||||
_create_task(tmp_project, "edit-scope-out", "implement")
|
||||
out, code = _run_status(["--can-edit", "--file", "/tmp/some_random_file.py"], tmp_project)
|
||||
assert code == 1
|
||||
assert "outside project" in out
|
||||
|
||||
def test_can_edit_json_output(self, tmp_project):
|
||||
_create_task(tmp_project, "edit-json", "implement")
|
||||
out, code = _run_status(["--can-edit", "--json"], tmp_project)
|
||||
assert code == 0
|
||||
assert '"allowed": true' in out.lower() or '"allowed": True' in out
|
||||
|
||||
def test_can_edit_json_denied(self, tmp_project):
|
||||
_create_task(tmp_project, "edit-json-denied", "research")
|
||||
out, code = _run_status(["--can-edit", "--json"], tmp_project)
|
||||
assert code == 1
|
||||
assert '"allowed": false' in out.lower() or '"allowed": False' in out
|
||||
|
||||
|
||||
class TestAudit:
|
||||
def test_audit_reports_manual_task(self, tmp_project):
|
||||
tasks_dir = tmp_project / ".automaton" / "tasks"
|
||||
task_dir = tasks_dir / "manual-only"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
out, code = _run_status(["--audit"], tmp_project)
|
||||
assert "manually created" in out
|
||||
|
||||
def test_audit_clean_task(self, tmp_project):
|
||||
task_dir = _create_task(tmp_project, "clean-audit", "research")
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
out, code = _run_status(["--audit"], tmp_project)
|
||||
assert "PASS" in out
|
||||
+264
-1
@@ -8,6 +8,9 @@ from automaton.dashboard.core.task import (
|
||||
determine_task_state,
|
||||
discover_tasks,
|
||||
parse_sub_tasks,
|
||||
parse_waves,
|
||||
parse_vram_config,
|
||||
WaveGroup,
|
||||
TaskState,
|
||||
)
|
||||
|
||||
@@ -39,7 +42,17 @@ def test_implementation_state(tmp_path: Path) -> None:
|
||||
task_dir = _make_task(
|
||||
tmp_path,
|
||||
"impl-task",
|
||||
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl"},
|
||||
{"SPEC.md": "# Spec", "TEST_PLAN.md": "# Tests", "IMPLEMENTATION.md": "# Impl"},
|
||||
)
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.BUG_FIND
|
||||
|
||||
|
||||
def test_implementation_from_test_plan(tmp_path: Path) -> None:
|
||||
task_dir = _make_task(
|
||||
tmp_path,
|
||||
"impl-test-plan",
|
||||
{"SPEC.md": "# Spec", "TEST_PLAN.md": "# Tests"},
|
||||
)
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.IMPLEMENT
|
||||
@@ -132,3 +145,253 @@ def test_discover_tasks_skips_subtasks_root(tmp_path: Path) -> None:
|
||||
assert len(tasks) == 1
|
||||
assert tasks[0].name == "parent"
|
||||
assert len(tasks[0].sub_tasks) == 1
|
||||
|
||||
|
||||
class TestVerdictParsing:
|
||||
"""Tests for structured verdict status parsing (R1-R3 of fix-verdict-parsing SPEC)."""
|
||||
|
||||
def test_pass_verdict_with_fail_in_findings(self, tmp_path: Path) -> None:
|
||||
ver = "## Status: PASS\n\nThe previous FAIL finding was resolved."
|
||||
task_dir = _make_task(tmp_path, "pass-with-fail", {"SPEC.md": "# Spec", "VERDICT.md": ver})
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.DONE
|
||||
|
||||
def test_pass_verdict_with_needs_review_in_body(self, tmp_path: Path) -> None:
|
||||
ver = "## Status: PASS\n\nNote: NEEDS_REVIEW was discussed but resolved."
|
||||
task_dir = _make_task(tmp_path, "pass-with-nr", {"SPEC.md": "# Spec", "VERDICT.md": ver})
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.DONE
|
||||
|
||||
def test_fail_verdict_structured(self, tmp_path: Path) -> None:
|
||||
ver = "## Status: FAIL\n\n2 tests PASS, 1 test FAIL."
|
||||
task_dir = _make_task(tmp_path, "fail-mentions-pass", {"SPEC.md": "# Spec", "VERDICT.md": ver})
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.BLOCKED
|
||||
|
||||
def test_needs_review_verdict_structured(self, tmp_path: Path) -> None:
|
||||
ver = "## Status: NEEDS_REVIEW\n\nSome items PASS but need review."
|
||||
task_dir = _make_task(tmp_path, "nr-mentions-pass", {"SPEC.md": "# Spec", "VERDICT.md": ver})
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.BLOCKED
|
||||
|
||||
def test_verdict_with_bold_status(self, tmp_path: Path) -> None:
|
||||
ver = "# Verdict\n\n- **Status**: PASS\n- **Timestamp**: 2025-01-01"
|
||||
task_dir = _make_task(tmp_path, "bold-status", {"SPEC.md": "# Spec", "VERDICT.md": ver})
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.DONE
|
||||
|
||||
def test_verdict_no_status_line(self, tmp_path: Path) -> None:
|
||||
ver = "# Verdict\nEverything looks good, PASS!"
|
||||
task_dir = _make_task(tmp_path, "no-status-line", {"SPEC.md": "# Spec", "VERDICT.md": ver})
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.DONE
|
||||
|
||||
def test_verdict_no_status_no_keywords(self, tmp_path: Path) -> None:
|
||||
ver = "# Verdict\n\nNeeds further discussion."
|
||||
task_dir = _make_task(tmp_path, "no-status-no-keywords", {"SPEC.md": "# Spec", "VERDICT.md": ver})
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.REFEREE
|
||||
|
||||
|
||||
class TestStateMachineAlignment:
|
||||
"""Tests for state machine alignment with orchestrate.md (R4)."""
|
||||
|
||||
def test_implementation_alone_shows_bug_find(self, tmp_path: Path) -> None:
|
||||
task_dir = _make_task(tmp_path, "impl-only", {"IMPLEMENTATION.md": "# Impl"})
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.BUG_FIND
|
||||
|
||||
def test_bug_report_without_adversarial(self, tmp_path: Path) -> None:
|
||||
task_dir = _make_task(
|
||||
tmp_path, "bug-only",
|
||||
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl", "BUG_REPORT.md": "# Bugs"},
|
||||
)
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.BUG_FIND
|
||||
|
||||
def test_both_bug_reports_shows_adv_bug_find(self, tmp_path: Path) -> None:
|
||||
task_dir = _make_task(
|
||||
tmp_path, "both-bugs",
|
||||
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl",
|
||||
"BUG_REPORT.md": "# Bugs", "ADVERSARIAL_BUG_REPORT.md": "# Adv"},
|
||||
)
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.ADV_BUG_FIND
|
||||
|
||||
def test_adv_bug_report_alone_shows_bug_find(self, tmp_path: Path) -> None:
|
||||
task_dir = _make_task(
|
||||
tmp_path, "adv-only",
|
||||
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl",
|
||||
"ADVERSARIAL_BUG_REPORT.md": "# Adv bugs only"},
|
||||
)
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.BUG_FIND
|
||||
|
||||
def test_spec_alone_shows_research(self, tmp_path: Path) -> None:
|
||||
task_dir = _make_task(tmp_path, "spec-only", {"SPEC.md": "# Spec"})
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.RESEARCH
|
||||
|
||||
def test_test_plan_shows_implement(self, tmp_path: Path) -> None:
|
||||
task_dir = _make_task(tmp_path, "testplan", {"SPEC.md": "# Spec", "TEST_PLAN.md": "# Tests"})
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.IMPLEMENT
|
||||
|
||||
def test_design_with_spec_shows_design(self, tmp_path: Path) -> None:
|
||||
task_dir = _make_task(tmp_path, "design-spec", {"SPEC.md": "# Spec", "DESIGN.md": "# Design"})
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.DESIGN
|
||||
|
||||
def test_doc_review_shows_doc_review(self, tmp_path: Path) -> None:
|
||||
task_dir = _make_task(
|
||||
tmp_path, "doc-review",
|
||||
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl",
|
||||
"BUG_REPORT.md": "# Bugs", "ADVERSARIAL_BUG_REPORT.md": "# Adv",
|
||||
"DOC_REVIEW.md": "# Docs"},
|
||||
)
|
||||
state, _ = determine_task_state(task_dir)
|
||||
assert state == TaskState.DOC_REVIEW
|
||||
|
||||
|
||||
class TestTaskNameValidation:
|
||||
"""Tests for filesystem-sourced task name validation."""
|
||||
|
||||
def test_valid_task_names(self, tmp_path: Path) -> None:
|
||||
for name in ["my-task", "task_1", "Task-Name-123"]:
|
||||
d = tmp_path / name
|
||||
d.mkdir()
|
||||
(d / "SPEC.md").write_text("# Spec")
|
||||
tasks = discover_tasks(tmp_path)
|
||||
assert len(tasks) == 3
|
||||
|
||||
def test_invalid_task_name_skipped(self, tmp_path: Path) -> None:
|
||||
valid = tmp_path / "good-task"
|
||||
valid.mkdir()
|
||||
(valid / "SPEC.md").write_text("# Spec")
|
||||
bad = tmp_path / "task with spaces"
|
||||
bad.mkdir()
|
||||
(bad / "SPEC.md").write_text("# Bad")
|
||||
tasks = discover_tasks(tmp_path)
|
||||
assert len(tasks) == 1
|
||||
assert tasks[0].name == "good-task"
|
||||
|
||||
|
||||
class TestParseWaves:
|
||||
def test_parse_waves_from_template(self) -> None:
|
||||
content = "# Decomposition: Test\n\n## Sub-Tasks\n\n### Wave 1 (Parallel)\n- subtask-a: Implement core\n- subtask-b: Implement model\n\n### Wave 2 (Dependent)\n- subtask-c: Implement UI\n- subtask-d: Implement tests\n"
|
||||
waves = parse_waves(content)
|
||||
assert len(waves) == 2
|
||||
assert waves[0].wave_number == 1
|
||||
assert waves[0].label == "Parallel"
|
||||
assert waves[0].sub_task_names == ["subtask-a", "subtask-b"]
|
||||
assert waves[1].wave_number == 2
|
||||
assert waves[1].label == "Dependent"
|
||||
assert waves[1].sub_task_names == ["subtask-c", "subtask-d"]
|
||||
|
||||
def test_parse_waves_empty(self) -> None:
|
||||
assert parse_waves("") == []
|
||||
assert parse_waves(None) == []
|
||||
|
||||
def test_parse_waves_no_waves(self) -> None:
|
||||
content = "# No waves here\nJust text\n"
|
||||
assert parse_waves(content) == []
|
||||
|
||||
def test_parse_waves_colon_format(self) -> None:
|
||||
content = "### Wave 1: Setup\n- task-alpha: Do setup\n### Wave 2: Execution\n- task-beta: Do execution\n"
|
||||
waves = parse_waves(content)
|
||||
assert len(waves) == 2
|
||||
assert waves[0].label == "Setup"
|
||||
assert waves[0].sub_task_names == ["task-alpha"]
|
||||
assert waves[1].label == "Execution"
|
||||
assert waves[1].sub_task_names == ["task-beta"]
|
||||
|
||||
|
||||
class TestDecompositionContent:
|
||||
def test_task_has_decomposition_content(self, tmp_path: Path) -> None:
|
||||
task_dir = tmp_path / "my-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
(task_dir / "DECOMPOSITION.md").write_text("### Wave 1 (Build)\n- sub-a: Build core\n")
|
||||
tasks = discover_tasks(tmp_path)
|
||||
assert len(tasks) == 1
|
||||
assert tasks[0].decomposition_content is not None
|
||||
assert "Wave 1" in tasks[0].decomposition_content
|
||||
assert len(tasks[0].waves) == 1
|
||||
assert tasks[0].waves[0].sub_task_names == ["sub-a"]
|
||||
|
||||
|
||||
class TestParentSpecAndVramConfig:
|
||||
def test_parent_spec_content(self, tmp_path: Path) -> None:
|
||||
task_dir = tmp_path / "my-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
(task_dir / "PARENT_SPEC.md").write_text("# Parent Context\nDetails here")
|
||||
tasks = discover_tasks(tmp_path)
|
||||
assert tasks[0].parent_spec_content is not None
|
||||
assert "Parent Context" in tasks[0].parent_spec_content
|
||||
|
||||
def test_vram_config_content(self, tmp_path: Path) -> None:
|
||||
task_dir = tmp_path / "my-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "SPEC.md").write_text("# Spec")
|
||||
(task_dir / "VRAM_CONFIG.md").write_text("# VRAM\nmodel: llama-3")
|
||||
tasks = discover_tasks(tmp_path)
|
||||
assert tasks[0].vram_config_content is not None
|
||||
assert "llama-3" in tasks[0].vram_config_content
|
||||
|
||||
def test_subtask_has_parent_spec(self, tmp_path: Path) -> None:
|
||||
tasks_dir = tmp_path
|
||||
parent = tasks_dir / "parent-task"
|
||||
parent.mkdir()
|
||||
(parent / "SPEC.md").write_text("# Parent")
|
||||
subtasks_dir = parent / "subtasks"
|
||||
subtasks_dir.mkdir()
|
||||
sub = subtasks_dir / "child-a"
|
||||
sub.mkdir()
|
||||
(sub / "SPEC.md").write_text("# Child")
|
||||
(sub / "PARENT_SPEC.md").write_text("# Parent Spec for child")
|
||||
(sub / "VRAM_CONFIG.md").write_text("# VRAM config")
|
||||
sub_tasks = parse_sub_tasks(parent)
|
||||
assert len(sub_tasks) == 1
|
||||
# discover_tasks doesn't recurse into subtasks for content, but parse_sub_tasks returns SubTask objects
|
||||
# Verify the files exist
|
||||
assert (sub / "PARENT_SPEC.md").exists()
|
||||
|
||||
|
||||
class TestStatusReason:
|
||||
def test_done_task_reason(self, tmp_path: Path) -> None:
|
||||
task_dir = tmp_path / "done-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "VERDICT.md").write_text("## Status: PASS\nAll good.\n")
|
||||
(task_dir / "SPEC.md").write_text("# Spec\n")
|
||||
tasks = discover_tasks(tmp_path)
|
||||
assert tasks[0].status_reason == "Verdict: PASS"
|
||||
|
||||
def test_blocked_fail_reason(self, tmp_path: Path) -> None:
|
||||
task_dir = tmp_path / "blocked-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "VERDICT.md").write_text("## Status: FAIL\nBroken.\n")
|
||||
(task_dir / "SPEC.md").write_text("# Spec\n")
|
||||
tasks = discover_tasks(tmp_path)
|
||||
assert "FAIL" in tasks[0].status_reason
|
||||
|
||||
def test_blocked_needs_review_reason(self, tmp_path: Path) -> None:
|
||||
task_dir = tmp_path / "review-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "VERDICT.md").write_text("## Status: NEEDS_REVIEW\nUnclear.\n")
|
||||
(task_dir / "SPEC.md").write_text("# Spec\n")
|
||||
tasks = discover_tasks(tmp_path)
|
||||
assert "NEEDS_REVIEW" in tasks[0].status_reason
|
||||
|
||||
def test_bug_find_implementation_reason(self, tmp_path: Path) -> None:
|
||||
task_dir = tmp_path / "impl-task"
|
||||
task_dir.mkdir()
|
||||
(task_dir / "IMPLEMENTATION.md").write_text("# Implementation\n")
|
||||
tasks = discover_tasks(tmp_path)
|
||||
assert "Implementation complete" in tasks[0].status_reason
|
||||
|
||||
def test_backlog_reason(self, tmp_path: Path) -> None:
|
||||
task_dir = tmp_path / "empty-task"
|
||||
task_dir.mkdir()
|
||||
tasks = discover_tasks(tmp_path)
|
||||
assert "No artifacts" in tasks[0].status_reason
|
||||
|
||||
Reference in New Issue
Block a user