- **Restore 82 completed tasks** from tasks/complete/ back to tasks/ top level (all <7 days old per the cleanup policy; premature bulk archive was fixed). - **Dashboard: fix scroll-reset on auto-refresh** — renderBoard rebuilds the board via innerHTML every 2s, destroying each column-body's scrollTop. Now snapshots column-body scrollTop + board.scrollLeft + view.scrollTop before rebuild and restores after (matched by PHASE_GROUPS index). - **Dashboard UI additions** (pre-existing unstaged work): approval section cards, transition buttons, inline artifact editor (textarea for writing missing SPEC/VERDICT/etc from the detail modal). - **Bind ornith as Implement model** — config.md: Model explicit to omlx/Ornith-1.0-35B-4bit-mlx, context window 32768. Interactive autopilot already used ornith via opencode default; now explicit. - **Fix cleanup stub** — automaton-cleanup.sh had a stale --project arg pointing at a pytest temp dir (test isolation leak). Rewired to point at ~/.automaton. - **Fix plist-isolation test** — test asserted host plist doesn't exist, but a real install creates it. Now snapshots mtime before run, asserts unchanged after (only a write during the test counts as bleed). - **New Playwright smoke test** (tests/test_dashboard_ui.py) — 2 tests: board renders tasks, column scroll survives auto-refresh tick. Verified the test fails without the scroll fix (scrollTop resets to 0). Skipped via importorskip when playwright is absent (main CI stays green). - **Clarify SI loop scope in README** — new-project onboarding section documents the framework-scoped self-improvement loop and options (leave/pause/create project loop). - **CHANGELOG** documents all changes including the known model-divergence gap (mde tasks marked complete but per-role model binding was never implemented).
254 lines
10 KiB
Python
254 lines
10 KiB
Python
"""Framework self-consistency tests.
|
|
|
|
These tests enforce structural rules on the framework itself — prompt
|
|
consistency, canonical paths, state machine integrity, and configuration
|
|
validity. They catch regressions that unit tests alone cannot.
|
|
"""
|
|
|
|
import re
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
ROOT = Path(__file__).resolve().parent.parent
|
|
PROMPTS_DIR = ROOT / "prompts"
|
|
RULES_FILE = ROOT / ".rules.md"
|
|
PYPROJECT_FILE = ROOT / "pyproject.toml"
|
|
CI_FILE = ROOT / ".gitea" / "workflows" / "ci.yml"
|
|
CSS_FILE = ROOT / "automaton" / "dashboard" / "html" / "styles.css"
|
|
JS_FILE = ROOT / "automaton" / "dashboard" / "html" / "dashboard.js"
|
|
|
|
|
|
class TestDeliveryPromptsHaveStopConditions:
|
|
"""R1.A: Every delivery prompt must contain a stop condition block."""
|
|
|
|
EXCLUDED = {"orchestrate.md", "compaction.md", "workflow.md", "subtask_management.md", "onboarding.md", "loop-implement.md", "loop-verifier.md", "loop-orchestrate.md"}
|
|
|
|
@pytest.fixture()
|
|
def delivery_prompts(self):
|
|
if not PROMPTS_DIR.exists():
|
|
pytest.skip("prompts/ directory not found")
|
|
return [
|
|
f
|
|
for f in sorted(PROMPTS_DIR.iterdir())
|
|
if f.is_file() and f.suffix == ".md" and f.name not in self.EXCLUDED
|
|
]
|
|
|
|
def test_each_prompt_has_stop_condition(self, delivery_prompts):
|
|
for prompt in delivery_prompts:
|
|
content = prompt.read_text()
|
|
has_stop = "## Stop Condition" in content or "STOP CONDITION" in content or "CONTRACT_MET" in content
|
|
assert has_stop, f"{prompt.name} is missing a stop condition block"
|
|
|
|
|
|
class TestNoHardcodedURLs:
|
|
"""R1.B: Prompts, contracts, and templates must not contain hardcoded repo URLs."""
|
|
|
|
ALLOWED_FILES = {"install.sh"}
|
|
|
|
def _check_dir(self, directory: Path, pattern: re.Pattern):
|
|
violations = []
|
|
if not directory.exists():
|
|
return violations
|
|
for f in directory.rglob("*.md"):
|
|
if f.name in self.ALLOWED_FILES:
|
|
continue
|
|
content = f.read_text()
|
|
for line_no, line in enumerate(content.splitlines(), 1):
|
|
if pattern.search(line):
|
|
violations.append(f"{f.relative_to(ROOT)}:{line_no}")
|
|
return violations
|
|
|
|
def test_no_hardcoded_ip_urls(self):
|
|
pattern = re.compile(r"\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}[:/]")
|
|
dirs = [PROMPTS_DIR, ROOT / "contracts", ROOT / "templates"]
|
|
violations = []
|
|
for d in dirs:
|
|
violations.extend(self._check_dir(d, pattern))
|
|
assert not violations, f"Hardcoded IP URLs found in: {violations}"
|
|
|
|
def test_no_hardcoded_localhost_ports(self):
|
|
pattern = re.compile(r"localhost:\d{4,5}")
|
|
dirs = [PROMPTS_DIR, ROOT / "contracts", ROOT / "templates"]
|
|
violations = []
|
|
for d in dirs:
|
|
violations.extend(self._check_dir(d, pattern))
|
|
assert not violations, f"Hardcoded localhost URLs found in: {violations}"
|
|
|
|
|
|
class TestRulesMdSections:
|
|
"""R1.C & R1.D: .rules.md must contain mandatory sections."""
|
|
|
|
MANDATORY_SECTIONS = [
|
|
"Task-Driven Development",
|
|
"VRAM",
|
|
"Changelog",
|
|
"Session Discipline",
|
|
"Scope Confinement",
|
|
"Artifact Integrity",
|
|
]
|
|
|
|
def test_mandatory_sections_present(self):
|
|
if not RULES_FILE.exists():
|
|
pytest.skip(".rules.md not found")
|
|
content = RULES_FILE.read_text()
|
|
for section in self.MANDATORY_SECTIONS:
|
|
assert section in content, f".rules.md missing mandatory section: {section}"
|
|
|
|
def test_self_improvement_has_example(self):
|
|
if not RULES_FILE.exists():
|
|
pytest.skip(".rules.md not found")
|
|
content = RULES_FILE.read_text()
|
|
assert "Past failure" in content, ".rules.md Self-Improvement section must reference at least one real failure mode"
|
|
|
|
|
|
class TestCanonicalTaskPaths:
|
|
"""R1.E: All task path references in prompts must use the canonical format."""
|
|
|
|
CANONICAL_PATTERN = re.compile(r"\{project\}/\.automaton/tasks/\w")
|
|
DEPRECATED_PATTERN = re.compile(r"\{project\}/tasks/[\w-]+/")
|
|
|
|
def test_no_deprecated_task_paths(self):
|
|
if not PROMPTS_DIR.exists():
|
|
pytest.skip("prompts/ directory not found")
|
|
violations = []
|
|
for f in sorted(PROMPTS_DIR.iterdir()):
|
|
if not f.is_file() or f.suffix != ".md":
|
|
continue
|
|
content = f.read_text()
|
|
for line_no, line in enumerate(content.splitlines(), 1):
|
|
if self.DEPRECATED_PATTERN.search(line):
|
|
violations.append(f"{f.name}:{line_no}: {line.strip()}")
|
|
assert not violations, f"Deprecated task paths found: {violations}"
|
|
|
|
|
|
class TestPyprojectNoStaleExtras:
|
|
"""R1.F: pyproject.toml must not reference inotify."""
|
|
|
|
def test_no_inotify_dependency(self):
|
|
if not PYPROJECT_FILE.exists():
|
|
pytest.skip("pyproject.toml not found")
|
|
content = PYPROJECT_FILE.read_text()
|
|
assert "inotify" not in content, "pyproject.toml still references inotify"
|
|
|
|
|
|
class TestDashboardCSSThemes:
|
|
"""R1.G: Light and dark themes must define the same variable set."""
|
|
|
|
def _extract_vars(self, section: str) -> set:
|
|
pattern = re.compile(r"--([\w-]+)\s*:", re.MULTILINE)
|
|
return set(pattern.findall(section))
|
|
|
|
def test_theme_variable_parity(self):
|
|
if not CSS_FILE.exists():
|
|
pytest.skip("styles.css not found")
|
|
content = CSS_FILE.read_text()
|
|
root_match = re.search(r":root\s*\{([^}]+)\}", content, re.DOTALL)
|
|
light_match = re.search(r'\[data-theme="light"\]\s*\{([^}]+)\}', content, re.DOTALL)
|
|
dark_match = re.search(r'\[data-theme="dark"\]\s*\{([^}]+)\}', content, re.DOTALL)
|
|
if not root_match:
|
|
pytest.skip(":root CSS variables not found")
|
|
root_vars = self._extract_vars(root_match.group(1))
|
|
theme_vars = set()
|
|
if light_match:
|
|
theme_vars |= self._extract_vars(light_match.group(1))
|
|
if dark_match:
|
|
theme_vars |= self._extract_vars(dark_match.group(1))
|
|
if not theme_vars:
|
|
pytest.skip("No theme sections found in CSS")
|
|
structural_vars = {"radius-sm", "radius-md", "radius-lg"}
|
|
themable_root_vars = root_vars - structural_vars
|
|
missing_from_themes = themable_root_vars - theme_vars
|
|
assert not missing_from_themes, f"CSS variables defined in :root but missing from theme overrides: {missing_from_themes}"
|
|
|
|
|
|
class TestVerdictParsingRegression:
|
|
"""R3: Verdict parsing regression tests for the critical false-BLOCKED bug."""
|
|
|
|
def test_pass_verdict_mentioning_fail_is_done(self):
|
|
from automaton.dashboard.core.task import determine_task_state, TaskState
|
|
from pathlib import Path
|
|
import tempfile
|
|
with tempfile.TemporaryDirectory() as tmp:
|
|
task_dir = Path(tmp) / "test-task"
|
|
task_dir.mkdir()
|
|
(task_dir / "VERDICT.md").write_text("## Status: PASS\n\nThe bug in the FAIL case is now fixed.\n")
|
|
state, _, _ = determine_task_state(task_dir)
|
|
assert state == TaskState.DONE, f"Expected DONE, got {state}"
|
|
|
|
def test_pass_verdict_with_needs_review_mention_is_done(self):
|
|
from automaton.dashboard.core.task import determine_task_state, TaskState
|
|
from pathlib import Path
|
|
import tempfile
|
|
with tempfile.TemporaryDirectory() as tmp:
|
|
task_dir = Path(tmp) / "test-task"
|
|
task_dir.mkdir()
|
|
(task_dir / "VERDICT.md").write_text("## Status: PASS\n\nPreviously flagged as NEEDS_REVIEW but resolved.\n")
|
|
state, _, _ = determine_task_state(task_dir)
|
|
assert state == TaskState.DONE, f"Expected DONE, got {state}"
|
|
|
|
def test_structured_fail_verdict_is_blocked(self):
|
|
from automaton.dashboard.core.task import determine_task_state, TaskState
|
|
from pathlib import Path
|
|
import tempfile
|
|
with tempfile.TemporaryDirectory() as tmp:
|
|
task_dir = Path(tmp) / "test-task"
|
|
task_dir.mkdir()
|
|
(task_dir / "VERDICT.md").write_text("## Status: FAIL\n\nThe implementation has a critical bug.\n")
|
|
state, _, _ = determine_task_state(task_dir)
|
|
assert state == TaskState.BLOCKED
|
|
|
|
def test_implementation_alone_is_implement(self):
|
|
from automaton.dashboard.core.task import determine_task_state, TaskState
|
|
from pathlib import Path
|
|
import tempfile
|
|
with tempfile.TemporaryDirectory() as tmp:
|
|
task_dir = Path(tmp) / "test-task"
|
|
task_dir.mkdir()
|
|
(task_dir / "IMPLEMENTATION.md").write_text("# Implementation\nDone.\n")
|
|
state, _, _ = determine_task_state(task_dir)
|
|
assert state == TaskState.IMPLEMENT
|
|
|
|
def test_empty_verdict_is_blocked(self):
|
|
from automaton.dashboard.core.task import determine_task_state, TaskState
|
|
from pathlib import Path
|
|
import tempfile
|
|
with tempfile.TemporaryDirectory() as tmp:
|
|
task_dir = Path(tmp) / "test-task"
|
|
task_dir.mkdir()
|
|
(task_dir / "VERDICT.md").write_text("")
|
|
state, _, _ = determine_task_state(task_dir)
|
|
assert state == TaskState.BLOCKED
|
|
|
|
def test_adv_bug_report_alone_is_bug_find(self):
|
|
from automaton.dashboard.core.task import determine_task_state, TaskState
|
|
from pathlib import Path
|
|
import tempfile
|
|
with tempfile.TemporaryDirectory() as tmp:
|
|
task_dir = Path(tmp) / "test-task"
|
|
task_dir.mkdir()
|
|
(task_dir / "ADVERSARIAL_BUG_REPORT.md").write_text("# Bug Report\nFound issue.\n")
|
|
state, _, _ = determine_task_state(task_dir)
|
|
assert state == TaskState.BUG_FIND
|
|
|
|
|
|
class TestCIWorkflowValidation:
|
|
"""R4: CI workflow must compile, test, and check shell scripts."""
|
|
|
|
def test_ci_runs_py_compile(self):
|
|
if not CI_FILE.exists():
|
|
pytest.skip("CI workflow not found")
|
|
content = CI_FILE.read_text()
|
|
assert "py_compile" in content, "CI must run py_compile"
|
|
|
|
def test_ci_runs_pytest(self):
|
|
if not CI_FILE.exists():
|
|
pytest.skip("CI workflow not found")
|
|
content = CI_FILE.read_text()
|
|
assert "pytest" in content, "CI must run pytest"
|
|
|
|
def test_ci_checks_shell_scripts(self):
|
|
if not CI_FILE.exists():
|
|
pytest.skip("CI workflow not found")
|
|
content = CI_FILE.read_text()
|
|
assert "bash -n" in content, "CI must syntax-check shell scripts" |