Restore archived tasks, fix dashboard scroll-reset, bind ornith, add Playwright smoke test

- **Restore 82 completed tasks** from tasks/complete/ back to tasks/ top
  level (all <7 days old per the cleanup policy; premature bulk archive
  was fixed).
- **Dashboard: fix scroll-reset on auto-refresh** — renderBoard rebuilds
  the board via innerHTML every 2s, destroying each column-body's
  scrollTop. Now snapshots column-body scrollTop + board.scrollLeft +
  view.scrollTop before rebuild and restores after (matched by
  PHASE_GROUPS index).
- **Dashboard UI additions** (pre-existing unstaged work): approval
  section cards, transition buttons, inline artifact editor (textarea for
  writing missing SPEC/VERDICT/etc from the detail modal).
- **Bind ornith as Implement model** — config.md: Model explicit to
  omlx/Ornith-1.0-35B-4bit-mlx, context window 32768. Interactive
  autopilot already used ornith via opencode default; now explicit.
- **Fix cleanup stub** — automaton-cleanup.sh had a stale --project arg
  pointing at a pytest temp dir (test isolation leak). Rewired to point
  at ~/.automaton.
- **Fix plist-isolation test** — test asserted host plist doesn't exist,
  but a real install creates it. Now snapshots mtime before run, asserts
  unchanged after (only a write during the test counts as bleed).
- **New Playwright smoke test** (tests/test_dashboard_ui.py) — 2 tests:
  board renders tasks, column scroll survives auto-refresh tick.
  Verified the test fails without the scroll fix (scrollTop resets to 0).
  Skipped via importorskip when playwright is absent (main CI stays
  green).
- **Clarify SI loop scope in README** — new-project onboarding section
  documents the framework-scoped self-improvement loop and options
  (leave/pause/create project loop).
- **CHANGELOG** documents all changes including the known model-divergence
  gap (mde tasks marked complete but per-role model binding was never
  implemented).
This commit is contained in:
Lap Tran
2026-06-26 10:05:18 -04:00
parent fe43b9e1fc
commit bc7daf8590
666 changed files with 15994 additions and 69 deletions
+1 -1
View File
@@ -15,7 +15,7 @@ def _task(name: str, tmp_path: Path, artifacts: dict[str, str]) -> object:
task_dir.mkdir(parents=True)
for filename, content in artifacts.items():
(task_dir / filename).write_text(content)
state, artifact_map = determine_task_state(task_dir)
state, _, artifact_map = determine_task_state(task_dir)
return Task(name=name, folder_path=task_dir, state=state, artifacts=artifact_map)
+6 -3
View File
@@ -134,6 +134,10 @@ class TestInstallCleanupScheduleIsolation:
def test_plist_written_to_override_dir_not_host(self, tmp_path):
agents = tmp_path / "agents"
agents.mkdir()
host = Path.home() / "Library" / "LaunchAgents" / "com.automaton.cleanup.plist"
# Snapshot host plist state before the run. A pre-existing real install
# is OK — only a write during this test would be a bleed.
host_before = host.stat().st_mtime_ns if host.exists() else None
res = self._run_cli(
["--install-cleanup-schedule", "--days", "5", "--interval", "120",
"--project", str(tmp_path)],
@@ -142,9 +146,8 @@ class TestInstallCleanupScheduleIsolation:
assert res.returncode == 0, res.stdout + res.stderr
plist = agents / "com.automaton.cleanup.plist"
assert plist.exists()
# Nothing bled to the real host LaunchAgents.
host = Path.home() / "Library" / "LaunchAgents" / "com.automaton.cleanup.plist"
assert not host.exists()
host_after = host.stat().st_mtime_ns if host.exists() else None
assert host_before == host_after, "host plist was modified by the test run"
content = plist.read_text()
assert "com.automaton.cleanup" in content
assert "<integer>120</integer>" in content
+115
View File
@@ -0,0 +1,115 @@
"""Playwright browser smoke test for the dashboard UI.
Verifies the dashboard boots, renders tasks, and preserves column scroll
position across auto-refresh (the fix for the scroll-reset bug). Skipped
when playwright/chromium is not installed so the main suite stays green
in CI environments without a browser.
Run standalone:
python3 -m pytest tests/test_dashboard_ui.py -v
"""
from __future__ import annotations
import socket
import threading
import time
from http.server import HTTPServer
from pathlib import Path
import pytest
from automaton.dashboard.ui.app import DashboardHandler, DashboardApp
playwright = pytest.importorskip("playwright")
from playwright.sync_api import sync_playwright # noqa: E402
def _free_port() -> int:
s = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
s.bind(("127.0.0.1", 0))
port = s.getsockname()[1]
s.close()
return port
def _make_done_tasks(project_root: Path, count: int) -> None:
tasks = project_root / ".automaton" / "tasks"
tasks.mkdir(parents=True, exist_ok=True)
for i in range(count):
d = tasks / f"done-task-{i:02d}"
d.mkdir(exist_ok=True)
(d / "SPEC.md").write_text(f"# Done task {i}\n" + ("line. " * 40) + "\n")
(d / ".state").write_text("complete\n")
@pytest.fixture
def dashboard_server(tmp_path: Path):
_make_done_tasks(tmp_path, 40)
app = DashboardApp(start_path=tmp_path)
assert app.initialize(), "dashboard failed to init"
port = _free_port()
DashboardHandler.config = app.config
DashboardHandler.project_root = app.project_root
DashboardHandler.scope = app.scope
server = HTTPServer(("127.0.0.1", port), DashboardHandler)
thread = threading.Thread(target=server.serve_forever, daemon=True)
thread.start()
base_url = f"http://127.0.0.1:{port}"
try:
yield base_url
finally:
server.shutdown()
server.server_close()
thread.join(timeout=5)
def _wait_for(predicate, timeout=10.0, interval=0.2):
deadline = time.time() + timeout
while time.time() < deadline:
if predicate():
return True
time.sleep(interval)
return False
class TestDashboardUI:
def test_board_renders_done_tasks(self, dashboard_server):
with sync_playwright() as p:
browser = p.chromium.launch()
page = browser.new_page()
page.goto(dashboard_server)
# Footer refresh counter increments once auto-refresh fires.
assert _wait_for(lambda: page.text_content("#refresh-count") not in (None, "0", ""))
# The Done (resolution) group contains our tasks.
cards = page.locator(".task-card")
assert _wait_for(lambda: cards.count() >= 40, timeout=15)
assert page.text_content("#done-tasks") == "40"
browser.close()
def test_refresh_preserves_column_scroll(self, dashboard_server):
"""The scroll-reset bug: re-render snapped a column's scrollTop to 0,
so users couldn't scroll the Done group down to review older tasks.
Verify the column-body scrollTop survives an auto-refresh tick."""
with sync_playwright() as p:
browser = p.chromium.launch()
# Wide viewport so columns fit and the Done column-body overflows
# vertically (rather than the board squeezing them horizontally).
page = browser.new_page(viewport={"width": 1400, "height": 500})
page.goto(dashboard_server)
assert _wait_for(lambda: page.locator(".task-card").count() >= 40, timeout=15)
# The Done (resolution) group is the last column.
bodies = page.locator(".column-body")
n = bodies.count()
done_body = bodies.nth(n - 1)
assert done_body.evaluate("el => el.scrollHeight > el.clientHeight")
# Scroll the column-body well past the top.
target = 400
done_body.evaluate("(el, y) => { el.scrollTop = y; }", target)
assert done_body.evaluate("el => el.scrollTop") == target
# Wait for an auto-refresh re-render (default interval 2s).
before = int(page.text_content("#refresh-count") or "0")
assert _wait_for(lambda: int(page.text_content("#refresh-count") or "0") > before, timeout=10)
# Scroll must survive the re-render (the bug fix).
assert done_body.evaluate("el => el.scrollTop") == target
browser.close()
+6 -6
View File
@@ -173,7 +173,7 @@ class TestVerdictParsingRegression:
task_dir = Path(tmp) / "test-task"
task_dir.mkdir()
(task_dir / "VERDICT.md").write_text("## Status: PASS\n\nThe bug in the FAIL case is now fixed.\n")
state, _ = determine_task_state(task_dir)
state, _, _ = determine_task_state(task_dir)
assert state == TaskState.DONE, f"Expected DONE, got {state}"
def test_pass_verdict_with_needs_review_mention_is_done(self):
@@ -184,7 +184,7 @@ class TestVerdictParsingRegression:
task_dir = Path(tmp) / "test-task"
task_dir.mkdir()
(task_dir / "VERDICT.md").write_text("## Status: PASS\n\nPreviously flagged as NEEDS_REVIEW but resolved.\n")
state, _ = determine_task_state(task_dir)
state, _, _ = determine_task_state(task_dir)
assert state == TaskState.DONE, f"Expected DONE, got {state}"
def test_structured_fail_verdict_is_blocked(self):
@@ -195,7 +195,7 @@ class TestVerdictParsingRegression:
task_dir = Path(tmp) / "test-task"
task_dir.mkdir()
(task_dir / "VERDICT.md").write_text("## Status: FAIL\n\nThe implementation has a critical bug.\n")
state, _ = determine_task_state(task_dir)
state, _, _ = determine_task_state(task_dir)
assert state == TaskState.BLOCKED
def test_implementation_alone_is_implement(self):
@@ -206,7 +206,7 @@ class TestVerdictParsingRegression:
task_dir = Path(tmp) / "test-task"
task_dir.mkdir()
(task_dir / "IMPLEMENTATION.md").write_text("# Implementation\nDone.\n")
state, _ = determine_task_state(task_dir)
state, _, _ = determine_task_state(task_dir)
assert state == TaskState.IMPLEMENT
def test_empty_verdict_is_blocked(self):
@@ -217,7 +217,7 @@ class TestVerdictParsingRegression:
task_dir = Path(tmp) / "test-task"
task_dir.mkdir()
(task_dir / "VERDICT.md").write_text("")
state, _ = determine_task_state(task_dir)
state, _, _ = determine_task_state(task_dir)
assert state == TaskState.BLOCKED
def test_adv_bug_report_alone_is_bug_find(self):
@@ -228,7 +228,7 @@ class TestVerdictParsingRegression:
task_dir = Path(tmp) / "test-task"
task_dir.mkdir()
(task_dir / "ADVERSARIAL_BUG_REPORT.md").write_text("# Bug Report\nFound issue.\n")
state, _ = determine_task_state(task_dir)
state, _, _ = determine_task_state(task_dir)
assert state == TaskState.BUG_FIND
+1 -1
View File
@@ -15,7 +15,7 @@ def _task(name: str, tmp_path: Path, artifacts: dict[str, str]) -> object:
task_dir.mkdir(parents=True)
for filename, content in artifacts.items():
(task_dir / filename).write_text(content)
state, artifact_map = determine_task_state(task_dir)
state, _, artifact_map = determine_task_state(task_dir)
return Task(name=name, folder_path=task_dir, state=state, artifacts=artifact_map)
+24 -24
View File
@@ -26,14 +26,14 @@ def _make_task(tmp_path: Path, name: str, artifacts: dict[str, str]) -> Path:
def test_backlog_state(tmp_path: Path) -> None:
task_dir = _make_task(tmp_path, "backlog-task", {})
state, artifacts = determine_task_state(task_dir)
state, phase_raw, artifacts = determine_task_state(task_dir)
assert state == TaskState.BACKLOG
assert not artifacts
def test_research_state(tmp_path: Path) -> None:
task_dir = _make_task(tmp_path, "research-task", {"SPEC.md": "# Spec"})
state, artifacts = determine_task_state(task_dir)
state, phase_raw, artifacts = determine_task_state(task_dir)
assert state == TaskState.RESEARCH
assert "SPEC.md" in artifacts
@@ -44,7 +44,7 @@ def test_implementation_state(tmp_path: Path) -> None:
"impl-task",
{"SPEC.md": "# Spec", "TEST_PLAN.md": "# Tests", "IMPLEMENTATION.md": "# Impl"},
)
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.IMPLEMENT
@@ -54,7 +54,7 @@ def test_implementation_from_test_plan(tmp_path: Path) -> None:
"impl-test-plan",
{"SPEC.md": "# Spec", "TEST_PLAN.md": "# Tests"},
)
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.TEST_DESIGN
@@ -64,7 +64,7 @@ def test_bug_find_state(tmp_path: Path) -> None:
"bug-task",
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl", "BUG_REPORT.md": "# Bugs"},
)
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.BUG_FIND
@@ -79,7 +79,7 @@ def test_adv_bug_find_state(tmp_path: Path) -> None:
"ADVERSARIAL_BUG_REPORT.md": "# Adv",
},
)
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.ADV_BUG_FIND
@@ -89,7 +89,7 @@ def test_done_state(tmp_path: Path) -> None:
"done-task",
{"SPEC.md": "# Spec", "VERDICT.md": "## Status: PASS"},
)
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.DONE
@@ -99,13 +99,13 @@ def test_blocked_state(tmp_path: Path) -> None:
"blocked-task",
{"SPEC.md": "# Spec", "VERDICT.md": "## Status: FAIL"},
)
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.BLOCKED
def test_empty_verdict_is_blocked(tmp_path: Path) -> None:
task_dir = _make_task(tmp_path, "empty-verdict", {"VERDICT.md": ""})
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.BLOCKED
@@ -166,43 +166,43 @@ class TestVerdictParsing:
def test_pass_verdict_with_fail_in_findings(self, tmp_path: Path) -> None:
ver = "## Status: PASS\n\nThe previous FAIL finding was resolved."
task_dir = _make_task(tmp_path, "pass-with-fail", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.DONE
def test_pass_verdict_with_needs_review_in_body(self, tmp_path: Path) -> None:
ver = "## Status: PASS\n\nNote: NEEDS_REVIEW was discussed but resolved."
task_dir = _make_task(tmp_path, "pass-with-nr", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.DONE
def test_fail_verdict_structured(self, tmp_path: Path) -> None:
ver = "## Status: FAIL\n\n2 tests PASS, 1 test FAIL."
task_dir = _make_task(tmp_path, "fail-mentions-pass", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.BLOCKED
def test_needs_review_verdict_structured(self, tmp_path: Path) -> None:
ver = "## Status: NEEDS_REVIEW\n\nSome items PASS but need review."
task_dir = _make_task(tmp_path, "nr-mentions-pass", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.BLOCKED
def test_verdict_with_bold_status(self, tmp_path: Path) -> None:
ver = "# Verdict\n\n- **Status**: PASS\n- **Timestamp**: 2025-01-01"
task_dir = _make_task(tmp_path, "bold-status", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.DONE
def test_verdict_no_status_line(self, tmp_path: Path) -> None:
ver = "# Verdict\nEverything looks good, PASS!"
task_dir = _make_task(tmp_path, "no-status-line", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.REFEREE # no structured ## Status: header → ambiguous
def test_verdict_no_status_no_keywords(self, tmp_path: Path) -> None:
ver = "# Verdict\n\nNeeds further discussion."
task_dir = _make_task(tmp_path, "no-status-no-keywords", {"SPEC.md": "# Spec", "VERDICT.md": ver})
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.REFEREE
@@ -211,7 +211,7 @@ class TestStateMachineAlignment:
def test_implementation_alone_shows_implement(self, tmp_path: Path) -> None:
task_dir = _make_task(tmp_path, "impl-only", {"IMPLEMENTATION.md": "# Impl"})
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.IMPLEMENT
def test_bug_report_without_adversarial(self, tmp_path: Path) -> None:
@@ -219,7 +219,7 @@ class TestStateMachineAlignment:
tmp_path, "bug-only",
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl", "BUG_REPORT.md": "# Bugs"},
)
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.BUG_FIND
def test_both_bug_reports_shows_adv_bug_find(self, tmp_path: Path) -> None:
@@ -228,7 +228,7 @@ class TestStateMachineAlignment:
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl",
"BUG_REPORT.md": "# Bugs", "ADVERSARIAL_BUG_REPORT.md": "# Adv"},
)
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.ADV_BUG_FIND
def test_adv_bug_report_alone_shows_bug_find(self, tmp_path: Path) -> None:
@@ -237,22 +237,22 @@ class TestStateMachineAlignment:
{"SPEC.md": "# Spec", "IMPLEMENTATION.md": "# Impl",
"ADVERSARIAL_BUG_REPORT.md": "# Adv bugs only"},
)
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.BUG_FIND
def test_spec_alone_shows_research(self, tmp_path: Path) -> None:
task_dir = _make_task(tmp_path, "spec-only", {"SPEC.md": "# Spec"})
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.RESEARCH
def test_test_plan_shows_test_design(self, tmp_path: Path) -> None:
task_dir = _make_task(tmp_path, "testplan", {"SPEC.md": "# Spec", "TEST_PLAN.md": "# Tests"})
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.TEST_DESIGN
def test_design_with_spec_shows_design(self, tmp_path: Path) -> None:
task_dir = _make_task(tmp_path, "design-spec", {"SPEC.md": "# Spec", "DESIGN.md": "# Design"})
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.DESIGN
def test_doc_review_shows_doc_review(self, tmp_path: Path) -> None:
@@ -262,7 +262,7 @@ class TestStateMachineAlignment:
"BUG_REPORT.md": "# Bugs", "ADVERSARIAL_BUG_REPORT.md": "# Adv",
"DOC_REVIEW.md": "# Docs"},
)
state, _ = determine_task_state(task_dir)
state, phase_raw, _ = determine_task_state(task_dir)
assert state == TaskState.DOC_REVIEW