From f980ccfe27c02280ceb4332f5090492b16206c48 Mon Sep 17 00:00:00 2001 From: Lap Tran Date: Fri, 26 Jun 2026 13:26:00 -0400 Subject: [PATCH] feat(dashboard): model badges on kanban cards + design doc update - task.py: Task dataclass gains models: dict[str, str], loaded from .state.models in discover_tasks() - app.py: models dict included in all task API responses - dashboard.js: model badges rendered between artifacts and subtask progress on kanban cards; ROLE_LABELS map for readable tooltips - styles.css: .task-card-models and .model-badge styles - design/loops/technical.md: document {model} substitution token, per-role model field, and model-divergence brake gate (gate #7) --- automaton/dashboard/core/task.py | 10 ++++++++++ automaton/dashboard/html/dashboard.js | 20 ++++++++++++++++++++ automaton/dashboard/html/styles.css | 2 ++ automaton/dashboard/ui/app.py | 1 + design/loops/technical.md | 5 +++++ scripts/automaton-cleanup.sh | 2 +- 6 files changed, 39 insertions(+), 1 deletion(-) diff --git a/automaton/dashboard/core/task.py b/automaton/dashboard/core/task.py index 2587681..ee6f9cd 100644 --- a/automaton/dashboard/core/task.py +++ b/automaton/dashboard/core/task.py @@ -133,6 +133,7 @@ class Task: vram_config_content: Optional[str] = None waves: list[WaveGroup] = field(default_factory=list) is_corrupted: bool = False + models: dict[str, str] = field(default_factory=dict) @property def display_name(self) -> str: @@ -690,6 +691,15 @@ def discover_tasks(tasks_dir: Path) -> list[Task]: task.parent_spec_content = parse_parent_spec(folder_path) task.vram_config_content = parse_vram_config(folder_path) + # Load .state.models for model divergence badges + state_models_path = folder_path / ".state.models" + if state_models_path.exists(): + try: + import json as _json + task.models = _json.loads(state_models_path.read_text(encoding="utf-8")) + except (OSError, IOError, _json.JSONDecodeError): + task.models = {} + tasks.append(task) # Sort by state (most advanced first) diff --git a/automaton/dashboard/html/dashboard.js b/automaton/dashboard/html/dashboard.js index e03d15f..9f11bd1 100644 --- a/automaton/dashboard/html/dashboard.js +++ b/automaton/dashboard/html/dashboard.js @@ -155,6 +155,17 @@ function renderBoard() { attachCardListeners(); } +const ROLE_LABELS = { + 'implement': 'Implement', + 'code_review': 'Code Review', + 'bug_find': 'Bug Find', + 'adversarial_bug_find': 'Adv Bug Find', + 'doc_review': 'Doc Review', + 'referee': 'Referee', + 'loop-implement': 'Loop Impl', + 'loop-verify': 'Loop Verify', +}; + const ARTIFACT_LABELS = { 'research': 'SPEC.md', 'decomposition': 'DECOMPOSITION.md', 'design': 'DESIGN.md', 'test_design': 'TEST_PLAN.md', @@ -182,11 +193,20 @@ function renderTaskCard(task) { const label = ARTIFACT_LABELS[col.id] || col.label; return `${label}`; }).join('')}`; + const modelKeys = Object.keys(task.models || {}); + const modelsHtml = modelKeys.length > 0 + ? `
${modelKeys.map(role => { + const m = task.models[role]; + const roleLabel = ROLE_LABELS[role] || role; + return `${m}`; + }).join('')}
` + : ''; return `
${escapeHtml(task.display_name)}${statusIcon}
${subLabel}
${task.status_reason ? `
${escapeHtml(task.status_reason)}
` : ''} ${artifactsHtml} + ${modelsHtml} ${progressHtml ? `` : ''} ${subtasksHtml}
`; diff --git a/automaton/dashboard/html/styles.css b/automaton/dashboard/html/styles.css index 13e576a..63dec58 100644 --- a/automaton/dashboard/html/styles.css +++ b/automaton/dashboard/html/styles.css @@ -332,6 +332,8 @@ kbd { .task-card-artifacts { display: flex; flex-wrap: wrap; gap: 4px; margin-top: 6px; } .artifact-badge { font-size: 10px; padding: 2px 8px; background: var(--bg-primary); border: 1px solid var(--border-color); border-radius: 4px; color: var(--text-secondary); font-family: 'SF Mono', 'Fira Code', monospace; font-weight: 500; } +.task-card-models { display: flex; flex-wrap: wrap; gap: 4px; margin-top: 6px; } +.model-badge { font-size: 10px; padding: 1px 6px; background: #e3f2fd; border: 1px solid #90caf9; border-radius: 4px; color: #1565c0; font-family: 'SF Mono', 'Fira Code', monospace; font-weight: 500; } /* Transition button — advance to next phase */ .transition-btn { display: inline-block; padding: 6px 16px; border: 1px solid var(--primary); border-radius: var(--radius-sm); cursor: pointer; font-size: 12px; font-weight: 500; background: rgba(74, 144, 226, 0.1); color: var(--primary); font-family: inherit; transition: all 0.15s; } diff --git a/automaton/dashboard/ui/app.py b/automaton/dashboard/ui/app.py index c7ed0e9..e547287 100644 --- a/automaton/dashboard/ui/app.py +++ b/automaton/dashboard/ui/app.py @@ -245,6 +245,7 @@ class DashboardHandler(SimpleHTTPRequestHandler): "is_approval_gated": t.is_approval_gated, "blocker": t.blocker, "waves": [{"wave_number": w.wave_number, "label": w.label, "sub_task_names": w.sub_task_names} for w in t.waves], + "models": t.models, } for t in tasks ] diff --git a/design/loops/technical.md b/design/loops/technical.md index 6b0087c..0b6ae82 100644 --- a/design/loops/technical.md +++ b/design/loops/technical.md @@ -103,6 +103,7 @@ Gate checks, in order: 4. **Task phase** — if `current_task` is set, that task's `.state` must still be one of the phases this loop is allowed to operate on. If the task has transitioned out (e.g. to `human_intervention` by some other path), halt as `human_intervention`. 5. **Worktree drift** — if worktree branch diverges from main in a way that indicates the loop wrote files outside its scope (checked via `git diff --name-only main...HEAD` restricted to `file_scope`), halt as `drift_detected`. 6. **Score plateau** — last N entries in `score_history` are flat or monotonically decreasing (where N = `score_plateau_window`). Trip → halt as `verifier_failed`. +7. **Model divergence** — in multi-LLM mode (2+ models in `models.json`), checks that the loop's implement and verify roles use different models. If they share the same model, halt as `human_intervention` (this prevents same-model verification / rubber-stamping within a loop tick). Single-LLM mode is exempt. Model is resolved from `roles[].model` if set, otherwise the manifest default. All halts atomically set `status=halted`, `halt_reason=`, write to `.state.log`, and call `--pause-loop`'s schedule-disable step (see §6). @@ -258,11 +259,13 @@ To bound `outputs/` directory growth (O5 from `add-loop-runner/BUG_REPORT.md`), v1.1's default `harness.command` is `opencode run` -- matching the framework's primary harness -- but the shape is generic. The runner substitutes the following tokens into the `command` list (single argv element per token, no shell expansion): +- `{model}` -- the model assigned to the role (from `roles[].model` or manifest default). Passed via `extras["model"]`. If the role has no model assignment, the token is left unsubstituted. - `{prompt}` -- resolved prompt file path (loop-local override or framework default). Kept for backwards compat and harnesses that prefer a file path. - `{prompt_content}` -- the resolved prompt file's text content as a single argv element. Safe under `subprocess.run` list mode; no shell quoting needed. Used by the default command since `opencode run` takes the message as a positional argument and has no `--prompt-file` flag. - `{cwd}` -- the working directory the harness should run in (the loop's project root or worktree). - `{output}`, `{artifact}` -- role-specific extras (the implement output path handed to verify). - `{verdict}`, `{current_task}`, `{current_phase}`, etc. -- other runtime extras; see `_resolve_prompt` below. +- `{model}` -- the model assigned to the role being invoked (from `roles[].model` in `loop.json`, or the manifest default). The runner passes it via the `extras["model"]` key. If the role has no explicit model, `{model}` is left as-is (no substitution). This allows per-role model pinning without hardcoding the model name in `harness.command`. The default command does NOT hardcode a `--model` flag; the spawned `opencode run` inherits the model from the project/user config. Users who want a per-loop model override (e.g. a local LLM for ticks) set `harness.command` in their `loop.json`: @@ -336,6 +339,8 @@ This means the harness receives a fully-resolved prompt file with all context ba } ``` +Each role in `roles` accepts an optional `"model"` field to pin a specific model for that role (e.g. `"implement": {"prompt": "loop-implement.md", "tier": 16000, "model": "model-a"}`). When set, the runner passes `model=` in the harness extras for that role, enabling `{model}` substitution in `harness.command`. This is how multi-LLM loops prevent same-model verification — see `CONFLICT_MATRIX` in `status.py`. + Installs default-on at `install.sh` time: `status.py --create-loop self-improvement --from-template self-improvement --project "$FRAMEWORK_DIR"` then `status.py --install-schedule self-improvement --interval 3600 --project "$FRAMEWORK_DIR"`. Both commands use `|| true` so the framework works even if loop creation fails. `update.sh` bootstraps the loop idempotently for existing users (checks `if [ ! -d "$FRAMEWORK_DIR/loops/self-improvement" ]`). Disabling: `status.py --pause-loop self-improvement --project ~/.automaton/`. ## 10. Tests (`tests/test_loops.py`) diff --git a/scripts/automaton-cleanup.sh b/scripts/automaton-cleanup.sh index fdc0af0..9b256f5 100755 --- a/scripts/automaton-cleanup.sh +++ b/scripts/automaton-cleanup.sh @@ -1,2 +1,2 @@ #!/usr/bin/env bash -python3 "/Users/laptran/.automaton/scripts/status.py" --cleanup-done --days 7 --project "/private/var/folders/f5/yv0dzbnx47x3yp8sc_2519gh0000gn/T/pytest-of-laptran/pytest-126/test_uninstall_via_disabled_re0" +python3 "/Users/laptran/.automaton/scripts/status.py" --cleanup-done --days 7 --project "/private/var/folders/f5/yv0dzbnx47x3yp8sc_2519gh0000gn/T/pytest-of-laptran/pytest-127/test_uninstall_via_disabled_re0"