feat(dashboard): model badges on kanban cards + design doc update

- task.py: Task dataclass gains models: dict[str, str], loaded from
  .state.models in discover_tasks()
- app.py: models dict included in all task API responses
- dashboard.js: model badges rendered between artifacts and subtask
  progress on kanban cards; ROLE_LABELS map for readable tooltips
- styles.css: .task-card-models and .model-badge styles
- design/loops/technical.md: document {model} substitution token,
  per-role model field, and model-divergence brake gate (gate #7)
This commit is contained in:
Lap Tran
2026-06-26 13:26:00 -04:00
parent 35e449b03e
commit f980ccfe27
6 changed files with 39 additions and 1 deletions
+10
View File
@@ -133,6 +133,7 @@ class Task:
vram_config_content: Optional[str] = None vram_config_content: Optional[str] = None
waves: list[WaveGroup] = field(default_factory=list) waves: list[WaveGroup] = field(default_factory=list)
is_corrupted: bool = False is_corrupted: bool = False
models: dict[str, str] = field(default_factory=dict)
@property @property
def display_name(self) -> str: def display_name(self) -> str:
@@ -690,6 +691,15 @@ def discover_tasks(tasks_dir: Path) -> list[Task]:
task.parent_spec_content = parse_parent_spec(folder_path) task.parent_spec_content = parse_parent_spec(folder_path)
task.vram_config_content = parse_vram_config(folder_path) task.vram_config_content = parse_vram_config(folder_path)
# Load .state.models for model divergence badges
state_models_path = folder_path / ".state.models"
if state_models_path.exists():
try:
import json as _json
task.models = _json.loads(state_models_path.read_text(encoding="utf-8"))
except (OSError, IOError, _json.JSONDecodeError):
task.models = {}
tasks.append(task) tasks.append(task)
# Sort by state (most advanced first) # Sort by state (most advanced first)
+20
View File
@@ -155,6 +155,17 @@ function renderBoard() {
attachCardListeners(); attachCardListeners();
} }
const ROLE_LABELS = {
'implement': 'Implement',
'code_review': 'Code Review',
'bug_find': 'Bug Find',
'adversarial_bug_find': 'Adv Bug Find',
'doc_review': 'Doc Review',
'referee': 'Referee',
'loop-implement': 'Loop Impl',
'loop-verify': 'Loop Verify',
};
const ARTIFACT_LABELS = { const ARTIFACT_LABELS = {
'research': 'SPEC.md', 'decomposition': 'DECOMPOSITION.md', 'research': 'SPEC.md', 'decomposition': 'DECOMPOSITION.md',
'design': 'DESIGN.md', 'test_design': 'TEST_PLAN.md', 'design': 'DESIGN.md', 'test_design': 'TEST_PLAN.md',
@@ -182,11 +193,20 @@ function renderTaskCard(task) {
const label = ARTIFACT_LABELS[col.id] || col.label; const label = ARTIFACT_LABELS[col.id] || col.label;
return `<span class="artifact-badge" title="${col.label}">${label}</span>`; return `<span class="artifact-badge" title="${col.label}">${label}</span>`;
}).join('')}</div>`; }).join('')}</div>`;
const modelKeys = Object.keys(task.models || {});
const modelsHtml = modelKeys.length > 0
? `<div class="task-card-models">${modelKeys.map(role => {
const m = task.models[role];
const roleLabel = ROLE_LABELS[role] || role;
return `<span class="model-badge" title="${roleLabel}: ${m}">${m}</span>`;
}).join('')}</div>`
: '';
return `<div class="task-card" data-task="${task.name}" data-status="${statusClass}"> return `<div class="task-card" data-task="${task.name}" data-status="${statusClass}">
<div class="task-card-header"><span class="task-card-name">${escapeHtml(task.display_name)}</span><span class="task-card-status ${statusClass}">${statusIcon}</span></div> <div class="task-card-header"><span class="task-card-name">${escapeHtml(task.display_name)}</span><span class="task-card-status ${statusClass}">${statusIcon}</span></div>
<div class="task-card-sublabel">${subLabel}</div> <div class="task-card-sublabel">${subLabel}</div>
${task.status_reason ? `<div class="task-card-reason">${escapeHtml(task.status_reason)}</div>` : ''} ${task.status_reason ? `<div class="task-card-reason">${escapeHtml(task.status_reason)}</div>` : ''}
${artifactsHtml} ${artifactsHtml}
${modelsHtml}
${progressHtml ? `<div class="task-card-footer"><span class="subtask-progress">${progressHtml}</span></div>` : ''} ${progressHtml ? `<div class="task-card-footer"><span class="subtask-progress">${progressHtml}</span></div>` : ''}
${subtasksHtml} ${subtasksHtml}
</div>`; </div>`;
+2
View File
@@ -332,6 +332,8 @@ kbd {
.task-card-artifacts { display: flex; flex-wrap: wrap; gap: 4px; margin-top: 6px; } .task-card-artifacts { display: flex; flex-wrap: wrap; gap: 4px; margin-top: 6px; }
.artifact-badge { font-size: 10px; padding: 2px 8px; background: var(--bg-primary); border: 1px solid var(--border-color); border-radius: 4px; color: var(--text-secondary); font-family: 'SF Mono', 'Fira Code', monospace; font-weight: 500; } .artifact-badge { font-size: 10px; padding: 2px 8px; background: var(--bg-primary); border: 1px solid var(--border-color); border-radius: 4px; color: var(--text-secondary); font-family: 'SF Mono', 'Fira Code', monospace; font-weight: 500; }
.task-card-models { display: flex; flex-wrap: wrap; gap: 4px; margin-top: 6px; }
.model-badge { font-size: 10px; padding: 1px 6px; background: #e3f2fd; border: 1px solid #90caf9; border-radius: 4px; color: #1565c0; font-family: 'SF Mono', 'Fira Code', monospace; font-weight: 500; }
/* Transition button — advance to next phase */ /* Transition button — advance to next phase */
.transition-btn { display: inline-block; padding: 6px 16px; border: 1px solid var(--primary); border-radius: var(--radius-sm); cursor: pointer; font-size: 12px; font-weight: 500; background: rgba(74, 144, 226, 0.1); color: var(--primary); font-family: inherit; transition: all 0.15s; } .transition-btn { display: inline-block; padding: 6px 16px; border: 1px solid var(--primary); border-radius: var(--radius-sm); cursor: pointer; font-size: 12px; font-weight: 500; background: rgba(74, 144, 226, 0.1); color: var(--primary); font-family: inherit; transition: all 0.15s; }
+1
View File
@@ -245,6 +245,7 @@ class DashboardHandler(SimpleHTTPRequestHandler):
"is_approval_gated": t.is_approval_gated, "is_approval_gated": t.is_approval_gated,
"blocker": t.blocker, "blocker": t.blocker,
"waves": [{"wave_number": w.wave_number, "label": w.label, "sub_task_names": w.sub_task_names} for w in t.waves], "waves": [{"wave_number": w.wave_number, "label": w.label, "sub_task_names": w.sub_task_names} for w in t.waves],
"models": t.models,
} }
for t in tasks for t in tasks
] ]
+5
View File
@@ -103,6 +103,7 @@ Gate checks, in order:
4. **Task phase** — if `current_task` is set, that task's `.state` must still be one of the phases this loop is allowed to operate on. If the task has transitioned out (e.g. to `human_intervention` by some other path), halt as `human_intervention`. 4. **Task phase** — if `current_task` is set, that task's `.state` must still be one of the phases this loop is allowed to operate on. If the task has transitioned out (e.g. to `human_intervention` by some other path), halt as `human_intervention`.
5. **Worktree drift** — if worktree branch diverges from main in a way that indicates the loop wrote files outside its scope (checked via `git diff --name-only main...HEAD` restricted to `file_scope`), halt as `drift_detected`. 5. **Worktree drift** — if worktree branch diverges from main in a way that indicates the loop wrote files outside its scope (checked via `git diff --name-only main...HEAD` restricted to `file_scope`), halt as `drift_detected`.
6. **Score plateau** — last N entries in `score_history` are flat or monotonically decreasing (where N = `score_plateau_window`). Trip → halt as `verifier_failed`. 6. **Score plateau** — last N entries in `score_history` are flat or monotonically decreasing (where N = `score_plateau_window`). Trip → halt as `verifier_failed`.
7. **Model divergence** — in multi-LLM mode (2+ models in `models.json`), checks that the loop's implement and verify roles use different models. If they share the same model, halt as `human_intervention` (this prevents same-model verification / rubber-stamping within a loop tick). Single-LLM mode is exempt. Model is resolved from `roles[<role>].model` if set, otherwise the manifest default.
All halts atomically set `status=halted`, `halt_reason=<reason>`, write to `.state.log`, and call `--pause-loop`'s schedule-disable step (see §6). All halts atomically set `status=halted`, `halt_reason=<reason>`, write to `.state.log`, and call `--pause-loop`'s schedule-disable step (see §6).
@@ -258,11 +259,13 @@ To bound `outputs/` directory growth (O5 from `add-loop-runner/BUG_REPORT.md`),
v1.1's default `harness.command` is `opencode run` -- matching the framework's primary harness -- but the shape is generic. The runner substitutes the following tokens into the `command` list (single argv element per token, no shell expansion): v1.1's default `harness.command` is `opencode run` -- matching the framework's primary harness -- but the shape is generic. The runner substitutes the following tokens into the `command` list (single argv element per token, no shell expansion):
- `{model}` -- the model assigned to the role (from `roles[<role>].model` or manifest default). Passed via `extras["model"]`. If the role has no model assignment, the token is left unsubstituted.
- `{prompt}` -- resolved prompt file path (loop-local override or framework default). Kept for backwards compat and harnesses that prefer a file path. - `{prompt}` -- resolved prompt file path (loop-local override or framework default). Kept for backwards compat and harnesses that prefer a file path.
- `{prompt_content}` -- the resolved prompt file's text content as a single argv element. Safe under `subprocess.run` list mode; no shell quoting needed. Used by the default command since `opencode run` takes the message as a positional argument and has no `--prompt-file` flag. - `{prompt_content}` -- the resolved prompt file's text content as a single argv element. Safe under `subprocess.run` list mode; no shell quoting needed. Used by the default command since `opencode run` takes the message as a positional argument and has no `--prompt-file` flag.
- `{cwd}` -- the working directory the harness should run in (the loop's project root or worktree). - `{cwd}` -- the working directory the harness should run in (the loop's project root or worktree).
- `{output}`, `{artifact}` -- role-specific extras (the implement output path handed to verify). - `{output}`, `{artifact}` -- role-specific extras (the implement output path handed to verify).
- `{verdict}`, `{current_task}`, `{current_phase}`, etc. -- other runtime extras; see `_resolve_prompt` below. - `{verdict}`, `{current_task}`, `{current_phase}`, etc. -- other runtime extras; see `_resolve_prompt` below.
- `{model}` -- the model assigned to the role being invoked (from `roles[<role>].model` in `loop.json`, or the manifest default). The runner passes it via the `extras["model"]` key. If the role has no explicit model, `{model}` is left as-is (no substitution). This allows per-role model pinning without hardcoding the model name in `harness.command`.
The default command does NOT hardcode a `--model` flag; the spawned `opencode run` inherits the model from the project/user config. Users who want a per-loop model override (e.g. a local LLM for ticks) set `harness.command` in their `loop.json`: The default command does NOT hardcode a `--model` flag; the spawned `opencode run` inherits the model from the project/user config. Users who want a per-loop model override (e.g. a local LLM for ticks) set `harness.command` in their `loop.json`:
@@ -336,6 +339,8 @@ This means the harness receives a fully-resolved prompt file with all context ba
} }
``` ```
Each role in `roles` accepts an optional `"model"` field to pin a specific model for that role (e.g. `"implement": {"prompt": "loop-implement.md", "tier": 16000, "model": "model-a"}`). When set, the runner passes `model=<value>` in the harness extras for that role, enabling `{model}` substitution in `harness.command`. This is how multi-LLM loops prevent same-model verification — see `CONFLICT_MATRIX` in `status.py`.
Installs default-on at `install.sh` time: `status.py --create-loop self-improvement --from-template self-improvement --project "$FRAMEWORK_DIR"` then `status.py --install-schedule self-improvement --interval 3600 --project "$FRAMEWORK_DIR"`. Both commands use `|| true` so the framework works even if loop creation fails. `update.sh` bootstraps the loop idempotently for existing users (checks `if [ ! -d "$FRAMEWORK_DIR/loops/self-improvement" ]`). Disabling: `status.py --pause-loop self-improvement --project ~/.automaton/`. Installs default-on at `install.sh` time: `status.py --create-loop self-improvement --from-template self-improvement --project "$FRAMEWORK_DIR"` then `status.py --install-schedule self-improvement --interval 3600 --project "$FRAMEWORK_DIR"`. Both commands use `|| true` so the framework works even if loop creation fails. `update.sh` bootstraps the loop idempotently for existing users (checks `if [ ! -d "$FRAMEWORK_DIR/loops/self-improvement" ]`). Disabling: `status.py --pause-loop self-improvement --project ~/.automaton/`.
## 10. Tests (`tests/test_loops.py`) ## 10. Tests (`tests/test_loops.py`)
+1 -1
View File
@@ -1,2 +1,2 @@
#!/usr/bin/env bash #!/usr/bin/env bash
python3 "/Users/laptran/.automaton/scripts/status.py" --cleanup-done --days 7 --project "/private/var/folders/f5/yv0dzbnx47x3yp8sc_2519gh0000gn/T/pytest-of-laptran/pytest-126/test_uninstall_via_disabled_re0" python3 "/Users/laptran/.automaton/scripts/status.py" --cleanup-done --days 7 --project "/private/var/folders/f5/yv0dzbnx47x3yp8sc_2519gh0000gn/T/pytest-of-laptran/pytest-127/test_uninstall_via_disabled_re0"