feat(model-divergence): full enforcement — manifest, transition, claim, audit, loop gates, detect script
Completes all 3 model-divergence enforcement subtasks:
- scripts/detect_models.py: probes opencode.json + localhost endpoints,
builds models.json with --json/--write/--force
- scripts/status.py: CONFLICT_MATRIX, --model flag, --transition --model,
--claim --model, --audit Category 6, model-divergence brake gate in
--check-gate, helpers for manifest loading and conflict checking
- scripts/loop-runner.py: _role_model() helper + {model} passed via extras
dict to _invoke_harness for implement, verify, orchestrate roles
- tests/test_model_divergence.py: 33 tests covering all enforcement layers
- Single-LLM mode: record model advisory, no conflict check
- Multi-LLM mode (2+ models): conflict matrix enforced at transition, claim,
and loop brake gate
- Project-level models.json preferred over global ~/.automaton/models.json
This commit is contained in:
@@ -1,2 +1,2 @@
|
||||
#!/usr/bin/env bash
|
||||
python3 "/Users/laptran/.automaton/scripts/status.py" --cleanup-done --days 7 --project "/private/var/folders/f5/yv0dzbnx47x3yp8sc_2519gh0000gn/T/pytest-of-laptran/pytest-120/test_uninstall_via_disabled_re0"
|
||||
python3 "/Users/laptran/.automaton/scripts/status.py" --cleanup-done --days 7 --project "/private/var/folders/f5/yv0dzbnx47x3yp8sc_2519gh0000gn/T/pytest-of-laptran/pytest-126/test_uninstall_via_disabled_re0"
|
||||
|
||||
@@ -0,0 +1,314 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Probe opencode.json and localhost endpoints to produce a candidate models.json.
|
||||
|
||||
Usage:
|
||||
python3 scripts/detect_models.py [--json] [--write]
|
||||
|
||||
Without --json, prints a human-readable report.
|
||||
With --json, emits the candidate models.json to stdout as the last JSON line.
|
||||
With --write, writes the candidate to ~/.automaton/models.json (idempotent,
|
||||
never overwrites an existing file unless --force is also given).
|
||||
|
||||
Probing strategy (stdlib only):
|
||||
1. Parse opencode.json (or opencode.jsonc) for configured provider+model pairs.
|
||||
2. Probe localhost endpoints to find locally-running LLM servers:
|
||||
- http://localhost:8080/v1/models (llama.cpp / generic OpenAI-compatible)
|
||||
- http://localhost:11434/api/tags (Ollama)
|
||||
- http://localhost:1234/v1/models (LM Studio)
|
||||
- http://localhost:8000/v1/models (vLLM)
|
||||
3. Merge results into a candidate models.json.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
AUTOMATON_DIR = Path.home() / ".automaton"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# opencode.json parsing
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _find_opencode_json() -> Optional[Path]:
|
||||
"""Locate the opencode config file (opencode.json or opencode.jsonc)."""
|
||||
candidates = [
|
||||
Path.cwd() / "opencode.json",
|
||||
Path.cwd() / "opencode.jsonc",
|
||||
AUTOMATON_DIR / "opencode.json",
|
||||
AUTOMATON_DIR / "opencode.jsonc",
|
||||
Path.home() / ".opencode.json",
|
||||
Path.home() / ".config" / "opencode" / "opencode.json",
|
||||
Path.home() / ".config" / "opencode" / "opencode.jsonc",
|
||||
]
|
||||
for p in candidates:
|
||||
if p.exists():
|
||||
return p
|
||||
return None
|
||||
|
||||
|
||||
def _parse_opencode_models(config_path: Path) -> list[dict]:
|
||||
"""Extract model entries from an opencode.json config.
|
||||
|
||||
Expected structure (common patterns):
|
||||
{
|
||||
"providers": {
|
||||
"opencode": { "model": "glm-4.6", ... },
|
||||
...
|
||||
}
|
||||
}
|
||||
or a flatter:
|
||||
{
|
||||
"model": "glm-4.6",
|
||||
...
|
||||
}
|
||||
"""
|
||||
try:
|
||||
content = config_path.read_text(encoding="utf-8")
|
||||
except OSError:
|
||||
return []
|
||||
# Strip JSONC comments (// line comments only, sufficient for our use)
|
||||
content = re.sub(r"//.*", "", content)
|
||||
try:
|
||||
data = json.loads(content)
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
if not isinstance(data, dict):
|
||||
return []
|
||||
models: list[dict] = []
|
||||
seen: set[str] = set()
|
||||
|
||||
# Check top-level "model" field (single-model config)
|
||||
single = data.get("model")
|
||||
if isinstance(single, str) and single not in seen:
|
||||
seen.add(single)
|
||||
models.append({"name": single, "provider": "opencode", "context_window": None, "location": "remote"})
|
||||
|
||||
# Check providers dict
|
||||
providers = data.get("providers") or {}
|
||||
for prov_name, prov_cfg in providers.items():
|
||||
if isinstance(prov_cfg, dict):
|
||||
model_name = prov_cfg.get("model")
|
||||
if isinstance(model_name, str) and model_name not in seen:
|
||||
seen.add(model_name)
|
||||
models.append({"name": model_name, "provider": prov_name, "context_window": None, "location": "remote"})
|
||||
|
||||
# Check "models" list (explicit model roster)
|
||||
model_list = data.get("models")
|
||||
if isinstance(model_list, list):
|
||||
for entry in model_list:
|
||||
if isinstance(entry, dict):
|
||||
name = entry.get("name") or entry.get("model")
|
||||
if isinstance(name, str) and name not in seen:
|
||||
seen.add(name)
|
||||
models.append({
|
||||
"name": name,
|
||||
"provider": entry.get("provider", "opencode"),
|
||||
"context_window": entry.get("context_window"),
|
||||
"location": entry.get("location", "remote"),
|
||||
})
|
||||
return models
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Localhost probing
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _fetch_json(url: str, timeout: int = 5) -> Optional[dict]:
|
||||
"""Fetch a JSON response from a URL using urllib (stdlib)."""
|
||||
import urllib.request
|
||||
import urllib.error
|
||||
try:
|
||||
req = urllib.request.Request(url, method="GET")
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
body = resp.read().decode("utf-8")
|
||||
return json.loads(body)
|
||||
except (OSError, urllib.error.URLError, json.JSONDecodeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _probe_ollama() -> list[dict]:
|
||||
"""Probe Ollama: GET http://localhost:11434/api/tags → models[].name"""
|
||||
data = _fetch_json("http://localhost:11434/api/tags")
|
||||
if not data:
|
||||
return []
|
||||
models_list = data.get("models") or []
|
||||
return [
|
||||
{"name": m.get("name"), "provider": "ollama", "context_window": None, "location": "http://localhost:11434"}
|
||||
for m in models_list
|
||||
if isinstance(m, dict) and isinstance(m.get("name"), str)
|
||||
]
|
||||
|
||||
|
||||
def _probe_openai_compatible(url: str, provider: str) -> list[dict]:
|
||||
"""Probe an OpenAI-compatible /v1/models endpoint."""
|
||||
data = _fetch_json(url)
|
||||
if not data:
|
||||
return []
|
||||
model_list = data.get("data") or []
|
||||
return [
|
||||
{"name": m.get("id"), "provider": provider, "context_window": None, "location": url}
|
||||
for m in model_list
|
||||
if isinstance(m, dict) and isinstance(m.get("id"), str)
|
||||
]
|
||||
|
||||
|
||||
_ENDPOINTS = [
|
||||
("http://localhost:8080/v1/models", "llama.cpp"),
|
||||
("http://localhost:11434/api/tags", "ollama"), # handled separately above
|
||||
("http://localhost:1234/v1/models", "lm-studio"),
|
||||
("http://localhost:8000/v1/models", "vllm"),
|
||||
]
|
||||
|
||||
|
||||
def _probe_localhost() -> list[dict]:
|
||||
"""Probe all known localhost endpoints and merge results."""
|
||||
seen_names: set[str] = set()
|
||||
models: list[dict] = []
|
||||
for url, provider in _ENDPOINTS:
|
||||
if provider == "ollama":
|
||||
result = _probe_ollama()
|
||||
else:
|
||||
result = _probe_openai_compatible(url, provider)
|
||||
for m in result:
|
||||
n = m.get("name")
|
||||
if isinstance(n, str) and n not in seen_names:
|
||||
seen_names.add(n)
|
||||
models.append(m)
|
||||
return models
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Merge & write
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def build_candidate_models(probe_local: bool = True) -> dict:
|
||||
"""Build a candidate models.json dict.
|
||||
|
||||
1. Parse models from opencode.json
|
||||
2. Optionally probe localhost endpoints
|
||||
3. Merge: opencode config models come first; local probes fill in gaps.
|
||||
4. Build result with default, advised, models[].
|
||||
"""
|
||||
opencode_path = _find_opencode_json()
|
||||
config_models: list[dict] = []
|
||||
if opencode_path:
|
||||
config_models = _parse_opencode_models(opencode_path)
|
||||
|
||||
local_models: list[dict] = []
|
||||
if probe_local:
|
||||
local_models = _probe_localhost()
|
||||
|
||||
# Merge: key by name, config models take priority (unordered)
|
||||
merged: dict[str, dict] = {}
|
||||
for m in config_models:
|
||||
n = m["name"]
|
||||
if n not in merged:
|
||||
merged[n] = m
|
||||
for m in local_models:
|
||||
n = m.get("name")
|
||||
if n and n not in merged:
|
||||
merged[n] = m
|
||||
|
||||
models_list = list(merged.values())
|
||||
|
||||
# Determine default: first config model, or first local model, or empty
|
||||
default_name: Optional[str] = None
|
||||
if config_models:
|
||||
default_name = config_models[0].get("name")
|
||||
elif local_models:
|
||||
default_name = local_models[0].get("name")
|
||||
|
||||
# Determine advised: if only 0-1 models, set advised=true; else false
|
||||
advised = len(models_list) <= 1
|
||||
|
||||
result: dict = {
|
||||
"schema_version": 1,
|
||||
"default": default_name,
|
||||
"advised": advised,
|
||||
"models": models_list,
|
||||
}
|
||||
|
||||
return result
|
||||
|
||||
|
||||
def write_models_file(candidate: dict, force: bool = False) -> bool:
|
||||
"""Write candidate models.json to AUTOMATON_DIR.
|
||||
|
||||
Never overwrites an existing file unless force=True.
|
||||
Returns True if written, False if skipped.
|
||||
"""
|
||||
target = AUTOMATON_DIR / "models.json"
|
||||
if target.exists() and not force:
|
||||
return False
|
||||
target.write_text(json.dumps(candidate, indent=2) + "\n")
|
||||
return True
|
||||
|
||||
|
||||
def format_report(candidate: dict) -> str:
|
||||
"""Human-readable report of the candidate models."""
|
||||
lines = []
|
||||
lines.append("=== Model Detection Report ===")
|
||||
lines.append("")
|
||||
source = "No opencode.json found" if not _find_opencode_json() else f"Config: {_find_opencode_json()}"
|
||||
lines.append(f"Source: {source}")
|
||||
lines.append("")
|
||||
models = candidate.get("models", [])
|
||||
if not models:
|
||||
lines.append("No models detected.")
|
||||
else:
|
||||
lines.append(f"Detected {len(models)} model(s):")
|
||||
for m in models:
|
||||
loc = m.get("location", "unknown")
|
||||
prov = m.get("provider", "?")
|
||||
ctx = m.get("context_window")
|
||||
ctx_str = f", context: {ctx}" if ctx else ""
|
||||
lines.append(f" - {m['name']} ({prov}, {loc}{ctx_str})")
|
||||
lines.append("")
|
||||
lines.append(f"Default: {candidate.get('default', 'none')}")
|
||||
lines.append(f"Advised: {candidate.get('advised', False)}")
|
||||
lines.append(f"Mode: {'multi-LLM' if len(models) >= 2 else 'single-LLM'}")
|
||||
lines.append("")
|
||||
target = AUTOMATON_DIR / "models.json"
|
||||
if target.exists():
|
||||
lines.append(f"models.json already exists at {target} (use --force to overwrite)")
|
||||
else:
|
||||
lines.append(f"Ready to write to {target} (use --write to create)")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
import argparse
|
||||
parser = argparse.ArgumentParser(description="Detect available LLM models and write models.json")
|
||||
parser.add_argument("--json", action="store_true", help="Output candidate JSON on last line")
|
||||
parser.add_argument("--write", action="store_true", help="Write candidate models.json to ~/.automaton/ (idempotent)")
|
||||
parser.add_argument("--force", action="store_true", help="Overwrite existing models.json")
|
||||
parser.add_argument("--no-probe", action="store_true", help="Skip localhost endpoint probing")
|
||||
args = parser.parse_args()
|
||||
|
||||
candidate = build_candidate_models(probe_local=not args.no_probe)
|
||||
|
||||
if args.write:
|
||||
written = write_models_file(candidate, force=args.force)
|
||||
if written:
|
||||
print(f"Written models.json to {AUTOMATON_DIR / 'models.json'}")
|
||||
else:
|
||||
print(f"Skipped: {AUTOMATON_DIR / 'models.json'} already exists (use --force to overwrite)")
|
||||
|
||||
if args.json:
|
||||
print(json.dumps(candidate))
|
||||
else:
|
||||
print(format_report(candidate))
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
+54
-15
@@ -375,6 +375,9 @@ def _invoke_harness(
|
||||
"""Build the harness command from loop.json and invoke it. Returns stdout.
|
||||
|
||||
extras: substitution tokens specific to this role ({artifact}, {verdict}, etc).
|
||||
If extras contains a "model" key, the ``{model}`` token in the harness
|
||||
command is substituted. The caller is responsible for passing the model
|
||||
via extras (extracted from loop.json role config or manifest default).
|
||||
"""
|
||||
resolved_prompt = prompt_path
|
||||
if loop_path is not None:
|
||||
@@ -524,6 +527,24 @@ def _role_prompt(cfg: dict, role: str) ->Optional[str]:
|
||||
return role_cfg.get("prompt")
|
||||
|
||||
|
||||
def _role_model(cfg: dict, role: str) -> Optional[str]:
|
||||
"""Get the model configured for a role in loop.json, or the manifest default."""
|
||||
roles = cfg.get("roles") or {}
|
||||
role_cfg = roles.get(role) or {}
|
||||
model = role_cfg.get("model")
|
||||
if model:
|
||||
return model
|
||||
models_file = AUTOMATON_DIR / "models.json"
|
||||
if models_file.exists():
|
||||
try:
|
||||
import json as _mj
|
||||
manifest = _mj.loads(models_file.read_text())
|
||||
return manifest.get("default")
|
||||
except (OSError, _mj.JSONDecodeError):
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Work sources (task add-goal-mode)
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -809,27 +830,39 @@ def cmd_tick(args, runner_state: Optional[dict] = None) -> dict:
|
||||
out_dir = _outputs_dir(loop_path)
|
||||
tick_num = state.get('iteration_count', 0) + 1
|
||||
impl_output = str(out_dir / f"tick{tick_num}-implement.json")
|
||||
impl_model = _role_model(cfg, "implement")
|
||||
implement_extras = {
|
||||
"output": impl_output,
|
||||
"current_task": current_task,
|
||||
"task_brief": task_brief,
|
||||
"acceptance_criteria": acceptance,
|
||||
"next_hint": next_hint,
|
||||
}
|
||||
if impl_model:
|
||||
implement_extras["model"] = impl_model
|
||||
implement_stdout = _invoke_harness(
|
||||
harness_cfg, "implement", implement_prompt, cwd,
|
||||
extras={"output": impl_output,
|
||||
"current_task": current_task,
|
||||
"task_brief": task_brief,
|
||||
"acceptance_criteria": acceptance,
|
||||
"next_hint": next_hint},
|
||||
extras=implement_extras,
|
||||
loop_path=loop_path, tick_num=tick_num)
|
||||
(Path(impl_output)).write_text(implement_stdout)
|
||||
|
||||
# Step 6: spawn Verify
|
||||
verify_prompt = _role_prompt(cfg, "verify") or ""
|
||||
verify_output = str(out_dir / f"tick{tick_num}-verify.json")
|
||||
verify_model = _role_model(cfg, "verify")
|
||||
verify_extras = {
|
||||
"output": verify_output,
|
||||
"artifact": impl_output,
|
||||
"current_task": current_task,
|
||||
"task_brief": task_brief,
|
||||
"acceptance_criteria": acceptance,
|
||||
"next_hint": next_hint,
|
||||
}
|
||||
if verify_model:
|
||||
verify_extras["model"] = verify_model
|
||||
verify_stdout = _invoke_harness(
|
||||
harness_cfg, "verify", verify_prompt, cwd,
|
||||
extras={"output": verify_output,
|
||||
"artifact": impl_output,
|
||||
"current_task": current_task,
|
||||
"task_brief": task_brief,
|
||||
"acceptance_criteria": acceptance,
|
||||
"next_hint": next_hint},
|
||||
extras=verify_extras,
|
||||
loop_path=loop_path, tick_num=tick_num)
|
||||
(Path(verify_output)).write_text(verify_stdout)
|
||||
|
||||
@@ -854,12 +887,18 @@ def cmd_tick(args, runner_state: Optional[dict] = None) -> dict:
|
||||
# Step 9: spawn Orchestrate
|
||||
orch_prompt = _role_prompt(cfg, "orchestrate") or ""
|
||||
orch_output = str(out_dir / f"tick{tick_num}-orchestrate.json")
|
||||
orch_model = _role_model(cfg, "orchestrate")
|
||||
orch_extras = {
|
||||
"output": orch_output,
|
||||
"verdict": json.dumps(verdict),
|
||||
"current_task": current_task,
|
||||
"current_phase": state.get("current_phase", ""),
|
||||
}
|
||||
if orch_model:
|
||||
orch_extras["model"] = orch_model
|
||||
orch_stdout = _invoke_harness(
|
||||
harness_cfg, "orchestrate", orch_prompt, cwd,
|
||||
extras={"output": orch_output,
|
||||
"verdict": json.dumps(verdict),
|
||||
"current_task": current_task,
|
||||
"current_phase": state.get("current_phase", "")},
|
||||
extras=orch_extras,
|
||||
loop_path=loop_path, tick_num=tick_num)
|
||||
(Path(orch_output)).write_text(orch_stdout)
|
||||
|
||||
|
||||
+261
-2
@@ -153,7 +153,19 @@ FORBIDDEN_ARTIFACTS = {
|
||||
}
|
||||
|
||||
NON_ARTIFACT_FILES = {".state", ".state.tmp", ".state.lock", ".state.approvals",
|
||||
".state.implementer", ".state.lastedit", "VRAM_CONFIG.md", "PARENT_SPEC.md", "REVIEW.md"}
|
||||
".state.implementer", ".state.lastedit", ".state.models",
|
||||
"VRAM_CONFIG.md", "PARENT_SPEC.md", "REVIEW.md"}
|
||||
|
||||
# Model-divergence enforcement
|
||||
CONFLICT_MATRIX = {
|
||||
"code_review": {"implement"},
|
||||
"bug_find": {"implement"},
|
||||
"adversarial_bug_find": {"implement", "bug_find"},
|
||||
"referee": {"implement", "bug_find", "adversarial_bug_find"},
|
||||
"loop-verify": {"loop-implement"},
|
||||
}
|
||||
|
||||
MODELS_JSON_FILE = "models.json"
|
||||
|
||||
PHASE_PRIORITY = {
|
||||
"referee": 12, "doc_review": 11, "adversarial_bug_find": 10,
|
||||
@@ -438,6 +450,124 @@ def _lock_timeout_seconds(project: Optional[str] = None) -> int:
|
||||
return val
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Model-divergence enforcement helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _load_models_manifest(project: Optional[str] = None) -> Optional[dict]:
|
||||
"""Load the models.json manifest for the given project.
|
||||
|
||||
Searches:
|
||||
1. project/.automaton/models.json
|
||||
2. ~/.automaton/models.json (fallback)
|
||||
|
||||
Returns None if no models.json exists (single-LLM mode, backward compatible).
|
||||
"""
|
||||
project_dir = _find_project_dir(project)
|
||||
candidates = [
|
||||
project_dir / ".automaton" / MODELS_JSON_FILE,
|
||||
AUTOMATON_DIR / MODELS_JSON_FILE,
|
||||
]
|
||||
for path in candidates:
|
||||
if path.exists():
|
||||
try:
|
||||
return json.loads(path.read_text())
|
||||
except (OSError, json.JSONDecodeError):
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _get_model_mode(manifest: Optional[dict]) -> str:
|
||||
"""Determine the model mode: 'single' or 'multi-llm'.
|
||||
|
||||
- Missing manifest → single-LLM (backward compatible)
|
||||
- 0-1 models → single-LLM
|
||||
- 2+ models → multi-LLM
|
||||
"""
|
||||
if manifest is None:
|
||||
return "single"
|
||||
models = manifest.get("models") or []
|
||||
if len(models) >= 2:
|
||||
return "multi-llm"
|
||||
return "single"
|
||||
|
||||
|
||||
def _check_conflict(state_models: dict, role: str, model: str, matrix: Optional[dict] = None) -> Optional[str]:
|
||||
"""Check if the given model conflicts with already-filled roles.
|
||||
|
||||
state_models: dict of {role: model_name} from .state.models
|
||||
role: the role being entered (e.g. 'code_review')
|
||||
model: the model name being assigned
|
||||
matrix: conflict matrix (defaults to CONFLICT_MATRIX)
|
||||
|
||||
Returns the name of the conflicting role, or None if no conflict.
|
||||
"""
|
||||
if matrix is None:
|
||||
matrix = CONFLICT_MATRIX
|
||||
if role not in matrix:
|
||||
return None
|
||||
conflicting_roles = matrix[role]
|
||||
for filled_role, filled_model in state_models.items():
|
||||
if filled_model == model and filled_role in conflicting_roles:
|
||||
return filled_role
|
||||
return None
|
||||
|
||||
|
||||
def _read_state_models(task_path: Path) -> dict:
|
||||
"""Read .state.models from the task directory. Returns {} if missing."""
|
||||
f = task_path / ".state.models"
|
||||
if not f.exists():
|
||||
return {}
|
||||
try:
|
||||
data = json.loads(f.read_text())
|
||||
if isinstance(data, dict):
|
||||
return data
|
||||
except (OSError, json.JSONDecodeError):
|
||||
pass
|
||||
return {}
|
||||
|
||||
|
||||
def _write_state_models(task_path: Path, state_models: dict) -> None:
|
||||
"""Write .state.models atomically."""
|
||||
tmp = task_path / ".state.models.tmp"
|
||||
tmp.write_text(json.dumps(state_models, indent=2, sort_keys=True) + "\n")
|
||||
tmp.replace(task_path / ".state.models")
|
||||
|
||||
|
||||
def _model_divergence_violations(project: Optional[str] = None) -> list[dict]:
|
||||
"""Scan all tasks for model-divergence violations.
|
||||
|
||||
Returns list of violation dicts:
|
||||
{"task": str, "message": str, "severity": "high", "resolved": False}
|
||||
"""
|
||||
manifest = _load_models_manifest(project)
|
||||
mode = _get_model_mode(manifest)
|
||||
if mode == "single":
|
||||
return []
|
||||
violations = []
|
||||
tasks = _all_task_dirs(project)
|
||||
for name, path in tasks:
|
||||
sm = _read_state_models(path)
|
||||
if not sm:
|
||||
continue
|
||||
for role, model in sm.items():
|
||||
if model is None:
|
||||
continue
|
||||
# D8: doc_review, code_review, bug_find have no cross-conflicts
|
||||
# with each other; only conflicts documented in CONFLICT_MATRIX apply.
|
||||
conflict = _check_conflict(sm, role, str(model))
|
||||
if conflict:
|
||||
violations.append({
|
||||
"task": name,
|
||||
"severity": "high",
|
||||
"message": f"model-divergence: role '{role}' uses model '{model}' "
|
||||
f"which conflicts with role '{conflict}' (same model)",
|
||||
"resolved": False,
|
||||
})
|
||||
return violations
|
||||
|
||||
|
||||
# --- Command implementations ---
|
||||
|
||||
def cmd_show_task(args):
|
||||
@@ -593,6 +723,60 @@ def cmd_transition(args):
|
||||
return 1
|
||||
if current == "human_intervention" and target == "complete":
|
||||
_auto_update_verdict_on_complete(task_path)
|
||||
# Model-divergence enforcement (Subtask 2)
|
||||
# When entering a phase that maps to a role, record the model
|
||||
target_base = _base_phase(target)
|
||||
ROLE_PHASES = {"implement", "code_review", "bug_find", "adversarial_bug_find", "doc_review", "referee"}
|
||||
if target_base in ROLE_PHASES and current != target:
|
||||
manifest = _load_models_manifest(args.project)
|
||||
mode = _get_model_mode(manifest)
|
||||
models_list = (manifest or {}).get("models") or []
|
||||
model_names = [m["name"] for m in models_list if isinstance(m, dict) and m.get("name")]
|
||||
state_models = _read_state_models(task_path)
|
||||
model_arg = getattr(args, "model", None)
|
||||
|
||||
if model_arg:
|
||||
# --model explicitly provided — record advisory in single mode, check in multi
|
||||
state_models[target_base] = model_arg
|
||||
if mode == "multi-llm":
|
||||
if model_arg not in model_names:
|
||||
print(f"ERROR: Model '{model_arg}' is not in models.json. Available: {', '.join(model_names)}")
|
||||
return 1
|
||||
conflict = _check_conflict(state_models, target_base, model_arg)
|
||||
if conflict:
|
||||
# Remove the entry we just added
|
||||
del state_models[target_base]
|
||||
print(f"ERROR: Model '{model_arg}' assigned to role '{target_base}' conflicts with "
|
||||
f"role '{conflict}' which already uses the same model. "
|
||||
f"Use --model <different-model> to specify a different model.")
|
||||
return 1
|
||||
elif mode == "multi-llm":
|
||||
# Auto-assign: try default, then next-available non-conflicting
|
||||
default = (manifest or {}).get("default")
|
||||
assigned = False
|
||||
if default and default in model_names:
|
||||
conflict = _check_conflict(state_models, target_base, default)
|
||||
if not conflict:
|
||||
state_models[target_base] = default
|
||||
assigned = True
|
||||
if not assigned:
|
||||
for m_name in model_names:
|
||||
if m_name == default:
|
||||
continue
|
||||
conflict = _check_conflict(state_models, target_base, m_name)
|
||||
if not conflict:
|
||||
state_models[target_base] = m_name
|
||||
assigned = True
|
||||
break
|
||||
if not assigned:
|
||||
print(f"ERROR: Cannot auto-assign a model for role '{target_base}'. "
|
||||
f"All available models conflict with already-filled roles. "
|
||||
f"Use --model <name> to override.")
|
||||
return 1
|
||||
|
||||
if model_arg or mode == "multi-llm":
|
||||
_write_state_models(task_path, state_models)
|
||||
|
||||
if current == "implement" and target == "code_review":
|
||||
lock_file = task_path / ".state.lock"
|
||||
if lock_file.exists():
|
||||
@@ -980,6 +1164,14 @@ def _audit_collect(args):
|
||||
"halt_reason": lhalt, "current_task": ltask,
|
||||
"violation": is_violation, "message": msg})
|
||||
|
||||
# Model-divergence violations (Category 6)
|
||||
for mv in _model_divergence_violations(args.project):
|
||||
violations.append({
|
||||
"category": 6, "severity": mv["severity"],
|
||||
"task": mv["task"], "message": mv["message"],
|
||||
"resolved": False,
|
||||
})
|
||||
|
||||
return {"violations": violations,
|
||||
"loops": loops,
|
||||
"total_tasks": len(tasks),
|
||||
@@ -1118,7 +1310,16 @@ def cmd_audit(args):
|
||||
if stuck_found == 0:
|
||||
print(f"[PASS] No stuck tasks (threshold: {stuck_threshold} min)")
|
||||
|
||||
print("\n=== Category 6: Loops ===")
|
||||
print("\n=== Category 6: Model-Divergence Violations ===")
|
||||
md_violations = _model_divergence_violations(args.project)
|
||||
if md_violations:
|
||||
for v in md_violations:
|
||||
print(f"[FAIL] {v['task']}: {v['message']}")
|
||||
violations += 1
|
||||
else:
|
||||
print("[PASS] No model-divergence violations found")
|
||||
|
||||
print("\n=== Category 7: Loops ===")
|
||||
violations += _audit_loops_block(args)
|
||||
|
||||
print(f"\n=== Summary ===")
|
||||
@@ -1435,6 +1636,26 @@ def cmd_claim(args):
|
||||
if implementer == args.agent:
|
||||
print(f"ERROR: Agent '{args.agent}' implemented this task and cannot claim the code_review phase. Reviewer must be different from implementer.")
|
||||
return 1
|
||||
|
||||
# Model-divergence check on claim (Subtask 2, multi-LLM only)
|
||||
model_arg = getattr(args, "model", None)
|
||||
if model_arg:
|
||||
manifest = _load_models_manifest(args.project)
|
||||
mode = _get_model_mode(manifest)
|
||||
if mode == "multi-llm":
|
||||
models_list = (manifest or {}).get("models") or []
|
||||
model_names = [m["name"] for m in models_list if isinstance(m, dict) and m.get("name")]
|
||||
if model_arg not in model_names:
|
||||
print(f"ERROR: Model '{model_arg}' is not in models.json. Available: {', '.join(model_names)}")
|
||||
return 2
|
||||
state_models = _read_state_models(task_path)
|
||||
conflict = _check_conflict(state_models, base, model_arg)
|
||||
if conflict:
|
||||
print(f"ERROR: Model '{model_arg}' for role '{base}' conflicts with "
|
||||
f"role '{conflict}' which already uses the same model. "
|
||||
f"Use --model <different-model>.")
|
||||
return 1
|
||||
|
||||
lock_file = task_path / ".state.lock"
|
||||
timeout_sec = _lock_timeout_seconds(args.project)
|
||||
if lock_file.exists():
|
||||
@@ -2600,6 +2821,42 @@ def _gate_worktree_drift(state: dict, cfg: dict, project: Optional[str]) -> Opti
|
||||
return None
|
||||
|
||||
|
||||
def _gate_model_divergence(state: dict, cfg: dict, project: Optional[str]) -> Optional[dict]:
|
||||
"""Model-divergence brake: in multi-LLM mode, verify and implement
|
||||
roles must use different models. This prevents same-model verification
|
||||
(rubber-stamping) within a loop tick."""
|
||||
manifest = _load_models_manifest(project)
|
||||
mode = _get_model_mode(manifest)
|
||||
if mode != "multi-llm":
|
||||
return None
|
||||
roles = cfg.get("roles") or {}
|
||||
impl_model = None
|
||||
verify_model = None
|
||||
impl_cfg = roles.get("implement") or {}
|
||||
verify_cfg = roles.get("verify") or {}
|
||||
impl_model = impl_cfg.get("model")
|
||||
verify_model = verify_cfg.get("model")
|
||||
# Fall back to manifest default if role has no explicit model
|
||||
if not impl_model or not verify_model:
|
||||
default = (manifest or {}).get("default")
|
||||
if not impl_model:
|
||||
impl_model = default
|
||||
if not verify_model:
|
||||
verify_model = default
|
||||
if impl_model and verify_model and impl_model == verify_model:
|
||||
return {
|
||||
"ok": False,
|
||||
"reason": "halted:model_conflict",
|
||||
"halt_reason": "human_intervention",
|
||||
"remaining_iterations": None,
|
||||
"remaining_budget_usd": None,
|
||||
"task_phase": None,
|
||||
"task_in_halt_loop": True,
|
||||
"out_of_scope_files": [],
|
||||
}
|
||||
return None
|
||||
|
||||
|
||||
def _gate_score_plateau(state: dict, cfg: dict) -> Optional[dict]:
|
||||
window = int(cfg.get("brakes", {}).get("score_plateau_window", 0))
|
||||
if window <= 0:
|
||||
@@ -2648,6 +2905,7 @@ def cmd_check_gate(args) -> int:
|
||||
_gate_task_phase(state, cfg, args.project),
|
||||
_gate_worktree_drift(state, cfg, args.project),
|
||||
_gate_score_plateau(state, cfg),
|
||||
_gate_model_divergence(state, cfg, args.project),
|
||||
]
|
||||
failure = next((g for g in gates if g is not None), None)
|
||||
if failure is None:
|
||||
@@ -2864,6 +3122,7 @@ def main():
|
||||
parser.add_argument("--days", type=int, help="Cleanup age threshold in days (default 7, used with --cleanup-done / --install-cleanup-schedule)")
|
||||
parser.add_argument("--dry-run", action="store_true", help="With --cleanup-done, list candidates without moving them")
|
||||
parser.add_argument("--version", action="store_true", help="Print framework version and exit")
|
||||
parser.add_argument("--model", metavar="NAME", help="Model name for model-divergence enforcement (used with --transition, --claim)")
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user