feat(model-divergence): full enforcement — manifest, transition, claim, audit, loop gates, detect script

Completes all 3 model-divergence enforcement subtasks:

- scripts/detect_models.py: probes opencode.json + localhost endpoints,
  builds models.json with --json/--write/--force
- scripts/status.py: CONFLICT_MATRIX, --model flag, --transition --model,
  --claim --model, --audit Category 6, model-divergence brake gate in
  --check-gate, helpers for manifest loading and conflict checking
- scripts/loop-runner.py: _role_model() helper + {model} passed via extras
  dict to _invoke_harness for implement, verify, orchestrate roles
- tests/test_model_divergence.py: 33 tests covering all enforcement layers
- Single-LLM mode: record model advisory, no conflict check
- Multi-LLM mode (2+ models): conflict matrix enforced at transition, claim,
  and loop brake gate
- Project-level models.json preferred over global ~/.automaton/models.json
This commit is contained in:
Lap Tran
2026-06-26 13:23:17 -04:00
parent bc7daf8590
commit 35e449b03e
648 changed files with 1120 additions and 15492 deletions
+261 -2
View File
@@ -153,7 +153,19 @@ FORBIDDEN_ARTIFACTS = {
}
NON_ARTIFACT_FILES = {".state", ".state.tmp", ".state.lock", ".state.approvals",
".state.implementer", ".state.lastedit", "VRAM_CONFIG.md", "PARENT_SPEC.md", "REVIEW.md"}
".state.implementer", ".state.lastedit", ".state.models",
"VRAM_CONFIG.md", "PARENT_SPEC.md", "REVIEW.md"}
# Model-divergence enforcement
CONFLICT_MATRIX = {
"code_review": {"implement"},
"bug_find": {"implement"},
"adversarial_bug_find": {"implement", "bug_find"},
"referee": {"implement", "bug_find", "adversarial_bug_find"},
"loop-verify": {"loop-implement"},
}
MODELS_JSON_FILE = "models.json"
PHASE_PRIORITY = {
"referee": 12, "doc_review": 11, "adversarial_bug_find": 10,
@@ -438,6 +450,124 @@ def _lock_timeout_seconds(project: Optional[str] = None) -> int:
return val
# ---------------------------------------------------------------------------
# Model-divergence enforcement helpers
# ---------------------------------------------------------------------------
def _load_models_manifest(project: Optional[str] = None) -> Optional[dict]:
"""Load the models.json manifest for the given project.
Searches:
1. project/.automaton/models.json
2. ~/.automaton/models.json (fallback)
Returns None if no models.json exists (single-LLM mode, backward compatible).
"""
project_dir = _find_project_dir(project)
candidates = [
project_dir / ".automaton" / MODELS_JSON_FILE,
AUTOMATON_DIR / MODELS_JSON_FILE,
]
for path in candidates:
if path.exists():
try:
return json.loads(path.read_text())
except (OSError, json.JSONDecodeError):
return None
return None
def _get_model_mode(manifest: Optional[dict]) -> str:
"""Determine the model mode: 'single' or 'multi-llm'.
- Missing manifest → single-LLM (backward compatible)
- 0-1 models → single-LLM
- 2+ models → multi-LLM
"""
if manifest is None:
return "single"
models = manifest.get("models") or []
if len(models) >= 2:
return "multi-llm"
return "single"
def _check_conflict(state_models: dict, role: str, model: str, matrix: Optional[dict] = None) -> Optional[str]:
"""Check if the given model conflicts with already-filled roles.
state_models: dict of {role: model_name} from .state.models
role: the role being entered (e.g. 'code_review')
model: the model name being assigned
matrix: conflict matrix (defaults to CONFLICT_MATRIX)
Returns the name of the conflicting role, or None if no conflict.
"""
if matrix is None:
matrix = CONFLICT_MATRIX
if role not in matrix:
return None
conflicting_roles = matrix[role]
for filled_role, filled_model in state_models.items():
if filled_model == model and filled_role in conflicting_roles:
return filled_role
return None
def _read_state_models(task_path: Path) -> dict:
"""Read .state.models from the task directory. Returns {} if missing."""
f = task_path / ".state.models"
if not f.exists():
return {}
try:
data = json.loads(f.read_text())
if isinstance(data, dict):
return data
except (OSError, json.JSONDecodeError):
pass
return {}
def _write_state_models(task_path: Path, state_models: dict) -> None:
"""Write .state.models atomically."""
tmp = task_path / ".state.models.tmp"
tmp.write_text(json.dumps(state_models, indent=2, sort_keys=True) + "\n")
tmp.replace(task_path / ".state.models")
def _model_divergence_violations(project: Optional[str] = None) -> list[dict]:
"""Scan all tasks for model-divergence violations.
Returns list of violation dicts:
{"task": str, "message": str, "severity": "high", "resolved": False}
"""
manifest = _load_models_manifest(project)
mode = _get_model_mode(manifest)
if mode == "single":
return []
violations = []
tasks = _all_task_dirs(project)
for name, path in tasks:
sm = _read_state_models(path)
if not sm:
continue
for role, model in sm.items():
if model is None:
continue
# D8: doc_review, code_review, bug_find have no cross-conflicts
# with each other; only conflicts documented in CONFLICT_MATRIX apply.
conflict = _check_conflict(sm, role, str(model))
if conflict:
violations.append({
"task": name,
"severity": "high",
"message": f"model-divergence: role '{role}' uses model '{model}' "
f"which conflicts with role '{conflict}' (same model)",
"resolved": False,
})
return violations
# --- Command implementations ---
def cmd_show_task(args):
@@ -593,6 +723,60 @@ def cmd_transition(args):
return 1
if current == "human_intervention" and target == "complete":
_auto_update_verdict_on_complete(task_path)
# Model-divergence enforcement (Subtask 2)
# When entering a phase that maps to a role, record the model
target_base = _base_phase(target)
ROLE_PHASES = {"implement", "code_review", "bug_find", "adversarial_bug_find", "doc_review", "referee"}
if target_base in ROLE_PHASES and current != target:
manifest = _load_models_manifest(args.project)
mode = _get_model_mode(manifest)
models_list = (manifest or {}).get("models") or []
model_names = [m["name"] for m in models_list if isinstance(m, dict) and m.get("name")]
state_models = _read_state_models(task_path)
model_arg = getattr(args, "model", None)
if model_arg:
# --model explicitly provided — record advisory in single mode, check in multi
state_models[target_base] = model_arg
if mode == "multi-llm":
if model_arg not in model_names:
print(f"ERROR: Model '{model_arg}' is not in models.json. Available: {', '.join(model_names)}")
return 1
conflict = _check_conflict(state_models, target_base, model_arg)
if conflict:
# Remove the entry we just added
del state_models[target_base]
print(f"ERROR: Model '{model_arg}' assigned to role '{target_base}' conflicts with "
f"role '{conflict}' which already uses the same model. "
f"Use --model <different-model> to specify a different model.")
return 1
elif mode == "multi-llm":
# Auto-assign: try default, then next-available non-conflicting
default = (manifest or {}).get("default")
assigned = False
if default and default in model_names:
conflict = _check_conflict(state_models, target_base, default)
if not conflict:
state_models[target_base] = default
assigned = True
if not assigned:
for m_name in model_names:
if m_name == default:
continue
conflict = _check_conflict(state_models, target_base, m_name)
if not conflict:
state_models[target_base] = m_name
assigned = True
break
if not assigned:
print(f"ERROR: Cannot auto-assign a model for role '{target_base}'. "
f"All available models conflict with already-filled roles. "
f"Use --model <name> to override.")
return 1
if model_arg or mode == "multi-llm":
_write_state_models(task_path, state_models)
if current == "implement" and target == "code_review":
lock_file = task_path / ".state.lock"
if lock_file.exists():
@@ -980,6 +1164,14 @@ def _audit_collect(args):
"halt_reason": lhalt, "current_task": ltask,
"violation": is_violation, "message": msg})
# Model-divergence violations (Category 6)
for mv in _model_divergence_violations(args.project):
violations.append({
"category": 6, "severity": mv["severity"],
"task": mv["task"], "message": mv["message"],
"resolved": False,
})
return {"violations": violations,
"loops": loops,
"total_tasks": len(tasks),
@@ -1118,7 +1310,16 @@ def cmd_audit(args):
if stuck_found == 0:
print(f"[PASS] No stuck tasks (threshold: {stuck_threshold} min)")
print("\n=== Category 6: Loops ===")
print("\n=== Category 6: Model-Divergence Violations ===")
md_violations = _model_divergence_violations(args.project)
if md_violations:
for v in md_violations:
print(f"[FAIL] {v['task']}: {v['message']}")
violations += 1
else:
print("[PASS] No model-divergence violations found")
print("\n=== Category 7: Loops ===")
violations += _audit_loops_block(args)
print(f"\n=== Summary ===")
@@ -1435,6 +1636,26 @@ def cmd_claim(args):
if implementer == args.agent:
print(f"ERROR: Agent '{args.agent}' implemented this task and cannot claim the code_review phase. Reviewer must be different from implementer.")
return 1
# Model-divergence check on claim (Subtask 2, multi-LLM only)
model_arg = getattr(args, "model", None)
if model_arg:
manifest = _load_models_manifest(args.project)
mode = _get_model_mode(manifest)
if mode == "multi-llm":
models_list = (manifest or {}).get("models") or []
model_names = [m["name"] for m in models_list if isinstance(m, dict) and m.get("name")]
if model_arg not in model_names:
print(f"ERROR: Model '{model_arg}' is not in models.json. Available: {', '.join(model_names)}")
return 2
state_models = _read_state_models(task_path)
conflict = _check_conflict(state_models, base, model_arg)
if conflict:
print(f"ERROR: Model '{model_arg}' for role '{base}' conflicts with "
f"role '{conflict}' which already uses the same model. "
f"Use --model <different-model>.")
return 1
lock_file = task_path / ".state.lock"
timeout_sec = _lock_timeout_seconds(args.project)
if lock_file.exists():
@@ -2600,6 +2821,42 @@ def _gate_worktree_drift(state: dict, cfg: dict, project: Optional[str]) -> Opti
return None
def _gate_model_divergence(state: dict, cfg: dict, project: Optional[str]) -> Optional[dict]:
"""Model-divergence brake: in multi-LLM mode, verify and implement
roles must use different models. This prevents same-model verification
(rubber-stamping) within a loop tick."""
manifest = _load_models_manifest(project)
mode = _get_model_mode(manifest)
if mode != "multi-llm":
return None
roles = cfg.get("roles") or {}
impl_model = None
verify_model = None
impl_cfg = roles.get("implement") or {}
verify_cfg = roles.get("verify") or {}
impl_model = impl_cfg.get("model")
verify_model = verify_cfg.get("model")
# Fall back to manifest default if role has no explicit model
if not impl_model or not verify_model:
default = (manifest or {}).get("default")
if not impl_model:
impl_model = default
if not verify_model:
verify_model = default
if impl_model and verify_model and impl_model == verify_model:
return {
"ok": False,
"reason": "halted:model_conflict",
"halt_reason": "human_intervention",
"remaining_iterations": None,
"remaining_budget_usd": None,
"task_phase": None,
"task_in_halt_loop": True,
"out_of_scope_files": [],
}
return None
def _gate_score_plateau(state: dict, cfg: dict) -> Optional[dict]:
window = int(cfg.get("brakes", {}).get("score_plateau_window", 0))
if window <= 0:
@@ -2648,6 +2905,7 @@ def cmd_check_gate(args) -> int:
_gate_task_phase(state, cfg, args.project),
_gate_worktree_drift(state, cfg, args.project),
_gate_score_plateau(state, cfg),
_gate_model_divergence(state, cfg, args.project),
]
failure = next((g for g in gates if g is not None), None)
if failure is None:
@@ -2864,6 +3122,7 @@ def main():
parser.add_argument("--days", type=int, help="Cleanup age threshold in days (default 7, used with --cleanup-done / --install-cleanup-schedule)")
parser.add_argument("--dry-run", action="store_true", help="With --cleanup-done, list candidates without moving them")
parser.add_argument("--version", action="store_true", help="Print framework version and exit")
parser.add_argument("--model", metavar="NAME", help="Model name for model-divergence enforcement (used with --transition, --claim)")
args = parser.parse_args()