264 lines
11 KiB
Python
264 lines
11 KiB
Python
"""Tests for Tier 1 context-sizing fixes (task fix-context-sizing).
|
|
|
|
Covers R1 (single headroom), R2 (honest quotients), R3 (--loop-mode refuse),
|
|
R4 (JSON fields), R6 (decompose.md tier acknowledgment).
|
|
|
|
R5 (config.md section) is a documentation requirement verified by a substring
|
|
check; R3's user-override-is-authoritative path is covered by reading
|
|
_parse_config_model + Override context window.
|
|
|
|
Run: python3 -m pytest tests/test_context_sizing.py -v
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import subprocess
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
FRAMEWORK_DIR = Path.home() / ".automaton"
|
|
VRAM_SCRIPT = FRAMEWORK_DIR / "scripts" / "vram_detect.py"
|
|
DECOMPOSE_MD = FRAMEWORK_DIR / "prompts" / "decompose.md"
|
|
CONFIG_MD = FRAMEWORK_DIR / "config.md"
|
|
|
|
|
|
def _run_vram_detect(*args: str) -> tuple[int, str]:
|
|
"""Run vram_detect.py with args. Returns (exit_code, stdout)."""
|
|
cmd = [sys.executable, str(VRAM_SCRIPT), *args]
|
|
result = subprocess.run(cmd, capture_output=True, text=True, timeout=30, check=False)
|
|
return result.returncode, result.stdout
|
|
|
|
|
|
def _parse_json_block(stdout: str) -> dict:
|
|
"""Extract the JSON block from vram_detect.py stdout (after === JSON Output ===)."""
|
|
marker = "=== JSON Output ==="
|
|
idx = stdout.find(marker)
|
|
assert idx >= 0, "no JSON output marker found"
|
|
rest = stdout[idx + len(marker):].strip()
|
|
return json.loads(rest)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R1: headroom applied exactly once
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_recommend_context_single_headroom():
|
|
"""Headroom is applied EXACTLY ONCE to derive max_peak_kb.
|
|
|
|
Regression: previously headroom was applied three times (once per
|
|
budget-construction site, once at the max_peak step), so a 25% headroom
|
|
acted as ~44% reduction. Now: recommended_kb is net of overhead, pre-headroom;
|
|
max_peak_kb is post-headroom.
|
|
"""
|
|
sys.path.insert(0, str(FRAMEWORK_DIR / "scripts"))
|
|
try:
|
|
import importlib
|
|
import vram_detect
|
|
importlib.reload(vram_detect)
|
|
config = {"auto_detect": True, "headroom_pct": 25, "target_context_kb": 0, "max_peak_kb": 0}
|
|
# GPU branch: 16GB VRAM -> 32000 tokens raw budget.
|
|
headroom_pct, recommended_kb, max_peak_kb = vram_detect.recommend_context(
|
|
gpu_vram_gb=16, ram_gb=64, model_context_kb=0,
|
|
overhead_tokens=0, config=config,
|
|
)
|
|
assert headroom_pct == 25
|
|
# recommended_kb is the raw budget (no headroom applied here).
|
|
assert recommended_kb == 16 * 2000
|
|
# max_peak_kb = recommended_kb * (100-25)/100 = 24000.
|
|
assert max_peak_kb == (16 * 2000) * 75 // 100
|
|
# Pre-fix formula would've produced 16*2000 * 0.75 * 0.75 = 18000.
|
|
assert max_peak_kb != (16 * 2000) * 75 // 100 * 75 // 100
|
|
finally:
|
|
sys.path.pop(0)
|
|
|
|
|
|
def test_recommend_context_overhead_subtracted_before_headroom():
|
|
"""net_kb subtracts overhead before headroom is applied; the test guards
|
|
against the old `max(0, ...)` clamp that hid negatives."""
|
|
sys.path.insert(0, str(FRAMEWORK_DIR / "scripts"))
|
|
try:
|
|
import importlib
|
|
import vram_detect
|
|
importlib.reload(vram_detect)
|
|
config = {"auto_detect": True, "headroom_pct": 25, "target_context_kb": 0, "max_peak_kb": 0}
|
|
# 16k VRAM (32000 tokens) with 40000 tokens overhead -> net = -8000.
|
|
headroom_pct, recommended_kb, max_peak_kb = vram_detect.recommend_context(
|
|
gpu_vram_gb=16, ram_gb=64, model_context_kb=0,
|
|
overhead_tokens=40000, config=config,
|
|
)
|
|
# Honest arithmetic. No clamp.
|
|
assert recommended_kb == 32000 - 40000 # -8000
|
|
assert max_peak_kb == -8000 * 75 // 100 # -6000
|
|
finally:
|
|
sys.path.pop(0)
|
|
|
|
|
|
def test_recommend_context_manual_override_applies_headroom_once():
|
|
"""Manual override path already applies headroom once; ensure the rewrite
|
|
preserves that semantics."""
|
|
sys.path.insert(0, str(FRAMEWORK_DIR / "scripts"))
|
|
try:
|
|
import importlib
|
|
import vram_detect
|
|
importlib.reload(vram_detect)
|
|
config = {"auto_detect": False, "headroom_pct": 25, "target_context_kb": 50000, "max_peak_kb": 0}
|
|
headroom_pct, recommended_kb, max_peak_kb = vram_detect.recommend_context(
|
|
gpu_vram_gb=100, ram_gb=100, model_context_kb=100000,
|
|
overhead_tokens=99999, config=config,
|
|
)
|
|
assert headroom_pct == 25
|
|
assert recommended_kb == 50000 # target unchanged
|
|
assert max_peak_kb == 50000 * 75 // 100 # headroom exactly once
|
|
finally:
|
|
sys.path.pop(0)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R2: no fake 8k/6k fallbacks
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_no_fake_defaults_when_budget_zero():
|
|
"""If model is unknown AND VRAM/RAM detection both return zero (impossible
|
|
in real CI but exercisable by passing an unknown model in non-loop mode),
|
|
the recommended_k / max_peak_k reported in JSON should reflect the real
|
|
math (not 8 / 6 fabricated defaults)."""
|
|
# Use a model name that will not match MODEL_CONTEXT_WINDOWS.
|
|
exit_code, stdout = _run_vram_detect("--model", "zzz-not-a-real-model-xyz")
|
|
assert exit_code == 0
|
|
payload = _parse_json_block(stdout)
|
|
# No fabricated 8 / 6. The real quotient (may be large if VRAM is non-zero,
|
|
# but the test asserts that the field equals recommended_kb // 1000, not 8).
|
|
assert payload["recommended_k"] == payload["recommended_kb"] // 1000
|
|
assert payload["max_peak_context_kb"] // 1000 == payload["max_peak_context_kb"] // 1000 # equality sanity
|
|
|
|
|
|
def test_no_max_zero_clamp_in_output():
|
|
"""Negative recommended_kb is reported honestly. We can't force a negative
|
|
in real CI, but we verify the output never contains the old clamp markers:
|
|
the function should never silently turn negative into 0."""
|
|
# The strict assertion is in test_recommend_context_overhead_subtracted_before_headroom.
|
|
# Here we just confirm a normal run's JSON doesn't echo an 'else 8' output
|
|
# when the budget is positive (the path we exercise).
|
|
exit_code, stdout = _run_vram_detect("--model", "gpt-4o")
|
|
assert exit_code == 0
|
|
payload = _parse_json_block(stdout)
|
|
# The recommended_k must equal the quotient, not a fabricated fallback.
|
|
assert payload["recommended_k"] == payload["recommended_kb"] // 1000
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R3: --loop-mode refuse paths
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_loop_mode_refuses_unknown_model():
|
|
"""--loop-mode on an unknown model exits 2 with a clear refuse message."""
|
|
exit_code, stdout = _run_vram_detect("--model", "zzz-unknown", "--loop-mode")
|
|
assert exit_code == 2
|
|
assert "model context window is unknown" in stdout.lower()
|
|
assert "loop-mode" in stdout.lower()
|
|
|
|
|
|
def test_loop_mode_refuses_sub_floor_budget():
|
|
"""--loop-mode tries hard to emulate a small context. We can't easily force
|
|
a sub-16k budget without mocking the whole detector, but we can check that
|
|
the refuse message is in the code path by exercising the unknown-model path
|
|
AND checking that a known model with low-context lookup would refuse if its
|
|
max_peak_kb < 16000.
|
|
|
|
Since real VRAM detection dominates (M5/32GB returns 43k), we instead
|
|
verify the LOOP_MODE_CONTEXT_FLOOR_KB constant equals 16000 — the gate is
|
|
structurally present and the refusal code is reachable via unknown-model."""
|
|
sys.path.insert(0, str(FRAMEWORK_DIR / "scripts"))
|
|
try:
|
|
import importlib
|
|
import vram_detect
|
|
importlib.reload(vram_detect)
|
|
assert vram_detect.LOOP_MODE_CONTEXT_FLOOR_KB == 16_000
|
|
finally:
|
|
sys.path.pop(0)
|
|
|
|
|
|
def test_loop_mode_passes_for_known_model():
|
|
"""A known model on this machine should pass --loop-mode (exit 0)."""
|
|
exit_code, stdout = _run_vram_detect("--model", "gpt-4o", "--loop-mode")
|
|
assert exit_code == 0
|
|
payload = _parse_json_block(stdout)
|
|
assert payload["loop_mode"] is True
|
|
assert payload["loop_mode_eligible"] is True
|
|
|
|
|
|
def test_non_loop_mode_does_not_refuse_unknown_model():
|
|
"""Non-loop callers keep prior behavior: unknown model just warns."""
|
|
exit_code, stdout = _run_vram_detect("--model", "zzz-unknown")
|
|
assert exit_code == 0 # warning only, no refuse
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R4: JSON fields present
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_json_includes_available_context_kb_and_eligible():
|
|
exit_code, stdout = _run_vram_detect("--model", "gpt-4o")
|
|
assert exit_code == 0
|
|
payload = _parse_json_block(stdout)
|
|
assert "available_context_kb" in payload
|
|
assert "loop_mode_eligible" in payload
|
|
assert "loop_mode" in payload
|
|
assert payload["available_context_kb"] == payload["max_peak_context_kb"]
|
|
assert isinstance(payload["loop_mode_eligible"], bool)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R5: config.md ## Loop Role Models section
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_config_md_includes_loop_role_models_section():
|
|
text = CONFIG_MD.read_text()
|
|
assert "## Loop Role Models" in text
|
|
assert "Implement:" in text
|
|
assert "Verify:" in text
|
|
assert "Orchestrate:" in text
|
|
assert "D12" in text or "D13" in text # design-decision reference present
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R6: decompose.md includes 4k tier and 16k floor
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_decompose_md_includes_4k_tier():
|
|
text = DECOMPOSE_MD.read_text()
|
|
# 4k must appear in both the peak-context guideline (bold marker) and
|
|
# the size-targets row (parenthetical marker, matching existing 8k/16k style).
|
|
assert "**4k VRAM**" in text
|
|
assert "(4k VRAM)" in text
|
|
|
|
|
|
def test_decompose_md_includes_16k_floor_refuse():
|
|
"""A non-negotiable '≤ 16k: refuse' line should now exist near the top
|
|
of the context budget guideline block."""
|
|
text = DECOMPOSE_MD.read_text()
|
|
assert "≤ 16k" in text
|
|
assert "REFUSE" in text.upper()
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Smoke test (no regression)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def test_vram_detect_compiles():
|
|
exit_code, _ = _run_vram_detect("--help")
|
|
assert exit_code == 0
|
|
|
|
|
|
def test_help_mentions_loop_mode():
|
|
_, stdout = _run_vram_detect("--help")
|
|
assert "--loop-mode" in stdout |