Files
automaton/tests/test_context_sizing.py
T

264 lines
11 KiB
Python

"""Tests for Tier 1 context-sizing fixes (task fix-context-sizing).
Covers R1 (single headroom), R2 (honest quotients), R3 (--loop-mode refuse),
R4 (JSON fields), R6 (decompose.md tier acknowledgment).
R5 (config.md section) is a documentation requirement verified by a substring
check; R3's user-override-is-authoritative path is covered by reading
_parse_config_model + Override context window.
Run: python3 -m pytest tests/test_context_sizing.py -v
"""
from __future__ import annotations
import json
import subprocess
import sys
from pathlib import Path
import pytest
FRAMEWORK_DIR = Path.home() / ".automaton"
VRAM_SCRIPT = FRAMEWORK_DIR / "scripts" / "vram_detect.py"
DECOMPOSE_MD = FRAMEWORK_DIR / "prompts" / "decompose.md"
CONFIG_MD = FRAMEWORK_DIR / "config.md"
def _run_vram_detect(*args: str) -> tuple[int, str]:
"""Run vram_detect.py with args. Returns (exit_code, stdout)."""
cmd = [sys.executable, str(VRAM_SCRIPT), *args]
result = subprocess.run(cmd, capture_output=True, text=True, timeout=30, check=False)
return result.returncode, result.stdout
def _parse_json_block(stdout: str) -> dict:
"""Extract the JSON block from vram_detect.py stdout (after === JSON Output ===)."""
marker = "=== JSON Output ==="
idx = stdout.find(marker)
assert idx >= 0, "no JSON output marker found"
rest = stdout[idx + len(marker):].strip()
return json.loads(rest)
# ---------------------------------------------------------------------------
# R1: headroom applied exactly once
# ---------------------------------------------------------------------------
def test_recommend_context_single_headroom():
"""Headroom is applied EXACTLY ONCE to derive max_peak_kb.
Regression: previously headroom was applied three times (once per
budget-construction site, once at the max_peak step), so a 25% headroom
acted as ~44% reduction. Now: recommended_kb is net of overhead, pre-headroom;
max_peak_kb is post-headroom.
"""
sys.path.insert(0, str(FRAMEWORK_DIR / "scripts"))
try:
import importlib
import vram_detect
importlib.reload(vram_detect)
config = {"auto_detect": True, "headroom_pct": 25, "target_context_kb": 0, "max_peak_kb": 0}
# GPU branch: 16GB VRAM -> 32000 tokens raw budget.
headroom_pct, recommended_kb, max_peak_kb = vram_detect.recommend_context(
gpu_vram_gb=16, ram_gb=64, model_context_kb=0,
overhead_tokens=0, config=config,
)
assert headroom_pct == 25
# recommended_kb is the raw budget (no headroom applied here).
assert recommended_kb == 16 * 2000
# max_peak_kb = recommended_kb * (100-25)/100 = 24000.
assert max_peak_kb == (16 * 2000) * 75 // 100
# Pre-fix formula would've produced 16*2000 * 0.75 * 0.75 = 18000.
assert max_peak_kb != (16 * 2000) * 75 // 100 * 75 // 100
finally:
sys.path.pop(0)
def test_recommend_context_overhead_subtracted_before_headroom():
"""net_kb subtracts overhead before headroom is applied; the test guards
against the old `max(0, ...)` clamp that hid negatives."""
sys.path.insert(0, str(FRAMEWORK_DIR / "scripts"))
try:
import importlib
import vram_detect
importlib.reload(vram_detect)
config = {"auto_detect": True, "headroom_pct": 25, "target_context_kb": 0, "max_peak_kb": 0}
# 16k VRAM (32000 tokens) with 40000 tokens overhead -> net = -8000.
headroom_pct, recommended_kb, max_peak_kb = vram_detect.recommend_context(
gpu_vram_gb=16, ram_gb=64, model_context_kb=0,
overhead_tokens=40000, config=config,
)
# Honest arithmetic. No clamp.
assert recommended_kb == 32000 - 40000 # -8000
assert max_peak_kb == -8000 * 75 // 100 # -6000
finally:
sys.path.pop(0)
def test_recommend_context_manual_override_applies_headroom_once():
"""Manual override path already applies headroom once; ensure the rewrite
preserves that semantics."""
sys.path.insert(0, str(FRAMEWORK_DIR / "scripts"))
try:
import importlib
import vram_detect
importlib.reload(vram_detect)
config = {"auto_detect": False, "headroom_pct": 25, "target_context_kb": 50000, "max_peak_kb": 0}
headroom_pct, recommended_kb, max_peak_kb = vram_detect.recommend_context(
gpu_vram_gb=100, ram_gb=100, model_context_kb=100000,
overhead_tokens=99999, config=config,
)
assert headroom_pct == 25
assert recommended_kb == 50000 # target unchanged
assert max_peak_kb == 50000 * 75 // 100 # headroom exactly once
finally:
sys.path.pop(0)
# ---------------------------------------------------------------------------
# R2: no fake 8k/6k fallbacks
# ---------------------------------------------------------------------------
def test_no_fake_defaults_when_budget_zero():
"""If model is unknown AND VRAM/RAM detection both return zero (impossible
in real CI but exercisable by passing an unknown model in non-loop mode),
the recommended_k / max_peak_k reported in JSON should reflect the real
math (not 8 / 6 fabricated defaults)."""
# Use a model name that will not match MODEL_CONTEXT_WINDOWS.
exit_code, stdout = _run_vram_detect("--model", "zzz-not-a-real-model-xyz")
assert exit_code == 0
payload = _parse_json_block(stdout)
# No fabricated 8 / 6. The real quotient (may be large if VRAM is non-zero,
# but the test asserts that the field equals recommended_kb // 1000, not 8).
assert payload["recommended_k"] == payload["recommended_kb"] // 1000
assert payload["max_peak_context_kb"] // 1000 == payload["max_peak_context_kb"] // 1000 # equality sanity
def test_no_max_zero_clamp_in_output():
"""Negative recommended_kb is reported honestly. We can't force a negative
in real CI, but we verify the output never contains the old clamp markers:
the function should never silently turn negative into 0."""
# The strict assertion is in test_recommend_context_overhead_subtracted_before_headroom.
# Here we just confirm a normal run's JSON doesn't echo an 'else 8' output
# when the budget is positive (the path we exercise).
exit_code, stdout = _run_vram_detect("--model", "gpt-4o")
assert exit_code == 0
payload = _parse_json_block(stdout)
# The recommended_k must equal the quotient, not a fabricated fallback.
assert payload["recommended_k"] == payload["recommended_kb"] // 1000
# ---------------------------------------------------------------------------
# R3: --loop-mode refuse paths
# ---------------------------------------------------------------------------
def test_loop_mode_refuses_unknown_model():
"""--loop-mode on an unknown model exits 2 with a clear refuse message."""
exit_code, stdout = _run_vram_detect("--model", "zzz-unknown", "--loop-mode")
assert exit_code == 2
assert "model context window is unknown" in stdout.lower()
assert "loop-mode" in stdout.lower()
def test_loop_mode_refuses_sub_floor_budget():
"""--loop-mode tries hard to emulate a small context. We can't easily force
a sub-16k budget without mocking the whole detector, but we can check that
the refuse message is in the code path by exercising the unknown-model path
AND checking that a known model with low-context lookup would refuse if its
max_peak_kb < 16000.
Since real VRAM detection dominates (M5/32GB returns 43k), we instead
verify the LOOP_MODE_CONTEXT_FLOOR_KB constant equals 16000 — the gate is
structurally present and the refusal code is reachable via unknown-model."""
sys.path.insert(0, str(FRAMEWORK_DIR / "scripts"))
try:
import importlib
import vram_detect
importlib.reload(vram_detect)
assert vram_detect.LOOP_MODE_CONTEXT_FLOOR_KB == 16_000
finally:
sys.path.pop(0)
def test_loop_mode_passes_for_known_model():
"""A known model on this machine should pass --loop-mode (exit 0)."""
exit_code, stdout = _run_vram_detect("--model", "gpt-4o", "--loop-mode")
assert exit_code == 0
payload = _parse_json_block(stdout)
assert payload["loop_mode"] is True
assert payload["loop_mode_eligible"] is True
def test_non_loop_mode_does_not_refuse_unknown_model():
"""Non-loop callers keep prior behavior: unknown model just warns."""
exit_code, stdout = _run_vram_detect("--model", "zzz-unknown")
assert exit_code == 0 # warning only, no refuse
# ---------------------------------------------------------------------------
# R4: JSON fields present
# ---------------------------------------------------------------------------
def test_json_includes_available_context_kb_and_eligible():
exit_code, stdout = _run_vram_detect("--model", "gpt-4o")
assert exit_code == 0
payload = _parse_json_block(stdout)
assert "available_context_kb" in payload
assert "loop_mode_eligible" in payload
assert "loop_mode" in payload
assert payload["available_context_kb"] == payload["max_peak_context_kb"]
assert isinstance(payload["loop_mode_eligible"], bool)
# ---------------------------------------------------------------------------
# R5: config.md ## Loop Role Models section
# ---------------------------------------------------------------------------
def test_config_md_includes_loop_role_models_section():
text = CONFIG_MD.read_text()
assert "## Loop Role Models" in text
assert "Implement:" in text
assert "Verify:" in text
assert "Orchestrate:" in text
assert "D12" in text or "D13" in text # design-decision reference present
# ---------------------------------------------------------------------------
# R6: decompose.md includes 4k tier and 16k floor
# ---------------------------------------------------------------------------
def test_decompose_md_includes_4k_tier():
text = DECOMPOSE_MD.read_text()
# 4k must appear in both the peak-context guideline (bold marker) and
# the size-targets row (parenthetical marker, matching existing 8k/16k style).
assert "**4k VRAM**" in text
assert "(4k VRAM)" in text
def test_decompose_md_includes_16k_floor_refuse():
"""A non-negotiable '≤ 16k: refuse' line should now exist near the top
of the context budget guideline block."""
text = DECOMPOSE_MD.read_text()
assert "≤ 16k" in text
assert "REFUSE" in text.upper()
# ---------------------------------------------------------------------------
# Smoke test (no regression)
# ---------------------------------------------------------------------------
def test_vram_detect_compiles():
exit_code, _ = _run_vram_detect("--help")
assert exit_code == 0
def test_help_mentions_loop_mode():
_, stdout = _run_vram_detect("--help")
assert "--loop-mode" in stdout