199 lines
7.7 KiB
Python
199 lines
7.7 KiB
Python
"""Pure-function tests for `parse_verdict` hardening (v1.1 task
|
|
`harden-parse-verdict`).
|
|
|
|
Covers: `pass` string coercion (`"true"`/`"false"` → bool, the O6 bug),
|
|
case-insensitive matching, fallback truthy semantics for other strings,
|
|
score clamping to `[0, 1]`, NaN handling, non-numeric score handling,
|
|
numeric-string score pass-through, fence-block path still works.
|
|
|
|
No subprocess, no fixtures, no monkeypatch. Just the function and
|
|
literal JSON strings.
|
|
"""
|
|
|
|
import importlib.util
|
|
import json
|
|
import math
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
_RUNNER_PATH = Path.home() / ".automaton" / "scripts" / "loop-runner.py"
|
|
_spec = importlib.util.spec_from_file_location("loop_runner", _RUNNER_PATH)
|
|
lr = importlib.util.module_from_spec(_spec)
|
|
_spec.loader.exec_module(lr)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Baseline — strict-JSON emitters (no behavior change)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestStrictBaseline:
|
|
def test_pass_true_bool(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": True, "score": 0.8}))
|
|
assert v is not None
|
|
assert v["pass"] is True
|
|
assert v["score"] == 0.8
|
|
|
|
def test_pass_false_bool(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": False, "score": 0.2}))
|
|
assert v is not None
|
|
assert v["pass"] is False
|
|
assert v["score"] == 0.2
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# O6 — `pass` as string coercion
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestPassStringCoercion:
|
|
def test_pass_true_string(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": "true", "score": 0.9}))
|
|
assert v is not None
|
|
# The O6 bug: bool("true") is True, but bool("false") is ALSO True
|
|
# (non-empty string is truthy). Verify our fix:
|
|
assert v["pass"] is True
|
|
|
|
def test_pass_false_string(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": "false", "score": 0.1}))
|
|
assert v is not None
|
|
# The O6 fix: "false" string → False (not the old bool("false")=True)
|
|
assert v["pass"] is False
|
|
|
|
def test_pass_string_case_insensitive(self):
|
|
for s_true in ("TRUE", "True", "tRuE"):
|
|
v = lr.parse_verdict(json.dumps({"pass": s_true, "score": 0.5}))
|
|
assert v["pass"] is True, f"failed for {s_true!r}"
|
|
for s_false in ("FALSE", "False", "fAlSe"):
|
|
v = lr.parse_verdict(json.dumps({"pass": s_false, "score": 0.5}))
|
|
assert v["pass"] is False, f"failed for {s_false!r}"
|
|
|
|
def test_pass_with_surrounding_whitespace(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": " true ", "score": 0.5}))
|
|
assert v["pass"] is True
|
|
v = lr.parse_verdict(json.dumps({"pass": " false ", "score": 0.5}))
|
|
assert v["pass"] is False
|
|
|
|
def test_empty_pass_string_is_false(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": "", "score": 0.5}))
|
|
assert v["pass"] is False # empty -> bool("") -> False (existing semantics)
|
|
|
|
def test_other_truthy_string_pass(self):
|
|
# Backwards compat: a string like "yes" falls through to bool("yes")
|
|
# which is True (non-empty string is truthy). Was the pre-fix
|
|
# behavior; we preserve it for non-true/non-false strings.
|
|
v = lr.parse_verdict(json.dumps({"pass": "yes", "score": 0.5}))
|
|
assert v["pass"] is True
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# R2 / R3 — score clamping and defensive numeric handling
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestScoreClamping:
|
|
def test_score_clamped_high(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": True, "score": 1.5}))
|
|
assert v["score"] == 1.0
|
|
|
|
def test_score_clamped_low(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": True, "score": -0.3}))
|
|
assert v["score"] == 0.0
|
|
|
|
def test_score_at_edges(self):
|
|
assert lr.parse_verdict(json.dumps({"pass": True, "score": 0.0}))["score"] == 0.0
|
|
assert lr.parse_verdict(json.dumps({"pass": True, "score": 1.0}))["score"] == 1.0
|
|
|
|
def test_score_nan_to_neutral(self):
|
|
# NaN — literal NaN token in JSON is not strict, but some
|
|
# post-JSON flows introduce it (Hermes-style recursive decode).
|
|
# Build the dict directly and json.dumps it; "NaN" round-trips
|
|
# through Python's json as the token "NaN". json.loads of the
|
|
# serialized form returns float("nan"). Use that.
|
|
text = '{"pass": true, "score": NaN}'
|
|
# Python's json.loads accepts "NaN" token by default; json.dumps
|
|
# writes it back. parse_verdict should detect via math.isfinite.
|
|
v = lr.parse_verdict(text)
|
|
assert v is not None
|
|
assert v["score"] == 0.5
|
|
assert v["pass"] is True
|
|
|
|
def test_score_infinity_to_neutral(self):
|
|
text = '{"pass": true, "score": Infinity}'
|
|
v = lr.parse_verdict(text)
|
|
assert v is not None
|
|
assert v["score"] == 0.5
|
|
|
|
def test_score_non_numeric_string(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": True, "score": "great"}))
|
|
assert v is not None
|
|
assert v["score"] == 0.5
|
|
assert v["pass"] is True
|
|
|
|
def test_score_numeric_string_ok(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": True, "score": "0.75"}))
|
|
assert v is not None
|
|
assert v["score"] == 0.75
|
|
|
|
def test_score_missing_defaults_to_zero(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": True}))
|
|
assert v is not None
|
|
assert v["score"] == 0.0 # .get("score", 0.0) fallback
|
|
|
|
def test_score_none_value_to_neutral(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": True, "score": None}))
|
|
# data.get("score", 0.0) returns None (key exists with None value);
|
|
# float(None) raises TypeError -> 0.5 default per R3.
|
|
assert v is not None
|
|
assert v["score"] == 0.5
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Existing behavior — fence-block path still parses (no regression)
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestFenceBlockStillWorks:
|
|
def test_existing_fence_block_behavior(self):
|
|
text = "```json\n{\"pass\": true, \"score\": 0.8}\n```"
|
|
v = lr.parse_verdict(text)
|
|
assert v is not None
|
|
assert v["pass"] is True
|
|
assert v["score"] == 0.8
|
|
|
|
def test_fenced_with_string_pass(self):
|
|
# Fence-block path also honors the new coercion.
|
|
text = "```json\n{\"pass\": \"false\", \"score\": 1.2}\n```"
|
|
v = lr.parse_verdict(text)
|
|
assert v is not None
|
|
assert v["pass"] is False
|
|
assert v["score"] == 1.0
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Optional keys still preserved
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
class TestOptionalKeysPreserved:
|
|
def test_reasons_and_next_hint(self):
|
|
text = json.dumps({
|
|
"pass": "false",
|
|
"score": 0.1,
|
|
"reasons": ["bug1", "bug2"],
|
|
"next_hint": "fix the parser edge case",
|
|
})
|
|
v = lr.parse_verdict(text)
|
|
assert v is not None
|
|
assert v["pass"] is False
|
|
assert v["reasons"] == ["bug1", "bug2"]
|
|
assert v["next_hint"] == "fix the parser edge case"
|
|
|
|
def test_missing_reasons_defaults_empty_list(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": True, "score": 0.8}))
|
|
assert v["reasons"] == []
|
|
|
|
def test_non_list_reasons_coerced_to_empty(self):
|
|
v = lr.parse_verdict(json.dumps({"pass": True, "score": 0.8, "reasons": "x"}))
|
|
assert v["reasons"] == [] |