atlas-iac/testing/tests/test_suite_claude_models.py

133 lines
6.7 KiB
Python
Raw Normal View History

"""Model upgrades preserve exact identity, capacity, and CLI isolation."""
import json
from pathlib import Path
import subprocess
import sys
import pytest
SCRIPTS = Path(__file__).resolve().parents[2] / "services/hermes/scripts"
sys.path.insert(0, str(SCRIPTS))
import suite_backends
from suite_cli_diagnostics import usage_counts
from suite_contract import (CLAUDE_MODELS, MODELS, Problem, preflight, prompt,
reasoning_selection, validate_request)
@pytest.mark.parametrize("model", CLAUDE_MODELS)
def test_only_configured_model_is_allowed_and_plugins_are_disabled(model):
"""Each invocation pins one model, with fallback and customizations disabled."""
command = suite_backends.claude_command(model, 30)
settings = json.loads(command[command.index("--settings") + 1])
assert command[command.index("--model") + 1] == model + "[1m]"
assert settings["availableModels"] == [model]
assert settings["switchModelsOnFlag"] is False
assert settings["disableAllHooks"] and settings["disableClaudeAiConnectors"]
assert settings["syncClaudeAiPlugins"] is False
assert settings["enabledPlugins"]["cc-plugin-agents-md@builtin"] is False
assert "--fallback-model" not in command
env = suite_backends.claude_environment("/jobs/fresh", "synthetic")
assert env["CLAUDE_CODE_MAX_OUTPUT_TOKENS"] == "64000"
assert env["CLAUDE_AGENT_SDK_DISABLE_BUILTIN_AGENTS"] == "1"
@pytest.mark.parametrize("model", ["claude-opus-5-5", "claude-sonnet-5-5"])
@pytest.mark.parametrize("failure", [None, "substitution", "capacity", "plugin"])
def test_new_runtime_envelope_remains_fail_closed(model, failure):
"""A requested name never substitutes for the actual runtime model identity."""
init = {"type": "system", "subtype": "init", "model": model + "[1m]",
"tools": ["StructuredOutput"], "mcp_servers": [], "plugins": []}
limits = {"contextWindow": 1000000, "maxOutputTokens": 128000,
"canonicalModel": model, "provider": "firstParty"}
result = {"type": "result", "subtype": "success", "is_error": False,
"modelUsage": {model + "[1m]": limits}, "usage": {},
"structured_output": {"groups": []}}
if failure == "substitution":
limits["canonicalModel"] = "claude-opus-4-8"
elif failure == "capacity":
limits["contextWindow"] = 200000
elif failure == "plugin":
init["plugins"] = [{"name": "unapproved"}]
raw = "\n".join(map(json.dumps, (init, result)))
if failure:
with pytest.raises(Problem):
suite_backends.parse_claude(raw, model, exit_code=0)
else:
_, metadata = suite_backends.parse_claude(raw, model, exit_code=0)
assert metadata["model"] == model
assert metadata["model_usage"][model + "[1m]"]["maxOutputTokens"] == 128000
def test_unknown_model_cannot_start_a_cli_process():
"""Operator configuration and command construction both reject unknown IDs."""
with pytest.raises(Problem):
suite_backends.claude_command("unapproved-model", 30)
result = subprocess.run([sys.executable, "-c", "import suite_contract"],
cwd=SCRIPTS, env={"PLANNING_CLAUDE_MODEL": "unapproved-model"},
capture_output=True, text=True)
assert result.returncode != 0
assert "Unsupported configured Claude model" in result.stderr
def test_thinking_usage_is_counted_without_retaining_unrecognized_fields():
"""New CLI usage details retain measurements and drop arbitrary content."""
assert usage_counts({"output_tokens_details": {"thinking_tokens": 123, "content": "private"}}) == {
"output_tokens_details": {"thinking_tokens": 123}}
def sized_request(count, source_bytes=None):
"""Build complete synthetic records with optional exact UTF-8 source size."""
value = {"campaign": "SYNTHETIC", "suite": "EFFORT", "cases": [
{"alias": f"CASE-{i:04d}", "description": "Synthetic parser objective."}
for i in range(count)],
"routing": {"allow_external": True, "allowed_external_providers": ["claude"]}}
if source_bytes is not None:
remaining = source_bytes - len(prompt(value).encode("utf-8"))
for case in value["cases"]:
take = min(remaining, 20000)
case["description"] += "\u03bc" * (take // 2) + "x" * (take % 2)
remaining -= take
assert remaining == 0
return validate_request(value, ["claude"])
@pytest.mark.parametrize("count,effort", [(14, "high"), (75, "high"), (99, "high"),
(100, "xhigh"), (363, "xhigh"), (400, "xhigh")])
def test_preflight_selects_effort_at_case_count_boundary(count, effort):
"""Small and large requests choose independently without changing the default."""
selected = preflight(sized_request(count))
assert selected["provider"] == "claude" and selected["reasoning"] == effort
assert selected["reasoning_selection"]["triggers"] == (["case_count"] if count >= 100 else [])
command = suite_backends.claude_command(selected["model"], 30, reasoning=selected["reasoning"])
assert command[command.index("--effort") + 1] == effort
assert MODELS["claude"]["reasoning"] == "high"
@pytest.mark.parametrize("size,effort", [(131071, "high"), (131072, "xhigh"), (131073, "xhigh")])
def test_complete_utf8_source_size_is_an_independent_trigger(size, effort):
"""Use encoded bytes, not characters or transport whitespace, at the boundary."""
value = sized_request(14, size)
selected = preflight(value)
assert selected["reasoning"] == effort
assert selected["reasoning_selection"]["source_bytes"] == size
assert len(prompt(value)) < size
assert selected["reasoning_selection"]["triggers"] == (["source_bytes"] if size >= 131072 else [])
reordered = {**value, "cases": list(reversed(value["cases"]))}
assert reasoning_selection(reordered, "claude") == reasoning_selection(value, "claude")
value["execution"]["max_cost_usd"] = 1
assert reasoning_selection(value, "claude")["reasoning"] == effort
def test_size_policy_cannot_broaden_routing_or_enable_client_effort_override():
"""Effort selection never adds an external destination to a local-only job."""
value = sized_request(363)
value["routing"] = {"allow_external": False, "allowed_external_providers": []}
with pytest.raises(Problem, match="capacity_or_unsupported_backend"):
preflight(value)
assert reasoning_selection(value, "local") == {"reasoning": "none"}
value["execution"]["reasoning"] = "xhigh"
with pytest.raises(Problem, match="invalid_request"):
validate_request(value, ["claude"])
with pytest.raises(Problem, match="unsupported_reasoning"):
suite_backends.claude_command(MODELS["claude"]["model"], 30, reasoning="max")