2026-09-29 15:50:36 -05:00
|
|
|
"""Model upgrades preserve exact identity, capacity, and CLI isolation."""
|
|
|
|
|
import json
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
import subprocess
|
|
|
|
|
import sys
|
|
|
|
|
|
|
|
|
|
import pytest
|
|
|
|
|
|
|
|
|
|
SCRIPTS = Path(__file__).resolve().parents[2] / "services/hermes/scripts"
|
|
|
|
|
sys.path.insert(0, str(SCRIPTS))
|
|
|
|
|
import suite_backends
|
|
|
|
|
from suite_cli_diagnostics import usage_counts
|
2026-09-29 16:05:30 -05:00
|
|
|
from suite_contract import (CLAUDE_MODELS, MODELS, Problem, preflight, prompt,
|
|
|
|
|
reasoning_selection, validate_request)
|
2026-09-29 15:50:36 -05:00
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize("model", CLAUDE_MODELS)
|
|
|
|
|
def test_only_configured_model_is_allowed_and_plugins_are_disabled(model):
|
|
|
|
|
"""Each invocation pins one model, with fallback and customizations disabled."""
|
|
|
|
|
command = suite_backends.claude_command(model, 30)
|
|
|
|
|
settings = json.loads(command[command.index("--settings") + 1])
|
|
|
|
|
assert command[command.index("--model") + 1] == model + "[1m]"
|
|
|
|
|
assert settings["availableModels"] == [model]
|
|
|
|
|
assert settings["switchModelsOnFlag"] is False
|
|
|
|
|
assert settings["disableAllHooks"] and settings["disableClaudeAiConnectors"]
|
|
|
|
|
assert settings["syncClaudeAiPlugins"] is False
|
|
|
|
|
assert settings["enabledPlugins"]["cc-plugin-agents-md@builtin"] is False
|
|
|
|
|
assert "--fallback-model" not in command
|
|
|
|
|
env = suite_backends.claude_environment("/jobs/fresh", "synthetic")
|
|
|
|
|
assert env["CLAUDE_CODE_MAX_OUTPUT_TOKENS"] == "64000"
|
|
|
|
|
assert env["CLAUDE_AGENT_SDK_DISABLE_BUILTIN_AGENTS"] == "1"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize("model", ["claude-opus-5-5", "claude-sonnet-5-5"])
|
|
|
|
|
@pytest.mark.parametrize("failure", [None, "substitution", "capacity", "plugin"])
|
|
|
|
|
def test_new_runtime_envelope_remains_fail_closed(model, failure):
|
|
|
|
|
"""A requested name never substitutes for the actual runtime model identity."""
|
|
|
|
|
init = {"type": "system", "subtype": "init", "model": model + "[1m]",
|
|
|
|
|
"tools": ["StructuredOutput"], "mcp_servers": [], "plugins": []}
|
|
|
|
|
limits = {"contextWindow": 1000000, "maxOutputTokens": 128000,
|
|
|
|
|
"canonicalModel": model, "provider": "firstParty"}
|
|
|
|
|
result = {"type": "result", "subtype": "success", "is_error": False,
|
|
|
|
|
"modelUsage": {model + "[1m]": limits}, "usage": {},
|
|
|
|
|
"structured_output": {"groups": []}}
|
|
|
|
|
if failure == "substitution":
|
|
|
|
|
limits["canonicalModel"] = "claude-opus-4-8"
|
|
|
|
|
elif failure == "capacity":
|
|
|
|
|
limits["contextWindow"] = 200000
|
|
|
|
|
elif failure == "plugin":
|
|
|
|
|
init["plugins"] = [{"name": "unapproved"}]
|
|
|
|
|
raw = "\n".join(map(json.dumps, (init, result)))
|
|
|
|
|
if failure:
|
|
|
|
|
with pytest.raises(Problem):
|
|
|
|
|
suite_backends.parse_claude(raw, model, exit_code=0)
|
|
|
|
|
else:
|
|
|
|
|
_, metadata = suite_backends.parse_claude(raw, model, exit_code=0)
|
|
|
|
|
assert metadata["model"] == model
|
|
|
|
|
assert metadata["model_usage"][model + "[1m]"]["maxOutputTokens"] == 128000
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_unknown_model_cannot_start_a_cli_process():
|
|
|
|
|
"""Operator configuration and command construction both reject unknown IDs."""
|
|
|
|
|
with pytest.raises(Problem):
|
|
|
|
|
suite_backends.claude_command("unapproved-model", 30)
|
|
|
|
|
result = subprocess.run([sys.executable, "-c", "import suite_contract"],
|
|
|
|
|
cwd=SCRIPTS, env={"PLANNING_CLAUDE_MODEL": "unapproved-model"},
|
|
|
|
|
capture_output=True, text=True)
|
|
|
|
|
assert result.returncode != 0
|
|
|
|
|
assert "Unsupported configured Claude model" in result.stderr
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_thinking_usage_is_counted_without_retaining_unrecognized_fields():
|
|
|
|
|
"""New CLI usage details retain measurements and drop arbitrary content."""
|
|
|
|
|
assert usage_counts({"output_tokens_details": {"thinking_tokens": 123, "content": "private"}}) == {
|
|
|
|
|
"output_tokens_details": {"thinking_tokens": 123}}
|
2026-09-29 16:05:30 -05:00
|
|
|
|
|
|
|
|
|
|
|
|
|
def sized_request(count, source_bytes=None):
|
|
|
|
|
"""Build complete synthetic records with optional exact UTF-8 source size."""
|
|
|
|
|
value = {"campaign": "SYNTHETIC", "suite": "EFFORT", "cases": [
|
|
|
|
|
{"alias": f"CASE-{i:04d}", "description": "Synthetic parser objective."}
|
|
|
|
|
for i in range(count)],
|
|
|
|
|
"routing": {"allow_external": True, "allowed_external_providers": ["claude"]}}
|
|
|
|
|
if source_bytes is not None:
|
|
|
|
|
remaining = source_bytes - len(prompt(value).encode("utf-8"))
|
|
|
|
|
for case in value["cases"]:
|
|
|
|
|
take = min(remaining, 20000)
|
|
|
|
|
case["description"] += "\u03bc" * (take // 2) + "x" * (take % 2)
|
|
|
|
|
remaining -= take
|
|
|
|
|
assert remaining == 0
|
|
|
|
|
return validate_request(value, ["claude"])
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize("count,effort", [(14, "high"), (75, "high"), (99, "high"),
|
|
|
|
|
(100, "xhigh"), (363, "xhigh"), (400, "xhigh")])
|
|
|
|
|
def test_preflight_selects_effort_at_case_count_boundary(count, effort):
|
|
|
|
|
"""Small and large requests choose independently without changing the default."""
|
|
|
|
|
selected = preflight(sized_request(count))
|
|
|
|
|
assert selected["provider"] == "claude" and selected["reasoning"] == effort
|
|
|
|
|
assert selected["reasoning_selection"]["triggers"] == (["case_count"] if count >= 100 else [])
|
|
|
|
|
command = suite_backends.claude_command(selected["model"], 30, reasoning=selected["reasoning"])
|
|
|
|
|
assert command[command.index("--effort") + 1] == effort
|
|
|
|
|
assert MODELS["claude"]["reasoning"] == "high"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize("size,effort", [(131071, "high"), (131072, "xhigh"), (131073, "xhigh")])
|
|
|
|
|
def test_complete_utf8_source_size_is_an_independent_trigger(size, effort):
|
|
|
|
|
"""Use encoded bytes, not characters or transport whitespace, at the boundary."""
|
|
|
|
|
value = sized_request(14, size)
|
|
|
|
|
selected = preflight(value)
|
|
|
|
|
assert selected["reasoning"] == effort
|
|
|
|
|
assert selected["reasoning_selection"]["source_bytes"] == size
|
|
|
|
|
assert len(prompt(value)) < size
|
|
|
|
|
assert selected["reasoning_selection"]["triggers"] == (["source_bytes"] if size >= 131072 else [])
|
|
|
|
|
reordered = {**value, "cases": list(reversed(value["cases"]))}
|
|
|
|
|
assert reasoning_selection(reordered, "claude") == reasoning_selection(value, "claude")
|
|
|
|
|
value["execution"]["max_cost_usd"] = 1
|
|
|
|
|
assert reasoning_selection(value, "claude")["reasoning"] == effort
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_size_policy_cannot_broaden_routing_or_enable_client_effort_override():
|
|
|
|
|
"""Effort selection never adds an external destination to a local-only job."""
|
|
|
|
|
value = sized_request(363)
|
|
|
|
|
value["routing"] = {"allow_external": False, "allowed_external_providers": []}
|
|
|
|
|
with pytest.raises(Problem, match="capacity_or_unsupported_backend"):
|
|
|
|
|
preflight(value)
|
|
|
|
|
assert reasoning_selection(value, "local") == {"reasoning": "none"}
|
|
|
|
|
value["execution"]["reasoning"] = "xhigh"
|
|
|
|
|
with pytest.raises(Problem, match="invalid_request"):
|
|
|
|
|
validate_request(value, ["claude"])
|
|
|
|
|
with pytest.raises(Problem, match="unsupported_reasoning"):
|
|
|
|
|
suite_backends.claude_command(MODELS["claude"]["model"], 30, reasoning="max")
|