hermes: select xhigh reasoning for large suite jobs
This commit is contained in:
parent
a281c99634
commit
ce3025fb9c
@ -7,10 +7,24 @@ server configuration option, not an automatic fallback. The HTTPS contract,
|
|||||||
credential scopes, implementation objective, five-case task cap,
|
credential scopes, implementation objective, five-case task cap,
|
||||||
30-minute deadline, and USD 30 CLI estimated-cost guard are unchanged.
|
30-minute deadline, and USD 30 CLI estimated-cost guard are unchanged.
|
||||||
|
|
||||||
Execution revision `suite-multipass-v6-20260929` raises effort to `high` for every
|
Execution revision `suite-multipass-v7-20260929` uses `high` by default and
|
||||||
new Claude invocation. Preflight `selection.reasoning` and the CLI `--effort`
|
`xhigh` for at least 100 cases OR at least 131,072 bytes (128 KiB) of complete
|
||||||
argument use the same configuration value. The comparison below used `medium`;
|
suite content. Byte size is the canonical UTF-8 JSON of `campaign`, `suite`, and
|
||||||
its runtime and cost measurements do not measure high effort. Both Opus 5.5 and
|
all `cases`, without transport whitespace, routing/execution fields, prompt text,
|
||||||
|
schemas, or generated proposals. The selected effort is fixed for every pass in
|
||||||
|
that job, including structured-output repair; later review prompts do not change
|
||||||
|
it. There is no classifier or auxiliary inference call to choose effort.
|
||||||
|
|
||||||
|
Capabilities expose the thresholds in `models.claude.reasoning_policy`.
|
||||||
|
Preflight and job metadata expose `selection.reasoning` plus
|
||||||
|
`selection.reasoning_selection` with the policy revision, measured source bytes,
|
||||||
|
case count, and triggered thresholds. Per-pass metadata and completed results
|
||||||
|
also contain `reasoning`. The policy revision is `suite-size-effort-v1-20260929`.
|
||||||
|
The client does not need a new request field. Explicit smaller job budgets remain
|
||||||
|
effective; the service fails instead of reducing effort or accepting partial work.
|
||||||
|
|
||||||
|
The comparison below used `medium`; its runtime and cost measurements do not
|
||||||
|
measure high or xhigh effort. Both Opus 5.5 and
|
||||||
Sonnet 5.5 support `low`, `medium`, `high`, `xhigh`, and `max`. High is the next
|
Sonnet 5.5 support `low`, `medium`, `high`, `xhigh`, and `max`. High is the next
|
||||||
step above medium; xhigh and max spend more reasoning tokens and may take longer,
|
step above medium; xhigh and max spend more reasoning tokens and may take longer,
|
||||||
without guaranteeing better grouping. No hosted inference or real roster rerun
|
without guaranteeing better grouping. No hosted inference or real roster rerun
|
||||||
@ -85,7 +99,7 @@ checksum verification leaves the worker unavailable instead of using another CLI
|
|||||||
|
|
||||||
The HTTP compatibility revision remains `suite-v6-20260929`, the prompt remains
|
The HTTP compatibility revision remains `suite-v6-20260929`, the prompt remains
|
||||||
`implementation-proximity-multipass-v4-20260929`, and the execution revision is
|
`implementation-proximity-multipass-v4-20260929`, and the execution revision is
|
||||||
`suite-multipass-v6-20260929`. Authenticated capabilities additionally expose
|
`suite-multipass-v7-20260929`. Authenticated capabilities additionally expose
|
||||||
`claude_model_options` and `model_selection: server_configuration`. No new client
|
`claude_model_options` and `model_selection: server_configuration`. No new client
|
||||||
request field is accepted or required.
|
request field is accepted or required.
|
||||||
|
|
||||||
|
|||||||
@ -11,7 +11,7 @@ most **64 characters**, unique after whitespace and case normalization.
|
|||||||
- Configuration: `suite-v6-20260929` (HTTP compatibility identifier).
|
- Configuration: `suite-v6-20260929` (HTTP compatibility identifier).
|
||||||
- Policy: `implementation-five-v1-20260929`.
|
- Policy: `implementation-five-v1-20260929`.
|
||||||
- Prompt: `implementation-proximity-multipass-v4-20260929`.
|
- Prompt: `implementation-proximity-multipass-v4-20260929`.
|
||||||
- Execution: `suite-multipass-v6-20260929`.
|
- Execution: `suite-multipass-v7-20260929`.
|
||||||
|
|
||||||
A single server-side job performs:
|
A single server-side job performs:
|
||||||
|
|
||||||
@ -46,13 +46,22 @@ missing review decisions still fail closed; no missing assignment is fabricated.
|
|||||||
Each invocation retains six CLI turns for structured output. Model review passes
|
Each invocation retains six CLI turns for structured output. Model review passes
|
||||||
and CLI turns are separate counters. All calls use the originally selected provider
|
and CLI turns are separate counters. All calls use the originally selected provider
|
||||||
and pinned model. The current default is `claude-opus-5-5[1m]`, with canonical runtime
|
and pinned model. The current default is `claude-opus-5-5[1m]`, with canonical runtime
|
||||||
identity checked as `claude-opus-5-5`, firstParty, high effort, reported 1M context
|
identity checked as `claude-opus-5-5`, firstParty, reported 1M context
|
||||||
and 128K model output ceiling, with requests limited to 64K output tokens.
|
and 128K model output ceiling, with requests limited to 64K output tokens.
|
||||||
The server invokes native Claude Code CLI 2.1.285 using
|
The server invokes native Claude Code CLI 2.1.285 using
|
||||||
the existing first-party OAuth account, not a separately configured API-key
|
the existing first-party OAuth account, not a separately configured API-key
|
||||||
account. The CLI itself communicates with Anthropic over HTTPS. No new provider
|
account. The CLI itself communicates with Anthropic over HTTPS. No new provider
|
||||||
fallback or tools are enabled.
|
fallback or tools are enabled.
|
||||||
|
|
||||||
|
Effort defaults to `high` and increases to `xhigh` for suites with at least 100
|
||||||
|
cases OR 128 KiB of canonical UTF-8 source JSON (campaign, suite, and all cases).
|
||||||
|
Preflight selects the effort once for the whole job; every discovery, review,
|
||||||
|
audit and structured-output repair uses that selection. Source size excludes
|
||||||
|
transport whitespace, prompts, schemas, and generated proposals. Thresholds are
|
||||||
|
published in capabilities; selection metadata includes counts, bytes, and triggers.
|
||||||
|
Existing deadlines and cost guards still apply and never cause a silent effort
|
||||||
|
reduction. See the [effort policy details](hermes_suite_claude55.md).
|
||||||
|
|
||||||
Sonnet 5.5 is also verified on the 14-case synthetic fixture and available through
|
Sonnet 5.5 is also verified on the 14-case synthetic fixture and available through
|
||||||
server configuration. See the [matched comparison](hermes_suite_claude55.md).
|
server configuration. See the [matched comparison](hermes_suite_claude55.md).
|
||||||
Earlier acceptance measurements below used Opus 4.8; they do not establish large
|
Earlier acceptance measurements below used Opus 4.8; they do not establish large
|
||||||
|
|||||||
@ -84,7 +84,7 @@ class Provider(BaseHTTPRequestHandler):
|
|||||||
'complete_system': system, 'max_tokens': body.get('max_tokens'),
|
'complete_system': system, 'max_tokens': body.get('max_tokens'),
|
||||||
'effort': body.get('output_config', {}).get('effort')})
|
'effort': body.get('output_config', {}).get('effort')})
|
||||||
assert complete and system
|
assert complete and system
|
||||||
assert seen[-1]['effort'] == MODELS['claude']['reasoning']
|
assert seen[-1]['effort'] == active['reasoning']
|
||||||
value = response_value()
|
value = response_value()
|
||||||
# Force one schema rejection; the actual CLI must repair it within turns.
|
# Force one schema rejection; the actual CLI must repair it within turns.
|
||||||
omitted = len(json.loads(active['input'])['suite']['cases']) == 14 and len(seen) == 1
|
omitted = len(json.loads(active['input'])['suite']['cases']) == 14 and len(seen) == 1
|
||||||
|
|||||||
@ -95,10 +95,13 @@ def local_generate(request, cancel, client_ip, *, invocation=None):
|
|||||||
"provenance": response.get("inference_provenance")}
|
"provenance": response.get("inference_provenance")}
|
||||||
|
|
||||||
|
|
||||||
def claude_command(model, max_cost):
|
def claude_command(model, max_cost, *, reasoning=None):
|
||||||
"""Use the native pinned binary, never the privileged Hermes shell wrapper."""
|
"""Use the native pinned binary, never the privileged Hermes shell wrapper."""
|
||||||
if model not in CLAUDE_MODELS:
|
if model not in CLAUDE_MODELS:
|
||||||
raise Problem("unsupported_backend", 422)
|
raise Problem("unsupported_backend", 422)
|
||||||
|
reasoning = MODELS["claude"]["reasoning"] if reasoning is None else reasoning
|
||||||
|
if reasoning not in {"high", "xhigh"}:
|
||||||
|
raise Problem("unsupported_reasoning", 422)
|
||||||
settings = {"enabledPlugins": {"agents-md@builtin": False,
|
settings = {"enabledPlugins": {"agents-md@builtin": False,
|
||||||
"cc-plugin-agents-md@builtin": False},
|
"cc-plugin-agents-md@builtin": False},
|
||||||
"disableAllHooks": True, "disableBundledSkills": True,
|
"disableAllHooks": True, "disableBundledSkills": True,
|
||||||
@ -109,7 +112,7 @@ def claude_command(model, max_cost):
|
|||||||
"--strict-mcp-config", "--mcp-config", '{"mcpServers":{}}',
|
"--strict-mcp-config", "--mcp-config", '{"mcpServers":{}}',
|
||||||
"--setting-sources", "", "--settings", encoded(settings).decode(), "--disable-slash-commands",
|
"--setting-sources", "", "--settings", encoded(settings).decode(), "--disable-slash-commands",
|
||||||
"--permission-mode", "dontAsk", "--no-chrome",
|
"--permission-mode", "dontAsk", "--no-chrome",
|
||||||
"--model", model + "[1m]", "--effort", MODELS["claude"]["reasoning"],
|
"--model", model + "[1m]", "--effort", reasoning,
|
||||||
"--max-budget-usd", str(max_cost),
|
"--max-budget-usd", str(max_cost),
|
||||||
"--max-turns", str(CLAUDE_MAX_TURNS), "--system-prompt", SYSTEM,
|
"--max-turns", str(CLAUDE_MAX_TURNS), "--system-prompt", SYSTEM,
|
||||||
"--json-schema", encoded(SCHEMA).decode()]
|
"--json-schema", encoded(SCHEMA).decode()]
|
||||||
@ -226,7 +229,8 @@ def claude_generate(request, cancel, *, invocation=None, progress=None):
|
|||||||
with tempfile.TemporaryDirectory(prefix="suite-", dir="/jobs") as directory:
|
with tempfile.TemporaryDirectory(prefix="suite-", dir="/jobs") as directory:
|
||||||
root = Path(directory)
|
root = Path(directory)
|
||||||
(root / "input").write_text(invocation["input"] if invocation else prompt(request))
|
(root / "input").write_text(invocation["input"] if invocation else prompt(request))
|
||||||
command = claude_command(model, request["execution"]["max_cost_usd"])
|
command = claude_command(model, request["execution"]["max_cost_usd"],
|
||||||
|
reasoning=invocation.get("reasoning") if invocation else None)
|
||||||
if invocation:
|
if invocation:
|
||||||
command[command.index("--system-prompt") + 1] = invocation["system"]
|
command[command.index("--system-prompt") + 1] = invocation["system"]
|
||||||
command[command.index("--json-schema") + 1] = encoded(invocation["schema"]).decode()
|
command[command.index("--json-schema") + 1] = encoded(invocation["schema"]).decode()
|
||||||
|
|||||||
@ -9,7 +9,7 @@ from collections import Counter
|
|||||||
|
|
||||||
REVISION = "suite-v6-20260929"
|
REVISION = "suite-v6-20260929"
|
||||||
PROMPT_REVISION = "implementation-proximity-multipass-v4-20260929"
|
PROMPT_REVISION = "implementation-proximity-multipass-v4-20260929"
|
||||||
EXECUTION_REVISION = "suite-multipass-v6-20260929"
|
EXECUTION_REVISION = "suite-multipass-v7-20260929"
|
||||||
CLAUDE_VERSION = "2.1.285"
|
CLAUDE_VERSION = "2.1.285"
|
||||||
CLAUDE_MODELS = {
|
CLAUDE_MODELS = {
|
||||||
"claude-opus-4-8": 64000,
|
"claude-opus-4-8": 64000,
|
||||||
@ -20,6 +20,13 @@ CLAUDE_MODEL = os.environ.get("PLANNING_CLAUDE_MODEL", "claude-opus-4-8")
|
|||||||
if CLAUDE_MODEL not in CLAUDE_MODELS:
|
if CLAUDE_MODEL not in CLAUDE_MODELS:
|
||||||
raise RuntimeError("Unsupported configured Claude model")
|
raise RuntimeError("Unsupported configured Claude model")
|
||||||
CLAUDE_MAX_TURNS = 6
|
CLAUDE_MAX_TURNS = 6
|
||||||
|
CLAUDE_REASONING_POLICY = {
|
||||||
|
"revision": "suite-size-effort-v1-20260929",
|
||||||
|
"default": "high", "large": "xhigh",
|
||||||
|
"case_count_at_least": 100, "source_bytes_at_least": 128 * 1024,
|
||||||
|
"match": "any", "scope": "whole_job",
|
||||||
|
"source_bytes_scope": "Canonical UTF-8 JSON of campaign, suite, and all cases",
|
||||||
|
}
|
||||||
MAX_BODY = 1 << 20
|
MAX_BODY = 1 << 20
|
||||||
MAX_RESULT = 1 << 20
|
MAX_RESULT = 1 << 20
|
||||||
MAX_CASES = 400
|
MAX_CASES = 400
|
||||||
@ -37,7 +44,8 @@ MODELS = {
|
|||||||
"cli_model": CLAUDE_MODEL + "[1m]",
|
"cli_model": CLAUDE_MODEL + "[1m]",
|
||||||
"output": 64000, "reported_output": CLAUDE_MODELS[CLAUDE_MODEL],
|
"output": 64000, "reported_output": CLAUDE_MODELS[CLAUDE_MODEL],
|
||||||
"overhead": 8192, "backend": "claude-code-" + CLAUDE_VERSION,
|
"overhead": 8192, "backend": "claude-code-" + CLAUDE_VERSION,
|
||||||
"enabled": True, "reasoning": "high", "max_turns": CLAUDE_MAX_TURNS},
|
"enabled": True, "reasoning": "high", "reasoning_policy": CLAUDE_REASONING_POLICY,
|
||||||
|
"max_turns": CLAUDE_MAX_TURNS},
|
||||||
"codex": {"model": "gpt-6-astra", "context": 258400,
|
"codex": {"model": "gpt-6-astra", "context": 258400,
|
||||||
"output": None, "overhead": None, "backend": "codex-subscription-broker",
|
"output": None, "overhead": None, "backend": "codex-subscription-broker",
|
||||||
"enabled": False, "reasoning": "medium",
|
"enabled": False, "reasoning": "medium",
|
||||||
@ -190,6 +198,23 @@ def prompt(request):
|
|||||||
return encoded({k: request[k] for k in ("campaign", "suite", "cases")}).decode()
|
return encoded({k: request[k] for k in ("campaign", "suite", "cases")}).decode()
|
||||||
|
|
||||||
|
|
||||||
|
def reasoning_selection(request, provider):
|
||||||
|
"""Choose effort from complete source size without a classifier or model call."""
|
||||||
|
if provider != "claude":
|
||||||
|
return {"reasoning": MODELS[provider]["reasoning"]}
|
||||||
|
policy = CLAUDE_REASONING_POLICY
|
||||||
|
count, size = len(request["cases"]), len(prompt(request).encode("utf-8"))
|
||||||
|
triggers = []
|
||||||
|
if count >= policy["case_count_at_least"]:
|
||||||
|
triggers.append("case_count")
|
||||||
|
if size >= policy["source_bytes_at_least"]:
|
||||||
|
triggers.append("source_bytes")
|
||||||
|
return {"reasoning": policy["large"] if triggers else policy["default"],
|
||||||
|
"reasoning_selection": {"policy_revision": policy["revision"],
|
||||||
|
"case_count": count, "source_bytes": size,
|
||||||
|
"triggers": triggers, "scope": policy["scope"]}}
|
||||||
|
|
||||||
|
|
||||||
def preflight(request):
|
def preflight(request):
|
||||||
"""Select one permitted provider that can admit the multi-pass policy."""
|
"""Select one permitted provider that can admit the multi-pass policy."""
|
||||||
from suite_multipass import preflight_workflow
|
from suite_multipass import preflight_workflow
|
||||||
|
|||||||
@ -8,7 +8,8 @@ import time
|
|||||||
import suite_backends
|
import suite_backends
|
||||||
from suite_assignments import decode
|
from suite_assignments import decode
|
||||||
from suite_contract import (EXECUTION_REVISION, MAX_BODY, MAX_RESULT, MODELS, PROMPT_REVISION,
|
from suite_contract import (EXECUTION_REVISION, MAX_BODY, MAX_RESULT, MODELS, PROMPT_REVISION,
|
||||||
PROMPT_SHA256, REVISION, Problem, digest, encoded, validate_partition)
|
PROMPT_SHA256, REVISION, Problem, digest, encoded, reasoning_selection,
|
||||||
|
validate_partition)
|
||||||
from suite_policy import (BASE_NAME_LIMIT, MAX_GROUP, POLICY_REVISION, invocation,
|
from suite_policy import (BASE_NAME_LIMIT, MAX_GROUP, POLICY_REVISION, invocation,
|
||||||
validate_natural)
|
validate_natural)
|
||||||
from suite_sizing import cap_families, review_summary
|
from suite_sizing import cap_families, review_summary
|
||||||
@ -67,7 +68,8 @@ def preflight_workflow(request):
|
|||||||
except Problem as exc:
|
except Problem as exc:
|
||||||
reasons[provider] = exc.code
|
reasons[provider] = exc.code
|
||||||
continue
|
continue
|
||||||
return {"provider": provider, **model, **initial, "configuration_revision": REVISION,
|
return {"provider": provider, **model, **initial, **reasoning_selection(request, provider),
|
||||||
|
"configuration_revision": REVISION,
|
||||||
"prompt_revision": PROMPT_REVISION, "prompt_sha256": PROMPT_SHA256,
|
"prompt_revision": PROMPT_REVISION, "prompt_sha256": PROMPT_SHA256,
|
||||||
"execution_revision": EXECUTION_REVISION, "policy_revision": POLICY_REVISION,
|
"execution_revision": EXECUTION_REVISION, "policy_revision": POLICY_REVISION,
|
||||||
"case_count": count, "source_sha256": digest(request["cases"]),
|
"case_count": count, "source_sha256": digest(request["cases"]),
|
||||||
@ -100,6 +102,7 @@ class Workflow:
|
|||||||
|
|
||||||
def __init__(self, request, selected, cancel, client_ip, progress=None):
|
def __init__(self, request, selected, cancel, client_ip, progress=None):
|
||||||
self.request, self.provider, self.cancel = request, selected["provider"], cancel
|
self.request, self.provider, self.cancel = request, selected["provider"], cancel
|
||||||
|
self.reasoning = selected.get("reasoning", MODELS[self.provider]["reasoning"])
|
||||||
self.client_ip, self.progress = client_ip, progress or (lambda _: None)
|
self.client_ip, self.progress = client_ip, progress or (lambda _: None)
|
||||||
self.started = time.monotonic()
|
self.started = time.monotonic()
|
||||||
self.deadline = self.started + request["execution"]["max_seconds"]
|
self.deadline = self.started + request["execution"]["max_seconds"]
|
||||||
@ -124,6 +127,8 @@ class Workflow:
|
|||||||
if len(self.records) >= MAX_PASSES:
|
if len(self.records) >= MAX_PASSES:
|
||||||
raise Problem("model_pass_limit", 502)
|
raise Problem("model_pass_limit", 502)
|
||||||
call = invocation(stage, source, context)
|
call = invocation(stage, source, context)
|
||||||
|
# Keep the preflight choice across independent ordering and larger review prompts.
|
||||||
|
call["reasoning"] = self.reasoning
|
||||||
limits = capacity(call, self.provider, len(source["cases"]), family_count)
|
limits = capacity(call, self.provider, len(source["cases"]), family_count)
|
||||||
remaining_seconds = self.deadline - time.monotonic()
|
remaining_seconds = self.deadline - time.monotonic()
|
||||||
remaining_cost = self.cost_limit - self.spent
|
remaining_cost = self.cost_limit - self.spent
|
||||||
@ -150,6 +155,7 @@ class Workflow:
|
|||||||
raise Problem("unsupported_backend", 422)
|
raise Problem("unsupported_backend", 422)
|
||||||
cost = metadata.get("cost_usd_estimate") if self.provider == "claude" else 0.0
|
cost = metadata.get("cost_usd_estimate") if self.provider == "claude" else 0.0
|
||||||
record = {"stage": stage, "provider": self.provider, "model": MODELS[self.provider]["model"],
|
record = {"stage": stage, "provider": self.provider, "model": MODELS[self.provider]["model"],
|
||||||
|
"reasoning": self.reasoning,
|
||||||
"wall_seconds": round(time.monotonic() - started, 3), **limits,
|
"wall_seconds": round(time.monotonic() - started, 3), **limits,
|
||||||
"system_sha256": hashlib.sha256(call["system"].encode()).hexdigest(),
|
"system_sha256": hashlib.sha256(call["system"].encode()).hexdigest(),
|
||||||
"schema_sha256": digest(call["schema"]),
|
"schema_sha256": digest(call["schema"]),
|
||||||
@ -194,6 +200,7 @@ class Workflow:
|
|||||||
review = review_summary(a, b, reconciled, reviewed, audited, final, divisions)
|
review = review_summary(a, b, reconciled, reviewed, audited, final, divisions)
|
||||||
self.checkpoint()
|
self.checkpoint()
|
||||||
metadata = {**self.last_metadata, "passes": self.records,
|
metadata = {**self.last_metadata, "passes": self.records,
|
||||||
|
"reasoning": self.reasoning,
|
||||||
"model_pass_count": len(self.records), "usage": sum_usage(self.records),
|
"model_pass_count": len(self.records), "usage": sum_usage(self.records),
|
||||||
"cli_diagnostics_scope": "last_model_pass",
|
"cli_diagnostics_scope": "last_model_pass",
|
||||||
"duration_api_ms": sum(r["duration_api_ms"] for r in self.records) if all(type(r["duration_api_ms"]) in (int, float) for r in self.records) else None,
|
"duration_api_ms": sum(r["duration_api_ms"] for r in self.records) if all(type(r["duration_api_ms"]) in (int, float) for r in self.records) else None,
|
||||||
|
|||||||
@ -36,7 +36,7 @@ spec:
|
|||||||
app: hermes-suite-planner
|
app: hermes-suite-planner
|
||||||
annotations:
|
annotations:
|
||||||
fluentbit.io/exclude: "true"
|
fluentbit.io/exclude: "true"
|
||||||
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v6-20260929
|
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v7-20260929
|
||||||
vault.hashicorp.com/agent-inject: "true"
|
vault.hashicorp.com/agent-inject: "true"
|
||||||
vault.hashicorp.com/agent-pre-populate-only: "true"
|
vault.hashicorp.com/agent-pre-populate-only: "true"
|
||||||
vault.hashicorp.com/agent-init-first: "true"
|
vault.hashicorp.com/agent-init-first: "true"
|
||||||
|
|||||||
@ -10,7 +10,8 @@ SCRIPTS = Path(__file__).resolve().parents[2] / "services/hermes/scripts"
|
|||||||
sys.path.insert(0, str(SCRIPTS))
|
sys.path.insert(0, str(SCRIPTS))
|
||||||
import suite_backends
|
import suite_backends
|
||||||
from suite_cli_diagnostics import usage_counts
|
from suite_cli_diagnostics import usage_counts
|
||||||
from suite_contract import CLAUDE_MODELS, Problem
|
from suite_contract import (CLAUDE_MODELS, MODELS, Problem, preflight, prompt,
|
||||||
|
reasoning_selection, validate_request)
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.parametrize("model", CLAUDE_MODELS)
|
@pytest.mark.parametrize("model", CLAUDE_MODELS)
|
||||||
@ -72,3 +73,60 @@ def test_thinking_usage_is_counted_without_retaining_unrecognized_fields():
|
|||||||
"""New CLI usage details retain measurements and drop arbitrary content."""
|
"""New CLI usage details retain measurements and drop arbitrary content."""
|
||||||
assert usage_counts({"output_tokens_details": {"thinking_tokens": 123, "content": "private"}}) == {
|
assert usage_counts({"output_tokens_details": {"thinking_tokens": 123, "content": "private"}}) == {
|
||||||
"output_tokens_details": {"thinking_tokens": 123}}
|
"output_tokens_details": {"thinking_tokens": 123}}
|
||||||
|
|
||||||
|
|
||||||
|
def sized_request(count, source_bytes=None):
|
||||||
|
"""Build complete synthetic records with optional exact UTF-8 source size."""
|
||||||
|
value = {"campaign": "SYNTHETIC", "suite": "EFFORT", "cases": [
|
||||||
|
{"alias": f"CASE-{i:04d}", "description": "Synthetic parser objective."}
|
||||||
|
for i in range(count)],
|
||||||
|
"routing": {"allow_external": True, "allowed_external_providers": ["claude"]}}
|
||||||
|
if source_bytes is not None:
|
||||||
|
remaining = source_bytes - len(prompt(value).encode("utf-8"))
|
||||||
|
for case in value["cases"]:
|
||||||
|
take = min(remaining, 20000)
|
||||||
|
case["description"] += "\u03bc" * (take // 2) + "x" * (take % 2)
|
||||||
|
remaining -= take
|
||||||
|
assert remaining == 0
|
||||||
|
return validate_request(value, ["claude"])
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("count,effort", [(14, "high"), (75, "high"), (99, "high"),
|
||||||
|
(100, "xhigh"), (363, "xhigh"), (400, "xhigh")])
|
||||||
|
def test_preflight_selects_effort_at_case_count_boundary(count, effort):
|
||||||
|
"""Small and large requests choose independently without changing the default."""
|
||||||
|
selected = preflight(sized_request(count))
|
||||||
|
assert selected["provider"] == "claude" and selected["reasoning"] == effort
|
||||||
|
assert selected["reasoning_selection"]["triggers"] == (["case_count"] if count >= 100 else [])
|
||||||
|
command = suite_backends.claude_command(selected["model"], 30, reasoning=selected["reasoning"])
|
||||||
|
assert command[command.index("--effort") + 1] == effort
|
||||||
|
assert MODELS["claude"]["reasoning"] == "high"
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize("size,effort", [(131071, "high"), (131072, "xhigh"), (131073, "xhigh")])
|
||||||
|
def test_complete_utf8_source_size_is_an_independent_trigger(size, effort):
|
||||||
|
"""Use encoded bytes, not characters or transport whitespace, at the boundary."""
|
||||||
|
value = sized_request(14, size)
|
||||||
|
selected = preflight(value)
|
||||||
|
assert selected["reasoning"] == effort
|
||||||
|
assert selected["reasoning_selection"]["source_bytes"] == size
|
||||||
|
assert len(prompt(value)) < size
|
||||||
|
assert selected["reasoning_selection"]["triggers"] == (["source_bytes"] if size >= 131072 else [])
|
||||||
|
reordered = {**value, "cases": list(reversed(value["cases"]))}
|
||||||
|
assert reasoning_selection(reordered, "claude") == reasoning_selection(value, "claude")
|
||||||
|
value["execution"]["max_cost_usd"] = 1
|
||||||
|
assert reasoning_selection(value, "claude")["reasoning"] == effort
|
||||||
|
|
||||||
|
|
||||||
|
def test_size_policy_cannot_broaden_routing_or_enable_client_effort_override():
|
||||||
|
"""Effort selection never adds an external destination to a local-only job."""
|
||||||
|
value = sized_request(363)
|
||||||
|
value["routing"] = {"allow_external": False, "allowed_external_providers": []}
|
||||||
|
with pytest.raises(Problem, match="capacity_or_unsupported_backend"):
|
||||||
|
preflight(value)
|
||||||
|
assert reasoning_selection(value, "local") == {"reasoning": "none"}
|
||||||
|
value["execution"]["reasoning"] = "xhigh"
|
||||||
|
with pytest.raises(Problem, match="invalid_request"):
|
||||||
|
validate_request(value, ["claude"])
|
||||||
|
with pytest.raises(Problem, match="unsupported_reasoning"):
|
||||||
|
suite_backends.claude_command(MODELS["claude"]["model"], 30, reasoning="max")
|
||||||
|
|||||||
@ -182,7 +182,7 @@ def test_actual_subprocess_paths_cleanup_and_safe_metadata(tmp_path, monkeypatch
|
|||||||
elif mode == "success":
|
elif mode == "success":
|
||||||
script += "time.sleep(0.25)"
|
script += "time.sleep(0.25)"
|
||||||
command = ["/not-installed-synthetic-cli"] if mode == "start" else [sys.executable, "-c", script]
|
command = ["/not-installed-synthetic-cli"] if mode == "start" else [sys.executable, "-c", script]
|
||||||
monkeypatch.setattr(suite_backends, "claude_command", lambda *_: command)
|
monkeypatch.setattr(suite_backends, "claude_command", lambda *_, **__: command)
|
||||||
value, cancel = request(), threading.Event()
|
value, cancel = request(), threading.Event()
|
||||||
if mode == "timeout":
|
if mode == "timeout":
|
||||||
value["execution"]["max_seconds"] = 0.02
|
value["execution"]["max_seconds"] = 0.02
|
||||||
|
|||||||
@ -139,6 +139,21 @@ def test_independent_orders_and_content_are_reproducible(monkeypatch):
|
|||||||
assert 'proposal_a' in json.loads(calls[2][1]['input'])['review_material']
|
assert 'proposal_a' in json.loads(calls[2][1]['input'])['review_material']
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.parametrize('count,effort', [(14, 'high'), (100, 'xhigh')])
|
||||||
|
def test_preflight_effort_is_preserved_through_all_five_passes(monkeypatch, count, effort):
|
||||||
|
"""Discovery, reconciliation, review and audit use the same selected effort."""
|
||||||
|
source = request(count)
|
||||||
|
family = natural(source, [[c['alias'] for c in source['cases']]])
|
||||||
|
calls = install_backend(monkeypatch, source, family)
|
||||||
|
selected = preflight(source)
|
||||||
|
result, metadata = workflow.generate(source, selected, threading.Event(), '192.168.22.8')
|
||||||
|
assert selected['reasoning'] == metadata['reasoning'] == effort
|
||||||
|
assert len(calls) == 5
|
||||||
|
assert all(call['reasoning'] == effort for _, call in calls)
|
||||||
|
assert all(p['reasoning'] == effort for p in metadata['passes'])
|
||||||
|
validate_result(result, source)
|
||||||
|
|
||||||
|
|
||||||
def test_progress_reports_activity_without_content_or_false_percentage(monkeypatch):
|
def test_progress_reports_activity_without_content_or_false_percentage(monkeypatch):
|
||||||
source = request(7)
|
source = request(7)
|
||||||
family = natural(source, [[c['alias'] for c in source['cases']]])
|
family = natural(source, [[c['alias'] for c in source['cases']]])
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user