hermes: select xhigh reasoning for large suite jobs

This commit is contained in:
jenkins 2026-09-29 16:05:30 -05:00
parent a281c99634
commit ce3025fb9c
10 changed files with 150 additions and 18 deletions

View File

@ -7,10 +7,24 @@ server configuration option, not an automatic fallback. The HTTPS contract,
credential scopes, implementation objective, five-case task cap,
30-minute deadline, and USD 30 CLI estimated-cost guard are unchanged.
Execution revision `suite-multipass-v6-20260929` raises effort to `high` for every
new Claude invocation. Preflight `selection.reasoning` and the CLI `--effort`
argument use the same configuration value. The comparison below used `medium`;
its runtime and cost measurements do not measure high effort. Both Opus 5.5 and
Execution revision `suite-multipass-v7-20260929` uses `high` by default and
`xhigh` for at least 100 cases OR at least 131,072 bytes (128 KiB) of complete
suite content. Byte size is the canonical UTF-8 JSON of `campaign`, `suite`, and
all `cases`, without transport whitespace, routing/execution fields, prompt text,
schemas, or generated proposals. The selected effort is fixed for every pass in
that job, including structured-output repair; later review prompts do not change
it. There is no classifier or auxiliary inference call to choose effort.
Capabilities expose the thresholds in `models.claude.reasoning_policy`.
Preflight and job metadata expose `selection.reasoning` plus
`selection.reasoning_selection` with the policy revision, measured source bytes,
case count, and triggered thresholds. Per-pass metadata and completed results
also contain `reasoning`. The policy revision is `suite-size-effort-v1-20260929`.
The client does not need a new request field. Explicit smaller job budgets remain
effective; the service fails instead of reducing effort or accepting partial work.
The comparison below used `medium`; its runtime and cost measurements do not
measure high or xhigh effort. Both Opus 5.5 and
Sonnet 5.5 support `low`, `medium`, `high`, `xhigh`, and `max`. High is the next
step above medium; xhigh and max spend more reasoning tokens and may take longer,
without guaranteeing better grouping. No hosted inference or real roster rerun
@ -85,7 +99,7 @@ checksum verification leaves the worker unavailable instead of using another CLI
The HTTP compatibility revision remains `suite-v6-20260929`, the prompt remains
`implementation-proximity-multipass-v4-20260929`, and the execution revision is
`suite-multipass-v6-20260929`. Authenticated capabilities additionally expose
`suite-multipass-v7-20260929`. Authenticated capabilities additionally expose
`claude_model_options` and `model_selection: server_configuration`. No new client
request field is accepted or required.

View File

@ -11,7 +11,7 @@ most **64 characters**, unique after whitespace and case normalization.
- Configuration: `suite-v6-20260929` (HTTP compatibility identifier).
- Policy: `implementation-five-v1-20260929`.
- Prompt: `implementation-proximity-multipass-v4-20260929`.
- Execution: `suite-multipass-v6-20260929`.
- Execution: `suite-multipass-v7-20260929`.
A single server-side job performs:
@ -46,13 +46,22 @@ missing review decisions still fail closed; no missing assignment is fabricated.
Each invocation retains six CLI turns for structured output. Model review passes
and CLI turns are separate counters. All calls use the originally selected provider
and pinned model. The current default is `claude-opus-5-5[1m]`, with canonical runtime
identity checked as `claude-opus-5-5`, firstParty, high effort, reported 1M context
identity checked as `claude-opus-5-5`, firstParty, reported 1M context
and 128K model output ceiling, with requests limited to 64K output tokens.
The server invokes native Claude Code CLI 2.1.285 using
the existing first-party OAuth account, not a separately configured API-key
account. The CLI itself communicates with Anthropic over HTTPS. No new provider
fallback or tools are enabled.
Effort defaults to `high` and increases to `xhigh` for suites with at least 100
cases OR 128 KiB of canonical UTF-8 source JSON (campaign, suite, and all cases).
Preflight selects the effort once for the whole job; every discovery, review,
audit and structured-output repair uses that selection. Source size excludes
transport whitespace, prompts, schemas, and generated proposals. Thresholds are
published in capabilities; selection metadata includes counts, bytes, and triggers.
Existing deadlines and cost guards still apply and never cause a silent effort
reduction. See the [effort policy details](hermes_suite_claude55.md).
Sonnet 5.5 is also verified on the 14-case synthetic fixture and available through
server configuration. See the [matched comparison](hermes_suite_claude55.md).
Earlier acceptance measurements below used Opus 4.8; they do not establish large

View File

@ -84,7 +84,7 @@ class Provider(BaseHTTPRequestHandler):
'complete_system': system, 'max_tokens': body.get('max_tokens'),
'effort': body.get('output_config', {}).get('effort')})
assert complete and system
assert seen[-1]['effort'] == MODELS['claude']['reasoning']
assert seen[-1]['effort'] == active['reasoning']
value = response_value()
# Force one schema rejection; the actual CLI must repair it within turns.
omitted = len(json.loads(active['input'])['suite']['cases']) == 14 and len(seen) == 1

View File

@ -95,10 +95,13 @@ def local_generate(request, cancel, client_ip, *, invocation=None):
"provenance": response.get("inference_provenance")}
def claude_command(model, max_cost):
def claude_command(model, max_cost, *, reasoning=None):
"""Use the native pinned binary, never the privileged Hermes shell wrapper."""
if model not in CLAUDE_MODELS:
raise Problem("unsupported_backend", 422)
reasoning = MODELS["claude"]["reasoning"] if reasoning is None else reasoning
if reasoning not in {"high", "xhigh"}:
raise Problem("unsupported_reasoning", 422)
settings = {"enabledPlugins": {"agents-md@builtin": False,
"cc-plugin-agents-md@builtin": False},
"disableAllHooks": True, "disableBundledSkills": True,
@ -109,7 +112,7 @@ def claude_command(model, max_cost):
"--strict-mcp-config", "--mcp-config", '{"mcpServers":{}}',
"--setting-sources", "", "--settings", encoded(settings).decode(), "--disable-slash-commands",
"--permission-mode", "dontAsk", "--no-chrome",
"--model", model + "[1m]", "--effort", MODELS["claude"]["reasoning"],
"--model", model + "[1m]", "--effort", reasoning,
"--max-budget-usd", str(max_cost),
"--max-turns", str(CLAUDE_MAX_TURNS), "--system-prompt", SYSTEM,
"--json-schema", encoded(SCHEMA).decode()]
@ -226,7 +229,8 @@ def claude_generate(request, cancel, *, invocation=None, progress=None):
with tempfile.TemporaryDirectory(prefix="suite-", dir="/jobs") as directory:
root = Path(directory)
(root / "input").write_text(invocation["input"] if invocation else prompt(request))
command = claude_command(model, request["execution"]["max_cost_usd"])
command = claude_command(model, request["execution"]["max_cost_usd"],
reasoning=invocation.get("reasoning") if invocation else None)
if invocation:
command[command.index("--system-prompt") + 1] = invocation["system"]
command[command.index("--json-schema") + 1] = encoded(invocation["schema"]).decode()

View File

@ -9,7 +9,7 @@ from collections import Counter
REVISION = "suite-v6-20260929"
PROMPT_REVISION = "implementation-proximity-multipass-v4-20260929"
EXECUTION_REVISION = "suite-multipass-v6-20260929"
EXECUTION_REVISION = "suite-multipass-v7-20260929"
CLAUDE_VERSION = "2.1.285"
CLAUDE_MODELS = {
"claude-opus-4-8": 64000,
@ -20,6 +20,13 @@ CLAUDE_MODEL = os.environ.get("PLANNING_CLAUDE_MODEL", "claude-opus-4-8")
if CLAUDE_MODEL not in CLAUDE_MODELS:
raise RuntimeError("Unsupported configured Claude model")
CLAUDE_MAX_TURNS = 6
CLAUDE_REASONING_POLICY = {
"revision": "suite-size-effort-v1-20260929",
"default": "high", "large": "xhigh",
"case_count_at_least": 100, "source_bytes_at_least": 128 * 1024,
"match": "any", "scope": "whole_job",
"source_bytes_scope": "Canonical UTF-8 JSON of campaign, suite, and all cases",
}
MAX_BODY = 1 << 20
MAX_RESULT = 1 << 20
MAX_CASES = 400
@ -37,7 +44,8 @@ MODELS = {
"cli_model": CLAUDE_MODEL + "[1m]",
"output": 64000, "reported_output": CLAUDE_MODELS[CLAUDE_MODEL],
"overhead": 8192, "backend": "claude-code-" + CLAUDE_VERSION,
"enabled": True, "reasoning": "high", "max_turns": CLAUDE_MAX_TURNS},
"enabled": True, "reasoning": "high", "reasoning_policy": CLAUDE_REASONING_POLICY,
"max_turns": CLAUDE_MAX_TURNS},
"codex": {"model": "gpt-6-astra", "context": 258400,
"output": None, "overhead": None, "backend": "codex-subscription-broker",
"enabled": False, "reasoning": "medium",
@ -190,6 +198,23 @@ def prompt(request):
return encoded({k: request[k] for k in ("campaign", "suite", "cases")}).decode()
def reasoning_selection(request, provider):
"""Choose effort from complete source size without a classifier or model call."""
if provider != "claude":
return {"reasoning": MODELS[provider]["reasoning"]}
policy = CLAUDE_REASONING_POLICY
count, size = len(request["cases"]), len(prompt(request).encode("utf-8"))
triggers = []
if count >= policy["case_count_at_least"]:
triggers.append("case_count")
if size >= policy["source_bytes_at_least"]:
triggers.append("source_bytes")
return {"reasoning": policy["large"] if triggers else policy["default"],
"reasoning_selection": {"policy_revision": policy["revision"],
"case_count": count, "source_bytes": size,
"triggers": triggers, "scope": policy["scope"]}}
def preflight(request):
"""Select one permitted provider that can admit the multi-pass policy."""
from suite_multipass import preflight_workflow

View File

@ -8,7 +8,8 @@ import time
import suite_backends
from suite_assignments import decode
from suite_contract import (EXECUTION_REVISION, MAX_BODY, MAX_RESULT, MODELS, PROMPT_REVISION,
PROMPT_SHA256, REVISION, Problem, digest, encoded, validate_partition)
PROMPT_SHA256, REVISION, Problem, digest, encoded, reasoning_selection,
validate_partition)
from suite_policy import (BASE_NAME_LIMIT, MAX_GROUP, POLICY_REVISION, invocation,
validate_natural)
from suite_sizing import cap_families, review_summary
@ -67,7 +68,8 @@ def preflight_workflow(request):
except Problem as exc:
reasons[provider] = exc.code
continue
return {"provider": provider, **model, **initial, "configuration_revision": REVISION,
return {"provider": provider, **model, **initial, **reasoning_selection(request, provider),
"configuration_revision": REVISION,
"prompt_revision": PROMPT_REVISION, "prompt_sha256": PROMPT_SHA256,
"execution_revision": EXECUTION_REVISION, "policy_revision": POLICY_REVISION,
"case_count": count, "source_sha256": digest(request["cases"]),
@ -100,6 +102,7 @@ class Workflow:
def __init__(self, request, selected, cancel, client_ip, progress=None):
self.request, self.provider, self.cancel = request, selected["provider"], cancel
self.reasoning = selected.get("reasoning", MODELS[self.provider]["reasoning"])
self.client_ip, self.progress = client_ip, progress or (lambda _: None)
self.started = time.monotonic()
self.deadline = self.started + request["execution"]["max_seconds"]
@ -124,6 +127,8 @@ class Workflow:
if len(self.records) >= MAX_PASSES:
raise Problem("model_pass_limit", 502)
call = invocation(stage, source, context)
# Keep the preflight choice across independent ordering and larger review prompts.
call["reasoning"] = self.reasoning
limits = capacity(call, self.provider, len(source["cases"]), family_count)
remaining_seconds = self.deadline - time.monotonic()
remaining_cost = self.cost_limit - self.spent
@ -150,6 +155,7 @@ class Workflow:
raise Problem("unsupported_backend", 422)
cost = metadata.get("cost_usd_estimate") if self.provider == "claude" else 0.0
record = {"stage": stage, "provider": self.provider, "model": MODELS[self.provider]["model"],
"reasoning": self.reasoning,
"wall_seconds": round(time.monotonic() - started, 3), **limits,
"system_sha256": hashlib.sha256(call["system"].encode()).hexdigest(),
"schema_sha256": digest(call["schema"]),
@ -194,6 +200,7 @@ class Workflow:
review = review_summary(a, b, reconciled, reviewed, audited, final, divisions)
self.checkpoint()
metadata = {**self.last_metadata, "passes": self.records,
"reasoning": self.reasoning,
"model_pass_count": len(self.records), "usage": sum_usage(self.records),
"cli_diagnostics_scope": "last_model_pass",
"duration_api_ms": sum(r["duration_api_ms"] for r in self.records) if all(type(r["duration_api_ms"]) in (int, float) for r in self.records) else None,

View File

@ -36,7 +36,7 @@ spec:
app: hermes-suite-planner
annotations:
fluentbit.io/exclude: "true"
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v6-20260929
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v7-20260929
vault.hashicorp.com/agent-inject: "true"
vault.hashicorp.com/agent-pre-populate-only: "true"
vault.hashicorp.com/agent-init-first: "true"

View File

@ -10,7 +10,8 @@ SCRIPTS = Path(__file__).resolve().parents[2] / "services/hermes/scripts"
sys.path.insert(0, str(SCRIPTS))
import suite_backends
from suite_cli_diagnostics import usage_counts
from suite_contract import CLAUDE_MODELS, Problem
from suite_contract import (CLAUDE_MODELS, MODELS, Problem, preflight, prompt,
reasoning_selection, validate_request)
@pytest.mark.parametrize("model", CLAUDE_MODELS)
@ -72,3 +73,60 @@ def test_thinking_usage_is_counted_without_retaining_unrecognized_fields():
"""New CLI usage details retain measurements and drop arbitrary content."""
assert usage_counts({"output_tokens_details": {"thinking_tokens": 123, "content": "private"}}) == {
"output_tokens_details": {"thinking_tokens": 123}}
def sized_request(count, source_bytes=None):
"""Build complete synthetic records with optional exact UTF-8 source size."""
value = {"campaign": "SYNTHETIC", "suite": "EFFORT", "cases": [
{"alias": f"CASE-{i:04d}", "description": "Synthetic parser objective."}
for i in range(count)],
"routing": {"allow_external": True, "allowed_external_providers": ["claude"]}}
if source_bytes is not None:
remaining = source_bytes - len(prompt(value).encode("utf-8"))
for case in value["cases"]:
take = min(remaining, 20000)
case["description"] += "\u03bc" * (take // 2) + "x" * (take % 2)
remaining -= take
assert remaining == 0
return validate_request(value, ["claude"])
@pytest.mark.parametrize("count,effort", [(14, "high"), (75, "high"), (99, "high"),
(100, "xhigh"), (363, "xhigh"), (400, "xhigh")])
def test_preflight_selects_effort_at_case_count_boundary(count, effort):
"""Small and large requests choose independently without changing the default."""
selected = preflight(sized_request(count))
assert selected["provider"] == "claude" and selected["reasoning"] == effort
assert selected["reasoning_selection"]["triggers"] == (["case_count"] if count >= 100 else [])
command = suite_backends.claude_command(selected["model"], 30, reasoning=selected["reasoning"])
assert command[command.index("--effort") + 1] == effort
assert MODELS["claude"]["reasoning"] == "high"
@pytest.mark.parametrize("size,effort", [(131071, "high"), (131072, "xhigh"), (131073, "xhigh")])
def test_complete_utf8_source_size_is_an_independent_trigger(size, effort):
"""Use encoded bytes, not characters or transport whitespace, at the boundary."""
value = sized_request(14, size)
selected = preflight(value)
assert selected["reasoning"] == effort
assert selected["reasoning_selection"]["source_bytes"] == size
assert len(prompt(value)) < size
assert selected["reasoning_selection"]["triggers"] == (["source_bytes"] if size >= 131072 else [])
reordered = {**value, "cases": list(reversed(value["cases"]))}
assert reasoning_selection(reordered, "claude") == reasoning_selection(value, "claude")
value["execution"]["max_cost_usd"] = 1
assert reasoning_selection(value, "claude")["reasoning"] == effort
def test_size_policy_cannot_broaden_routing_or_enable_client_effort_override():
"""Effort selection never adds an external destination to a local-only job."""
value = sized_request(363)
value["routing"] = {"allow_external": False, "allowed_external_providers": []}
with pytest.raises(Problem, match="capacity_or_unsupported_backend"):
preflight(value)
assert reasoning_selection(value, "local") == {"reasoning": "none"}
value["execution"]["reasoning"] = "xhigh"
with pytest.raises(Problem, match="invalid_request"):
validate_request(value, ["claude"])
with pytest.raises(Problem, match="unsupported_reasoning"):
suite_backends.claude_command(MODELS["claude"]["model"], 30, reasoning="max")

View File

@ -182,7 +182,7 @@ def test_actual_subprocess_paths_cleanup_and_safe_metadata(tmp_path, monkeypatch
elif mode == "success":
script += "time.sleep(0.25)"
command = ["/not-installed-synthetic-cli"] if mode == "start" else [sys.executable, "-c", script]
monkeypatch.setattr(suite_backends, "claude_command", lambda *_: command)
monkeypatch.setattr(suite_backends, "claude_command", lambda *_, **__: command)
value, cancel = request(), threading.Event()
if mode == "timeout":
value["execution"]["max_seconds"] = 0.02

View File

@ -139,6 +139,21 @@ def test_independent_orders_and_content_are_reproducible(monkeypatch):
assert 'proposal_a' in json.loads(calls[2][1]['input'])['review_material']
@pytest.mark.parametrize('count,effort', [(14, 'high'), (100, 'xhigh')])
def test_preflight_effort_is_preserved_through_all_five_passes(monkeypatch, count, effort):
"""Discovery, reconciliation, review and audit use the same selected effort."""
source = request(count)
family = natural(source, [[c['alias'] for c in source['cases']]])
calls = install_backend(monkeypatch, source, family)
selected = preflight(source)
result, metadata = workflow.generate(source, selected, threading.Event(), '192.168.22.8')
assert selected['reasoning'] == metadata['reasoning'] == effort
assert len(calls) == 5
assert all(call['reasoning'] == effort for _, call in calls)
assert all(p['reasoning'] == effort for p in metadata['passes'])
validate_result(result, source)
def test_progress_reports_activity_without_content_or_false_percentage(monkeypatch):
source = request(7)
family = natural(source, [[c['alias'] for c in source['cases']]])