hermes: raise suite planning reasoning effort to high

This commit is contained in:
jenkins 2026-09-29 15:57:56 -05:00
parent bb4c0b6951
commit a281c99634
6 changed files with 29 additions and 12 deletions

View File

@ -4,9 +4,19 @@ The suite planner now pins native Claude Code 2.1.285. Opus 5.5 and Sonnet 5.5
both returned their exact canonical model identities on the existing first-party
OAuth account. The default deployment selects `claude-opus-5-5`; Sonnet is a
server configuration option, not an automatic fallback. The HTTPS contract,
credential scopes, implementation objective, medium effort, five-case task cap,
credential scopes, implementation objective, five-case task cap,
30-minute deadline, and USD 30 CLI estimated-cost guard are unchanged.
Execution revision `suite-multipass-v6-20260929` raises effort to `high` for every
new Claude invocation. Preflight `selection.reasoning` and the CLI `--effort`
argument use the same configuration value. The comparison below used `medium`;
its runtime and cost measurements do not measure high effort. Both Opus 5.5 and
Sonnet 5.5 support `low`, `medium`, `high`, `xhigh`, and `max`. High is the next
step above medium; xhigh and max spend more reasoning tokens and may take longer,
without guaranteeing better grouping. No hosted inference or real roster rerun
is needed to apply this setting. The HTTP compatibility and prompt revisions
remain unchanged.
## Matched 14-case results
Each run used the same 9,851-byte normalized synthetic request, complete source
@ -75,7 +85,7 @@ checksum verification leaves the worker unavailable instead of using another CLI
The HTTP compatibility revision remains `suite-v6-20260929`, the prompt remains
`implementation-proximity-multipass-v4-20260929`, and the execution revision is
`suite-multipass-v5-20260929`. Authenticated capabilities additionally expose
`suite-multipass-v6-20260929`. Authenticated capabilities additionally expose
`claude_model_options` and `model_selection: server_configuration`. No new client
request field is accepted or required.

View File

@ -11,7 +11,7 @@ most **64 characters**, unique after whitespace and case normalization.
- Configuration: `suite-v6-20260929` (HTTP compatibility identifier).
- Policy: `implementation-five-v1-20260929`.
- Prompt: `implementation-proximity-multipass-v4-20260929`.
- Execution: `suite-multipass-v5-20260929`.
- Execution: `suite-multipass-v6-20260929`.
A single server-side job performs:
@ -46,7 +46,7 @@ missing review decisions still fail closed; no missing assignment is fabricated.
Each invocation retains six CLI turns for structured output. Model review passes
and CLI turns are separate counters. All calls use the originally selected provider
and pinned model. The current default is `claude-opus-5-5[1m]`, with canonical runtime
identity checked as `claude-opus-5-5`, firstParty, medium effort, reported 1M context
identity checked as `claude-opus-5-5`, firstParty, high effort, reported 1M context
and 128K model output ceiling, with requests limited to 64K output tokens.
The server invokes native Claude Code CLI 2.1.285 using
the existing first-party OAuth account, not a separately configured API-key

View File

@ -4,6 +4,7 @@
Execute inside the planner using Python stdin. No hosted inference is performed.
This checks full-field transmission and fresh CLI sessions, not grouping quality.
"""
import argparse
import json
import os
from http.server import BaseHTTPRequestHandler, HTTPServer
@ -12,7 +13,7 @@ import threading
sys.path.insert(0, os.environ.get('SUITE_PROBE_MODULE_DIR', '/opt/planner'))
import suite_backends
from suite_contract import validate_request, validate_result
from suite_contract import MODELS, validate_request, validate_result
from suite_multipass import generate, preflight_workflow
from suite_synthetic import fixture
@ -80,8 +81,10 @@ class Provider(BaseHTTPRequestHandler):
complete = active['input'] in list(strings(body.get('messages', [])))
system = any(active['system'] in s for s in strings(body.get('system', [])))
seen.append({'stage': active['stage'], 'bytes': len(raw), 'complete_input': complete,
'complete_system': system, 'max_tokens': body.get('max_tokens')})
'complete_system': system, 'max_tokens': body.get('max_tokens'),
'effort': body.get('output_config', {}).get('effort')})
assert complete and system
assert seen[-1]['effort'] == MODELS['claude']['reasoning']
value = response_value()
# Force one schema rejection; the actual CLI must repair it within turns.
omitted = len(json.loads(active['input'])['suite']['cases']) == 14 and len(seen) == 1
@ -89,7 +92,7 @@ class Provider(BaseHTTPRequestHandler):
del value['assignments'][next(iter(value['assignments']))]
seen[-1]['missing_assignment_injected'] = omitted
block = {'type': 'tool_use', 'id': 'mock-output', 'name': 'StructuredOutput', 'input': {}}
message = {'id': 'mock', 'type': 'message', 'role': 'assistant', 'model': 'claude-opus-4-8',
message = {'id': 'mock', 'type': 'message', 'role': 'assistant', 'model': MODELS['claude']['model'],
'content': [], 'stop_reason': None, 'stop_sequence': None,
'usage': {'input_tokens': 10, 'output_tokens': 0}}
events = [
@ -111,6 +114,9 @@ class Provider(BaseHTTPRequestHandler):
def main():
"""Run complete suites through the installed binary with fake loopback auth."""
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--sizes', type=int, nargs='+', choices=(14, 75, 363), default=[14, 75, 363])
args = parser.parse_args()
server = HTTPServer(('127.0.0.1', 0), Provider)
threading.Thread(target=server.serve_forever, daemon=True).start()
original_env = suite_backends.claude_environment
@ -129,7 +135,7 @@ def main():
suite_backends.claude_environment = environment
suite_backends.claude_generate = backend
try:
for size in (14, 75, 363):
for size in args.sizes:
seen.clear()
request = fixture(size)[0]
request['routing'] = {'allow_external': True, 'allowed_external_providers': ['claude']}

View File

@ -109,7 +109,8 @@ def claude_command(model, max_cost):
"--strict-mcp-config", "--mcp-config", '{"mcpServers":{}}',
"--setting-sources", "", "--settings", encoded(settings).decode(), "--disable-slash-commands",
"--permission-mode", "dontAsk", "--no-chrome",
"--model", model + "[1m]", "--effort", "medium", "--max-budget-usd", str(max_cost),
"--model", model + "[1m]", "--effort", MODELS["claude"]["reasoning"],
"--max-budget-usd", str(max_cost),
"--max-turns", str(CLAUDE_MAX_TURNS), "--system-prompt", SYSTEM,
"--json-schema", encoded(SCHEMA).decode()]

View File

@ -9,7 +9,7 @@ from collections import Counter
REVISION = "suite-v6-20260929"
PROMPT_REVISION = "implementation-proximity-multipass-v4-20260929"
EXECUTION_REVISION = "suite-multipass-v5-20260929"
EXECUTION_REVISION = "suite-multipass-v6-20260929"
CLAUDE_VERSION = "2.1.285"
CLAUDE_MODELS = {
"claude-opus-4-8": 64000,
@ -37,7 +37,7 @@ MODELS = {
"cli_model": CLAUDE_MODEL + "[1m]",
"output": 64000, "reported_output": CLAUDE_MODELS[CLAUDE_MODEL],
"overhead": 8192, "backend": "claude-code-" + CLAUDE_VERSION,
"enabled": True, "reasoning": "medium", "max_turns": CLAUDE_MAX_TURNS},
"enabled": True, "reasoning": "high", "max_turns": CLAUDE_MAX_TURNS},
"codex": {"model": "gpt-6-astra", "context": 258400,
"output": None, "overhead": None, "backend": "codex-subscription-broker",
"enabled": False, "reasoning": "medium",

View File

@ -36,7 +36,7 @@ spec:
app: hermes-suite-planner
annotations:
fluentbit.io/exclude: "true"
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v5-20260929
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v6-20260929
vault.hashicorp.com/agent-inject: "true"
vault.hashicorp.com/agent-pre-populate-only: "true"
vault.hashicorp.com/agent-init-first: "true"