hermes: raise suite planning reasoning effort to high
This commit is contained in:
parent
bb4c0b6951
commit
a281c99634
@ -4,9 +4,19 @@ The suite planner now pins native Claude Code 2.1.285. Opus 5.5 and Sonnet 5.5
|
||||
both returned their exact canonical model identities on the existing first-party
|
||||
OAuth account. The default deployment selects `claude-opus-5-5`; Sonnet is a
|
||||
server configuration option, not an automatic fallback. The HTTPS contract,
|
||||
credential scopes, implementation objective, medium effort, five-case task cap,
|
||||
credential scopes, implementation objective, five-case task cap,
|
||||
30-minute deadline, and USD 30 CLI estimated-cost guard are unchanged.
|
||||
|
||||
Execution revision `suite-multipass-v6-20260929` raises effort to `high` for every
|
||||
new Claude invocation. Preflight `selection.reasoning` and the CLI `--effort`
|
||||
argument use the same configuration value. The comparison below used `medium`;
|
||||
its runtime and cost measurements do not measure high effort. Both Opus 5.5 and
|
||||
Sonnet 5.5 support `low`, `medium`, `high`, `xhigh`, and `max`. High is the next
|
||||
step above medium; xhigh and max spend more reasoning tokens and may take longer,
|
||||
without guaranteeing better grouping. No hosted inference or real roster rerun
|
||||
is needed to apply this setting. The HTTP compatibility and prompt revisions
|
||||
remain unchanged.
|
||||
|
||||
## Matched 14-case results
|
||||
|
||||
Each run used the same 9,851-byte normalized synthetic request, complete source
|
||||
@ -75,7 +85,7 @@ checksum verification leaves the worker unavailable instead of using another CLI
|
||||
|
||||
The HTTP compatibility revision remains `suite-v6-20260929`, the prompt remains
|
||||
`implementation-proximity-multipass-v4-20260929`, and the execution revision is
|
||||
`suite-multipass-v5-20260929`. Authenticated capabilities additionally expose
|
||||
`suite-multipass-v6-20260929`. Authenticated capabilities additionally expose
|
||||
`claude_model_options` and `model_selection: server_configuration`. No new client
|
||||
request field is accepted or required.
|
||||
|
||||
|
||||
@ -11,7 +11,7 @@ most **64 characters**, unique after whitespace and case normalization.
|
||||
- Configuration: `suite-v6-20260929` (HTTP compatibility identifier).
|
||||
- Policy: `implementation-five-v1-20260929`.
|
||||
- Prompt: `implementation-proximity-multipass-v4-20260929`.
|
||||
- Execution: `suite-multipass-v5-20260929`.
|
||||
- Execution: `suite-multipass-v6-20260929`.
|
||||
|
||||
A single server-side job performs:
|
||||
|
||||
@ -46,7 +46,7 @@ missing review decisions still fail closed; no missing assignment is fabricated.
|
||||
Each invocation retains six CLI turns for structured output. Model review passes
|
||||
and CLI turns are separate counters. All calls use the originally selected provider
|
||||
and pinned model. The current default is `claude-opus-5-5[1m]`, with canonical runtime
|
||||
identity checked as `claude-opus-5-5`, firstParty, medium effort, reported 1M context
|
||||
identity checked as `claude-opus-5-5`, firstParty, high effort, reported 1M context
|
||||
and 128K model output ceiling, with requests limited to 64K output tokens.
|
||||
The server invokes native Claude Code CLI 2.1.285 using
|
||||
the existing first-party OAuth account, not a separately configured API-key
|
||||
|
||||
@ -4,6 +4,7 @@
|
||||
Execute inside the planner using Python stdin. No hosted inference is performed.
|
||||
This checks full-field transmission and fresh CLI sessions, not grouping quality.
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
from http.server import BaseHTTPRequestHandler, HTTPServer
|
||||
@ -12,7 +13,7 @@ import threading
|
||||
|
||||
sys.path.insert(0, os.environ.get('SUITE_PROBE_MODULE_DIR', '/opt/planner'))
|
||||
import suite_backends
|
||||
from suite_contract import validate_request, validate_result
|
||||
from suite_contract import MODELS, validate_request, validate_result
|
||||
from suite_multipass import generate, preflight_workflow
|
||||
from suite_synthetic import fixture
|
||||
|
||||
@ -80,8 +81,10 @@ class Provider(BaseHTTPRequestHandler):
|
||||
complete = active['input'] in list(strings(body.get('messages', [])))
|
||||
system = any(active['system'] in s for s in strings(body.get('system', [])))
|
||||
seen.append({'stage': active['stage'], 'bytes': len(raw), 'complete_input': complete,
|
||||
'complete_system': system, 'max_tokens': body.get('max_tokens')})
|
||||
'complete_system': system, 'max_tokens': body.get('max_tokens'),
|
||||
'effort': body.get('output_config', {}).get('effort')})
|
||||
assert complete and system
|
||||
assert seen[-1]['effort'] == MODELS['claude']['reasoning']
|
||||
value = response_value()
|
||||
# Force one schema rejection; the actual CLI must repair it within turns.
|
||||
omitted = len(json.loads(active['input'])['suite']['cases']) == 14 and len(seen) == 1
|
||||
@ -89,7 +92,7 @@ class Provider(BaseHTTPRequestHandler):
|
||||
del value['assignments'][next(iter(value['assignments']))]
|
||||
seen[-1]['missing_assignment_injected'] = omitted
|
||||
block = {'type': 'tool_use', 'id': 'mock-output', 'name': 'StructuredOutput', 'input': {}}
|
||||
message = {'id': 'mock', 'type': 'message', 'role': 'assistant', 'model': 'claude-opus-4-8',
|
||||
message = {'id': 'mock', 'type': 'message', 'role': 'assistant', 'model': MODELS['claude']['model'],
|
||||
'content': [], 'stop_reason': None, 'stop_sequence': None,
|
||||
'usage': {'input_tokens': 10, 'output_tokens': 0}}
|
||||
events = [
|
||||
@ -111,6 +114,9 @@ class Provider(BaseHTTPRequestHandler):
|
||||
|
||||
def main():
|
||||
"""Run complete suites through the installed binary with fake loopback auth."""
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument('--sizes', type=int, nargs='+', choices=(14, 75, 363), default=[14, 75, 363])
|
||||
args = parser.parse_args()
|
||||
server = HTTPServer(('127.0.0.1', 0), Provider)
|
||||
threading.Thread(target=server.serve_forever, daemon=True).start()
|
||||
original_env = suite_backends.claude_environment
|
||||
@ -129,7 +135,7 @@ def main():
|
||||
suite_backends.claude_environment = environment
|
||||
suite_backends.claude_generate = backend
|
||||
try:
|
||||
for size in (14, 75, 363):
|
||||
for size in args.sizes:
|
||||
seen.clear()
|
||||
request = fixture(size)[0]
|
||||
request['routing'] = {'allow_external': True, 'allowed_external_providers': ['claude']}
|
||||
|
||||
@ -109,7 +109,8 @@ def claude_command(model, max_cost):
|
||||
"--strict-mcp-config", "--mcp-config", '{"mcpServers":{}}',
|
||||
"--setting-sources", "", "--settings", encoded(settings).decode(), "--disable-slash-commands",
|
||||
"--permission-mode", "dontAsk", "--no-chrome",
|
||||
"--model", model + "[1m]", "--effort", "medium", "--max-budget-usd", str(max_cost),
|
||||
"--model", model + "[1m]", "--effort", MODELS["claude"]["reasoning"],
|
||||
"--max-budget-usd", str(max_cost),
|
||||
"--max-turns", str(CLAUDE_MAX_TURNS), "--system-prompt", SYSTEM,
|
||||
"--json-schema", encoded(SCHEMA).decode()]
|
||||
|
||||
|
||||
@ -9,7 +9,7 @@ from collections import Counter
|
||||
|
||||
REVISION = "suite-v6-20260929"
|
||||
PROMPT_REVISION = "implementation-proximity-multipass-v4-20260929"
|
||||
EXECUTION_REVISION = "suite-multipass-v5-20260929"
|
||||
EXECUTION_REVISION = "suite-multipass-v6-20260929"
|
||||
CLAUDE_VERSION = "2.1.285"
|
||||
CLAUDE_MODELS = {
|
||||
"claude-opus-4-8": 64000,
|
||||
@ -37,7 +37,7 @@ MODELS = {
|
||||
"cli_model": CLAUDE_MODEL + "[1m]",
|
||||
"output": 64000, "reported_output": CLAUDE_MODELS[CLAUDE_MODEL],
|
||||
"overhead": 8192, "backend": "claude-code-" + CLAUDE_VERSION,
|
||||
"enabled": True, "reasoning": "medium", "max_turns": CLAUDE_MAX_TURNS},
|
||||
"enabled": True, "reasoning": "high", "max_turns": CLAUDE_MAX_TURNS},
|
||||
"codex": {"model": "gpt-6-astra", "context": 258400,
|
||||
"output": None, "overhead": None, "backend": "codex-subscription-broker",
|
||||
"enabled": False, "reasoning": "medium",
|
||||
|
||||
@ -36,7 +36,7 @@ spec:
|
||||
app: hermes-suite-planner
|
||||
annotations:
|
||||
fluentbit.io/exclude: "true"
|
||||
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v5-20260929
|
||||
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v6-20260929
|
||||
vault.hashicorp.com/agent-inject: "true"
|
||||
vault.hashicorp.com/agent-pre-populate-only: "true"
|
||||
vault.hashicorp.com/agent-init-first: "true"
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user