From a281c996345c0167a751989269f40acbcdd82d36 Mon Sep 17 00:00:00 2001 From: jenkins Date: Tue, 29 Sep 2026 15:57:56 -0500 Subject: [PATCH] hermes: raise suite planning reasoning effort to high --- docs/hermes_suite_claude55.md | 14 ++++++++++++-- docs/hermes_suite_multipass.md | 4 ++-- .../ops/hermes_suite_multipass_transport_probe.py | 14 ++++++++++---- services/hermes/scripts/suite_backends.py | 3 ++- services/hermes/scripts/suite_contract.py | 4 ++-- services/hermes/suite-planner-deployment.yaml | 2 +- 6 files changed, 29 insertions(+), 12 deletions(-) diff --git a/docs/hermes_suite_claude55.md b/docs/hermes_suite_claude55.md index fc434a6c..4c7f69df 100644 --- a/docs/hermes_suite_claude55.md +++ b/docs/hermes_suite_claude55.md @@ -4,9 +4,19 @@ The suite planner now pins native Claude Code 2.1.285. Opus 5.5 and Sonnet 5.5 both returned their exact canonical model identities on the existing first-party OAuth account. The default deployment selects `claude-opus-5-5`; Sonnet is a server configuration option, not an automatic fallback. The HTTPS contract, -credential scopes, implementation objective, medium effort, five-case task cap, +credential scopes, implementation objective, five-case task cap, 30-minute deadline, and USD 30 CLI estimated-cost guard are unchanged. +Execution revision `suite-multipass-v6-20260929` raises effort to `high` for every +new Claude invocation. Preflight `selection.reasoning` and the CLI `--effort` +argument use the same configuration value. The comparison below used `medium`; +its runtime and cost measurements do not measure high effort. Both Opus 5.5 and +Sonnet 5.5 support `low`, `medium`, `high`, `xhigh`, and `max`. High is the next +step above medium; xhigh and max spend more reasoning tokens and may take longer, +without guaranteeing better grouping. No hosted inference or real roster rerun +is needed to apply this setting. The HTTP compatibility and prompt revisions +remain unchanged. + ## Matched 14-case results Each run used the same 9,851-byte normalized synthetic request, complete source @@ -75,7 +85,7 @@ checksum verification leaves the worker unavailable instead of using another CLI The HTTP compatibility revision remains `suite-v6-20260929`, the prompt remains `implementation-proximity-multipass-v4-20260929`, and the execution revision is -`suite-multipass-v5-20260929`. Authenticated capabilities additionally expose +`suite-multipass-v6-20260929`. Authenticated capabilities additionally expose `claude_model_options` and `model_selection: server_configuration`. No new client request field is accepted or required. diff --git a/docs/hermes_suite_multipass.md b/docs/hermes_suite_multipass.md index a6df3a05..000e96fe 100644 --- a/docs/hermes_suite_multipass.md +++ b/docs/hermes_suite_multipass.md @@ -11,7 +11,7 @@ most **64 characters**, unique after whitespace and case normalization. - Configuration: `suite-v6-20260929` (HTTP compatibility identifier). - Policy: `implementation-five-v1-20260929`. - Prompt: `implementation-proximity-multipass-v4-20260929`. -- Execution: `suite-multipass-v5-20260929`. +- Execution: `suite-multipass-v6-20260929`. A single server-side job performs: @@ -46,7 +46,7 @@ missing review decisions still fail closed; no missing assignment is fabricated. Each invocation retains six CLI turns for structured output. Model review passes and CLI turns are separate counters. All calls use the originally selected provider and pinned model. The current default is `claude-opus-5-5[1m]`, with canonical runtime -identity checked as `claude-opus-5-5`, firstParty, medium effort, reported 1M context +identity checked as `claude-opus-5-5`, firstParty, high effort, reported 1M context and 128K model output ceiling, with requests limited to 64K output tokens. The server invokes native Claude Code CLI 2.1.285 using the existing first-party OAuth account, not a separately configured API-key diff --git a/scripts/ops/hermes_suite_multipass_transport_probe.py b/scripts/ops/hermes_suite_multipass_transport_probe.py index cd27eed1..57f9dfe5 100755 --- a/scripts/ops/hermes_suite_multipass_transport_probe.py +++ b/scripts/ops/hermes_suite_multipass_transport_probe.py @@ -4,6 +4,7 @@ Execute inside the planner using Python stdin. No hosted inference is performed. This checks full-field transmission and fresh CLI sessions, not grouping quality. """ +import argparse import json import os from http.server import BaseHTTPRequestHandler, HTTPServer @@ -12,7 +13,7 @@ import threading sys.path.insert(0, os.environ.get('SUITE_PROBE_MODULE_DIR', '/opt/planner')) import suite_backends -from suite_contract import validate_request, validate_result +from suite_contract import MODELS, validate_request, validate_result from suite_multipass import generate, preflight_workflow from suite_synthetic import fixture @@ -80,8 +81,10 @@ class Provider(BaseHTTPRequestHandler): complete = active['input'] in list(strings(body.get('messages', []))) system = any(active['system'] in s for s in strings(body.get('system', []))) seen.append({'stage': active['stage'], 'bytes': len(raw), 'complete_input': complete, - 'complete_system': system, 'max_tokens': body.get('max_tokens')}) + 'complete_system': system, 'max_tokens': body.get('max_tokens'), + 'effort': body.get('output_config', {}).get('effort')}) assert complete and system + assert seen[-1]['effort'] == MODELS['claude']['reasoning'] value = response_value() # Force one schema rejection; the actual CLI must repair it within turns. omitted = len(json.loads(active['input'])['suite']['cases']) == 14 and len(seen) == 1 @@ -89,7 +92,7 @@ class Provider(BaseHTTPRequestHandler): del value['assignments'][next(iter(value['assignments']))] seen[-1]['missing_assignment_injected'] = omitted block = {'type': 'tool_use', 'id': 'mock-output', 'name': 'StructuredOutput', 'input': {}} - message = {'id': 'mock', 'type': 'message', 'role': 'assistant', 'model': 'claude-opus-4-8', + message = {'id': 'mock', 'type': 'message', 'role': 'assistant', 'model': MODELS['claude']['model'], 'content': [], 'stop_reason': None, 'stop_sequence': None, 'usage': {'input_tokens': 10, 'output_tokens': 0}} events = [ @@ -111,6 +114,9 @@ class Provider(BaseHTTPRequestHandler): def main(): """Run complete suites through the installed binary with fake loopback auth.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument('--sizes', type=int, nargs='+', choices=(14, 75, 363), default=[14, 75, 363]) + args = parser.parse_args() server = HTTPServer(('127.0.0.1', 0), Provider) threading.Thread(target=server.serve_forever, daemon=True).start() original_env = suite_backends.claude_environment @@ -129,7 +135,7 @@ def main(): suite_backends.claude_environment = environment suite_backends.claude_generate = backend try: - for size in (14, 75, 363): + for size in args.sizes: seen.clear() request = fixture(size)[0] request['routing'] = {'allow_external': True, 'allowed_external_providers': ['claude']} diff --git a/services/hermes/scripts/suite_backends.py b/services/hermes/scripts/suite_backends.py index d7806f6a..b484d297 100644 --- a/services/hermes/scripts/suite_backends.py +++ b/services/hermes/scripts/suite_backends.py @@ -109,7 +109,8 @@ def claude_command(model, max_cost): "--strict-mcp-config", "--mcp-config", '{"mcpServers":{}}', "--setting-sources", "", "--settings", encoded(settings).decode(), "--disable-slash-commands", "--permission-mode", "dontAsk", "--no-chrome", - "--model", model + "[1m]", "--effort", "medium", "--max-budget-usd", str(max_cost), + "--model", model + "[1m]", "--effort", MODELS["claude"]["reasoning"], + "--max-budget-usd", str(max_cost), "--max-turns", str(CLAUDE_MAX_TURNS), "--system-prompt", SYSTEM, "--json-schema", encoded(SCHEMA).decode()] diff --git a/services/hermes/scripts/suite_contract.py b/services/hermes/scripts/suite_contract.py index 8eccc997..a80f988c 100644 --- a/services/hermes/scripts/suite_contract.py +++ b/services/hermes/scripts/suite_contract.py @@ -9,7 +9,7 @@ from collections import Counter REVISION = "suite-v6-20260929" PROMPT_REVISION = "implementation-proximity-multipass-v4-20260929" -EXECUTION_REVISION = "suite-multipass-v5-20260929" +EXECUTION_REVISION = "suite-multipass-v6-20260929" CLAUDE_VERSION = "2.1.285" CLAUDE_MODELS = { "claude-opus-4-8": 64000, @@ -37,7 +37,7 @@ MODELS = { "cli_model": CLAUDE_MODEL + "[1m]", "output": 64000, "reported_output": CLAUDE_MODELS[CLAUDE_MODEL], "overhead": 8192, "backend": "claude-code-" + CLAUDE_VERSION, - "enabled": True, "reasoning": "medium", "max_turns": CLAUDE_MAX_TURNS}, + "enabled": True, "reasoning": "high", "max_turns": CLAUDE_MAX_TURNS}, "codex": {"model": "gpt-6-astra", "context": 258400, "output": None, "overhead": None, "backend": "codex-subscription-broker", "enabled": False, "reasoning": "medium", diff --git a/services/hermes/suite-planner-deployment.yaml b/services/hermes/suite-planner-deployment.yaml index ec0b56d8..31dc4a48 100644 --- a/services/hermes/suite-planner-deployment.yaml +++ b/services/hermes/suite-planner-deployment.yaml @@ -36,7 +36,7 @@ spec: app: hermes-suite-planner annotations: fluentbit.io/exclude: "true" - ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v5-20260929 + ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v6-20260929 vault.hashicorp.com/agent-inject: "true" vault.hashicorp.com/agent-pre-populate-only: "true" vault.hashicorp.com/agent-init-first: "true"