From ce3025fb9cb8e31ca79528dfd18c1d861fd6e2bb Mon Sep 17 00:00:00 2001 From: jenkins Date: Tue, 29 Sep 2026 16:05:30 -0500 Subject: [PATCH] hermes: select xhigh reasoning for large suite jobs --- docs/hermes_suite_claude55.md | 24 ++++++-- docs/hermes_suite_multipass.md | 13 +++- .../hermes_suite_multipass_transport_probe.py | 2 +- services/hermes/scripts/suite_backends.py | 10 +++- services/hermes/scripts/suite_contract.py | 29 ++++++++- services/hermes/scripts/suite_multipass.py | 11 +++- services/hermes/suite-planner-deployment.yaml | 2 +- testing/tests/test_suite_claude_models.py | 60 ++++++++++++++++++- testing/tests/test_suite_cli_diagnostics.py | 2 +- testing/tests/test_suite_multipass.py | 15 +++++ 10 files changed, 150 insertions(+), 18 deletions(-) diff --git a/docs/hermes_suite_claude55.md b/docs/hermes_suite_claude55.md index 4c7f69df..54ab3398 100644 --- a/docs/hermes_suite_claude55.md +++ b/docs/hermes_suite_claude55.md @@ -7,10 +7,24 @@ server configuration option, not an automatic fallback. The HTTPS contract, credential scopes, implementation objective, five-case task cap, 30-minute deadline, and USD 30 CLI estimated-cost guard are unchanged. -Execution revision `suite-multipass-v6-20260929` raises effort to `high` for every -new Claude invocation. Preflight `selection.reasoning` and the CLI `--effort` -argument use the same configuration value. The comparison below used `medium`; -its runtime and cost measurements do not measure high effort. Both Opus 5.5 and +Execution revision `suite-multipass-v7-20260929` uses `high` by default and +`xhigh` for at least 100 cases OR at least 131,072 bytes (128 KiB) of complete +suite content. Byte size is the canonical UTF-8 JSON of `campaign`, `suite`, and +all `cases`, without transport whitespace, routing/execution fields, prompt text, +schemas, or generated proposals. The selected effort is fixed for every pass in +that job, including structured-output repair; later review prompts do not change +it. There is no classifier or auxiliary inference call to choose effort. + +Capabilities expose the thresholds in `models.claude.reasoning_policy`. +Preflight and job metadata expose `selection.reasoning` plus +`selection.reasoning_selection` with the policy revision, measured source bytes, +case count, and triggered thresholds. Per-pass metadata and completed results +also contain `reasoning`. The policy revision is `suite-size-effort-v1-20260929`. +The client does not need a new request field. Explicit smaller job budgets remain +effective; the service fails instead of reducing effort or accepting partial work. + +The comparison below used `medium`; its runtime and cost measurements do not +measure high or xhigh effort. Both Opus 5.5 and Sonnet 5.5 support `low`, `medium`, `high`, `xhigh`, and `max`. High is the next step above medium; xhigh and max spend more reasoning tokens and may take longer, without guaranteeing better grouping. No hosted inference or real roster rerun @@ -85,7 +99,7 @@ checksum verification leaves the worker unavailable instead of using another CLI The HTTP compatibility revision remains `suite-v6-20260929`, the prompt remains `implementation-proximity-multipass-v4-20260929`, and the execution revision is -`suite-multipass-v6-20260929`. Authenticated capabilities additionally expose +`suite-multipass-v7-20260929`. Authenticated capabilities additionally expose `claude_model_options` and `model_selection: server_configuration`. No new client request field is accepted or required. diff --git a/docs/hermes_suite_multipass.md b/docs/hermes_suite_multipass.md index 000e96fe..db57e3f4 100644 --- a/docs/hermes_suite_multipass.md +++ b/docs/hermes_suite_multipass.md @@ -11,7 +11,7 @@ most **64 characters**, unique after whitespace and case normalization. - Configuration: `suite-v6-20260929` (HTTP compatibility identifier). - Policy: `implementation-five-v1-20260929`. - Prompt: `implementation-proximity-multipass-v4-20260929`. -- Execution: `suite-multipass-v6-20260929`. +- Execution: `suite-multipass-v7-20260929`. A single server-side job performs: @@ -46,13 +46,22 @@ missing review decisions still fail closed; no missing assignment is fabricated. Each invocation retains six CLI turns for structured output. Model review passes and CLI turns are separate counters. All calls use the originally selected provider and pinned model. The current default is `claude-opus-5-5[1m]`, with canonical runtime -identity checked as `claude-opus-5-5`, firstParty, high effort, reported 1M context +identity checked as `claude-opus-5-5`, firstParty, reported 1M context and 128K model output ceiling, with requests limited to 64K output tokens. The server invokes native Claude Code CLI 2.1.285 using the existing first-party OAuth account, not a separately configured API-key account. The CLI itself communicates with Anthropic over HTTPS. No new provider fallback or tools are enabled. +Effort defaults to `high` and increases to `xhigh` for suites with at least 100 +cases OR 128 KiB of canonical UTF-8 source JSON (campaign, suite, and all cases). +Preflight selects the effort once for the whole job; every discovery, review, +audit and structured-output repair uses that selection. Source size excludes +transport whitespace, prompts, schemas, and generated proposals. Thresholds are +published in capabilities; selection metadata includes counts, bytes, and triggers. +Existing deadlines and cost guards still apply and never cause a silent effort +reduction. See the [effort policy details](hermes_suite_claude55.md). + Sonnet 5.5 is also verified on the 14-case synthetic fixture and available through server configuration. See the [matched comparison](hermes_suite_claude55.md). Earlier acceptance measurements below used Opus 4.8; they do not establish large diff --git a/scripts/ops/hermes_suite_multipass_transport_probe.py b/scripts/ops/hermes_suite_multipass_transport_probe.py index 57f9dfe5..7b43b2e0 100755 --- a/scripts/ops/hermes_suite_multipass_transport_probe.py +++ b/scripts/ops/hermes_suite_multipass_transport_probe.py @@ -84,7 +84,7 @@ class Provider(BaseHTTPRequestHandler): 'complete_system': system, 'max_tokens': body.get('max_tokens'), 'effort': body.get('output_config', {}).get('effort')}) assert complete and system - assert seen[-1]['effort'] == MODELS['claude']['reasoning'] + assert seen[-1]['effort'] == active['reasoning'] value = response_value() # Force one schema rejection; the actual CLI must repair it within turns. omitted = len(json.loads(active['input'])['suite']['cases']) == 14 and len(seen) == 1 diff --git a/services/hermes/scripts/suite_backends.py b/services/hermes/scripts/suite_backends.py index b484d297..4eb28ed9 100644 --- a/services/hermes/scripts/suite_backends.py +++ b/services/hermes/scripts/suite_backends.py @@ -95,10 +95,13 @@ def local_generate(request, cancel, client_ip, *, invocation=None): "provenance": response.get("inference_provenance")} -def claude_command(model, max_cost): +def claude_command(model, max_cost, *, reasoning=None): """Use the native pinned binary, never the privileged Hermes shell wrapper.""" if model not in CLAUDE_MODELS: raise Problem("unsupported_backend", 422) + reasoning = MODELS["claude"]["reasoning"] if reasoning is None else reasoning + if reasoning not in {"high", "xhigh"}: + raise Problem("unsupported_reasoning", 422) settings = {"enabledPlugins": {"agents-md@builtin": False, "cc-plugin-agents-md@builtin": False}, "disableAllHooks": True, "disableBundledSkills": True, @@ -109,7 +112,7 @@ def claude_command(model, max_cost): "--strict-mcp-config", "--mcp-config", '{"mcpServers":{}}', "--setting-sources", "", "--settings", encoded(settings).decode(), "--disable-slash-commands", "--permission-mode", "dontAsk", "--no-chrome", - "--model", model + "[1m]", "--effort", MODELS["claude"]["reasoning"], + "--model", model + "[1m]", "--effort", reasoning, "--max-budget-usd", str(max_cost), "--max-turns", str(CLAUDE_MAX_TURNS), "--system-prompt", SYSTEM, "--json-schema", encoded(SCHEMA).decode()] @@ -226,7 +229,8 @@ def claude_generate(request, cancel, *, invocation=None, progress=None): with tempfile.TemporaryDirectory(prefix="suite-", dir="/jobs") as directory: root = Path(directory) (root / "input").write_text(invocation["input"] if invocation else prompt(request)) - command = claude_command(model, request["execution"]["max_cost_usd"]) + command = claude_command(model, request["execution"]["max_cost_usd"], + reasoning=invocation.get("reasoning") if invocation else None) if invocation: command[command.index("--system-prompt") + 1] = invocation["system"] command[command.index("--json-schema") + 1] = encoded(invocation["schema"]).decode() diff --git a/services/hermes/scripts/suite_contract.py b/services/hermes/scripts/suite_contract.py index a80f988c..9814e2fb 100644 --- a/services/hermes/scripts/suite_contract.py +++ b/services/hermes/scripts/suite_contract.py @@ -9,7 +9,7 @@ from collections import Counter REVISION = "suite-v6-20260929" PROMPT_REVISION = "implementation-proximity-multipass-v4-20260929" -EXECUTION_REVISION = "suite-multipass-v6-20260929" +EXECUTION_REVISION = "suite-multipass-v7-20260929" CLAUDE_VERSION = "2.1.285" CLAUDE_MODELS = { "claude-opus-4-8": 64000, @@ -20,6 +20,13 @@ CLAUDE_MODEL = os.environ.get("PLANNING_CLAUDE_MODEL", "claude-opus-4-8") if CLAUDE_MODEL not in CLAUDE_MODELS: raise RuntimeError("Unsupported configured Claude model") CLAUDE_MAX_TURNS = 6 +CLAUDE_REASONING_POLICY = { + "revision": "suite-size-effort-v1-20260929", + "default": "high", "large": "xhigh", + "case_count_at_least": 100, "source_bytes_at_least": 128 * 1024, + "match": "any", "scope": "whole_job", + "source_bytes_scope": "Canonical UTF-8 JSON of campaign, suite, and all cases", +} MAX_BODY = 1 << 20 MAX_RESULT = 1 << 20 MAX_CASES = 400 @@ -37,7 +44,8 @@ MODELS = { "cli_model": CLAUDE_MODEL + "[1m]", "output": 64000, "reported_output": CLAUDE_MODELS[CLAUDE_MODEL], "overhead": 8192, "backend": "claude-code-" + CLAUDE_VERSION, - "enabled": True, "reasoning": "high", "max_turns": CLAUDE_MAX_TURNS}, + "enabled": True, "reasoning": "high", "reasoning_policy": CLAUDE_REASONING_POLICY, + "max_turns": CLAUDE_MAX_TURNS}, "codex": {"model": "gpt-6-astra", "context": 258400, "output": None, "overhead": None, "backend": "codex-subscription-broker", "enabled": False, "reasoning": "medium", @@ -190,6 +198,23 @@ def prompt(request): return encoded({k: request[k] for k in ("campaign", "suite", "cases")}).decode() +def reasoning_selection(request, provider): + """Choose effort from complete source size without a classifier or model call.""" + if provider != "claude": + return {"reasoning": MODELS[provider]["reasoning"]} + policy = CLAUDE_REASONING_POLICY + count, size = len(request["cases"]), len(prompt(request).encode("utf-8")) + triggers = [] + if count >= policy["case_count_at_least"]: + triggers.append("case_count") + if size >= policy["source_bytes_at_least"]: + triggers.append("source_bytes") + return {"reasoning": policy["large"] if triggers else policy["default"], + "reasoning_selection": {"policy_revision": policy["revision"], + "case_count": count, "source_bytes": size, + "triggers": triggers, "scope": policy["scope"]}} + + def preflight(request): """Select one permitted provider that can admit the multi-pass policy.""" from suite_multipass import preflight_workflow diff --git a/services/hermes/scripts/suite_multipass.py b/services/hermes/scripts/suite_multipass.py index 07b6f46a..1de7964c 100644 --- a/services/hermes/scripts/suite_multipass.py +++ b/services/hermes/scripts/suite_multipass.py @@ -8,7 +8,8 @@ import time import suite_backends from suite_assignments import decode from suite_contract import (EXECUTION_REVISION, MAX_BODY, MAX_RESULT, MODELS, PROMPT_REVISION, - PROMPT_SHA256, REVISION, Problem, digest, encoded, validate_partition) + PROMPT_SHA256, REVISION, Problem, digest, encoded, reasoning_selection, + validate_partition) from suite_policy import (BASE_NAME_LIMIT, MAX_GROUP, POLICY_REVISION, invocation, validate_natural) from suite_sizing import cap_families, review_summary @@ -67,7 +68,8 @@ def preflight_workflow(request): except Problem as exc: reasons[provider] = exc.code continue - return {"provider": provider, **model, **initial, "configuration_revision": REVISION, + return {"provider": provider, **model, **initial, **reasoning_selection(request, provider), + "configuration_revision": REVISION, "prompt_revision": PROMPT_REVISION, "prompt_sha256": PROMPT_SHA256, "execution_revision": EXECUTION_REVISION, "policy_revision": POLICY_REVISION, "case_count": count, "source_sha256": digest(request["cases"]), @@ -100,6 +102,7 @@ class Workflow: def __init__(self, request, selected, cancel, client_ip, progress=None): self.request, self.provider, self.cancel = request, selected["provider"], cancel + self.reasoning = selected.get("reasoning", MODELS[self.provider]["reasoning"]) self.client_ip, self.progress = client_ip, progress or (lambda _: None) self.started = time.monotonic() self.deadline = self.started + request["execution"]["max_seconds"] @@ -124,6 +127,8 @@ class Workflow: if len(self.records) >= MAX_PASSES: raise Problem("model_pass_limit", 502) call = invocation(stage, source, context) + # Keep the preflight choice across independent ordering and larger review prompts. + call["reasoning"] = self.reasoning limits = capacity(call, self.provider, len(source["cases"]), family_count) remaining_seconds = self.deadline - time.monotonic() remaining_cost = self.cost_limit - self.spent @@ -150,6 +155,7 @@ class Workflow: raise Problem("unsupported_backend", 422) cost = metadata.get("cost_usd_estimate") if self.provider == "claude" else 0.0 record = {"stage": stage, "provider": self.provider, "model": MODELS[self.provider]["model"], + "reasoning": self.reasoning, "wall_seconds": round(time.monotonic() - started, 3), **limits, "system_sha256": hashlib.sha256(call["system"].encode()).hexdigest(), "schema_sha256": digest(call["schema"]), @@ -194,6 +200,7 @@ class Workflow: review = review_summary(a, b, reconciled, reviewed, audited, final, divisions) self.checkpoint() metadata = {**self.last_metadata, "passes": self.records, + "reasoning": self.reasoning, "model_pass_count": len(self.records), "usage": sum_usage(self.records), "cli_diagnostics_scope": "last_model_pass", "duration_api_ms": sum(r["duration_api_ms"] for r in self.records) if all(type(r["duration_api_ms"]) in (int, float) for r in self.records) else None, diff --git a/services/hermes/suite-planner-deployment.yaml b/services/hermes/suite-planner-deployment.yaml index 31dc4a48..1a4aae16 100644 --- a/services/hermes/suite-planner-deployment.yaml +++ b/services/hermes/suite-planner-deployment.yaml @@ -36,7 +36,7 @@ spec: app: hermes-suite-planner annotations: fluentbit.io/exclude: "true" - ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v6-20260929 + ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v7-20260929 vault.hashicorp.com/agent-inject: "true" vault.hashicorp.com/agent-pre-populate-only: "true" vault.hashicorp.com/agent-init-first: "true" diff --git a/testing/tests/test_suite_claude_models.py b/testing/tests/test_suite_claude_models.py index 859d8d13..652cd33f 100644 --- a/testing/tests/test_suite_claude_models.py +++ b/testing/tests/test_suite_claude_models.py @@ -10,7 +10,8 @@ SCRIPTS = Path(__file__).resolve().parents[2] / "services/hermes/scripts" sys.path.insert(0, str(SCRIPTS)) import suite_backends from suite_cli_diagnostics import usage_counts -from suite_contract import CLAUDE_MODELS, Problem +from suite_contract import (CLAUDE_MODELS, MODELS, Problem, preflight, prompt, + reasoning_selection, validate_request) @pytest.mark.parametrize("model", CLAUDE_MODELS) @@ -72,3 +73,60 @@ def test_thinking_usage_is_counted_without_retaining_unrecognized_fields(): """New CLI usage details retain measurements and drop arbitrary content.""" assert usage_counts({"output_tokens_details": {"thinking_tokens": 123, "content": "private"}}) == { "output_tokens_details": {"thinking_tokens": 123}} + + +def sized_request(count, source_bytes=None): + """Build complete synthetic records with optional exact UTF-8 source size.""" + value = {"campaign": "SYNTHETIC", "suite": "EFFORT", "cases": [ + {"alias": f"CASE-{i:04d}", "description": "Synthetic parser objective."} + for i in range(count)], + "routing": {"allow_external": True, "allowed_external_providers": ["claude"]}} + if source_bytes is not None: + remaining = source_bytes - len(prompt(value).encode("utf-8")) + for case in value["cases"]: + take = min(remaining, 20000) + case["description"] += "\u03bc" * (take // 2) + "x" * (take % 2) + remaining -= take + assert remaining == 0 + return validate_request(value, ["claude"]) + + +@pytest.mark.parametrize("count,effort", [(14, "high"), (75, "high"), (99, "high"), + (100, "xhigh"), (363, "xhigh"), (400, "xhigh")]) +def test_preflight_selects_effort_at_case_count_boundary(count, effort): + """Small and large requests choose independently without changing the default.""" + selected = preflight(sized_request(count)) + assert selected["provider"] == "claude" and selected["reasoning"] == effort + assert selected["reasoning_selection"]["triggers"] == (["case_count"] if count >= 100 else []) + command = suite_backends.claude_command(selected["model"], 30, reasoning=selected["reasoning"]) + assert command[command.index("--effort") + 1] == effort + assert MODELS["claude"]["reasoning"] == "high" + + +@pytest.mark.parametrize("size,effort", [(131071, "high"), (131072, "xhigh"), (131073, "xhigh")]) +def test_complete_utf8_source_size_is_an_independent_trigger(size, effort): + """Use encoded bytes, not characters or transport whitespace, at the boundary.""" + value = sized_request(14, size) + selected = preflight(value) + assert selected["reasoning"] == effort + assert selected["reasoning_selection"]["source_bytes"] == size + assert len(prompt(value)) < size + assert selected["reasoning_selection"]["triggers"] == (["source_bytes"] if size >= 131072 else []) + reordered = {**value, "cases": list(reversed(value["cases"]))} + assert reasoning_selection(reordered, "claude") == reasoning_selection(value, "claude") + value["execution"]["max_cost_usd"] = 1 + assert reasoning_selection(value, "claude")["reasoning"] == effort + + +def test_size_policy_cannot_broaden_routing_or_enable_client_effort_override(): + """Effort selection never adds an external destination to a local-only job.""" + value = sized_request(363) + value["routing"] = {"allow_external": False, "allowed_external_providers": []} + with pytest.raises(Problem, match="capacity_or_unsupported_backend"): + preflight(value) + assert reasoning_selection(value, "local") == {"reasoning": "none"} + value["execution"]["reasoning"] = "xhigh" + with pytest.raises(Problem, match="invalid_request"): + validate_request(value, ["claude"]) + with pytest.raises(Problem, match="unsupported_reasoning"): + suite_backends.claude_command(MODELS["claude"]["model"], 30, reasoning="max") diff --git a/testing/tests/test_suite_cli_diagnostics.py b/testing/tests/test_suite_cli_diagnostics.py index 39a720fd..c8b9dceb 100644 --- a/testing/tests/test_suite_cli_diagnostics.py +++ b/testing/tests/test_suite_cli_diagnostics.py @@ -182,7 +182,7 @@ def test_actual_subprocess_paths_cleanup_and_safe_metadata(tmp_path, monkeypatch elif mode == "success": script += "time.sleep(0.25)" command = ["/not-installed-synthetic-cli"] if mode == "start" else [sys.executable, "-c", script] - monkeypatch.setattr(suite_backends, "claude_command", lambda *_: command) + monkeypatch.setattr(suite_backends, "claude_command", lambda *_, **__: command) value, cancel = request(), threading.Event() if mode == "timeout": value["execution"]["max_seconds"] = 0.02 diff --git a/testing/tests/test_suite_multipass.py b/testing/tests/test_suite_multipass.py index c822ac78..4ffa9fbf 100644 --- a/testing/tests/test_suite_multipass.py +++ b/testing/tests/test_suite_multipass.py @@ -139,6 +139,21 @@ def test_independent_orders_and_content_are_reproducible(monkeypatch): assert 'proposal_a' in json.loads(calls[2][1]['input'])['review_material'] +@pytest.mark.parametrize('count,effort', [(14, 'high'), (100, 'xhigh')]) +def test_preflight_effort_is_preserved_through_all_five_passes(monkeypatch, count, effort): + """Discovery, reconciliation, review and audit use the same selected effort.""" + source = request(count) + family = natural(source, [[c['alias'] for c in source['cases']]]) + calls = install_backend(monkeypatch, source, family) + selected = preflight(source) + result, metadata = workflow.generate(source, selected, threading.Event(), '192.168.22.8') + assert selected['reasoning'] == metadata['reasoning'] == effort + assert len(calls) == 5 + assert all(call['reasoning'] == effort for _, call in calls) + assert all(p['reasoning'] == effort for p in metadata['passes']) + validate_result(result, source) + + def test_progress_reports_activity_without_content_or_false_percentage(monkeypatch): source = request(7) family = natural(source, [[c['alias'] for c in source['cases']]])