From 3036064bf7ad46fb5b2c7a02207e9f050698d766 Mon Sep 17 00:00:00 2001 From: jenkins Date: Tue, 29 Sep 2026 14:56:18 -0500 Subject: [PATCH] hermes: raise suite job limits to thirty minutes and thirty dollars --- .../suite_planning_request.schema.json | 8 ++++---- docs/hermes_suite_multipass.md | 10 ++++++++-- docs/hermes_suite_planning.md | 18 +++++++++--------- services/hermes/scripts/suite_contract.py | 6 +++--- services/hermes/suite-planner-deployment.yaml | 2 +- testing/tests/test_suite_multipass.py | 14 +++++++------- 6 files changed, 32 insertions(+), 26 deletions(-) diff --git a/docs/contracts/suite_planning_request.schema.json b/docs/contracts/suite_planning_request.schema.json index 493e88eb..0b73f224 100644 --- a/docs/contracts/suite_planning_request.schema.json +++ b/docs/contracts/suite_planning_request.schema.json @@ -206,14 +206,14 @@ "max_seconds": { "type": "integer", "minimum": 10, - "maximum": 1200, - "default": 1200 + "maximum": 1800, + "default": 1800 }, "max_cost_usd": { "type": "number", "exclusiveMinimum": 0, - "maximum": 10, - "default": 10, + "maximum": 30, + "default": 30, "description": "CLI estimated-cost guard, not subscription billing. Shared across all passes." } } diff --git a/docs/hermes_suite_multipass.md b/docs/hermes_suite_multipass.md index 304535c6..eef5b4f4 100644 --- a/docs/hermes_suite_multipass.md +++ b/docs/hermes_suite_multipass.md @@ -11,7 +11,7 @@ most **64 characters**, unique after whitespace and case normalization. - Configuration: `suite-v6-20260929` (HTTP compatibility identifier). - Policy: `implementation-five-v1-20260929`. - Prompt: `implementation-proximity-multipass-v4-20260929`. -- Execution: `suite-multipass-v3-20260929`. +- Execution: `suite-multipass-v4-20260929`. A single server-side job performs: @@ -136,7 +136,7 @@ invented percentage, because different passes have very different runtimes. ## Shared bounds and failure behavior -The user-approved 1200-second maximum job deadline and USD 10 CLI estimated-cost +The user-approved 1800-second maximum job deadline and USD 30 CLI estimated-cost guard apply to **all calls combined**, including routing time. These are also the defaults when omitted. Explicit lower client limits remain effective. This guard uses Claude CLI estimated-cost accounting; it is not subscription billing. Each invocation receives @@ -312,3 +312,9 @@ measurements are in [the acceptance evidence](evidence/hermes_suite_multipass_20 Routine planner logs contained only allowed operational fields, the completed job's SQLite metadata contained no result/review/case content, and temporary CLI job directories were empty after completion. + +Execution revision `suite-multipass-v4-20260929` raises only the shared job defaults +and maximums to 1800 seconds and USD 30 in CLI estimate accounting. Explicit lower +client limits remain effective. The existing request fields are unchanged. No +inference or regression tests were rerun for this limit-only update, as requested; +the acceptance measurements above retain their original revisions and bounds. diff --git a/docs/hermes_suite_planning.md b/docs/hermes_suite_planning.md index f593ae4c..3053b091 100644 --- a/docs/hermes_suite_planning.md +++ b/docs/hermes_suite_planning.md @@ -13,8 +13,8 @@ The endpoint and request fields are unchanged. Optional top-level `review_summar is returned only with the authorized result. Configuration remains `suite-v6-20260929`; policy, prompt, and execution revisions identify the changed grouping behavior. The prior single-pass acceptance results below are historical and do not measure -the new workflow. The new process shares the user-approved 1200-second deadline -and USD 10 CLI estimate guard across all invocations. This is not subscription +the new workflow. The new process shares the user-approved 1800-second deadline +and USD 30 CLI estimate guard across all invocations. This is not subscription billing. Progress updates every five seconds during a CLI pass. The separate local-model endpoint is unchanged; the 8K local route cannot admit the new multi-pass reconciliation schema. @@ -200,7 +200,7 @@ LOCAL_ONLY_FIELD: token SYNTHETIC_EXTERNAL_FIELD: synthetic_token APPROVED_OPERATIONAL_FIELD: operational_token INITIAL_CONCURRENCY: 1; busy submissions return 429; no waiting queue -JOB_TIMEOUT: up to 1200 seconds +JOB_TIMEOUT: up to 1800 seconds HTTP_TIMEOUT: client 45 seconds; submit/status do not wait for inference TLS: existing worker.bstein.dev certificate; normal trusted CA verification ``` @@ -253,8 +253,8 @@ UTF-8 byte limits, unique aliases, ownership, permissions, and exact result cove }, "execution": { "strategy": "whole_suite", - "max_seconds": 1200, - "max_cost_usd": 10 + "max_seconds": 1800, + "max_cost_usd": 30 } } ``` @@ -273,8 +273,8 @@ Missing routing policy means local-only. External provider names are `claude` an unknown providers, malformed booleans, duplicate names, and permission escalation are rejected before routing. `allow_external=false` cannot include providers. `whole_suite` is the only strategy. No batching, summarization, or reconciliation -is silently substituted. `max_seconds` is 10-1200. The Claude CLI guard is at most -USD 10 in its estimated usage accounting; this is not a verified subscription +is silently substituted. `max_seconds` is 10-1800. The Claude CLI guard is at most +USD 30 in its estimated usage accounting; this is not a verified subscription billing ceiling and can overshoot within a single provider call. The gateway deterministically selects a fitting permitted backend, preferring @@ -464,7 +464,7 @@ Small local requests reserve 1,024 overhead tokens and 2,048 output tokens withi 8,192 context. The realistic 14/75/363 fixtures exceed this conservative local whole-suite budget and must use the approved hosted path or fail preflight. -All CLI invocations share at most 1200 seconds per suite job; cancellation and timeouts terminate its +All CLI invocations share at most 1800 seconds per suite job; cancellation and timeouts terminate its process group. Async HTTP calls finish promptly, with a 30-second body-read timeout, 40-second ingress response-header timeout, and recommended 45-second client timeout. Switchyard's ten-minute internal request limit carries only a tiny immediate routing @@ -515,7 +515,7 @@ import json with open('synthetic-suite.json') as stream: request = json.load(stream) request['routing'] = {'allow_external': True, 'allowed_external_providers': ['claude']} -request['execution'] = {'strategy': 'whole_suite', 'max_seconds': 1200, 'max_cost_usd': 10} +request['execution'] = {'strategy': 'whole_suite', 'max_seconds': 1800, 'max_cost_usd': 30} with open('synthetic-request.json', 'w') as stream: json.dump(request, stream) PY diff --git a/services/hermes/scripts/suite_contract.py b/services/hermes/scripts/suite_contract.py index 67347f93..404e3446 100644 --- a/services/hermes/scripts/suite_contract.py +++ b/services/hermes/scripts/suite_contract.py @@ -8,13 +8,13 @@ from collections import Counter REVISION = "suite-v6-20260929" PROMPT_REVISION = "implementation-proximity-multipass-v4-20260929" -EXECUTION_REVISION = "suite-multipass-v3-20260929" +EXECUTION_REVISION = "suite-multipass-v4-20260929" CLAUDE_MAX_TURNS = 6 MAX_BODY = 1 << 20 MAX_RESULT = 1 << 20 MAX_CASES = 400 -TIMEOUT = 1200 -COST_LIMIT = 10.0 +TIMEOUT = 1800 +COST_LIMIT = 30.0 RESULT_TTL = 3600 FIELDS = {"description", "success_criteria", "preconditions", "operating_condition", "case_type", "verification_method", "target", "swci", "verifies", diff --git a/services/hermes/suite-planner-deployment.yaml b/services/hermes/suite-planner-deployment.yaml index e9ebd9e4..66177a1b 100644 --- a/services/hermes/suite-planner-deployment.yaml +++ b/services/hermes/suite-planner-deployment.yaml @@ -36,7 +36,7 @@ spec: app: hermes-suite-planner annotations: fluentbit.io/exclude: "true" - ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v3-20260929 + ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v4-20260929 vault.hashicorp.com/agent-inject: "true" vault.hashicorp.com/agent-pre-populate-only: "true" vault.hashicorp.com/agent-init-first: "true" diff --git a/testing/tests/test_suite_multipass.py b/testing/tests/test_suite_multipass.py index b93e0933..c822ac78 100644 --- a/testing/tests/test_suite_multipass.py +++ b/testing/tests/test_suite_multipass.py @@ -119,7 +119,7 @@ def test_coherent_families_are_balanced_not_semantically_fragmented(monkeypatch, assert [g['name'] for g in result['groups']] == [f'Reset recovery ({i}/{len(sizes)})' for i in range(1,len(sizes)+1)] assert all('Work-size part' in g['description'] for g in result['groups']) assert metadata['review_summary']['decision_audit'][0]['decision'] == 'keep' - assert [call[0]['execution']['max_cost_usd'] for call in calls] == pytest.approx([10,9.9,9.8,9.7,9.6]) + assert [call[0]['execution']['max_cost_usd'] for call in calls] == pytest.approx([30,29.9,29.8,29.7,29.6]) assert all(calls[i+1][0]['execution']['max_seconds'] < calls[i][0]['execution']['max_seconds'] for i in range(4)) validate_result(result, source) @@ -149,28 +149,28 @@ def test_progress_reports_activity_without_content_or_false_percentage(monkeypat assert len(active) == 5 assert [v['completed_model_passes'] for v in active] == [0,1,2,3,4] assert all(v['maximum_model_passes'] == 5 and v['cli_output_bytes'] == 120 for v in active) - assert all(v['heartbeat_at'] > 0 and v['job_remaining_seconds'] <= 1200 for v in active) + assert all(v['heartbeat_at'] > 0 and v['job_remaining_seconds'] <= 1800 for v in active) assert all(v['last_cli_activity_seconds_ago'] == 0 for v in active) assert updates[-1]['cli_running'] is False assert CANARY not in json.dumps(updates) -def test_twenty_minute_job_limit_retains_smaller_client_deadlines(): +def test_thirty_minute_job_limit_retains_smaller_client_deadlines(): source = request(7) - assert source['execution']['max_seconds'] == 1200 + assert source['execution']['max_seconds'] == 1800 source['execution']['max_seconds'] = 900 assert validate_request(source, ['claude'])['execution']['max_seconds'] == 900 - source['execution']['max_seconds'] = 1201 + source['execution']['max_seconds'] = 1801 with pytest.raises(Problem, match='invalid_timeout'): validate_request(source, ['claude']) def test_cli_estimate_guard_default_and_client_override(): source = request(7) - assert source['execution']['max_cost_usd'] == 10 + assert source['execution']['max_cost_usd'] == 30 source['execution']['max_cost_usd'] = 5 assert validate_request(source, ['claude'])['execution']['max_cost_usd'] == 5 - source['execution']['max_cost_usd'] = 10.1 + source['execution']['max_cost_usd'] = 30.1 with pytest.raises(Problem, match='invalid_cost_limit'): validate_request(source, ['claude'])