From 83446c23dce13dcd7780a45612d03eeac3088407 Mon Sep 17 00:00:00 2001 From: jenkins Date: Tue, 29 Sep 2026 23:01:36 -0500 Subject: [PATCH] fix(hermes): allow explicit one-hour suite job budgets --- .../suite_planning_request.schema.json | 2 +- .../hermes_suite_budget_20260929.json | 1377 +++++++++++++++++ docs/hermes_suite_claude55.md | 8 +- docs/hermes_suite_job_budget.md | 129 ++ docs/hermes_suite_multipass.md | 17 +- docs/hermes_suite_planning.md | 53 +- scripts/ops/hermes_suite_budget_probe.py | 110 ++ .../hermes_suite_multipass_transport_probe.py | 5 +- services/hermes/scripts/suite_api.py | 8 +- services/hermes/scripts/suite_backends.py | 7 +- services/hermes/scripts/suite_contract.py | 7 +- services/hermes/scripts/suite_jobs.py | 15 +- services/hermes/scripts/suite_multipass.py | 13 +- services/hermes/suite-planner-deployment.yaml | 2 +- testing/tests/test_suite_cli_diagnostics.py | 19 +- testing/tests/test_suite_deadlines.py | 119 ++ testing/tests/test_suite_multipass.py | 8 +- testing/tests/test_suite_planning.py | 4 +- 18 files changed, 1837 insertions(+), 66 deletions(-) create mode 100644 docs/evidence/hermes_suite_budget_20260929.json create mode 100644 docs/hermes_suite_job_budget.md create mode 100755 scripts/ops/hermes_suite_budget_probe.py create mode 100644 testing/tests/test_suite_deadlines.py diff --git a/docs/contracts/suite_planning_request.schema.json b/docs/contracts/suite_planning_request.schema.json index 0b73f224..c676e358 100644 --- a/docs/contracts/suite_planning_request.schema.json +++ b/docs/contracts/suite_planning_request.schema.json @@ -206,7 +206,7 @@ "max_seconds": { "type": "integer", "minimum": 10, - "maximum": 1800, + "maximum": 3600, "default": 1800 }, "max_cost_usd": { diff --git a/docs/evidence/hermes_suite_budget_20260929.json b/docs/evidence/hermes_suite_budget_20260929.json new file mode 100644 index 00000000..3cafc259 --- /dev/null +++ b/docs/evidence/hermes_suite_budget_20260929.json @@ -0,0 +1,1377 @@ +{ + "failed_real_job_safe_metadata": { + "completed_pass_usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 1131975, + "output_tokens": 214423, + "output_tokens_details": { + "thinking_tokens": 153293 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "cost_limit_usd_estimate": 30.0, + "cost_used_usd_estimate": 8.81636, + "error_code": "timeout", + "failed_stage": "large_family_review", + "interrupted_pass_cost_usd_estimate": null, + "interrupted_pass_timeout_seconds": 60.72140585165471, + "interrupted_pass_usage": null, + "job_id": "9af0db7c71b04882a3a2c0449e83b74c", + "passes": [ + { + "allocated_seconds": 1799.9636647687294, + "cli_turns": 3, + "cost_usd_estimate": 2.635496, + "model": "claude-opus-5-5", + "reasoning": "xhigh", + "stage": "proposal_a", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 369479, + "output_tokens": 57879, + "output_tokens_details": { + "thinking_tokens": 41501 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 505.343 + }, + { + "allocated_seconds": 1294.6144117447548, + "cli_turns": 3, + "cost_usd_estimate": 2.415476, + "model": "claude-opus-5-5", + "reasoning": "xhigh", + "stage": "proposal_b", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 364164, + "output_tokens": 47941, + "output_tokens_details": { + "thinking_tokens": 34819 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 385.938 + }, + { + "allocated_seconds": 908.6719134976156, + "cli_turns": 3, + "cost_usd_estimate": 3.7653879999999997, + "model": "claude-opus-5-5", + "reasoning": "xhigh", + "stage": "reconciliation", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 398332, + "output_tokens": 108603, + "output_tokens_details": { + "thinking_tokens": 76973 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 847.944 + } + ], + "status": "failed", + "wall_seconds": 1800.064 + }, + "regression_tests": { + "accelerated_clock_budget_seconds": 3600, + "accelerated_clock_completed_seconds": 2510, + "passed": 163 + }, + "scope": "Safe operational metadata only. No real job rerun. Benchmark is synthetic 75 cases at high effort; real 363-case job was xhigh.", + "synthetic_async_benchmark": { + "automatic_job_retries": 0, + "capabilities_limits": { + "default_max_seconds": 1800, + "max_cost_usd": 30.0, + "max_seconds": 3600 + }, + "case_count": 75, + "configuration_revision": "suite-v6-20260929", + "counts": { + "final_singletons": 3, + "final_tasks": 21, + "natural_families": 9, + "natural_singletons": 3 + }, + "exact_final_coverage": true, + "execution_revision": "suite-multipass-v9-20260929", + "http_timeout_seconds": 45, + "idempotent_replay_verified": true, + "job_metadata": { + "attempted_destinations": [ + "switchyard:atlas/planning/claude", + "claude:claude-opus-5-5" + ], + "cli_diagnostics": { + "api_error_status": null, + "assistant_api_error_seen": false, + "assistant_error_code": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "cli_transport_error": null, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.29248, + "duration_api_ms": 43762, + "execution_revision": "suite-multipass-v9-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "high", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 3442.3193380017765, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 42965, + "output_tokens": 6031, + "output_tokens_details": { + "thinking_tokens": 477 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_diagnostics_scope": "last_model_pass", + "compaction": false, + "compaction_signal": "CLI events and disabled compaction", + "configuration_revision": "suite-v6-20260929", + "cost_usd_estimate": 1.449292, + "created_at": 1790740611.8458564, + "duration_api_ms": 199463, + "execution": { + "max_cost_usd": 30.0, + "max_seconds": 3600, + "strategy": "whole_suite" + }, + "execution_progress": { + "cli_running": false, + "completed_model_passes": 5, + "cost_limit_usd_estimate": 30.0, + "cost_used_usd_estimate": 1.449292, + "current_pass": "decision_audit", + "heartbeat_at": 1790740813.7992806, + "job_elapsed_seconds": 201.953, + "job_remaining_seconds": 3398.0, + "maximum_model_passes": 5, + "pass_elapsed_seconds": 44.3, + "passes": [ + { + "allocated_cost_usd": 30.0, + "allocated_seconds": 3599.965945621021, + "case_order_sha256": "86166f136829af715c2a8400aa4f3672b0f608690ef6b4eb47b20bd69da1dbf6", + "cli_diagnostics": { + "api_error_status": null, + "assistant_api_error_seen": false, + "assistant_error_code": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "cli_transport_error": null, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.132708, + "duration_api_ms": 15041, + "execution_revision": "suite-multipass-v9-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "high", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 3599.965945621021, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 23737, + "output_tokens": 1888, + "output_tokens_details": { + "thinking_tokens": 299 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 2, + "cost_usd_estimate": 0.132708, + "duration_api_ms": 15041, + "input_bytes": 66341, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 74533, + "input_token_count": null, + "model": "claude-opus-5-5", + "output_reservation_tokens": 18816, + "output_reservation_verified": false, + "provider": "claude", + "reasoning": "high", + "schema_sha256": "60f11f8f897d645c1dce01f904172111879802cb3112daba65d9a6d5117861c5", + "stage": "proposal_a", + "system_sha256": "5df3f59dd1c4182d42fef8b58db69f0b18a9ff19ebaa7217363622484a29a04a", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 23737, + "output_tokens": 1888, + "output_tokens_details": { + "thinking_tokens": 299 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 15.56 + }, + { + "allocated_cost_usd": 29.867292, + "allocated_seconds": 3584.4048600518145, + "case_order_sha256": "cfaae948234b695780c2e876572ed0b62699ecbd7818fc4dbc66ac8dbe1582cf", + "cli_diagnostics": { + "api_error_status": null, + "assistant_api_error_seen": false, + "assistant_error_code": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "cli_transport_error": null, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.273548, + "duration_api_ms": 28369, + "execution_revision": "suite-multipass-v9-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "high", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 3584.4048600518145, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 3, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 49672, + "output_tokens": 3743, + "output_tokens_details": { + "thinking_tokens": 451 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 3, + "cost_usd_estimate": 0.273548, + "duration_api_ms": 28369, + "input_bytes": 66341, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 74533, + "input_token_count": null, + "model": "claude-opus-5-5", + "output_reservation_tokens": 18816, + "output_reservation_verified": false, + "provider": "claude", + "reasoning": "high", + "schema_sha256": "60f11f8f897d645c1dce01f904172111879802cb3112daba65d9a6d5117861c5", + "stage": "proposal_b", + "system_sha256": "5df3f59dd1c4182d42fef8b58db69f0b18a9ff19ebaa7217363622484a29a04a", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 49672, + "output_tokens": 3743, + "output_tokens_details": { + "thinking_tokens": 451 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 28.815 + }, + { + "allocated_cost_usd": 29.593744, + "allocated_seconds": 3555.5887848697603, + "case_order_sha256": "86166f136829af715c2a8400aa4f3672b0f608690ef6b4eb47b20bd69da1dbf6", + "cli_diagnostics": { + "api_error_status": null, + "assistant_api_error_seen": false, + "assistant_error_code": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "cli_transport_error": null, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.475608, + "duration_api_ms": 67661, + "execution_revision": "suite-multipass-v9-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "high", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 3555.5887848697603, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 3, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 70267, + "output_tokens": 9727, + "output_tokens_details": { + "thinking_tokens": 851 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 3, + "cost_usd_estimate": 0.475608, + "duration_api_ms": 67661, + "input_bytes": 86996, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 95188, + "input_token_count": null, + "model": "claude-opus-5-5", + "output_reservation_tokens": 16272, + "output_reservation_verified": false, + "provider": "claude", + "reasoning": "high", + "schema_sha256": "e110dabf89a21e52e7f697f0a5dfec4b0cac375862617af22f57773cab61a9b5", + "stage": "reconciliation", + "system_sha256": "7d85f607438ad80803f5f7a519b61355387886ed2f2cf7ab3671354628ffc6d5", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 70267, + "output_tokens": 9727, + "output_tokens_details": { + "thinking_tokens": 851 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 68.171 + }, + { + "allocated_cost_usd": 29.118136, + "allocated_seconds": 3487.413067241665, + "case_order_sha256": "86166f136829af715c2a8400aa4f3672b0f608690ef6b4eb47b20bd69da1dbf6", + "cli_diagnostics": { + "api_error_status": null, + "assistant_api_error_seen": false, + "assistant_error_code": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "cli_transport_error": null, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.27494799999999997, + "duration_api_ms": 44630, + "execution_revision": "suite-multipass-v9-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "high", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 3487.413067241665, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 37117, + "output_tokens": 6324, + "output_tokens_details": { + "thinking_tokens": 745 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 2, + "cost_usd_estimate": 0.27494799999999997, + "duration_api_ms": 44630, + "input_bytes": 97593, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 105785, + "input_token_count": null, + "model": "claude-opus-5-5", + "output_reservation_tokens": 16272, + "output_reservation_verified": false, + "provider": "claude", + "reasoning": "high", + "schema_sha256": "e0a213b4ae38c1220617d24273e83e5437ebdee96677ab8d0667736faa248c27", + "stage": "large_family_review", + "system_sha256": "ba3c856dfc67482e1b35841388b27e19087832942f757519449397ef7636a238", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 37117, + "output_tokens": 6324, + "output_tokens_details": { + "thinking_tokens": 745 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 45.09 + }, + { + "allocated_cost_usd": 28.843188, + "allocated_seconds": 3442.3193380017765, + "case_order_sha256": "86166f136829af715c2a8400aa4f3672b0f608690ef6b4eb47b20bd69da1dbf6", + "cli_diagnostics": { + "api_error_status": null, + "assistant_api_error_seen": false, + "assistant_error_code": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "cli_transport_error": null, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.29248, + "duration_api_ms": 43762, + "execution_revision": "suite-multipass-v9-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "high", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 3442.3193380017765, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 42965, + "output_tokens": 6031, + "output_tokens_details": { + "thinking_tokens": 477 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 2, + "cost_usd_estimate": 0.29248, + "duration_api_ms": 43762, + "input_bytes": 112731, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 120923, + "input_token_count": null, + "model": "claude-opus-5-5", + "output_reservation_tokens": 16272, + "output_reservation_verified": false, + "provider": "claude", + "reasoning": "high", + "schema_sha256": "e0a213b4ae38c1220617d24273e83e5437ebdee96677ab8d0667736faa248c27", + "stage": "decision_audit", + "system_sha256": "a3123bd136a1416160db1c17d29c1a63ff42c7fe49656bab5d0e5b5fd709d2e8", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 42965, + "output_tokens": 6031, + "output_tokens_details": { + "thinking_tokens": 477 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 44.266 + } + ] + }, + "execution_revision": "suite-multipass-v9-20260929", + "final_task_count": 21, + "job_id": "df1759e9836f4471bb8706cb600929e1", + "model": "claude-opus-5-5", + "model_pass_count": 5, + "model_usage": { + "claude-opus-5-5[1m]": { + "canonicalModel": "claude-opus-5-5", + "contextWindow": 1000000, + "maxOutputTokens": 128000, + "provider": "firstParty" + } + }, + "natural_family_count": 9, + "passes": [ + { + "allocated_cost_usd": 30.0, + "allocated_seconds": 3599.965945621021, + "case_order_sha256": "86166f136829af715c2a8400aa4f3672b0f608690ef6b4eb47b20bd69da1dbf6", + "cli_diagnostics": { + "api_error_status": null, + "assistant_api_error_seen": false, + "assistant_error_code": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "cli_transport_error": null, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.132708, + "duration_api_ms": 15041, + "execution_revision": "suite-multipass-v9-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "high", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 3599.965945621021, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 23737, + "output_tokens": 1888, + "output_tokens_details": { + "thinking_tokens": 299 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 2, + "cost_usd_estimate": 0.132708, + "duration_api_ms": 15041, + "input_bytes": 66341, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 74533, + "input_token_count": null, + "model": "claude-opus-5-5", + "output_reservation_tokens": 18816, + "output_reservation_verified": false, + "provider": "claude", + "reasoning": "high", + "schema_sha256": "60f11f8f897d645c1dce01f904172111879802cb3112daba65d9a6d5117861c5", + "stage": "proposal_a", + "system_sha256": "5df3f59dd1c4182d42fef8b58db69f0b18a9ff19ebaa7217363622484a29a04a", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 23737, + "output_tokens": 1888, + "output_tokens_details": { + "thinking_tokens": 299 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 15.56 + }, + { + "allocated_cost_usd": 29.867292, + "allocated_seconds": 3584.4048600518145, + "case_order_sha256": "cfaae948234b695780c2e876572ed0b62699ecbd7818fc4dbc66ac8dbe1582cf", + "cli_diagnostics": { + "api_error_status": null, + "assistant_api_error_seen": false, + "assistant_error_code": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "cli_transport_error": null, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.273548, + "duration_api_ms": 28369, + "execution_revision": "suite-multipass-v9-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "high", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 3584.4048600518145, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 3, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 49672, + "output_tokens": 3743, + "output_tokens_details": { + "thinking_tokens": 451 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 3, + "cost_usd_estimate": 0.273548, + "duration_api_ms": 28369, + "input_bytes": 66341, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 74533, + "input_token_count": null, + "model": "claude-opus-5-5", + "output_reservation_tokens": 18816, + "output_reservation_verified": false, + "provider": "claude", + "reasoning": "high", + "schema_sha256": "60f11f8f897d645c1dce01f904172111879802cb3112daba65d9a6d5117861c5", + "stage": "proposal_b", + "system_sha256": "5df3f59dd1c4182d42fef8b58db69f0b18a9ff19ebaa7217363622484a29a04a", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 49672, + "output_tokens": 3743, + "output_tokens_details": { + "thinking_tokens": 451 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 28.815 + }, + { + "allocated_cost_usd": 29.593744, + "allocated_seconds": 3555.5887848697603, + "case_order_sha256": "86166f136829af715c2a8400aa4f3672b0f608690ef6b4eb47b20bd69da1dbf6", + "cli_diagnostics": { + "api_error_status": null, + "assistant_api_error_seen": false, + "assistant_error_code": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "cli_transport_error": null, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.475608, + "duration_api_ms": 67661, + "execution_revision": "suite-multipass-v9-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "high", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 3555.5887848697603, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 3, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 70267, + "output_tokens": 9727, + "output_tokens_details": { + "thinking_tokens": 851 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 3, + "cost_usd_estimate": 0.475608, + "duration_api_ms": 67661, + "input_bytes": 86996, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 95188, + "input_token_count": null, + "model": "claude-opus-5-5", + "output_reservation_tokens": 16272, + "output_reservation_verified": false, + "provider": "claude", + "reasoning": "high", + "schema_sha256": "e110dabf89a21e52e7f697f0a5dfec4b0cac375862617af22f57773cab61a9b5", + "stage": "reconciliation", + "system_sha256": "7d85f607438ad80803f5f7a519b61355387886ed2f2cf7ab3671354628ffc6d5", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 70267, + "output_tokens": 9727, + "output_tokens_details": { + "thinking_tokens": 851 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 68.171 + }, + { + "allocated_cost_usd": 29.118136, + "allocated_seconds": 3487.413067241665, + "case_order_sha256": "86166f136829af715c2a8400aa4f3672b0f608690ef6b4eb47b20bd69da1dbf6", + "cli_diagnostics": { + "api_error_status": null, + "assistant_api_error_seen": false, + "assistant_error_code": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "cli_transport_error": null, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.27494799999999997, + "duration_api_ms": 44630, + "execution_revision": "suite-multipass-v9-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "high", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 3487.413067241665, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 37117, + "output_tokens": 6324, + "output_tokens_details": { + "thinking_tokens": 745 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 2, + "cost_usd_estimate": 0.27494799999999997, + "duration_api_ms": 44630, + "input_bytes": 97593, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 105785, + "input_token_count": null, + "model": "claude-opus-5-5", + "output_reservation_tokens": 16272, + "output_reservation_verified": false, + "provider": "claude", + "reasoning": "high", + "schema_sha256": "e0a213b4ae38c1220617d24273e83e5437ebdee96677ab8d0667736faa248c27", + "stage": "large_family_review", + "system_sha256": "ba3c856dfc67482e1b35841388b27e19087832942f757519449397ef7636a238", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 37117, + "output_tokens": 6324, + "output_tokens_details": { + "thinking_tokens": 745 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 45.09 + }, + { + "allocated_cost_usd": 28.843188, + "allocated_seconds": 3442.3193380017765, + "case_order_sha256": "86166f136829af715c2a8400aa4f3672b0f608690ef6b4eb47b20bd69da1dbf6", + "cli_diagnostics": { + "api_error_status": null, + "assistant_api_error_seen": false, + "assistant_error_code": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "cli_transport_error": null, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.29248, + "duration_api_ms": 43762, + "execution_revision": "suite-multipass-v9-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "high", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 3442.3193380017765, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 42965, + "output_tokens": 6031, + "output_tokens_details": { + "thinking_tokens": 477 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 2, + "cost_usd_estimate": 0.29248, + "duration_api_ms": 43762, + "input_bytes": 112731, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 120923, + "input_token_count": null, + "model": "claude-opus-5-5", + "output_reservation_tokens": 16272, + "output_reservation_verified": false, + "provider": "claude", + "reasoning": "high", + "schema_sha256": "e0a213b4ae38c1220617d24273e83e5437ebdee96677ab8d0667736faa248c27", + "stage": "decision_audit", + "system_sha256": "a3123bd136a1416160db1c17d29c1a63ff42c7fe49656bab5d0e5b5fd709d2e8", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 42965, + "output_tokens": 6031, + "output_tokens_details": { + "thinking_tokens": 477 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 44.266 + } + ], + "policy_revision": "implementation-five-v1-20260929", + "prompt_revision": "implementation-proximity-multipass-v4-20260929", + "prompt_sha256": "253314b009ea82fc44695260ece8edd501ab60dc40ab3cc9e1ac9a0acb2017c0", + "reasoning": "high", + "result_retention_seconds": 3600, + "routing": { + "allow_external": true, + "allowed_external_providers": [ + "claude" + ] + }, + "selection": { + "backend": "claude-code-2.1.285", + "case_count": 75, + "cli_model": "claude-opus-5-5[1m]", + "configuration_revision": "suite-v6-20260929", + "context": 1000000, + "enabled": true, + "execution_revision": "suite-multipass-v9-20260929", + "input_bytes": 66341, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 74533, + "input_token_count": null, + "later_pass_capacity_verified": false, + "later_pass_checks": "before_each_invocation", + "max_final_group_cases": 5, + "max_final_name_characters": 64, + "max_turns": 6, + "maximum_model_passes": 5, + "minimum_model_passes": 3, + "model": "claude-opus-5-5", + "output": 64000, + "output_reservation_tokens": 18816, + "output_reservation_verified": false, + "overhead": 8192, + "policy_revision": "implementation-five-v1-20260929", + "prompt_revision": "implementation-proximity-multipass-v4-20260929", + "prompt_sha256": "253314b009ea82fc44695260ece8edd501ab60dc40ab3cc9e1ac9a0acb2017c0", + "provider": "claude", + "reasoning": "high", + "reasoning_policy": { + "case_count_at_least": 100, + "default": "high", + "large": "xhigh", + "match": "any", + "revision": "suite-size-effort-v1-20260929", + "scope": "whole_job", + "source_bytes_at_least": 131072, + "source_bytes_scope": "Canonical UTF-8 JSON of campaign, suite, and all cases" + }, + "reasoning_selection": { + "case_count": 75, + "policy_revision": "suite-size-effort-v1-20260929", + "scope": "whole_job", + "source_bytes": 55836, + "triggers": [] + }, + "reported_output": 128000, + "source_sha256": "3cca99c228210deee67c4f7a4674bad55283440c5662f8add1c5182409447d32" + }, + "singleton_statistics": { + "final_singletons": 3, + "natural_singletons": 3 + }, + "status": "completed", + "temporary_files_deleted": true, + "truncation": false, + "turns": 12, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 223758, + "output_tokens": 27713, + "output_tokens_details": { + "thinking_tokens": 2823 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 201.953 + }, + "natural_quality": { + "coverage": true, + "false_merge_pairs": 0, + "families": 9, + "missed_merge_pairs": 0, + "pair_precision": 1.0, + "pair_recall": 1.0 + }, + "preflight": { + "availability": "checked_at_dispatch", + "execution": { + "max_cost_usd": 30.0, + "max_seconds": 3600, + "strategy": "whole_suite" + }, + "routing": { + "allow_external": true, + "allowed_external_providers": [ + "claude" + ] + }, + "selection": { + "backend": "claude-code-2.1.285", + "case_count": 75, + "cli_model": "claude-opus-5-5[1m]", + "configuration_revision": "suite-v6-20260929", + "context": 1000000, + "enabled": true, + "execution_revision": "suite-multipass-v9-20260929", + "input_bytes": 66341, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 74533, + "input_token_count": null, + "later_pass_capacity_verified": false, + "later_pass_checks": "before_each_invocation", + "max_final_group_cases": 5, + "max_final_name_characters": 64, + "max_turns": 6, + "maximum_model_passes": 5, + "minimum_model_passes": 3, + "model": "claude-opus-5-5", + "output": 64000, + "output_reservation_tokens": 18816, + "output_reservation_verified": false, + "overhead": 8192, + "policy_revision": "implementation-five-v1-20260929", + "prompt_revision": "implementation-proximity-multipass-v4-20260929", + "prompt_sha256": "253314b009ea82fc44695260ece8edd501ab60dc40ab3cc9e1ac9a0acb2017c0", + "provider": "claude", + "reasoning": "high", + "reasoning_policy": { + "case_count_at_least": 100, + "default": "high", + "large": "xhigh", + "match": "any", + "revision": "suite-size-effort-v1-20260929", + "scope": "whole_job", + "source_bytes_at_least": 131072, + "source_bytes_scope": "Canonical UTF-8 JSON of campaign, suite, and all cases" + }, + "reasoning_selection": { + "case_count": 75, + "policy_revision": "suite-size-effort-v1-20260929", + "scope": "whole_job", + "source_bytes": 55836, + "triggers": [] + }, + "reported_output": 128000, + "source_sha256": "3cca99c228210deee67c4f7a4674bad55283440c5662f8add1c5182409447d32" + }, + "status": "eligible" + }, + "prompt_revision": "implementation-proximity-multipass-v4-20260929", + "request_bytes": 55986, + "status": "completed", + "test": "isolated_async_synthetic_budget_benchmark", + "wall_seconds": 202.407 + } +} diff --git a/docs/hermes_suite_claude55.md b/docs/hermes_suite_claude55.md index 54ab3398..a131b27e 100644 --- a/docs/hermes_suite_claude55.md +++ b/docs/hermes_suite_claude55.md @@ -5,9 +5,11 @@ both returned their exact canonical model identities on the existing first-party OAuth account. The default deployment selects `claude-opus-5-5`; Sonnet is a server configuration option, not an automatic fallback. The HTTPS contract, credential scopes, implementation objective, five-case task cap, -30-minute deadline, and USD 30 CLI estimated-cost guard are unchanged. +30-minute default deadline, and USD 30 CLI estimated-cost guard are unchanged. +An explicit `execution.max_seconds: 3600` now permits a 60-minute total job; see +[current budget evidence](hermes_suite_job_budget.md). -Execution revision `suite-multipass-v7-20260929` uses `high` by default and +Execution revision `suite-multipass-v9-20260929` preserves the effort policy: `high` by default and `xhigh` for at least 100 cases OR at least 131,072 bytes (128 KiB) of complete suite content. Byte size is the canonical UTF-8 JSON of `campaign`, `suite`, and all `cases`, without transport whitespace, routing/execution fields, prompt text, @@ -99,7 +101,7 @@ checksum verification leaves the worker unavailable instead of using another CLI The HTTP compatibility revision remains `suite-v6-20260929`, the prompt remains `implementation-proximity-multipass-v4-20260929`, and the execution revision is -`suite-multipass-v7-20260929`. Authenticated capabilities additionally expose +`suite-multipass-v9-20260929`. Authenticated capabilities additionally expose `claude_model_options` and `model_selection: server_configuration`. No new client request field is accepted or required. diff --git a/docs/hermes_suite_job_budget.md b/docs/hermes_suite_job_budget.md new file mode 100644 index 00000000..159ec274 --- /dev/null +++ b/docs/hermes_suite_job_budget.md @@ -0,0 +1,129 @@ +# Suite job time and cost budget + +New explicitly requested jobs may use `execution.max_seconds: 3600`. Omitting +the field still selects 1800 seconds; valid values are integers from 10 to 3600. +The USD 30 CLI estimated-cost maximum/default is unchanged. These figures are +CLI accounting guards, not proof of subscription billing charges. + +```json +{ + "execution": { + "strategy": "whole_suite", + "max_seconds": 3600, + "max_cost_usd": 30 + } +} +``` + +This is the execution section of the existing request, not a complete request. +Use a fresh job attempt/idempotency key when intentionally resubmitting a failed +job. No real suite was rerun during this change. + +## Runtime and revisions + +- HTTP compatibility configuration: `suite-v6-20260929`, unchanged. +- Execution: `suite-multipass-v9-20260929`. +- Prompt: `implementation-proximity-multipass-v4-20260929`, unchanged. +- Policy: `implementation-five-v1-20260929`, unchanged. +- Model: `claude-opus-5-5`, CLI setting `claude-opus-5-5[1m]`. +- Native Claude Code: 2.1.285, existing pinned binary and first-party OAuth. + +Opus 5.5 was intentionally selected in the earlier user-authorized model +upgrade, not by this budget change. Completed real-job CLI results verified +Opus 5.5, firstParty, 1M context and 128K reported model output ceiling. The +service still requests at most 64K output per model turn, with at most six CLI +turns per invocation. Aggregated invocation output can therefore exceed 64K. +The source default now agrees with the existing deployment pin. A job never +changes model or falls back; mismatched runtime identity is rejected. + +Effort remains high by default, xhigh for 100+ cases or 128 KiB+ canonical +source content. All passes keep the same selected effort. This is not changed +to make a slow job fit. + +## One shared deadline + +Request validation and the published JSON schema allow 3600 only when explicitly +requested. Capabilities expose `max_seconds: 3600` and +`default_max_seconds: 1800`. Preflight returns the normalized `execution` object, +and job metadata records it. No new client request field is required. + +One absolute monotonic deadline starts before routing, passes through the +orchestrator, and reaches each CLI watchdog. Each call receives only the +remaining time and cost. Status uses that same deadline, including a clamped +zero remaining time at expiry. No pass independently receives another hour. + +Shared deadline expiry reports `job_time_budget_exhausted`, including when the +watchdog stops an active CLI process. A distinct provider timeout keeps its +provider error classification; a standalone subprocess watchdog can still +report `timeout`. Output and coverage validation remain strict. + +The API remains asynchronous at `https://worker.bstein.dev/suite-planning`. +Client submission/polling stays at 45 seconds. The 30-second HTTP body-read and +40-second ingress response-header timeouts, ingress, authentication, routing, +concurrency, and USD 30 guard are unchanged. + +## Failed 363-case job: observed usage + +Safe metadata for `9af0db7c71b04882a3a2c0449e83b74c` shows: + +| Completed pass | Seconds | Input tokens | Output tokens | Thinking tokens | Estimated USD | +| --- | ---: | ---: | ---: | ---: | ---: | +| proposal_a | 505.343 | 369479 | 57879 | 41501 | 2.635496 | +| proposal_b | 385.938 | 364164 | 47941 | 34819 | 2.415476 | +| reconciliation | 847.944 | 398332 | 108603 | 76973 | 3.765388 | +| Recorded total | 1739.225 | 1131975 | 214423 | 153293 | 8.816360 | + +`execution_progress.cost_used_usd_estimate` was **8.81636**, against 30.0. +Cache counters and provider web-tool counts were zero. Each completed invocation +used three CLI turns on Opus 5.5 at xhigh effort. These token counts aggregate +multiple turns/passes; they are not one context-window measurement. + +`large_family_review` received 60.72140585165471 seconds. The job failed at +1800.064 wall seconds when the old watchdog terminated it, reporting exit 143, +termination reason `timeout`, and no terminal CLI event. Its token usage and +cost are unknown, so 8.81636 is the recorded completed-pass cost, not an exact +total including interrupted work. The decision audit never started. + +If each remaining review takes as long and costs as much as reconciliation, +a fresh full job would take approximately 3435 seconds (57.3 minutes) and +USD 16.35 in CLI accounting. This is a scenario estimate, not a verified upper +bound. It supports the requested 3600-second allowance and retaining USD 30; +there is no evidence requiring a higher cost guard. Content difficulty and +structured-output repair can still exhaust either guard. + +## Verification and rollback + +Regression checks include explicit opt-in/default/schema bounds, an accelerated +five-pass job lasting 2510 simulated seconds under one 3600-second deadline, +declining allocations after routing, shared-watchdog classification and zero +remaining time, distinct provider timeouts, and existing coverage/isolation tests. +No actual hour-long wait is represented as tested by the accelerated clock. + +The pre-deployment benchmark uses an isolated loopback async API and a temporary +database with only the public synthetic 75-case fixture. It exercises the actual +pinned hosted CLI/account, preflight, idempotent submission and 45-second HTTP +polling. Its metadata is recorded separately in the accompanying evidence. It +does not establish laptop connectivity or guarantee the real 363-case runtime. + +The synthetic job completed all five passes in **201.953 server wall seconds** +(202.407 seconds including client preflight/polling), with 223758 input tokens, +27713 output tokens including 2823 thinking tokens, and USD **1.449292** estimated +cost. The request was 55986 serialized bytes. Runtime identity was verified as +Opus 5.5, firstParty, at high effort. There were no automatic job retries. + +All 75 aliases appeared exactly once. The natural partition matched all nine +synthetic mechanisms with zero false/missed merge pairs; sizing produced 21 +final tasks with three singletons. This membership score is distinct from a +comprehensive semantic review of generated descriptions. + +The API accepted an explicit 3600-second preflight/submission and rejected 3601. +Per-pass remaining-time allocations were 3599.966, 3584.405, 3555.589, 3487.413, +and 3442.319 seconds; the final watchdog received the last allocation. +Idempotent replay returned the same job. HTTP polling retained 45 seconds. +All 163 regression tests passed; Kustomize render and client dry-run passed. +Flux diff affects only the planner script ConfigMap and rollout annotation. +Full safe measurements: [budget evidence](evidence/hermes_suite_budget_20260929.json). + +Rollback by reverting the v9 implementation commit through Git and reconciling +the Hermes Flux Kustomization while no job is active. This restores the 1800-second +maximum and old timeout classification. Do not manually edit the deployment. diff --git a/docs/hermes_suite_multipass.md b/docs/hermes_suite_multipass.md index db57e3f4..4a617193 100644 --- a/docs/hermes_suite_multipass.md +++ b/docs/hermes_suite_multipass.md @@ -11,7 +11,7 @@ most **64 characters**, unique after whitespace and case normalization. - Configuration: `suite-v6-20260929` (HTTP compatibility identifier). - Policy: `implementation-five-v1-20260929`. - Prompt: `implementation-proximity-multipass-v4-20260929`. -- Execution: `suite-multipass-v7-20260929`. +- Execution: `suite-multipass-v9-20260929`. A single server-side job performs: @@ -151,9 +151,10 @@ invented percentage, because different passes have very different runtimes. ## Shared bounds and failure behavior -The user-approved 1800-second maximum job deadline and USD 30 CLI estimated-cost -guard apply to **all calls combined**, including routing time. These are also the -defaults when omitted. Explicit lower client limits remain effective. This guard +An explicitly requested job may run up to 3600 seconds. The default remains +1800 seconds. The USD 30 CLI estimated-cost guard is unchanged. One absolute +deadline applies to **all calls combined**, including routing time. Explicit +lower client limits remain effective. This guard uses Claude CLI estimated-cost accounting; it is not subscription billing. Each invocation receives only the remaining time and estimated-cost allowance. Available usage/cost is accumulated after every call; unknown cost accounting stops further hosted calls. @@ -333,3 +334,11 @@ and maximums to 1800 seconds and USD 30 in CLI estimate accounting. Explicit low client limits remain effective. The existing request fields are unchanged. No inference or regression tests were rerun for this limit-only update, as requested; the acceptance measurements above retain their original revisions and bounds. + +Execution revision `suite-multipass-v9-20260929` supersedes the maximum above: +3600 seconds is opt-in through `execution.max_seconds`, with the default still +1800 and the estimated-cost maximum/default still USD 30. The shared deadline +reaches every subprocess watchdog and status update. Its expiry reports +`job_time_budget_exhausted`; distinct provider/standalone subprocess timeouts keep +their own classification. The asynchronous HTTP timeout remains 45 seconds. +See [current measurements and handoff](hermes_suite_job_budget.md). diff --git a/docs/hermes_suite_planning.md b/docs/hermes_suite_planning.md index 3053b091..e2afff3d 100644 --- a/docs/hermes_suite_planning.md +++ b/docs/hermes_suite_planning.md @@ -13,11 +13,18 @@ The endpoint and request fields are unchanged. Optional top-level `review_summar is returned only with the authorized result. Configuration remains `suite-v6-20260929`; policy, prompt, and execution revisions identify the changed grouping behavior. The prior single-pass acceptance results below are historical and do not measure -the new workflow. The new process shares the user-approved 1800-second deadline -and USD 30 CLI estimate guard across all invocations. This is not subscription +the new workflow. New jobs may explicitly request up to 3600 seconds; the default +remains 1800 seconds. One deadline and the USD 30 CLI estimate guard cover all invocations. This is not subscription billing. Progress updates every five seconds during a CLI pass. The separate local-model endpoint is unchanged; the 8K local route cannot admit the new multi-pass reconciliation schema. +The active model is **`claude-opus-5-5`**, invoked as `claude-opus-5-5[1m]` +using native Claude Code 2.1.285. This was the user-requested 5.5 upgrade; +older Opus 4.8 acceptance records below are historical. Effort is high, or +xhigh at 100+ cases or 128 KiB+ source content. The model and effort stay fixed +throughout a job. Current execution revision: `suite-multipass-v9-20260929`. +See [current deadline and cost evidence](hermes_suite_job_budget.md). + ## Earlier CLI failure diagnostics update (historical) Execution revision `claude-diagnostics-turns-v1-20260929` adds content-free @@ -87,7 +94,7 @@ The focused unit suite passes 36 tests. These checks establish the synthetic beh not the quality of a real-suite rerun. See the [prompt-update acceptance evidence](evidence/hermes_suite_prompt_v2_20260929.json). -## Acceptance results, 2026-09-29 +## Historical Opus 4.8 acceptance results, 2026-09-29 The final runs use `suite-v6-20260929`, deployed commit `299b9ed1`, including case-type fields. These supersede the preliminary runs without that field. @@ -200,7 +207,7 @@ LOCAL_ONLY_FIELD: token SYNTHETIC_EXTERNAL_FIELD: synthetic_token APPROVED_OPERATIONAL_FIELD: operational_token INITIAL_CONCURRENCY: 1; busy submissions return 429; no waiting queue -JOB_TIMEOUT: up to 1800 seconds +JOB_TIMEOUT: default 1800 seconds; explicitly request up to 3600 seconds HTTP_TIMEOUT: client 45 seconds; submit/status do not wait for inference TLS: existing worker.bstein.dev certificate; normal trusted CA verification ``` @@ -273,7 +280,7 @@ Missing routing policy means local-only. External provider names are `claude` an unknown providers, malformed booleans, duplicate names, and permission escalation are rejected before routing. `allow_external=false` cannot include providers. `whole_suite` is the only strategy. No batching, summarization, or reconciliation -is silently substituted. `max_seconds` is 10-1800. The Claude CLI guard is at most +is silently substituted. `max_seconds` is 10-3600 (default 1800). The Claude CLI guard is at most USD 30 in its estimated usage accounting; this is not a verified subscription billing ceiling and can overshoot within a single provider call. @@ -387,26 +394,23 @@ result, and independently audit aliases. No importer replacement is needed. | Property | Codex | Claude | Existing local API | | --- | --- | --- | --- | -| Installed client | Codex CLI 0.154.0 | Claude Code 2.1.226, pinned binary SHA-256 | Ollama 0.13.5 | +| Installed client | Codex CLI 0.154.0 | Claude Code 2.1.285, pinned binary SHA-256 | Ollama 0.13.5 | | Existing broker mode | Direct subscription Responses transport, not `codex exec` | Native CLI print mode behind a wrapper | Native `/api/generate` | | New planner backend | Disabled pending effective output-budget verification | Fresh native CLI process, wrapper bypass avoided | Unchanged model-gate LAN listener | | Auth | ChatGPT Pro claim verified locally | Existing first-party OAuth setup token; associated credential metadata says Max 20x | Scoped local bearer | -| Visible catalog | `gpt-6-astra`, `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5` | Fable 5, Opus 5 with 1M option, Sonnet 5, Haiku 4.5 | Pinned Qwen 2.5 14B Q4 | -| Exact selected ID | `gpt-6-astra` reserved, not enabled | `claude-opus-4-8` | `qwen2.5:14b-instruct-q4_0` | +| Visible catalog | `gpt-6-astra`, `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5` | Verified Opus 5.5 and Sonnet 5.5; server pin selects Opus 5.5 | Pinned Qwen 2.5 14B Q4 | +| Exact selected ID | `gpt-6-astra` reserved, not enabled | `claude-opus-5-5` | `qwen2.5:14b-instruct-q4_0` | | Context evidence | Local account model cache 272,000, 95% effective = 258,400 | Actual CLI response: 1,000,000 | Verified serving configuration: 8,192 | -| Output control | Existing broker removes public token-limit fields; effective maximum unverified | Actual CLI reports 64,000; environment pins that ceiling | 2,048 | +| Output control | Existing broker removes public token-limit fields; effective maximum unverified | Actual CLI reports 128,000; service requests 64,000 per model turn | 2,048 | | New planner concurrency | None | One across the whole planning service | Shares the existing serialized GPU backend | | Hardware | Provider hosted; broker on titan-22 | Provider hosted; CLI on titan-22 | RTX 3080 10GB on titan-24 | -The Claude live catalog resolves `claude-fable-5[1m]` to `claude-fable-5`, -`default`/`opus[1m]` to `claude-opus-5[1m]`, `sonnet` to `claude-sonnet-5`, -and `haiku` to `claude-haiku-4-5-20251001`. Catalog availability is not proof -that every model has been exercised. Fable 5 suite requests reported `claude-opus-4-8` at runtime and were rejected. -The initial working route therefore explicitly pins the observed Opus 4.8 model; -the advertised Fable name is not verified for full-suite execution. -Only the tiny Opus/Fable probes and recorded -planner acceptance jobs were executed. No Claude agent wrote infrastructure code. -Subscription quotas and account retention settings are not verified by model discovery. +The current exact Opus 5.5 and Sonnet 5.5 runtime identities were verified through +the first-party account. The deployment intentionally pins Opus 5.5; aliases and +catalog advertisements are not substituted for the runtime identity check. +The earlier Fable/Opus 4.8 discovery is historical. Subscription quotas and +provider/account retention settings remain unverified. No Claude agent writes +infrastructure code; hosted tests contain only synthetic cases. The legacy Codex and Claude wrapper scripts enable permission bypass; the Claude wrapper also enables automatic compaction and shared settings. Neither is used by @@ -415,15 +419,17 @@ but that separate CLI execution mode has not been approved as a whole-suite back The direct Codex broker uses `store=false`, streaming Responses, and a 900-second read timeout. That flag does not establish provider Zero Data Retention. -Claude invocation, with the schema supplied by the server: +Abbreviated server-side Claude invocation (the actual command builder also supplies +strict isolation settings disabling bundled plugins, hooks, connectors and model +switching; see `suite_backends.claude_command`): ``` /opt/cli/claude -p --output-format stream-json --verbose \ --no-session-persistence --safe-mode --tools '' \ --strict-mcp-config --mcp-config '{"mcpServers":{}}' \ --setting-sources '' --disable-slash-commands --permission-mode dontAsk \ - --no-chrome --model 'claude-opus-4-8[1m]' --effort medium \ - --max-budget-usd 5 --max-turns 6 \ + --no-chrome --model 'claude-opus-5-5[1m]' --effort high \ + --max-budget-usd 30 --max-turns 6 \ --system-prompt '' --json-schema '' ``` @@ -464,7 +470,8 @@ Small local requests reserve 1,024 overhead tokens and 2,048 output tokens withi 8,192 context. The realistic 14/75/363 fixtures exceed this conservative local whole-suite budget and must use the approved hosted path or fail preflight. -All CLI invocations share at most 1800 seconds per suite job; cancellation and timeouts terminate its +All CLI invocations share at most the requested 3600 seconds per suite job +(default 1800 seconds); cancellation and timeouts terminate its process group. Async HTTP calls finish promptly, with a 30-second body-read timeout, 40-second ingress response-header timeout, and recommended 45-second client timeout. Switchyard's ten-minute internal request limit carries only a tiny immediate routing @@ -515,7 +522,7 @@ import json with open('synthetic-suite.json') as stream: request = json.load(stream) request['routing'] = {'allow_external': True, 'allowed_external_providers': ['claude']} -request['execution'] = {'strategy': 'whole_suite', 'max_seconds': 1800, 'max_cost_usd': 30} +request['execution'] = {'strategy': 'whole_suite', 'max_seconds': 3600, 'max_cost_usd': 30} with open('synthetic-request.json', 'w') as stream: json.dump(request, stream) PY diff --git a/scripts/ops/hermes_suite_budget_probe.py b/scripts/ops/hermes_suite_budget_probe.py new file mode 100755 index 00000000..9b81cca0 --- /dev/null +++ b/scripts/ops/hermes_suite_budget_probe.py @@ -0,0 +1,110 @@ +#!/usr/bin/env python3 +"""Benchmark one synthetic async job in an isolated loopback API before rollout. + +Uses the installed pinned CLI/account and scoped synthetic credential. Accepts +only built-in synthetic fixtures, never an input file or real job identifier. +HTTP requests retain the ordinary 45-second timeout. No service configuration +or live job database is modified by this probe. +""" +import argparse +from http.server import ThreadingHTTPServer +import json +import os +from pathlib import Path +import sys +import tempfile +import threading +import time +from urllib.error import HTTPError +from urllib.request import ProxyHandler, Request, build_opener + +sys.path.insert(0, os.environ.get("SUITE_PROBE_MODULE_DIR", "/opt/planner")) +from suite_api import Handler +from suite_contract import EXECUTION_REVISION, PROMPT_REVISION, REVISION, encoded, validate_result +from suite_jobs import Jobs +from suite_synthetic import fixture, score + + +def main(): + """Submit once, poll safely, and report measured metadata and independent scores.""" + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--size", type=int, choices=(14, 75, 363), default=75) + parser.add_argument("--max-seconds", type=int, default=3600) + args = parser.parse_args() + request, expected = fixture(args.size) + request["routing"] = {"allow_external": True, "allowed_external_providers": ["claude"]} + request["execution"] = {"strategy": "whole_suite", "max_seconds": args.max_seconds, "max_cost_usd": 30} + token = Path("/vault/secrets/synthetic-token").read_text().strip() + opener = build_opener(ProxyHandler({})) + with tempfile.TemporaryDirectory(prefix="suite-budget-probe-", dir="/jobs") as directory: + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + server.jobs = Jobs(Path(directory) / "jobs.sqlite") + threading.Thread(target=server.serve_forever, daemon=True).start() + + def http(path, value=None, key=None): + headers = {"Authorization": "Bearer " + token, "X-Forwarded-For": "192.168.22.8", + "Content-Type": "application/json"} + if key: + headers["Idempotency-Key"] = key + call = Request("http://127.0.0.1:" + str(server.server_port) + path, + data=encoded(value) if value is not None else None, headers=headers) + try: + with opener.open(call, timeout=45) as response: + return response.status, json.load(response) + except HTTPError as exc: + with exc: + return exc.code, json.load(exc) + + started, last_report = time.monotonic(), 0.0 + report = {"test": "isolated_async_synthetic_budget_benchmark", "case_count": args.size, + "request_bytes": len(encoded(request)), "configuration_revision": REVISION, + "execution_revision": EXECUTION_REVISION, "prompt_revision": PROMPT_REVISION, + "http_timeout_seconds": 45, "automatic_job_retries": 0} + try: + status, capabilities = http("/v1/capabilities") + assert status == 200 and capabilities["max_seconds"] == 3600 + assert capabilities["default_max_seconds"] == 1800 + invalid = {**request, "execution": {**request["execution"], "max_seconds": 3601}} + status, rejected = http("/v1/preflight", invalid) + assert status == 400 and rejected["error"]["code"] == "invalid_timeout" + status, checked = http("/v1/preflight", request) + assert status == 200 and checked["execution"]["max_seconds"] == args.max_seconds + report.update(capabilities_limits={k: capabilities[k] for k in ( + "max_seconds", "default_max_seconds", "max_cost_usd")}, preflight=checked) + status, job = http("/v1/jobs", request, "synthetic-long-budget-probe") + assert status == 202 and job["execution"]["max_seconds"] == args.max_seconds + path = "/v1/jobs/" + job["job_id"] + replay_status, replay = http("/v1/jobs", request, "synthetic-long-budget-probe") + assert replay_status == 200 and replay["job_id"] == job["job_id"] + report["idempotent_replay_verified"] = True + while job["status"] in {"accepted", "running", "cancelling"}: + if time.monotonic() - last_report >= 20: + progress = job.get("execution_progress", {}) + print(json.dumps({k: progress.get(k) for k in ( + "current_pass", "completed_model_passes", "job_elapsed_seconds", + "job_remaining_seconds", "cost_used_usd_estimate")}), file=sys.stderr, flush=True) + last_report = time.monotonic() + time.sleep(2) + _, job = http(path) + report["status"] = job["status"] + report["job_metadata"] = job + if job["status"] == "completed": + status, result = http(path + "/result") + assert status == 200 + validate_result(result["result"], request) + natural = {"groups": result["review_summary"]["natural_families"]} + report.update(natural_quality=score(natural, expected), + counts=result["review_summary"]["counts"], exact_final_coverage=True) + allocations = [p["allocated_seconds"] for p in job["passes"]] + assert allocations[0] > 1800 + assert all(a > b for a, b in zip(allocations, allocations[1:])) + assert job["cli_diagnostics"]["subprocess_timeout_seconds"] == allocations[-1] + report["wall_seconds"] = round(time.monotonic() - started, 3) + print(json.dumps(report, sort_keys=True), flush=True) + return 0 if report["status"] == "completed" else 1 + finally: + server.shutdown() + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/ops/hermes_suite_multipass_transport_probe.py b/scripts/ops/hermes_suite_multipass_transport_probe.py index 7b43b2e0..0cfe6458 100755 --- a/scripts/ops/hermes_suite_multipass_transport_probe.py +++ b/scripts/ops/hermes_suite_multipass_transport_probe.py @@ -127,10 +127,11 @@ def main(): env['ANTHROPIC_BASE_URL'] = 'http://127.0.0.1:' + str(server.server_port) return env - def backend(request, cancel, *, invocation, progress=None): + def backend(request, cancel, *, invocation, progress=None, job_deadline=None): active.clear() active.update(invocation) - return original_generate(request, cancel, invocation=invocation, progress=progress) + return original_generate(request, cancel, invocation=invocation, progress=progress, + job_deadline=job_deadline) suite_backends.claude_environment = environment suite_backends.claude_generate = backend diff --git a/services/hermes/scripts/suite_api.py b/services/hermes/scripts/suite_api.py index 11fe4f0c..d391fd61 100644 --- a/services/hermes/scripts/suite_api.py +++ b/services/hermes/scripts/suite_api.py @@ -12,7 +12,7 @@ import re import subprocess import threading -from suite_contract import (CLAUDE_MODELS, CLAUDE_VERSION, COST_LIMIT, EXECUTION_REVISION, MAX_BODY, MAX_CASES, MAX_RESULT, MODELS, PROMPT_REVISION, +from suite_contract import (CLAUDE_MODELS, CLAUDE_VERSION, COST_LIMIT, EXECUTION_REVISION, MAX_BODY, MAX_CASES, MAX_JOB_SECONDS, MAX_RESULT, MODELS, PROMPT_REVISION, PROMPT_SHA256, REVISION, TIMEOUT, Problem, encoded, preflight, validate_request) from suite_jobs import Jobs @@ -137,7 +137,8 @@ class Handler(BaseHTTPRequestHandler): if generalized_claude_approved() and owner == "operational-token" else [], "strategy": ["whole_suite"], "max_request_bytes": MAX_BODY, "max_result_bytes": MAX_RESULT, "max_cases": MAX_CASES, - "max_seconds": TIMEOUT, "concurrency": 1, "queue": False, + "max_seconds": MAX_JOB_SECONDS, "default_max_seconds": TIMEOUT, + "concurrency": 1, "queue": False, "max_cost_usd": COST_LIMIT, "cost_basis": "CLI estimate guard; not subscription billing", "progress_interval_seconds": 5, "result_retention_seconds": 3600, "idempotency_retention_seconds": 604800, @@ -150,7 +151,8 @@ class Handler(BaseHTTPRequestHandler): selected = preflight(request) if self.path == "/v1/preflight": return self.send(200, {"status": "eligible", "routing": request["routing"], - "selection": selected, "availability": "checked_at_dispatch"}) + "selection": selected, "execution": request["execution"], + "availability": "checked_at_dispatch"}) key = self.headers.get("Idempotency-Key", "") if not re.fullmatch(r"[A-Za-z0-9_-]{8,128}", key): raise Problem("idempotency_key_required") diff --git a/services/hermes/scripts/suite_backends.py b/services/hermes/scripts/suite_backends.py index def8b5af..d971779d 100644 --- a/services/hermes/scripts/suite_backends.py +++ b/services/hermes/scripts/suite_backends.py @@ -230,7 +230,7 @@ def parse_claude(raw, expected_model, **process_info): "turns": diagnostics["turns"], "cli_diagnostics": diagnostics} -def claude_generate(request, cancel, *, invocation=None, progress=None): +def claude_generate(request, cancel, *, invocation=None, progress=None, job_deadline=None): """Run one fresh job in tmpfs; input, output, configuration and caches expire together.""" token = Path("/vault/secrets/claude-token").read_text().strip() if not token: @@ -263,7 +263,10 @@ def claude_generate(request, cancel, *, invocation=None, progress=None): while process.poll() is None: if cancel.wait(0.1): raise Problem("cancelled", 409) - if time.monotonic() > deadline: + now = time.monotonic() + if job_deadline is not None and now >= job_deadline: + raise Problem("job_time_budget_exhausted", 504) + if now >= deadline: raise Problem("timeout", 504) activity = (root / "output").stat() if progress and time.monotonic() >= next_progress: diff --git a/services/hermes/scripts/suite_contract.py b/services/hermes/scripts/suite_contract.py index a6f468c0..fd83ef2d 100644 --- a/services/hermes/scripts/suite_contract.py +++ b/services/hermes/scripts/suite_contract.py @@ -9,14 +9,14 @@ from collections import Counter REVISION = "suite-v6-20260929" PROMPT_REVISION = "implementation-proximity-multipass-v4-20260929" -EXECUTION_REVISION = "suite-multipass-v8-20260929" +EXECUTION_REVISION = "suite-multipass-v9-20260929" CLAUDE_VERSION = "2.1.285" CLAUDE_MODELS = { "claude-opus-4-8": 64000, "claude-opus-5-5": 128000, "claude-sonnet-5-5": 128000, } -CLAUDE_MODEL = os.environ.get("PLANNING_CLAUDE_MODEL", "claude-opus-4-8") +CLAUDE_MODEL = os.environ.get("PLANNING_CLAUDE_MODEL", "claude-opus-5-5") if CLAUDE_MODEL not in CLAUDE_MODELS: raise RuntimeError("Unsupported configured Claude model") CLAUDE_MAX_TURNS = 6 @@ -31,6 +31,7 @@ MAX_BODY = 1 << 20 MAX_RESULT = 1 << 20 MAX_CASES = 400 TIMEOUT = 1800 +MAX_JOB_SECONDS = 3600 COST_LIMIT = 30.0 RESULT_TTL = 3600 FIELDS = {"description", "success_criteria", "preconditions", "operating_condition", @@ -163,7 +164,7 @@ def validate_request(raw, permissions): raise Problem("unsupported_strategy", 422) seconds = execution.get("max_seconds", TIMEOUT) cost = execution.get("max_cost_usd", COST_LIMIT) - if type(seconds) is not int or not 10 <= seconds <= TIMEOUT: + if type(seconds) is not int or not 10 <= seconds <= MAX_JOB_SECONDS: raise Problem("invalid_timeout") if type(cost) not in (int, float) or not 0 < cost <= COST_LIMIT: raise Problem("invalid_cost_limit") diff --git a/services/hermes/scripts/suite_jobs.py b/services/hermes/scripts/suite_jobs.py index 2a23feb0..8c878b44 100644 --- a/services/hermes/scripts/suite_jobs.py +++ b/services/hermes/scripts/suite_jobs.py @@ -85,7 +85,7 @@ class Jobs: "execution_revision": EXECUTION_REVISION, "policy_revision": POLICY_REVISION, "prompt_revision": PROMPT_REVISION, "prompt_sha256": PROMPT_SHA256, - "selection": selected, "attempted_destinations": [], + "selection": selected, "execution": request["execution"], "attempted_destinations": [], "compaction": None, "truncation": None, "usage": None, "result_retention_seconds": RESULT_TTL, "created_at": time.time()} @@ -112,6 +112,7 @@ class Jobs: def run(self, job_id, owner, request, selected, client_ip): """Authorize a fixed route, then run one bounded multi-pass suite job.""" started = time.monotonic() + deadline = started + request["execution"]["max_seconds"] provider = selected["provider"] document = self.get(job_id, owner) event = self.cancels[job_id] @@ -125,9 +126,9 @@ class Jobs: suite_backends.switchyard_decision(provider) if event.is_set(): raise Problem("cancelled", 409) - remaining = request["execution"]["max_seconds"] - (time.monotonic() - started) + remaining = deadline - time.monotonic() if remaining <= 0: - raise Problem("timeout", 504) + raise Problem("job_time_budget_exhausted", 504, failure_stage="routing") effective = {**request, "execution": {**request["execution"], "max_seconds": remaining}} # This selection is fixed before dispatch; no exception invokes a fallback. if provider not in {"local", "claude"}: @@ -141,7 +142,8 @@ class Jobs: document["execution_progress"] = value with self.lock: self._save(job_id, document) - result, metadata = suite_multipass.generate(effective, selected, event, client_ip, progress) + result, metadata = suite_multipass.generate(effective, selected, event, client_ip, progress, + started=started, deadline=deadline) review = metadata.pop("review_summary") document.update(metadata) try: @@ -151,6 +153,8 @@ class Jobs: failure_stage="result_validation") from None if event.is_set(): raise Problem("cancelled", 409) + if time.monotonic() >= deadline: + raise Problem("job_time_budget_exhausted", 504, failure_stage="result_validation") document.update(metadata, status="completed") if len(encoded({**document, "result": result, "review_summary": review})) > MAX_RESULT: raise Problem("response_too_large", 502) @@ -176,7 +180,8 @@ class Jobs: document["wall_seconds"] = round(time.monotonic() - started, 3) if "execution_progress" in document: document["execution_progress"].update(cli_running=False, heartbeat_at=time.time(), - job_elapsed_seconds=document["wall_seconds"]) + job_elapsed_seconds=document["wall_seconds"], + job_remaining_seconds=round(max(0, deadline - time.monotonic()), 1)) with self.lock: if event.is_set() and document["status"] == "completed": document.update(status="cancelled", error={"code": "cancelled"}) diff --git a/services/hermes/scripts/suite_multipass.py b/services/hermes/scripts/suite_multipass.py index 1de7964c..b18ab3ac 100644 --- a/services/hermes/scripts/suite_multipass.py +++ b/services/hermes/scripts/suite_multipass.py @@ -100,12 +100,12 @@ def sum_usage(records): class Workflow: """Keep content in memory and allow only one pinned provider across model passes.""" - def __init__(self, request, selected, cancel, client_ip, progress=None): + def __init__(self, request, selected, cancel, client_ip, progress=None, *, started=None, deadline=None): self.request, self.provider, self.cancel = request, selected["provider"], cancel self.reasoning = selected.get("reasoning", MODELS[self.provider]["reasoning"]) self.client_ip, self.progress = client_ip, progress or (lambda _: None) - self.started = time.monotonic() - self.deadline = self.started + request["execution"]["max_seconds"] + self.started = time.monotonic() if started is None else started + self.deadline = self.started + request["execution"]["max_seconds"] if deadline is None else deadline self.cost_limit, self.spent = request["execution"]["max_cost_usd"], 0.0 self.records = [] self.last_metadata = {} @@ -148,7 +148,8 @@ class Workflow: "cost_limit_usd_estimate": self.cost_limit, **(activity or {})}) report() if self.provider == "claude": - value, metadata = suite_backends.claude_generate(effective, self.cancel, invocation=call, progress=report) + value, metadata = suite_backends.claude_generate(effective, self.cancel, invocation=call, + progress=report, job_deadline=self.deadline) elif self.provider == "local": value, metadata = suite_backends.local_generate(effective, self.cancel, self.client_ip, invocation=call) else: @@ -214,9 +215,9 @@ class Workflow: return final, metadata -def generate(request, selected, cancel, client_ip, progress=None): +def generate(request, selected, cancel, client_ip, progress=None, *, started=None, deadline=None): """Never expose partial partitions as completion or broaden a failed route.""" - workflow = Workflow(request, selected, cancel, client_ip, progress) + workflow = Workflow(request, selected, cancel, client_ip, progress, started=started, deadline=deadline) try: return workflow.execute() except Problem as exc: diff --git a/services/hermes/suite-planner-deployment.yaml b/services/hermes/suite-planner-deployment.yaml index ed34dddd..4bb8b54b 100644 --- a/services/hermes/suite-planner-deployment.yaml +++ b/services/hermes/suite-planner-deployment.yaml @@ -36,7 +36,7 @@ spec: app: hermes-suite-planner annotations: fluentbit.io/exclude: "true" - ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v8-20260929 + ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v9-20260929 vault.hashicorp.com/agent-inject: "true" vault.hashicorp.com/agent-pre-populate-only: "true" vault.hashicorp.com/agent-init-first: "true" diff --git a/testing/tests/test_suite_cli_diagnostics.py b/testing/tests/test_suite_cli_diagnostics.py index 09eda314..5f29e174 100644 --- a/testing/tests/test_suite_cli_diagnostics.py +++ b/testing/tests/test_suite_cli_diagnostics.py @@ -4,6 +4,7 @@ from pathlib import Path import signal import sys import threading +import time import pytest @@ -11,7 +12,7 @@ sys.path.insert(0, str(Path(__file__).resolve().parents[2] / "services/hermes/sc import suite_backends import suite_multipass from suite_cli_diagnostics import snapshot -from suite_contract import CLAUDE_MAX_TURNS, MODELS, Problem, preflight, validate_request +from suite_contract import CLAUDE_MAX_TURNS, CLAUDE_MODELS, MODELS, Problem, preflight, validate_request from suite_jobs import Jobs MODEL = MODELS["claude"]["model"] @@ -25,7 +26,7 @@ def envelope(**changes): final = {"type": "result", "subtype": "success", "is_error": False, "num_turns": 2, "stop_reason": "tool_use", "usage": {"input_tokens": 120, "output_tokens": 50, CANARY: CANARY}, - "modelUsage": {MODEL: {"contextWindow": 1000000, "maxOutputTokens": 64000, + "modelUsage": {MODEL: {"contextWindow": 1000000, "maxOutputTokens": CLAUDE_MODELS[MODEL], "canonicalModel": MODEL, "provider": "firstParty", CANARY: CANARY}}, "structured_output": {"groups": [{"name": "Synthetic", "description": CANARY, "members": ["CASE-1"]}]}, @@ -129,7 +130,7 @@ def test_explicit_provider_stop_distinct_from_turn_limit(reason): @pytest.mark.parametrize("raw,stage,code", [ ("", "missing_init_or_final_event", "incomplete_generation"), - ('{"type":"system","subtype":"init","model":"claude-opus-4-8"}', "missing_init_or_final_event", "incomplete_generation"), + (json.dumps({"type": "system", "subtype": "init", "model": MODEL}), "missing_init_or_final_event", "incomplete_generation"), ("not json " + CANARY, "cli_event_json", "invalid_json_result"), ("[]", "cli_event_json", "invalid_json_result"), ]) @@ -190,7 +191,7 @@ def test_validation_failure_retains_usage_without_accepting_content(tmp_path, mo result["groups"][0]["members"].append("CASE-1") monkeypatch.setattr(suite_backends, "switchyard_decision", lambda *_: None) metadata["review_summary"] = {} - monkeypatch.setattr(suite_multipass, "generate", lambda *_: (result, metadata)) + monkeypatch.setattr(suite_multipass, "generate", lambda *_, **__: (result, metadata)) jobs = Jobs(tmp_path / "jobs.sqlite") job, _ = jobs.submit("owner", "synthetic-validation", value, selected, "192.168.22.8", launch=False) jobs.run(job["job_id"], "owner", value, selected, "192.168.22.8") @@ -207,6 +208,7 @@ def test_validation_failure_retains_usage_without_accepting_content(tmp_path, mo ("signal", "incomplete_generation", "process_exit"), ("empty", "backend_unavailable", "process_exit"), ("timeout", "timeout", "subprocess"), ("cancel", "cancelled", "subprocess"), + ("job_timeout", "job_time_budget_exhausted", "subprocess"), ("oversize", "response_too_large", "subprocess"), ("start", "backend_unavailable", "process_start"), ]) @@ -217,7 +219,7 @@ def test_actual_subprocess_paths_cleanup_and_safe_metadata(tmp_path, monkeypatch lambda **_: real_tempdir(prefix="suite-", dir=tmp_path)) monkeypatch.setattr(Path, "read_text", lambda p, *a, **k: "fake-token" if str(p) == "/vault/secrets/claude-token" else real_read(p, *a, **k)) script = "import os,signal,time; print(" + repr(envelope()) + ",flush=True); " - if mode in {"timeout", "cancel"}: + if mode in {"timeout", "job_timeout", "cancel"}: script += "time.sleep(30)" elif mode == "signal": script += "os.kill(os.getpid(),signal.SIGKILL)" @@ -238,13 +240,14 @@ def test_actual_subprocess_paths_cleanup_and_safe_metadata(tmp_path, monkeypatch cancel.set() if code: with pytest.raises(Problem, match=code) as raised: - suite_backends.claude_generate(value, cancel) + suite_backends.claude_generate(value, cancel, + job_deadline=time.monotonic() + 0.02 if mode == "job_timeout" else None) details = raised.value.details assert details["failure_stage"] == stage assert details["reasoning_effort"] == MODELS["claude"]["reasoning"] - if mode in {"timeout", "cancel"}: + if mode in {"timeout", "job_timeout", "cancel"}: assert details["termination_signal"] == signal.SIGTERM - assert details["termination_reason"] == mode.replace("cancel", "cancelled") + assert details["termination_reason"] == code assert CANARY not in json.dumps(details) else: updates = [] diff --git a/testing/tests/test_suite_deadlines.py b/testing/tests/test_suite_deadlines.py new file mode 100644 index 00000000..896b3039 --- /dev/null +++ b/testing/tests/test_suite_deadlines.py @@ -0,0 +1,119 @@ +"""Explicit long budgets share one deadline across routing, passes and progress.""" +import json +import threading +from pathlib import Path + +import pytest + +import suite_backends +import suite_jobs +import suite_multipass +from suite_contract import MAX_JOB_SECONDS, Problem, TIMEOUT, preflight, validate_request +from test_suite_multipass import install_backend, natural, request + + +@pytest.mark.parametrize("seconds", [1801, 2700, 3600]) +def test_explicit_long_budget_matches_published_schema(seconds): + """Admission accepts opt-in durations; omitted limits retain their old default.""" + source = request(7) + assert source["execution"]["max_seconds"] == TIMEOUT == 1800 + source["execution"]["max_seconds"] = seconds + source = validate_request(source, ["claude"]) + assert preflight(source)["provider"] == "claude" + schema = json.loads((Path(__file__).resolve().parents[2] / + "docs/contracts/suite_planning_request.schema.json").read_text()) + limit = schema["properties"]["execution"]["properties"]["max_seconds"] + assert limit["maximum"] == MAX_JOB_SECONDS == 3600 and limit["default"] == TIMEOUT + + +@pytest.mark.parametrize("seconds", [3601, 3600.1, True, "3600", 9]) +def test_invalid_budget_fails_before_inference(seconds): + source = request(7) + source["execution"]["max_seconds"] = seconds + with pytest.raises(Problem, match="invalid_timeout"): + validate_request(source, ["claude"]) + + +def test_job_passes_1800_seconds_with_one_absolute_deadline(tmp_path, monkeypatch): + """Advance synthetic time through a complete five-pass job beyond the old cap.""" + source = request(7) + source["execution"]["max_seconds"] = 3600 + family = natural(source, [[c["alias"] for c in source["cases"]]]) + install_backend(monkeypatch, source, family) + original = suite_backends.claude_generate + clock, allocations, deadlines = [100.0], [], [] + monkeypatch.setattr(suite_jobs.time, "monotonic", lambda: clock[0]) + + def routing(_): + clock[0] += 10 + + def backend(value, cancel, **kwargs): + allocations.append(value["execution"]["max_seconds"]) + deadlines.append(kwargs["job_deadline"]) + result = original(value, cancel, **kwargs) + clock[0] += 500 + return result + + monkeypatch.setattr(suite_backends, "switchyard_decision", routing) + monkeypatch.setattr(suite_backends, "claude_generate", backend) + jobs = suite_jobs.Jobs(tmp_path / "jobs.sqlite") + selected = preflight(source) + job, _ = jobs.submit("owner", "long-budget", source, selected, "192.168.22.8", launch=False) + jobs.run(job["job_id"], "owner", source, selected, "192.168.22.8") + result = jobs.get(job["job_id"], "owner") + assert result["status"] == "completed" and result["wall_seconds"] == 2510 + assert allocations == [3590, 3090, 2590, 2090, 1590] + assert deadlines == [3700] * 5 + assert result["execution_progress"]["job_remaining_seconds"] == 1090 + assert result["execution_progress"]["job_elapsed_seconds"] == 2510 + assert result["execution"]["max_seconds"] == 3600 + + +@pytest.mark.parametrize("provider_error", ["backend_timeout_or_unavailable", "timeout"]) +def test_distinct_backend_timeout_is_not_reclassified(monkeypatch, provider_error): + """Only the shared-deadline watchdog identifies job time exhaustion.""" + source = request(7) + def fail(*_, **__): + raise Problem(provider_error, 504, failure_stage="subprocess") + monkeypatch.setattr(suite_backends, "claude_generate", fail) + with pytest.raises(Problem, match="^" + provider_error + "$"): + suite_multipass.generate(source, preflight(source), threading.Event(), "192.168.22.8") + + +def test_routing_consumes_job_budget_and_stops_before_provider(tmp_path, monkeypatch): + source = request(7) + source["execution"]["max_seconds"] = 10 + clock = [100.0] + monkeypatch.setattr(suite_jobs.time, "monotonic", lambda: clock[0]) + monkeypatch.setattr(suite_backends, "switchyard_decision", lambda _: clock.__setitem__(0, 111)) + monkeypatch.setattr(suite_backends, "claude_generate", lambda *a, **k: pytest.fail("provider launched")) + jobs = suite_jobs.Jobs(tmp_path / "jobs.sqlite") + selected = preflight(source) + job, _ = jobs.submit("owner", "routing-budget", source, selected, "192.168.22.8", launch=False) + jobs.run(job["job_id"], "owner", source, selected, "192.168.22.8") + result = jobs.get(job["job_id"], "owner") + assert result["error"]["code"] == "job_time_budget_exhausted" + assert result["error"]["details"]["failure_stage"] == "routing" + + +def test_watchdog_job_expiry_reaches_status_with_zero_remaining(tmp_path, monkeypatch): + """The backend's shared-deadline error survives orchestration and persistence.""" + source = request(7) + source["execution"]["max_seconds"] = 3600 + clock = [100.0] + monkeypatch.setattr(suite_jobs.time, "monotonic", lambda: clock[0]) + monkeypatch.setattr(suite_backends, "switchyard_decision", lambda _: None) + def expired(value, cancel, *, invocation, progress, job_deadline): + assert value["execution"]["max_seconds"] == 3600 + clock[0] = job_deadline + raise Problem("job_time_budget_exhausted", 504, failure_stage="subprocess") + monkeypatch.setattr(suite_backends, "claude_generate", expired) + jobs = suite_jobs.Jobs(tmp_path / "jobs.sqlite") + selected = preflight(source) + job, _ = jobs.submit("owner", "watchdog-budget", source, selected, "192.168.22.8", launch=False) + jobs.run(job["job_id"], "owner", source, selected, "192.168.22.8") + result = jobs.get(job["job_id"], "owner") + assert result["status"] == "failed" and not jobs.results + assert result["error"]["code"] == "job_time_budget_exhausted" + assert result["error"]["details"]["failure_stage"] == "subprocess" + assert result["execution_progress"]["job_remaining_seconds"] == 0 diff --git a/testing/tests/test_suite_multipass.py b/testing/tests/test_suite_multipass.py index 4ffa9fbf..746da623 100644 --- a/testing/tests/test_suite_multipass.py +++ b/testing/tests/test_suite_multipass.py @@ -88,7 +88,7 @@ def install_backend(monkeypatch, source, reconciled, *, a=None, b=None, reviewed audited = review(audited or reviewed, reconciled) answers = {'proposal_a': a or public(reconciled), 'proposal_b': b or public(reconciled), 'reconciliation': reconciled, 'large_family_review': reviewed, 'decision_audit': audited} - def backend(value, cancel, *, invocation, progress=None): + def backend(value, cancel, *, invocation, progress=None, job_deadline=None): payload = json.loads(invocation['input']) assert {c['alias']: c for c in payload['suite']['cases']} == {c['alias']: c for c in source['cases']} assert payload['suite']['campaign'] == source['campaign'] @@ -170,12 +170,14 @@ def test_progress_reports_activity_without_content_or_false_percentage(monkeypat assert CANARY not in json.dumps(updates) -def test_thirty_minute_job_limit_retains_smaller_client_deadlines(): +def test_one_hour_opt_in_retains_thirty_minute_default(): source = request(7) assert source['execution']['max_seconds'] == 1800 source['execution']['max_seconds'] = 900 assert validate_request(source, ['claude'])['execution']['max_seconds'] == 900 - source['execution']['max_seconds'] = 1801 + source['execution']['max_seconds'] = 3600 + assert validate_request(source, ['claude'])['execution']['max_seconds'] == 3600 + source['execution']['max_seconds'] = 3601 with pytest.raises(Problem, match='invalid_timeout'): validate_request(source, ['claude']) diff --git a/testing/tests/test_suite_planning.py b/testing/tests/test_suite_planning.py index d79f1c85..04d702c5 100644 --- a/testing/tests/test_suite_planning.py +++ b/testing/tests/test_suite_planning.py @@ -10,7 +10,7 @@ import pytest sys.path.insert(0, str(Path(__file__).resolve().parents[2] / "services/hermes/scripts")) import suite_api import suite_backends -from suite_contract import (MODELS, PROMPT_REVISION, PROMPT_SHA256, SYSTEM, Problem, +from suite_contract import (CLAUDE_MODELS, MODELS, PROMPT_REVISION, PROMPT_SHA256, SYSTEM, Problem, preflight, prompt, validate_partition, validate_request, validate_result) from suite_jobs import Jobs from suite_synthetic import fixture, score @@ -243,7 +243,7 @@ def test_actual_cli_envelope_and_runtime_guards(failure, code): model = MODELS["claude"]["model"] init = {"type": "system", "subtype": "init", "model": model + "[1m]", "tools": ["StructuredOutput"], "mcp_servers": [], "plugins": []} - limits = {"contextWindow": 1000000, "maxOutputTokens": 64000, + limits = {"contextWindow": 1000000, "maxOutputTokens": CLAUDE_MODELS[model], "canonicalModel": model, "provider": "firstParty"} result = {"type": "result", "subtype": "success", "is_error": False, "modelUsage": {model + "[1m]": limits}, "usage": {},