diff --git a/docs/evidence/hermes_suite_claude55_20260929.json b/docs/evidence/hermes_suite_claude55_20260929.json new file mode 100644 index 00000000..5307bf55 --- /dev/null +++ b/docs/evidence/hermes_suite_claude55_20260929.json @@ -0,0 +1,2120 @@ +{ + "scope": "Public synthetic 14-case fixture only; no real roster input", + "cli_version": "2.1.285", + "cli_sha256": "33dad1ec615a2e08cc78b494f05c110e49916de2c79d78ec8799ebf46b233d29", + "effort": "medium", + "cost_basis": "CLI estimate, not observed subscription billing", + "runs": [ + { + "automatic_retries": 0, + "case_count": 14, + "execution_revision": "suite-multipass-v5-20260929", + "metadata": { + "cli_diagnostics": { + "api_error_status": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.08560400000000001, + "duration_api_ms": 19424, + "execution_revision": "suite-multipass-v5-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "medium", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 1780.7227951879613, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 9501, + "output_tokens": 2380, + "output_tokens_details": { + "thinking_tokens": 0 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_diagnostics_scope": "last_model_pass", + "compaction": false, + "compaction_signal": "CLI events and disabled compaction", + "cost_usd_estimate": 0.172604, + "duration_api_ms": 37620, + "final_task_count": 9, + "model": "claude-opus-5-5", + "model_pass_count": 3, + "model_usage": { + "claude-opus-5-5[1m]": { + "canonicalModel": "claude-opus-5-5", + "contextWindow": 1000000, + "maxOutputTokens": 128000, + "provider": "firstParty" + } + }, + "natural_family_count": 9, + "passes": [ + { + "allocated_cost_usd": 30.0, + "allocated_seconds": 1799.999867352657, + "case_order_sha256": "8c031138e0e155f0e44e28ee8126fcecd5c75316de0c40cc48f8d12550751ecd", + "cli_diagnostics": { + "api_error_status": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.043396000000000004, + "duration_api_ms": 9106, + "execution_revision": "suite-multipass-v5-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "medium", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 1799.999867352657, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 6214, + "output_tokens": 927, + "output_tokens_details": { + "thinking_tokens": 0 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 2, + "cost_usd_estimate": 0.043396000000000004, + "duration_api_ms": 9106, + "input_bytes": 16056, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 24248, + "input_token_count": null, + "model": "claude-opus-5-5", + "output_reservation_tokens": 11008, + "output_reservation_verified": false, + "provider": "claude", + "schema_sha256": "7ce70c8fecb046cbbbb146f34eb3cf9a8d245a3d72aea150ef1461fc6452bb63", + "stage": "proposal_a", + "system_sha256": "5df3f59dd1c4182d42fef8b58db69f0b18a9ff19ebaa7217363622484a29a04a", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 6214, + "output_tokens": 927, + "output_tokens_details": { + "thinking_tokens": 0 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 9.739 + }, + { + "allocated_cost_usd": 29.956604, + "allocated_seconds": 1790.2606900590472, + "case_order_sha256": "3ee82f66d96c5f70da1c64a9d191f699ec95dd02eadf8f7185eed2f34ff107b3", + "cli_diagnostics": { + "api_error_status": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.043604000000000004, + "duration_api_ms": 9090, + "execution_revision": "suite-multipass-v5-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "medium", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 1790.2606900590472, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 6221, + "output_tokens": 936, + "output_tokens_details": { + "thinking_tokens": 0 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 2, + "cost_usd_estimate": 0.043604000000000004, + "duration_api_ms": 9090, + "input_bytes": 16056, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 24248, + "input_token_count": null, + "model": "claude-opus-5-5", + "output_reservation_tokens": 11008, + "output_reservation_verified": false, + "provider": "claude", + "schema_sha256": "7ce70c8fecb046cbbbb146f34eb3cf9a8d245a3d72aea150ef1461fc6452bb63", + "stage": "proposal_b", + "system_sha256": "5df3f59dd1c4182d42fef8b58db69f0b18a9ff19ebaa7217363622484a29a04a", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 6221, + "output_tokens": 936, + "output_tokens_details": { + "thinking_tokens": 0 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 9.537 + }, + { + "allocated_cost_usd": 29.913, + "allocated_seconds": 1780.7227951879613, + "case_order_sha256": "8c031138e0e155f0e44e28ee8126fcecd5c75316de0c40cc48f8d12550751ecd", + "cli_diagnostics": { + "api_error_status": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.08560400000000001, + "duration_api_ms": 19424, + "execution_revision": "suite-multipass-v5-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "medium", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 1780.7227951879613, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 9501, + "output_tokens": 2380, + "output_tokens_details": { + "thinking_tokens": 0 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 2, + "cost_usd_estimate": 0.08560400000000001, + "duration_api_ms": 19424, + "input_bytes": 24641, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 32833, + "input_token_count": null, + "model": "claude-opus-5-5", + "output_reservation_tokens": 13344, + "output_reservation_verified": false, + "provider": "claude", + "schema_sha256": "def1e9458d237e8b66435fee29b2c263e6d20cc32f16e2622fe79101cc081f91", + "stage": "reconciliation", + "system_sha256": "7d85f607438ad80803f5f7a519b61355387886ed2f2cf7ab3671354628ffc6d5", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 9501, + "output_tokens": 2380, + "output_tokens_details": { + "thinking_tokens": 0 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 19.977 + } + ], + "policy_revision": "implementation-five-v1-20260929", + "review_summary": { + "capacity_divisions": [], + "counts": { + "final_singletons": 4, + "final_tasks": 9, + "natural_families": 9, + "natural_singletons": 4 + }, + "decision_audit": [], + "large_family_review": [], + "natural_families": [ + { + "common_work": "Control the thermal chamber cycle and take calibrated dimensional gauge readings", + "description": "Cycles a thermal chamber and measures enclosure expansion with a calibrated gauge against dimensional tolerance. The procedure details are unknown.", + "evidence": [ + { + "alias": "CASE-0001", + "field": "success_criteria", + "quote": "Expansion stays within the dimensional tolerance using a calibrated gauge" + } + ], + "members": [ + "CASE-0001" + ], + "name": "Thermal chamber expansion gauging", + "rationale": "This is the only case that needs a chamber and dimensional metrology. No other case uses that equipment.", + "uncertainty": "Neither the cycle profile nor the gauge interface is stated.", + "variation_sets": [ + [ + "CASE-0001" + ] + ] + }, + { + "common_work": "Drive the pulse source, capture the reset line digitally, and check timing against the supplied tolerance", + "description": "A controllable pulse source stops and resumes watchdog pulses. A digital capture fixture records reset-line timing, which is checked against the deadline. The two cases have identical content.", + "evidence": [ + { + "alias": "CASE-0002", + "field": "success_criteria", + "quote": "Measure reset-line timing with a digital capture fixture and check the deadline." + }, + { + "alias": "CASE-0013", + "field": "preconditions", + "quote": "Use a controllable pulse source and reset-line recorder" + } + ], + "members": [ + "CASE-0002", + "CASE-0013" + ], + "name": "Watchdog pulse reset-line timing", + "rationale": "Both cases are identical and use the same hardware-in-the-loop stimulus and capture setup.", + "uncertainty": "The two cases have identical content, so it is unclear how they are meant to differ.", + "variation_sets": [ + [ + "CASE-0002", + "CASE-0013" + ] + ] + }, + { + "common_work": "Load the report parser and severity policy fixture, count findings and compare rule IDs", + "description": "An offline parser reads watchdog static-analysis reports, counts findings by severity and checks rule IDs against a policy table. No firmware is run. The members differ only in their report inputs.", + "evidence": [ + { + "alias": "CASE-0003", + "field": "success_criteria", + "quote": "Count findings by severity and compare each rule identifier against the policy table." + } + ], + "members": [ + "CASE-0003", + "CASE-0009" + ], + "name": "Static-analysis report parsing", + "rationale": "The criteria are identical. The members differ only by case type and input set.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0003", + "CASE-0009" + ] + ] + }, + { + "common_work": "Build configuration text fixtures, call the parser and assert on the returned parser objects", + "description": "Configuration text is passed straight to the parser without the network stack. Tests assert accepted values or diagnostic positions. The members are nominal and boundary input variations.", + "evidence": [ + { + "alias": "CASE-0004", + "field": "success_criteria", + "quote": "Assert accepted values or diagnostic positions using returned parser objects." + } + ], + "members": [ + "CASE-0004", + "CASE-0010" + ], + "name": "Configuration parser text fixtures", + "rationale": "Both cases use the same parser harness. Only the inputs differ.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0004", + "CASE-0010" + ] + ] + }, + { + "common_work": "Run concurrent task drivers against the bounded queue fixture, with sequence-number assertions", + "description": "Concurrent producer and consumer drivers fill a bounded queue. Tests observe queue depth, rejected writes, sequence ordering and recovery after draining. The members are nominal and boundary variations.", + "evidence": [ + { + "alias": "CASE-0005", + "field": "success_criteria", + "quote": "Observe queue depth, rejected writes, delivery ordering, and recovery after draining." + } + ], + "members": [ + "CASE-0005", + "CASE-0011" + ], + "name": "Bounded queue concurrency drivers", + "rationale": "Both cases use the same concurrency harness and observations. Only the inputs differ.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0005", + "CASE-0011" + ] + ] + }, + { + "common_work": "Set up identity fixtures and the in-memory policy store, then check decisions and audit events against the matrix", + "description": "Identity fixtures and an in-memory policy store feed role-scoped operations to the decision function. Allow/deny results and audit fields are compared to the permissions matrix. The members are nominal and boundary.", + "evidence": [ + { + "alias": "CASE-0006", + "field": "success_criteria", + "quote": "Compare allow or deny decisions and audit-event fields against the permissions matrix." + } + ], + "members": [ + "CASE-0006", + "CASE-0012" + ], + "name": "Authorization decision matrix", + "rationale": "Both cases share the same harness and assertions. Only the input data differs.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0006", + "CASE-0012" + ] + ] + }, + { + "common_work": "Build byte-array frames, call the decoder and compare the decoded object's fields", + "description": "A byte-array builder creates encoded watchdog status frames for a decoder call. Tests compare decoded fields, checksum status and the rejection code. No hardware is needed.", + "evidence": [ + { + "alias": "CASE-0007", + "field": "success_criteria", + "quote": "Capture the decoded object and compare fields, checksum status, and rejection code." + } + ], + "members": [ + "CASE-0007" + ], + "name": "Status frame decoder byte vectors", + "rationale": "This is a pure software decoder harness, unlike the hardware pulse cases and the report parser.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0007" + ] + ] + }, + { + "common_work": "Record acoustic output, run spectral analysis and compare peaks to the threshold", + "description": "Acoustic output is recorded in an anechoic fixture and spectral peaks are compared to a threshold that depends on frequency. The procedure details are unknown.", + "evidence": [ + { + "alias": "CASE-0008", + "field": "success_criteria", + "quote": "Spectral peak magnitude stays below the supplied frequency-dependent threshold" + } + ], + "members": [ + "CASE-0008" + ], + "name": "Anechoic acoustic spectrum capture", + "rationale": "This case needs its own acoustic measurement equipment.", + "uncertainty": "The fixture procedure is unknown.", + "variation_sets": [ + [ + "CASE-0008" + ] + ] + }, + { + "common_work": "Run two clean container builds, remove allowed metadata and compare digests", + "description": "The same source is rebuilt twice in clean containers, and artifact digests are compared after removing allowed timestamp metadata. The container setup details are unknown.", + "evidence": [ + { + "alias": "CASE-0014", + "field": "success_criteria", + "quote": "Compare artifact digests after removing only explicitly allowed timestamp metadata" + } + ], + "members": [ + "CASE-0014" + ], + "name": "Reproducible build digest comparison", + "rationale": "This case uses a build-pipeline workflow that no other case shares.", + "uncertainty": "The container setup is unknown. The case is labelled fault injection, but no fault mechanism is described.", + "variation_sets": [ + [ + "CASE-0014" + ] + ] + } + ], + "policy_revision": "implementation-five-v1-20260929", + "proposal_disagreements": { + "aliases": [], + "pair_count": 0, + "proposal_a": [ + [ + "CASE-0001" + ], + [ + "CASE-0002", + "CASE-0013" + ], + [ + "CASE-0003", + "CASE-0009" + ], + [ + "CASE-0004", + "CASE-0010" + ], + [ + "CASE-0005", + "CASE-0011" + ], + [ + "CASE-0006", + "CASE-0012" + ], + [ + "CASE-0007" + ], + [ + "CASE-0008" + ], + [ + "CASE-0014" + ] + ], + "proposal_b": [ + [ + "CASE-0001" + ], + [ + "CASE-0002", + "CASE-0013" + ], + [ + "CASE-0003", + "CASE-0009" + ], + [ + "CASE-0004", + "CASE-0010" + ], + [ + "CASE-0005", + "CASE-0011" + ], + [ + "CASE-0006", + "CASE-0012" + ], + [ + "CASE-0007" + ], + [ + "CASE-0008" + ], + [ + "CASE-0014" + ] + ] + }, + "reconciled_families": [ + { + "common_work": "Control the thermal chamber cycle and take calibrated dimensional gauge readings", + "description": "Cycles a thermal chamber and measures enclosure expansion with a calibrated gauge against dimensional tolerance. The procedure details are unknown.", + "evidence": [ + { + "alias": "CASE-0001", + "field": "success_criteria", + "quote": "Expansion stays within the dimensional tolerance using a calibrated gauge" + } + ], + "members": [ + "CASE-0001" + ], + "name": "Thermal chamber expansion gauging", + "rationale": "This is the only case that needs a chamber and dimensional metrology. No other case uses that equipment.", + "uncertainty": "Neither the cycle profile nor the gauge interface is stated.", + "variation_sets": [ + [ + "CASE-0001" + ] + ] + }, + { + "common_work": "Drive the pulse source, capture the reset line digitally, and check timing against the supplied tolerance", + "description": "A controllable pulse source stops and resumes watchdog pulses. A digital capture fixture records reset-line timing, which is checked against the deadline. The two cases have identical content.", + "evidence": [ + { + "alias": "CASE-0002", + "field": "success_criteria", + "quote": "Measure reset-line timing with a digital capture fixture and check the deadline." + }, + { + "alias": "CASE-0013", + "field": "preconditions", + "quote": "Use a controllable pulse source and reset-line recorder" + } + ], + "members": [ + "CASE-0002", + "CASE-0013" + ], + "name": "Watchdog pulse reset-line timing", + "rationale": "Both cases are identical and use the same hardware-in-the-loop stimulus and capture setup.", + "uncertainty": "The two cases have identical content, so it is unclear how they are meant to differ.", + "variation_sets": [ + [ + "CASE-0002", + "CASE-0013" + ] + ] + }, + { + "common_work": "Load the report parser and severity policy fixture, count findings and compare rule IDs", + "description": "An offline parser reads watchdog static-analysis reports, counts findings by severity and checks rule IDs against a policy table. No firmware is run. The members differ only in their report inputs.", + "evidence": [ + { + "alias": "CASE-0003", + "field": "success_criteria", + "quote": "Count findings by severity and compare each rule identifier against the policy table." + } + ], + "members": [ + "CASE-0003", + "CASE-0009" + ], + "name": "Static-analysis report parsing", + "rationale": "The criteria are identical. The members differ only by case type and input set.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0003", + "CASE-0009" + ] + ] + }, + { + "common_work": "Build configuration text fixtures, call the parser and assert on the returned parser objects", + "description": "Configuration text is passed straight to the parser without the network stack. Tests assert accepted values or diagnostic positions. The members are nominal and boundary input variations.", + "evidence": [ + { + "alias": "CASE-0004", + "field": "success_criteria", + "quote": "Assert accepted values or diagnostic positions using returned parser objects." + } + ], + "members": [ + "CASE-0004", + "CASE-0010" + ], + "name": "Configuration parser text fixtures", + "rationale": "Both cases use the same parser harness. Only the inputs differ.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0004", + "CASE-0010" + ] + ] + }, + { + "common_work": "Run concurrent task drivers against the bounded queue fixture, with sequence-number assertions", + "description": "Concurrent producer and consumer drivers fill a bounded queue. Tests observe queue depth, rejected writes, sequence ordering and recovery after draining. The members are nominal and boundary variations.", + "evidence": [ + { + "alias": "CASE-0005", + "field": "success_criteria", + "quote": "Observe queue depth, rejected writes, delivery ordering, and recovery after draining." + } + ], + "members": [ + "CASE-0005", + "CASE-0011" + ], + "name": "Bounded queue concurrency drivers", + "rationale": "Both cases use the same concurrency harness and observations. Only the inputs differ.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0005", + "CASE-0011" + ] + ] + }, + { + "common_work": "Set up identity fixtures and the in-memory policy store, then check decisions and audit events against the matrix", + "description": "Identity fixtures and an in-memory policy store feed role-scoped operations to the decision function. Allow/deny results and audit fields are compared to the permissions matrix. The members are nominal and boundary.", + "evidence": [ + { + "alias": "CASE-0006", + "field": "success_criteria", + "quote": "Compare allow or deny decisions and audit-event fields against the permissions matrix." + } + ], + "members": [ + "CASE-0006", + "CASE-0012" + ], + "name": "Authorization decision matrix", + "rationale": "Both cases share the same harness and assertions. Only the input data differs.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0006", + "CASE-0012" + ] + ] + }, + { + "common_work": "Build byte-array frames, call the decoder and compare the decoded object's fields", + "description": "A byte-array builder creates encoded watchdog status frames for a decoder call. Tests compare decoded fields, checksum status and the rejection code. No hardware is needed.", + "evidence": [ + { + "alias": "CASE-0007", + "field": "success_criteria", + "quote": "Capture the decoded object and compare fields, checksum status, and rejection code." + } + ], + "members": [ + "CASE-0007" + ], + "name": "Status frame decoder byte vectors", + "rationale": "This is a pure software decoder harness, unlike the hardware pulse cases and the report parser.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0007" + ] + ] + }, + { + "common_work": "Record acoustic output, run spectral analysis and compare peaks to the threshold", + "description": "Acoustic output is recorded in an anechoic fixture and spectral peaks are compared to a threshold that depends on frequency. The procedure details are unknown.", + "evidence": [ + { + "alias": "CASE-0008", + "field": "success_criteria", + "quote": "Spectral peak magnitude stays below the supplied frequency-dependent threshold" + } + ], + "members": [ + "CASE-0008" + ], + "name": "Anechoic acoustic spectrum capture", + "rationale": "This case needs its own acoustic measurement equipment.", + "uncertainty": "The fixture procedure is unknown.", + "variation_sets": [ + [ + "CASE-0008" + ] + ] + }, + { + "common_work": "Run two clean container builds, remove allowed metadata and compare digests", + "description": "The same source is rebuilt twice in clean containers, and artifact digests are compared after removing allowed timestamp metadata. The container setup details are unknown.", + "evidence": [ + { + "alias": "CASE-0014", + "field": "success_criteria", + "quote": "Compare artifact digests after removing only explicitly allowed timestamp metadata" + } + ], + "members": [ + "CASE-0014" + ], + "name": "Reproducible build digest comparison", + "rationale": "This case uses a build-pipeline workflow that no other case shares.", + "uncertainty": "The container setup is unknown. The case is labelled fault injection, but no fault mechanism is described.", + "variation_sets": [ + [ + "CASE-0014" + ] + ] + } + ], + "sizing_is_not_semantic_evidence": true, + "unresolved_uncertainties": [ + { + "family_name": "Thermal chamber expansion gauging", + "members": [ + "CASE-0001" + ], + "uncertainty": "Neither the cycle profile nor the gauge interface is stated." + }, + { + "family_name": "Watchdog pulse reset-line timing", + "members": [ + "CASE-0002", + "CASE-0013" + ], + "uncertainty": "The two cases have identical content, so it is unclear how they are meant to differ." + }, + { + "family_name": "Anechoic acoustic spectrum capture", + "members": [ + "CASE-0008" + ], + "uncertainty": "The fixture procedure is unknown." + }, + { + "family_name": "Reproducible build digest comparison", + "members": [ + "CASE-0014" + ], + "uncertainty": "The container setup is unknown. The case is labelled fault injection, but no fault mechanism is described." + } + ] + }, + "singleton_statistics": { + "final_singletons": 4, + "natural_singletons": 4 + }, + "temporary_files_deleted": true, + "truncation": false, + "turns": 6, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 21936, + "output_tokens": 4243, + "output_tokens_details": { + "thinking_tokens": 0 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "model": "claude-opus-5-5", + "prompt_revision": "implementation-proximity-multipass-v4-20260929", + "quality": { + "coverage": true, + "false_merge_pairs": 0, + "families": 9, + "missed_merge_pairs": 0, + "pair_precision": 1.0, + "pair_recall": 1.0 + }, + "request_bytes": 9851, + "result": { + "groups": [ + { + "description": "Acoustic output is recorded in an anechoic fixture and spectral peaks are compared to a threshold that depends on frequency. The procedure details are unknown.", + "members": [ + "CASE-0008" + ], + "name": "Anechoic acoustic spectrum capture" + }, + { + "description": "Identity fixtures and an in-memory policy store feed role-scoped operations to the decision function. Allow/deny results and audit fields are compared to the permissions matrix. The members are nominal and boundary.", + "members": [ + "CASE-0006", + "CASE-0012" + ], + "name": "Authorization decision matrix" + }, + { + "description": "Concurrent producer and consumer drivers fill a bounded queue. Tests observe queue depth, rejected writes, sequence ordering and recovery after draining. The members are nominal and boundary variations.", + "members": [ + "CASE-0005", + "CASE-0011" + ], + "name": "Bounded queue concurrency drivers" + }, + { + "description": "Configuration text is passed straight to the parser without the network stack. Tests assert accepted values or diagnostic positions. The members are nominal and boundary input variations.", + "members": [ + "CASE-0004", + "CASE-0010" + ], + "name": "Configuration parser text fixtures" + }, + { + "description": "The same source is rebuilt twice in clean containers, and artifact digests are compared after removing allowed timestamp metadata. The container setup details are unknown.", + "members": [ + "CASE-0014" + ], + "name": "Reproducible build digest comparison" + }, + { + "description": "An offline parser reads watchdog static-analysis reports, counts findings by severity and checks rule IDs against a policy table. No firmware is run. The members differ only in their report inputs.", + "members": [ + "CASE-0003", + "CASE-0009" + ], + "name": "Static-analysis report parsing" + }, + { + "description": "A byte-array builder creates encoded watchdog status frames for a decoder call. Tests compare decoded fields, checksum status and the rejection code. No hardware is needed.", + "members": [ + "CASE-0007" + ], + "name": "Status frame decoder byte vectors" + }, + { + "description": "Cycles a thermal chamber and measures enclosure expansion with a calibrated gauge against dimensional tolerance. The procedure details are unknown.", + "members": [ + "CASE-0001" + ], + "name": "Thermal chamber expansion gauging" + }, + { + "description": "A controllable pulse source stops and resumes watchdog pulses. A digital capture fixture records reset-line timing, which is checked against the deadline. The two cases have identical content.", + "members": [ + "CASE-0002", + "CASE-0013" + ], + "name": "Watchdog pulse reset-line timing" + } + ] + }, + "selection": { + "backend": "claude-code-2.1.285", + "case_count": 14, + "cli_model": "claude-opus-5-5[1m]", + "configuration_revision": "suite-v6-20260929", + "context": 1000000, + "enabled": true, + "execution_revision": "suite-multipass-v5-20260929", + "input_bytes": 16056, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 24248, + "input_token_count": null, + "later_pass_capacity_verified": false, + "later_pass_checks": "before_each_invocation", + "max_final_group_cases": 5, + "max_final_name_characters": 64, + "max_turns": 6, + "maximum_model_passes": 5, + "minimum_model_passes": 3, + "model": "claude-opus-5-5", + "output": 64000, + "output_reservation_tokens": 11008, + "output_reservation_verified": false, + "overhead": 8192, + "policy_revision": "implementation-five-v1-20260929", + "prompt_revision": "implementation-proximity-multipass-v4-20260929", + "prompt_sha256": "253314b009ea82fc44695260ece8edd501ab60dc40ab3cc9e1ac9a0acb2017c0", + "provider": "claude", + "reasoning": "medium", + "reported_output": 128000, + "source_sha256": "99fd35cb089e57285670f6def4339a86754316e67ce3e53ec7efa95968534ca9" + }, + "singletons": 4, + "status": "completed", + "test_path": "server_side_multipass_backend", + "wall_seconds": 39.256 + }, + { + "automatic_retries": 0, + "case_count": 14, + "execution_revision": "suite-multipass-v5-20260929", + "metadata": { + "cli_diagnostics": { + "api_error_status": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.040824, + "duration_api_ms": 13213, + "execution_revision": "suite-multipass-v5-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "medium", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 1788.8493417161517, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 9192, + "output_tokens": 2244, + "output_tokens_details": { + "thinking_tokens": 186 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_diagnostics_scope": "last_model_pass", + "compaction": false, + "compaction_signal": "CLI events and disabled compaction", + "cost_usd_estimate": 0.081016, + "duration_api_ms": 23449, + "final_task_count": 9, + "model": "claude-sonnet-5-5", + "model_pass_count": 3, + "model_usage": { + "claude-sonnet-5-5[1m]": { + "canonicalModel": "claude-sonnet-5-5", + "contextWindow": 1000000, + "maxOutputTokens": 128000, + "provider": "firstParty" + } + }, + "natural_family_count": 9, + "passes": [ + { + "allocated_cost_usd": 30.0, + "allocated_seconds": 1799.9998973542824, + "case_order_sha256": "8c031138e0e155f0e44e28ee8126fcecd5c75316de0c40cc48f8d12550751ecd", + "cli_diagnostics": { + "api_error_status": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.01975, + "duration_api_ms": 4733, + "execution_revision": "suite-multipass-v5-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "medium", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 1799.9998973542824, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 6120, + "output_tokens": 751, + "output_tokens_details": { + "thinking_tokens": 0 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 2, + "cost_usd_estimate": 0.01975, + "duration_api_ms": 4733, + "input_bytes": 16056, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 24248, + "input_token_count": null, + "model": "claude-sonnet-5-5", + "output_reservation_tokens": 11008, + "output_reservation_verified": false, + "provider": "claude", + "schema_sha256": "7ce70c8fecb046cbbbb146f34eb3cf9a8d245a3d72aea150ef1461fc6452bb63", + "stage": "proposal_a", + "system_sha256": "5df3f59dd1c4182d42fef8b58db69f0b18a9ff19ebaa7217363622484a29a04a", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 6120, + "output_tokens": 751, + "output_tokens_details": { + "thinking_tokens": 0 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 5.223 + }, + { + "allocated_cost_usd": 29.98025, + "allocated_seconds": 1794.7759290733375, + "case_order_sha256": "3ee82f66d96c5f70da1c64a9d191f699ec95dd02eadf8f7185eed2f34ff107b3", + "cli_diagnostics": { + "api_error_status": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.020442000000000002, + "duration_api_ms": 5503, + "execution_revision": "suite-multipass-v5-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "medium", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 1794.7759290733375, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 6126, + "output_tokens": 819, + "output_tokens_details": { + "thinking_tokens": 0 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 2, + "cost_usd_estimate": 0.020442000000000002, + "duration_api_ms": 5503, + "input_bytes": 16056, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 24248, + "input_token_count": null, + "model": "claude-sonnet-5-5", + "output_reservation_tokens": 11008, + "output_reservation_verified": false, + "provider": "claude", + "schema_sha256": "7ce70c8fecb046cbbbb146f34eb3cf9a8d245a3d72aea150ef1461fc6452bb63", + "stage": "proposal_b", + "system_sha256": "5df3f59dd1c4182d42fef8b58db69f0b18a9ff19ebaa7217363622484a29a04a", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 6126, + "output_tokens": 819, + "output_tokens_details": { + "thinking_tokens": 0 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 5.925 + }, + { + "allocated_cost_usd": 29.959808, + "allocated_seconds": 1788.8493417161517, + "case_order_sha256": "8c031138e0e155f0e44e28ee8126fcecd5c75316de0c40cc48f8d12550751ecd", + "cli_diagnostics": { + "api_error_status": null, + "assistant_json_text_present": false, + "assistant_structured_tool_input_present": true, + "compaction_event_seen": false, + "configured_max_output_tokens": 64000, + "cost_usd_estimate": 0.040824, + "duration_api_ms": 13213, + "execution_revision": "suite-multipass-v5-20260929", + "exit_code": 0, + "final_event_seen": true, + "final_event_subtype": "success", + "final_event_type": "result", + "final_is_error": false, + "final_json_text_present": true, + "init_event_seen": true, + "invalid_event_count": 0, + "last_assistant_stop_reason": null, + "max_turns": 6, + "observed_model_limits": [ + { + "contextWindow": 1000000, + "maxOutputTokens": 128000 + } + ], + "provider_stop_reason": "tool_use", + "provider_timeout_seconds": null, + "reasoning_effort": "medium", + "reasoning_token_limit": null, + "structured_output_is_object": true, + "structured_output_location": "result.structured_output", + "structured_output_present": true, + "structured_retry_limit_reached": false, + "subprocess_timeout_seconds": 1788.8493417161517, + "termination_reason": "exited", + "termination_signal": null, + "turn_limit_reached": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 9192, + "output_tokens": 2244, + "output_tokens_details": { + "thinking_tokens": 186 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "cli_turns": 2, + "cost_usd_estimate": 0.040824, + "duration_api_ms": 13213, + "input_bytes": 23933, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 32125, + "input_token_count": null, + "model": "claude-sonnet-5-5", + "output_reservation_tokens": 13344, + "output_reservation_verified": false, + "provider": "claude", + "schema_sha256": "def1e9458d237e8b66435fee29b2c263e6d20cc32f16e2622fe79101cc081f91", + "stage": "reconciliation", + "system_sha256": "7d85f607438ad80803f5f7a519b61355387886ed2f2cf7ab3671354628ffc6d5", + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 9192, + "output_tokens": 2244, + "output_tokens_details": { + "thinking_tokens": 186 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + }, + "wall_seconds": 13.647 + } + ], + "policy_revision": "implementation-five-v1-20260929", + "review_summary": { + "capacity_divisions": [], + "counts": { + "final_singletons": 4, + "final_tasks": 9, + "natural_families": 9, + "natural_singletons": 4 + }, + "decision_audit": [], + "large_family_review": [], + "natural_families": [ + { + "common_work": "Thermal chamber cycling with calibrated gauge dimensional measurement.", + "description": "Single case: thermal chamber cycling with calibrated gauge measurement of enclosure dimensional expansion against tolerance.", + "evidence": [ + { + "alias": "CASE-0001", + "field": "success_criteria", + "quote": "Expansion stays within the dimensional tolerance using a calibrated gauge" + } + ], + "members": [ + "CASE-0001" + ], + "name": "Thermal chamber expansion gauge measurement", + "rationale": "Unique physical chamber and gauge equipment; no other case shares this machinery.", + "uncertainty": "Procedure details are stated as unknown.", + "variation_sets": [ + [ + "CASE-0001" + ] + ] + }, + { + "common_work": "Pulse source control, reset-line digital capture fixture, deadline assertion.", + "description": "Controllable pulse source stopped/resumed with digital capture of reset-line timing and deadline check; CASE-0002 and CASE-0013 are identical in content.", + "evidence": [ + { + "alias": "CASE-0002", + "field": "success_criteria", + "quote": "Measure reset-line timing with a digital capture fixture and check the deadline." + } + ], + "members": [ + "CASE-0002", + "CASE-0013" + ], + "name": "Watchdog pulse reset-timing capture", + "rationale": "Identical hardware-in-loop stimulus and capture machinery.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0002", + "CASE-0013" + ] + ] + }, + { + "common_work": "Report parser and policy fixture; severity counts and rule ID policy comparison.", + "description": "Offline static-analysis report parser with severity policy fixture; count findings by severity and check rule IDs; nominal vs boundary inputs.", + "evidence": [ + { + "alias": "CASE-0003", + "field": "success_criteria", + "quote": "Count findings by severity and compare each rule identifier against the policy table." + } + ], + "members": [ + "CASE-0003", + "CASE-0009" + ], + "name": "Static-analysis report parsing", + "rationale": "Same parser and assertion structure; only inputs differ.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0003" + ], + [ + "CASE-0009" + ] + ] + }, + { + "common_work": "Config text fixtures, direct parser call, assertions on returned parser objects.", + "description": "Call configuration parser with text fixtures; assert accepted values or diagnostic positions; nominal and boundary inputs.", + "evidence": [ + { + "alias": "CASE-0004", + "field": "success_criteria", + "quote": "Assert accepted values or diagnostic positions using returned parser objects." + } + ], + "members": [ + "CASE-0004", + "CASE-0010" + ], + "name": "Configuration parser input tests", + "rationale": "Same driver and assertions; inputs vary.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0004" + ], + [ + "CASE-0010" + ] + ] + }, + { + "common_work": "Concurrent task drivers, bounded queue fixture, sequence-number assertions.", + "description": "Concurrent producer/consumer task drivers against a bounded queue; observe depth, rejected writes, ordering, drain recovery; nominal vs boundary.", + "evidence": [ + { + "alias": "CASE-0005", + "field": "success_criteria", + "quote": "Observe queue depth, rejected writes, delivery ordering, and recovery after draining." + } + ], + "members": [ + "CASE-0005", + "CASE-0011" + ], + "name": "Bounded queue producer-consumer tests", + "rationale": "Identical machinery and observations.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0005" + ], + [ + "CASE-0011" + ] + ] + }, + { + "common_work": "Identity fixtures, in-memory policy store, decision and audit comparison to matrix.", + "description": "Role-scoped operations submitted with identity fixtures and in-memory policy store; compare allow/deny and audit fields to permissions matrix.", + "evidence": [ + { + "alias": "CASE-0006", + "field": "success_criteria", + "quote": "Compare allow or deny decisions and audit-event fields against the permissions matrix." + } + ], + "members": [ + "CASE-0006", + "CASE-0012" + ], + "name": "Authorization decision matrix tests", + "rationale": "Same driver and assertion structure; inputs vary.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0006" + ], + [ + "CASE-0012" + ] + ] + }, + { + "common_work": "Byte-array frame builder, decoder call, decoded object comparison.", + "description": "Single case: byte-array builder feeding the decoder; compare fields, checksum status, rejection code.", + "evidence": [ + { + "alias": "CASE-0007", + "field": "success_criteria", + "quote": "Capture the decoded object and compare fields, checksum status, and rejection code." + } + ], + "members": [ + "CASE-0007" + ], + "name": "Status frame decoder tests", + "rationale": "Distinct decoder machinery from the other parsers.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0007" + ] + ] + }, + { + "common_work": "Anechoic fixture recording and spectral peak analysis.", + "description": "Single case: acoustic recording in anechoic fixture with spectral peak threshold check.", + "evidence": [ + { + "alias": "CASE-0008", + "field": "success_criteria", + "quote": "Spectral peak magnitude stays below the supplied frequency-dependent threshold" + } + ], + "members": [ + "CASE-0008" + ], + "name": "Anechoic acoustic spectral measurement", + "rationale": "Unique acoustic equipment and spectral analysis.", + "uncertainty": "Procedure details are unknown.", + "variation_sets": [ + [ + "CASE-0008" + ] + ] + }, + { + "common_work": "Clean container builds and digest comparison.", + "description": "Single case: rebuild same source twice in clean containers and compare artifact digests excluding allowed timestamps.", + "evidence": [ + { + "alias": "CASE-0014", + "field": "success_criteria", + "quote": "Compare artifact digests after removing only explicitly allowed timestamp metadata" + } + ], + "members": [ + "CASE-0014" + ], + "name": "Reproducible build digest comparison", + "rationale": "Unique build-system workflow.", + "uncertainty": "Procedure details are unknown.", + "variation_sets": [ + [ + "CASE-0014" + ] + ] + } + ], + "policy_revision": "implementation-five-v1-20260929", + "proposal_disagreements": { + "aliases": [], + "pair_count": 0, + "proposal_a": [ + [ + "CASE-0001" + ], + [ + "CASE-0002", + "CASE-0013" + ], + [ + "CASE-0003", + "CASE-0009" + ], + [ + "CASE-0004", + "CASE-0010" + ], + [ + "CASE-0005", + "CASE-0011" + ], + [ + "CASE-0006", + "CASE-0012" + ], + [ + "CASE-0007" + ], + [ + "CASE-0008" + ], + [ + "CASE-0014" + ] + ], + "proposal_b": [ + [ + "CASE-0001" + ], + [ + "CASE-0002", + "CASE-0013" + ], + [ + "CASE-0003", + "CASE-0009" + ], + [ + "CASE-0004", + "CASE-0010" + ], + [ + "CASE-0005", + "CASE-0011" + ], + [ + "CASE-0006", + "CASE-0012" + ], + [ + "CASE-0007" + ], + [ + "CASE-0008" + ], + [ + "CASE-0014" + ] + ] + }, + "reconciled_families": [ + { + "common_work": "Thermal chamber cycling with calibrated gauge dimensional measurement.", + "description": "Single case: thermal chamber cycling with calibrated gauge measurement of enclosure dimensional expansion against tolerance.", + "evidence": [ + { + "alias": "CASE-0001", + "field": "success_criteria", + "quote": "Expansion stays within the dimensional tolerance using a calibrated gauge" + } + ], + "members": [ + "CASE-0001" + ], + "name": "Thermal chamber expansion gauge measurement", + "rationale": "Unique physical chamber and gauge equipment; no other case shares this machinery.", + "uncertainty": "Procedure details are stated as unknown.", + "variation_sets": [ + [ + "CASE-0001" + ] + ] + }, + { + "common_work": "Pulse source control, reset-line digital capture fixture, deadline assertion.", + "description": "Controllable pulse source stopped/resumed with digital capture of reset-line timing and deadline check; CASE-0002 and CASE-0013 are identical in content.", + "evidence": [ + { + "alias": "CASE-0002", + "field": "success_criteria", + "quote": "Measure reset-line timing with a digital capture fixture and check the deadline." + } + ], + "members": [ + "CASE-0002", + "CASE-0013" + ], + "name": "Watchdog pulse reset-timing capture", + "rationale": "Identical hardware-in-loop stimulus and capture machinery.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0002", + "CASE-0013" + ] + ] + }, + { + "common_work": "Report parser and policy fixture; severity counts and rule ID policy comparison.", + "description": "Offline static-analysis report parser with severity policy fixture; count findings by severity and check rule IDs; nominal vs boundary inputs.", + "evidence": [ + { + "alias": "CASE-0003", + "field": "success_criteria", + "quote": "Count findings by severity and compare each rule identifier against the policy table." + } + ], + "members": [ + "CASE-0003", + "CASE-0009" + ], + "name": "Static-analysis report parsing", + "rationale": "Same parser and assertion structure; only inputs differ.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0003" + ], + [ + "CASE-0009" + ] + ] + }, + { + "common_work": "Config text fixtures, direct parser call, assertions on returned parser objects.", + "description": "Call configuration parser with text fixtures; assert accepted values or diagnostic positions; nominal and boundary inputs.", + "evidence": [ + { + "alias": "CASE-0004", + "field": "success_criteria", + "quote": "Assert accepted values or diagnostic positions using returned parser objects." + } + ], + "members": [ + "CASE-0004", + "CASE-0010" + ], + "name": "Configuration parser input tests", + "rationale": "Same driver and assertions; inputs vary.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0004" + ], + [ + "CASE-0010" + ] + ] + }, + { + "common_work": "Concurrent task drivers, bounded queue fixture, sequence-number assertions.", + "description": "Concurrent producer/consumer task drivers against a bounded queue; observe depth, rejected writes, ordering, drain recovery; nominal vs boundary.", + "evidence": [ + { + "alias": "CASE-0005", + "field": "success_criteria", + "quote": "Observe queue depth, rejected writes, delivery ordering, and recovery after draining." + } + ], + "members": [ + "CASE-0005", + "CASE-0011" + ], + "name": "Bounded queue producer-consumer tests", + "rationale": "Identical machinery and observations.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0005" + ], + [ + "CASE-0011" + ] + ] + }, + { + "common_work": "Identity fixtures, in-memory policy store, decision and audit comparison to matrix.", + "description": "Role-scoped operations submitted with identity fixtures and in-memory policy store; compare allow/deny and audit fields to permissions matrix.", + "evidence": [ + { + "alias": "CASE-0006", + "field": "success_criteria", + "quote": "Compare allow or deny decisions and audit-event fields against the permissions matrix." + } + ], + "members": [ + "CASE-0006", + "CASE-0012" + ], + "name": "Authorization decision matrix tests", + "rationale": "Same driver and assertion structure; inputs vary.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0006" + ], + [ + "CASE-0012" + ] + ] + }, + { + "common_work": "Byte-array frame builder, decoder call, decoded object comparison.", + "description": "Single case: byte-array builder feeding the decoder; compare fields, checksum status, rejection code.", + "evidence": [ + { + "alias": "CASE-0007", + "field": "success_criteria", + "quote": "Capture the decoded object and compare fields, checksum status, and rejection code." + } + ], + "members": [ + "CASE-0007" + ], + "name": "Status frame decoder tests", + "rationale": "Distinct decoder machinery from the other parsers.", + "uncertainty": "", + "variation_sets": [ + [ + "CASE-0007" + ] + ] + }, + { + "common_work": "Anechoic fixture recording and spectral peak analysis.", + "description": "Single case: acoustic recording in anechoic fixture with spectral peak threshold check.", + "evidence": [ + { + "alias": "CASE-0008", + "field": "success_criteria", + "quote": "Spectral peak magnitude stays below the supplied frequency-dependent threshold" + } + ], + "members": [ + "CASE-0008" + ], + "name": "Anechoic acoustic spectral measurement", + "rationale": "Unique acoustic equipment and spectral analysis.", + "uncertainty": "Procedure details are unknown.", + "variation_sets": [ + [ + "CASE-0008" + ] + ] + }, + { + "common_work": "Clean container builds and digest comparison.", + "description": "Single case: rebuild same source twice in clean containers and compare artifact digests excluding allowed timestamps.", + "evidence": [ + { + "alias": "CASE-0014", + "field": "success_criteria", + "quote": "Compare artifact digests after removing only explicitly allowed timestamp metadata" + } + ], + "members": [ + "CASE-0014" + ], + "name": "Reproducible build digest comparison", + "rationale": "Unique build-system workflow.", + "uncertainty": "Procedure details are unknown.", + "variation_sets": [ + [ + "CASE-0014" + ] + ] + } + ], + "sizing_is_not_semantic_evidence": true, + "unresolved_uncertainties": [ + { + "family_name": "Thermal chamber expansion gauge measurement", + "members": [ + "CASE-0001" + ], + "uncertainty": "Procedure details are stated as unknown." + }, + { + "family_name": "Anechoic acoustic spectral measurement", + "members": [ + "CASE-0008" + ], + "uncertainty": "Procedure details are unknown." + }, + { + "family_name": "Reproducible build digest comparison", + "members": [ + "CASE-0014" + ], + "uncertainty": "Procedure details are unknown." + } + ] + }, + "singleton_statistics": { + "final_singletons": 4, + "natural_singletons": 4 + }, + "temporary_files_deleted": true, + "truncation": false, + "turns": 6, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "input_tokens": 21438, + "output_tokens": 3814, + "output_tokens_details": { + "thinking_tokens": 186 + }, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + } + } + }, + "model": "claude-sonnet-5-5", + "prompt_revision": "implementation-proximity-multipass-v4-20260929", + "quality": { + "coverage": true, + "false_merge_pairs": 0, + "families": 9, + "missed_merge_pairs": 0, + "pair_precision": 1.0, + "pair_recall": 1.0 + }, + "request_bytes": 9851, + "result": { + "groups": [ + { + "description": "Single case: acoustic recording in anechoic fixture with spectral peak threshold check.", + "members": [ + "CASE-0008" + ], + "name": "Anechoic acoustic spectral measurement" + }, + { + "description": "Role-scoped operations submitted with identity fixtures and in-memory policy store; compare allow/deny and audit fields to permissions matrix.", + "members": [ + "CASE-0006", + "CASE-0012" + ], + "name": "Authorization decision matrix tests" + }, + { + "description": "Concurrent producer/consumer task drivers against a bounded queue; observe depth, rejected writes, ordering, drain recovery; nominal vs boundary.", + "members": [ + "CASE-0005", + "CASE-0011" + ], + "name": "Bounded queue producer-consumer tests" + }, + { + "description": "Call configuration parser with text fixtures; assert accepted values or diagnostic positions; nominal and boundary inputs.", + "members": [ + "CASE-0004", + "CASE-0010" + ], + "name": "Configuration parser input tests" + }, + { + "description": "Single case: rebuild same source twice in clean containers and compare artifact digests excluding allowed timestamps.", + "members": [ + "CASE-0014" + ], + "name": "Reproducible build digest comparison" + }, + { + "description": "Offline static-analysis report parser with severity policy fixture; count findings by severity and check rule IDs; nominal vs boundary inputs.", + "members": [ + "CASE-0003", + "CASE-0009" + ], + "name": "Static-analysis report parsing" + }, + { + "description": "Single case: byte-array builder feeding the decoder; compare fields, checksum status, rejection code.", + "members": [ + "CASE-0007" + ], + "name": "Status frame decoder tests" + }, + { + "description": "Single case: thermal chamber cycling with calibrated gauge measurement of enclosure dimensional expansion against tolerance.", + "members": [ + "CASE-0001" + ], + "name": "Thermal chamber expansion gauge measurement" + }, + { + "description": "Controllable pulse source stopped/resumed with digital capture of reset-line timing and deadline check; CASE-0002 and CASE-0013 are identical in content.", + "members": [ + "CASE-0002", + "CASE-0013" + ], + "name": "Watchdog pulse reset-timing capture" + } + ] + }, + "selection": { + "backend": "claude-code-2.1.285", + "case_count": 14, + "cli_model": "claude-sonnet-5-5[1m]", + "configuration_revision": "suite-v6-20260929", + "context": 1000000, + "enabled": true, + "execution_revision": "suite-multipass-v5-20260929", + "input_bytes": 16056, + "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer", + "input_token_bound": 24248, + "input_token_count": null, + "later_pass_capacity_verified": false, + "later_pass_checks": "before_each_invocation", + "max_final_group_cases": 5, + "max_final_name_characters": 64, + "max_turns": 6, + "maximum_model_passes": 5, + "minimum_model_passes": 3, + "model": "claude-sonnet-5-5", + "output": 64000, + "output_reservation_tokens": 11008, + "output_reservation_verified": false, + "overhead": 8192, + "policy_revision": "implementation-five-v1-20260929", + "prompt_revision": "implementation-proximity-multipass-v4-20260929", + "prompt_sha256": "253314b009ea82fc44695260ece8edd501ab60dc40ab3cc9e1ac9a0acb2017c0", + "provider": "claude", + "reasoning": "medium", + "reported_output": 128000, + "source_sha256": "99fd35cb089e57285670f6def4339a86754316e67ce3e53ec7efa95968534ca9" + }, + "singletons": 4, + "status": "completed", + "test_path": "server_side_multipass_backend", + "wall_seconds": 24.799 + } + ], + "review": { + "membership": "Both match all nine expected mechanisms; four valid singletons; duplicate-content aliases preserved.", + "descriptions": "No cross-family objectives observed. Opus more explicitly retains unknown procedure details for thermal, acoustic, and build fixtures. Sonnet is more concise; its result descriptions omit those uncertainty notes.", + "limitations": [ + "One run per model, synthetic 14-case suite only.", + "This fixture requires three passes; no family exceeds five, so the two oversized-family review passes are not exercised by these hosted runs.", + "First tiny Opus readiness probe failed with zero usage and no runtime model metadata; a separate fresh probe succeeded. The failed probe did not retain an actionable provider cause.", + "No automatic retry occurred in either measured grouping run.", + "The CLI-reported thinking count is recorded as reported; it is not a measure of reasoning quality." + ] + } +} diff --git a/docs/hermes_suite_claude55.md b/docs/hermes_suite_claude55.md new file mode 100644 index 00000000..44b5c8d1 --- /dev/null +++ b/docs/hermes_suite_claude55.md @@ -0,0 +1,110 @@ +# Claude 5.5 suite comparison + +The suite planner now pins native Claude Code 2.1.285. Opus 5.5 and Sonnet 5.5 +both returned their exact canonical model identities on the existing first-party +OAuth account. The default deployment selects `claude-opus-5-5`; Sonnet is a +server configuration option, not an automatic fallback. The HTTPS contract, +credential scopes, implementation objective, medium effort, five-case task cap, +30-minute deadline, and USD 30 CLI estimated-cost guard are unchanged. + +## Matched 14-case results + +Each run used the same 9,851-byte normalized synthetic request, complete source +fields, schema, and three-pass workflow (two independent proposals and one +whole-suite reconciliation). Each completed in six CLI turns without a retry. +No family exceeded five, so these hosted tests did not need the two extra +oversized-family review passes. + +| Measurement | Opus 5.5 | Sonnet 5.5 | +| --- | ---: | ---: | +| Wall seconds | 39.256 | 24.799 | +| Aggregate input tokens | 21,936 | 21,438 | +| Aggregate output tokens | 4,243 | 3,814 | +| CLI estimated USD | 0.172604 | 0.081016 | +| Natural families / final tasks | 9 / 9 | 9 / 9 | +| Singleton families | 4 | 4 | +| Exact alias coverage | Yes | Yes | +| Incorrect merge / missed merge pairs | 0 / 0 | 0 / 0 | + +Both produced identical memberships matching the fixture's independent mechanism +labels, including distant related cases, identical text under different aliases, +and watchdog terminology shared by different test machinery. Review found no +description importing another family's objectives. Opus explicitly retained +unknown procedure details for the thermal, acoustic, and build singleton fixtures. +Sonnet's descriptions were more concise and omitted those uncertainty notes. + +Membership quality tied on this small fixture. Sonnet used about 53% less CLI +estimated cost and finished about 37% sooner. This is one observation per model, +not evidence that their performance or quality is equivalent on a difficult +363-case suite. The amounts are CLI accounting estimates, not subscription +charges or proof of a dollar-denominated account bill. + +## Runtime limits and isolation + +Both CLI result envelopes report a 1,000,000-token context and 128,000-token model +output ceiling. The service deliberately continues to request at most 64,000 +output tokens per invocation. `selection.output` is that service limit; +`selection.reported_output` and `model_usage[*].maxOutputTokens` describe the +reported model ceiling. The native CLI loopback check captured `max_tokens=64000` +and complete source/system text on all four requests, including one deliberately +invalid structured assignment followed by repair. + +The newer CLI loads a built-in instruction plugin even in safe mode unless it is +explicitly disabled. The service now disables it, all hooks, bundled skills, +provider connectors, plugin synchronization, built-in subagents, and git +instructions. Each CLI invocation allowlists only its pinned model and turns off +automatic model switching when a request is flagged. Runtime identity, empty +plugins/MCP lists, and tool isolation are still checked before accepting output. +No provider or model fallback was added. The first tiny Opus readiness probe +failed with zero usage and insufficient retained diagnostics; a fresh probe +succeeded. Neither measured suite job needed a retry. + +The artifact is downloaded from the official release host at startup with a +pinned version, 240,327,864-byte size, and SHA-256: + +```text +33dad1ec615a2e08cc78b494f05c110e49916de2c79d78ec8799ebf46b233d29 +``` + +This removes the planner's dependency on the shared agent tools volume. Other +Hermes CLIs and the isolated local-model endpoint are unchanged. A cold start now +requires the official release download host to be reachable; failed download or +checksum verification leaves the worker unavailable instead of using another CLI. + +## Configuration and rollback + +The HTTP compatibility revision remains `suite-v6-20260929`, the prompt remains +`implementation-proximity-multipass-v4-20260929`, and the execution revision is +`suite-multipass-v5-20260929`. Authenticated capabilities additionally expose +`claude_model_options` and `model_selection: server_configuration`. No new client +request field is accepted or required. + +To select Sonnet, change the existing Flux manifest environment entry and its +rollout annotation, commit, push, and reconcile Hermes while no job is active: + +```yaml +- name: PLANNING_CLAUDE_MODEL + value: claude-sonnet-5-5 +``` + +The allowlisted IDs are `claude-opus-4-8`, `claude-opus-5-5`, and +`claude-sonnet-5-5`. Reverting the upgrade commit restores the former CLI staging, +model pin, and execution revision. Reconcile with: + +```bash +flux reconcile kustomization hermes --namespace flux-system --with-source +``` + +Do not restart during a job: provider work is never automatically retried, and +in-memory results expire on worker restart. The server comparison runs used the +actual multi-pass backend inside isolated temporary directories, not laptop +connectivity or HTTP submission. Results and concise review are retained in +[synthetic evidence](evidence/hermes_suite_claude55_20260929.json). The operator +probe accepts no input file and constructs only the fixed synthetic 14-case +fixture. Focused mocked regression tests passed: 122. Kustomize render and client +dry-run passed; Flux diff was limited to the planner Deployment and ConfigMap. +The documented combined server-side/client dry-run flags are incompatible with +the installed kubectl, so the client dry-run was run without `--server-side`. + +Official references: [model configuration](https://code.claude.com/docs/en/model-config) +and [CLI settings](https://code.claude.com/docs/en/settings-reference). diff --git a/docs/hermes_suite_multipass.md b/docs/hermes_suite_multipass.md index eef5b4f4..a6df3a05 100644 --- a/docs/hermes_suite_multipass.md +++ b/docs/hermes_suite_multipass.md @@ -11,7 +11,7 @@ most **64 characters**, unique after whitespace and case normalization. - Configuration: `suite-v6-20260929` (HTTP compatibility identifier). - Policy: `implementation-five-v1-20260929`. - Prompt: `implementation-proximity-multipass-v4-20260929`. -- Execution: `suite-multipass-v4-20260929`. +- Execution: `suite-multipass-v5-20260929`. A single server-side job performs: @@ -45,13 +45,19 @@ missing review decisions still fail closed; no missing assignment is fabricated. Each invocation retains six CLI turns for structured output. Model review passes and CLI turns are separate counters. All calls use the originally selected provider -and pinned model. The model remains `claude-opus-4-8[1m]`, with canonical runtime -identity checked as `claude-opus-4-8`, firstParty, medium effort, reported 1M context -and 64K output ceiling. The server invokes native Claude Code CLI 2.1.226 using +and pinned model. The current default is `claude-opus-5-5[1m]`, with canonical runtime +identity checked as `claude-opus-5-5`, firstParty, medium effort, reported 1M context +and 128K model output ceiling, with requests limited to 64K output tokens. +The server invokes native Claude Code CLI 2.1.285 using the existing first-party OAuth account, not a separately configured API-key account. The CLI itself communicates with Anthropic over HTTPS. No new provider fallback or tools are enabled. +Sonnet 5.5 is also verified on the 14-case synthetic fixture and available through +server configuration. See the [matched comparison](hermes_suite_claude55.md). +Earlier acceptance measurements below used Opus 4.8; they do not establish large +suite capacity or quality for the new models. + ## Deterministic task sizing and names For each remaining coherent large family, `k = ceil(n / 5)` and diff --git a/scripts/ops/hermes_suite_model_compare.py b/scripts/ops/hermes_suite_model_compare.py new file mode 100644 index 00000000..79ed81ad --- /dev/null +++ b/scripts/ops/hermes_suite_model_compare.py @@ -0,0 +1,65 @@ +#!/usr/bin/env python3 +"""Compare one configured Claude model on the public synthetic 14-case suite. + +Run server-side with the planner modules and scoped OAuth mount. This invokes +the same multi-pass backend without creating an HTTP job. Output is synthetic +evidence only; progress contains metadata. Never accepts a roster input file. +""" +import json +import os +import sys +import threading +import time + +sys.path.insert(0, os.environ.get("SUITE_PROBE_MODULE_DIR", "/opt/planner")) +import suite_backends +from suite_contract import (EXECUTION_REVISION, MODELS, PROMPT_REVISION, Problem, + encoded, validate_request, validate_result) +from suite_multipass import generate, preflight_workflow +from suite_synthetic import fixture, score + + +def main(): + """Run one fixed fixture and emit normalized results plus measured metadata.""" + model = MODELS["claude"]["model"] + if model not in {"claude-opus-5-5", "claude-sonnet-5-5"}: + raise SystemExit("Configure one of the two approved comparison models") + if os.environ.get("SUITE_PROBE_CLAUDE_BIN"): + suite_backends.CLAUDE_BIN = os.environ["SUITE_PROBE_CLAUDE_BIN"] + request, expected = fixture(14) + request["routing"] = {"allow_external": True, "allowed_external_providers": ["claude"]} + request = validate_request(request, ["claude"]) + selected = preflight_workflow(request) + started, last_report = time.monotonic(), 0.0 + last_stage = None + + def progress(value): + """Emit stage and elapsed time without model-generated content.""" + nonlocal last_report, last_stage + stage = value["current_pass"] + if stage != last_stage or time.monotonic() - last_report >= 20: + print(json.dumps({"model": model, "stage": stage, + "elapsed_seconds": value["job_elapsed_seconds"], + "completed_passes": value["completed_model_passes"]}), + file=sys.stderr, flush=True) + last_report, last_stage = time.monotonic(), stage + + report = {"model": model, "execution_revision": EXECUTION_REVISION, + "prompt_revision": PROMPT_REVISION, "selection": selected, + "request_bytes": len(encoded(request)), "case_count": 14, + "automatic_retries": 0, "test_path": "server_side_multipass_backend"} + try: + result, metadata = generate(request, selected, threading.Event(), "192.168.22.8", progress) + validate_result(result, request) + report.update(status="completed", result=result, metadata=metadata, + quality=score(result, expected), + singletons=sum(len(g["members"]) == 1 for g in result["groups"])) + except Problem as exc: + report.update(status="failed", **exc.document()) + report["wall_seconds"] = round(time.monotonic() - started, 3) + print(json.dumps(report, sort_keys=True), flush=True) + return 0 if report["status"] == "completed" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/services/hermes/scripts/suite_api.py b/services/hermes/scripts/suite_api.py index 3239bf1c..11fe4f0c 100644 --- a/services/hermes/scripts/suite_api.py +++ b/services/hermes/scripts/suite_api.py @@ -12,7 +12,7 @@ import re import subprocess import threading -from suite_contract import (COST_LIMIT, EXECUTION_REVISION, MAX_BODY, MAX_CASES, MAX_RESULT, MODELS, PROMPT_REVISION, +from suite_contract import (CLAUDE_MODELS, CLAUDE_VERSION, COST_LIMIT, EXECUTION_REVISION, MAX_BODY, MAX_CASES, MAX_RESULT, MODELS, PROMPT_REVISION, PROMPT_SHA256, REVISION, TIMEOUT, Problem, encoded, preflight, validate_request) from suite_jobs import Jobs @@ -124,6 +124,7 @@ class Handler(BaseHTTPRequestHandler): "prompt_revision": PROMPT_REVISION, "prompt_sha256": PROMPT_SHA256}) if method == "GET" and self.path == "/v1/capabilities": return self.send(200, {"configuration_revision": REVISION, "models": MODELS, + "claude_model_options": list(CLAUDE_MODELS), "model_selection": "server_configuration", "execution_revision": EXECUTION_REVISION, "policy_revision": POLICY_REVISION, "max_final_group_cases": 5, "max_final_name_characters": 64, "model_passes": {"minimum": 3, "maximum": 5}, @@ -214,7 +215,7 @@ def main(): raise SystemExit("Pinned CLI binary mismatch") version = subprocess.run(["/opt/cli/claude", "--version"], capture_output=True, text=True, timeout=10, check=True) - if version.stdout.strip() != "2.1.226 (Claude Code)": + if version.stdout.strip() != CLAUDE_VERSION + " (Claude Code)": raise SystemExit("Pinned CLI version mismatch") private = ThreadingHTTPServer(("0.0.0.0", 9001), Decision) threading.Thread(target=private.serve_forever, daemon=True).start() diff --git a/services/hermes/scripts/suite_backends.py b/services/hermes/scripts/suite_backends.py index ff5f4d5d..d7806f6a 100644 --- a/services/hermes/scripts/suite_backends.py +++ b/services/hermes/scripts/suite_backends.py @@ -11,7 +11,7 @@ import time from urllib.error import HTTPError, URLError from urllib.request import HTTPRedirectHandler, ProxyHandler, Request, build_opener -from suite_contract import CLAUDE_MAX_TURNS, MODELS, SCHEMA, SYSTEM, Problem, encoded, prompt +from suite_contract import CLAUDE_MAX_TURNS, CLAUDE_MODELS, MODELS, SCHEMA, SYSTEM, Problem, encoded, prompt from suite_cli_diagnostics import snapshot SWITCHYARD = "http://hermes-switchyard.hermes.svc.cluster.local:9005/v1/chat/completions" @@ -97,12 +97,19 @@ def local_generate(request, cancel, client_ip, *, invocation=None): def claude_command(model, max_cost): """Use the native pinned binary, never the privileged Hermes shell wrapper.""" + if model not in CLAUDE_MODELS: + raise Problem("unsupported_backend", 422) + settings = {"enabledPlugins": {"agents-md@builtin": False, + "cc-plugin-agents-md@builtin": False}, + "disableAllHooks": True, "disableBundledSkills": True, + "disableClaudeAiConnectors": True, "syncClaudeAiPlugins": False, + "availableModels": [model], "switchModelsOnFlag": False} return [CLAUDE_BIN, "-p", "--output-format", "stream-json", "--verbose", "--no-session-persistence", "--safe-mode", "--tools", "", "--strict-mcp-config", "--mcp-config", '{"mcpServers":{}}', - "--setting-sources", "", "--disable-slash-commands", + "--setting-sources", "", "--settings", encoded(settings).decode(), "--disable-slash-commands", "--permission-mode", "dontAsk", "--no-chrome", - "--model", MODELS["claude"]["cli_model"], "--effort", "medium", "--max-budget-usd", str(max_cost), + "--model", model + "[1m]", "--effort", "medium", "--max-budget-usd", str(max_cost), "--max-turns", str(CLAUDE_MAX_TURNS), "--system-prompt", SYSTEM, "--json-schema", encoded(SCHEMA).decode()] @@ -117,7 +124,9 @@ def claude_environment(directory, token): "DISABLE_ERROR_REPORTING", "DISABLE_AUTOUPDATER", "DISABLE_UPDATES", "DISABLE_PROMPT_CACHING", "CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC", "CLAUDE_CODE_DISABLE_BACKGROUND_TASKS", "CLAUDE_CODE_DISABLE_TERMINAL_TITLE", - "CLAUDE_CODE_DISABLE_AUTO_MEMORY", "CLAUDE_CODE_SKIP_PROMPT_HISTORY"): + "CLAUDE_CODE_DISABLE_AUTO_MEMORY", "CLAUDE_CODE_SKIP_PROMPT_HISTORY", + "CLAUDE_AGENT_SDK_DISABLE_BUILTIN_AGENTS", "CLAUDE_CODE_DISABLE_GIT_INSTRUCTIONS", + "CLAUDE_CODE_DISABLE_BUNDLED_SKILLS"): env[key] = "1" return env @@ -180,7 +189,8 @@ def parse_claude(raw, expected_model, **process_info): if not isinstance(models, dict) or not models or set(models) - aliases: fail("model_changed", "runtime_model") for limits in models.values(): - if (not isinstance(limits, dict) or limits.get("contextWindow") != 1000000 or limits.get("maxOutputTokens") != 64000 + if (not isinstance(limits, dict) or limits.get("contextWindow") != 1000000 + or limits.get("maxOutputTokens") != CLAUDE_MODELS.get(expected_model) or limits.get("canonicalModel") != expected_model or limits.get("provider") != "firstParty"): fail("backend_capabilities_changed", "runtime_model_limits") @@ -196,7 +206,7 @@ def parse_claude(raw, expected_model, **process_info): fail("incomplete_generation", "process_exit") # Persist measured metadata only, including on later schema/coverage failure. safe_models = {name: {"canonicalModel": expected_model, "provider": "firstParty", - "contextWindow": 1000000, "maxOutputTokens": 64000} + "contextWindow": 1000000, "maxOutputTokens": CLAUDE_MODELS[expected_model]} for name in models} return result, {"model": expected_model, "compaction": False, "truncation": False, "compaction_signal": "CLI events and disabled compaction", diff --git a/services/hermes/scripts/suite_cli_diagnostics.py b/services/hermes/scripts/suite_cli_diagnostics.py index bfd55c81..157c9a6a 100644 --- a/services/hermes/scripts/suite_cli_diagnostics.py +++ b/services/hermes/scripts/suite_cli_diagnostics.py @@ -30,6 +30,7 @@ def usage_counts(value): "input_tokens", "output_tokens", "cache_read_input_tokens", "cache_creation_input_tokens") if key in value} for key, fields in (("server_tool_use", ("web_search_requests", "web_fetch_requests")), + ("output_tokens_details", ("thinking_tokens",)), ("cache_creation", ("ephemeral_1h_input_tokens", "ephemeral_5m_input_tokens"))): if isinstance(value.get(key), dict): result[key] = {field: number(value[key][field]) for field in fields if field in value[key]} diff --git a/services/hermes/scripts/suite_contract.py b/services/hermes/scripts/suite_contract.py index 404e3446..8eccc997 100644 --- a/services/hermes/scripts/suite_contract.py +++ b/services/hermes/scripts/suite_contract.py @@ -3,12 +3,22 @@ from __future__ import annotations import hashlib import json +import os import re from collections import Counter REVISION = "suite-v6-20260929" PROMPT_REVISION = "implementation-proximity-multipass-v4-20260929" -EXECUTION_REVISION = "suite-multipass-v4-20260929" +EXECUTION_REVISION = "suite-multipass-v5-20260929" +CLAUDE_VERSION = "2.1.285" +CLAUDE_MODELS = { + "claude-opus-4-8": 64000, + "claude-opus-5-5": 128000, + "claude-sonnet-5-5": 128000, +} +CLAUDE_MODEL = os.environ.get("PLANNING_CLAUDE_MODEL", "claude-opus-4-8") +if CLAUDE_MODEL not in CLAUDE_MODELS: + raise RuntimeError("Unsupported configured Claude model") CLAUDE_MAX_TURNS = 6 MAX_BODY = 1 << 20 MAX_RESULT = 1 << 20 @@ -23,9 +33,10 @@ MODELS = { "local": {"model": "qwen2.5:14b-instruct-q4_0", "context": 8192, "output": 2048, "overhead": 1024, "backend": "ollama-model-gate", "enabled": True, "reasoning": "none"}, - "claude": {"model": "claude-opus-4-8", "context": 1000000, - "cli_model": "claude-opus-4-8[1m]", - "output": 64000, "overhead": 8192, "backend": "claude-code-2.1.226", + "claude": {"model": CLAUDE_MODEL, "context": 1000000, + "cli_model": CLAUDE_MODEL + "[1m]", + "output": 64000, "reported_output": CLAUDE_MODELS[CLAUDE_MODEL], + "overhead": 8192, "backend": "claude-code-" + CLAUDE_VERSION, "enabled": True, "reasoning": "medium", "max_turns": CLAUDE_MAX_TURNS}, "codex": {"model": "gpt-6-astra", "context": 258400, "output": None, "overhead": None, "backend": "codex-subscription-broker", diff --git a/services/hermes/suite-planner-deployment.yaml b/services/hermes/suite-planner-deployment.yaml index 66177a1b..ec0b56d8 100644 --- a/services/hermes/suite-planner-deployment.yaml +++ b/services/hermes/suite-planner-deployment.yaml @@ -36,7 +36,7 @@ spec: app: hermes-suite-planner annotations: fluentbit.io/exclude: "true" - ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v4-20260929 + ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v5-20260929 vault.hashicorp.com/agent-inject: "true" vault.hashicorp.com/agent-pre-populate-only: "true" vault.hashicorp.com/agent-init-first: "true" @@ -80,7 +80,7 @@ spec: automountServiceAccountToken: false enableServiceLinks: false terminationGracePeriodSeconds: 15 - # The installed amd64 CLI is copied from the existing RWO tools volume. + # The pinned native CLI is amd64; retain the existing worker placement. nodeSelector: kubernetes.io/hostname: titan-22 securityContext: @@ -97,18 +97,27 @@ spec: command: [python, -c] args: - | - import hashlib,pathlib,shutil - source=pathlib.Path('/installed/lib/node_modules/@anthropic-ai/claude-code/bin/claude.exe') - assert hashlib.sha256(source.read_bytes()).hexdigest() == '4e9bec1177ce9690e8bd988b710ac24105e70da428dd094c5adcbbe786a55555' - shutil.copyfile(source, '/opt/cli/claude') - pathlib.Path('/opt/cli/claude').chmod(0o555) + import hashlib,pathlib,urllib.request + target=pathlib.Path('/opt/cli/claude') + url='https://downloads.claude.ai/claude-code-releases/2.1.285/linux-x64/claude' + digest=hashlib.sha256() + size=0 + with urllib.request.urlopen(url, timeout=120) as source, target.open('wb') as output: + while data:=source.read(1048576): + size+=len(data) + if size > 240327864: + raise RuntimeError('CLI artifact exceeds pinned size') + digest.update(data) + output.write(data) + if size != 240327864 or digest.hexdigest() != '33dad1ec615a2e08cc78b494f05c110e49916de2c79d78ec8799ebf46b233d29': + raise RuntimeError('CLI artifact verification failed') + target.chmod(0o555) securityContext: allowPrivilegeEscalation: false readOnlyRootFilesystem: true capabilities: drop: [ALL] volumeMounts: - - {name: installed, mountPath: /installed, subPath: tools, readOnly: true} - {name: cli, mountPath: /opt/cli} resources: requests: {cpu: 25m, memory: 256Mi} @@ -122,7 +131,8 @@ spec: - {name: PYTHONUNBUFFERED, value: "1"} # User approved generalized CASE records on this Claude account, 2026-09-29. - {name: PLANNING_GENERALIZED_CLAUDE_APPROVED, value: "true"} - - {name: PLANNING_CLAUDE_SHA256, value: 4e9bec1177ce9690e8bd988b710ac24105e70da428dd094c5adcbbe786a55555} + - {name: PLANNING_CLAUDE_SHA256, value: 33dad1ec615a2e08cc78b494f05c110e49916de2c79d78ec8799ebf46b233d29} + - {name: PLANNING_CLAUDE_MODEL, value: claude-opus-5-5} ports: - {name: http, containerPort: 9000} - {name: decision, containerPort: 9001} @@ -148,10 +158,6 @@ spec: - name: scripts configMap: name: hermes-suite-planner - - name: installed - persistentVolumeClaim: - claimName: hermes-agent-home - readOnly: true - name: cli emptyDir: {medium: Memory, sizeLimit: 512Mi} - name: jobs diff --git a/testing/tests/test_suite_claude_models.py b/testing/tests/test_suite_claude_models.py new file mode 100644 index 00000000..859d8d13 --- /dev/null +++ b/testing/tests/test_suite_claude_models.py @@ -0,0 +1,74 @@ +"""Model upgrades preserve exact identity, capacity, and CLI isolation.""" +import json +from pathlib import Path +import subprocess +import sys + +import pytest + +SCRIPTS = Path(__file__).resolve().parents[2] / "services/hermes/scripts" +sys.path.insert(0, str(SCRIPTS)) +import suite_backends +from suite_cli_diagnostics import usage_counts +from suite_contract import CLAUDE_MODELS, Problem + + +@pytest.mark.parametrize("model", CLAUDE_MODELS) +def test_only_configured_model_is_allowed_and_plugins_are_disabled(model): + """Each invocation pins one model, with fallback and customizations disabled.""" + command = suite_backends.claude_command(model, 30) + settings = json.loads(command[command.index("--settings") + 1]) + assert command[command.index("--model") + 1] == model + "[1m]" + assert settings["availableModels"] == [model] + assert settings["switchModelsOnFlag"] is False + assert settings["disableAllHooks"] and settings["disableClaudeAiConnectors"] + assert settings["syncClaudeAiPlugins"] is False + assert settings["enabledPlugins"]["cc-plugin-agents-md@builtin"] is False + assert "--fallback-model" not in command + env = suite_backends.claude_environment("/jobs/fresh", "synthetic") + assert env["CLAUDE_CODE_MAX_OUTPUT_TOKENS"] == "64000" + assert env["CLAUDE_AGENT_SDK_DISABLE_BUILTIN_AGENTS"] == "1" + + +@pytest.mark.parametrize("model", ["claude-opus-5-5", "claude-sonnet-5-5"]) +@pytest.mark.parametrize("failure", [None, "substitution", "capacity", "plugin"]) +def test_new_runtime_envelope_remains_fail_closed(model, failure): + """A requested name never substitutes for the actual runtime model identity.""" + init = {"type": "system", "subtype": "init", "model": model + "[1m]", + "tools": ["StructuredOutput"], "mcp_servers": [], "plugins": []} + limits = {"contextWindow": 1000000, "maxOutputTokens": 128000, + "canonicalModel": model, "provider": "firstParty"} + result = {"type": "result", "subtype": "success", "is_error": False, + "modelUsage": {model + "[1m]": limits}, "usage": {}, + "structured_output": {"groups": []}} + if failure == "substitution": + limits["canonicalModel"] = "claude-opus-4-8" + elif failure == "capacity": + limits["contextWindow"] = 200000 + elif failure == "plugin": + init["plugins"] = [{"name": "unapproved"}] + raw = "\n".join(map(json.dumps, (init, result))) + if failure: + with pytest.raises(Problem): + suite_backends.parse_claude(raw, model, exit_code=0) + else: + _, metadata = suite_backends.parse_claude(raw, model, exit_code=0) + assert metadata["model"] == model + assert metadata["model_usage"][model + "[1m]"]["maxOutputTokens"] == 128000 + + +def test_unknown_model_cannot_start_a_cli_process(): + """Operator configuration and command construction both reject unknown IDs.""" + with pytest.raises(Problem): + suite_backends.claude_command("unapproved-model", 30) + result = subprocess.run([sys.executable, "-c", "import suite_contract"], + cwd=SCRIPTS, env={"PLANNING_CLAUDE_MODEL": "unapproved-model"}, + capture_output=True, text=True) + assert result.returncode != 0 + assert "Unsupported configured Claude model" in result.stderr + + +def test_thinking_usage_is_counted_without_retaining_unrecognized_fields(): + """New CLI usage details retain measurements and drop arbitrary content.""" + assert usage_counts({"output_tokens_details": {"thinking_tokens": 123, "content": "private"}}) == { + "output_tokens_details": {"thinking_tokens": 123}}