hermes: recover suite passes and adapt large-suite analysis
This commit is contained in:
parent
cef01b0d72
commit
9255a64b23
@ -206,15 +206,17 @@
|
||||
"max_seconds": {
|
||||
"type": "integer",
|
||||
"minimum": 10,
|
||||
"maximum": 3600,
|
||||
"maximum": 7200,
|
||||
"default": 1800
|
||||
},
|
||||
"max_cost_usd": {
|
||||
"type": "number",
|
||||
"type": [
|
||||
"number",
|
||||
"null"
|
||||
],
|
||||
"exclusiveMinimum": 0,
|
||||
"maximum": 30,
|
||||
"default": 30,
|
||||
"description": "CLI estimated-cost guard, not subscription billing. Shared across all passes."
|
||||
"deprecated": true,
|
||||
"description": "Accepted for compatibility only. Estimated cost is informational and never enforced."
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
File diff suppressed because one or more lines are too long
524
docs/evidence/hermes_suite_recovery_20260930.json
Normal file
524
docs/evidence/hermes_suite_recovery_20260930.json
Normal file
@ -0,0 +1,524 @@
|
||||
{
|
||||
"hosted_synthetic_14": [
|
||||
{
|
||||
"cost_usd_estimate": 0.42616,
|
||||
"final_groups": [
|
||||
{
|
||||
"description": "Work-size part 1/2 of one family. Byte-array frame builder, decoder-call fixture, capture of decoded object, comparison of fields, checksum status and rejection code; no hardware",
|
||||
"members": [
|
||||
"CASE-REC-02",
|
||||
"CASE-REC-04",
|
||||
"CASE-REC-08",
|
||||
"CASE-REC-10"
|
||||
],
|
||||
"name": "Status frame decoder harness (1/2)"
|
||||
},
|
||||
{
|
||||
"description": "Work-size part 2/2 of one family. Byte-array frame builder, decoder-call fixture, capture of decoded object, comparison of fields, checksum status and rejection code; no hardware",
|
||||
"members": [
|
||||
"CASE-REC-00",
|
||||
"CASE-REC-06",
|
||||
"CASE-REC-12"
|
||||
],
|
||||
"name": "Status frame decoder harness (2/2)"
|
||||
},
|
||||
{
|
||||
"description": "Work-size part 1/2 of one family. Controllable pulse source driver, reset-line recorder/digital capture fixture, reset timing extraction, deadline check against supplied tolerance",
|
||||
"members": [
|
||||
"CASE-REC-01",
|
||||
"CASE-REC-05",
|
||||
"CASE-REC-07",
|
||||
"CASE-REC-11"
|
||||
],
|
||||
"name": "Watchdog pulse reset timing rig (1/2)"
|
||||
},
|
||||
{
|
||||
"description": "Work-size part 2/2 of one family. Controllable pulse source driver, reset-line recorder/digital capture fixture, reset timing extraction, deadline check against supplied tolerance",
|
||||
"members": [
|
||||
"CASE-REC-03",
|
||||
"CASE-REC-09",
|
||||
"CASE-REC-13"
|
||||
],
|
||||
"name": "Watchdog pulse reset timing rig (2/2)"
|
||||
}
|
||||
],
|
||||
"mode": "direct",
|
||||
"natural_groups": [
|
||||
{
|
||||
"common_work": "Byte-array frame builder, decoder-call fixture, capture of decoded object, comparison of fields, checksum status and rejection code; no hardware",
|
||||
"description": "Offline tests: a byte-array builder feeds encoded watchdog status frames to a decoder-call fixture. Checks decoded fields, checksum status and rejection code. Nominal and fault cases differ only in bytes and expected results.",
|
||||
"evidence": [
|
||||
{
|
||||
"alias": "CASE-REC-00",
|
||||
"field": "success_criteria",
|
||||
"quote": "Capture the decoded object and compare fields, checksum status, and rejection code"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-REC-02",
|
||||
"field": "preconditions",
|
||||
"quote": "no live hardware is required"
|
||||
}
|
||||
],
|
||||
"members": [
|
||||
"CASE-REC-00",
|
||||
"CASE-REC-02",
|
||||
"CASE-REC-04",
|
||||
"CASE-REC-06",
|
||||
"CASE-REC-08",
|
||||
"CASE-REC-10",
|
||||
"CASE-REC-12"
|
||||
],
|
||||
"name": "Status frame decoder harness",
|
||||
"rationale": "All members have the same preconditions, the same stimulus path and the same success criteria. Fault-injection cases change only the frame bytes and the expected checksum or rejection outcome. They reuse the fixture and the assertion structure.",
|
||||
"uncertainty": "Frame contents, the faults to inject and the expected rejection codes are not stated. Complex corruption patterns may need extra builder helpers.",
|
||||
"variation_sets": [
|
||||
[
|
||||
"CASE-REC-02",
|
||||
"CASE-REC-04",
|
||||
"CASE-REC-08",
|
||||
"CASE-REC-10"
|
||||
],
|
||||
[
|
||||
"CASE-REC-00",
|
||||
"CASE-REC-06",
|
||||
"CASE-REC-12"
|
||||
]
|
||||
]
|
||||
},
|
||||
{
|
||||
"common_work": "Controllable pulse source driver, reset-line recorder/digital capture fixture, reset timing extraction, deadline check against supplied tolerance",
|
||||
"description": "Hardware tests: a controllable pulse source stops and resumes watchdog pulses at the controller input. A digital capture fixture records reset-line timing against the supplied deadline. Nominal and fault cases differ in stimulus only.",
|
||||
"evidence": [
|
||||
{
|
||||
"alias": "CASE-REC-01",
|
||||
"field": "success_criteria",
|
||||
"quote": "Measure reset-line timing with a digital capture fixture and check the deadline"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-REC-03",
|
||||
"field": "preconditions",
|
||||
"quote": "Use a controllable pulse source and reset-line recorder"
|
||||
}
|
||||
],
|
||||
"members": [
|
||||
"CASE-REC-01",
|
||||
"CASE-REC-03",
|
||||
"CASE-REC-05",
|
||||
"CASE-REC-07",
|
||||
"CASE-REC-09",
|
||||
"CASE-REC-11",
|
||||
"CASE-REC-13"
|
||||
],
|
||||
"name": "Watchdog pulse reset timing rig",
|
||||
"rationale": "All members use the same physical rig, the same stop/resume pulse sequence and the same timing measurement and deadline assertion. The fault-injection label adds no new equipment or new way of observing. This is separate from the offline decoder work.",
|
||||
"uncertainty": "Deadline and tolerance values are left out. The records don't say how fault cases change the pulse stimulus, for example stop duration or irregular pulses. Such changes might need extra pulse-source control.",
|
||||
"variation_sets": [
|
||||
[
|
||||
"CASE-REC-01",
|
||||
"CASE-REC-05",
|
||||
"CASE-REC-07",
|
||||
"CASE-REC-11",
|
||||
"CASE-REC-13"
|
||||
],
|
||||
[
|
||||
"CASE-REC-03",
|
||||
"CASE-REC-09"
|
||||
]
|
||||
]
|
||||
}
|
||||
],
|
||||
"passes": [
|
||||
{
|
||||
"cli_turns": 2,
|
||||
"input_bytes": 11122,
|
||||
"max_turns": 4,
|
||||
"stage": "proposal_a",
|
||||
"wall_seconds": 5.724
|
||||
},
|
||||
{
|
||||
"cli_turns": 3,
|
||||
"input_bytes": 11122,
|
||||
"max_turns": 4,
|
||||
"stage": "proposal_b",
|
||||
"wall_seconds": 9.437
|
||||
},
|
||||
{
|
||||
"cli_turns": 3,
|
||||
"input_bytes": 16696,
|
||||
"max_turns": 4,
|
||||
"stage": "reconciliation",
|
||||
"wall_seconds": 23.197
|
||||
},
|
||||
{
|
||||
"cli_turns": 3,
|
||||
"input_bytes": 20816,
|
||||
"max_turns": 4,
|
||||
"stage": "large_family_review",
|
||||
"wall_seconds": 26.503
|
||||
},
|
||||
{
|
||||
"cli_turns": 2,
|
||||
"input_bytes": 25569,
|
||||
"max_turns": 4,
|
||||
"stage": "decision_audit",
|
||||
"wall_seconds": 14.143
|
||||
}
|
||||
],
|
||||
"quality": {
|
||||
"coverage": true,
|
||||
"false_merge_pairs": 0,
|
||||
"families": 2,
|
||||
"missed_merge_pairs": 0,
|
||||
"pair_precision": 1.0,
|
||||
"pair_recall": 1.0
|
||||
},
|
||||
"seconds": 79.00923439487815,
|
||||
"status": "completed",
|
||||
"usage": {
|
||||
"cache_creation": {
|
||||
"ephemeral_1h_input_tokens": 0,
|
||||
"ephemeral_5m_input_tokens": 0
|
||||
},
|
||||
"cache_creation_input_tokens": 0,
|
||||
"cache_read_input_tokens": 0,
|
||||
"input_tokens": 60130,
|
||||
"output_tokens": 9282,
|
||||
"output_tokens_details": {
|
||||
"thinking_tokens": 604
|
||||
},
|
||||
"server_tool_use": {
|
||||
"web_fetch_requests": 0,
|
||||
"web_search_requests": 0
|
||||
}
|
||||
}
|
||||
},
|
||||
{
|
||||
"cost_usd_estimate": 0.801868,
|
||||
"final_groups": [
|
||||
{
|
||||
"description": "Work-size part 1/2 of one family. Byte-array frame builder, hardware-free decoder-call fixture, capture of decoded object, and field/checksum/rejection-code comparison",
|
||||
"members": [
|
||||
"CASE-REC-02",
|
||||
"CASE-REC-04",
|
||||
"CASE-REC-08",
|
||||
"CASE-REC-10"
|
||||
],
|
||||
"name": "Status Frame Decoder Byte Harness (1/2)"
|
||||
},
|
||||
{
|
||||
"description": "Work-size part 2/2 of one family. Byte-array frame builder, hardware-free decoder-call fixture, capture of decoded object, and field/checksum/rejection-code comparison",
|
||||
"members": [
|
||||
"CASE-REC-00",
|
||||
"CASE-REC-06",
|
||||
"CASE-REC-12"
|
||||
],
|
||||
"name": "Status Frame Decoder Byte Harness (2/2)"
|
||||
},
|
||||
{
|
||||
"description": "Work-size part 1/2 of one family. Controllable pulse source at controller input, digital capture of the reset line, and a deadline check within the supplied timing tolerance",
|
||||
"members": [
|
||||
"CASE-REC-01",
|
||||
"CASE-REC-05",
|
||||
"CASE-REC-07",
|
||||
"CASE-REC-11"
|
||||
],
|
||||
"name": "Watchdog Pulse Reset-Line Timing Capture (1/2)"
|
||||
},
|
||||
{
|
||||
"description": "Work-size part 2/2 of one family. Controllable pulse source at controller input, digital capture of the reset line, and a deadline check within the supplied timing tolerance",
|
||||
"members": [
|
||||
"CASE-REC-03",
|
||||
"CASE-REC-09",
|
||||
"CASE-REC-13"
|
||||
],
|
||||
"name": "Watchdog Pulse Reset-Line Timing Capture (2/2)"
|
||||
}
|
||||
],
|
||||
"mode": "hierarchical",
|
||||
"natural_groups": [
|
||||
{
|
||||
"common_work": "Byte-array frame builder, hardware-free decoder-call fixture, capture of decoded object, and field/checksum/rejection-code comparison",
|
||||
"description": "Software-only tests: a byte-array builder encodes watchdog status frames and the decoder is called. Decoded fields, checksum status and rejection code are asserted. Frames are nominal or fault-injected, but the faults used are not stated.",
|
||||
"evidence": [
|
||||
{
|
||||
"alias": "CASE-REC-00",
|
||||
"field": "success_criteria",
|
||||
"quote": "Capture the decoded object and compare fields, checksum status, and rejection code"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-REC-02",
|
||||
"field": "preconditions",
|
||||
"quote": "Use a byte-array builder and a decoder-call fixture; no live hardware is required"
|
||||
}
|
||||
],
|
||||
"members": [
|
||||
"CASE-REC-00",
|
||||
"CASE-REC-02",
|
||||
"CASE-REC-04",
|
||||
"CASE-REC-06",
|
||||
"CASE-REC-08",
|
||||
"CASE-REC-10",
|
||||
"CASE-REC-12"
|
||||
],
|
||||
"name": "Status Frame Decoder Byte Harness",
|
||||
"rationale": "All members have the same preconditions and success_criteria. Nominal and fault-injected cases change only the frame bytes and the expected decode or rejection result. Once the builder and comparison helper exist, each extra case is a cheap input and assertion variation.",
|
||||
"uncertainty": "The source does not state which corruptions the fault-injection frames use or which rejection codes they expect. Identical text may still hide different frame contents.",
|
||||
"variation_sets": [
|
||||
[
|
||||
"CASE-REC-02",
|
||||
"CASE-REC-04",
|
||||
"CASE-REC-08",
|
||||
"CASE-REC-10"
|
||||
],
|
||||
[
|
||||
"CASE-REC-00",
|
||||
"CASE-REC-06",
|
||||
"CASE-REC-12"
|
||||
]
|
||||
]
|
||||
},
|
||||
{
|
||||
"common_work": "Controllable pulse source at controller input, digital capture of the reset line, and a deadline check within the supplied timing tolerance",
|
||||
"description": "Hardware tests: a pulse source stops and resumes watchdog pulses at the controller input. A digital capture records reset-line timing for a deadline check within tolerance. Deadlines and fault pulse patterns are not stated.",
|
||||
"evidence": [
|
||||
{
|
||||
"alias": "CASE-REC-01",
|
||||
"field": "success_criteria",
|
||||
"quote": "Measure reset-line timing with a digital capture fixture and check the deadline"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-REC-03",
|
||||
"field": "preconditions",
|
||||
"quote": "Use a controllable pulse source and reset-line recorder; timing tolerance is supplied"
|
||||
}
|
||||
],
|
||||
"members": [
|
||||
"CASE-REC-01",
|
||||
"CASE-REC-03",
|
||||
"CASE-REC-05",
|
||||
"CASE-REC-07",
|
||||
"CASE-REC-09",
|
||||
"CASE-REC-11",
|
||||
"CASE-REC-13"
|
||||
],
|
||||
"name": "Watchdog Pulse Reset-Line Timing Capture",
|
||||
"rationale": "All members share the same physical stimulus equipment and digital-capture timing measurement. This is materially different from the software decoder harness. Nominal and fault cases differ only in pulse pattern and expected timing, using the same equipment and workflow.",
|
||||
"uncertainty": "Deadline values, tolerances and the fault-injection pulse patterns are not stated. Fault cases might need stimulus control beyond simply stopping and resuming pulses.",
|
||||
"variation_sets": [
|
||||
[
|
||||
"CASE-REC-01",
|
||||
"CASE-REC-05",
|
||||
"CASE-REC-07",
|
||||
"CASE-REC-11",
|
||||
"CASE-REC-13"
|
||||
],
|
||||
[
|
||||
"CASE-REC-03",
|
||||
"CASE-REC-09"
|
||||
]
|
||||
]
|
||||
}
|
||||
],
|
||||
"passes": [
|
||||
{
|
||||
"cli_turns": 2,
|
||||
"input_bytes": 9394,
|
||||
"max_turns": 4,
|
||||
"stage": "implementation_profiles",
|
||||
"wall_seconds": 10.335
|
||||
},
|
||||
{
|
||||
"cli_turns": 2,
|
||||
"input_bytes": 9386,
|
||||
"max_turns": 4,
|
||||
"stage": "implementation_profiles",
|
||||
"wall_seconds": 10.134
|
||||
},
|
||||
{
|
||||
"cli_turns": 3,
|
||||
"input_bytes": 9386,
|
||||
"max_turns": 4,
|
||||
"stage": "implementation_profiles",
|
||||
"wall_seconds": 19.774
|
||||
},
|
||||
{
|
||||
"cli_turns": 3,
|
||||
"input_bytes": 7049,
|
||||
"max_turns": 4,
|
||||
"stage": "implementation_profiles",
|
||||
"wall_seconds": 12.544
|
||||
},
|
||||
{
|
||||
"cli_turns": 2,
|
||||
"input_bytes": 17434,
|
||||
"max_turns": 4,
|
||||
"stage": "profile_proposal_a",
|
||||
"wall_seconds": 5.923
|
||||
},
|
||||
{
|
||||
"cli_turns": 2,
|
||||
"input_bytes": 17434,
|
||||
"max_turns": 4,
|
||||
"stage": "profile_proposal_b",
|
||||
"wall_seconds": 5.723
|
||||
},
|
||||
{
|
||||
"cli_turns": 3,
|
||||
"input_bytes": 18258,
|
||||
"max_turns": 4,
|
||||
"stage": "profile_reconciliation",
|
||||
"wall_seconds": 9.535
|
||||
},
|
||||
{
|
||||
"cli_turns": 3,
|
||||
"input_bytes": 16167,
|
||||
"max_turns": 4,
|
||||
"stage": "source_review",
|
||||
"wall_seconds": 21.383
|
||||
},
|
||||
{
|
||||
"cli_turns": 3,
|
||||
"input_bytes": 20937,
|
||||
"max_turns": 4,
|
||||
"stage": "large_family_review",
|
||||
"wall_seconds": 26.395
|
||||
},
|
||||
{
|
||||
"cli_turns": 3,
|
||||
"input_bytes": 25745,
|
||||
"max_turns": 4,
|
||||
"stage": "decision_audit",
|
||||
"wall_seconds": 26.992
|
||||
}
|
||||
],
|
||||
"quality": {
|
||||
"coverage": true,
|
||||
"false_merge_pairs": 0,
|
||||
"families": 2,
|
||||
"missed_merge_pairs": 0,
|
||||
"pair_precision": 1.0,
|
||||
"pair_recall": 1.0
|
||||
},
|
||||
"seconds": 148.75216588703915,
|
||||
"status": "completed",
|
||||
"usage": {
|
||||
"cache_creation": {
|
||||
"ephemeral_1h_input_tokens": 0,
|
||||
"ephemeral_5m_input_tokens": 0
|
||||
},
|
||||
"cache_creation_input_tokens": 0,
|
||||
"cache_read_input_tokens": 0,
|
||||
"input_tokens": 111617,
|
||||
"output_tokens": 17770,
|
||||
"output_tokens_details": {
|
||||
"thinking_tokens": 635
|
||||
},
|
||||
"server_tool_use": {
|
||||
"web_fetch_requests": 0,
|
||||
"web_search_requests": 0
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"mocked_synthetic_363": [
|
||||
{
|
||||
"mode": "direct",
|
||||
"backend": "mocked-independent-fixture-oracle",
|
||||
"elapsed_seconds": 6.596287971013226,
|
||||
"source_size_metadata": {
|
||||
"source_bytes": 410322,
|
||||
"description": {
|
||||
"min": 113,
|
||||
"max": 650,
|
||||
"median": 133,
|
||||
"p95": 641
|
||||
},
|
||||
"preconditions": {
|
||||
"min": 108,
|
||||
"max": 694,
|
||||
"median": 131,
|
||||
"p95": 682
|
||||
},
|
||||
"case_type": {
|
||||
"min": 7,
|
||||
"max": 15,
|
||||
"median": 8,
|
||||
"p95": 15
|
||||
},
|
||||
"success_criteria": {
|
||||
"min": 149,
|
||||
"max": 1009,
|
||||
"median": 164,
|
||||
"p95": 1002
|
||||
}
|
||||
},
|
||||
"request_bytes": 410474,
|
||||
"result_bytes": 175646,
|
||||
"model_calls": 5,
|
||||
"natural_families": 45,
|
||||
"final_tasks": 90,
|
||||
"max_group_size": 5,
|
||||
"fixture_assignment_validation": {
|
||||
"coverage": true,
|
||||
"families": 45,
|
||||
"pair_precision": 1.0,
|
||||
"pair_recall": 1.0,
|
||||
"false_merge_pairs": 0,
|
||||
"missed_merge_pairs": 0
|
||||
},
|
||||
"measured_provider_usage": null,
|
||||
"measured_provider_cost": null
|
||||
},
|
||||
{
|
||||
"mode": "hierarchical",
|
||||
"backend": "mocked-independent-fixture-oracle",
|
||||
"elapsed_seconds": 6.578743355930783,
|
||||
"source_size_metadata": {
|
||||
"source_bytes": 410322,
|
||||
"description": {
|
||||
"min": 113,
|
||||
"max": 650,
|
||||
"median": 133,
|
||||
"p95": 641
|
||||
},
|
||||
"preconditions": {
|
||||
"min": 108,
|
||||
"max": 694,
|
||||
"median": 131,
|
||||
"p95": 682
|
||||
},
|
||||
"case_type": {
|
||||
"min": 7,
|
||||
"max": 15,
|
||||
"median": 8,
|
||||
"p95": 15
|
||||
},
|
||||
"success_criteria": {
|
||||
"min": 149,
|
||||
"max": 1009,
|
||||
"median": 164,
|
||||
"p95": 1002
|
||||
}
|
||||
},
|
||||
"request_bytes": 410474,
|
||||
"result_bytes": 224822,
|
||||
"model_calls": 26,
|
||||
"natural_families": 45,
|
||||
"final_tasks": 90,
|
||||
"max_group_size": 5,
|
||||
"fixture_assignment_validation": {
|
||||
"coverage": true,
|
||||
"families": 45,
|
||||
"pair_precision": 1.0,
|
||||
"pair_recall": 1.0,
|
||||
"false_merge_pairs": 0,
|
||||
"missed_merge_pairs": 0
|
||||
},
|
||||
"measured_provider_usage": null,
|
||||
"measured_provider_cost": null
|
||||
}
|
||||
],
|
||||
"unit_tests": 200,
|
||||
"real_suite_rerun": false,
|
||||
"large_hosted_quality_verified": false
|
||||
}
|
||||
@ -1,5 +1,9 @@
|
||||
# Suite job time and cost budget
|
||||
|
||||
Current limits and recovery behavior supersede the historical details below. See
|
||||
[the adaptive execution handoff](hermes_suite_recovery_20260930.md): 7200 seconds,
|
||||
no estimated-cost cutoff, bounded pass recovery and optional hierarchical execution.
|
||||
|
||||
New explicitly requested jobs may use `execution.max_seconds: 3600`. Omitting
|
||||
the field still selects 1800 seconds; valid values are integers from 10 to 3600.
|
||||
The USD 30 CLI estimated-cost maximum/default is unchanged. These figures are
|
||||
|
||||
@ -1,5 +1,9 @@
|
||||
# Multi-pass implementation grouping
|
||||
|
||||
Current limits and recovery behavior supersede the historical details below. See
|
||||
[the adaptive execution handoff](hermes_suite_recovery_20260930.md): 7200 seconds,
|
||||
no estimated-cost cutoff, bounded pass recovery and optional hierarchical execution.
|
||||
|
||||
The suite planner keeps the same HTTPS endpoint, request fields, authentication,
|
||||
provider permissions, generalized-data approval, and asynchronous job lifecycle.
|
||||
The required result object remains `result.groups`, with `name`, `description`,
|
||||
|
||||
@ -4,26 +4,20 @@ This optional API groups one complete campaign/suite into implementation familie
|
||||
The existing `/local-model` API, GPU allocation, serving model, and limits are unchanged.
|
||||
Deployment uses Flux; the application and workbook stay on the laptop.
|
||||
|
||||
## Current multi-pass policy
|
||||
## Current adaptive policy
|
||||
|
||||
New jobs use [the multi-pass implementation policy](hermes_suite_multipass.md):
|
||||
two independent full-suite proposals, reconciliation, up to two bounded semantic
|
||||
reviews, and a programmatic five-case final cap. Names are at most 64 characters.
|
||||
The endpoint and request fields are unchanged. Optional top-level `review_summary`
|
||||
is returned only with the authorized result. Configuration remains `suite-v6-20260929`;
|
||||
policy, prompt, and execution revisions identify the changed grouping behavior.
|
||||
The prior single-pass acceptance results below are historical and do not measure
|
||||
the new workflow. New jobs may explicitly request up to 3600 seconds; the default
|
||||
remains 1800 seconds. One deadline and the USD 30 CLI estimate guard cover all invocations. This is not subscription
|
||||
billing. Progress updates every five seconds during a CLI pass. The separate local-model endpoint is unchanged;
|
||||
the 8K local route cannot admit the new multi-pass reconciliation schema.
|
||||
The current [recovery and adaptive-analysis handoff](hermes_suite_recovery_20260930.md)
|
||||
records the exact latest failure, deployed behavior, tests and client settings.
|
||||
New jobs accept up to 7200 seconds, with one shared deadline and no estimated-cost
|
||||
cutoff. Legacy `max_cost_usd` is accepted but unenforced. The HTTP compatibility
|
||||
revision stays `suite-v6-20260929`; execution is `suite-adaptive-v11-20260930`.
|
||||
|
||||
The active model is **`claude-opus-5-5`**, invoked as `claude-opus-5-5[1m]`
|
||||
using native Claude Code 2.1.285. This was the user-requested 5.5 upgrade;
|
||||
older Opus 4.8 acceptance records below are historical. Effort is high, or
|
||||
xhigh at 100+ cases or 128 KiB+ source content. The model and effort stay fixed
|
||||
throughout a job. Current execution revision: `suite-multipass-v10-20260930`.
|
||||
See the [current capacity fix and client settings](hermes_suite_capacity_20260930.md).
|
||||
Small suites prefer two full-suite proposals and reconciliation. High-risk suites
|
||||
use compact profiles, two independent global profile proposals, global
|
||||
reconciliation and full-source verification before semantic review and the hard
|
||||
five-case final cap. Recovery preserves validated active-job checkpoints. No
|
||||
provider fallback, source-content logs or partial completion is enabled. Pinned
|
||||
model remains `claude-opus-5-5` through native CLI 2.1.285, high/xhigh by suite size.
|
||||
|
||||
## Earlier CLI failure diagnostics update (historical)
|
||||
|
||||
|
||||
221
docs/hermes_suite_recovery_20260930.md
Normal file
221
docs/hermes_suite_recovery_20260930.md
Normal file
@ -0,0 +1,221 @@
|
||||
# Suite recovery and adaptive analysis
|
||||
|
||||
## Latest real-job diagnosis
|
||||
|
||||
Exact failed job: `90ab61bf285e4c89a41d893d4fee649d`, located by descending
|
||||
persisted creation time, failed status and 363-case count. Its request selected
|
||||
3600 seconds and an estimated-cost guard of 30. No real source was inspected or
|
||||
rerun. The record is not the earlier decision-audit capacity failure.
|
||||
|
||||
| Measurement | Retained evidence |
|
||||
| --- | --- |
|
||||
| Error | `incomplete_generation`, `failure_stage=cli_final_result` |
|
||||
| Failed pass | `proposal_b`; one completed model pass |
|
||||
| CLI exit / signal | 1 / null; `nonzero_exit` |
|
||||
| Terminal event | `result`, subtype `success`, **is_error=true** |
|
||||
| Provider / assistant stop | `stop_sequence` / `stop_sequence` |
|
||||
| API error indicator | true; allowlisted category `unknown`; HTTP status unknown |
|
||||
| Structured result | Absent from terminal output, tool input and JSON text candidates |
|
||||
| Turns / ceiling | 1 / 6; no turn-limit or structured-retry-limit outcome |
|
||||
| B input / output usage | CLI reported 0 / 0; does not independently prove no provider work occurred |
|
||||
| B observed model capacity | Unknown; failed terminal event had no modelUsage limits |
|
||||
| B admission | Same 439270-byte input/system/schema size as A; conservative bound 447462; six full outputs reserved: 831462 under 1M |
|
||||
| B configured output / reasoning | 64000 per turn / xhigh; reasoning-token limit unknown |
|
||||
| B output bytes | Last heartbeat observed 54366; exact terminal byte count was not retained |
|
||||
| B elapsed / timeout | Last heartbeat 411.8 s; approximately 412.9 s including boundary overhead / 2791.629775 s watchdog |
|
||||
| Provider timeout | Unknown |
|
||||
| Job elapsed / remaining | 1221.233 s / 2378.8 s |
|
||||
| Total input / output | 319884 / 98177; output includes 82581 thinking tokens |
|
||||
| Cost used / remaining old guard | USD 3.243076 / 26.756924 estimated |
|
||||
| Route | Claude CLI, first-party OAuth; selected `claude-opus-5-5[1m]` |
|
||||
| Runtime model evidence | A confirmed firstParty canonical `claude-opus-5-5`; B initialization matched requested model, but no successful terminal canonical-model verification |
|
||||
| CLI | 2.1.285 |
|
||||
| Old configuration / prompt | `suite-v6-20260929` / `implementation-proximity-multipass-v5-20260930` |
|
||||
| Old execution / policy | `suite-multipass-v10-20260930` / `implementation-five-v1-20260929` |
|
||||
|
||||
Proposal A completed in 808.317 seconds, three CLI turns, exit zero and valid
|
||||
structured output. B produced an unsuccessful API-error event and exited nonzero.
|
||||
The underlying provider cause is **unknown**: raw CLI text was deliberately deleted.
|
||||
No retained evidence identifies rate limiting, context/output exhaustion, timeout,
|
||||
turn exhaustion or an adapter rejecting valid structured content. The old adapter
|
||||
collapsed this API error into `incomplete_generation` and the workflow had no retry.
|
||||
The new classifier treats an unclassified API error as eligible for a bounded
|
||||
retry, with `provider_failure_transience=unconfirmed`; that is not proof it was
|
||||
transient. New diagnostics retain exact output bytes and per-attempt elapsed time.
|
||||
|
||||
## Time and cost
|
||||
|
||||
The maximum accepted `execution.max_seconds` is **7200**; omission still uses 1800.
|
||||
All profile, grouping, review, retry and repair calls share one absolute deadline.
|
||||
Submit/status/result HTTP timeouts remain 45 seconds on the laptop.
|
||||
|
||||
The cost guard is **removed**, following the user's subsequent instruction. The
|
||||
CLI receives no `--max-budget-usd`. Legacy positive finite `max_cost_usd` values
|
||||
or null are accepted but never enforced. Capability/status metadata reports
|
||||
`cost_guard_enforced=false` and a null cost ceiling. Measured estimates remain
|
||||
informational; unknown usage/cost does not prevent further work. These estimates
|
||||
are not a statement of subscription billing charges.
|
||||
|
||||
7200 seconds is the recommended next limit. Five passes at A's 808.317 seconds
|
||||
would total 4041.585 seconds, leaving 3158.415 seconds for variability/recovery.
|
||||
Hierarchical execution has a different call mix, so this is a planning reference,
|
||||
not a guarantee or a measured hierarchical runtime. No fixed deadline guarantees
|
||||
success through a prolonged provider outage.
|
||||
|
||||
Later stages retain explicit time floors. Global stages reserve
|
||||
`min(500, max(10, job_seconds / 8))` seconds per remaining global stage; pending
|
||||
profile batches reserve 30 seconds each, and source/review batches 60 seconds.
|
||||
These are admission/watchdog reservations, not guaranteed runtimes. Each attempt
|
||||
receives only time remaining after future reservations. Insufficient remaining
|
||||
reservation returns `insufficient_remaining_pass_budget`; the absolute deadline
|
||||
returns `job_time_budget_exhausted`. No per-pass allowance resets the job clock.
|
||||
|
||||
## Recovery and checkpoints
|
||||
|
||||
At most three attempts per logical pass, each a fresh isolated CLI session with
|
||||
the same provider, model, complete pass source and objective. Retry backoff is
|
||||
2 then 4 seconds, cancellable. CLI turns start at four and can increase to five
|
||||
then six only when complete-input context admission allows. No compaction,
|
||||
truncation, provider fallback or partial result is accepted.
|
||||
|
||||
Retriable outcomes include unsuccessful/missing terminal generation, CLI turn or
|
||||
structured-output retry exhaustion, rate limiting, temporary transport failures,
|
||||
nonzero transient exits, and distinct pass timeouts. One additional structured
|
||||
assignment/JSON repair attempt is allowed. Successful final JSON text envelopes
|
||||
are recognized after the normal completion, runtime-model and isolation guards;
|
||||
normal schema/membership validation still applies.
|
||||
|
||||
Authentication/authorization, model/isolation changes, cancellation, invalid
|
||||
requests, confirmed capacity violations and deterministic evidence/review failures
|
||||
are not blindly retried. Capacity or exhausted generation can select hierarchical
|
||||
analysis instead of repeating the same full-suite request. Final errors identify
|
||||
turn exhaustion, generation retries, output/context limits, provider failure,
|
||||
pass timeout or insufficient remaining budget. Status contains attempts, terminal
|
||||
metadata and retained checkpoint names, never generated content.
|
||||
|
||||
Every **validated** output is checkpointed in active-job memory. Identity binds
|
||||
the complete request hash, credential owner/scope, routing, provider, canonical
|
||||
model, configuration/prompt/policy/execution revisions, exact pass source, schema,
|
||||
objective and relevant prior-output hashes. Altering any input prevents reuse.
|
||||
Completed predecessors survive recovery of a later pass. A matching checkpoint
|
||||
restores a copy without another model call. Contents never enter SQLite, logs,
|
||||
telemetry or ordinary status; all checkpoints are removed when the job ends or
|
||||
its deadline expires (at most 7200 seconds).
|
||||
|
||||
**Restart limitation:** checkpoints are not durable. A worker restart marks an
|
||||
in-flight job `interrupted_no_retry`; no automatic provider resubmission occurs.
|
||||
A fresh client attempt after restart regenerates its passes. This deliberately
|
||||
uses the requested active-job recovery option instead of adding disk copies of
|
||||
source-derived outputs. Results retain the existing one-hour volatile retention.
|
||||
|
||||
## Adaptive execution
|
||||
|
||||
Small requests prefer direct independent proposals, whole-suite reconciliation,
|
||||
large-family review and decision audit. Initial hierarchical selection occurs for
|
||||
250+ cases, 256 KiB+ canonical source, or a p95 source field of 8192+ bytes. Direct
|
||||
capacity failure also selects hierarchical mode before inference. Retriable direct
|
||||
generation exhaustion or confirmed output/context limits can switch modes within
|
||||
the same job and remaining deadline; the transition is recorded.
|
||||
|
||||
Hierarchical mode:
|
||||
|
||||
1. Extract profiles in deterministic batches of at most 24 cases or about 48 KiB.
|
||||
A single larger case remains intact and must pass capacity admission. Each
|
||||
profile preserves target, success-criteria work, setup, stimulus, observations,
|
||||
machinery, variations, uncertainty, exact alias and original source evidence.
|
||||
2. Run two independent proposals over **every profile in the complete suite**,
|
||||
with an alternative reproducible order for the second. Reconcile globally;
|
||||
profile-processing batches never determine family membership.
|
||||
3. Reopen all original generalized fields for every candidate family in bounded
|
||||
whole-family source reviews. Verify original evidence, split unsupported
|
||||
compatibility, and reject ownership drift. Only then accept natural families.
|
||||
4. Perform the existing oversized-family review, decision audit, exact coverage,
|
||||
unique short names and deterministic balanced five-case packaging.
|
||||
|
||||
Profiles have explicit size bounds; they are not arbitrary silent summarization.
|
||||
No profile can be missing, invented, or silently truncated. At the current
|
||||
400-case service maximum, bounded profiles plus membership-only proposal context
|
||||
fit one global comparison request. A worst-case 400-alias ASCII regression admitted
|
||||
profile reconciliation at 618260 input/system/schema bytes, with five reserved CLI
|
||||
outputs. Every real expanded request is checked again. The bounded contract avoids
|
||||
needing overlapping-neighborhood reconciliation; no such fallback is claimed.
|
||||
A single full-source family that cannot fit intact still fails explicitly, as do
|
||||
excessive profile/source/review batches. Transport and case ceilings remain 1 MiB
|
||||
and 400, not unlimited input support.
|
||||
|
||||
Bounds: 32 profile batches, 16 source-review batches, eight batches per semantic
|
||||
review, 64 validated logical calls and 128 model invocations including retries.
|
||||
All share the same deadline. Optional metadata includes `execution_mode`,
|
||||
`strategy_events`, `direct_pass_attempts`, `retry_count`, `checkpointed_passes`,
|
||||
`checkpoint_reused`, profile/source batch counts, `cross_batch_review_count`,
|
||||
`model_pass_count`, `natural_family_count`, `final_task_count` and per-attempt
|
||||
measurements. Content-free source byte distributions now support better future
|
||||
fault fixtures. Engineering rationales remain only in authorized `review_summary`.
|
||||
|
||||
## Verification
|
||||
|
||||
200 focused local tests passed. They cover isolated retries, turn adaptation,
|
||||
assignment repair, non-retriable errors, preserved predecessor checkpoints,
|
||||
checkpoint scope/expiry/restart semantics, global profile coverage, complete source
|
||||
reopening, no fallback, shared deadlines, reserved future time, and cost estimates
|
||||
exceeding the old guard without stopping work. Manifest render and client dry run
|
||||
passed; Flux diff was reviewed.
|
||||
|
||||
Hosted native Opus 5.5 comparison used the same synthetic 14 cases, interleaved
|
||||
between byte-array decoder tests and physical pulse/reset timing tests. Duplicate
|
||||
text retained distinct aliases. Hierarchical mode deliberately used four profile
|
||||
batches, then global comparisons, full-source review and both semantic reviews.
|
||||
|
||||
| Hosted execution | Seconds | Calls | Input / output tokens | CLI estimate | Natural / final groups |
|
||||
| --- | ---: | ---: | --- | ---: | --- |
|
||||
| Direct | 79.009 | 5 | 60130 / 9282 | USD 0.426160 | 2 / 4 |
|
||||
| Hierarchical | 148.752 | 10 | 111617 / 17770 | USD 0.801868 | 2 / 4 |
|
||||
|
||||
Both had exact alias coverage, zero incorrect merge pairs, zero unnecessary
|
||||
semantic split pairs, no retries and valid balanced 4+3 packages per family.
|
||||
Descriptions kept decoder and hardware-capture objectives separate. The direct
|
||||
answer inferred that fault cases varied only in stimulus; the hierarchical answer
|
||||
more accurately retained uncertainty about unspecified fault patterns. This is a
|
||||
small quality check, not proof of real-suite quality or runtime.
|
||||
|
||||
The 363-case regression used 410322 source bytes, matching the failed record, with
|
||||
45 interleaved machinery patterns and distant identical records. Actual historical
|
||||
per-field lengths were not retained, so that distribution cannot be matched or
|
||||
claimed equivalent. Both modes completed with **mocked model answers from the
|
||||
independent fixture map**, producing 45 natural families and 90 final tasks,
|
||||
maximum five members, balanced sizes and unique short names. Direct used five
|
||||
calls (175646 response bytes); hierarchical used 26 (224822 response bytes).
|
||||
Mocked execution was about 6.6 seconds per mode; these are local orchestration
|
||||
measurements, not hosted latency, cost or a model-quality result. No new hosted
|
||||
363-case benchmark or real-suite rerun was performed.
|
||||
|
||||
Detailed [synthetic evidence](evidence/hermes_suite_recovery_20260930.json) and
|
||||
[the retained safe failure record](evidence/hermes_suite_failure_90ab61bf_metadata.json)
|
||||
are separate. No source fields or raw provider messages from the real suite are
|
||||
included in either artifact.
|
||||
|
||||
## Client handoff and revisions
|
||||
|
||||
No new request fields. `execution.strategy="whole_suite"` still submits one
|
||||
complete suite and receives one combined `result.groups` result. Authentication,
|
||||
endpoint, credential scopes, pinned Opus 5.5, effort thresholds and routing remain.
|
||||
|
||||
```python
|
||||
AI_MAX_SECONDS = 7200
|
||||
AI_JOB_REVISION = "suite-adaptive-v11-20260930-1"
|
||||
```
|
||||
|
||||
Generate a **new Idempotency-Key**. `AI_MAX_COST_USD` may remain at its existing
|
||||
value if the client sends it; no client edit is needed to disable server cost
|
||||
limits. Do not reuse the failed job's key.
|
||||
|
||||
- Configuration: `suite-v6-20260929`, HTTP compatibility unchanged.
|
||||
- Execution: `suite-adaptive-v11-20260930`.
|
||||
- Prompt: `implementation-proximity-adaptive-v6-20260930`.
|
||||
- Grouping policy: `implementation-five-v1-20260929`, unchanged.
|
||||
- Strategy: `suite-adaptive-selection-v1-20260930`.
|
||||
- Capacity reservation: `suite-context-turns-v1-20260930`.
|
||||
|
||||
Rollback while idle: revert the recovery deployment Git commit and reconcile the
|
||||
`hermes` Flux Kustomization. Do not edit live resources. Reverting reinstates the
|
||||
old 3600-second maximum and cost guard. Save any desired volatile results first.
|
||||
@ -29,11 +29,11 @@ def main():
|
||||
"""Submit once, poll safely, and report measured metadata and independent scores."""
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--size", type=int, choices=(14, 75, 363), default=75)
|
||||
parser.add_argument("--max-seconds", type=int, default=3600)
|
||||
parser.add_argument("--max-seconds", type=int, default=7200)
|
||||
args = parser.parse_args()
|
||||
request, expected = fixture(args.size)
|
||||
request["routing"] = {"allow_external": True, "allowed_external_providers": ["claude"]}
|
||||
request["execution"] = {"strategy": "whole_suite", "max_seconds": args.max_seconds, "max_cost_usd": 30}
|
||||
request["execution"] = {"strategy": "whole_suite", "max_seconds": args.max_seconds, "max_cost_usd": None}
|
||||
token = Path("/vault/secrets/synthetic-token").read_text().strip()
|
||||
opener = build_opener(ProxyHandler({}))
|
||||
with tempfile.TemporaryDirectory(prefix="suite-budget-probe-", dir="/jobs") as directory:
|
||||
@ -62,9 +62,9 @@ def main():
|
||||
"http_timeout_seconds": 45, "automatic_job_retries": 0}
|
||||
try:
|
||||
status, capabilities = http("/v1/capabilities")
|
||||
assert status == 200 and capabilities["max_seconds"] == 3600
|
||||
assert status == 200 and capabilities["max_seconds"] == 7200
|
||||
assert capabilities["default_max_seconds"] == 1800
|
||||
invalid = {**request, "execution": {**request["execution"], "max_seconds": 3601}}
|
||||
invalid = {**request, "execution": {**request["execution"], "max_seconds": 7201}}
|
||||
status, rejected = http("/v1/preflight", invalid)
|
||||
assert status == 400 and rejected["error"]["code"] == "invalid_timeout"
|
||||
status, checked = http("/v1/preflight", request)
|
||||
|
||||
@ -35,7 +35,7 @@ def main():
|
||||
raw, expected = fixture()
|
||||
source_bytes = len(encoded(raw))
|
||||
raw["routing"] = {"allow_external": True, "allowed_external_providers": ["claude"]}
|
||||
raw["execution"] = {"strategy": "whole_suite", "max_seconds": 3600, "max_cost_usd": 30}
|
||||
raw["execution"] = {"strategy": "whole_suite", "max_seconds": 7200}
|
||||
request = validate_request(raw, ["claude"])
|
||||
selected = preflight(request)
|
||||
report = {"test": "synthetic_capacity_363", "source_bytes": source_bytes,
|
||||
@ -61,7 +61,10 @@ def main():
|
||||
result = jobs.get(job["job_id"], "synthetic-capacity-probe", result=True)
|
||||
validate_result(result["result"], request)
|
||||
stages = {p["stage"] for p in job["passes"]}
|
||||
assert stages == {"proposal_a", "proposal_b", "reconciliation", "large_family_review", "decision_audit"}
|
||||
assert {"large_family_review", "decision_audit"} <= stages
|
||||
assert ({"proposal_a", "proposal_b", "reconciliation"} <= stages or
|
||||
{"implementation_profiles", "profile_proposal_a", "profile_proposal_b",
|
||||
"profile_reconciliation", "source_review"} <= stages)
|
||||
review = result["review_summary"]
|
||||
assert all(d["part_sizes"] == balanced_sizes(d["natural_case_count"]) for d in review["capacity_divisions"])
|
||||
report.update(response_bytes=len(encoded(result)), exact_coverage=True,
|
||||
|
||||
67
scripts/ops/hermes_suite_recovery_probe.py
Executable file
67
scripts/ops/hermes_suite_recovery_probe.py
Executable file
@ -0,0 +1,67 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Compare direct and hierarchical execution on a fixed synthetic 14-case suite.
|
||||
|
||||
Run inside the planner with staged modules. No real source or job ID is accepted.
|
||||
Reports operational metadata plus synthetic-only quality evidence. No live job DB.
|
||||
"""
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
|
||||
sys.path.insert(0, os.environ.get('SUITE_PROBE_MODULE_DIR', '/opt/planner'))
|
||||
import suite_profiles
|
||||
from suite_contract import preflight, validate_request, validate_result
|
||||
from suite_multipass import generate
|
||||
from suite_synthetic import PATTERNS, score
|
||||
|
||||
|
||||
def main():
|
||||
"""Use interleaved mechanisms, identical records, and cross-processing-batch peers."""
|
||||
cases, expected = [], {}
|
||||
for i in range(14):
|
||||
family, action, assertion, setup = PATTERNS[i % 2]
|
||||
alias = 'CASE-REC-%02d' % i
|
||||
cases.append({'alias': alias, 'description': action,
|
||||
'success_criteria': assertion, 'preconditions': setup,
|
||||
'case_type': 'nominal' if i % 3 else 'fault injection'})
|
||||
expected[alias] = family
|
||||
cases[-2] = {**cases[0], 'alias': cases[-2]['alias']}
|
||||
request = validate_request({'campaign':'SYNTHETIC', 'suite':'RECOVERY-14', 'cases':cases,
|
||||
'routing':{'allow_external':True,'allowed_external_providers':['claude']},
|
||||
'execution':{'strategy':'whole_suite','max_seconds':900}}, ['claude'])
|
||||
suite_profiles.BATCH_CASES = 4 # Test processing-boundary independence explicitly.
|
||||
for mode in ('direct','hierarchical'):
|
||||
last = [0]
|
||||
def progress(p):
|
||||
if time.monotonic()-last[0] >= 20:
|
||||
print(json.dumps({'mode':mode, **{k:p.get(k) for k in (
|
||||
'current_pass','completed_model_passes','job_elapsed_seconds','retry_count')}}),
|
||||
file=sys.stderr, flush=True)
|
||||
last[0] = time.monotonic()
|
||||
started = time.monotonic()
|
||||
try:
|
||||
result, metadata = generate(request, {**preflight(request),'execution_mode':mode},
|
||||
threading.Event(), '192.168.22.8', progress,
|
||||
credential_scope={'owner':'synthetic-recovery-probe','providers':['claude']})
|
||||
validate_result(result, request)
|
||||
quality = score({'groups':metadata['review_summary']['natural_families']}, expected)
|
||||
print(json.dumps({'mode':mode,'status':'completed','seconds':time.monotonic()-started,
|
||||
'quality':quality,'usage':metadata['usage'],'cost_usd_estimate':metadata['cost_usd_estimate'],
|
||||
'passes':[{k:r.get(k) for k in ('stage','wall_seconds','cli_turns','max_turns','input_bytes')}
|
||||
for r in metadata['passes']],
|
||||
'natural_groups':metadata['review_summary']['natural_families'],
|
||||
'final_groups':result['groups']}, sort_keys=True),flush=True)
|
||||
except Exception as exc:
|
||||
if hasattr(exc,'document'):
|
||||
print(json.dumps({'mode':mode,'status':'failed',**exc.document()}),flush=True)
|
||||
else:
|
||||
print(json.dumps({'mode':mode,'status':'failed','error_type':type(exc).__name__}),flush=True)
|
||||
return 1
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
raise SystemExit(main())
|
||||
@ -66,6 +66,9 @@ configMapGenerator:
|
||||
- suite_sizing.py=scripts/suite_sizing.py
|
||||
- suite_multipass.py=scripts/suite_multipass.py
|
||||
- suite_review_batches.py=scripts/suite_review_batches.py
|
||||
- suite_recovery.py=scripts/suite_recovery.py
|
||||
- suite_profiles.py=scripts/suite_profiles.py
|
||||
- suite_hierarchical.py=scripts/suite_hierarchical.py
|
||||
- suite_contract.py=scripts/suite_contract.py
|
||||
- suite_synthetic.py=scripts/suite_synthetic.py
|
||||
options:
|
||||
|
||||
@ -17,7 +17,10 @@ from suite_contract import (CLAUDE_MODELS, CLAUDE_VERSION, COST_LIMIT, EXECUTION
|
||||
preflight, validate_request)
|
||||
from suite_jobs import Jobs
|
||||
from suite_policy import POLICY_REVISION
|
||||
from suite_review_batches import MAX_MODEL_CALLS, MAX_REVIEW_BATCHES
|
||||
from suite_review_batches import MAX_REVIEW_BATCHES
|
||||
from suite_recovery import MAX_INVOCATIONS, MAX_ATTEMPTS
|
||||
from suite_hierarchical import STRATEGY_REVISION
|
||||
MAX_MODEL_CALLS = MAX_INVOCATIONS
|
||||
from suite_synthetic import allowed_synthetic, fixture
|
||||
|
||||
|
||||
@ -129,7 +132,10 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"execution_revision": EXECUTION_REVISION,
|
||||
"policy_revision": POLICY_REVISION, "max_final_group_cases": 5,
|
||||
"max_final_name_characters": 64, "model_passes": {"minimum": 3, "maximum": MAX_MODEL_CALLS},
|
||||
"logical_stages": 5, "automatic_review_batching": True,
|
||||
"logical_stages": 5, "execution_modes": ["direct", "hierarchical"],
|
||||
"strategy_revision": STRATEGY_REVISION, "max_attempts_per_pass": MAX_ATTEMPTS,
|
||||
"checkpoint_storage": "active_job_memory", "checkpoint_restart_recovery": False,
|
||||
"checkpoint_retention": "until_job_end_or_deadline", "automatic_review_batching": True,
|
||||
"max_batches_per_review_stage": MAX_REVIEW_BATCHES,
|
||||
"review_summary_location": "GET /v1/jobs/<id>/result: top-level review_summary",
|
||||
"prompt_revision": PROMPT_REVISION, "prompt_sha256": PROMPT_SHA256,
|
||||
@ -142,7 +148,8 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"max_result_bytes": MAX_RESULT, "max_cases": MAX_CASES,
|
||||
"max_seconds": MAX_JOB_SECONDS, "default_max_seconds": TIMEOUT,
|
||||
"concurrency": 1, "queue": False,
|
||||
"max_cost_usd": COST_LIMIT, "cost_basis": "CLI estimate guard; not subscription billing",
|
||||
"max_cost_usd": COST_LIMIT, "cost_guard_enforced": False,
|
||||
"cost_basis": "CLI estimate, informational only; not subscription billing",
|
||||
"progress_interval_seconds": 5,
|
||||
"result_retention_seconds": 3600, "idempotency_retention_seconds": 604800,
|
||||
"tokenizer": None, "provider_retention_verified": False})
|
||||
|
||||
@ -115,7 +115,6 @@ def claude_command(model, max_cost, *, reasoning=None, max_turns=CLAUDE_MAX_TURN
|
||||
"--setting-sources", "", "--settings", encoded(settings).decode(), "--disable-slash-commands",
|
||||
"--permission-mode", "dontAsk", "--no-chrome",
|
||||
"--model", model + "[1m]", "--effort", reasoning,
|
||||
"--max-budget-usd", str(max_cost),
|
||||
"--max-turns", str(max_turns), "--system-prompt", SYSTEM,
|
||||
"--json-schema", encoded(SCHEMA).decode()]
|
||||
|
||||
@ -180,17 +179,27 @@ def parse_claude(raw, expected_model, **process_info):
|
||||
if event.get("type") == "result":
|
||||
final = event
|
||||
if not initialized or not final:
|
||||
fail("incomplete_generation", "missing_init_or_final_event")
|
||||
fail("pass_missing_terminal_event", "missing_init_or_final_event")
|
||||
if final.get("stop_reason") == "model_context_window_exceeded":
|
||||
fail("pass_context_limit", "provider_stop")
|
||||
if final.get("stop_reason") == "max_tokens" or diagnostics["assistant_error_code"] == "max_output_tokens":
|
||||
fail("pass_output_limit", "provider_stop")
|
||||
if final.get("is_error") or final.get("subtype") != "success":
|
||||
if final.get("subtype") == "error_max_budget_usd":
|
||||
fail("job_cost_budget_exhausted", "cli_final_result")
|
||||
fail("unexpected_cli_budget_limit", "cli_final_result")
|
||||
status = final.get("api_error_status")
|
||||
assistant_error = diagnostics["assistant_error_code"]
|
||||
transport_error = diagnostics["cli_transport_error"]
|
||||
code = "incomplete_generation"
|
||||
code = "pass_provider_transient_failure" if diagnostics["assistant_api_error_seen"] else "incomplete_generation"
|
||||
if code == "pass_provider_transient_failure":
|
||||
diagnostics["provider_failure_transience"] = "unconfirmed"
|
||||
if final.get("subtype") == "error_max_turns":
|
||||
fail("pass_turn_limit", "cli_final_result")
|
||||
if final.get("subtype") == "error_max_structured_output_retries":
|
||||
fail("pass_schema_repair", "cli_final_result")
|
||||
if status == 429 or assistant_error == "rate_limit":
|
||||
code = "rate_limit"
|
||||
elif status in (401, 403) or assistant_error in {"authentication_failed", "oauth_org_not_allowed"}:
|
||||
elif status in (401, 403) or assistant_error in {"authentication_failed", "oauth_org_not_allowed", "account_on_hold", "billing_error", "model_not_found"}:
|
||||
code = "provider_authentication"
|
||||
elif transport_error == "request_timeout":
|
||||
code = "backend_timeout_or_unavailable"
|
||||
@ -199,8 +208,6 @@ def parse_claude(raw, expected_model, **process_info):
|
||||
or (type(status) is int and 500 <= status <= 599)):
|
||||
code = "backend_unavailable"
|
||||
fail(code, "cli_final_result")
|
||||
if final.get("stop_reason") in {"max_tokens", "model_context_window_exceeded"}:
|
||||
fail("incomplete_generation", "provider_stop")
|
||||
models = final.get("modelUsage", {})
|
||||
if not isinstance(models, dict) or not models or set(models) - aliases:
|
||||
fail("model_changed", "runtime_model")
|
||||
@ -216,6 +223,10 @@ def parse_claude(raw, expected_model, **process_info):
|
||||
if any(usage.get("server_tool_use", {}).values()):
|
||||
fail("worker_isolation_failed", "provider_tools")
|
||||
result = final.get("structured_output")
|
||||
if not isinstance(result, dict) and diagnostics["final_json_text_present"]:
|
||||
result = json.loads(final["result"])
|
||||
diagnostics.update(structured_output_present=True, structured_output_is_object=True,
|
||||
structured_output_location="result.result_json")
|
||||
if not isinstance(result, dict):
|
||||
fail("invalid_json_result", "structured_output_extraction")
|
||||
if process_info.get("exit_code"):
|
||||
@ -270,7 +281,7 @@ def claude_generate(request, cancel, *, invocation=None, progress=None, job_dead
|
||||
if job_deadline is not None and now >= job_deadline:
|
||||
raise Problem("job_time_budget_exhausted", 504)
|
||||
if now >= deadline:
|
||||
raise Problem("timeout", 504)
|
||||
raise Problem("pass_timeout", 504)
|
||||
activity = (root / "output").stat()
|
||||
if progress and time.monotonic() >= next_progress:
|
||||
# File activity proves CLI activity, not semantic completion.
|
||||
@ -291,7 +302,8 @@ def claude_generate(request, cancel, *, invocation=None, progress=None, job_dead
|
||||
raw = output.read(OUTPUT_BYTES).decode("utf-8", errors="replace")
|
||||
info = {"exit_code": process.returncode, "subprocess_timeout_seconds": seconds,
|
||||
"termination_reason": failure.code if failure else None,
|
||||
"reasoning_effort": effort, "max_turns": max_turns}
|
||||
"reasoning_effort": effort, "max_turns": max_turns,
|
||||
"output_bytes": (root / "output").stat().st_size}
|
||||
if failure:
|
||||
raise Problem(failure.code, failure.status, failure_stage="subprocess",
|
||||
**snapshot(raw, **info))
|
||||
|
||||
@ -56,7 +56,7 @@ def object_text(value):
|
||||
|
||||
|
||||
def snapshot(raw, *, exit_code=None, subprocess_timeout_seconds=None, termination_reason=None,
|
||||
reasoning_effort=None, max_turns=CLAUDE_MAX_TURNS):
|
||||
reasoning_effort=None, max_turns=CLAUDE_MAX_TURNS, output_bytes=None):
|
||||
"""Return bounded diagnostic fields from a CLI stream, including failed runs."""
|
||||
final, last_assistant = None, {}
|
||||
initialized = compacted = tool_candidate = text_candidate = False
|
||||
@ -106,6 +106,7 @@ def snapshot(raw, *, exit_code=None, subprocess_timeout_seconds=None, terminatio
|
||||
for entry in limits.values() if isinstance(entry, dict)] if isinstance(limits, dict) else []
|
||||
return {
|
||||
"execution_revision": EXECUTION_REVISION,
|
||||
"output_bytes": number(output_bytes) if output_bytes is not None else len(raw.encode()),
|
||||
"exit_code": exit_code if type(exit_code) is int and exit_code >= 0 else None,
|
||||
"termination_signal": -exit_code if type(exit_code) is int and exit_code < 0 else None,
|
||||
"termination_reason": termination_reason or (
|
||||
|
||||
@ -8,8 +8,8 @@ import re
|
||||
from collections import Counter
|
||||
|
||||
REVISION = "suite-v6-20260929"
|
||||
PROMPT_REVISION = "implementation-proximity-multipass-v5-20260930"
|
||||
EXECUTION_REVISION = "suite-multipass-v10-20260930"
|
||||
PROMPT_REVISION = "implementation-proximity-adaptive-v6-20260930"
|
||||
EXECUTION_REVISION = "suite-adaptive-v11-20260930"
|
||||
CLAUDE_VERSION = "2.1.285"
|
||||
CLAUDE_MODELS = {
|
||||
"claude-opus-4-8": 64000,
|
||||
@ -32,8 +32,8 @@ MAX_BODY = 1 << 20
|
||||
MAX_RESULT = 1 << 20
|
||||
MAX_CASES = 400
|
||||
TIMEOUT = 1800
|
||||
MAX_JOB_SECONDS = 3600
|
||||
COST_LIMIT = 30.0
|
||||
MAX_JOB_SECONDS = 7200
|
||||
COST_LIMIT = None # Compatibility metadata: estimates are never enforcement limits.
|
||||
RESULT_TTL = 3600
|
||||
FIELDS = {"description", "success_criteria", "preconditions", "operating_condition",
|
||||
"case_type", "verification_method", "target", "swci", "verifies",
|
||||
@ -164,10 +164,10 @@ def validate_request(raw, permissions):
|
||||
if execution.get("strategy", "whole_suite") != "whole_suite":
|
||||
raise Problem("unsupported_strategy", 422)
|
||||
seconds = execution.get("max_seconds", TIMEOUT)
|
||||
cost = execution.get("max_cost_usd", COST_LIMIT)
|
||||
cost = execution.get("max_cost_usd")
|
||||
if type(seconds) is not int or not 10 <= seconds <= MAX_JOB_SECONDS:
|
||||
raise Problem("invalid_timeout")
|
||||
if type(cost) not in (int, float) or not 0 < cost <= COST_LIMIT:
|
||||
if cost is not None and (type(cost) not in (int, float) or not 0 < cost < float("inf")):
|
||||
raise Problem("invalid_cost_limit")
|
||||
for key in ("campaign", "suite"):
|
||||
if type(raw[key]) is not str or not 1 <= len(raw[key]) <= 128:
|
||||
@ -192,7 +192,7 @@ def validate_request(raw, permissions):
|
||||
return {**raw, "routing": {"allow_external": external,
|
||||
"allowed_external_providers": providers},
|
||||
"execution": {"strategy": "whole_suite", "max_seconds": seconds,
|
||||
"max_cost_usd": float(cost)}}
|
||||
"max_cost_usd": float(cost) if cost is not None else None}}
|
||||
|
||||
|
||||
def prompt(request):
|
||||
|
||||
111
services/hermes/scripts/suite_hierarchical.py
Normal file
111
services/hermes/scripts/suite_hierarchical.py
Normal file
@ -0,0 +1,111 @@
|
||||
"""Global profile comparison followed by full-source family verification."""
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
|
||||
from suite_contract import Problem, encoded, validate_partition
|
||||
from suite_policy import BASE_NAME_LIMIT, invocation, validate_natural
|
||||
from suite_profiles import batches, profile_call, profile_source, size_metadata, validate_profiles
|
||||
|
||||
STRATEGY_REVISION = "suite-adaptive-selection-v1-20260930"
|
||||
MAX_SOURCE_BATCHES = 16
|
||||
MAX_LOGICAL_CALLS = 64
|
||||
FALLBACK_ERRORS = {"pass_generation_retries_exhausted", "pass_turn_limit_exhausted",
|
||||
"pass_output_limit", "pass_context_limit", "pass_capacity", "pass_request_too_large",
|
||||
"pass_provider_transient_failure", "pass_timeout"}
|
||||
|
||||
|
||||
def selection(source):
|
||||
"""Use content-free deterministic risk thresholds before any paid inference."""
|
||||
sizes = size_metadata(source)
|
||||
reasons = []
|
||||
if len(source["cases"]) >= 250:
|
||||
reasons.append("case_count_at_least_250")
|
||||
if sizes["source_bytes"] >= 256 * 1024:
|
||||
reasons.append("source_bytes_at_least_262144")
|
||||
if any(sizes[f]["p95"] >= 8192 for f in ("description", "preconditions", "success_criteria")):
|
||||
reasons.append("p95_field_bytes_at_least_8192")
|
||||
return {"execution_mode": "hierarchical" if reasons else "direct", "strategy_reasons": reasons,
|
||||
"strategy_revision": STRATEGY_REVISION, "source_size_metadata": sizes}
|
||||
|
||||
|
||||
def minimal_partition(value):
|
||||
"""Keep exact memberships; the complete profiles supply mechanism evidence."""
|
||||
return {"groups": [{"members": g["members"]} for g in value["groups"]]}
|
||||
|
||||
|
||||
def source_batches(source, partition):
|
||||
"""Reopen intact candidate families, keeping all member source fields."""
|
||||
by_alias = {c["alias"]: c for c in source["cases"]}
|
||||
packed, current = [], []
|
||||
for group in sorted(partition["groups"], key=lambda g: sorted(g["members"])):
|
||||
candidate = current + [group]
|
||||
records = [by_alias[a] for g in candidate for a in g["members"]]
|
||||
if current and (len(records) > 80 or len(encoded(records)) > 96 * 1024):
|
||||
packed.append(current)
|
||||
current = []
|
||||
current.append(group)
|
||||
if current:
|
||||
packed.append(current)
|
||||
if len(packed) > MAX_SOURCE_BATCHES:
|
||||
raise Problem("hierarchical_source_capacity", 422, batch_count=len(packed))
|
||||
result = []
|
||||
for groups in packed:
|
||||
aliases = {a for g in groups for a in g["members"]}
|
||||
subset = {**source, "cases": [c for c in source["cases"] if c["alias"] in aliases]}
|
||||
result.append((subset, {"candidate_partition": {"groups": groups}}))
|
||||
return result
|
||||
|
||||
|
||||
def validate_source(value, source, context):
|
||||
"""Require original evidence and forbid accidental cross-candidate review drift."""
|
||||
validate_natural(value, source)
|
||||
originals = [set(g["members"]) for g in context["candidate_partition"]["groups"]]
|
||||
if any(sum(set(g["members"]) <= members for members in originals) != 1 for g in value["groups"]):
|
||||
raise Problem("source_review_crossed_family_boundary", 502)
|
||||
|
||||
|
||||
def execute(workflow, source, capacity):
|
||||
"""Independent global perspectives never inherit profile-processing boundaries."""
|
||||
workflow.execution_mode = "hierarchical"
|
||||
workflow.max_calls = MAX_LOGICAL_CALLS
|
||||
pieces = batches(source)
|
||||
workflow.profile_batch_count = len(pieces)
|
||||
calls = [profile_call(piece, i) for i, piece in enumerate(pieces, 1)]
|
||||
for piece, call in zip(pieces, calls):
|
||||
capacity(call, workflow.provider, len(piece["cases"]))
|
||||
profiles = {}
|
||||
source_floor = 60 * max(1, math.ceil(len(source["cases"])/80),
|
||||
math.ceil(len(encoded(source["cases"]))/98304))
|
||||
for i, (piece, call) in enumerate(zip(pieces, calls)):
|
||||
workflow.future_seconds = (len(pieces)-i-1)*30 + 5*workflow.global_floor + source_floor
|
||||
value = workflow.call("implementation_profiles", piece, custom_call=call,
|
||||
validator=lambda v, p=piece: validate_profiles(v, p))
|
||||
profiles.update(value["profiles"])
|
||||
compact = profile_source(source, profiles)
|
||||
# Each proposal sees the entire profile set in an independent order. No
|
||||
# processing batch partition or prior proposal is supplied to proposal B.
|
||||
workflow.future_seconds = 4*workflow.global_floor + source_floor
|
||||
a = workflow.call("profile_proposal_a", compact)
|
||||
from suite_multipass import ordered_request
|
||||
workflow.future_seconds = 3*workflow.global_floor + source_floor
|
||||
b = workflow.call("profile_proposal_b", ordered_request(compact, True))
|
||||
workflow.future_seconds = 2*workflow.global_floor + source_floor
|
||||
provisional = workflow.call("profile_reconciliation", compact,
|
||||
{"proposal_a": minimal_partition(a), "proposal_b": minimal_partition(b)})
|
||||
workflow.cross_batch_review_count += 1
|
||||
checks = source_batches(source, provisional)
|
||||
# Every full-source verification is admitted before any such call launches.
|
||||
for piece, context in checks:
|
||||
capacity(invocation("source_review", piece, context), workflow.provider,
|
||||
len(piece["cases"]), len(context["candidate_partition"]["groups"]))
|
||||
confirmed = {"groups": []}
|
||||
for i, (piece, context) in enumerate(checks):
|
||||
workflow.future_seconds = 2*workflow.global_floor + (len(checks)-i-1)*60
|
||||
value = workflow.call("source_review", piece, context,
|
||||
len(context["candidate_partition"]["groups"]),
|
||||
validator=lambda v, p=piece, c=context: validate_source(v, p, c))
|
||||
confirmed["groups"].extend(value["groups"])
|
||||
validate_natural(confirmed, source)
|
||||
workflow.source_review_batch_count = len(checks)
|
||||
return a, b, confirmed
|
||||
@ -143,7 +143,9 @@ class Jobs:
|
||||
with self.lock:
|
||||
self._save(job_id, document)
|
||||
result, metadata = suite_multipass.generate(effective, selected, event, client_ip, progress,
|
||||
started=started, deadline=deadline)
|
||||
started=started, deadline=deadline,
|
||||
credential_scope={"owner": owner, "routing": request["routing"],
|
||||
"scope_revision": "generalized-claude-v1"})
|
||||
review = metadata.pop("review_summary")
|
||||
document.update(metadata)
|
||||
try:
|
||||
|
||||
@ -13,9 +13,13 @@ from suite_contract import (EXECUTION_REVISION, MAX_BODY, MAX_RESULT, MODELS, PR
|
||||
from suite_policy import (BASE_NAME_LIMIT, MAX_GROUP, POLICY_REVISION, invocation,
|
||||
validate_natural)
|
||||
from suite_sizing import cap_families, review_summary
|
||||
from suite_review_batches import MAX_MODEL_CALLS, review_stage
|
||||
from suite_review_batches import review_stage
|
||||
from suite_recovery import Checkpoints, MAX_INVOCATIONS, allocate, recover
|
||||
from suite_hierarchical import MAX_LOGICAL_CALLS, FALLBACK_ERRORS, execute as hierarchical, selection
|
||||
from collections import Counter
|
||||
|
||||
MAX_PASSES = 5
|
||||
MAX_MODEL_CALLS = MAX_INVOCATIONS
|
||||
CAPACITY_REVISION = "suite-context-turns-v1-20260930"
|
||||
|
||||
|
||||
@ -36,7 +40,9 @@ def capacity(call, provider, count, family_count=None):
|
||||
input_bytes = len(call["input"].encode()) + len(call["system"].encode()) + len(encoded(call["schema"]))
|
||||
if input_bytes > MAX_BODY:
|
||||
raise Problem("pass_request_too_large", 413, review_pass=call["stage"], input_bytes=input_bytes)
|
||||
if family_count is None:
|
||||
if call["stage"] == "implementation_profiles":
|
||||
estimate = 1024 + 1400 * count
|
||||
elif family_count is None:
|
||||
estimate = 1024 + 128 * count
|
||||
else:
|
||||
estimate = 1024 + 48 * count + 384 * family_count
|
||||
@ -75,15 +81,27 @@ def preflight_workflow(request):
|
||||
if not model["enabled"]:
|
||||
reasons[provider] = "unverified_output_capacity"
|
||||
continue
|
||||
strategy = selection(request) if provider == "claude" else {"execution_mode": "direct"}
|
||||
try:
|
||||
initial = capacity(invocation("proposal_a", request), provider, count)
|
||||
# Even a minimum-size reconciliation must fit; its real proposals and
|
||||
# every subsequent request get checked again before any model call.
|
||||
capacity(invocation("reconciliation", request), provider, count, 1)
|
||||
if strategy["execution_mode"] == "direct":
|
||||
try:
|
||||
initial = capacity(invocation("proposal_a", request), provider, count)
|
||||
capacity(invocation("reconciliation", request), provider, count, 1)
|
||||
except Problem:
|
||||
if provider != "claude":
|
||||
raise
|
||||
strategy.update(execution_mode="hierarchical", strategy_reasons=["direct_capacity"])
|
||||
if strategy["execution_mode"] == "hierarchical":
|
||||
from suite_profiles import batches, profile_call
|
||||
pieces = batches(request)
|
||||
checked = [capacity(profile_call(piece, i), provider, len(piece["cases"]))
|
||||
for i, piece in enumerate(pieces, 1)]
|
||||
initial = checked[0]
|
||||
strategy["profile_batch_count"] = len(pieces)
|
||||
except Problem as exc:
|
||||
reasons[provider] = exc.code
|
||||
continue
|
||||
return {"provider": provider, **model, **initial, **reasoning_selection(request, provider),
|
||||
return {"provider": provider, **model, **initial, **strategy, **reasoning_selection(request, provider),
|
||||
"configuration_revision": REVISION,
|
||||
"prompt_revision": PROMPT_REVISION, "prompt_sha256": PROMPT_SHA256,
|
||||
"execution_revision": EXECUTION_REVISION, "policy_revision": POLICY_REVISION,
|
||||
@ -91,6 +109,8 @@ def preflight_workflow(request):
|
||||
"minimum_model_passes": 3, "maximum_model_passes": MAX_MODEL_CALLS,
|
||||
"logical_stages": MAX_PASSES, "automatic_review_batching": True,
|
||||
"max_final_group_cases": MAX_GROUP, "max_final_name_characters": 64,
|
||||
"cost_guard_enforced": False, "maximum_attempts_per_pass": 3,
|
||||
"checkpoint_storage": "active_job_memory", "checkpoint_restart_recovery": False,
|
||||
"later_pass_capacity_verified": False, "later_pass_checks": "before_each_invocation"}
|
||||
raise Problem("capacity_or_unsupported_backend", 422, candidates=reasons)
|
||||
|
||||
@ -114,130 +134,202 @@ def sum_usage(records):
|
||||
|
||||
|
||||
class Workflow:
|
||||
"""Keep content in memory and allow only one pinned provider across model passes."""
|
||||
"""One pinned provider, shared deadline, isolated attempts and scoped checkpoints."""
|
||||
|
||||
def __init__(self, request, selected, cancel, client_ip, progress=None, *, started=None, deadline=None):
|
||||
def __init__(self, request, selected, cancel, client_ip, progress=None, *, started=None,
|
||||
deadline=None, credential_scope=None):
|
||||
self.request, self.provider, self.cancel = request, selected["provider"], cancel
|
||||
self.reasoning = selected.get("reasoning", MODELS[self.provider]["reasoning"])
|
||||
self.client_ip, self.progress = client_ip, progress or (lambda _: None)
|
||||
self.started = time.monotonic() if started is None else started
|
||||
self.deadline = self.started + request["execution"]["max_seconds"] if deadline is None else deadline
|
||||
self.cost_limit, self.spent = request["execution"]["max_cost_usd"], 0.0
|
||||
self.records = []
|
||||
self.last_metadata = {}
|
||||
self.stage = "proposal_a"
|
||||
self.max_calls = MAX_PASSES
|
||||
self.spent, self.cost_known = 0.0, True
|
||||
self.records, self.attempts, self.last_metadata = [], [], {}
|
||||
self.stage, self.max_calls = "proposal_a", MAX_LOGICAL_CALLS
|
||||
self.execution_mode = selected.get("execution_mode", "direct")
|
||||
self.strategy_events = [{"mode": self.execution_mode, "reason": "preflight"}]
|
||||
self.retries = Counter()
|
||||
self.profile_batch_count = self.cross_batch_review_count = self.source_review_batch_count = 0
|
||||
self.future_seconds = 0
|
||||
self.global_floor = min(500, max(10, request["execution"]["max_seconds"] / 8))
|
||||
self.checkpoints = Checkpoints(request, self.provider, credential_scope, self.deadline)
|
||||
|
||||
def checkpoint(self):
|
||||
"""Fail closed on cancellation or exhaustion of the shared job budgets."""
|
||||
"""Cancellation and time are hard limits; cost is informational only."""
|
||||
if self.cancel.is_set():
|
||||
raise Problem("cancelled", 409)
|
||||
if time.monotonic() >= self.deadline:
|
||||
raise Problem("job_time_budget_exhausted", 504)
|
||||
if self.spent > self.cost_limit:
|
||||
raise Problem("job_cost_budget_exhausted", 502)
|
||||
if len(self.attempts) >= MAX_INVOCATIONS:
|
||||
raise Problem("model_pass_limit", 502)
|
||||
|
||||
def call(self, stage, source, context=None, family_count=None):
|
||||
"""Run one fresh complete-input invocation, retaining metadata before validation."""
|
||||
def safe_state(self):
|
||||
"""Never place cached outputs, profiles or rationales into status metadata."""
|
||||
return {"execution_mode": self.execution_mode, "strategy_events": self.strategy_events,
|
||||
"direct_pass_attempts": sum(a["execution_mode"] == "direct" for a in self.attempts),
|
||||
"retry_count": dict(self.retries), "checkpointed_passes": self.checkpoints.names(),
|
||||
"checkpoint_reused": list(self.checkpoints.reused),
|
||||
"profile_batch_count": self.profile_batch_count,
|
||||
"source_review_batch_count": self.source_review_batch_count,
|
||||
"cross_batch_review_count": self.cross_batch_review_count,
|
||||
"model_pass_count": len(self.attempts), "cost_guard_enforced": False,
|
||||
"cost_used_usd_estimate": round(self.spent, 8) if self.cost_known else None,
|
||||
"cost_limit_usd_estimate": None,
|
||||
"remaining_seconds": round(max(0, self.deadline-time.monotonic()), 3)}
|
||||
|
||||
def call(self, stage, source, context=None, family_count=None, *, custom_call=None, validator=None):
|
||||
"""Recover one logical pass without regenerating validated predecessors."""
|
||||
self.stage = stage
|
||||
self.checkpoint()
|
||||
if len(self.records) >= self.max_calls:
|
||||
raise Problem("model_pass_limit", 502)
|
||||
call = invocation(stage, source, context)
|
||||
# Keep the preflight choice across independent ordering and larger review prompts.
|
||||
call = custom_call or invocation(stage, source, context)
|
||||
call["reasoning"] = self.reasoning
|
||||
if context and context.get("review_batch"):
|
||||
call["review_batch"] = context["review_batch"]
|
||||
started = time.monotonic()
|
||||
def report(activity=None):
|
||||
now = time.monotonic()
|
||||
self.progress({"current_pass": stage, "completed_model_passes": len(self.records),
|
||||
"maximum_model_passes": self.max_calls, "passes": self.records,
|
||||
"maximum_model_passes": MAX_INVOCATIONS, "passes": list(self.records),
|
||||
"attempts": list(self.attempts), **self.safe_state(),
|
||||
"review_batch": (context or {}).get("review_batch"),
|
||||
"heartbeat_at": time.time(), "pass_elapsed_seconds": round(now - started, 1),
|
||||
"job_elapsed_seconds": round(now - self.started, 1),
|
||||
"job_remaining_seconds": round(max(0, self.deadline - now), 1),
|
||||
"cost_used_usd_estimate": round(self.spent, 8),
|
||||
"cost_limit_usd_estimate": self.cost_limit, **(activity or {})})
|
||||
"profile_batch_index": call.get("batch_index"),
|
||||
"heartbeat_at": time.time(), "pass_elapsed_seconds": round(now-started, 1),
|
||||
"job_elapsed_seconds": round(now-self.started, 1),
|
||||
"job_remaining_seconds": round(max(0, self.deadline-now), 1),
|
||||
**(activity or {})})
|
||||
report({"cli_running": False, "pass_state": "capacity_preflight"})
|
||||
limits = capacity(call, self.provider, len(source["cases"]), family_count)
|
||||
call["max_turns"] = limits["max_turns"]
|
||||
remaining_seconds = self.deadline - time.monotonic()
|
||||
remaining_cost = self.cost_limit - self.spent
|
||||
if remaining_cost <= 0:
|
||||
raise Problem("job_cost_budget_exhausted", 502)
|
||||
effective = {**source, "execution": {**source["execution"],
|
||||
"max_seconds": remaining_seconds, "max_cost_usd": remaining_cost}}
|
||||
if self.provider == "claude":
|
||||
value, metadata = suite_backends.claude_generate(effective, self.cancel, invocation=call,
|
||||
progress=report, job_deadline=self.deadline)
|
||||
elif self.provider == "local":
|
||||
value, metadata = suite_backends.local_generate(effective, self.cancel, self.client_ip, invocation=call)
|
||||
else:
|
||||
raise Problem("unsupported_backend", 422)
|
||||
cost = metadata.get("cost_usd_estimate") if self.provider == "claude" else 0.0
|
||||
record = {"stage": stage, "provider": self.provider, "model": MODELS[self.provider]["model"],
|
||||
"reasoning": self.reasoning,
|
||||
"wall_seconds": round(time.monotonic() - started, 3), **limits,
|
||||
"case_count": len(source["cases"]), "review_batch": (context or {}).get("review_batch"),
|
||||
"system_sha256": hashlib.sha256(call["system"].encode()).hexdigest(),
|
||||
"schema_sha256": digest(call["schema"]),
|
||||
"case_order_sha256": digest([c["alias"] for c in source["cases"]]),
|
||||
"allocated_seconds": remaining_seconds, "allocated_cost_usd": remaining_cost,
|
||||
"usage": metadata.get("usage"), "cli_turns": metadata.get("turns"),
|
||||
"duration_api_ms": metadata.get("duration_api_ms"),
|
||||
"cost_usd_estimate": cost, "cli_diagnostics": metadata.get("cli_diagnostics")}
|
||||
self.records.append(record)
|
||||
self.last_metadata = metadata
|
||||
if type(cost) not in (int, float) or not math.isfinite(cost) or cost < 0:
|
||||
raise Problem("budget_accounting_unavailable", 502)
|
||||
self.spent += cost
|
||||
def validate(value):
|
||||
if validator:
|
||||
return validator(value)
|
||||
if stage in {"proposal_a", "proposal_b", "profile_proposal_a", "profile_proposal_b", "profile_reconciliation"}:
|
||||
return validate_partition(value, source, name_limit=BASE_NAME_LIMIT, unique_names=False)
|
||||
originals = None
|
||||
if stage in {"large_family_review", "decision_audit"}:
|
||||
originals = context.get("original_partition", context.get("natural_partition"))["groups"]
|
||||
return validate_natural(value, source, originals)
|
||||
before = len(self.records)
|
||||
value = recover(self, call, source, limits, validate, report)
|
||||
if len(self.records) == before and self.attempts:
|
||||
# Only a successfully validated output becomes a completed model pass.
|
||||
successful = next((a for a in reversed(self.attempts) if a["stage"] == stage and a["status"] == "generated"), None)
|
||||
if successful:
|
||||
successful["status"] = "validated"
|
||||
self.records.append(dict(successful))
|
||||
report({"cli_running": False, "pass_state": "validated"})
|
||||
return value
|
||||
|
||||
def invoke(self, call, source, limits, attempt, report):
|
||||
"""Account for successful and failed attempts before returning any output."""
|
||||
self.checkpoint()
|
||||
report({"cli_running": False})
|
||||
return decode(value, call, source)
|
||||
seconds = allocate(self)
|
||||
effective = {**source, "execution": {**source["execution"], "max_seconds": seconds,
|
||||
"max_cost_usd": None}}
|
||||
started = time.monotonic()
|
||||
record = {"stage": call["stage"], "attempt": attempt, "execution_mode": self.execution_mode,
|
||||
"provider": self.provider, "model": MODELS[self.provider]["model"],
|
||||
"reasoning": self.reasoning, **limits, "case_count": len(source["cases"]),
|
||||
"review_batch": call.get("review_batch"), "profile_batch_index": call.get("batch_index"),
|
||||
"allocated_seconds": seconds, "reserved_future_seconds": self.future_seconds,
|
||||
"allocated_cost_usd": None, "system_sha256": hashlib.sha256(call["system"].encode()).hexdigest(),
|
||||
"schema_sha256": digest(call["schema"]),
|
||||
"case_order_sha256": digest([c["alias"] for c in source["cases"]])}
|
||||
metadata, error = {}, None
|
||||
try:
|
||||
if self.provider == "claude":
|
||||
value, metadata = suite_backends.claude_generate(effective, self.cancel, invocation=call,
|
||||
progress=report, job_deadline=self.deadline)
|
||||
elif self.provider == "local":
|
||||
value, metadata = suite_backends.local_generate(effective, self.cancel, self.client_ip, invocation=call)
|
||||
else:
|
||||
raise Problem("unsupported_backend", 422)
|
||||
record["status"] = "generated"
|
||||
except Problem as exc:
|
||||
error = exc
|
||||
metadata = {**exc.details, "cli_diagnostics": exc.details}
|
||||
record.update(status="failed", error_code=exc.code)
|
||||
finally:
|
||||
cost = metadata.get("cost_usd_estimate") if self.provider == "claude" else 0.0
|
||||
if type(cost) in (float, int) and math.isfinite(cost) and cost >= 0:
|
||||
self.spent += cost
|
||||
else:
|
||||
self.cost_known = False
|
||||
cost = None
|
||||
record.update(wall_seconds=round(time.monotonic()-started, 3), usage=metadata.get("usage"),
|
||||
cli_turns=metadata.get("turns"), duration_api_ms=metadata.get("duration_api_ms"),
|
||||
cost_usd_estimate=cost, cli_diagnostics=metadata.get("cli_diagnostics"))
|
||||
self.attempts.append(record)
|
||||
self.last_metadata = metadata
|
||||
if error:
|
||||
raise error
|
||||
self.checkpoint()
|
||||
try:
|
||||
return value if call["stage"] == "implementation_profiles" else decode(value, call, source)
|
||||
except Problem as exc:
|
||||
record.update(status="failed", error_code=exc.code)
|
||||
raise
|
||||
|
||||
def direct(self, source):
|
||||
"""Reserve a viable time floor for every subsequent required stage."""
|
||||
self.future_seconds = 4*self.global_floor
|
||||
a = self.call("proposal_a", source)
|
||||
self.future_seconds = 3*self.global_floor
|
||||
b = self.call("proposal_b", ordered_request(source, True))
|
||||
self.future_seconds = 2*self.global_floor
|
||||
c = self.call("reconciliation", source, {"proposal_a": a, "proposal_b": b},
|
||||
max(len(a["groups"]), len(b["groups"])))
|
||||
return a, b, c
|
||||
|
||||
def execute(self):
|
||||
"""Perform independent discovery, reconciliation, and bounded semantic reviews."""
|
||||
"""Choose direct or hierarchical analysis, then review and size once."""
|
||||
source = ordered_request(self.request)
|
||||
a = self.call("proposal_a", source)
|
||||
validate_partition(a, source, name_limit=BASE_NAME_LIMIT, unique_names=False)
|
||||
b = self.call("proposal_b", ordered_request(source, True))
|
||||
validate_partition(b, source, name_limit=BASE_NAME_LIMIT, unique_names=False)
|
||||
count = max(len(a["groups"]), len(b["groups"]))
|
||||
reconciled = self.call("reconciliation", source, {"proposal_a": a, "proposal_b": b}, count)
|
||||
validate_natural(reconciled, source)
|
||||
if self.execution_mode == "direct":
|
||||
try:
|
||||
a, b, reconciled = self.direct(source)
|
||||
except Problem as exc:
|
||||
if self.provider != "claude" or exc.code not in FALLBACK_ERRORS:
|
||||
raise
|
||||
self.strategy_events.append({"mode": "hierarchical", "reason": exc.code, "after_pass": self.stage})
|
||||
a, b, reconciled = hierarchical(self, source, capacity)
|
||||
else:
|
||||
a, b, reconciled = hierarchical(self, source, capacity)
|
||||
large = [g for g in reconciled["groups"] if len(g["members"]) > MAX_GROUP]
|
||||
reviewed = audited = None
|
||||
if large:
|
||||
self.future_seconds = self.global_floor
|
||||
reviewed = review_stage(self, "large_family_review", source, reconciled, None, capacity)
|
||||
self.future_seconds = 0
|
||||
audited = review_stage(self, "decision_audit", source, reconciled, reviewed, capacity)
|
||||
natural = audited or reconciled
|
||||
final, divisions = cap_families(natural, source)
|
||||
review = review_summary(a, b, reconciled, reviewed, audited, final, divisions)
|
||||
self.checkpoint()
|
||||
metadata = {**self.last_metadata, "passes": self.records,
|
||||
"reasoning": self.reasoning,
|
||||
"model_pass_count": len(self.records), "usage": sum_usage(self.records),
|
||||
"cli_diagnostics_scope": "last_model_pass",
|
||||
"duration_api_ms": sum(r["duration_api_ms"] for r in self.records) if all(type(r["duration_api_ms"]) in (int, float) for r in self.records) else None,
|
||||
"cost_usd_estimate": round(self.spent, 8),
|
||||
"turns": sum(r["cli_turns"] for r in self.records) if all(type(r["cli_turns"]) is int for r in self.records) else None,
|
||||
metadata = {**self.last_metadata, **self.safe_state(), "passes": self.records, "attempts": self.attempts,
|
||||
"reasoning": self.reasoning, "usage": sum_usage(self.attempts),
|
||||
"cli_diagnostics_scope": "last_model_attempt", "cost_usd_estimate": self.safe_state()["cost_used_usd_estimate"],
|
||||
"turns": sum(a["cli_turns"] for a in self.attempts) if all(type(a["cli_turns"]) is int for a in self.attempts) else None,
|
||||
"natural_family_count": len(natural["groups"]), "final_task_count": len(final["groups"]),
|
||||
"singleton_statistics": {k: v for k, v in review["counts"].items() if "singleton" in k},
|
||||
"singleton_statistics": {k:v for k,v in review["counts"].items() if "singleton" in k},
|
||||
"policy_revision": POLICY_REVISION, "review_summary": review}
|
||||
if len(encoded({"result": final, **metadata})) > MAX_RESULT - 16384:
|
||||
raise Problem("response_too_large", 502)
|
||||
return final, metadata
|
||||
|
||||
|
||||
def generate(request, selected, cancel, client_ip, progress=None, *, started=None, deadline=None):
|
||||
"""Never expose partial partitions as completion or broaden a failed route."""
|
||||
workflow = Workflow(request, selected, cancel, client_ip, progress, started=started, deadline=deadline)
|
||||
def generate(request, selected, cancel, client_ip, progress=None, *, started=None, deadline=None, credential_scope=None):
|
||||
"""No partial completion, provider fallback, or persistent source-derived cache."""
|
||||
workflow = Workflow(request, selected, cancel, client_ip, progress, started=started,
|
||||
deadline=deadline, credential_scope=credential_scope)
|
||||
try:
|
||||
return workflow.execute()
|
||||
except Problem as exc:
|
||||
details = {"failure_stage": "multi_pass_orchestration", **exc.details, "review_pass": workflow.stage,
|
||||
"completed_model_passes": len(workflow.records), "passes": workflow.records,
|
||||
"aggregate_usage": sum_usage(workflow.records + ([{"usage": exc.details["usage"]}] if isinstance(exc.details.get("usage"), dict) else [])),
|
||||
"aggregate_cost_usd_estimate": None if exc.code == "budget_accounting_unavailable" else round(workflow.spent, 8)}
|
||||
if exc.details.get("cost_usd_estimate") is not None:
|
||||
details["aggregate_cost_usd_estimate"] += exc.details["cost_usd_estimate"]
|
||||
details = {"failure_stage": "multi_pass_orchestration", **exc.details, **workflow.safe_state(),
|
||||
"review_pass": workflow.stage, "completed_model_passes": len(workflow.records),
|
||||
"passes": workflow.records, "attempts_metadata": workflow.attempts,
|
||||
"aggregate_usage": sum_usage(workflow.attempts),
|
||||
"aggregate_cost_usd_estimate": workflow.safe_state()["cost_used_usd_estimate"]}
|
||||
raise Problem(exc.code, exc.status, **details) from None
|
||||
finally:
|
||||
workflow.checkpoints.clear()
|
||||
|
||||
@ -98,10 +98,13 @@ REVIEW_SCHEMA["properties"]["decisions"] = {"type": "array", "minItems": 1, "ite
|
||||
|
||||
def invocation(stage, request, context=None):
|
||||
"""Build one full-input pass; independent discovery has no proposal context."""
|
||||
original_stage = stage
|
||||
stage = {"profile_proposal_a": "proposal_a", "profile_proposal_b": "proposal_b",
|
||||
"profile_reconciliation": "proposal_a", "source_review": "reconciliation"}.get(stage, stage)
|
||||
if stage not in STAGES:
|
||||
raise ValueError("unknown internal stage")
|
||||
discovery = stage in STAGES[:2]
|
||||
if discovery and context:
|
||||
if discovery and context and original_stage != "profile_reconciliation":
|
||||
raise ValueError("discovery must be independent")
|
||||
schema = DISCOVERY_SCHEMA if discovery else NATURAL_SCHEMA if stage == "reconciliation" else REVIEW_SCHEMA
|
||||
instruction = DISCOVERY if discovery else RECONCILE
|
||||
@ -109,6 +112,21 @@ def invocation(stage, request, context=None):
|
||||
instruction += "\n" + REVIEW
|
||||
if stage == "decision_audit":
|
||||
instruction += "\n" + REVIEW + "\n" + AUDIT
|
||||
if request.get("_profile_source"):
|
||||
instruction += ("\nThe supplied cases are compact implementation profiles, not verbatim source. "
|
||||
"Compare across ALL profiles and batches. Original fields will be reopened "
|
||||
"before accepting natural families. Retain uncertainty.")
|
||||
if original_stage == "profile_reconciliation":
|
||||
instruction += ("\nReconcile the two supplied independent profile partitions across every alias. "
|
||||
"Compare memberships and machinery, not votes or connected components.")
|
||||
if original_stage == "source_review":
|
||||
instruction = instruction.replace(
|
||||
"Independently reassess two complete candidate partitions using ALL original cases.",
|
||||
"Reassess the supplied candidate partition using ALL original cases in this review.")
|
||||
instruction += ("\nReopen ALL original fields for every candidate family supplied here. "
|
||||
"Reject compression-induced compatibility; split genuinely different machinery. "
|
||||
"Do not cross candidate-family boundaries in this source verification pass. "
|
||||
"Use original source evidence, not profile quotations.")
|
||||
source = {k: request[k] for k in ("campaign", "suite", "cases")}
|
||||
context = dict(context or {})
|
||||
if context.get("review_batch"):
|
||||
@ -123,7 +141,7 @@ def invocation(stage, request, context=None):
|
||||
keys = decision_keys(context) if stage in STAGES[3:] else {}
|
||||
if keys:
|
||||
context["review_keys"] = keys
|
||||
return {"stage": stage,
|
||||
return {"stage": original_stage,
|
||||
"schema": wire_schema(schema, request, context), "review_keys": keys,
|
||||
"input": encoded({"suite": source, "review_material": context}).decode(),
|
||||
"system": SYSTEM + "\n\nPASS INSTRUCTIONS:\n" + instruction + "\n\n" + WIRE_INSTRUCTIONS}
|
||||
|
||||
108
services/hermes/scripts/suite_profiles.py
Normal file
108
services/hermes/scripts/suite_profiles.py
Normal file
@ -0,0 +1,108 @@
|
||||
"""Compact implementation profiles with exact aliases and source-grounded evidence."""
|
||||
from __future__ import annotations
|
||||
|
||||
from suite_contract import SYSTEM, Problem, encoded
|
||||
from suite_policy import evidence
|
||||
|
||||
PROFILE_FIELDS = {"target": 64, "work": 192, "setup": 96, "stimulus": 96,
|
||||
"observations": 128, "machinery": 96, "variations": 96, "uncertainty": 96}
|
||||
PROFILE_BYTES = 1152
|
||||
BATCH_CASES = 24
|
||||
BATCH_BYTES = 48 * 1024
|
||||
MAX_PROFILE_BATCHES = 32
|
||||
PROFILE_INSTRUCTIONS = (
|
||||
"Extract a compact implementation profile for EVERY alias. These processing batches "
|
||||
"are NOT families. Work must capture primary success-criteria implications; setup "
|
||||
"preserves material preconditions; stimulus captures controls/actions; observations "
|
||||
"captures measurement/assertions; machinery identifies special test mechanisms; "
|
||||
"variations and uncertainty preserve qualitative distinctions and missing details. "
|
||||
"Use null for unknowns. State only supplied facts, no invented implementation. "
|
||||
"Keep profiles concise, under 1152 UTF-8 bytes each including evidence. Include one "
|
||||
"short exact evidence quote from success_criteria if present, otherwise description. "
|
||||
"Do not drop distinctions to force a merge. Output profiles keyed by exact alias."
|
||||
)
|
||||
|
||||
|
||||
def batches(source):
|
||||
"""Pack complete cases deterministically; never split or truncate one case."""
|
||||
result, current = [], []
|
||||
for case in sorted(source["cases"], key=lambda c: c["alias"]):
|
||||
if current and (len(current) >= BATCH_CASES or len(encoded(current + [case])) > BATCH_BYTES):
|
||||
result.append({**source, "cases": current})
|
||||
current = []
|
||||
current.append(case)
|
||||
if current:
|
||||
result.append({**source, "cases": current})
|
||||
if len(result) > MAX_PROFILE_BATCHES:
|
||||
raise Problem("profile_batch_capacity", 422, batch_count=len(result))
|
||||
return result
|
||||
|
||||
|
||||
def profile_call(source, index):
|
||||
"""Supply every generalized source field and require one result per alias."""
|
||||
item = {"type": "object", "additionalProperties": False,
|
||||
"required": list(PROFILE_FIELDS) + ["evidence"], "properties": {
|
||||
k: {"type": ["string", "null"], "maxLength": size} for k, size in PROFILE_FIELDS.items()}}
|
||||
item["properties"]["evidence"] = {"type": "object", "additionalProperties": False,
|
||||
"required": ["field", "quote"], "properties": {
|
||||
"field": {"enum": ["success_criteria", "description"]},
|
||||
"quote": {"type": "string", "minLength": 1, "maxLength": 96}}}
|
||||
aliases = sorted(c["alias"] for c in source["cases"])
|
||||
return {"stage": "implementation_profiles", "batch_index": index,
|
||||
"input": encoded({k: source[k] for k in ("campaign", "suite", "cases")}).decode(),
|
||||
"system": SYSTEM + "\n" + PROFILE_INSTRUCTIONS,
|
||||
"schema": {"type": "object", "additionalProperties": False, "required": ["profiles"],
|
||||
"properties": {"profiles": {"type": "object", "additionalProperties": False,
|
||||
"required": aliases, "properties": {a: item for a in aliases}}}}}
|
||||
|
||||
|
||||
def validate_profiles(value, source):
|
||||
"""Check profile coverage, size, nullable fields and exact source evidence."""
|
||||
by_alias = {c["alias"]: c for c in source["cases"]}
|
||||
if not isinstance(value, dict) or set(value) != {"profiles"} or not isinstance(value["profiles"], dict):
|
||||
raise Problem("invalid_profile_output", 502)
|
||||
if set(value["profiles"]) != set(by_alias):
|
||||
raise Problem("invalid_case_assignments", 502, failure_stage="profile_aliases")
|
||||
for alias, profile in value["profiles"].items():
|
||||
if not isinstance(profile, dict) or set(profile) != set(PROFILE_FIELDS) | {"evidence"}:
|
||||
raise Problem("invalid_profile_output", 502)
|
||||
for key, limit in PROFILE_FIELDS.items():
|
||||
text = profile[key]
|
||||
if text is not None and (type(text) is not str or len(text) > limit):
|
||||
raise Problem("invalid_profile_output", 502)
|
||||
item = profile["evidence"]
|
||||
if not isinstance(item, dict) or set(item) != {"field", "quote"}:
|
||||
raise Problem("invalid_profile_output", 502)
|
||||
expected = "success_criteria" if by_alias[alias].get("success_criteria") else "description"
|
||||
if item["field"] != expected:
|
||||
raise Problem("invalid_review_evidence", 502)
|
||||
evidence([{**item, "alias": alias}], {alias}, by_alias)
|
||||
if len(encoded(profile)) > PROFILE_BYTES:
|
||||
raise Problem("profile_capacity", 422)
|
||||
return value
|
||||
|
||||
|
||||
def profile_source(source, profiles):
|
||||
"""Use compact fields for global comparison; originals are reopened afterwards."""
|
||||
cases = []
|
||||
for alias in sorted(profiles):
|
||||
p = profiles[alias]
|
||||
def join(keys):
|
||||
return "; ".join(k + ": " + (p[k] or "unknown") for k in keys)
|
||||
cases.append({"alias": alias, "description": join(("target", "stimulus")),
|
||||
"success_criteria": join(("work", "observations", "machinery")),
|
||||
"preconditions": join(("setup",)),
|
||||
"operating_condition": join(("variations", "uncertainty"))})
|
||||
if any(len(encoded(c)) > 1280 for c in cases):
|
||||
raise Problem("profile_capacity", 422)
|
||||
return {**source, "cases": cases, "_profile_source": True}
|
||||
|
||||
|
||||
def size_metadata(source):
|
||||
"""Retain byte distributions only, allowing future representative fault tests."""
|
||||
result = {"source_bytes": len(encoded({k: source[k] for k in ("campaign", "suite", "cases")}))}
|
||||
for field in ("description", "preconditions", "case_type", "success_criteria"):
|
||||
values = sorted(len((c.get(field) or "").encode()) for c in source["cases"])
|
||||
result[field] = {"min": min(values), "max": max(values), "median": values[len(values)//2],
|
||||
"p95": values[int((len(values)-1)*.95)]}
|
||||
return result
|
||||
124
services/hermes/scripts/suite_recovery.py
Normal file
124
services/hermes/scripts/suite_recovery.py
Normal file
@ -0,0 +1,124 @@
|
||||
"""Bounded isolated retries and volatile, identity-bound validated checkpoints."""
|
||||
from __future__ import annotations
|
||||
|
||||
import copy
|
||||
import time
|
||||
from collections import Counter
|
||||
|
||||
from suite_contract import (EXECUTION_REVISION, MODELS, PROMPT_REVISION, REVISION,
|
||||
Problem, digest)
|
||||
from suite_policy import POLICY_REVISION
|
||||
|
||||
MAX_ATTEMPTS = 3
|
||||
MAX_INVOCATIONS = 128
|
||||
RETRIABLE = {
|
||||
"incomplete_generation", "pass_missing_terminal_event", "pass_turn_limit",
|
||||
"pass_schema_repair", "pass_provider_transient_failure", "rate_limit",
|
||||
"backend_unavailable", "backend_timeout_or_unavailable", "pass_timeout", "timeout",
|
||||
}
|
||||
REPAIRABLE = {"invalid_json_result", "invalid_case_assignments"}
|
||||
NONREPAIR_STAGES = {"runtime_model", "runtime_model_limits", "cli_initialization", "cli_usage"}
|
||||
|
||||
|
||||
class Checkpoints:
|
||||
"""Keep validated derived outputs only in this job's memory, until its deadline.
|
||||
|
||||
No disk persistence, cross-job reuse, restart recovery or status content access.
|
||||
Complete call/context hashes include relevant prior-pass outputs.
|
||||
"""
|
||||
|
||||
def __init__(self, request, provider, scope, deadline):
|
||||
self.identity = {"request": digest(request), "provider": provider,
|
||||
"model": MODELS[provider]["model"], "routing": request["routing"],
|
||||
"credential_scope": scope, "configuration": REVISION,
|
||||
"prompt": PROMPT_REVISION, "policy": POLICY_REVISION,
|
||||
"execution": EXECUTION_REVISION}
|
||||
self.deadline, self.values, self.reused = deadline, {}, []
|
||||
|
||||
def key(self, call):
|
||||
"""Bind the whole source, schema, objective and prior results to each pass."""
|
||||
return digest({**self.identity, "call": call})
|
||||
|
||||
def get(self, key, stage):
|
||||
"""Return an isolated copy; expired checkpoints can never be restored."""
|
||||
if time.monotonic() >= self.deadline:
|
||||
self.values.clear()
|
||||
raise Problem("job_time_budget_exhausted", 504)
|
||||
if key not in self.values:
|
||||
return None
|
||||
self.reused.append(stage)
|
||||
return copy.deepcopy(self.values[key][1])
|
||||
|
||||
def put(self, key, stage, value):
|
||||
self.values[key] = (stage, copy.deepcopy(value))
|
||||
|
||||
def names(self):
|
||||
return sorted({stage for stage, _ in self.values.values()})
|
||||
|
||||
def clear(self):
|
||||
self.values.clear()
|
||||
|
||||
|
||||
def recover(workflow, call, source, limits, validate, report):
|
||||
"""Retry only eligible failures; every attempt is fresh and fully accounted."""
|
||||
stage = call["stage"]
|
||||
key = workflow.checkpoints.key(call)
|
||||
cached = workflow.checkpoints.get(key, stage)
|
||||
if cached is not None:
|
||||
validate(cached)
|
||||
return cached
|
||||
repair_used = False
|
||||
for attempt in range(1, MAX_ATTEMPTS + 1):
|
||||
workflow.checkpoint()
|
||||
turns = min(limits["max_turns"], 3 + attempt)
|
||||
actual = {**call, "max_turns": turns}
|
||||
if attempt > 1:
|
||||
# No prior generated content is fed back; only the unchanged objective
|
||||
# and complete source enter a new isolated CLI session.
|
||||
report({"retry_attempt": attempt, "retry_reason": previous.code, "cli_running": False})
|
||||
if workflow.cancel.wait(min(10, 2 ** (attempt - 1))):
|
||||
raise Problem("cancelled", 409)
|
||||
try:
|
||||
value = workflow.invoke(actual, source, {**limits, "max_turns": turns}, attempt, report)
|
||||
validate(value)
|
||||
except Problem as exc:
|
||||
previous = exc
|
||||
if workflow.attempts and workflow.attempts[-1]["stage"] == stage:
|
||||
workflow.attempts[-1].update(status="failed", error_code=exc.code)
|
||||
can_repair = (exc.code in REPAIRABLE and not repair_used
|
||||
and exc.details.get("failure_stage") not in NONREPAIR_STAGES)
|
||||
retry = exc.code in RETRIABLE or can_repair
|
||||
if not retry:
|
||||
raise
|
||||
if can_repair:
|
||||
repair_used = True
|
||||
if exc.code == "pass_turn_limit" and turns >= limits["max_turns"]:
|
||||
code = "pass_turn_limit_exhausted"
|
||||
break
|
||||
if attempt == MAX_ATTEMPTS:
|
||||
code = ("pass_turn_limit_exhausted" if exc.code == "pass_turn_limit" else
|
||||
"pass_provider_transient_failure" if exc.code in {
|
||||
"rate_limit", "backend_unavailable", "backend_timeout_or_unavailable",
|
||||
"pass_provider_transient_failure"} else
|
||||
"pass_timeout" if exc.code in {"timeout", "pass_timeout"} else
|
||||
"pass_generation_retries_exhausted")
|
||||
break
|
||||
workflow.retries[stage] += 1
|
||||
continue
|
||||
workflow.checkpoints.put(key, stage, value)
|
||||
report({"cli_running": False, "pass_state": "validated"})
|
||||
return value
|
||||
raise Problem(code, 502, **{**previous.details, "attempts": attempt,
|
||||
"last_error_code": previous.code, "review_pass": stage}) from None
|
||||
|
||||
|
||||
def allocate(workflow):
|
||||
"""Protect explicit later-pass time floors within the single absolute deadline."""
|
||||
remaining = workflow.deadline - time.monotonic()
|
||||
reserved = workflow.future_seconds
|
||||
if remaining <= 0:
|
||||
raise Problem("job_time_budget_exhausted", 504)
|
||||
if remaining - reserved < 1:
|
||||
raise Problem("insufficient_remaining_pass_budget", 422,
|
||||
remaining_seconds=round(remaining, 3), reserved_future_seconds=reserved)
|
||||
return remaining - reserved
|
||||
@ -3,7 +3,7 @@ from suite_contract import Problem
|
||||
from suite_policy import MAX_GROUP, invocation, validate_natural
|
||||
|
||||
MAX_REVIEW_BATCHES = 8
|
||||
MAX_MODEL_CALLS = 3 + 2 * MAX_REVIEW_BATCHES
|
||||
MAX_MODEL_CALLS = 64
|
||||
|
||||
|
||||
def material(stage, original, reviewed=None):
|
||||
@ -75,15 +75,18 @@ def review_stage(workflow, stage, source, original, reviewed, capacity):
|
||||
# Admit every batch before launching any paid review call.
|
||||
for subset, context, count in calls:
|
||||
capacity(invocation(stage, subset, context), workflow.provider, len(subset["cases"]), count)
|
||||
workflow.max_calls += len(calls) - 1
|
||||
workflow.max_calls = MAX_MODEL_CALLS
|
||||
if workflow.max_calls > MAX_MODEL_CALLS:
|
||||
raise Problem("model_pass_limit", 502)
|
||||
combined = {"groups": [g for g in original["groups"] if len(g["members"]) <= MAX_GROUP], "decisions": []}
|
||||
for originals, (subset, context, count) in zip(batches, calls):
|
||||
base_reserve = workflow.future_seconds
|
||||
for index, (originals, (subset, context, count)) in enumerate(zip(batches, calls)):
|
||||
workflow.future_seconds = base_reserve + (len(calls)-index-1)*60
|
||||
value = workflow.call(stage, subset, context, count)
|
||||
validate_natural(value, subset, originals)
|
||||
combined["groups"].extend(value["groups"])
|
||||
combined["decisions"].extend(value["decisions"])
|
||||
workflow.future_seconds = base_reserve
|
||||
# Recheck global naming, membership, boundaries and all original decisions.
|
||||
validate_natural(combined, source, original["groups"])
|
||||
return combined
|
||||
|
||||
@ -36,7 +36,7 @@ spec:
|
||||
app: hermes-suite-planner
|
||||
annotations:
|
||||
fluentbit.io/exclude: "true"
|
||||
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v10-20260930
|
||||
ai.bstein.dev/config-rev: suite-v6-adaptive-v11-20260930
|
||||
vault.hashicorp.com/agent-inject: "true"
|
||||
vault.hashicorp.com/agent-pre-populate-only: "true"
|
||||
vault.hashicorp.com/agent-init-first: "true"
|
||||
|
||||
@ -114,8 +114,8 @@ def test_automatic_review_batches_preserve_global_discovery_and_full_validation(
|
||||
assert calls[3]["aliases"].isdisjoint(calls[4]["aliases"])
|
||||
assert calls[3]["aliases"] == calls[5]["aliases"] and calls[4]["aliases"] == calls[6]["aliases"]
|
||||
assert len({c["deadline"] for c in calls}) == 1
|
||||
assert calls[-1]["execution"]["max_cost_usd"] == pytest.approx(29.4)
|
||||
assert updates[-1]["maximum_model_passes"] == 7
|
||||
assert all(c['execution']['max_cost_usd'] is None for c in calls)
|
||||
assert updates[-1]["maximum_model_passes"] == 128
|
||||
assert metadata["model_pass_count"] == 7 and len(result["groups"]) == 4
|
||||
validate_result(result, source)
|
||||
|
||||
|
||||
@ -42,9 +42,9 @@ def request():
|
||||
|
||||
|
||||
@pytest.mark.parametrize("subtype,code", [
|
||||
("error_max_turns", "incomplete_generation"),
|
||||
("error_max_structured_output_retries", "incomplete_generation"),
|
||||
("error_max_budget_usd", "job_cost_budget_exhausted"),
|
||||
("error_max_turns", "pass_turn_limit"),
|
||||
("error_max_structured_output_retries", "pass_schema_repair"),
|
||||
("error_max_budget_usd", "unexpected_cli_budget_limit"),
|
||||
("error_during_execution", "incomplete_generation"),
|
||||
(CANARY, "incomplete_generation"),
|
||||
])
|
||||
@ -75,8 +75,8 @@ def test_provider_error_status_is_preserved(status, code):
|
||||
("authentication_failed", "provider_authentication"),
|
||||
("oauth_org_not_allowed", "provider_authentication"),
|
||||
("rate_limit", "rate_limit"), ("server_error", "backend_unavailable"),
|
||||
("overloaded", "backend_unavailable"), ("unknown", "incomplete_generation"),
|
||||
(CANARY, "incomplete_generation"),
|
||||
("overloaded", "backend_unavailable"), ("unknown", "pass_provider_transient_failure"),
|
||||
(CANARY, "pass_provider_transient_failure"),
|
||||
])
|
||||
def test_error_flag_on_success_subtype_uses_safe_assistant_category(category, code):
|
||||
"""A result subtype alone does not mean the CLI completed successfully."""
|
||||
@ -101,7 +101,7 @@ def test_error_flag_on_success_subtype_uses_safe_assistant_category(category, co
|
||||
True, "connection_refused", "backend_unavailable"),
|
||||
("API Error: Request timed out.", True, "request_timeout", "backend_timeout_or_unavailable"),
|
||||
("API Error: Connection error.", False, None, "incomplete_generation"),
|
||||
("API Error: Connection error. " + CANARY, True, None, "incomplete_generation"),
|
||||
("API Error: Connection error. " + CANARY, True, None, "pass_provider_transient_failure"),
|
||||
])
|
||||
def test_transport_signatures_require_exact_cli_error_marker(text, marked, transport, code):
|
||||
"""Provider text never becomes a diagnostic string or a substring classifier."""
|
||||
@ -121,7 +121,7 @@ def test_reasoning_effort_reports_actual_configuration_or_unknown(effort):
|
||||
|
||||
@pytest.mark.parametrize("reason", ["max_tokens", "model_context_window_exceeded"])
|
||||
def test_explicit_provider_stop_distinct_from_turn_limit(reason):
|
||||
with pytest.raises(Problem, match="incomplete_generation") as raised:
|
||||
with pytest.raises(Problem, match="pass_output_limit|pass_context_limit") as raised:
|
||||
suite_backends.parse_claude(envelope(stop_reason=reason), MODEL, exit_code=0)
|
||||
assert raised.value.details["failure_stage"] == "provider_stop"
|
||||
assert raised.value.details["provider_stop_reason"] == reason
|
||||
@ -129,8 +129,8 @@ def test_explicit_provider_stop_distinct_from_turn_limit(reason):
|
||||
|
||||
|
||||
@pytest.mark.parametrize("raw,stage,code", [
|
||||
("", "missing_init_or_final_event", "incomplete_generation"),
|
||||
(json.dumps({"type": "system", "subtype": "init", "model": MODEL}), "missing_init_or_final_event", "incomplete_generation"),
|
||||
("", "missing_init_or_final_event", "pass_missing_terminal_event"),
|
||||
(json.dumps({"type": "system", "subtype": "init", "model": MODEL}), "missing_init_or_final_event", "pass_missing_terminal_event"),
|
||||
("not json " + CANARY, "cli_event_json", "invalid_json_result"),
|
||||
("[]", "cli_event_json", "invalid_json_result"),
|
||||
])
|
||||
@ -147,13 +147,10 @@ def test_text_or_tool_candidate_is_detected_but_not_silently_accepted():
|
||||
{"type": "tool_use", "name": "StructuredOutput", "input": value},
|
||||
{"type": "text", "text": json.dumps(value)}]}}
|
||||
raw = json.dumps(assistant) + "\n" + envelope(structured_output=None, result=json.dumps(value))
|
||||
with pytest.raises(Problem, match="invalid_json_result") as raised:
|
||||
suite_backends.parse_claude(raw, MODEL, exit_code=0)
|
||||
details = raised.value.details
|
||||
assert details["failure_stage"] == "structured_output_extraction"
|
||||
assert details["assistant_structured_tool_input_present"] is True
|
||||
assert details["assistant_json_text_present"] is True
|
||||
assert details["final_json_text_present"] is True
|
||||
result, metadata = suite_backends.parse_claude(raw, MODEL, exit_code=0)
|
||||
assert result == value
|
||||
details = metadata['cli_diagnostics']
|
||||
assert details['structured_output_location'] == 'result.result_json'
|
||||
assert CANARY not in json.dumps(details)
|
||||
|
||||
|
||||
@ -207,7 +204,7 @@ def test_validation_failure_retains_usage_without_accepting_content(tmp_path, mo
|
||||
("success", None, None), ("nonzero", "incomplete_generation", "process_exit"),
|
||||
("signal", "incomplete_generation", "process_exit"),
|
||||
("empty", "backend_unavailable", "process_exit"),
|
||||
("timeout", "timeout", "subprocess"), ("cancel", "cancelled", "subprocess"),
|
||||
("timeout", "pass_timeout", "subprocess"), ("cancel", "cancelled", "subprocess"),
|
||||
("job_timeout", "job_time_budget_exhausted", "subprocess"),
|
||||
("oversize", "response_too_large", "subprocess"),
|
||||
("start", "backend_unavailable", "process_start"),
|
||||
@ -273,7 +270,7 @@ def test_failed_cli_usage_survives_job_cleanup_without_content(tmp_path, monkeyp
|
||||
jobs.run(document["job_id"], "owner", value, selected, "192.168.22.8")
|
||||
failed = jobs.get(document["job_id"], "owner")
|
||||
assert failed["error"]["details"]["turn_limit_reached"] is True
|
||||
assert failed["usage"]["output_tokens"] == 50
|
||||
assert failed["usage"]["output_tokens"] == 50 * len(failed["error"]["details"]["attempts_metadata"])
|
||||
assert not jobs.results and not jobs.active
|
||||
replay, created = jobs.submit("owner", "synthetic-failure", value, selected, "192.168.22.8", launch=False)
|
||||
assert not created and replay["job_id"] == document["job_id"]
|
||||
|
||||
@ -8,11 +8,12 @@ import pytest
|
||||
import suite_backends
|
||||
import suite_jobs
|
||||
import suite_multipass
|
||||
import suite_recovery
|
||||
from suite_contract import MAX_JOB_SECONDS, Problem, TIMEOUT, preflight, validate_request
|
||||
from test_suite_multipass import install_backend, natural, request
|
||||
|
||||
|
||||
@pytest.mark.parametrize("seconds", [1801, 2700, 3600])
|
||||
@pytest.mark.parametrize("seconds", [1801, 2700, 3600, 7200])
|
||||
def test_explicit_long_budget_matches_published_schema(seconds):
|
||||
"""Admission accepts opt-in durations; omitted limits retain their old default."""
|
||||
source = request(7)
|
||||
@ -23,10 +24,10 @@ def test_explicit_long_budget_matches_published_schema(seconds):
|
||||
schema = json.loads((Path(__file__).resolve().parents[2] /
|
||||
"docs/contracts/suite_planning_request.schema.json").read_text())
|
||||
limit = schema["properties"]["execution"]["properties"]["max_seconds"]
|
||||
assert limit["maximum"] == MAX_JOB_SECONDS == 3600 and limit["default"] == TIMEOUT
|
||||
assert limit["maximum"] == MAX_JOB_SECONDS == 7200 and limit["default"] == TIMEOUT
|
||||
|
||||
|
||||
@pytest.mark.parametrize("seconds", [3601, 3600.1, True, "3600", 9])
|
||||
@pytest.mark.parametrize("seconds", [7201, 7200.1, True, "7200", 9])
|
||||
def test_invalid_budget_fails_before_inference(seconds):
|
||||
source = request(7)
|
||||
source["execution"]["max_seconds"] = seconds
|
||||
@ -62,7 +63,7 @@ def test_job_passes_1800_seconds_with_one_absolute_deadline(tmp_path, monkeypatc
|
||||
jobs.run(job["job_id"], "owner", source, selected, "192.168.22.8")
|
||||
result = jobs.get(job["job_id"], "owner")
|
||||
assert result["status"] == "completed" and result["wall_seconds"] == 2510
|
||||
assert allocations == [3590, 3090, 2590, 2090, 1590]
|
||||
assert allocations == [1795, 1743.75, 1692.5, 1641.25, 1590]
|
||||
assert deadlines == [3700] * 5
|
||||
assert result["execution_progress"]["job_remaining_seconds"] == 1090
|
||||
assert result["execution_progress"]["job_elapsed_seconds"] == 2510
|
||||
@ -76,7 +77,9 @@ def test_distinct_backend_timeout_is_not_reclassified(monkeypatch, provider_erro
|
||||
def fail(*_, **__):
|
||||
raise Problem(provider_error, 504, failure_stage="subprocess")
|
||||
monkeypatch.setattr(suite_backends, "claude_generate", fail)
|
||||
with pytest.raises(Problem, match="^" + provider_error + "$"):
|
||||
monkeypatch.setattr(suite_multipass, "FALLBACK_ERRORS", set())
|
||||
monkeypatch.setattr(suite_recovery, "MAX_ATTEMPTS", 1)
|
||||
with pytest.raises(Problem, match="^(pass_provider_transient_failure|pass_timeout)$"):
|
||||
suite_multipass.generate(source, preflight(source), threading.Event(), "192.168.22.8")
|
||||
|
||||
|
||||
@ -104,7 +107,7 @@ def test_watchdog_job_expiry_reaches_status_with_zero_remaining(tmp_path, monkey
|
||||
monkeypatch.setattr(suite_jobs.time, "monotonic", lambda: clock[0])
|
||||
monkeypatch.setattr(suite_backends, "switchyard_decision", lambda _: None)
|
||||
def expired(value, cancel, *, invocation, progress, job_deadline):
|
||||
assert value["execution"]["max_seconds"] == 3600
|
||||
assert value["execution"]["max_seconds"] == 1800
|
||||
clock[0] = job_deadline
|
||||
raise Problem("job_time_budget_exhausted", 504, failure_stage="subprocess")
|
||||
monkeypatch.setattr(suite_backends, "claude_generate", expired)
|
||||
|
||||
@ -119,8 +119,8 @@ def test_coherent_families_are_balanced_not_semantically_fragmented(monkeypatch,
|
||||
assert [g['name'] for g in result['groups']] == [f'Reset recovery ({i}/{len(sizes)})' for i in range(1,len(sizes)+1)]
|
||||
assert all('Work-size part' in g['description'] for g in result['groups'])
|
||||
assert metadata['review_summary']['decision_audit'][0]['decision'] == 'keep'
|
||||
assert [call[0]['execution']['max_cost_usd'] for call in calls] == pytest.approx([30,29.9,29.8,29.7,29.6])
|
||||
assert all(calls[i+1][0]['execution']['max_seconds'] < calls[i][0]['execution']['max_seconds'] for i in range(4))
|
||||
assert [call[0]['execution']['max_cost_usd'] for call in calls] == [None]*5
|
||||
assert all(0 < c[0]['execution']['max_seconds'] <= source['execution']['max_seconds'] for c in calls)
|
||||
validate_result(result, source)
|
||||
|
||||
|
||||
@ -163,7 +163,7 @@ def test_progress_reports_activity_without_content_or_false_percentage(monkeypat
|
||||
active = [value for value in updates if value.get('cli_running')]
|
||||
assert len(active) == 5
|
||||
assert [v['completed_model_passes'] for v in active] == [0,1,2,3,4]
|
||||
assert all(v['maximum_model_passes'] == 5 and v['cli_output_bytes'] == 120 for v in active)
|
||||
assert all(v['maximum_model_passes'] == 128 and v['cli_output_bytes'] == 120 for v in active)
|
||||
assert all(v['heartbeat_at'] > 0 and v['job_remaining_seconds'] <= 1800 for v in active)
|
||||
assert all(v['last_cli_activity_seconds_ago'] == 0 for v in active)
|
||||
assert updates[-1]['cli_running'] is False
|
||||
@ -177,19 +177,18 @@ def test_one_hour_opt_in_retains_thirty_minute_default():
|
||||
assert validate_request(source, ['claude'])['execution']['max_seconds'] == 900
|
||||
source['execution']['max_seconds'] = 3600
|
||||
assert validate_request(source, ['claude'])['execution']['max_seconds'] == 3600
|
||||
source['execution']['max_seconds'] = 3601
|
||||
source['execution']['max_seconds'] = 7201
|
||||
with pytest.raises(Problem, match='invalid_timeout'):
|
||||
validate_request(source, ['claude'])
|
||||
|
||||
|
||||
def test_cli_estimate_guard_default_and_client_override():
|
||||
source = request(7)
|
||||
assert source['execution']['max_cost_usd'] == 30
|
||||
assert source['execution']['max_cost_usd'] is None
|
||||
source['execution']['max_cost_usd'] = 5
|
||||
assert validate_request(source, ['claude'])['execution']['max_cost_usd'] == 5
|
||||
source['execution']['max_cost_usd'] = 30.1
|
||||
with pytest.raises(Problem, match='invalid_cost_limit'):
|
||||
validate_request(source, ['claude'])
|
||||
source['execution']['max_cost_usd'] = 100000
|
||||
assert validate_request(source, ['claude'])['execution']['max_cost_usd'] == 100000
|
||||
|
||||
|
||||
@pytest.mark.parametrize('size', [14,75,363])
|
||||
@ -343,7 +342,7 @@ def test_existing_suite_sizes_have_separate_natural_and_task_counts(monkeypatch,
|
||||
for alias,family in expected.items(): buckets.setdefault(family,[]).append(alias)
|
||||
families = natural(source,list(buckets.values()),list(buckets))
|
||||
install_backend(monkeypatch,source,families)
|
||||
result, metadata = workflow.generate(source,preflight(source),threading.Event(),'192.168.22.8')
|
||||
result, metadata = workflow.generate(source,{**preflight(source), 'execution_mode':'direct'},threading.Event(),'192.168.22.8')
|
||||
assert metadata['natural_family_count'] == 9
|
||||
assert metadata['final_task_count'] == sum(len(balanced_sizes(len(p))) for p in buckets.values())
|
||||
assert all(1 <= len(g['members']) <= 5 for g in result['groups'])
|
||||
@ -351,16 +350,15 @@ def test_existing_suite_sizes_have_separate_natural_and_task_counts(monkeypatch,
|
||||
assert len(encoded({'result':result,**metadata})) < 1<<20
|
||||
|
||||
|
||||
def test_whole_job_cost_budget_and_failure_no_partial_answer(monkeypatch):
|
||||
def test_cost_estimates_never_stop_a_job(monkeypatch):
|
||||
source = request(14)
|
||||
source['execution']['max_cost_usd'] = 5
|
||||
family = natural(source,[[c['alias'] for c in source['cases']]])
|
||||
calls = install_backend(monkeypatch,source,family,costs=[1,2,3])
|
||||
with pytest.raises(Problem,match='job_cost_budget_exhausted') as raised:
|
||||
workflow.generate(source,preflight(source),threading.Event(),'192.168.22.8')
|
||||
assert [c[0]['execution']['max_cost_usd'] for c in calls] == [5,4,2]
|
||||
assert raised.value.details['completed_model_passes'] == 3
|
||||
assert CANARY not in json.dumps(raised.value.document())
|
||||
calls = install_backend(monkeypatch,source,family,costs=[10]*5)
|
||||
result, metadata = workflow.generate(source,preflight(source),threading.Event(),'192.168.22.8')
|
||||
assert metadata['cost_usd_estimate'] == 50
|
||||
assert all(c[0]['execution']['max_cost_usd'] is None for c in calls)
|
||||
validate_result(result,source)
|
||||
|
||||
|
||||
def test_expanded_reconciliation_capacity_checked_before_launch(monkeypatch):
|
||||
@ -374,6 +372,7 @@ def test_expanded_reconciliation_capacity_checked_before_launch(monkeypatch):
|
||||
call['input'] += 'x'*(1<<20)
|
||||
return original(call,*args)
|
||||
monkeypatch.setattr(workflow,'capacity',capacity)
|
||||
monkeypatch.setattr(workflow,'FALLBACK_ERRORS',set())
|
||||
with pytest.raises(Problem,match='pass_request_too_large'):
|
||||
workflow.generate(source,selected,threading.Event(),'192.168.22.8')
|
||||
assert len(calls) == 2
|
||||
@ -423,19 +422,18 @@ def test_shared_deadline_stops_after_prior_calls(monkeypatch):
|
||||
return answer
|
||||
monkeypatch.setattr(workflow.time,'monotonic',lambda:clock[0])
|
||||
monkeypatch.setattr(suite_backends,'claude_generate',advancing)
|
||||
with pytest.raises(Problem,match='job_time_budget_exhausted'):
|
||||
workflow.generate(source,preflight(source),threading.Event(),'192.168.22.8')
|
||||
assert len(calls) == 2
|
||||
assert [c[0]['execution']['max_seconds'] for c in calls] == [60,29]
|
||||
|
||||
|
||||
def test_missing_cost_measurement_stops_future_paid_calls(monkeypatch):
|
||||
source = request(7)
|
||||
family = natural(source,[[c['alias'] for c in source['cases']]])
|
||||
calls = install_backend(monkeypatch,source,family,costs=[None])
|
||||
with pytest.raises(Problem,match='budget_accounting_unavailable'):
|
||||
with pytest.raises(Problem,match='insufficient_remaining_pass_budget'):
|
||||
workflow.generate(source,preflight(source),threading.Event(),'192.168.22.8')
|
||||
assert len(calls) == 1
|
||||
assert calls[0][0]['execution']['max_seconds'] == 20
|
||||
|
||||
|
||||
def test_missing_cost_measurement_stays_unknown_without_stopping(monkeypatch):
|
||||
source = request(7)
|
||||
family = natural(source,[[c['alias'] for c in source['cases']]])
|
||||
calls = install_backend(monkeypatch,source,family,costs=[None]*5)
|
||||
_, metadata = workflow.generate(source,preflight(source),threading.Event(),'192.168.22.8')
|
||||
assert len(calls) == 5 and metadata['cost_usd_estimate'] is None
|
||||
|
||||
|
||||
def test_cancellation_stops_between_model_passes(monkeypatch):
|
||||
|
||||
@ -229,7 +229,7 @@ def test_prompt_provenance_is_stored_and_replay_does_not_relabel(tmp_path, monke
|
||||
|
||||
def test_compaction_and_incomplete_detection():
|
||||
for events, code in [([{"type": "system", "subtype": "compact_boundary"}], "compaction_detected"),
|
||||
([], "incomplete_generation")]:
|
||||
([], "pass_missing_terminal_event")]:
|
||||
with pytest.raises(Problem, match=code):
|
||||
suite_backends.parse_claude("\n".join(json.dumps(e) for e in events), "claude-fable-5")
|
||||
|
||||
@ -237,7 +237,7 @@ def test_compaction_and_incomplete_detection():
|
||||
@pytest.mark.parametrize("failure,code", [
|
||||
(None, None), ("model", "model_changed"),
|
||||
("context", "backend_capabilities_changed"),
|
||||
("tools", "worker_isolation_failed"), ("output", "incomplete_generation"),
|
||||
("tools", "worker_isolation_failed"), ("output", "pass_output_limit"),
|
||||
])
|
||||
def test_actual_cli_envelope_and_runtime_guards(failure, code):
|
||||
model = MODELS["claude"]["model"]
|
||||
|
||||
244
testing/tests/test_suite_recovery.py
Normal file
244
testing/tests/test_suite_recovery.py
Normal file
@ -0,0 +1,244 @@
|
||||
"""Fault injection exercises real orchestration without hosted model requests."""
|
||||
import copy
|
||||
import json
|
||||
import threading
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[2] / 'scripts/ops'))
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parents[2] / 'services/hermes/scripts'))
|
||||
import suite_backends
|
||||
import suite_hierarchical
|
||||
import suite_multipass as workflow
|
||||
import suite_recovery
|
||||
from suite_contract import MODELS, Problem, preflight, validate_request, validate_result
|
||||
from suite_profiles import PROFILE_FIELDS, profile_source, validate_profiles
|
||||
from suite_recovery import Checkpoints
|
||||
from suite_sizing import balanced_sizes
|
||||
from suite_synthetic import fixture, score
|
||||
from test_suite_multipass import natural, public, review, wire_value
|
||||
from suite_capacity_fixture import fixture as realistic_fixture
|
||||
|
||||
|
||||
class NoWait:
|
||||
"""Cancellation remains testable without real retry sleeps."""
|
||||
def is_set(self):
|
||||
return False
|
||||
def wait(self, _):
|
||||
return False
|
||||
|
||||
|
||||
def source(size=14, large=False):
|
||||
raw, expected = realistic_fixture() if large else fixture(size)
|
||||
raw['routing'] = {'allow_external': True, 'allowed_external_providers': ['claude']}
|
||||
raw['execution'] = {'strategy': 'whole_suite', 'max_seconds': 7200, 'max_cost_usd': 60}
|
||||
return validate_request(raw, ['claude']), expected
|
||||
|
||||
|
||||
def oracle(monkeypatch, expected, failures=None):
|
||||
"""Use an independent fixture map to check cross-batch control flow and validation."""
|
||||
calls = []
|
||||
failures = dict(failures or {})
|
||||
def backend(request, cancel, *, invocation, progress=None, job_deadline=None):
|
||||
stage = invocation['stage']
|
||||
calls.append({'stage': stage, 'aliases': sorted(c['alias'] for c in request['cases']),
|
||||
'seconds': request['execution']['max_seconds'], 'deadline': job_deadline,
|
||||
'turns': invocation['max_turns'], 'source': copy.deepcopy(request['cases'])})
|
||||
if failures.get(stage, 0):
|
||||
failures[stage] -= 1
|
||||
raise Problem('incomplete_generation', 502, final_event_subtype='success',
|
||||
final_is_error=True, exit_code=1, assistant_api_error_seen=True,
|
||||
structured_output_present=False, turns=1, max_turns=invocation['max_turns'],
|
||||
cost_usd_estimate=0, usage={'input_tokens':0,'output_tokens':0})
|
||||
if stage == 'implementation_profiles':
|
||||
value = {'profiles': {}}
|
||||
for c in request['cases']:
|
||||
name = expected[c['alias']]
|
||||
p = {k: None for k in PROFILE_FIELDS}
|
||||
p.update(target=name, work='Implement '+name+' tests.', setup=name+' fixture',
|
||||
stimulus=name+' stimulus', observations=name+' measurements',
|
||||
machinery=name, evidence={'field':'success_criteria', 'quote':c['success_criteria'][:90]})
|
||||
value['profiles'][c['alias']] = p
|
||||
else:
|
||||
parts = {}
|
||||
for c in request['cases']:
|
||||
parts.setdefault(expected[c['alias']], []).append(c['alias'])
|
||||
decision = natural(request, list(parts.values()), list(parts))
|
||||
if stage in ('proposal_a','proposal_b','profile_proposal_a','profile_proposal_b','profile_reconciliation'):
|
||||
decision = public(decision)
|
||||
if stage in ('large_family_review','decision_audit'):
|
||||
decision = review(decision, decision)
|
||||
value = wire_value(decision, invocation)
|
||||
return value, {'model':MODELS['claude']['model'], 'cost_usd_estimate': 25.0,
|
||||
'usage':{'input_tokens':100,'output_tokens':50}, 'turns':2, 'duration_api_ms':2,
|
||||
'compaction':False, 'truncation':False, 'cli_diagnostics':{'exit_code':0}}
|
||||
monkeypatch.setattr(suite_backends, 'claude_generate', backend)
|
||||
return calls
|
||||
|
||||
|
||||
def run(monkeypatch, *, mode='direct', size=14, large=False, failures=None):
|
||||
request, expected = source(size, large)
|
||||
calls = oracle(monkeypatch, expected, failures)
|
||||
selected = {**preflight(request), 'execution_mode':mode}
|
||||
updates = []
|
||||
result, metadata = workflow.generate(request, selected, NoWait(), '192.168.22.8', updates.append,
|
||||
credential_scope={'owner':'synthetic-test','providers':['claude']})
|
||||
validate_result(result, request)
|
||||
return request, expected, calls, result, metadata, updates
|
||||
|
||||
|
||||
def test_proposal_b_retry_keeps_validated_a_and_same_source(monkeypatch):
|
||||
request, expected, calls, result, meta, updates = run(monkeypatch, failures={'proposal_b':1})
|
||||
assert [c['stage'] for c in calls].count('proposal_a') == 1
|
||||
assert [c['stage'] for c in calls].count('proposal_b') == 2
|
||||
attempts = [c for c in calls if c['stage']=='proposal_b']
|
||||
assert attempts[0]['source'] == attempts[1]['source']
|
||||
assert meta['retry_count']['proposal_b'] == 1
|
||||
assert 'proposal_a' in meta['checkpointed_passes']
|
||||
assert meta['cost_usd_estimate'] > 60 and meta['cost_guard_enforced'] is False
|
||||
assert all(c['deadline']==calls[0]['deadline'] for c in calls)
|
||||
assert score({'groups':meta['review_summary']['natural_families']}, expected)['false_merge_pairs']==0
|
||||
|
||||
|
||||
def test_review_failure_does_not_regenerate_prior_passes(monkeypatch):
|
||||
_, _, calls, _, meta, _ = run(monkeypatch, size=75, failures={'large_family_review':1})
|
||||
for name in ('proposal_a','proposal_b','reconciliation'):
|
||||
assert [c['stage'] for c in calls].count(name)==1
|
||||
assert [c['stage'] for c in calls].count('large_family_review')==2
|
||||
assert meta['retry_count']['large_family_review']==1
|
||||
|
||||
|
||||
@pytest.mark.parametrize('mode', ['direct','hierarchical'])
|
||||
def test_large_comparable_fixture_complete_and_globally_reconciled(monkeypatch, mode):
|
||||
request, expected, calls, result, meta, updates = run(monkeypatch, mode=mode, large=True)
|
||||
assert len(request['cases'])==363
|
||||
assert max(len(g['members']) for g in result['groups'])<=5
|
||||
assert len(result['groups'])==90 and meta['natural_family_count']==45
|
||||
quality=score({'groups':meta['review_summary']['natural_families']}, expected)
|
||||
assert quality['false_merge_pairs']==quality['missed_merge_pairs']==0
|
||||
assert all(d['part_sizes']==balanced_sizes(d['natural_case_count']) for d in meta['review_summary']['capacity_divisions'])
|
||||
if mode=='hierarchical':
|
||||
assert meta['profile_batch_count']==16
|
||||
for c in calls:
|
||||
if c['stage'] in ('profile_proposal_a','profile_proposal_b','profile_reconciliation'):
|
||||
assert len(c['aliases'])==363
|
||||
originals={c['alias']:c for c in request['cases']}
|
||||
reopened=[c for call in calls if call['stage']=='source_review' for c in call['source']]
|
||||
assert {c['alias']:c for c in reopened}==originals
|
||||
assert meta['cross_batch_review_count']==1
|
||||
assert 'Implement ' not in json.dumps(updates)
|
||||
|
||||
|
||||
def test_failed_direct_switches_only_after_bounded_recovery(monkeypatch):
|
||||
_, _, calls, _, meta, _ = run(monkeypatch, failures={'proposal_b':3})
|
||||
assert [c['stage'] for c in calls].count('proposal_a')==1
|
||||
assert [c['stage'] for c in calls].count('proposal_b')==3
|
||||
assert meta['execution_mode']=='hierarchical'
|
||||
assert meta['strategy_events'][-1]['reason']=='pass_generation_retries_exhausted'
|
||||
|
||||
|
||||
@pytest.mark.parametrize('code', ['provider_authentication','provider_forbidden','cancelled',
|
||||
'invalid_review_evidence','worker_isolation_failed','job_time_budget_exhausted'])
|
||||
def test_nonretriable_conditions_stop_without_fallback(monkeypatch, code):
|
||||
request, _ = source()
|
||||
seen=[]
|
||||
def fail(*args, **kw):
|
||||
seen.append(kw['invocation']['stage'])
|
||||
raise Problem(code, 502)
|
||||
monkeypatch.setattr(suite_backends,'claude_generate',fail)
|
||||
with pytest.raises(Problem, match='^'+code+'$'):
|
||||
workflow.generate(request, preflight(request), NoWait(),'192.168.22.8')
|
||||
assert seen==['proposal_a']
|
||||
|
||||
|
||||
def test_turn_recovery_grows_only_within_capacity(monkeypatch):
|
||||
request, expected = source()
|
||||
calls=oracle(monkeypatch,expected)
|
||||
original=suite_backends.claude_generate
|
||||
turns=[]
|
||||
def backend(*args,**kw):
|
||||
turns.append(kw['invocation']['max_turns'])
|
||||
if len(turns)<3:
|
||||
raise Problem('pass_turn_limit',502,cost_usd_estimate=0,turns=turns[-1],
|
||||
max_turns=turns[-1],usage={'input_tokens':1,'output_tokens':1})
|
||||
return original(*args,**kw)
|
||||
monkeypatch.setattr(suite_backends,'claude_generate',backend)
|
||||
workflow.generate(request,preflight(request),NoWait(),'192.168.22.8')
|
||||
assert turns[:3]==[4,5,6]
|
||||
|
||||
|
||||
def test_checkpoint_scope_hash_and_restart_behavior():
|
||||
request,_=source()
|
||||
call={'stage':'proposal_a','input':'synthetic','prior_hash':'a'}
|
||||
store=Checkpoints(request,'claude',{'owner':'a'},float('inf'))
|
||||
key=store.key(call);store.put(key,'proposal_a',{'groups':[]})
|
||||
assert store.get(key,'proposal_a')=={'groups':[]}
|
||||
assert store.key({**call,'prior_hash':'b'})!=key
|
||||
assert Checkpoints(request,'claude',{'owner':'b'},float('inf')).key(call)!=key
|
||||
assert Checkpoints(request,'claude',{'owner':'a'},float('inf')).get(key,'proposal_a') is None
|
||||
store.clear();assert store.get(key,'proposal_a') is None
|
||||
|
||||
|
||||
def test_cli_cost_guard_is_absent_and_legacy_request_cost_is_unenforced():
|
||||
request,_=source()
|
||||
request['execution']['max_cost_usd']=1e9
|
||||
assert validate_request(request,['claude'])['execution']['max_cost_usd']==1e9
|
||||
assert '--max-budget-usd' not in suite_backends.claude_command(MODELS['claude']['model'],60)
|
||||
assert preflight(request)['cost_guard_enforced'] is False
|
||||
|
||||
|
||||
def test_full_profile_set_fits_global_comparison_at_service_case_limit():
|
||||
"""Bounded profiles keep every alias globally visible; no neighborhood shortcut."""
|
||||
from suite_policy import invocation
|
||||
request = {'campaign':'SYNTHETIC','suite':'CAPACITY','cases':[
|
||||
{'alias':'CASE-'+str(i).zfill(48),'description':'Synthetic'} for i in range(400)]}
|
||||
profiles = {c['alias']:{k:'x'*n for k,n in PROFILE_FIELDS.items()} for c in request['cases']}
|
||||
compact = profile_source(request, profiles)
|
||||
proposal = {'groups':[{'name':'x'*56,'members':[c['alias']]} for c in request['cases']]}
|
||||
call = invocation('profile_reconciliation',compact,{'proposal_a':proposal,'proposal_b':proposal})
|
||||
result = workflow.capacity(call,'claude',400)
|
||||
assert result['max_turns'] >= 4 and result['context_reserved_tokens'] <= 1000000
|
||||
assert len(call['schema']['properties']['assignments']['required']) == 400
|
||||
|
||||
|
||||
def test_profile_evidence_and_exact_aliases_are_independently_validated():
|
||||
request,expected=source()
|
||||
c=request['cases'][0]
|
||||
p={k:None for k in PROFILE_FIELDS}
|
||||
p['evidence']={'field':'success_criteria','quote':c['success_criteria'][:90]}
|
||||
subset={**request,'cases':[c]}
|
||||
validate_profiles({'profiles':{c['alias']:p}},subset)
|
||||
with pytest.raises(Problem,match='invalid_case_assignments'):
|
||||
validate_profiles({'profiles':{'CASE-invented':p}},subset)
|
||||
p['evidence']['quote']='unsupported secret instructions'
|
||||
with pytest.raises(Problem,match='invalid_review_evidence'):
|
||||
validate_profiles({'profiles':{c['alias']:p}},subset)
|
||||
|
||||
|
||||
def test_checkpoint_hit_launches_no_new_model_call(monkeypatch):
|
||||
request,expected=source()
|
||||
calls=oracle(monkeypatch,expected)
|
||||
worker=workflow.Workflow(request,preflight(request),NoWait(),'192.168.22.8',credential_scope='synthetic')
|
||||
first=worker.call('proposal_a',request)
|
||||
second=worker.call('proposal_a',request)
|
||||
assert first==second and len(calls)==1
|
||||
assert worker.safe_state()['checkpoint_reused']==['proposal_a']
|
||||
assert len(worker.records)==1
|
||||
|
||||
|
||||
def test_one_structured_assignment_repair_preserves_source(monkeypatch):
|
||||
request,expected=source()
|
||||
calls=oracle(monkeypatch,expected)
|
||||
original=suite_backends.claude_generate
|
||||
def backend(*args,**kwargs):
|
||||
value,metadata=original(*args,**kwargs)
|
||||
if len(calls)==1:
|
||||
value['assignments'].pop(next(iter(value['assignments'])))
|
||||
return value,metadata
|
||||
monkeypatch.setattr(suite_backends,'claude_generate',backend)
|
||||
result,meta=workflow.generate(request,preflight(request),NoWait(),'192.168.22.8')
|
||||
assert meta['retry_count']=={'proposal_a':1}
|
||||
assert calls[0]['source']==calls[1]['source']
|
||||
validate_result(result,request)
|
||||
Loading…
x
Reference in New Issue
Block a user