hermes: recover suite passes and adapt large-suite analysis

This commit is contained in:
jenkins 2026-09-30 01:52:19 -05:00
parent cef01b0d72
commit 9255a64b23
29 changed files with 1739 additions and 196 deletions

View File

@ -206,15 +206,17 @@
"max_seconds": {
"type": "integer",
"minimum": 10,
"maximum": 3600,
"maximum": 7200,
"default": 1800
},
"max_cost_usd": {
"type": "number",
"type": [
"number",
"null"
],
"exclusiveMinimum": 0,
"maximum": 30,
"default": 30,
"description": "CLI estimated-cost guard, not subscription billing. Shared across all passes."
"deprecated": true,
"description": "Accepted for compatibility only. Estimated cost is informational and never enforced."
}
}
}

File diff suppressed because one or more lines are too long

View File

@ -0,0 +1,524 @@
{
"hosted_synthetic_14": [
{
"cost_usd_estimate": 0.42616,
"final_groups": [
{
"description": "Work-size part 1/2 of one family. Byte-array frame builder, decoder-call fixture, capture of decoded object, comparison of fields, checksum status and rejection code; no hardware",
"members": [
"CASE-REC-02",
"CASE-REC-04",
"CASE-REC-08",
"CASE-REC-10"
],
"name": "Status frame decoder harness (1/2)"
},
{
"description": "Work-size part 2/2 of one family. Byte-array frame builder, decoder-call fixture, capture of decoded object, comparison of fields, checksum status and rejection code; no hardware",
"members": [
"CASE-REC-00",
"CASE-REC-06",
"CASE-REC-12"
],
"name": "Status frame decoder harness (2/2)"
},
{
"description": "Work-size part 1/2 of one family. Controllable pulse source driver, reset-line recorder/digital capture fixture, reset timing extraction, deadline check against supplied tolerance",
"members": [
"CASE-REC-01",
"CASE-REC-05",
"CASE-REC-07",
"CASE-REC-11"
],
"name": "Watchdog pulse reset timing rig (1/2)"
},
{
"description": "Work-size part 2/2 of one family. Controllable pulse source driver, reset-line recorder/digital capture fixture, reset timing extraction, deadline check against supplied tolerance",
"members": [
"CASE-REC-03",
"CASE-REC-09",
"CASE-REC-13"
],
"name": "Watchdog pulse reset timing rig (2/2)"
}
],
"mode": "direct",
"natural_groups": [
{
"common_work": "Byte-array frame builder, decoder-call fixture, capture of decoded object, comparison of fields, checksum status and rejection code; no hardware",
"description": "Offline tests: a byte-array builder feeds encoded watchdog status frames to a decoder-call fixture. Checks decoded fields, checksum status and rejection code. Nominal and fault cases differ only in bytes and expected results.",
"evidence": [
{
"alias": "CASE-REC-00",
"field": "success_criteria",
"quote": "Capture the decoded object and compare fields, checksum status, and rejection code"
},
{
"alias": "CASE-REC-02",
"field": "preconditions",
"quote": "no live hardware is required"
}
],
"members": [
"CASE-REC-00",
"CASE-REC-02",
"CASE-REC-04",
"CASE-REC-06",
"CASE-REC-08",
"CASE-REC-10",
"CASE-REC-12"
],
"name": "Status frame decoder harness",
"rationale": "All members have the same preconditions, the same stimulus path and the same success criteria. Fault-injection cases change only the frame bytes and the expected checksum or rejection outcome. They reuse the fixture and the assertion structure.",
"uncertainty": "Frame contents, the faults to inject and the expected rejection codes are not stated. Complex corruption patterns may need extra builder helpers.",
"variation_sets": [
[
"CASE-REC-02",
"CASE-REC-04",
"CASE-REC-08",
"CASE-REC-10"
],
[
"CASE-REC-00",
"CASE-REC-06",
"CASE-REC-12"
]
]
},
{
"common_work": "Controllable pulse source driver, reset-line recorder/digital capture fixture, reset timing extraction, deadline check against supplied tolerance",
"description": "Hardware tests: a controllable pulse source stops and resumes watchdog pulses at the controller input. A digital capture fixture records reset-line timing against the supplied deadline. Nominal and fault cases differ in stimulus only.",
"evidence": [
{
"alias": "CASE-REC-01",
"field": "success_criteria",
"quote": "Measure reset-line timing with a digital capture fixture and check the deadline"
},
{
"alias": "CASE-REC-03",
"field": "preconditions",
"quote": "Use a controllable pulse source and reset-line recorder"
}
],
"members": [
"CASE-REC-01",
"CASE-REC-03",
"CASE-REC-05",
"CASE-REC-07",
"CASE-REC-09",
"CASE-REC-11",
"CASE-REC-13"
],
"name": "Watchdog pulse reset timing rig",
"rationale": "All members use the same physical rig, the same stop/resume pulse sequence and the same timing measurement and deadline assertion. The fault-injection label adds no new equipment or new way of observing. This is separate from the offline decoder work.",
"uncertainty": "Deadline and tolerance values are left out. The records don't say how fault cases change the pulse stimulus, for example stop duration or irregular pulses. Such changes might need extra pulse-source control.",
"variation_sets": [
[
"CASE-REC-01",
"CASE-REC-05",
"CASE-REC-07",
"CASE-REC-11",
"CASE-REC-13"
],
[
"CASE-REC-03",
"CASE-REC-09"
]
]
}
],
"passes": [
{
"cli_turns": 2,
"input_bytes": 11122,
"max_turns": 4,
"stage": "proposal_a",
"wall_seconds": 5.724
},
{
"cli_turns": 3,
"input_bytes": 11122,
"max_turns": 4,
"stage": "proposal_b",
"wall_seconds": 9.437
},
{
"cli_turns": 3,
"input_bytes": 16696,
"max_turns": 4,
"stage": "reconciliation",
"wall_seconds": 23.197
},
{
"cli_turns": 3,
"input_bytes": 20816,
"max_turns": 4,
"stage": "large_family_review",
"wall_seconds": 26.503
},
{
"cli_turns": 2,
"input_bytes": 25569,
"max_turns": 4,
"stage": "decision_audit",
"wall_seconds": 14.143
}
],
"quality": {
"coverage": true,
"false_merge_pairs": 0,
"families": 2,
"missed_merge_pairs": 0,
"pair_precision": 1.0,
"pair_recall": 1.0
},
"seconds": 79.00923439487815,
"status": "completed",
"usage": {
"cache_creation": {
"ephemeral_1h_input_tokens": 0,
"ephemeral_5m_input_tokens": 0
},
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"input_tokens": 60130,
"output_tokens": 9282,
"output_tokens_details": {
"thinking_tokens": 604
},
"server_tool_use": {
"web_fetch_requests": 0,
"web_search_requests": 0
}
}
},
{
"cost_usd_estimate": 0.801868,
"final_groups": [
{
"description": "Work-size part 1/2 of one family. Byte-array frame builder, hardware-free decoder-call fixture, capture of decoded object, and field/checksum/rejection-code comparison",
"members": [
"CASE-REC-02",
"CASE-REC-04",
"CASE-REC-08",
"CASE-REC-10"
],
"name": "Status Frame Decoder Byte Harness (1/2)"
},
{
"description": "Work-size part 2/2 of one family. Byte-array frame builder, hardware-free decoder-call fixture, capture of decoded object, and field/checksum/rejection-code comparison",
"members": [
"CASE-REC-00",
"CASE-REC-06",
"CASE-REC-12"
],
"name": "Status Frame Decoder Byte Harness (2/2)"
},
{
"description": "Work-size part 1/2 of one family. Controllable pulse source at controller input, digital capture of the reset line, and a deadline check within the supplied timing tolerance",
"members": [
"CASE-REC-01",
"CASE-REC-05",
"CASE-REC-07",
"CASE-REC-11"
],
"name": "Watchdog Pulse Reset-Line Timing Capture (1/2)"
},
{
"description": "Work-size part 2/2 of one family. Controllable pulse source at controller input, digital capture of the reset line, and a deadline check within the supplied timing tolerance",
"members": [
"CASE-REC-03",
"CASE-REC-09",
"CASE-REC-13"
],
"name": "Watchdog Pulse Reset-Line Timing Capture (2/2)"
}
],
"mode": "hierarchical",
"natural_groups": [
{
"common_work": "Byte-array frame builder, hardware-free decoder-call fixture, capture of decoded object, and field/checksum/rejection-code comparison",
"description": "Software-only tests: a byte-array builder encodes watchdog status frames and the decoder is called. Decoded fields, checksum status and rejection code are asserted. Frames are nominal or fault-injected, but the faults used are not stated.",
"evidence": [
{
"alias": "CASE-REC-00",
"field": "success_criteria",
"quote": "Capture the decoded object and compare fields, checksum status, and rejection code"
},
{
"alias": "CASE-REC-02",
"field": "preconditions",
"quote": "Use a byte-array builder and a decoder-call fixture; no live hardware is required"
}
],
"members": [
"CASE-REC-00",
"CASE-REC-02",
"CASE-REC-04",
"CASE-REC-06",
"CASE-REC-08",
"CASE-REC-10",
"CASE-REC-12"
],
"name": "Status Frame Decoder Byte Harness",
"rationale": "All members have the same preconditions and success_criteria. Nominal and fault-injected cases change only the frame bytes and the expected decode or rejection result. Once the builder and comparison helper exist, each extra case is a cheap input and assertion variation.",
"uncertainty": "The source does not state which corruptions the fault-injection frames use or which rejection codes they expect. Identical text may still hide different frame contents.",
"variation_sets": [
[
"CASE-REC-02",
"CASE-REC-04",
"CASE-REC-08",
"CASE-REC-10"
],
[
"CASE-REC-00",
"CASE-REC-06",
"CASE-REC-12"
]
]
},
{
"common_work": "Controllable pulse source at controller input, digital capture of the reset line, and a deadline check within the supplied timing tolerance",
"description": "Hardware tests: a pulse source stops and resumes watchdog pulses at the controller input. A digital capture records reset-line timing for a deadline check within tolerance. Deadlines and fault pulse patterns are not stated.",
"evidence": [
{
"alias": "CASE-REC-01",
"field": "success_criteria",
"quote": "Measure reset-line timing with a digital capture fixture and check the deadline"
},
{
"alias": "CASE-REC-03",
"field": "preconditions",
"quote": "Use a controllable pulse source and reset-line recorder; timing tolerance is supplied"
}
],
"members": [
"CASE-REC-01",
"CASE-REC-03",
"CASE-REC-05",
"CASE-REC-07",
"CASE-REC-09",
"CASE-REC-11",
"CASE-REC-13"
],
"name": "Watchdog Pulse Reset-Line Timing Capture",
"rationale": "All members share the same physical stimulus equipment and digital-capture timing measurement. This is materially different from the software decoder harness. Nominal and fault cases differ only in pulse pattern and expected timing, using the same equipment and workflow.",
"uncertainty": "Deadline values, tolerances and the fault-injection pulse patterns are not stated. Fault cases might need stimulus control beyond simply stopping and resuming pulses.",
"variation_sets": [
[
"CASE-REC-01",
"CASE-REC-05",
"CASE-REC-07",
"CASE-REC-11",
"CASE-REC-13"
],
[
"CASE-REC-03",
"CASE-REC-09"
]
]
}
],
"passes": [
{
"cli_turns": 2,
"input_bytes": 9394,
"max_turns": 4,
"stage": "implementation_profiles",
"wall_seconds": 10.335
},
{
"cli_turns": 2,
"input_bytes": 9386,
"max_turns": 4,
"stage": "implementation_profiles",
"wall_seconds": 10.134
},
{
"cli_turns": 3,
"input_bytes": 9386,
"max_turns": 4,
"stage": "implementation_profiles",
"wall_seconds": 19.774
},
{
"cli_turns": 3,
"input_bytes": 7049,
"max_turns": 4,
"stage": "implementation_profiles",
"wall_seconds": 12.544
},
{
"cli_turns": 2,
"input_bytes": 17434,
"max_turns": 4,
"stage": "profile_proposal_a",
"wall_seconds": 5.923
},
{
"cli_turns": 2,
"input_bytes": 17434,
"max_turns": 4,
"stage": "profile_proposal_b",
"wall_seconds": 5.723
},
{
"cli_turns": 3,
"input_bytes": 18258,
"max_turns": 4,
"stage": "profile_reconciliation",
"wall_seconds": 9.535
},
{
"cli_turns": 3,
"input_bytes": 16167,
"max_turns": 4,
"stage": "source_review",
"wall_seconds": 21.383
},
{
"cli_turns": 3,
"input_bytes": 20937,
"max_turns": 4,
"stage": "large_family_review",
"wall_seconds": 26.395
},
{
"cli_turns": 3,
"input_bytes": 25745,
"max_turns": 4,
"stage": "decision_audit",
"wall_seconds": 26.992
}
],
"quality": {
"coverage": true,
"false_merge_pairs": 0,
"families": 2,
"missed_merge_pairs": 0,
"pair_precision": 1.0,
"pair_recall": 1.0
},
"seconds": 148.75216588703915,
"status": "completed",
"usage": {
"cache_creation": {
"ephemeral_1h_input_tokens": 0,
"ephemeral_5m_input_tokens": 0
},
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"input_tokens": 111617,
"output_tokens": 17770,
"output_tokens_details": {
"thinking_tokens": 635
},
"server_tool_use": {
"web_fetch_requests": 0,
"web_search_requests": 0
}
}
}
],
"mocked_synthetic_363": [
{
"mode": "direct",
"backend": "mocked-independent-fixture-oracle",
"elapsed_seconds": 6.596287971013226,
"source_size_metadata": {
"source_bytes": 410322,
"description": {
"min": 113,
"max": 650,
"median": 133,
"p95": 641
},
"preconditions": {
"min": 108,
"max": 694,
"median": 131,
"p95": 682
},
"case_type": {
"min": 7,
"max": 15,
"median": 8,
"p95": 15
},
"success_criteria": {
"min": 149,
"max": 1009,
"median": 164,
"p95": 1002
}
},
"request_bytes": 410474,
"result_bytes": 175646,
"model_calls": 5,
"natural_families": 45,
"final_tasks": 90,
"max_group_size": 5,
"fixture_assignment_validation": {
"coverage": true,
"families": 45,
"pair_precision": 1.0,
"pair_recall": 1.0,
"false_merge_pairs": 0,
"missed_merge_pairs": 0
},
"measured_provider_usage": null,
"measured_provider_cost": null
},
{
"mode": "hierarchical",
"backend": "mocked-independent-fixture-oracle",
"elapsed_seconds": 6.578743355930783,
"source_size_metadata": {
"source_bytes": 410322,
"description": {
"min": 113,
"max": 650,
"median": 133,
"p95": 641
},
"preconditions": {
"min": 108,
"max": 694,
"median": 131,
"p95": 682
},
"case_type": {
"min": 7,
"max": 15,
"median": 8,
"p95": 15
},
"success_criteria": {
"min": 149,
"max": 1009,
"median": 164,
"p95": 1002
}
},
"request_bytes": 410474,
"result_bytes": 224822,
"model_calls": 26,
"natural_families": 45,
"final_tasks": 90,
"max_group_size": 5,
"fixture_assignment_validation": {
"coverage": true,
"families": 45,
"pair_precision": 1.0,
"pair_recall": 1.0,
"false_merge_pairs": 0,
"missed_merge_pairs": 0
},
"measured_provider_usage": null,
"measured_provider_cost": null
}
],
"unit_tests": 200,
"real_suite_rerun": false,
"large_hosted_quality_verified": false
}

View File

@ -1,5 +1,9 @@
# Suite job time and cost budget
Current limits and recovery behavior supersede the historical details below. See
[the adaptive execution handoff](hermes_suite_recovery_20260930.md): 7200 seconds,
no estimated-cost cutoff, bounded pass recovery and optional hierarchical execution.
New explicitly requested jobs may use `execution.max_seconds: 3600`. Omitting
the field still selects 1800 seconds; valid values are integers from 10 to 3600.
The USD 30 CLI estimated-cost maximum/default is unchanged. These figures are

View File

@ -1,5 +1,9 @@
# Multi-pass implementation grouping
Current limits and recovery behavior supersede the historical details below. See
[the adaptive execution handoff](hermes_suite_recovery_20260930.md): 7200 seconds,
no estimated-cost cutoff, bounded pass recovery and optional hierarchical execution.
The suite planner keeps the same HTTPS endpoint, request fields, authentication,
provider permissions, generalized-data approval, and asynchronous job lifecycle.
The required result object remains `result.groups`, with `name`, `description`,

View File

@ -4,26 +4,20 @@ This optional API groups one complete campaign/suite into implementation familie
The existing `/local-model` API, GPU allocation, serving model, and limits are unchanged.
Deployment uses Flux; the application and workbook stay on the laptop.
## Current multi-pass policy
## Current adaptive policy
New jobs use [the multi-pass implementation policy](hermes_suite_multipass.md):
two independent full-suite proposals, reconciliation, up to two bounded semantic
reviews, and a programmatic five-case final cap. Names are at most 64 characters.
The endpoint and request fields are unchanged. Optional top-level `review_summary`
is returned only with the authorized result. Configuration remains `suite-v6-20260929`;
policy, prompt, and execution revisions identify the changed grouping behavior.
The prior single-pass acceptance results below are historical and do not measure
the new workflow. New jobs may explicitly request up to 3600 seconds; the default
remains 1800 seconds. One deadline and the USD 30 CLI estimate guard cover all invocations. This is not subscription
billing. Progress updates every five seconds during a CLI pass. The separate local-model endpoint is unchanged;
the 8K local route cannot admit the new multi-pass reconciliation schema.
The current [recovery and adaptive-analysis handoff](hermes_suite_recovery_20260930.md)
records the exact latest failure, deployed behavior, tests and client settings.
New jobs accept up to 7200 seconds, with one shared deadline and no estimated-cost
cutoff. Legacy `max_cost_usd` is accepted but unenforced. The HTTP compatibility
revision stays `suite-v6-20260929`; execution is `suite-adaptive-v11-20260930`.
The active model is **`claude-opus-5-5`**, invoked as `claude-opus-5-5[1m]`
using native Claude Code 2.1.285. This was the user-requested 5.5 upgrade;
older Opus 4.8 acceptance records below are historical. Effort is high, or
xhigh at 100+ cases or 128 KiB+ source content. The model and effort stay fixed
throughout a job. Current execution revision: `suite-multipass-v10-20260930`.
See the [current capacity fix and client settings](hermes_suite_capacity_20260930.md).
Small suites prefer two full-suite proposals and reconciliation. High-risk suites
use compact profiles, two independent global profile proposals, global
reconciliation and full-source verification before semantic review and the hard
five-case final cap. Recovery preserves validated active-job checkpoints. No
provider fallback, source-content logs or partial completion is enabled. Pinned
model remains `claude-opus-5-5` through native CLI 2.1.285, high/xhigh by suite size.
## Earlier CLI failure diagnostics update (historical)

View File

@ -0,0 +1,221 @@
# Suite recovery and adaptive analysis
## Latest real-job diagnosis
Exact failed job: `90ab61bf285e4c89a41d893d4fee649d`, located by descending
persisted creation time, failed status and 363-case count. Its request selected
3600 seconds and an estimated-cost guard of 30. No real source was inspected or
rerun. The record is not the earlier decision-audit capacity failure.
| Measurement | Retained evidence |
| --- | --- |
| Error | `incomplete_generation`, `failure_stage=cli_final_result` |
| Failed pass | `proposal_b`; one completed model pass |
| CLI exit / signal | 1 / null; `nonzero_exit` |
| Terminal event | `result`, subtype `success`, **is_error=true** |
| Provider / assistant stop | `stop_sequence` / `stop_sequence` |
| API error indicator | true; allowlisted category `unknown`; HTTP status unknown |
| Structured result | Absent from terminal output, tool input and JSON text candidates |
| Turns / ceiling | 1 / 6; no turn-limit or structured-retry-limit outcome |
| B input / output usage | CLI reported 0 / 0; does not independently prove no provider work occurred |
| B observed model capacity | Unknown; failed terminal event had no modelUsage limits |
| B admission | Same 439270-byte input/system/schema size as A; conservative bound 447462; six full outputs reserved: 831462 under 1M |
| B configured output / reasoning | 64000 per turn / xhigh; reasoning-token limit unknown |
| B output bytes | Last heartbeat observed 54366; exact terminal byte count was not retained |
| B elapsed / timeout | Last heartbeat 411.8 s; approximately 412.9 s including boundary overhead / 2791.629775 s watchdog |
| Provider timeout | Unknown |
| Job elapsed / remaining | 1221.233 s / 2378.8 s |
| Total input / output | 319884 / 98177; output includes 82581 thinking tokens |
| Cost used / remaining old guard | USD 3.243076 / 26.756924 estimated |
| Route | Claude CLI, first-party OAuth; selected `claude-opus-5-5[1m]` |
| Runtime model evidence | A confirmed firstParty canonical `claude-opus-5-5`; B initialization matched requested model, but no successful terminal canonical-model verification |
| CLI | 2.1.285 |
| Old configuration / prompt | `suite-v6-20260929` / `implementation-proximity-multipass-v5-20260930` |
| Old execution / policy | `suite-multipass-v10-20260930` / `implementation-five-v1-20260929` |
Proposal A completed in 808.317 seconds, three CLI turns, exit zero and valid
structured output. B produced an unsuccessful API-error event and exited nonzero.
The underlying provider cause is **unknown**: raw CLI text was deliberately deleted.
No retained evidence identifies rate limiting, context/output exhaustion, timeout,
turn exhaustion or an adapter rejecting valid structured content. The old adapter
collapsed this API error into `incomplete_generation` and the workflow had no retry.
The new classifier treats an unclassified API error as eligible for a bounded
retry, with `provider_failure_transience=unconfirmed`; that is not proof it was
transient. New diagnostics retain exact output bytes and per-attempt elapsed time.
## Time and cost
The maximum accepted `execution.max_seconds` is **7200**; omission still uses 1800.
All profile, grouping, review, retry and repair calls share one absolute deadline.
Submit/status/result HTTP timeouts remain 45 seconds on the laptop.
The cost guard is **removed**, following the user's subsequent instruction. The
CLI receives no `--max-budget-usd`. Legacy positive finite `max_cost_usd` values
or null are accepted but never enforced. Capability/status metadata reports
`cost_guard_enforced=false` and a null cost ceiling. Measured estimates remain
informational; unknown usage/cost does not prevent further work. These estimates
are not a statement of subscription billing charges.
7200 seconds is the recommended next limit. Five passes at A's 808.317 seconds
would total 4041.585 seconds, leaving 3158.415 seconds for variability/recovery.
Hierarchical execution has a different call mix, so this is a planning reference,
not a guarantee or a measured hierarchical runtime. No fixed deadline guarantees
success through a prolonged provider outage.
Later stages retain explicit time floors. Global stages reserve
`min(500, max(10, job_seconds / 8))` seconds per remaining global stage; pending
profile batches reserve 30 seconds each, and source/review batches 60 seconds.
These are admission/watchdog reservations, not guaranteed runtimes. Each attempt
receives only time remaining after future reservations. Insufficient remaining
reservation returns `insufficient_remaining_pass_budget`; the absolute deadline
returns `job_time_budget_exhausted`. No per-pass allowance resets the job clock.
## Recovery and checkpoints
At most three attempts per logical pass, each a fresh isolated CLI session with
the same provider, model, complete pass source and objective. Retry backoff is
2 then 4 seconds, cancellable. CLI turns start at four and can increase to five
then six only when complete-input context admission allows. No compaction,
truncation, provider fallback or partial result is accepted.
Retriable outcomes include unsuccessful/missing terminal generation, CLI turn or
structured-output retry exhaustion, rate limiting, temporary transport failures,
nonzero transient exits, and distinct pass timeouts. One additional structured
assignment/JSON repair attempt is allowed. Successful final JSON text envelopes
are recognized after the normal completion, runtime-model and isolation guards;
normal schema/membership validation still applies.
Authentication/authorization, model/isolation changes, cancellation, invalid
requests, confirmed capacity violations and deterministic evidence/review failures
are not blindly retried. Capacity or exhausted generation can select hierarchical
analysis instead of repeating the same full-suite request. Final errors identify
turn exhaustion, generation retries, output/context limits, provider failure,
pass timeout or insufficient remaining budget. Status contains attempts, terminal
metadata and retained checkpoint names, never generated content.
Every **validated** output is checkpointed in active-job memory. Identity binds
the complete request hash, credential owner/scope, routing, provider, canonical
model, configuration/prompt/policy/execution revisions, exact pass source, schema,
objective and relevant prior-output hashes. Altering any input prevents reuse.
Completed predecessors survive recovery of a later pass. A matching checkpoint
restores a copy without another model call. Contents never enter SQLite, logs,
telemetry or ordinary status; all checkpoints are removed when the job ends or
its deadline expires (at most 7200 seconds).
**Restart limitation:** checkpoints are not durable. A worker restart marks an
in-flight job `interrupted_no_retry`; no automatic provider resubmission occurs.
A fresh client attempt after restart regenerates its passes. This deliberately
uses the requested active-job recovery option instead of adding disk copies of
source-derived outputs. Results retain the existing one-hour volatile retention.
## Adaptive execution
Small requests prefer direct independent proposals, whole-suite reconciliation,
large-family review and decision audit. Initial hierarchical selection occurs for
250+ cases, 256 KiB+ canonical source, or a p95 source field of 8192+ bytes. Direct
capacity failure also selects hierarchical mode before inference. Retriable direct
generation exhaustion or confirmed output/context limits can switch modes within
the same job and remaining deadline; the transition is recorded.
Hierarchical mode:
1. Extract profiles in deterministic batches of at most 24 cases or about 48 KiB.
A single larger case remains intact and must pass capacity admission. Each
profile preserves target, success-criteria work, setup, stimulus, observations,
machinery, variations, uncertainty, exact alias and original source evidence.
2. Run two independent proposals over **every profile in the complete suite**,
with an alternative reproducible order for the second. Reconcile globally;
profile-processing batches never determine family membership.
3. Reopen all original generalized fields for every candidate family in bounded
whole-family source reviews. Verify original evidence, split unsupported
compatibility, and reject ownership drift. Only then accept natural families.
4. Perform the existing oversized-family review, decision audit, exact coverage,
unique short names and deterministic balanced five-case packaging.
Profiles have explicit size bounds; they are not arbitrary silent summarization.
No profile can be missing, invented, or silently truncated. At the current
400-case service maximum, bounded profiles plus membership-only proposal context
fit one global comparison request. A worst-case 400-alias ASCII regression admitted
profile reconciliation at 618260 input/system/schema bytes, with five reserved CLI
outputs. Every real expanded request is checked again. The bounded contract avoids
needing overlapping-neighborhood reconciliation; no such fallback is claimed.
A single full-source family that cannot fit intact still fails explicitly, as do
excessive profile/source/review batches. Transport and case ceilings remain 1 MiB
and 400, not unlimited input support.
Bounds: 32 profile batches, 16 source-review batches, eight batches per semantic
review, 64 validated logical calls and 128 model invocations including retries.
All share the same deadline. Optional metadata includes `execution_mode`,
`strategy_events`, `direct_pass_attempts`, `retry_count`, `checkpointed_passes`,
`checkpoint_reused`, profile/source batch counts, `cross_batch_review_count`,
`model_pass_count`, `natural_family_count`, `final_task_count` and per-attempt
measurements. Content-free source byte distributions now support better future
fault fixtures. Engineering rationales remain only in authorized `review_summary`.
## Verification
200 focused local tests passed. They cover isolated retries, turn adaptation,
assignment repair, non-retriable errors, preserved predecessor checkpoints,
checkpoint scope/expiry/restart semantics, global profile coverage, complete source
reopening, no fallback, shared deadlines, reserved future time, and cost estimates
exceeding the old guard without stopping work. Manifest render and client dry run
passed; Flux diff was reviewed.
Hosted native Opus 5.5 comparison used the same synthetic 14 cases, interleaved
between byte-array decoder tests and physical pulse/reset timing tests. Duplicate
text retained distinct aliases. Hierarchical mode deliberately used four profile
batches, then global comparisons, full-source review and both semantic reviews.
| Hosted execution | Seconds | Calls | Input / output tokens | CLI estimate | Natural / final groups |
| --- | ---: | ---: | --- | ---: | --- |
| Direct | 79.009 | 5 | 60130 / 9282 | USD 0.426160 | 2 / 4 |
| Hierarchical | 148.752 | 10 | 111617 / 17770 | USD 0.801868 | 2 / 4 |
Both had exact alias coverage, zero incorrect merge pairs, zero unnecessary
semantic split pairs, no retries and valid balanced 4+3 packages per family.
Descriptions kept decoder and hardware-capture objectives separate. The direct
answer inferred that fault cases varied only in stimulus; the hierarchical answer
more accurately retained uncertainty about unspecified fault patterns. This is a
small quality check, not proof of real-suite quality or runtime.
The 363-case regression used 410322 source bytes, matching the failed record, with
45 interleaved machinery patterns and distant identical records. Actual historical
per-field lengths were not retained, so that distribution cannot be matched or
claimed equivalent. Both modes completed with **mocked model answers from the
independent fixture map**, producing 45 natural families and 90 final tasks,
maximum five members, balanced sizes and unique short names. Direct used five
calls (175646 response bytes); hierarchical used 26 (224822 response bytes).
Mocked execution was about 6.6 seconds per mode; these are local orchestration
measurements, not hosted latency, cost or a model-quality result. No new hosted
363-case benchmark or real-suite rerun was performed.
Detailed [synthetic evidence](evidence/hermes_suite_recovery_20260930.json) and
[the retained safe failure record](evidence/hermes_suite_failure_90ab61bf_metadata.json)
are separate. No source fields or raw provider messages from the real suite are
included in either artifact.
## Client handoff and revisions
No new request fields. `execution.strategy="whole_suite"` still submits one
complete suite and receives one combined `result.groups` result. Authentication,
endpoint, credential scopes, pinned Opus 5.5, effort thresholds and routing remain.
```python
AI_MAX_SECONDS = 7200
AI_JOB_REVISION = "suite-adaptive-v11-20260930-1"
```
Generate a **new Idempotency-Key**. `AI_MAX_COST_USD` may remain at its existing
value if the client sends it; no client edit is needed to disable server cost
limits. Do not reuse the failed job's key.
- Configuration: `suite-v6-20260929`, HTTP compatibility unchanged.
- Execution: `suite-adaptive-v11-20260930`.
- Prompt: `implementation-proximity-adaptive-v6-20260930`.
- Grouping policy: `implementation-five-v1-20260929`, unchanged.
- Strategy: `suite-adaptive-selection-v1-20260930`.
- Capacity reservation: `suite-context-turns-v1-20260930`.
Rollback while idle: revert the recovery deployment Git commit and reconcile the
`hermes` Flux Kustomization. Do not edit live resources. Reverting reinstates the
old 3600-second maximum and cost guard. Save any desired volatile results first.

View File

@ -29,11 +29,11 @@ def main():
"""Submit once, poll safely, and report measured metadata and independent scores."""
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--size", type=int, choices=(14, 75, 363), default=75)
parser.add_argument("--max-seconds", type=int, default=3600)
parser.add_argument("--max-seconds", type=int, default=7200)
args = parser.parse_args()
request, expected = fixture(args.size)
request["routing"] = {"allow_external": True, "allowed_external_providers": ["claude"]}
request["execution"] = {"strategy": "whole_suite", "max_seconds": args.max_seconds, "max_cost_usd": 30}
request["execution"] = {"strategy": "whole_suite", "max_seconds": args.max_seconds, "max_cost_usd": None}
token = Path("/vault/secrets/synthetic-token").read_text().strip()
opener = build_opener(ProxyHandler({}))
with tempfile.TemporaryDirectory(prefix="suite-budget-probe-", dir="/jobs") as directory:
@ -62,9 +62,9 @@ def main():
"http_timeout_seconds": 45, "automatic_job_retries": 0}
try:
status, capabilities = http("/v1/capabilities")
assert status == 200 and capabilities["max_seconds"] == 3600
assert status == 200 and capabilities["max_seconds"] == 7200
assert capabilities["default_max_seconds"] == 1800
invalid = {**request, "execution": {**request["execution"], "max_seconds": 3601}}
invalid = {**request, "execution": {**request["execution"], "max_seconds": 7201}}
status, rejected = http("/v1/preflight", invalid)
assert status == 400 and rejected["error"]["code"] == "invalid_timeout"
status, checked = http("/v1/preflight", request)

View File

@ -35,7 +35,7 @@ def main():
raw, expected = fixture()
source_bytes = len(encoded(raw))
raw["routing"] = {"allow_external": True, "allowed_external_providers": ["claude"]}
raw["execution"] = {"strategy": "whole_suite", "max_seconds": 3600, "max_cost_usd": 30}
raw["execution"] = {"strategy": "whole_suite", "max_seconds": 7200}
request = validate_request(raw, ["claude"])
selected = preflight(request)
report = {"test": "synthetic_capacity_363", "source_bytes": source_bytes,
@ -61,7 +61,10 @@ def main():
result = jobs.get(job["job_id"], "synthetic-capacity-probe", result=True)
validate_result(result["result"], request)
stages = {p["stage"] for p in job["passes"]}
assert stages == {"proposal_a", "proposal_b", "reconciliation", "large_family_review", "decision_audit"}
assert {"large_family_review", "decision_audit"} <= stages
assert ({"proposal_a", "proposal_b", "reconciliation"} <= stages or
{"implementation_profiles", "profile_proposal_a", "profile_proposal_b",
"profile_reconciliation", "source_review"} <= stages)
review = result["review_summary"]
assert all(d["part_sizes"] == balanced_sizes(d["natural_case_count"]) for d in review["capacity_divisions"])
report.update(response_bytes=len(encoded(result)), exact_coverage=True,

View File

@ -0,0 +1,67 @@
#!/usr/bin/env python3
"""Compare direct and hierarchical execution on a fixed synthetic 14-case suite.
Run inside the planner with staged modules. No real source or job ID is accepted.
Reports operational metadata plus synthetic-only quality evidence. No live job DB.
"""
import json
import os
from pathlib import Path
import sys
import threading
import time
sys.path.insert(0, os.environ.get('SUITE_PROBE_MODULE_DIR', '/opt/planner'))
import suite_profiles
from suite_contract import preflight, validate_request, validate_result
from suite_multipass import generate
from suite_synthetic import PATTERNS, score
def main():
"""Use interleaved mechanisms, identical records, and cross-processing-batch peers."""
cases, expected = [], {}
for i in range(14):
family, action, assertion, setup = PATTERNS[i % 2]
alias = 'CASE-REC-%02d' % i
cases.append({'alias': alias, 'description': action,
'success_criteria': assertion, 'preconditions': setup,
'case_type': 'nominal' if i % 3 else 'fault injection'})
expected[alias] = family
cases[-2] = {**cases[0], 'alias': cases[-2]['alias']}
request = validate_request({'campaign':'SYNTHETIC', 'suite':'RECOVERY-14', 'cases':cases,
'routing':{'allow_external':True,'allowed_external_providers':['claude']},
'execution':{'strategy':'whole_suite','max_seconds':900}}, ['claude'])
suite_profiles.BATCH_CASES = 4 # Test processing-boundary independence explicitly.
for mode in ('direct','hierarchical'):
last = [0]
def progress(p):
if time.monotonic()-last[0] >= 20:
print(json.dumps({'mode':mode, **{k:p.get(k) for k in (
'current_pass','completed_model_passes','job_elapsed_seconds','retry_count')}}),
file=sys.stderr, flush=True)
last[0] = time.monotonic()
started = time.monotonic()
try:
result, metadata = generate(request, {**preflight(request),'execution_mode':mode},
threading.Event(), '192.168.22.8', progress,
credential_scope={'owner':'synthetic-recovery-probe','providers':['claude']})
validate_result(result, request)
quality = score({'groups':metadata['review_summary']['natural_families']}, expected)
print(json.dumps({'mode':mode,'status':'completed','seconds':time.monotonic()-started,
'quality':quality,'usage':metadata['usage'],'cost_usd_estimate':metadata['cost_usd_estimate'],
'passes':[{k:r.get(k) for k in ('stage','wall_seconds','cli_turns','max_turns','input_bytes')}
for r in metadata['passes']],
'natural_groups':metadata['review_summary']['natural_families'],
'final_groups':result['groups']}, sort_keys=True),flush=True)
except Exception as exc:
if hasattr(exc,'document'):
print(json.dumps({'mode':mode,'status':'failed',**exc.document()}),flush=True)
else:
print(json.dumps({'mode':mode,'status':'failed','error_type':type(exc).__name__}),flush=True)
return 1
return 0
if __name__ == '__main__':
raise SystemExit(main())

View File

@ -66,6 +66,9 @@ configMapGenerator:
- suite_sizing.py=scripts/suite_sizing.py
- suite_multipass.py=scripts/suite_multipass.py
- suite_review_batches.py=scripts/suite_review_batches.py
- suite_recovery.py=scripts/suite_recovery.py
- suite_profiles.py=scripts/suite_profiles.py
- suite_hierarchical.py=scripts/suite_hierarchical.py
- suite_contract.py=scripts/suite_contract.py
- suite_synthetic.py=scripts/suite_synthetic.py
options:

View File

@ -17,7 +17,10 @@ from suite_contract import (CLAUDE_MODELS, CLAUDE_VERSION, COST_LIMIT, EXECUTION
preflight, validate_request)
from suite_jobs import Jobs
from suite_policy import POLICY_REVISION
from suite_review_batches import MAX_MODEL_CALLS, MAX_REVIEW_BATCHES
from suite_review_batches import MAX_REVIEW_BATCHES
from suite_recovery import MAX_INVOCATIONS, MAX_ATTEMPTS
from suite_hierarchical import STRATEGY_REVISION
MAX_MODEL_CALLS = MAX_INVOCATIONS
from suite_synthetic import allowed_synthetic, fixture
@ -129,7 +132,10 @@ class Handler(BaseHTTPRequestHandler):
"execution_revision": EXECUTION_REVISION,
"policy_revision": POLICY_REVISION, "max_final_group_cases": 5,
"max_final_name_characters": 64, "model_passes": {"minimum": 3, "maximum": MAX_MODEL_CALLS},
"logical_stages": 5, "automatic_review_batching": True,
"logical_stages": 5, "execution_modes": ["direct", "hierarchical"],
"strategy_revision": STRATEGY_REVISION, "max_attempts_per_pass": MAX_ATTEMPTS,
"checkpoint_storage": "active_job_memory", "checkpoint_restart_recovery": False,
"checkpoint_retention": "until_job_end_or_deadline", "automatic_review_batching": True,
"max_batches_per_review_stage": MAX_REVIEW_BATCHES,
"review_summary_location": "GET /v1/jobs/<id>/result: top-level review_summary",
"prompt_revision": PROMPT_REVISION, "prompt_sha256": PROMPT_SHA256,
@ -142,7 +148,8 @@ class Handler(BaseHTTPRequestHandler):
"max_result_bytes": MAX_RESULT, "max_cases": MAX_CASES,
"max_seconds": MAX_JOB_SECONDS, "default_max_seconds": TIMEOUT,
"concurrency": 1, "queue": False,
"max_cost_usd": COST_LIMIT, "cost_basis": "CLI estimate guard; not subscription billing",
"max_cost_usd": COST_LIMIT, "cost_guard_enforced": False,
"cost_basis": "CLI estimate, informational only; not subscription billing",
"progress_interval_seconds": 5,
"result_retention_seconds": 3600, "idempotency_retention_seconds": 604800,
"tokenizer": None, "provider_retention_verified": False})

View File

@ -115,7 +115,6 @@ def claude_command(model, max_cost, *, reasoning=None, max_turns=CLAUDE_MAX_TURN
"--setting-sources", "", "--settings", encoded(settings).decode(), "--disable-slash-commands",
"--permission-mode", "dontAsk", "--no-chrome",
"--model", model + "[1m]", "--effort", reasoning,
"--max-budget-usd", str(max_cost),
"--max-turns", str(max_turns), "--system-prompt", SYSTEM,
"--json-schema", encoded(SCHEMA).decode()]
@ -180,17 +179,27 @@ def parse_claude(raw, expected_model, **process_info):
if event.get("type") == "result":
final = event
if not initialized or not final:
fail("incomplete_generation", "missing_init_or_final_event")
fail("pass_missing_terminal_event", "missing_init_or_final_event")
if final.get("stop_reason") == "model_context_window_exceeded":
fail("pass_context_limit", "provider_stop")
if final.get("stop_reason") == "max_tokens" or diagnostics["assistant_error_code"] == "max_output_tokens":
fail("pass_output_limit", "provider_stop")
if final.get("is_error") or final.get("subtype") != "success":
if final.get("subtype") == "error_max_budget_usd":
fail("job_cost_budget_exhausted", "cli_final_result")
fail("unexpected_cli_budget_limit", "cli_final_result")
status = final.get("api_error_status")
assistant_error = diagnostics["assistant_error_code"]
transport_error = diagnostics["cli_transport_error"]
code = "incomplete_generation"
code = "pass_provider_transient_failure" if diagnostics["assistant_api_error_seen"] else "incomplete_generation"
if code == "pass_provider_transient_failure":
diagnostics["provider_failure_transience"] = "unconfirmed"
if final.get("subtype") == "error_max_turns":
fail("pass_turn_limit", "cli_final_result")
if final.get("subtype") == "error_max_structured_output_retries":
fail("pass_schema_repair", "cli_final_result")
if status == 429 or assistant_error == "rate_limit":
code = "rate_limit"
elif status in (401, 403) or assistant_error in {"authentication_failed", "oauth_org_not_allowed"}:
elif status in (401, 403) or assistant_error in {"authentication_failed", "oauth_org_not_allowed", "account_on_hold", "billing_error", "model_not_found"}:
code = "provider_authentication"
elif transport_error == "request_timeout":
code = "backend_timeout_or_unavailable"
@ -199,8 +208,6 @@ def parse_claude(raw, expected_model, **process_info):
or (type(status) is int and 500 <= status <= 599)):
code = "backend_unavailable"
fail(code, "cli_final_result")
if final.get("stop_reason") in {"max_tokens", "model_context_window_exceeded"}:
fail("incomplete_generation", "provider_stop")
models = final.get("modelUsage", {})
if not isinstance(models, dict) or not models or set(models) - aliases:
fail("model_changed", "runtime_model")
@ -216,6 +223,10 @@ def parse_claude(raw, expected_model, **process_info):
if any(usage.get("server_tool_use", {}).values()):
fail("worker_isolation_failed", "provider_tools")
result = final.get("structured_output")
if not isinstance(result, dict) and diagnostics["final_json_text_present"]:
result = json.loads(final["result"])
diagnostics.update(structured_output_present=True, structured_output_is_object=True,
structured_output_location="result.result_json")
if not isinstance(result, dict):
fail("invalid_json_result", "structured_output_extraction")
if process_info.get("exit_code"):
@ -270,7 +281,7 @@ def claude_generate(request, cancel, *, invocation=None, progress=None, job_dead
if job_deadline is not None and now >= job_deadline:
raise Problem("job_time_budget_exhausted", 504)
if now >= deadline:
raise Problem("timeout", 504)
raise Problem("pass_timeout", 504)
activity = (root / "output").stat()
if progress and time.monotonic() >= next_progress:
# File activity proves CLI activity, not semantic completion.
@ -291,7 +302,8 @@ def claude_generate(request, cancel, *, invocation=None, progress=None, job_dead
raw = output.read(OUTPUT_BYTES).decode("utf-8", errors="replace")
info = {"exit_code": process.returncode, "subprocess_timeout_seconds": seconds,
"termination_reason": failure.code if failure else None,
"reasoning_effort": effort, "max_turns": max_turns}
"reasoning_effort": effort, "max_turns": max_turns,
"output_bytes": (root / "output").stat().st_size}
if failure:
raise Problem(failure.code, failure.status, failure_stage="subprocess",
**snapshot(raw, **info))

View File

@ -56,7 +56,7 @@ def object_text(value):
def snapshot(raw, *, exit_code=None, subprocess_timeout_seconds=None, termination_reason=None,
reasoning_effort=None, max_turns=CLAUDE_MAX_TURNS):
reasoning_effort=None, max_turns=CLAUDE_MAX_TURNS, output_bytes=None):
"""Return bounded diagnostic fields from a CLI stream, including failed runs."""
final, last_assistant = None, {}
initialized = compacted = tool_candidate = text_candidate = False
@ -106,6 +106,7 @@ def snapshot(raw, *, exit_code=None, subprocess_timeout_seconds=None, terminatio
for entry in limits.values() if isinstance(entry, dict)] if isinstance(limits, dict) else []
return {
"execution_revision": EXECUTION_REVISION,
"output_bytes": number(output_bytes) if output_bytes is not None else len(raw.encode()),
"exit_code": exit_code if type(exit_code) is int and exit_code >= 0 else None,
"termination_signal": -exit_code if type(exit_code) is int and exit_code < 0 else None,
"termination_reason": termination_reason or (

View File

@ -8,8 +8,8 @@ import re
from collections import Counter
REVISION = "suite-v6-20260929"
PROMPT_REVISION = "implementation-proximity-multipass-v5-20260930"
EXECUTION_REVISION = "suite-multipass-v10-20260930"
PROMPT_REVISION = "implementation-proximity-adaptive-v6-20260930"
EXECUTION_REVISION = "suite-adaptive-v11-20260930"
CLAUDE_VERSION = "2.1.285"
CLAUDE_MODELS = {
"claude-opus-4-8": 64000,
@ -32,8 +32,8 @@ MAX_BODY = 1 << 20
MAX_RESULT = 1 << 20
MAX_CASES = 400
TIMEOUT = 1800
MAX_JOB_SECONDS = 3600
COST_LIMIT = 30.0
MAX_JOB_SECONDS = 7200
COST_LIMIT = None # Compatibility metadata: estimates are never enforcement limits.
RESULT_TTL = 3600
FIELDS = {"description", "success_criteria", "preconditions", "operating_condition",
"case_type", "verification_method", "target", "swci", "verifies",
@ -164,10 +164,10 @@ def validate_request(raw, permissions):
if execution.get("strategy", "whole_suite") != "whole_suite":
raise Problem("unsupported_strategy", 422)
seconds = execution.get("max_seconds", TIMEOUT)
cost = execution.get("max_cost_usd", COST_LIMIT)
cost = execution.get("max_cost_usd")
if type(seconds) is not int or not 10 <= seconds <= MAX_JOB_SECONDS:
raise Problem("invalid_timeout")
if type(cost) not in (int, float) or not 0 < cost <= COST_LIMIT:
if cost is not None and (type(cost) not in (int, float) or not 0 < cost < float("inf")):
raise Problem("invalid_cost_limit")
for key in ("campaign", "suite"):
if type(raw[key]) is not str or not 1 <= len(raw[key]) <= 128:
@ -192,7 +192,7 @@ def validate_request(raw, permissions):
return {**raw, "routing": {"allow_external": external,
"allowed_external_providers": providers},
"execution": {"strategy": "whole_suite", "max_seconds": seconds,
"max_cost_usd": float(cost)}}
"max_cost_usd": float(cost) if cost is not None else None}}
def prompt(request):

View File

@ -0,0 +1,111 @@
"""Global profile comparison followed by full-source family verification."""
from __future__ import annotations
import math
from suite_contract import Problem, encoded, validate_partition
from suite_policy import BASE_NAME_LIMIT, invocation, validate_natural
from suite_profiles import batches, profile_call, profile_source, size_metadata, validate_profiles
STRATEGY_REVISION = "suite-adaptive-selection-v1-20260930"
MAX_SOURCE_BATCHES = 16
MAX_LOGICAL_CALLS = 64
FALLBACK_ERRORS = {"pass_generation_retries_exhausted", "pass_turn_limit_exhausted",
"pass_output_limit", "pass_context_limit", "pass_capacity", "pass_request_too_large",
"pass_provider_transient_failure", "pass_timeout"}
def selection(source):
"""Use content-free deterministic risk thresholds before any paid inference."""
sizes = size_metadata(source)
reasons = []
if len(source["cases"]) >= 250:
reasons.append("case_count_at_least_250")
if sizes["source_bytes"] >= 256 * 1024:
reasons.append("source_bytes_at_least_262144")
if any(sizes[f]["p95"] >= 8192 for f in ("description", "preconditions", "success_criteria")):
reasons.append("p95_field_bytes_at_least_8192")
return {"execution_mode": "hierarchical" if reasons else "direct", "strategy_reasons": reasons,
"strategy_revision": STRATEGY_REVISION, "source_size_metadata": sizes}
def minimal_partition(value):
"""Keep exact memberships; the complete profiles supply mechanism evidence."""
return {"groups": [{"members": g["members"]} for g in value["groups"]]}
def source_batches(source, partition):
"""Reopen intact candidate families, keeping all member source fields."""
by_alias = {c["alias"]: c for c in source["cases"]}
packed, current = [], []
for group in sorted(partition["groups"], key=lambda g: sorted(g["members"])):
candidate = current + [group]
records = [by_alias[a] for g in candidate for a in g["members"]]
if current and (len(records) > 80 or len(encoded(records)) > 96 * 1024):
packed.append(current)
current = []
current.append(group)
if current:
packed.append(current)
if len(packed) > MAX_SOURCE_BATCHES:
raise Problem("hierarchical_source_capacity", 422, batch_count=len(packed))
result = []
for groups in packed:
aliases = {a for g in groups for a in g["members"]}
subset = {**source, "cases": [c for c in source["cases"] if c["alias"] in aliases]}
result.append((subset, {"candidate_partition": {"groups": groups}}))
return result
def validate_source(value, source, context):
"""Require original evidence and forbid accidental cross-candidate review drift."""
validate_natural(value, source)
originals = [set(g["members"]) for g in context["candidate_partition"]["groups"]]
if any(sum(set(g["members"]) <= members for members in originals) != 1 for g in value["groups"]):
raise Problem("source_review_crossed_family_boundary", 502)
def execute(workflow, source, capacity):
"""Independent global perspectives never inherit profile-processing boundaries."""
workflow.execution_mode = "hierarchical"
workflow.max_calls = MAX_LOGICAL_CALLS
pieces = batches(source)
workflow.profile_batch_count = len(pieces)
calls = [profile_call(piece, i) for i, piece in enumerate(pieces, 1)]
for piece, call in zip(pieces, calls):
capacity(call, workflow.provider, len(piece["cases"]))
profiles = {}
source_floor = 60 * max(1, math.ceil(len(source["cases"])/80),
math.ceil(len(encoded(source["cases"]))/98304))
for i, (piece, call) in enumerate(zip(pieces, calls)):
workflow.future_seconds = (len(pieces)-i-1)*30 + 5*workflow.global_floor + source_floor
value = workflow.call("implementation_profiles", piece, custom_call=call,
validator=lambda v, p=piece: validate_profiles(v, p))
profiles.update(value["profiles"])
compact = profile_source(source, profiles)
# Each proposal sees the entire profile set in an independent order. No
# processing batch partition or prior proposal is supplied to proposal B.
workflow.future_seconds = 4*workflow.global_floor + source_floor
a = workflow.call("profile_proposal_a", compact)
from suite_multipass import ordered_request
workflow.future_seconds = 3*workflow.global_floor + source_floor
b = workflow.call("profile_proposal_b", ordered_request(compact, True))
workflow.future_seconds = 2*workflow.global_floor + source_floor
provisional = workflow.call("profile_reconciliation", compact,
{"proposal_a": minimal_partition(a), "proposal_b": minimal_partition(b)})
workflow.cross_batch_review_count += 1
checks = source_batches(source, provisional)
# Every full-source verification is admitted before any such call launches.
for piece, context in checks:
capacity(invocation("source_review", piece, context), workflow.provider,
len(piece["cases"]), len(context["candidate_partition"]["groups"]))
confirmed = {"groups": []}
for i, (piece, context) in enumerate(checks):
workflow.future_seconds = 2*workflow.global_floor + (len(checks)-i-1)*60
value = workflow.call("source_review", piece, context,
len(context["candidate_partition"]["groups"]),
validator=lambda v, p=piece, c=context: validate_source(v, p, c))
confirmed["groups"].extend(value["groups"])
validate_natural(confirmed, source)
workflow.source_review_batch_count = len(checks)
return a, b, confirmed

View File

@ -143,7 +143,9 @@ class Jobs:
with self.lock:
self._save(job_id, document)
result, metadata = suite_multipass.generate(effective, selected, event, client_ip, progress,
started=started, deadline=deadline)
started=started, deadline=deadline,
credential_scope={"owner": owner, "routing": request["routing"],
"scope_revision": "generalized-claude-v1"})
review = metadata.pop("review_summary")
document.update(metadata)
try:

View File

@ -13,9 +13,13 @@ from suite_contract import (EXECUTION_REVISION, MAX_BODY, MAX_RESULT, MODELS, PR
from suite_policy import (BASE_NAME_LIMIT, MAX_GROUP, POLICY_REVISION, invocation,
validate_natural)
from suite_sizing import cap_families, review_summary
from suite_review_batches import MAX_MODEL_CALLS, review_stage
from suite_review_batches import review_stage
from suite_recovery import Checkpoints, MAX_INVOCATIONS, allocate, recover
from suite_hierarchical import MAX_LOGICAL_CALLS, FALLBACK_ERRORS, execute as hierarchical, selection
from collections import Counter
MAX_PASSES = 5
MAX_MODEL_CALLS = MAX_INVOCATIONS
CAPACITY_REVISION = "suite-context-turns-v1-20260930"
@ -36,7 +40,9 @@ def capacity(call, provider, count, family_count=None):
input_bytes = len(call["input"].encode()) + len(call["system"].encode()) + len(encoded(call["schema"]))
if input_bytes > MAX_BODY:
raise Problem("pass_request_too_large", 413, review_pass=call["stage"], input_bytes=input_bytes)
if family_count is None:
if call["stage"] == "implementation_profiles":
estimate = 1024 + 1400 * count
elif family_count is None:
estimate = 1024 + 128 * count
else:
estimate = 1024 + 48 * count + 384 * family_count
@ -75,15 +81,27 @@ def preflight_workflow(request):
if not model["enabled"]:
reasons[provider] = "unverified_output_capacity"
continue
strategy = selection(request) if provider == "claude" else {"execution_mode": "direct"}
try:
initial = capacity(invocation("proposal_a", request), provider, count)
# Even a minimum-size reconciliation must fit; its real proposals and
# every subsequent request get checked again before any model call.
capacity(invocation("reconciliation", request), provider, count, 1)
if strategy["execution_mode"] == "direct":
try:
initial = capacity(invocation("proposal_a", request), provider, count)
capacity(invocation("reconciliation", request), provider, count, 1)
except Problem:
if provider != "claude":
raise
strategy.update(execution_mode="hierarchical", strategy_reasons=["direct_capacity"])
if strategy["execution_mode"] == "hierarchical":
from suite_profiles import batches, profile_call
pieces = batches(request)
checked = [capacity(profile_call(piece, i), provider, len(piece["cases"]))
for i, piece in enumerate(pieces, 1)]
initial = checked[0]
strategy["profile_batch_count"] = len(pieces)
except Problem as exc:
reasons[provider] = exc.code
continue
return {"provider": provider, **model, **initial, **reasoning_selection(request, provider),
return {"provider": provider, **model, **initial, **strategy, **reasoning_selection(request, provider),
"configuration_revision": REVISION,
"prompt_revision": PROMPT_REVISION, "prompt_sha256": PROMPT_SHA256,
"execution_revision": EXECUTION_REVISION, "policy_revision": POLICY_REVISION,
@ -91,6 +109,8 @@ def preflight_workflow(request):
"minimum_model_passes": 3, "maximum_model_passes": MAX_MODEL_CALLS,
"logical_stages": MAX_PASSES, "automatic_review_batching": True,
"max_final_group_cases": MAX_GROUP, "max_final_name_characters": 64,
"cost_guard_enforced": False, "maximum_attempts_per_pass": 3,
"checkpoint_storage": "active_job_memory", "checkpoint_restart_recovery": False,
"later_pass_capacity_verified": False, "later_pass_checks": "before_each_invocation"}
raise Problem("capacity_or_unsupported_backend", 422, candidates=reasons)
@ -114,130 +134,202 @@ def sum_usage(records):
class Workflow:
"""Keep content in memory and allow only one pinned provider across model passes."""
"""One pinned provider, shared deadline, isolated attempts and scoped checkpoints."""
def __init__(self, request, selected, cancel, client_ip, progress=None, *, started=None, deadline=None):
def __init__(self, request, selected, cancel, client_ip, progress=None, *, started=None,
deadline=None, credential_scope=None):
self.request, self.provider, self.cancel = request, selected["provider"], cancel
self.reasoning = selected.get("reasoning", MODELS[self.provider]["reasoning"])
self.client_ip, self.progress = client_ip, progress or (lambda _: None)
self.started = time.monotonic() if started is None else started
self.deadline = self.started + request["execution"]["max_seconds"] if deadline is None else deadline
self.cost_limit, self.spent = request["execution"]["max_cost_usd"], 0.0
self.records = []
self.last_metadata = {}
self.stage = "proposal_a"
self.max_calls = MAX_PASSES
self.spent, self.cost_known = 0.0, True
self.records, self.attempts, self.last_metadata = [], [], {}
self.stage, self.max_calls = "proposal_a", MAX_LOGICAL_CALLS
self.execution_mode = selected.get("execution_mode", "direct")
self.strategy_events = [{"mode": self.execution_mode, "reason": "preflight"}]
self.retries = Counter()
self.profile_batch_count = self.cross_batch_review_count = self.source_review_batch_count = 0
self.future_seconds = 0
self.global_floor = min(500, max(10, request["execution"]["max_seconds"] / 8))
self.checkpoints = Checkpoints(request, self.provider, credential_scope, self.deadline)
def checkpoint(self):
"""Fail closed on cancellation or exhaustion of the shared job budgets."""
"""Cancellation and time are hard limits; cost is informational only."""
if self.cancel.is_set():
raise Problem("cancelled", 409)
if time.monotonic() >= self.deadline:
raise Problem("job_time_budget_exhausted", 504)
if self.spent > self.cost_limit:
raise Problem("job_cost_budget_exhausted", 502)
if len(self.attempts) >= MAX_INVOCATIONS:
raise Problem("model_pass_limit", 502)
def call(self, stage, source, context=None, family_count=None):
"""Run one fresh complete-input invocation, retaining metadata before validation."""
def safe_state(self):
"""Never place cached outputs, profiles or rationales into status metadata."""
return {"execution_mode": self.execution_mode, "strategy_events": self.strategy_events,
"direct_pass_attempts": sum(a["execution_mode"] == "direct" for a in self.attempts),
"retry_count": dict(self.retries), "checkpointed_passes": self.checkpoints.names(),
"checkpoint_reused": list(self.checkpoints.reused),
"profile_batch_count": self.profile_batch_count,
"source_review_batch_count": self.source_review_batch_count,
"cross_batch_review_count": self.cross_batch_review_count,
"model_pass_count": len(self.attempts), "cost_guard_enforced": False,
"cost_used_usd_estimate": round(self.spent, 8) if self.cost_known else None,
"cost_limit_usd_estimate": None,
"remaining_seconds": round(max(0, self.deadline-time.monotonic()), 3)}
def call(self, stage, source, context=None, family_count=None, *, custom_call=None, validator=None):
"""Recover one logical pass without regenerating validated predecessors."""
self.stage = stage
self.checkpoint()
if len(self.records) >= self.max_calls:
raise Problem("model_pass_limit", 502)
call = invocation(stage, source, context)
# Keep the preflight choice across independent ordering and larger review prompts.
call = custom_call or invocation(stage, source, context)
call["reasoning"] = self.reasoning
if context and context.get("review_batch"):
call["review_batch"] = context["review_batch"]
started = time.monotonic()
def report(activity=None):
now = time.monotonic()
self.progress({"current_pass": stage, "completed_model_passes": len(self.records),
"maximum_model_passes": self.max_calls, "passes": self.records,
"maximum_model_passes": MAX_INVOCATIONS, "passes": list(self.records),
"attempts": list(self.attempts), **self.safe_state(),
"review_batch": (context or {}).get("review_batch"),
"heartbeat_at": time.time(), "pass_elapsed_seconds": round(now - started, 1),
"job_elapsed_seconds": round(now - self.started, 1),
"job_remaining_seconds": round(max(0, self.deadline - now), 1),
"cost_used_usd_estimate": round(self.spent, 8),
"cost_limit_usd_estimate": self.cost_limit, **(activity or {})})
"profile_batch_index": call.get("batch_index"),
"heartbeat_at": time.time(), "pass_elapsed_seconds": round(now-started, 1),
"job_elapsed_seconds": round(now-self.started, 1),
"job_remaining_seconds": round(max(0, self.deadline-now), 1),
**(activity or {})})
report({"cli_running": False, "pass_state": "capacity_preflight"})
limits = capacity(call, self.provider, len(source["cases"]), family_count)
call["max_turns"] = limits["max_turns"]
remaining_seconds = self.deadline - time.monotonic()
remaining_cost = self.cost_limit - self.spent
if remaining_cost <= 0:
raise Problem("job_cost_budget_exhausted", 502)
effective = {**source, "execution": {**source["execution"],
"max_seconds": remaining_seconds, "max_cost_usd": remaining_cost}}
if self.provider == "claude":
value, metadata = suite_backends.claude_generate(effective, self.cancel, invocation=call,
progress=report, job_deadline=self.deadline)
elif self.provider == "local":
value, metadata = suite_backends.local_generate(effective, self.cancel, self.client_ip, invocation=call)
else:
raise Problem("unsupported_backend", 422)
cost = metadata.get("cost_usd_estimate") if self.provider == "claude" else 0.0
record = {"stage": stage, "provider": self.provider, "model": MODELS[self.provider]["model"],
"reasoning": self.reasoning,
"wall_seconds": round(time.monotonic() - started, 3), **limits,
"case_count": len(source["cases"]), "review_batch": (context or {}).get("review_batch"),
"system_sha256": hashlib.sha256(call["system"].encode()).hexdigest(),
"schema_sha256": digest(call["schema"]),
"case_order_sha256": digest([c["alias"] for c in source["cases"]]),
"allocated_seconds": remaining_seconds, "allocated_cost_usd": remaining_cost,
"usage": metadata.get("usage"), "cli_turns": metadata.get("turns"),
"duration_api_ms": metadata.get("duration_api_ms"),
"cost_usd_estimate": cost, "cli_diagnostics": metadata.get("cli_diagnostics")}
self.records.append(record)
self.last_metadata = metadata
if type(cost) not in (int, float) or not math.isfinite(cost) or cost < 0:
raise Problem("budget_accounting_unavailable", 502)
self.spent += cost
def validate(value):
if validator:
return validator(value)
if stage in {"proposal_a", "proposal_b", "profile_proposal_a", "profile_proposal_b", "profile_reconciliation"}:
return validate_partition(value, source, name_limit=BASE_NAME_LIMIT, unique_names=False)
originals = None
if stage in {"large_family_review", "decision_audit"}:
originals = context.get("original_partition", context.get("natural_partition"))["groups"]
return validate_natural(value, source, originals)
before = len(self.records)
value = recover(self, call, source, limits, validate, report)
if len(self.records) == before and self.attempts:
# Only a successfully validated output becomes a completed model pass.
successful = next((a for a in reversed(self.attempts) if a["stage"] == stage and a["status"] == "generated"), None)
if successful:
successful["status"] = "validated"
self.records.append(dict(successful))
report({"cli_running": False, "pass_state": "validated"})
return value
def invoke(self, call, source, limits, attempt, report):
"""Account for successful and failed attempts before returning any output."""
self.checkpoint()
report({"cli_running": False})
return decode(value, call, source)
seconds = allocate(self)
effective = {**source, "execution": {**source["execution"], "max_seconds": seconds,
"max_cost_usd": None}}
started = time.monotonic()
record = {"stage": call["stage"], "attempt": attempt, "execution_mode": self.execution_mode,
"provider": self.provider, "model": MODELS[self.provider]["model"],
"reasoning": self.reasoning, **limits, "case_count": len(source["cases"]),
"review_batch": call.get("review_batch"), "profile_batch_index": call.get("batch_index"),
"allocated_seconds": seconds, "reserved_future_seconds": self.future_seconds,
"allocated_cost_usd": None, "system_sha256": hashlib.sha256(call["system"].encode()).hexdigest(),
"schema_sha256": digest(call["schema"]),
"case_order_sha256": digest([c["alias"] for c in source["cases"]])}
metadata, error = {}, None
try:
if self.provider == "claude":
value, metadata = suite_backends.claude_generate(effective, self.cancel, invocation=call,
progress=report, job_deadline=self.deadline)
elif self.provider == "local":
value, metadata = suite_backends.local_generate(effective, self.cancel, self.client_ip, invocation=call)
else:
raise Problem("unsupported_backend", 422)
record["status"] = "generated"
except Problem as exc:
error = exc
metadata = {**exc.details, "cli_diagnostics": exc.details}
record.update(status="failed", error_code=exc.code)
finally:
cost = metadata.get("cost_usd_estimate") if self.provider == "claude" else 0.0
if type(cost) in (float, int) and math.isfinite(cost) and cost >= 0:
self.spent += cost
else:
self.cost_known = False
cost = None
record.update(wall_seconds=round(time.monotonic()-started, 3), usage=metadata.get("usage"),
cli_turns=metadata.get("turns"), duration_api_ms=metadata.get("duration_api_ms"),
cost_usd_estimate=cost, cli_diagnostics=metadata.get("cli_diagnostics"))
self.attempts.append(record)
self.last_metadata = metadata
if error:
raise error
self.checkpoint()
try:
return value if call["stage"] == "implementation_profiles" else decode(value, call, source)
except Problem as exc:
record.update(status="failed", error_code=exc.code)
raise
def direct(self, source):
"""Reserve a viable time floor for every subsequent required stage."""
self.future_seconds = 4*self.global_floor
a = self.call("proposal_a", source)
self.future_seconds = 3*self.global_floor
b = self.call("proposal_b", ordered_request(source, True))
self.future_seconds = 2*self.global_floor
c = self.call("reconciliation", source, {"proposal_a": a, "proposal_b": b},
max(len(a["groups"]), len(b["groups"])))
return a, b, c
def execute(self):
"""Perform independent discovery, reconciliation, and bounded semantic reviews."""
"""Choose direct or hierarchical analysis, then review and size once."""
source = ordered_request(self.request)
a = self.call("proposal_a", source)
validate_partition(a, source, name_limit=BASE_NAME_LIMIT, unique_names=False)
b = self.call("proposal_b", ordered_request(source, True))
validate_partition(b, source, name_limit=BASE_NAME_LIMIT, unique_names=False)
count = max(len(a["groups"]), len(b["groups"]))
reconciled = self.call("reconciliation", source, {"proposal_a": a, "proposal_b": b}, count)
validate_natural(reconciled, source)
if self.execution_mode == "direct":
try:
a, b, reconciled = self.direct(source)
except Problem as exc:
if self.provider != "claude" or exc.code not in FALLBACK_ERRORS:
raise
self.strategy_events.append({"mode": "hierarchical", "reason": exc.code, "after_pass": self.stage})
a, b, reconciled = hierarchical(self, source, capacity)
else:
a, b, reconciled = hierarchical(self, source, capacity)
large = [g for g in reconciled["groups"] if len(g["members"]) > MAX_GROUP]
reviewed = audited = None
if large:
self.future_seconds = self.global_floor
reviewed = review_stage(self, "large_family_review", source, reconciled, None, capacity)
self.future_seconds = 0
audited = review_stage(self, "decision_audit", source, reconciled, reviewed, capacity)
natural = audited or reconciled
final, divisions = cap_families(natural, source)
review = review_summary(a, b, reconciled, reviewed, audited, final, divisions)
self.checkpoint()
metadata = {**self.last_metadata, "passes": self.records,
"reasoning": self.reasoning,
"model_pass_count": len(self.records), "usage": sum_usage(self.records),
"cli_diagnostics_scope": "last_model_pass",
"duration_api_ms": sum(r["duration_api_ms"] for r in self.records) if all(type(r["duration_api_ms"]) in (int, float) for r in self.records) else None,
"cost_usd_estimate": round(self.spent, 8),
"turns": sum(r["cli_turns"] for r in self.records) if all(type(r["cli_turns"]) is int for r in self.records) else None,
metadata = {**self.last_metadata, **self.safe_state(), "passes": self.records, "attempts": self.attempts,
"reasoning": self.reasoning, "usage": sum_usage(self.attempts),
"cli_diagnostics_scope": "last_model_attempt", "cost_usd_estimate": self.safe_state()["cost_used_usd_estimate"],
"turns": sum(a["cli_turns"] for a in self.attempts) if all(type(a["cli_turns"]) is int for a in self.attempts) else None,
"natural_family_count": len(natural["groups"]), "final_task_count": len(final["groups"]),
"singleton_statistics": {k: v for k, v in review["counts"].items() if "singleton" in k},
"singleton_statistics": {k:v for k,v in review["counts"].items() if "singleton" in k},
"policy_revision": POLICY_REVISION, "review_summary": review}
if len(encoded({"result": final, **metadata})) > MAX_RESULT - 16384:
raise Problem("response_too_large", 502)
return final, metadata
def generate(request, selected, cancel, client_ip, progress=None, *, started=None, deadline=None):
"""Never expose partial partitions as completion or broaden a failed route."""
workflow = Workflow(request, selected, cancel, client_ip, progress, started=started, deadline=deadline)
def generate(request, selected, cancel, client_ip, progress=None, *, started=None, deadline=None, credential_scope=None):
"""No partial completion, provider fallback, or persistent source-derived cache."""
workflow = Workflow(request, selected, cancel, client_ip, progress, started=started,
deadline=deadline, credential_scope=credential_scope)
try:
return workflow.execute()
except Problem as exc:
details = {"failure_stage": "multi_pass_orchestration", **exc.details, "review_pass": workflow.stage,
"completed_model_passes": len(workflow.records), "passes": workflow.records,
"aggregate_usage": sum_usage(workflow.records + ([{"usage": exc.details["usage"]}] if isinstance(exc.details.get("usage"), dict) else [])),
"aggregate_cost_usd_estimate": None if exc.code == "budget_accounting_unavailable" else round(workflow.spent, 8)}
if exc.details.get("cost_usd_estimate") is not None:
details["aggregate_cost_usd_estimate"] += exc.details["cost_usd_estimate"]
details = {"failure_stage": "multi_pass_orchestration", **exc.details, **workflow.safe_state(),
"review_pass": workflow.stage, "completed_model_passes": len(workflow.records),
"passes": workflow.records, "attempts_metadata": workflow.attempts,
"aggregate_usage": sum_usage(workflow.attempts),
"aggregate_cost_usd_estimate": workflow.safe_state()["cost_used_usd_estimate"]}
raise Problem(exc.code, exc.status, **details) from None
finally:
workflow.checkpoints.clear()

View File

@ -98,10 +98,13 @@ REVIEW_SCHEMA["properties"]["decisions"] = {"type": "array", "minItems": 1, "ite
def invocation(stage, request, context=None):
"""Build one full-input pass; independent discovery has no proposal context."""
original_stage = stage
stage = {"profile_proposal_a": "proposal_a", "profile_proposal_b": "proposal_b",
"profile_reconciliation": "proposal_a", "source_review": "reconciliation"}.get(stage, stage)
if stage not in STAGES:
raise ValueError("unknown internal stage")
discovery = stage in STAGES[:2]
if discovery and context:
if discovery and context and original_stage != "profile_reconciliation":
raise ValueError("discovery must be independent")
schema = DISCOVERY_SCHEMA if discovery else NATURAL_SCHEMA if stage == "reconciliation" else REVIEW_SCHEMA
instruction = DISCOVERY if discovery else RECONCILE
@ -109,6 +112,21 @@ def invocation(stage, request, context=None):
instruction += "\n" + REVIEW
if stage == "decision_audit":
instruction += "\n" + REVIEW + "\n" + AUDIT
if request.get("_profile_source"):
instruction += ("\nThe supplied cases are compact implementation profiles, not verbatim source. "
"Compare across ALL profiles and batches. Original fields will be reopened "
"before accepting natural families. Retain uncertainty.")
if original_stage == "profile_reconciliation":
instruction += ("\nReconcile the two supplied independent profile partitions across every alias. "
"Compare memberships and machinery, not votes or connected components.")
if original_stage == "source_review":
instruction = instruction.replace(
"Independently reassess two complete candidate partitions using ALL original cases.",
"Reassess the supplied candidate partition using ALL original cases in this review.")
instruction += ("\nReopen ALL original fields for every candidate family supplied here. "
"Reject compression-induced compatibility; split genuinely different machinery. "
"Do not cross candidate-family boundaries in this source verification pass. "
"Use original source evidence, not profile quotations.")
source = {k: request[k] for k in ("campaign", "suite", "cases")}
context = dict(context or {})
if context.get("review_batch"):
@ -123,7 +141,7 @@ def invocation(stage, request, context=None):
keys = decision_keys(context) if stage in STAGES[3:] else {}
if keys:
context["review_keys"] = keys
return {"stage": stage,
return {"stage": original_stage,
"schema": wire_schema(schema, request, context), "review_keys": keys,
"input": encoded({"suite": source, "review_material": context}).decode(),
"system": SYSTEM + "\n\nPASS INSTRUCTIONS:\n" + instruction + "\n\n" + WIRE_INSTRUCTIONS}

View File

@ -0,0 +1,108 @@
"""Compact implementation profiles with exact aliases and source-grounded evidence."""
from __future__ import annotations
from suite_contract import SYSTEM, Problem, encoded
from suite_policy import evidence
PROFILE_FIELDS = {"target": 64, "work": 192, "setup": 96, "stimulus": 96,
"observations": 128, "machinery": 96, "variations": 96, "uncertainty": 96}
PROFILE_BYTES = 1152
BATCH_CASES = 24
BATCH_BYTES = 48 * 1024
MAX_PROFILE_BATCHES = 32
PROFILE_INSTRUCTIONS = (
"Extract a compact implementation profile for EVERY alias. These processing batches "
"are NOT families. Work must capture primary success-criteria implications; setup "
"preserves material preconditions; stimulus captures controls/actions; observations "
"captures measurement/assertions; machinery identifies special test mechanisms; "
"variations and uncertainty preserve qualitative distinctions and missing details. "
"Use null for unknowns. State only supplied facts, no invented implementation. "
"Keep profiles concise, under 1152 UTF-8 bytes each including evidence. Include one "
"short exact evidence quote from success_criteria if present, otherwise description. "
"Do not drop distinctions to force a merge. Output profiles keyed by exact alias."
)
def batches(source):
"""Pack complete cases deterministically; never split or truncate one case."""
result, current = [], []
for case in sorted(source["cases"], key=lambda c: c["alias"]):
if current and (len(current) >= BATCH_CASES or len(encoded(current + [case])) > BATCH_BYTES):
result.append({**source, "cases": current})
current = []
current.append(case)
if current:
result.append({**source, "cases": current})
if len(result) > MAX_PROFILE_BATCHES:
raise Problem("profile_batch_capacity", 422, batch_count=len(result))
return result
def profile_call(source, index):
"""Supply every generalized source field and require one result per alias."""
item = {"type": "object", "additionalProperties": False,
"required": list(PROFILE_FIELDS) + ["evidence"], "properties": {
k: {"type": ["string", "null"], "maxLength": size} for k, size in PROFILE_FIELDS.items()}}
item["properties"]["evidence"] = {"type": "object", "additionalProperties": False,
"required": ["field", "quote"], "properties": {
"field": {"enum": ["success_criteria", "description"]},
"quote": {"type": "string", "minLength": 1, "maxLength": 96}}}
aliases = sorted(c["alias"] for c in source["cases"])
return {"stage": "implementation_profiles", "batch_index": index,
"input": encoded({k: source[k] for k in ("campaign", "suite", "cases")}).decode(),
"system": SYSTEM + "\n" + PROFILE_INSTRUCTIONS,
"schema": {"type": "object", "additionalProperties": False, "required": ["profiles"],
"properties": {"profiles": {"type": "object", "additionalProperties": False,
"required": aliases, "properties": {a: item for a in aliases}}}}}
def validate_profiles(value, source):
"""Check profile coverage, size, nullable fields and exact source evidence."""
by_alias = {c["alias"]: c for c in source["cases"]}
if not isinstance(value, dict) or set(value) != {"profiles"} or not isinstance(value["profiles"], dict):
raise Problem("invalid_profile_output", 502)
if set(value["profiles"]) != set(by_alias):
raise Problem("invalid_case_assignments", 502, failure_stage="profile_aliases")
for alias, profile in value["profiles"].items():
if not isinstance(profile, dict) or set(profile) != set(PROFILE_FIELDS) | {"evidence"}:
raise Problem("invalid_profile_output", 502)
for key, limit in PROFILE_FIELDS.items():
text = profile[key]
if text is not None and (type(text) is not str or len(text) > limit):
raise Problem("invalid_profile_output", 502)
item = profile["evidence"]
if not isinstance(item, dict) or set(item) != {"field", "quote"}:
raise Problem("invalid_profile_output", 502)
expected = "success_criteria" if by_alias[alias].get("success_criteria") else "description"
if item["field"] != expected:
raise Problem("invalid_review_evidence", 502)
evidence([{**item, "alias": alias}], {alias}, by_alias)
if len(encoded(profile)) > PROFILE_BYTES:
raise Problem("profile_capacity", 422)
return value
def profile_source(source, profiles):
"""Use compact fields for global comparison; originals are reopened afterwards."""
cases = []
for alias in sorted(profiles):
p = profiles[alias]
def join(keys):
return "; ".join(k + ": " + (p[k] or "unknown") for k in keys)
cases.append({"alias": alias, "description": join(("target", "stimulus")),
"success_criteria": join(("work", "observations", "machinery")),
"preconditions": join(("setup",)),
"operating_condition": join(("variations", "uncertainty"))})
if any(len(encoded(c)) > 1280 for c in cases):
raise Problem("profile_capacity", 422)
return {**source, "cases": cases, "_profile_source": True}
def size_metadata(source):
"""Retain byte distributions only, allowing future representative fault tests."""
result = {"source_bytes": len(encoded({k: source[k] for k in ("campaign", "suite", "cases")}))}
for field in ("description", "preconditions", "case_type", "success_criteria"):
values = sorted(len((c.get(field) or "").encode()) for c in source["cases"])
result[field] = {"min": min(values), "max": max(values), "median": values[len(values)//2],
"p95": values[int((len(values)-1)*.95)]}
return result

View File

@ -0,0 +1,124 @@
"""Bounded isolated retries and volatile, identity-bound validated checkpoints."""
from __future__ import annotations
import copy
import time
from collections import Counter
from suite_contract import (EXECUTION_REVISION, MODELS, PROMPT_REVISION, REVISION,
Problem, digest)
from suite_policy import POLICY_REVISION
MAX_ATTEMPTS = 3
MAX_INVOCATIONS = 128
RETRIABLE = {
"incomplete_generation", "pass_missing_terminal_event", "pass_turn_limit",
"pass_schema_repair", "pass_provider_transient_failure", "rate_limit",
"backend_unavailable", "backend_timeout_or_unavailable", "pass_timeout", "timeout",
}
REPAIRABLE = {"invalid_json_result", "invalid_case_assignments"}
NONREPAIR_STAGES = {"runtime_model", "runtime_model_limits", "cli_initialization", "cli_usage"}
class Checkpoints:
"""Keep validated derived outputs only in this job's memory, until its deadline.
No disk persistence, cross-job reuse, restart recovery or status content access.
Complete call/context hashes include relevant prior-pass outputs.
"""
def __init__(self, request, provider, scope, deadline):
self.identity = {"request": digest(request), "provider": provider,
"model": MODELS[provider]["model"], "routing": request["routing"],
"credential_scope": scope, "configuration": REVISION,
"prompt": PROMPT_REVISION, "policy": POLICY_REVISION,
"execution": EXECUTION_REVISION}
self.deadline, self.values, self.reused = deadline, {}, []
def key(self, call):
"""Bind the whole source, schema, objective and prior results to each pass."""
return digest({**self.identity, "call": call})
def get(self, key, stage):
"""Return an isolated copy; expired checkpoints can never be restored."""
if time.monotonic() >= self.deadline:
self.values.clear()
raise Problem("job_time_budget_exhausted", 504)
if key not in self.values:
return None
self.reused.append(stage)
return copy.deepcopy(self.values[key][1])
def put(self, key, stage, value):
self.values[key] = (stage, copy.deepcopy(value))
def names(self):
return sorted({stage for stage, _ in self.values.values()})
def clear(self):
self.values.clear()
def recover(workflow, call, source, limits, validate, report):
"""Retry only eligible failures; every attempt is fresh and fully accounted."""
stage = call["stage"]
key = workflow.checkpoints.key(call)
cached = workflow.checkpoints.get(key, stage)
if cached is not None:
validate(cached)
return cached
repair_used = False
for attempt in range(1, MAX_ATTEMPTS + 1):
workflow.checkpoint()
turns = min(limits["max_turns"], 3 + attempt)
actual = {**call, "max_turns": turns}
if attempt > 1:
# No prior generated content is fed back; only the unchanged objective
# and complete source enter a new isolated CLI session.
report({"retry_attempt": attempt, "retry_reason": previous.code, "cli_running": False})
if workflow.cancel.wait(min(10, 2 ** (attempt - 1))):
raise Problem("cancelled", 409)
try:
value = workflow.invoke(actual, source, {**limits, "max_turns": turns}, attempt, report)
validate(value)
except Problem as exc:
previous = exc
if workflow.attempts and workflow.attempts[-1]["stage"] == stage:
workflow.attempts[-1].update(status="failed", error_code=exc.code)
can_repair = (exc.code in REPAIRABLE and not repair_used
and exc.details.get("failure_stage") not in NONREPAIR_STAGES)
retry = exc.code in RETRIABLE or can_repair
if not retry:
raise
if can_repair:
repair_used = True
if exc.code == "pass_turn_limit" and turns >= limits["max_turns"]:
code = "pass_turn_limit_exhausted"
break
if attempt == MAX_ATTEMPTS:
code = ("pass_turn_limit_exhausted" if exc.code == "pass_turn_limit" else
"pass_provider_transient_failure" if exc.code in {
"rate_limit", "backend_unavailable", "backend_timeout_or_unavailable",
"pass_provider_transient_failure"} else
"pass_timeout" if exc.code in {"timeout", "pass_timeout"} else
"pass_generation_retries_exhausted")
break
workflow.retries[stage] += 1
continue
workflow.checkpoints.put(key, stage, value)
report({"cli_running": False, "pass_state": "validated"})
return value
raise Problem(code, 502, **{**previous.details, "attempts": attempt,
"last_error_code": previous.code, "review_pass": stage}) from None
def allocate(workflow):
"""Protect explicit later-pass time floors within the single absolute deadline."""
remaining = workflow.deadline - time.monotonic()
reserved = workflow.future_seconds
if remaining <= 0:
raise Problem("job_time_budget_exhausted", 504)
if remaining - reserved < 1:
raise Problem("insufficient_remaining_pass_budget", 422,
remaining_seconds=round(remaining, 3), reserved_future_seconds=reserved)
return remaining - reserved

View File

@ -3,7 +3,7 @@ from suite_contract import Problem
from suite_policy import MAX_GROUP, invocation, validate_natural
MAX_REVIEW_BATCHES = 8
MAX_MODEL_CALLS = 3 + 2 * MAX_REVIEW_BATCHES
MAX_MODEL_CALLS = 64
def material(stage, original, reviewed=None):
@ -75,15 +75,18 @@ def review_stage(workflow, stage, source, original, reviewed, capacity):
# Admit every batch before launching any paid review call.
for subset, context, count in calls:
capacity(invocation(stage, subset, context), workflow.provider, len(subset["cases"]), count)
workflow.max_calls += len(calls) - 1
workflow.max_calls = MAX_MODEL_CALLS
if workflow.max_calls > MAX_MODEL_CALLS:
raise Problem("model_pass_limit", 502)
combined = {"groups": [g for g in original["groups"] if len(g["members"]) <= MAX_GROUP], "decisions": []}
for originals, (subset, context, count) in zip(batches, calls):
base_reserve = workflow.future_seconds
for index, (originals, (subset, context, count)) in enumerate(zip(batches, calls)):
workflow.future_seconds = base_reserve + (len(calls)-index-1)*60
value = workflow.call(stage, subset, context, count)
validate_natural(value, subset, originals)
combined["groups"].extend(value["groups"])
combined["decisions"].extend(value["decisions"])
workflow.future_seconds = base_reserve
# Recheck global naming, membership, boundaries and all original decisions.
validate_natural(combined, source, original["groups"])
return combined

View File

@ -36,7 +36,7 @@ spec:
app: hermes-suite-planner
annotations:
fluentbit.io/exclude: "true"
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v10-20260930
ai.bstein.dev/config-rev: suite-v6-adaptive-v11-20260930
vault.hashicorp.com/agent-inject: "true"
vault.hashicorp.com/agent-pre-populate-only: "true"
vault.hashicorp.com/agent-init-first: "true"

View File

@ -114,8 +114,8 @@ def test_automatic_review_batches_preserve_global_discovery_and_full_validation(
assert calls[3]["aliases"].isdisjoint(calls[4]["aliases"])
assert calls[3]["aliases"] == calls[5]["aliases"] and calls[4]["aliases"] == calls[6]["aliases"]
assert len({c["deadline"] for c in calls}) == 1
assert calls[-1]["execution"]["max_cost_usd"] == pytest.approx(29.4)
assert updates[-1]["maximum_model_passes"] == 7
assert all(c['execution']['max_cost_usd'] is None for c in calls)
assert updates[-1]["maximum_model_passes"] == 128
assert metadata["model_pass_count"] == 7 and len(result["groups"]) == 4
validate_result(result, source)

View File

@ -42,9 +42,9 @@ def request():
@pytest.mark.parametrize("subtype,code", [
("error_max_turns", "incomplete_generation"),
("error_max_structured_output_retries", "incomplete_generation"),
("error_max_budget_usd", "job_cost_budget_exhausted"),
("error_max_turns", "pass_turn_limit"),
("error_max_structured_output_retries", "pass_schema_repair"),
("error_max_budget_usd", "unexpected_cli_budget_limit"),
("error_during_execution", "incomplete_generation"),
(CANARY, "incomplete_generation"),
])
@ -75,8 +75,8 @@ def test_provider_error_status_is_preserved(status, code):
("authentication_failed", "provider_authentication"),
("oauth_org_not_allowed", "provider_authentication"),
("rate_limit", "rate_limit"), ("server_error", "backend_unavailable"),
("overloaded", "backend_unavailable"), ("unknown", "incomplete_generation"),
(CANARY, "incomplete_generation"),
("overloaded", "backend_unavailable"), ("unknown", "pass_provider_transient_failure"),
(CANARY, "pass_provider_transient_failure"),
])
def test_error_flag_on_success_subtype_uses_safe_assistant_category(category, code):
"""A result subtype alone does not mean the CLI completed successfully."""
@ -101,7 +101,7 @@ def test_error_flag_on_success_subtype_uses_safe_assistant_category(category, co
True, "connection_refused", "backend_unavailable"),
("API Error: Request timed out.", True, "request_timeout", "backend_timeout_or_unavailable"),
("API Error: Connection error.", False, None, "incomplete_generation"),
("API Error: Connection error. " + CANARY, True, None, "incomplete_generation"),
("API Error: Connection error. " + CANARY, True, None, "pass_provider_transient_failure"),
])
def test_transport_signatures_require_exact_cli_error_marker(text, marked, transport, code):
"""Provider text never becomes a diagnostic string or a substring classifier."""
@ -121,7 +121,7 @@ def test_reasoning_effort_reports_actual_configuration_or_unknown(effort):
@pytest.mark.parametrize("reason", ["max_tokens", "model_context_window_exceeded"])
def test_explicit_provider_stop_distinct_from_turn_limit(reason):
with pytest.raises(Problem, match="incomplete_generation") as raised:
with pytest.raises(Problem, match="pass_output_limit|pass_context_limit") as raised:
suite_backends.parse_claude(envelope(stop_reason=reason), MODEL, exit_code=0)
assert raised.value.details["failure_stage"] == "provider_stop"
assert raised.value.details["provider_stop_reason"] == reason
@ -129,8 +129,8 @@ def test_explicit_provider_stop_distinct_from_turn_limit(reason):
@pytest.mark.parametrize("raw,stage,code", [
("", "missing_init_or_final_event", "incomplete_generation"),
(json.dumps({"type": "system", "subtype": "init", "model": MODEL}), "missing_init_or_final_event", "incomplete_generation"),
("", "missing_init_or_final_event", "pass_missing_terminal_event"),
(json.dumps({"type": "system", "subtype": "init", "model": MODEL}), "missing_init_or_final_event", "pass_missing_terminal_event"),
("not json " + CANARY, "cli_event_json", "invalid_json_result"),
("[]", "cli_event_json", "invalid_json_result"),
])
@ -147,13 +147,10 @@ def test_text_or_tool_candidate_is_detected_but_not_silently_accepted():
{"type": "tool_use", "name": "StructuredOutput", "input": value},
{"type": "text", "text": json.dumps(value)}]}}
raw = json.dumps(assistant) + "\n" + envelope(structured_output=None, result=json.dumps(value))
with pytest.raises(Problem, match="invalid_json_result") as raised:
suite_backends.parse_claude(raw, MODEL, exit_code=0)
details = raised.value.details
assert details["failure_stage"] == "structured_output_extraction"
assert details["assistant_structured_tool_input_present"] is True
assert details["assistant_json_text_present"] is True
assert details["final_json_text_present"] is True
result, metadata = suite_backends.parse_claude(raw, MODEL, exit_code=0)
assert result == value
details = metadata['cli_diagnostics']
assert details['structured_output_location'] == 'result.result_json'
assert CANARY not in json.dumps(details)
@ -207,7 +204,7 @@ def test_validation_failure_retains_usage_without_accepting_content(tmp_path, mo
("success", None, None), ("nonzero", "incomplete_generation", "process_exit"),
("signal", "incomplete_generation", "process_exit"),
("empty", "backend_unavailable", "process_exit"),
("timeout", "timeout", "subprocess"), ("cancel", "cancelled", "subprocess"),
("timeout", "pass_timeout", "subprocess"), ("cancel", "cancelled", "subprocess"),
("job_timeout", "job_time_budget_exhausted", "subprocess"),
("oversize", "response_too_large", "subprocess"),
("start", "backend_unavailable", "process_start"),
@ -273,7 +270,7 @@ def test_failed_cli_usage_survives_job_cleanup_without_content(tmp_path, monkeyp
jobs.run(document["job_id"], "owner", value, selected, "192.168.22.8")
failed = jobs.get(document["job_id"], "owner")
assert failed["error"]["details"]["turn_limit_reached"] is True
assert failed["usage"]["output_tokens"] == 50
assert failed["usage"]["output_tokens"] == 50 * len(failed["error"]["details"]["attempts_metadata"])
assert not jobs.results and not jobs.active
replay, created = jobs.submit("owner", "synthetic-failure", value, selected, "192.168.22.8", launch=False)
assert not created and replay["job_id"] == document["job_id"]

View File

@ -8,11 +8,12 @@ import pytest
import suite_backends
import suite_jobs
import suite_multipass
import suite_recovery
from suite_contract import MAX_JOB_SECONDS, Problem, TIMEOUT, preflight, validate_request
from test_suite_multipass import install_backend, natural, request
@pytest.mark.parametrize("seconds", [1801, 2700, 3600])
@pytest.mark.parametrize("seconds", [1801, 2700, 3600, 7200])
def test_explicit_long_budget_matches_published_schema(seconds):
"""Admission accepts opt-in durations; omitted limits retain their old default."""
source = request(7)
@ -23,10 +24,10 @@ def test_explicit_long_budget_matches_published_schema(seconds):
schema = json.loads((Path(__file__).resolve().parents[2] /
"docs/contracts/suite_planning_request.schema.json").read_text())
limit = schema["properties"]["execution"]["properties"]["max_seconds"]
assert limit["maximum"] == MAX_JOB_SECONDS == 3600 and limit["default"] == TIMEOUT
assert limit["maximum"] == MAX_JOB_SECONDS == 7200 and limit["default"] == TIMEOUT
@pytest.mark.parametrize("seconds", [3601, 3600.1, True, "3600", 9])
@pytest.mark.parametrize("seconds", [7201, 7200.1, True, "7200", 9])
def test_invalid_budget_fails_before_inference(seconds):
source = request(7)
source["execution"]["max_seconds"] = seconds
@ -62,7 +63,7 @@ def test_job_passes_1800_seconds_with_one_absolute_deadline(tmp_path, monkeypatc
jobs.run(job["job_id"], "owner", source, selected, "192.168.22.8")
result = jobs.get(job["job_id"], "owner")
assert result["status"] == "completed" and result["wall_seconds"] == 2510
assert allocations == [3590, 3090, 2590, 2090, 1590]
assert allocations == [1795, 1743.75, 1692.5, 1641.25, 1590]
assert deadlines == [3700] * 5
assert result["execution_progress"]["job_remaining_seconds"] == 1090
assert result["execution_progress"]["job_elapsed_seconds"] == 2510
@ -76,7 +77,9 @@ def test_distinct_backend_timeout_is_not_reclassified(monkeypatch, provider_erro
def fail(*_, **__):
raise Problem(provider_error, 504, failure_stage="subprocess")
monkeypatch.setattr(suite_backends, "claude_generate", fail)
with pytest.raises(Problem, match="^" + provider_error + "$"):
monkeypatch.setattr(suite_multipass, "FALLBACK_ERRORS", set())
monkeypatch.setattr(suite_recovery, "MAX_ATTEMPTS", 1)
with pytest.raises(Problem, match="^(pass_provider_transient_failure|pass_timeout)$"):
suite_multipass.generate(source, preflight(source), threading.Event(), "192.168.22.8")
@ -104,7 +107,7 @@ def test_watchdog_job_expiry_reaches_status_with_zero_remaining(tmp_path, monkey
monkeypatch.setattr(suite_jobs.time, "monotonic", lambda: clock[0])
monkeypatch.setattr(suite_backends, "switchyard_decision", lambda _: None)
def expired(value, cancel, *, invocation, progress, job_deadline):
assert value["execution"]["max_seconds"] == 3600
assert value["execution"]["max_seconds"] == 1800
clock[0] = job_deadline
raise Problem("job_time_budget_exhausted", 504, failure_stage="subprocess")
monkeypatch.setattr(suite_backends, "claude_generate", expired)

View File

@ -119,8 +119,8 @@ def test_coherent_families_are_balanced_not_semantically_fragmented(monkeypatch,
assert [g['name'] for g in result['groups']] == [f'Reset recovery ({i}/{len(sizes)})' for i in range(1,len(sizes)+1)]
assert all('Work-size part' in g['description'] for g in result['groups'])
assert metadata['review_summary']['decision_audit'][0]['decision'] == 'keep'
assert [call[0]['execution']['max_cost_usd'] for call in calls] == pytest.approx([30,29.9,29.8,29.7,29.6])
assert all(calls[i+1][0]['execution']['max_seconds'] < calls[i][0]['execution']['max_seconds'] for i in range(4))
assert [call[0]['execution']['max_cost_usd'] for call in calls] == [None]*5
assert all(0 < c[0]['execution']['max_seconds'] <= source['execution']['max_seconds'] for c in calls)
validate_result(result, source)
@ -163,7 +163,7 @@ def test_progress_reports_activity_without_content_or_false_percentage(monkeypat
active = [value for value in updates if value.get('cli_running')]
assert len(active) == 5
assert [v['completed_model_passes'] for v in active] == [0,1,2,3,4]
assert all(v['maximum_model_passes'] == 5 and v['cli_output_bytes'] == 120 for v in active)
assert all(v['maximum_model_passes'] == 128 and v['cli_output_bytes'] == 120 for v in active)
assert all(v['heartbeat_at'] > 0 and v['job_remaining_seconds'] <= 1800 for v in active)
assert all(v['last_cli_activity_seconds_ago'] == 0 for v in active)
assert updates[-1]['cli_running'] is False
@ -177,19 +177,18 @@ def test_one_hour_opt_in_retains_thirty_minute_default():
assert validate_request(source, ['claude'])['execution']['max_seconds'] == 900
source['execution']['max_seconds'] = 3600
assert validate_request(source, ['claude'])['execution']['max_seconds'] == 3600
source['execution']['max_seconds'] = 3601
source['execution']['max_seconds'] = 7201
with pytest.raises(Problem, match='invalid_timeout'):
validate_request(source, ['claude'])
def test_cli_estimate_guard_default_and_client_override():
source = request(7)
assert source['execution']['max_cost_usd'] == 30
assert source['execution']['max_cost_usd'] is None
source['execution']['max_cost_usd'] = 5
assert validate_request(source, ['claude'])['execution']['max_cost_usd'] == 5
source['execution']['max_cost_usd'] = 30.1
with pytest.raises(Problem, match='invalid_cost_limit'):
validate_request(source, ['claude'])
source['execution']['max_cost_usd'] = 100000
assert validate_request(source, ['claude'])['execution']['max_cost_usd'] == 100000
@pytest.mark.parametrize('size', [14,75,363])
@ -343,7 +342,7 @@ def test_existing_suite_sizes_have_separate_natural_and_task_counts(monkeypatch,
for alias,family in expected.items(): buckets.setdefault(family,[]).append(alias)
families = natural(source,list(buckets.values()),list(buckets))
install_backend(monkeypatch,source,families)
result, metadata = workflow.generate(source,preflight(source),threading.Event(),'192.168.22.8')
result, metadata = workflow.generate(source,{**preflight(source), 'execution_mode':'direct'},threading.Event(),'192.168.22.8')
assert metadata['natural_family_count'] == 9
assert metadata['final_task_count'] == sum(len(balanced_sizes(len(p))) for p in buckets.values())
assert all(1 <= len(g['members']) <= 5 for g in result['groups'])
@ -351,16 +350,15 @@ def test_existing_suite_sizes_have_separate_natural_and_task_counts(monkeypatch,
assert len(encoded({'result':result,**metadata})) < 1<<20
def test_whole_job_cost_budget_and_failure_no_partial_answer(monkeypatch):
def test_cost_estimates_never_stop_a_job(monkeypatch):
source = request(14)
source['execution']['max_cost_usd'] = 5
family = natural(source,[[c['alias'] for c in source['cases']]])
calls = install_backend(monkeypatch,source,family,costs=[1,2,3])
with pytest.raises(Problem,match='job_cost_budget_exhausted') as raised:
workflow.generate(source,preflight(source),threading.Event(),'192.168.22.8')
assert [c[0]['execution']['max_cost_usd'] for c in calls] == [5,4,2]
assert raised.value.details['completed_model_passes'] == 3
assert CANARY not in json.dumps(raised.value.document())
calls = install_backend(monkeypatch,source,family,costs=[10]*5)
result, metadata = workflow.generate(source,preflight(source),threading.Event(),'192.168.22.8')
assert metadata['cost_usd_estimate'] == 50
assert all(c[0]['execution']['max_cost_usd'] is None for c in calls)
validate_result(result,source)
def test_expanded_reconciliation_capacity_checked_before_launch(monkeypatch):
@ -374,6 +372,7 @@ def test_expanded_reconciliation_capacity_checked_before_launch(monkeypatch):
call['input'] += 'x'*(1<<20)
return original(call,*args)
monkeypatch.setattr(workflow,'capacity',capacity)
monkeypatch.setattr(workflow,'FALLBACK_ERRORS',set())
with pytest.raises(Problem,match='pass_request_too_large'):
workflow.generate(source,selected,threading.Event(),'192.168.22.8')
assert len(calls) == 2
@ -423,19 +422,18 @@ def test_shared_deadline_stops_after_prior_calls(monkeypatch):
return answer
monkeypatch.setattr(workflow.time,'monotonic',lambda:clock[0])
monkeypatch.setattr(suite_backends,'claude_generate',advancing)
with pytest.raises(Problem,match='job_time_budget_exhausted'):
workflow.generate(source,preflight(source),threading.Event(),'192.168.22.8')
assert len(calls) == 2
assert [c[0]['execution']['max_seconds'] for c in calls] == [60,29]
def test_missing_cost_measurement_stops_future_paid_calls(monkeypatch):
source = request(7)
family = natural(source,[[c['alias'] for c in source['cases']]])
calls = install_backend(monkeypatch,source,family,costs=[None])
with pytest.raises(Problem,match='budget_accounting_unavailable'):
with pytest.raises(Problem,match='insufficient_remaining_pass_budget'):
workflow.generate(source,preflight(source),threading.Event(),'192.168.22.8')
assert len(calls) == 1
assert calls[0][0]['execution']['max_seconds'] == 20
def test_missing_cost_measurement_stays_unknown_without_stopping(monkeypatch):
source = request(7)
family = natural(source,[[c['alias'] for c in source['cases']]])
calls = install_backend(monkeypatch,source,family,costs=[None]*5)
_, metadata = workflow.generate(source,preflight(source),threading.Event(),'192.168.22.8')
assert len(calls) == 5 and metadata['cost_usd_estimate'] is None
def test_cancellation_stops_between_model_passes(monkeypatch):

View File

@ -229,7 +229,7 @@ def test_prompt_provenance_is_stored_and_replay_does_not_relabel(tmp_path, monke
def test_compaction_and_incomplete_detection():
for events, code in [([{"type": "system", "subtype": "compact_boundary"}], "compaction_detected"),
([], "incomplete_generation")]:
([], "pass_missing_terminal_event")]:
with pytest.raises(Problem, match=code):
suite_backends.parse_claude("\n".join(json.dumps(e) for e in events), "claude-fable-5")
@ -237,7 +237,7 @@ def test_compaction_and_incomplete_detection():
@pytest.mark.parametrize("failure,code", [
(None, None), ("model", "model_changed"),
("context", "backend_capabilities_changed"),
("tools", "worker_isolation_failed"), ("output", "incomplete_generation"),
("tools", "worker_isolation_failed"), ("output", "pass_output_limit"),
])
def test_actual_cli_envelope_and_runtime_guards(failure, code):
model = MODELS["claude"]["model"]

View File

@ -0,0 +1,244 @@
"""Fault injection exercises real orchestration without hosted model requests."""
import copy
import json
import threading
from pathlib import Path
import sys
import pytest
sys.path.insert(0, str(Path(__file__).resolve().parents[2] / 'scripts/ops'))
sys.path.insert(0, str(Path(__file__).resolve().parents[2] / 'services/hermes/scripts'))
import suite_backends
import suite_hierarchical
import suite_multipass as workflow
import suite_recovery
from suite_contract import MODELS, Problem, preflight, validate_request, validate_result
from suite_profiles import PROFILE_FIELDS, profile_source, validate_profiles
from suite_recovery import Checkpoints
from suite_sizing import balanced_sizes
from suite_synthetic import fixture, score
from test_suite_multipass import natural, public, review, wire_value
from suite_capacity_fixture import fixture as realistic_fixture
class NoWait:
"""Cancellation remains testable without real retry sleeps."""
def is_set(self):
return False
def wait(self, _):
return False
def source(size=14, large=False):
raw, expected = realistic_fixture() if large else fixture(size)
raw['routing'] = {'allow_external': True, 'allowed_external_providers': ['claude']}
raw['execution'] = {'strategy': 'whole_suite', 'max_seconds': 7200, 'max_cost_usd': 60}
return validate_request(raw, ['claude']), expected
def oracle(monkeypatch, expected, failures=None):
"""Use an independent fixture map to check cross-batch control flow and validation."""
calls = []
failures = dict(failures or {})
def backend(request, cancel, *, invocation, progress=None, job_deadline=None):
stage = invocation['stage']
calls.append({'stage': stage, 'aliases': sorted(c['alias'] for c in request['cases']),
'seconds': request['execution']['max_seconds'], 'deadline': job_deadline,
'turns': invocation['max_turns'], 'source': copy.deepcopy(request['cases'])})
if failures.get(stage, 0):
failures[stage] -= 1
raise Problem('incomplete_generation', 502, final_event_subtype='success',
final_is_error=True, exit_code=1, assistant_api_error_seen=True,
structured_output_present=False, turns=1, max_turns=invocation['max_turns'],
cost_usd_estimate=0, usage={'input_tokens':0,'output_tokens':0})
if stage == 'implementation_profiles':
value = {'profiles': {}}
for c in request['cases']:
name = expected[c['alias']]
p = {k: None for k in PROFILE_FIELDS}
p.update(target=name, work='Implement '+name+' tests.', setup=name+' fixture',
stimulus=name+' stimulus', observations=name+' measurements',
machinery=name, evidence={'field':'success_criteria', 'quote':c['success_criteria'][:90]})
value['profiles'][c['alias']] = p
else:
parts = {}
for c in request['cases']:
parts.setdefault(expected[c['alias']], []).append(c['alias'])
decision = natural(request, list(parts.values()), list(parts))
if stage in ('proposal_a','proposal_b','profile_proposal_a','profile_proposal_b','profile_reconciliation'):
decision = public(decision)
if stage in ('large_family_review','decision_audit'):
decision = review(decision, decision)
value = wire_value(decision, invocation)
return value, {'model':MODELS['claude']['model'], 'cost_usd_estimate': 25.0,
'usage':{'input_tokens':100,'output_tokens':50}, 'turns':2, 'duration_api_ms':2,
'compaction':False, 'truncation':False, 'cli_diagnostics':{'exit_code':0}}
monkeypatch.setattr(suite_backends, 'claude_generate', backend)
return calls
def run(monkeypatch, *, mode='direct', size=14, large=False, failures=None):
request, expected = source(size, large)
calls = oracle(monkeypatch, expected, failures)
selected = {**preflight(request), 'execution_mode':mode}
updates = []
result, metadata = workflow.generate(request, selected, NoWait(), '192.168.22.8', updates.append,
credential_scope={'owner':'synthetic-test','providers':['claude']})
validate_result(result, request)
return request, expected, calls, result, metadata, updates
def test_proposal_b_retry_keeps_validated_a_and_same_source(monkeypatch):
request, expected, calls, result, meta, updates = run(monkeypatch, failures={'proposal_b':1})
assert [c['stage'] for c in calls].count('proposal_a') == 1
assert [c['stage'] for c in calls].count('proposal_b') == 2
attempts = [c for c in calls if c['stage']=='proposal_b']
assert attempts[0]['source'] == attempts[1]['source']
assert meta['retry_count']['proposal_b'] == 1
assert 'proposal_a' in meta['checkpointed_passes']
assert meta['cost_usd_estimate'] > 60 and meta['cost_guard_enforced'] is False
assert all(c['deadline']==calls[0]['deadline'] for c in calls)
assert score({'groups':meta['review_summary']['natural_families']}, expected)['false_merge_pairs']==0
def test_review_failure_does_not_regenerate_prior_passes(monkeypatch):
_, _, calls, _, meta, _ = run(monkeypatch, size=75, failures={'large_family_review':1})
for name in ('proposal_a','proposal_b','reconciliation'):
assert [c['stage'] for c in calls].count(name)==1
assert [c['stage'] for c in calls].count('large_family_review')==2
assert meta['retry_count']['large_family_review']==1
@pytest.mark.parametrize('mode', ['direct','hierarchical'])
def test_large_comparable_fixture_complete_and_globally_reconciled(monkeypatch, mode):
request, expected, calls, result, meta, updates = run(monkeypatch, mode=mode, large=True)
assert len(request['cases'])==363
assert max(len(g['members']) for g in result['groups'])<=5
assert len(result['groups'])==90 and meta['natural_family_count']==45
quality=score({'groups':meta['review_summary']['natural_families']}, expected)
assert quality['false_merge_pairs']==quality['missed_merge_pairs']==0
assert all(d['part_sizes']==balanced_sizes(d['natural_case_count']) for d in meta['review_summary']['capacity_divisions'])
if mode=='hierarchical':
assert meta['profile_batch_count']==16
for c in calls:
if c['stage'] in ('profile_proposal_a','profile_proposal_b','profile_reconciliation'):
assert len(c['aliases'])==363
originals={c['alias']:c for c in request['cases']}
reopened=[c for call in calls if call['stage']=='source_review' for c in call['source']]
assert {c['alias']:c for c in reopened}==originals
assert meta['cross_batch_review_count']==1
assert 'Implement ' not in json.dumps(updates)
def test_failed_direct_switches_only_after_bounded_recovery(monkeypatch):
_, _, calls, _, meta, _ = run(monkeypatch, failures={'proposal_b':3})
assert [c['stage'] for c in calls].count('proposal_a')==1
assert [c['stage'] for c in calls].count('proposal_b')==3
assert meta['execution_mode']=='hierarchical'
assert meta['strategy_events'][-1]['reason']=='pass_generation_retries_exhausted'
@pytest.mark.parametrize('code', ['provider_authentication','provider_forbidden','cancelled',
'invalid_review_evidence','worker_isolation_failed','job_time_budget_exhausted'])
def test_nonretriable_conditions_stop_without_fallback(monkeypatch, code):
request, _ = source()
seen=[]
def fail(*args, **kw):
seen.append(kw['invocation']['stage'])
raise Problem(code, 502)
monkeypatch.setattr(suite_backends,'claude_generate',fail)
with pytest.raises(Problem, match='^'+code+'$'):
workflow.generate(request, preflight(request), NoWait(),'192.168.22.8')
assert seen==['proposal_a']
def test_turn_recovery_grows_only_within_capacity(monkeypatch):
request, expected = source()
calls=oracle(monkeypatch,expected)
original=suite_backends.claude_generate
turns=[]
def backend(*args,**kw):
turns.append(kw['invocation']['max_turns'])
if len(turns)<3:
raise Problem('pass_turn_limit',502,cost_usd_estimate=0,turns=turns[-1],
max_turns=turns[-1],usage={'input_tokens':1,'output_tokens':1})
return original(*args,**kw)
monkeypatch.setattr(suite_backends,'claude_generate',backend)
workflow.generate(request,preflight(request),NoWait(),'192.168.22.8')
assert turns[:3]==[4,5,6]
def test_checkpoint_scope_hash_and_restart_behavior():
request,_=source()
call={'stage':'proposal_a','input':'synthetic','prior_hash':'a'}
store=Checkpoints(request,'claude',{'owner':'a'},float('inf'))
key=store.key(call);store.put(key,'proposal_a',{'groups':[]})
assert store.get(key,'proposal_a')=={'groups':[]}
assert store.key({**call,'prior_hash':'b'})!=key
assert Checkpoints(request,'claude',{'owner':'b'},float('inf')).key(call)!=key
assert Checkpoints(request,'claude',{'owner':'a'},float('inf')).get(key,'proposal_a') is None
store.clear();assert store.get(key,'proposal_a') is None
def test_cli_cost_guard_is_absent_and_legacy_request_cost_is_unenforced():
request,_=source()
request['execution']['max_cost_usd']=1e9
assert validate_request(request,['claude'])['execution']['max_cost_usd']==1e9
assert '--max-budget-usd' not in suite_backends.claude_command(MODELS['claude']['model'],60)
assert preflight(request)['cost_guard_enforced'] is False
def test_full_profile_set_fits_global_comparison_at_service_case_limit():
"""Bounded profiles keep every alias globally visible; no neighborhood shortcut."""
from suite_policy import invocation
request = {'campaign':'SYNTHETIC','suite':'CAPACITY','cases':[
{'alias':'CASE-'+str(i).zfill(48),'description':'Synthetic'} for i in range(400)]}
profiles = {c['alias']:{k:'x'*n for k,n in PROFILE_FIELDS.items()} for c in request['cases']}
compact = profile_source(request, profiles)
proposal = {'groups':[{'name':'x'*56,'members':[c['alias']]} for c in request['cases']]}
call = invocation('profile_reconciliation',compact,{'proposal_a':proposal,'proposal_b':proposal})
result = workflow.capacity(call,'claude',400)
assert result['max_turns'] >= 4 and result['context_reserved_tokens'] <= 1000000
assert len(call['schema']['properties']['assignments']['required']) == 400
def test_profile_evidence_and_exact_aliases_are_independently_validated():
request,expected=source()
c=request['cases'][0]
p={k:None for k in PROFILE_FIELDS}
p['evidence']={'field':'success_criteria','quote':c['success_criteria'][:90]}
subset={**request,'cases':[c]}
validate_profiles({'profiles':{c['alias']:p}},subset)
with pytest.raises(Problem,match='invalid_case_assignments'):
validate_profiles({'profiles':{'CASE-invented':p}},subset)
p['evidence']['quote']='unsupported secret instructions'
with pytest.raises(Problem,match='invalid_review_evidence'):
validate_profiles({'profiles':{c['alias']:p}},subset)
def test_checkpoint_hit_launches_no_new_model_call(monkeypatch):
request,expected=source()
calls=oracle(monkeypatch,expected)
worker=workflow.Workflow(request,preflight(request),NoWait(),'192.168.22.8',credential_scope='synthetic')
first=worker.call('proposal_a',request)
second=worker.call('proposal_a',request)
assert first==second and len(calls)==1
assert worker.safe_state()['checkpoint_reused']==['proposal_a']
assert len(worker.records)==1
def test_one_structured_assignment_repair_preserves_source(monkeypatch):
request,expected=source()
calls=oracle(monkeypatch,expected)
original=suite_backends.claude_generate
def backend(*args,**kwargs):
value,metadata=original(*args,**kwargs)
if len(calls)==1:
value['assignments'].pop(next(iter(value['assignments'])))
return value,metadata
monkeypatch.setattr(suite_backends,'claude_generate',backend)
result,meta=workflow.generate(request,preflight(request),NoWait(),'192.168.22.8')
assert meta['retry_count']=={'proposal_a':1}
assert calls[0]['source']==calls[1]['source']
validate_result(result,request)