diff --git a/docs/hermes_suite_capacity_20260930.md b/docs/hermes_suite_capacity_20260930.md new file mode 100644 index 00000000..b48cf83e --- /dev/null +++ b/docs/hermes_suite_capacity_20260930.md @@ -0,0 +1,112 @@ +# Suite review capacity fix + +The latest failed 363-case job was `3707c784abdb48a3ac625b2f965d1310`, created +2026-09-30 at 04:06:17.079260 UTC. Its persisted error was `pass_capacity` in +`decision_audit`, after four successful model invocations. The previous progress +display was stale: large-family review succeeded, then audit admission failed +before launching a CLI process. + +## Established failure + +The audit's complete input, system instructions and schema occupied 638,566 +UTF-8 bytes. The conservative capacity check added 8,192 harness tokens and +reserved six possible 64,000-token CLI outputs: 1,030,758 against 1,000,000. +This was a server reservation failure, not measured provider context exhaustion. +No tokenizer measurement was available. Audit output reservation was 43,920, +within the 64,000 output setting. Transport size was within the 1 MiB bound. + +The audit had no subprocess exit/signal, terminal event, provider stop reason, +turn count, output usage or timeout: no audit CLI process was launched. Do not +attribute the preceding successful review's diagnostics to this rejected audit. + +| Completed pass | Seconds | CLI turns | Input tokens | Output tokens | CLI USD estimate | +| --- | ---: | ---: | ---: | ---: | ---: | +| proposal_a | 451.994 | 2 | 159912 | 54847 | 1.736588 | +| proposal_b | 460.241 | 3 | 372684 | 57051 | 2.631756 | +| reconciliation | 465.694 | 2 | 199517 | 60783 | 2.013728 | +| large_family_review | 302.696 | 2 | 220788 | 41005 | 1.703252 | + +All four exited zero, terminal subtype `success`, structured output at +`result.structured_output`, observed stop reason `tool_use`, no compaction. +Counts aggregate multiple CLI turns; they are not individual context occupancy. +Total usage: 952901 input and 213686 output tokens, including 143418 thinking +tokens. Runtime validation confirmed firstParty `claude-opus-5-5`, native CLI +2.1.285, xhigh effort, reported context 1M and model output ceiling 128K. The +configured output allowance remained 64K per turn and turn ceiling six. + +The job spent 1680.696 seconds and USD 8.085324 estimated, leaving 1919.3 seconds +and USD 21.914676. Time and cost were not implicated. At reconciliation, budgeting +each remaining pass at the slowest/costliest completed pass plus 30 percent gives +3002.112 seconds and USD 15.139259. With the successful review now measured, the +same method projects 2790.307 seconds and USD 13.932204. Keep 3600 seconds and +USD 30 for variance; these are CLI estimates, not a claim of subscription charges. + +## Changes + +- Reserve four to six full 64K outputs according to the complete pass size. + The exact failed audit now admits five turns: reservation 966758, headroom + 33242. Every invocation retains at least four turns for structured-output repair. +- If a whole review still exceeds capacity, automatically batch whole already + reconciled natural families. Both proposals and reconciliation still consider + the complete suite. Every family member retains all supplied source fields. +- Admit every review batch before launching it; maximum eight batches per review + stage, nineteen calls total, under the same single deadline and cost budget. + Combine and validate exact global coverage, original family boundaries, names, + evidence and final five-case packaging before marking the job complete. +- Report the attempted stage before capacity checks. Safe failure details now + include the capacity reason, input byte/bound counts, context/output limits + and available/configured/minimum turns. Per-pass metadata records actual turn + allowance, reservation/headroom and optional `review_batch` information. + CLI diagnostics add maximum observed per-message input/output counts when + available. Prompt/response bodies and raw provider errors remain excluded. + +This is bounded review batching, not unlimited input support. An initial proposal +or reconciliation that cannot fit, or one natural family too large to review +intact, still fails explicitly. No source text is silently compacted or truncated; +no independent arbitrary batches become permanent grouping boundaries. + +## Verification and limits + +181 focused local tests passed. The native pinned CLI passed a loopback mock +transport regression at the exact failed 638566-byte audit size, including an +injected missing assignment repaired in three CLI turns under the new five-turn +ceiling. Complete input and system instructions reached the mock transport; all +five stages completed, with 90 final tasks and normal validation. This is a parser, +transport and capacity regression, not a hosted grouping-quality benchmark. + +A separate hosted 363-case synthetic stress run was stopped at the user's request +during large-family review. Its first three passes finished; their cumulative CLI +estimate was USD 3.00636. The in-flight call's terminal usage was not obtained. +It is not a completed acceptance test. No real case material was rerun. Matching +the failed source's total bytes did not establish equal semantic difficulty or +field-length distribution, which the historical record did not retain. + +## Client handoff + +Endpoint, authentication, routing restrictions, provider/model, request fields, +required `result.groups` shape and 45-second submission/poll HTTP timeout stay +unchanged. Review evidence remains in the authorized result's top-level +`review_summary`; operational metadata is available in job status and `passes`. + +```python +AI_MAX_SECONDS = 3600 +AI_MAX_COST_USD = 30 +AI_JOB_REVISION = "suite-v10-capacity-20260930-1" +``` + +Use a fresh `Idempotency-Key` for the new real attempt. Reusing the failed key +returns the failed job; it does not rerun under the new revision. + +- Configuration: `suite-v6-20260929` (unchanged HTTP compatibility identifier). +- Execution: `suite-multipass-v10-20260930`. +- Prompt: `implementation-proximity-multipass-v5-20260930`. +- Policy: `implementation-five-v1-20260929`. +- Capacity: `suite-context-turns-v1-20260930`. + +The prompt revision adds only the bounded review-batch scope instruction; the +success-criteria-first implementation objective remains. The base prompt hash +is unchanged; per-pass system hashes identify the actual stage instructions. + +Rollback: when no jobs are active, revert the capacity-fix Git commit and reconcile +the `hermes` Flux Kustomization. Do not edit live ConfigMaps or deployments. A +rollout clears volatile results, so save wanted completed results first. diff --git a/docs/hermes_suite_job_budget.md b/docs/hermes_suite_job_budget.md index 159ec274..395232ee 100644 --- a/docs/hermes_suite_job_budget.md +++ b/docs/hermes_suite_job_budget.md @@ -19,11 +19,14 @@ This is the execution section of the existing request, not a complete request. Use a fresh job attempt/idempotency key when intentionally resubmitting a failed job. No real suite was rerun during this change. +See the [latest real-job capacity diagnosis](hermes_suite_capacity_20260930.md). +The budget verification below is historical; the time and cost limits remain unchanged. + ## Runtime and revisions - HTTP compatibility configuration: `suite-v6-20260929`, unchanged. -- Execution: `suite-multipass-v9-20260929`. -- Prompt: `implementation-proximity-multipass-v4-20260929`, unchanged. +- Execution: `suite-multipass-v10-20260930`. +- Prompt: `implementation-proximity-multipass-v5-20260930`. - Policy: `implementation-five-v1-20260929`, unchanged. - Model: `claude-opus-5-5`, CLI setting `claude-opus-5-5[1m]`. - Native Claude Code: 2.1.285, existing pinned binary and first-party OAuth. diff --git a/docs/hermes_suite_multipass.md b/docs/hermes_suite_multipass.md index 4a617193..79ce865d 100644 --- a/docs/hermes_suite_multipass.md +++ b/docs/hermes_suite_multipass.md @@ -10,8 +10,8 @@ most **64 characters**, unique after whitespace and case normalization. - Configuration: `suite-v6-20260929` (HTTP compatibility identifier). - Policy: `implementation-five-v1-20260929`. -- Prompt: `implementation-proximity-multipass-v4-20260929`. -- Execution: `suite-multipass-v9-20260929`. +- Prompt: `implementation-proximity-multipass-v5-20260930`. +- Execution: `suite-multipass-v10-20260930`. A single server-side job performs: @@ -43,7 +43,10 @@ This avoids free-form membership lists silently omitting or repeating aliases. The public groups schema is unchanged. Invalid keys, indexes, empty groups, and missing review decisions still fail closed; no missing assignment is fabricated. -Each invocation retains six CLI turns for structured output. Model review passes +Each invocation reserves four to six CLI turns according to its complete input +size. Oversized review stages use bounded whole-family batches after complete-suite +reconciliation. See the [capacity diagnosis and fix](hermes_suite_capacity_20260930.md). +Model review passes and CLI turns are separate counters. All calls use the originally selected provider and pinned model. The current default is `claude-opus-5-5[1m]`, with canonical runtime identity checked as `claude-opus-5-5`, firstParty, reported 1M context @@ -335,7 +338,7 @@ client limits remain effective. The existing request fields are unchanged. No inference or regression tests were rerun for this limit-only update, as requested; the acceptance measurements above retain their original revisions and bounds. -Execution revision `suite-multipass-v9-20260929` supersedes the maximum above: +Execution revision `suite-multipass-v10-20260930` supersedes the maximum above: 3600 seconds is opt-in through `execution.max_seconds`, with the default still 1800 and the estimated-cost maximum/default still USD 30. The shared deadline reaches every subprocess watchdog and status update. Its expiry reports diff --git a/docs/hermes_suite_planning.md b/docs/hermes_suite_planning.md index e2afff3d..fe53b386 100644 --- a/docs/hermes_suite_planning.md +++ b/docs/hermes_suite_planning.md @@ -22,8 +22,8 @@ The active model is **`claude-opus-5-5`**, invoked as `claude-opus-5-5[1m]` using native Claude Code 2.1.285. This was the user-requested 5.5 upgrade; older Opus 4.8 acceptance records below are historical. Effort is high, or xhigh at 100+ cases or 128 KiB+ source content. The model and effort stay fixed -throughout a job. Current execution revision: `suite-multipass-v9-20260929`. -See [current deadline and cost evidence](hermes_suite_job_budget.md). +throughout a job. Current execution revision: `suite-multipass-v10-20260930`. +See the [current capacity fix and client settings](hermes_suite_capacity_20260930.md). ## Earlier CLI failure diagnostics update (historical) diff --git a/scripts/ops/hermes_suite_capacity_probe.py b/scripts/ops/hermes_suite_capacity_probe.py new file mode 100755 index 00000000..89c688ec --- /dev/null +++ b/scripts/ops/hermes_suite_capacity_probe.py @@ -0,0 +1,76 @@ +#!/usr/bin/env python3 +"""Run only the fixed 363-case capacity fixture through one async suite job. + +Server-side acceptance uses an isolated metadata DB and the deployed pinned CLI. +No real job/source input is accepted. Output is operational metadata and scores. +""" +import json +import os +from pathlib import Path +import sys +import tempfile +import time + +sys.path.insert(0, os.environ.get("SUITE_PROBE_MODULE_DIR", "/opt/planner")) +from suite_capacity_fixture import fixture +from suite_contract import encoded, preflight, validate_request, validate_result +from suite_jobs import Jobs +from suite_sizing import balanced_sizes +from suite_synthetic import score + + +def field_sizes(cases): + """Record only byte-length distribution, never source field values.""" + result = {} + for field in ("description", "preconditions", "success_criteria", "case_type"): + sizes = sorted(len(c[field].encode()) for c in cases) + result[field] = {"min": min(sizes), "median": sizes[len(sizes)//2], + "p95": sizes[int((len(sizes)-1)*.95)], "max": max(sizes), + "mean": round(sum(sizes)/len(sizes), 2)} + return result + + +def main(): + """Require all five logical stages; retain no generated descriptions in report.""" + raw, expected = fixture() + source_bytes = len(encoded(raw)) + raw["routing"] = {"allow_external": True, "allowed_external_providers": ["claude"]} + raw["execution"] = {"strategy": "whole_suite", "max_seconds": 3600, "max_cost_usd": 30} + request = validate_request(raw, ["claude"]) + selected = preflight(request) + report = {"test": "synthetic_capacity_363", "source_bytes": source_bytes, + "request_bytes": len(encoded(request)), "case_count": len(request["cases"]), + "field_bytes": field_sizes(request["cases"]), "preflight": selected, + "real_field_distribution_known": False, "automatic_job_retries": 0} + with tempfile.TemporaryDirectory(prefix="suite-capacity-probe-", dir="/jobs") as directory: + jobs = Jobs(Path(directory) / "jobs.sqlite") + job, _ = jobs.submit("synthetic-capacity-probe", "synthetic-capacity-363", request, + selected, "192.168.22.8") + last_report = 0.0 + while job["status"] in {"accepted", "running", "cancelling"}: + if time.monotonic() - last_report >= 20: + p = job.get("execution_progress", {}) + print(json.dumps({k: p.get(k) for k in ( + "current_pass", "review_batch", "completed_model_passes", "job_elapsed_seconds", + "job_remaining_seconds", "cost_used_usd_estimate")}), file=sys.stderr, flush=True) + last_report = time.monotonic() + time.sleep(2) + job = jobs.get(job["job_id"], "synthetic-capacity-probe") + report.update(status=job["status"], metadata=job) + if job["status"] == "completed": + result = jobs.get(job["job_id"], "synthetic-capacity-probe", result=True) + validate_result(result["result"], request) + stages = {p["stage"] for p in job["passes"]} + assert stages == {"proposal_a", "proposal_b", "reconciliation", "large_family_review", "decision_audit"} + review = result["review_summary"] + assert all(d["part_sizes"] == balanced_sizes(d["natural_case_count"]) for d in review["capacity_divisions"]) + report.update(response_bytes=len(encoded(result)), exact_coverage=True, + max_group_size=max(len(g["members"]) for g in result["result"]["groups"]), + names_valid=True, balanced_packaging=True, counts=review["counts"], + natural_quality=score({"groups": review["natural_families"]}, expected)) + print(json.dumps(report, sort_keys=True), flush=True) + return 0 if report["status"] == "completed" else 1 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/ops/hermes_suite_multipass_transport_probe.py b/scripts/ops/hermes_suite_multipass_transport_probe.py index 0cfe6458..b3ddfbca 100755 --- a/scripts/ops/hermes_suite_multipass_transport_probe.py +++ b/scripts/ops/hermes_suite_multipass_transport_probe.py @@ -19,6 +19,8 @@ from suite_synthetic import fixture active = {} seen = [] +expected_assignments = {} +force_audit_bytes = None def strings(value): @@ -36,7 +38,8 @@ def strings(value): def response_value(): """Create schema-valid mock replies from independent synthetic expectations.""" source = json.loads(active['input'])['suite'] - expected = fixture(len(source['cases']))[1] + expected = {a: family for a, family in expected_assignments.items() + if a in {c['alias'] for c in source['cases']}} buckets = {} records = {c['alias']: c for c in source['cases']} for alias, family in expected.items(): @@ -87,7 +90,9 @@ class Provider(BaseHTTPRequestHandler): assert seen[-1]['effort'] == active['reasoning'] value = response_value() # Force one schema rejection; the actual CLI must repair it within turns. - omitted = len(json.loads(active['input'])['suite']['cases']) == 14 and len(seen) == 1 + omitted = (len(json.loads(active['input'])['suite']['cases']) == 14 and len(seen) == 1 + or force_audit_bytes is not None and active['stage'] == 'decision_audit' + and sum(r['stage'] == 'decision_audit' for r in seen) == 1) if omitted: del value['assignments'][next(iter(value['assignments']))] seen[-1]['missing_assignment_injected'] = omitted @@ -116,7 +121,11 @@ def main(): """Run complete suites through the installed binary with fake loopback auth.""" parser = argparse.ArgumentParser(description=__doc__) parser.add_argument('--sizes', type=int, nargs='+', choices=(14, 75, 363), default=[14, 75, 363]) + parser.add_argument('--capacity-fixture', action='store_true') + parser.add_argument('--audit-input-bytes', type=int) args = parser.parse_args() + global expected_assignments, force_audit_bytes + force_audit_bytes = args.audit_input_bytes server = HTTPServer(('127.0.0.1', 0), Provider) threading.Thread(target=server.serve_forever, daemon=True).start() original_env = suite_backends.claude_environment @@ -135,16 +144,33 @@ def main(): suite_backends.claude_environment = environment suite_backends.claude_generate = backend + if force_audit_bytes is not None: + import suite_multipass + from suite_contract import encoded + original_capacity = suite_multipass.capacity + def padded_capacity(call, *parameters): + if call['stage'] == 'decision_audit': + current = len(call['input'].encode()) + len(call['system'].encode()) + len(encoded(call['schema'])) + assert current <= force_audit_bytes + call['input'] += ' ' * (force_audit_bytes - current) + return original_capacity(call, *parameters) + suite_multipass.capacity = padded_capacity try: for size in args.sizes: seen.clear() - request = fixture(size)[0] + if args.capacity_fixture: + from suite_capacity_fixture import fixture as capacity_fixture + request, expected_assignments = capacity_fixture() + else: + request, expected_assignments = fixture(size) request['routing'] = {'allow_external': True, 'allowed_external_providers': ['claude']} request = validate_request(request, ['claude']) result, metadata = generate(request, preflight_workflow(request), threading.Event(), '192.168.22.8') validate_result(result, request) print(json.dumps({'size': size, 'model_passes': metadata['model_pass_count'], - 'final_tasks': len(result['groups']), 'requests': seen}), flush=True) + 'final_tasks': len(result['groups']), 'requests': seen, + 'passes': [{k:p.get(k) for k in ('stage','max_turns','cli_turns','input_bytes', + 'context_reserved_tokens','context_headroom_tokens')} for p in metadata['passes']]}), flush=True) finally: server.shutdown() diff --git a/scripts/ops/suite_capacity_fixture.py b/scripts/ops/suite_capacity_fixture.py new file mode 100644 index 00000000..0be7085b --- /dev/null +++ b/scripts/ops/suite_capacity_fixture.py @@ -0,0 +1,100 @@ +"""Public synthetic machinery cases; no roster data or source-derived wording.""" +from collections import Counter + +from suite_contract import encoded + +# Each row specifies different control/measurement machinery, not subsystem labels. +MECHANISMS = ( + ("Reset pulse timing", "Stop and restart a hardware watchdog pulse generator", "Measure reset-pin edges with a logic analyzer", "pulse generator and digital capture driver"), + ("Network loss recovery", "Drop selected datagrams through a network impairment bridge", "Correlate packet captures with retransmission state", "network impairment bridge and packet-capture adapter"), + ("SPI waveform decoding", "Replay chip-select and clock waveform sequences", "Decode sampled SPI edge ordering and returned words", "SPI waveform synthesizer and edge decoder"), + ("I2C contention handling", "Hold the I2C data line during competing controller transfers", "Observe bus arbitration loss and recovery pulses", "I2C open-drain bus fault injector"), + ("CAN arbitration", "Schedule competing CAN frames with controlled identifiers", "Measure arbitration winners and error-frame counters", "CAN bus traffic generator and bus monitor"), + ("UART framing errors", "Transmit serial bytes with altered stop-bit durations", "Correlate serial framing flags with received characters", "UART timing generator and serial capture adapter"), + ("USB reconnect enumeration", "Disconnect and reconnect a USB endpoint through a relay", "Capture enumeration transactions and endpoint availability", "USB relay controller and protocol analyzer"), + ("DMA guard protection", "Issue PCIe DMA transfers spanning mapped guard regions", "Inspect IOMMU faults and memory guard signatures", "PCIe DMA exerciser and IOMMU fault collector"), + ("GPIO debounce timing", "Apply contact-bounce transitions to an input pin", "Measure accepted event timing against the bounce waveform", "GPIO bounce sequencer and event timestamp recorder"), + ("ADC transfer accuracy", "Sweep calibrated voltage into an analog converter", "Fit conversion residuals against sampled reference voltages", "precision voltage source and ADC sampling adapter"), + ("DAC spectral purity", "Command analog waveform output at several frequencies", "Calculate harmonic distortion from oscilloscope samples", "oscilloscope waveform reader and FFT analysis helper"), + ("Supply transient recovery", "Apply short controlled supply-voltage dips", "Measure rail recovery and brownout indications", "programmable supply transient controller and voltage probes"), + ("Thermal expansion", "Cycle the enclosure in a thermal chamber", "Measure enclosure dimensions with a calibrated displacement gauge", "thermal chamber controller and displacement gauge reader"), + ("Acoustic spectrum", "Run the actuator under controlled load in an anechoic fixture", "Compare microphone spectral peaks with the acoustic envelope", "microphone acquisition chain and acoustic spectrum analyzer"), + ("Vibration resonance", "Sweep a shaker through the mechanical excitation range", "Estimate resonant peaks from accelerometer transfer functions", "shaker controller and accelerometer acquisition system"), + ("RF emissions", "Operate the transmitter inside a shielded emissions fixture", "Measure out-of-band power with a swept spectrum analyzer", "RF chamber fixture and spectrum analyzer adapter"), + ("Display luminance", "Render grayscale patches on the display", "Compare luminance uniformity using a calibrated photometer", "photometer positioning stage and luminance reader"), + ("Camera geometric calibration", "Capture a calibrated optical checkerboard at controlled poses", "Calculate reprojection errors from detected calibration corners", "camera capture adapter and geometric calibration solver"), + ("Touch trajectory response", "Move a robotic stylus along a specified screen trajectory", "Compare reported touch coordinates with robot encoder positions", "robotic stylus controller and coordinate synchronizer"), + ("Battery discharge", "Discharge a battery through programmed current profiles", "Integrate current and compare cutoff voltage timing", "electronic load controller and coulomb-counting acquisition"), + ("GNSS acquisition", "Generate synthetic satellite signals with controlled clock offsets", "Measure acquisition latency and reported navigation residuals", "GNSS RF simulator and receiver telemetry adapter"), + ("PTP synchronization", "Inject timestamp asymmetry into precision time exchanges", "Compare hardware clock offsets against a reference clock", "PTP timestamp injector and clock comparison fixture"), + ("Filesystem crash consistency", "Cut block-device writes at selected journal boundaries", "Remount the image and verify filesystem invariants", "block-device crash emulator and filesystem image checker"), + ("ECC fault handling", "Inject single-bit and double-bit memory corruption", "Read ECC syndrome registers and corrected memory values", "memory error injector and ECC register access adapter"), + ("Allocation exhaustion", "Force allocator failures at selected allocation calls", "Inspect returned errors and retained allocation ownership", "allocator interposition hooks and ownership tracker"), + ("Thread race detection", "Replay conflicting memory accesses with controlled scheduling", "Compare race-detector reports with happens-before expectations", "thread scheduling harness and race-detector report adapter"), + ("Deadlock graph detection", "Force opposing lock-acquisition orders in worker threads", "Inspect blocked-thread stacks and wait-for graph cycles", "lock-order scheduler and thread-stack graph collector"), + ("Queue saturation", "Drive bounded producers and consumers at controlled rates", "Check queue depth, rejection ordering, and recovery after drain", "concurrent queue drivers and sequence-number observation fixture"), + ("Flash endurance evidence", "Execute repeated flash erase/program cycles", "Compare bad-block telemetry and corrected-read counters", "flash cycling controller and device health telemetry decoder"), + ("Certificate path validation", "Construct certificate chains with varied constraints", "Assert validation decisions and reported path violations", "certificate graph builder and trust-store fixture"), + ("Firmware rollback recovery", "Interrupt a firmware image installation between boot-bank writes", "Inspect boot-bank selection and retained image signatures", "boot-bank flash interrupter and bootloader console collector"), + ("Secure boot measurement", "Start measured boot with altered signed boot components", "Read TPM event logs and verify PCR extension chains", "TPM access adapter and measured-boot event-log verifier"), + ("Authorization decisions", "Submit role-scoped operations to the policy decision function", "Compare permit/deny results with the permissions matrix", "identity fixture builder and in-memory policy store"), + ("Audit log chain integrity", "Rotate tamper-evident audit segments after controlled events", "Recompute segment hash links and verify event continuity", "audit segment reader and hash-chain verifier"), + ("AST rule analysis", "Run syntax-tree checks against small source-code fixtures", "Compare AST findings and source spans with expected rule matches", "compiler AST adapter and rule-location comparator"), + ("Binary symbol inspection", "Build object files with controlled symbol visibility", "Inspect ELF symbol tables and relocation references", "ELF section parser and symbol/relocation checker"), + ("Branch coverage tracing", "Execute instrumented branch paths with selected inputs", "Compare trace counters with the branch transition matrix", "coverage instrumentation driver and binary trace decoder"), + ("Reproducible builds", "Rebuild the same source in isolated clean build containers", "Compare artifact digests after permitted timestamp normalization", "container build runner and reproducibility diff helper"), + ("Configuration parsing", "Load text configuration fixtures through the parser", "Assert parsed values or diagnostic source positions", "configuration text builder and parser-call fixture"), + ("Binary frame decoding", "Feed escaped binary frames through the frame decoder", "Compare decoded fields, checksum flags, and rejection codes", "byte-array frame builder and decoder-call fixture"), + ("Transaction rollback", "Inject connection failures at database transaction boundaries", "Query committed rows and verify transaction isolation invariants", "database fault proxy and transaction-state query fixture"), + ("Backup restoration", "Restore a snapshot into a fresh isolated service instance", "Compare restored object manifests and recovery checkpoints", "snapshot restoration runner and manifest comparison adapter"), + ("HTTP backpressure", "Throttle readers of a long HTTP response stream", "Measure socket send-buffer pressure and cancellation cleanup", "HTTP slow-reader driver and socket instrumentation collector"), + ("WebSocket reconnection", "Break established WebSocket sessions during sequenced messages", "Compare resumed sequence markers and duplicate-delivery flags", "WebSocket reconnect driver and message sequence tracker"), + ("GPU kernel accuracy", "Launch numerical GPU kernels on generated input tensors", "Compare device results with high-precision host reference arrays", "GPU launch adapter and numeric tolerance comparator"), +) + + +def fixture(): + """Return 363 interleaved objectives and an independent machinery partition. + + Exact total source bytes match retained metadata. Per-field real lengths were + not retained, so the synthetic distribution is explicit, not claimed identical. + """ + cases, expected = [], {} + for i in range(363): + name, action, observation, machinery = MECHANISMS[i % len(MECHANISMS)] + case = {"alias": f"CASE-CAP-{i:04d}", "case_type": ("nominal", "boundary", "fault injection")[(i // 45) % 3], + "description": action + ". Repeat the same test operation with alternate inputs and initial states.", + "preconditions": "Initialize the " + machinery + ". Restore the fixture to its known baseline before the test.", + "success_criteria": observation + ". Compare observed results with supplied expected values. Retain evidence for each individual objective.", + "verification_method": "test", "verifies": "[reference]"} + cases.append(case) + expected[case["alias"]] = name + # Duplicate content at distant positions retains two membership objectives. + cases[-2] = {**cases[1], "alias": cases[-2]["alias"]} + expected[cases[-2]["alias"]] = expected[cases[1]["alias"]] + source = {"campaign": "SYNTHETIC", "suite": "CAPACITY-363", "cases": cases} + target = 410322 + additions = { + "description": " The case varies supplied settings without introducing an additional control or measurement interface.", + "preconditions": " Confirm the driver is ready and the observation buffer is empty. Record the fixture configuration for the run.", + "success_criteria": " Use the existing observations to assert the selected expected outcome. Preserve diagnostic evidence and distinguish an unavailable observation from a failed assertion.", + } + while len(encoded(source)) < target: + for i, case in enumerate(cases): + field = ("description", "preconditions", "success_criteria")[(i // 7) % 3] + remaining = target - len(encoded(source)) + if remaining <= 0: + break + # Fixed sentence variations provide a range of realistic field lengths. + case[field] += additions[field][:remaining] + # Preserve one identical pair after variable-length expansion without changing size. + before = len(encoded(source)) + for field in ("description", "preconditions", "success_criteria"): + cases[-2][field], cases[1][field] = cases[1][field], cases[1][field] + delta = target - len(encoded(source)) + cases[-1]["description"] += " " * max(0, delta) + if delta < 0: + cases[-1]["description"] = cases[-1]["description"][:delta] + assert len(encoded(source)) == target and before == target + assert min(Counter(expected.values()).values()) > 5 + return source, expected diff --git a/services/hermes/kustomization.yaml b/services/hermes/kustomization.yaml index caa1005f..c6f09059 100644 --- a/services/hermes/kustomization.yaml +++ b/services/hermes/kustomization.yaml @@ -65,6 +65,7 @@ configMapGenerator: - suite_policy.py=scripts/suite_policy.py - suite_sizing.py=scripts/suite_sizing.py - suite_multipass.py=scripts/suite_multipass.py + - suite_review_batches.py=scripts/suite_review_batches.py - suite_contract.py=scripts/suite_contract.py - suite_synthetic.py=scripts/suite_synthetic.py options: diff --git a/services/hermes/scripts/suite_api.py b/services/hermes/scripts/suite_api.py index d391fd61..2be9537e 100644 --- a/services/hermes/scripts/suite_api.py +++ b/services/hermes/scripts/suite_api.py @@ -17,6 +17,7 @@ from suite_contract import (CLAUDE_MODELS, CLAUDE_VERSION, COST_LIMIT, EXECUTION preflight, validate_request) from suite_jobs import Jobs from suite_policy import POLICY_REVISION +from suite_review_batches import MAX_MODEL_CALLS, MAX_REVIEW_BATCHES from suite_synthetic import allowed_synthetic, fixture @@ -127,7 +128,9 @@ class Handler(BaseHTTPRequestHandler): "claude_model_options": list(CLAUDE_MODELS), "model_selection": "server_configuration", "execution_revision": EXECUTION_REVISION, "policy_revision": POLICY_REVISION, "max_final_group_cases": 5, - "max_final_name_characters": 64, "model_passes": {"minimum": 3, "maximum": 5}, + "max_final_name_characters": 64, "model_passes": {"minimum": 3, "maximum": MAX_MODEL_CALLS}, + "logical_stages": 5, "automatic_review_batching": True, + "max_batches_per_review_stage": MAX_REVIEW_BATCHES, "review_summary_location": "GET /v1/jobs//result: top-level review_summary", "prompt_revision": PROMPT_REVISION, "prompt_sha256": PROMPT_SHA256, "allowed_external_providers": providers, diff --git a/services/hermes/scripts/suite_backends.py b/services/hermes/scripts/suite_backends.py index d971779d..bc34228b 100644 --- a/services/hermes/scripts/suite_backends.py +++ b/services/hermes/scripts/suite_backends.py @@ -11,7 +11,7 @@ import time from urllib.error import HTTPError, URLError from urllib.request import HTTPRedirectHandler, ProxyHandler, Request, build_opener -from suite_contract import CLAUDE_MAX_TURNS, CLAUDE_MODELS, MODELS, SCHEMA, SYSTEM, Problem, encoded, prompt +from suite_contract import CLAUDE_MAX_TURNS, CLAUDE_MIN_TURNS, CLAUDE_MODELS, MODELS, SCHEMA, SYSTEM, Problem, encoded, prompt from suite_cli_diagnostics import snapshot SWITCHYARD = "http://hermes-switchyard.hermes.svc.cluster.local:9005/v1/chat/completions" @@ -95,13 +95,15 @@ def local_generate(request, cancel, client_ip, *, invocation=None): "provenance": response.get("inference_provenance")} -def claude_command(model, max_cost, *, reasoning=None): +def claude_command(model, max_cost, *, reasoning=None, max_turns=CLAUDE_MAX_TURNS): """Use the native pinned binary, never the privileged Hermes shell wrapper.""" if model not in CLAUDE_MODELS: raise Problem("unsupported_backend", 422) reasoning = MODELS["claude"]["reasoning"] if reasoning is None else reasoning if reasoning not in {"high", "xhigh"}: raise Problem("unsupported_reasoning", 422) + if type(max_turns) is not int or not CLAUDE_MIN_TURNS <= max_turns <= CLAUDE_MAX_TURNS: + raise Problem("unsupported_turn_limit", 422) settings = {"enabledPlugins": {"agents-md@builtin": False, "cc-plugin-agents-md@builtin": False}, "disableAllHooks": True, "disableBundledSkills": True, @@ -114,7 +116,7 @@ def claude_command(model, max_cost, *, reasoning=None): "--permission-mode", "dontAsk", "--no-chrome", "--model", model + "[1m]", "--effort", reasoning, "--max-budget-usd", str(max_cost), - "--max-turns", str(CLAUDE_MAX_TURNS), "--system-prompt", SYSTEM, + "--max-turns", str(max_turns), "--system-prompt", SYSTEM, "--json-schema", encoded(SCHEMA).decode()] @@ -237,11 +239,12 @@ def claude_generate(request, cancel, *, invocation=None, progress=None, job_dead raise Problem("provider_authentication", 503) model = MODELS["claude"]["model"] effort = invocation.get("reasoning", MODELS["claude"]["reasoning"]) if invocation else MODELS["claude"]["reasoning"] + max_turns = invocation.get("max_turns", CLAUDE_MAX_TURNS) if invocation else CLAUDE_MAX_TURNS with tempfile.TemporaryDirectory(prefix="suite-", dir="/jobs") as directory: root = Path(directory) (root / "input").write_text(invocation["input"] if invocation else prompt(request)) command = claude_command(model, request["execution"]["max_cost_usd"], - reasoning=effort) + reasoning=effort, max_turns=max_turns) if invocation: command[command.index("--system-prompt") + 1] = invocation["system"] command[command.index("--json-schema") + 1] = encoded(invocation["schema"]).decode() @@ -256,7 +259,7 @@ def claude_generate(request, cancel, *, invocation=None, progress=None, job_dead except OSError: raise Problem("backend_unavailable", 503, failure_stage="process_start", **snapshot("", subprocess_timeout_seconds=seconds, - reasoning_effort=effort)) from None + reasoning_effort=effort, max_turns=max_turns)) from None deadline = time.monotonic() + seconds next_progress = 0.0 try: @@ -287,7 +290,8 @@ def claude_generate(request, cancel, *, invocation=None, progress=None, job_dead with (root / "output").open("rb") as output: raw = output.read(OUTPUT_BYTES).decode("utf-8", errors="replace") info = {"exit_code": process.returncode, "subprocess_timeout_seconds": seconds, - "termination_reason": failure.code if failure else None, "reasoning_effort": effort} + "termination_reason": failure.code if failure else None, + "reasoning_effort": effort, "max_turns": max_turns} if failure: raise Problem(failure.code, failure.status, failure_stage="subprocess", **snapshot(raw, **info)) diff --git a/services/hermes/scripts/suite_cli_diagnostics.py b/services/hermes/scripts/suite_cli_diagnostics.py index db8dd709..1ca26dfa 100644 --- a/services/hermes/scripts/suite_cli_diagnostics.py +++ b/services/hermes/scripts/suite_cli_diagnostics.py @@ -56,12 +56,13 @@ def object_text(value): def snapshot(raw, *, exit_code=None, subprocess_timeout_seconds=None, termination_reason=None, - reasoning_effort=None): + reasoning_effort=None, max_turns=CLAUDE_MAX_TURNS): """Return bounded diagnostic fields from a CLI stream, including failed runs.""" final, last_assistant = None, {} initialized = compacted = tool_candidate = text_candidate = False assistant_error_code, transport_error = None, None assistant_api_error_seen = False + message_input, message_output = [], [] invalid_events = 0 for line in raw.splitlines(): try: @@ -78,6 +79,11 @@ def snapshot(raw, *, exit_code=None, subprocess_timeout_seconds=None, terminatio final = event if event.get("type") == "assistant" and isinstance(event.get("message"), dict): last_assistant = event["message"] + measured = usage_counts(last_assistant.get("usage")) or {} + if measured.get("input_tokens") is not None: + message_input.append(measured["input_tokens"]) + if measured.get("output_tokens") is not None: + message_output.append(measured["output_tokens"]) if event.get("error") is not None: assistant_error_code = enum(event["error"], ASSISTANT_ERRORS) api_error = event.get("is_api_error_message") is True @@ -107,7 +113,7 @@ def snapshot(raw, *, exit_code=None, subprocess_timeout_seconds=None, terminatio "nonzero_exit" if exit_code else "exited" if exit_code == 0 else None), "subprocess_timeout_seconds": number(subprocess_timeout_seconds), "provider_timeout_seconds": None, - "max_turns": CLAUDE_MAX_TURNS, + "max_turns": max_turns, "turns": number(result.get("num_turns")), "turn_limit_reached": subtype == "error_max_turns" if subtype in SUBTYPES else None, "structured_retry_limit_reached": subtype == "error_max_structured_output_retries" if subtype in SUBTYPES else None, @@ -131,6 +137,8 @@ def snapshot(raw, *, exit_code=None, subprocess_timeout_seconds=None, terminatio "final_json_text_present": object_text(result.get("result")), "compaction_event_seen": compacted, "usage": usage_counts(result.get("usage")), + "max_message_input_tokens": max(message_input, default=None), + "max_message_output_tokens": max(message_output, default=None), "duration_api_ms": number(result.get("duration_api_ms")), "cost_usd_estimate": number(result.get("total_cost_usd")), "observed_model_limits": limits[:8], diff --git a/services/hermes/scripts/suite_contract.py b/services/hermes/scripts/suite_contract.py index fd83ef2d..0746a9a7 100644 --- a/services/hermes/scripts/suite_contract.py +++ b/services/hermes/scripts/suite_contract.py @@ -8,8 +8,8 @@ import re from collections import Counter REVISION = "suite-v6-20260929" -PROMPT_REVISION = "implementation-proximity-multipass-v4-20260929" -EXECUTION_REVISION = "suite-multipass-v9-20260929" +PROMPT_REVISION = "implementation-proximity-multipass-v5-20260930" +EXECUTION_REVISION = "suite-multipass-v10-20260930" CLAUDE_VERSION = "2.1.285" CLAUDE_MODELS = { "claude-opus-4-8": 64000, @@ -20,6 +20,7 @@ CLAUDE_MODEL = os.environ.get("PLANNING_CLAUDE_MODEL", "claude-opus-5-5") if CLAUDE_MODEL not in CLAUDE_MODELS: raise RuntimeError("Unsupported configured Claude model") CLAUDE_MAX_TURNS = 6 +CLAUDE_MIN_TURNS = 4 CLAUDE_REASONING_POLICY = { "revision": "suite-size-effort-v1-20260929", "default": "high", "large": "xhigh", @@ -46,7 +47,7 @@ MODELS = { "output": 64000, "reported_output": CLAUDE_MODELS[CLAUDE_MODEL], "overhead": 8192, "backend": "claude-code-" + CLAUDE_VERSION, "enabled": True, "reasoning": "high", "reasoning_policy": CLAUDE_REASONING_POLICY, - "max_turns": CLAUDE_MAX_TURNS}, + "max_turns": CLAUDE_MAX_TURNS, "min_turns": CLAUDE_MIN_TURNS}, "codex": {"model": "gpt-6-astra", "context": 258400, "output": None, "overhead": None, "backend": "codex-subscription-broker", "enabled": False, "reasoning": "medium", diff --git a/services/hermes/scripts/suite_multipass.py b/services/hermes/scripts/suite_multipass.py index b18ab3ac..f49ebced 100644 --- a/services/hermes/scripts/suite_multipass.py +++ b/services/hermes/scripts/suite_multipass.py @@ -13,8 +13,10 @@ from suite_contract import (EXECUTION_REVISION, MAX_BODY, MAX_RESULT, MODELS, PR from suite_policy import (BASE_NAME_LIMIT, MAX_GROUP, POLICY_REVISION, invocation, validate_natural) from suite_sizing import cap_families, review_summary +from suite_review_batches import MAX_MODEL_CALLS, review_stage MAX_PASSES = 5 +CAPACITY_REVISION = "suite-context-turns-v1-20260930" def ordered_request(request, alternative=False): @@ -40,10 +42,23 @@ def capacity(call, provider, count, family_count=None): estimate = 1024 + 48 * count + 384 * family_count reserve = estimate + (8192 if provider == "claude" else 0) bound = input_bytes + model["overhead"] - if reserve > model["output"] or bound + model["output"] * model.get("max_turns", 1) > model["context"]: - raise Problem("pass_capacity", 422, review_pass=call["stage"], input_bytes=input_bytes, - output_reservation_tokens=reserve) + ceiling, minimum = model.get("max_turns", 1), model.get("min_turns", 1) + # Reserve a full output for every possible CLI turn, while retaining bounded + # repair headroom. Do not reject a complete input merely to reserve unused turns. + turns = min(ceiling, max(0, (model["context"] - bound) // model["output"])) + details = {"review_pass": call["stage"], "input_bytes": input_bytes, + "input_token_bound": bound, "context_limit": model["context"], + "output_reservation_tokens": reserve, "max_output_tokens": model["output"], + "configured_max_turns": ceiling, "minimum_max_turns": minimum, + "available_max_turns": turns, "capacity_revision": CAPACITY_REVISION} + if reserve > model["output"] or turns < minimum: + raise Problem("pass_capacity", 422, **details, capacity_reason= + "output_reservation" if reserve > model["output"] else "context_reservation") return {"input_bytes": input_bytes, "input_token_bound": bound, "input_token_count": None, + "max_turns": turns, "configured_max_turns": ceiling, "minimum_max_turns": minimum, + "context_reserved_tokens": bound + model["output"] * turns, + "context_headroom_tokens": model["context"] - bound - model["output"] * turns, + "capacity_revision": CAPACITY_REVISION, "output_reservation_tokens": reserve, "output_reservation_verified": False, "input_count_method": "Complete UTF-8 input/system/schema byte bound plus harness overhead; not a tokenizer"} @@ -73,7 +88,8 @@ def preflight_workflow(request): "prompt_revision": PROMPT_REVISION, "prompt_sha256": PROMPT_SHA256, "execution_revision": EXECUTION_REVISION, "policy_revision": POLICY_REVISION, "case_count": count, "source_sha256": digest(request["cases"]), - "minimum_model_passes": 3, "maximum_model_passes": MAX_PASSES, + "minimum_model_passes": 3, "maximum_model_passes": MAX_MODEL_CALLS, + "logical_stages": MAX_PASSES, "automatic_review_batching": True, "max_final_group_cases": MAX_GROUP, "max_final_name_characters": 64, "later_pass_capacity_verified": False, "later_pass_checks": "before_each_invocation"} raise Problem("capacity_or_unsupported_backend", 422, candidates=reasons) @@ -110,6 +126,7 @@ class Workflow: self.records = [] self.last_metadata = {} self.stage = "proposal_a" + self.max_calls = MAX_PASSES def checkpoint(self): """Fail closed on cancellation or exhaustion of the shared job budgets.""" @@ -124,29 +141,31 @@ class Workflow: """Run one fresh complete-input invocation, retaining metadata before validation.""" self.stage = stage self.checkpoint() - if len(self.records) >= MAX_PASSES: + if len(self.records) >= self.max_calls: raise Problem("model_pass_limit", 502) call = invocation(stage, source, context) # Keep the preflight choice across independent ordering and larger review prompts. call["reasoning"] = self.reasoning + started = time.monotonic() + def report(activity=None): + now = time.monotonic() + self.progress({"current_pass": stage, "completed_model_passes": len(self.records), + "maximum_model_passes": self.max_calls, "passes": self.records, + "review_batch": (context or {}).get("review_batch"), + "heartbeat_at": time.time(), "pass_elapsed_seconds": round(now - started, 1), + "job_elapsed_seconds": round(now - self.started, 1), + "job_remaining_seconds": round(max(0, self.deadline - now), 1), + "cost_used_usd_estimate": round(self.spent, 8), + "cost_limit_usd_estimate": self.cost_limit, **(activity or {})}) + report({"cli_running": False, "pass_state": "capacity_preflight"}) limits = capacity(call, self.provider, len(source["cases"]), family_count) + call["max_turns"] = limits["max_turns"] remaining_seconds = self.deadline - time.monotonic() remaining_cost = self.cost_limit - self.spent if remaining_cost <= 0: raise Problem("job_cost_budget_exhausted", 502) effective = {**source, "execution": {**source["execution"], "max_seconds": remaining_seconds, "max_cost_usd": remaining_cost}} - started = time.monotonic() - def report(activity=None): - now = time.monotonic() - self.progress({"current_pass": stage, "completed_model_passes": len(self.records), - "maximum_model_passes": MAX_PASSES, "passes": self.records, - "heartbeat_at": time.time(), "pass_elapsed_seconds": round(now - started, 1), - "job_elapsed_seconds": round(now - self.started, 1), - "job_remaining_seconds": round(max(0, self.deadline - now), 1), - "cost_used_usd_estimate": round(self.spent, 8), - "cost_limit_usd_estimate": self.cost_limit, **(activity or {})}) - report() if self.provider == "claude": value, metadata = suite_backends.claude_generate(effective, self.cancel, invocation=call, progress=report, job_deadline=self.deadline) @@ -158,6 +177,7 @@ class Workflow: record = {"stage": stage, "provider": self.provider, "model": MODELS[self.provider]["model"], "reasoning": self.reasoning, "wall_seconds": round(time.monotonic() - started, 3), **limits, + "case_count": len(source["cases"]), "review_batch": (context or {}).get("review_batch"), "system_sha256": hashlib.sha256(call["system"].encode()).hexdigest(), "schema_sha256": digest(call["schema"]), "case_order_sha256": digest([c["alias"] for c in source["cases"]]), @@ -187,15 +207,8 @@ class Workflow: large = [g for g in reconciled["groups"] if len(g["members"]) > MAX_GROUP] reviewed = audited = None if large: - reviewed = self.call("large_family_review", source, - {"natural_partition": reconciled, "oversized_families": [g["members"] for g in large]}, - len(reconciled["groups"])) - validate_natural(reviewed, source, reconciled["groups"]) - audited = self.call("decision_audit", source, - {"original_partition": reconciled, "reviewed_partition": reviewed, - "oversized_families": [g["members"] for g in large]}, - len(reviewed["groups"])) - validate_natural(audited, source, reconciled["groups"]) + reviewed = review_stage(self, "large_family_review", source, reconciled, None, capacity) + audited = review_stage(self, "decision_audit", source, reconciled, reviewed, capacity) natural = audited or reconciled final, divisions = cap_families(natural, source) review = review_summary(a, b, reconciled, reviewed, audited, final, divisions) diff --git a/services/hermes/scripts/suite_policy.py b/services/hermes/scripts/suite_policy.py index de9b96b4..6cffd2b0 100644 --- a/services/hermes/scripts/suite_policy.py +++ b/services/hermes/scripts/suite_policy.py @@ -111,6 +111,15 @@ def invocation(stage, request, context=None): instruction += "\n" + REVIEW + "\n" + AUDIT source = {k: request[k] for k in ("campaign", "suite", "cases")} context = dict(context or {}) + if context.get("review_batch"): + instruction += ( + "\nThis is a capacity-bounded review batch of already reconciled natural families. " + "Both independent proposals and reconciliation considered the whole suite. " + "All source fields for every member of these families are supplied here. " + "Other families are reviewed separately. Return exactly this batch's aliases; " + "do not rediscover cross-family groupings or invent missing outside cases. " + "This execution batch introduces no additional semantic grouping boundary." + ) keys = decision_keys(context) if stage in STAGES[3:] else {} if keys: context["review_keys"] = keys diff --git a/services/hermes/scripts/suite_review_batches.py b/services/hermes/scripts/suite_review_batches.py new file mode 100644 index 00000000..88536955 --- /dev/null +++ b/services/hermes/scripts/suite_review_batches.py @@ -0,0 +1,89 @@ +"""Bounded review batching after complete-suite discovery and reconciliation.""" +from suite_contract import Problem +from suite_policy import MAX_GROUP, invocation, validate_natural + +MAX_REVIEW_BATCHES = 8 +MAX_MODEL_CALLS = 3 + 2 * MAX_REVIEW_BATCHES + + +def material(stage, original, reviewed=None): + """Retain every original oversized family for semantic review and its audit.""" + context = {"oversized_families": [g["members"] for g in original["groups"] if len(g["members"]) > MAX_GROUP]} + if stage == "large_family_review": + context["natural_partition"] = original + else: + context.update(original_partition=original, reviewed_partition=reviewed) + return context + + +def batch_request(stage, source, originals, reviewed, index, total): + """Slice only reconciled family boundaries, retaining all their source fields.""" + aliases = {a for g in originals for a in g["members"]} + subset = {**source, "cases": [c for c in source["cases"] if c["alias"] in aliases]} + prior = None if reviewed is None else { + "groups": [g for g in reviewed["groups"] if set(g["members"]) <= aliases], + "decisions": [d for d in reviewed["decisions"] if set(d["source_members"]) <= aliases]} + context = material(stage, {"groups": originals}, prior) + context["review_batch"] = {"index": index, "total": total, + "case_count": len(aliases), "whole_suite_case_count": len(source["cases"])} + count = len(prior["groups"]) if prior else len(originals) + return subset, context, count + + +def review_stage(workflow, stage, source, original, reviewed, capacity): + """Use one full review when it fits, otherwise bounded whole-family batches. + + Discovery and reconciliation already considered every suite case together. + Reviews may refine only within an original family under the existing policy; + their execution batches therefore introduce no new semantic boundaries. + """ + context = material(stage, original, reviewed) + count = len((reviewed or original)["groups"]) + try: + result = workflow.call(stage, source, context, count) + except Problem as exc: + if exc.code not in {"pass_capacity", "pass_request_too_large"}: + raise + else: + validate_natural(result, source, original["groups"]) + return result + + large = sorted((g for g in original["groups"] if len(g["members"]) > MAX_GROUP), + key=lambda g: tuple(sorted(g["members"]))) + batches, current = [], [] + for family in large: + candidate = current + [family] + subset, context, count = batch_request(stage, source, candidate, reviewed, 8, 8) + try: + capacity(invocation(stage, subset, context), workflow.provider, len(subset["cases"]), count) + except Problem as exc: + if exc.code not in {"pass_capacity", "pass_request_too_large"}: + raise + if not current: + raise Problem("review_family_capacity", 422, review_pass=stage, + case_count=len(family["members"]), failure_stage="review_batch_preflight") from None + batches.append(current) + current = [family] + else: + current = candidate + if current: + batches.append(current) + if not batches or len(batches) > MAX_REVIEW_BATCHES: + raise Problem("review_batch_limit", 422, review_pass=stage, batch_count=len(batches)) + calls = [batch_request(stage, source, group, reviewed, i, len(batches)) + for i, group in enumerate(batches, 1)] + # Admit every batch before launching any paid review call. + for subset, context, count in calls: + capacity(invocation(stage, subset, context), workflow.provider, len(subset["cases"]), count) + workflow.max_calls += len(calls) - 1 + if workflow.max_calls > MAX_MODEL_CALLS: + raise Problem("model_pass_limit", 502) + combined = {"groups": [g for g in original["groups"] if len(g["members"]) <= MAX_GROUP], "decisions": []} + for originals, (subset, context, count) in zip(batches, calls): + value = workflow.call(stage, subset, context, count) + validate_natural(value, subset, originals) + combined["groups"].extend(value["groups"]) + combined["decisions"].extend(value["decisions"]) + # Recheck global naming, membership, boundaries and all original decisions. + validate_natural(combined, source, original["groups"]) + return combined diff --git a/services/hermes/suite-planner-deployment.yaml b/services/hermes/suite-planner-deployment.yaml index 4bb8b54b..8d378463 100644 --- a/services/hermes/suite-planner-deployment.yaml +++ b/services/hermes/suite-planner-deployment.yaml @@ -36,7 +36,7 @@ spec: app: hermes-suite-planner annotations: fluentbit.io/exclude: "true" - ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v9-20260929 + ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v10-20260930 vault.hashicorp.com/agent-inject: "true" vault.hashicorp.com/agent-pre-populate-only: "true" vault.hashicorp.com/agent-init-first: "true" diff --git a/testing/tests/test_suite_capacity.py b/testing/tests/test_suite_capacity.py new file mode 100644 index 00000000..d72f96a0 --- /dev/null +++ b/testing/tests/test_suite_capacity.py @@ -0,0 +1,138 @@ +"""Conservative context reservation and post-reconciliation review batching.""" +import json +import threading + +import pytest + +import suite_backends +import suite_multipass as workflow +from suite_cli_diagnostics import snapshot +from suite_contract import MODELS, Problem, encoded, preflight, validate_result +from test_suite_multipass import natural, public, request, review, wire_value + + +def call_with_bytes(size): + """Construct a content-free pass at an exact serialized capacity boundary.""" + return {"stage": "decision_audit", "input": "x" * (size - 2), "system": "", "schema": {}} + + +def test_exact_historical_audit_reservation_fits_five_turns(): + result = workflow.capacity(call_with_bytes(638566), "claude", 363, 45) + assert result["input_token_bound"] == 646758 + assert 646758 + 6 * 64000 == 1030758 + assert result["max_turns"] == 5 and result["configured_max_turns"] == 6 + assert result["context_reserved_tokens"] == 966758 + assert result["context_headroom_tokens"] == 33242 + assert result["output_reservation_tokens"] == 43920 + + +@pytest.mark.parametrize("size,turns", [(607808, 6), (607809, 5), (671808, 5), + (671809, 4), (735808, 4)]) +def test_context_admission_keeps_full_outputs_and_bounded_repair(size, turns): + result = workflow.capacity(call_with_bytes(size), "claude", 363, 45) + assert result["max_turns"] == turns + assert result["context_reserved_tokens"] <= MODELS["claude"]["context"] + command = suite_backends.claude_command(MODELS["claude"]["model"], 30, max_turns=turns) + assert command[command.index("--max-turns") + 1] == str(turns) + + +@pytest.mark.parametrize("size,count,reason", [(735809, 45, "context_reservation"), + (638566, 200, "output_reservation")]) +def test_capacity_failure_is_explicit_when_minimum_headroom_cannot_fit(size, count, reason): + with pytest.raises(Problem, match="pass_capacity") as raised: + workflow.capacity(call_with_bytes(size), "claude", 363, count) + details = raised.value.details + assert details["capacity_reason"] == reason and details["context_limit"] == 1000000 + assert details["max_output_tokens"] == 64000 and details["minimum_max_turns"] == 4 + assert "x" * 100 not in json.dumps(raised.value.document()) + + +@pytest.mark.parametrize("turns", [0, 3, 7, True, 5.0]) +def test_invalid_cli_turn_limits_never_launch(turns): + with pytest.raises(Problem, match="unsupported_turn_limit"): + suite_backends.claude_command(MODELS["claude"]["model"], 30, max_turns=turns) + + +def test_actual_turn_ceiling_and_message_measurements_are_content_free(): + raw = json.dumps({"type": "assistant", "message": {"usage": { + "input_tokens": 1234, "output_tokens": 987, "private": "DO_NOT_RETAIN"}, + "content": [{"type": "text", "text": "DO_NOT_RETAIN"}]}}) + result = snapshot(raw, max_turns=5) + assert result["max_turns"] == 5 + assert result["max_message_input_tokens"] == 1234 and result["max_message_output_tokens"] == 987 + assert "DO_NOT_RETAIN" not in json.dumps(result) + assert snapshot("")["max_message_input_tokens"] is None + + +def install_review_fixture(monkeypatch, families=2, per_batch=7, collision=False): + """Mock only model answers; run real batch planning, admission and validation.""" + source = request(families * 7) + aliases = [c["alias"] for c in source["cases"]] + original = natural(source, [aliases[i:i+7] for i in range(0, len(aliases), 7)]) + checks, calls = [], [] + actual_capacity = workflow.capacity + + def capacity(call, provider, count, family_count=None): + if call["stage"] in {"large_family_review", "decision_audit"} and count > per_batch: + raise Problem("pass_capacity", 422, review_pass=call["stage"], capacity_reason="context_reservation") + checks.append((call["stage"], count)) + return actual_capacity(call, provider, count, family_count) + + def backend(value, cancel, *, invocation, progress=None, job_deadline=None): + subset = {c["alias"] for c in value["cases"]} + original_cases = {c["alias"]: c for c in source["cases"]} + assert all(c == original_cases[c["alias"]] for c in value["cases"]) + assert {c["alias"] for c in json.loads(invocation["input"])["suite"]["cases"]} == subset + stage = invocation["stage"] + chosen = {"groups": [g for g in original["groups"] if set(g["members"]) <= subset]} + if stage.startswith("proposal"): + chosen = public(chosen) + elif stage in {"large_family_review", "decision_audit"}: + chosen = review(chosen, chosen) + if collision and stage == "decision_audit": + for group in chosen["groups"]: + group["name"] = "Colliding mechanism" + calls.append({"stage": stage, "aliases": subset, "execution": value["execution"], + "deadline": job_deadline, "batch": json.loads(invocation["input"])["review_material"].get("review_batch")}) + return wire_value(chosen, invocation), {"cost_usd_estimate": .1, "usage": {"input_tokens": 10, "output_tokens": 10}, + "model": MODELS["claude"]["model"], "turns": 2, "duration_api_ms": 1, + "compaction": False, "truncation": False, "cli_diagnostics": {"exit_code": 0}} + + monkeypatch.setattr(workflow, "capacity", capacity) + monkeypatch.setattr(suite_backends, "claude_generate", backend) + return source, calls + + +def test_automatic_review_batches_preserve_global_discovery_and_full_validation(monkeypatch): + source, calls = install_review_fixture(monkeypatch) + updates = [] + result, metadata = workflow.generate(source, preflight(source), threading.Event(), "192.168.22.8", updates.append) + assert [c["stage"] for c in calls] == ["proposal_a", "proposal_b", "reconciliation", + "large_family_review", "large_family_review", "decision_audit", "decision_audit"] + assert all(len(c["aliases"]) == 14 for c in calls[:3]) + assert all(len(c["aliases"]) == 7 for c in calls[3:]) + assert calls[3]["aliases"].isdisjoint(calls[4]["aliases"]) + assert calls[3]["aliases"] == calls[5]["aliases"] and calls[4]["aliases"] == calls[6]["aliases"] + assert len({c["deadline"] for c in calls}) == 1 + assert calls[-1]["execution"]["max_cost_usd"] == pytest.approx(29.4) + assert updates[-1]["maximum_model_passes"] == 7 + assert metadata["model_pass_count"] == 7 and len(result["groups"]) == 4 + validate_result(result, source) + + +@pytest.mark.parametrize("families,per_batch,collision,code", [ + (2, 7, True, "duplicate_family_name"), + (1, 6, False, "review_family_capacity"), + (9, 7, False, "review_batch_limit"), +]) +def test_batch_failure_never_accepts_partial_or_arbitrary_split(monkeypatch, families, per_batch, collision, code): + source, calls = install_review_fixture(monkeypatch, families, per_batch, collision) + updates = [] + with pytest.raises(Problem, match=code) as raised: + workflow.generate(source, preflight(source), threading.Event(), "192.168.22.8", updates.append) + stage = "decision_audit" if collision else "large_family_review" + assert raised.value.details["review_pass"] == updates[-1]["current_pass"] == stage + if not collision: + assert len(calls) == 3 + assert updates[-1]["pass_state"] == "capacity_preflight" + assert updates[-1]["cli_running"] is False