From 15c05cb7fbf49d746decabaf967c484cd003e05e Mon Sep 17 00:00:00 2001 From: jenkins Date: Tue, 29 Sep 2026 13:53:46 -0500 Subject: [PATCH] hermes: report suite progress and extend approved job bounds --- .../suite_planning_request.schema.json | 9 +- docs/hermes_suite_multipass.md | 25 +- .../hermes_suite_multipass_transport_probe.py | 4 +- services/hermes/scripts/suite_api.py | 4 +- services/hermes/scripts/suite_backends.py | 14 +- services/hermes/scripts/suite_contract.py | 9 +- services/hermes/scripts/suite_jobs.py | 3 + services/hermes/scripts/suite_multipass.py | 18 +- services/hermes/suite-planner-deployment.yaml | 2 +- testing/fixtures/suite_balanced_families.json | 331 ++++++++++++++++++ testing/tests/test_suite_cli_diagnostics.py | 10 +- testing/tests/test_suite_multipass.py | 43 ++- 12 files changed, 446 insertions(+), 26 deletions(-) create mode 100644 testing/fixtures/suite_balanced_families.json diff --git a/docs/contracts/suite_planning_request.schema.json b/docs/contracts/suite_planning_request.schema.json index d3719f8c..493e88eb 100644 --- a/docs/contracts/suite_planning_request.schema.json +++ b/docs/contracts/suite_planning_request.schema.json @@ -206,14 +206,15 @@ "max_seconds": { "type": "integer", "minimum": 10, - "maximum": 900, - "default": 900 + "maximum": 1200, + "default": 1200 }, "max_cost_usd": { "type": "number", "exclusiveMinimum": 0, - "maximum": 5, - "default": 5 + "maximum": 10, + "default": 10, + "description": "CLI estimated-cost guard, not subscription billing. Shared across all passes." } } } diff --git a/docs/hermes_suite_multipass.md b/docs/hermes_suite_multipass.md index 0de5f5fb..4055570f 100644 --- a/docs/hermes_suite_multipass.md +++ b/docs/hermes_suite_multipass.md @@ -11,7 +11,7 @@ most **64 characters**, unique after whitespace and case normalization. - Configuration: `suite-v6-20260929` (HTTP compatibility identifier). - Policy: `implementation-five-v1-20260929`. - Prompt: `implementation-proximity-multipass-v3-20260929`. -- Execution: `suite-multipass-v1-20260929`. +- Execution: `suite-multipass-v2-20260929`. A single server-side job performs: @@ -98,11 +98,20 @@ capacity check, remaining budgets, usage, timing, and CLI diagnostics. The top-l `cli_diagnostics` concerns the last invocation; `turns` is the sum of reported CLI turns, while `model_pass_count` counts complete model review invocations. `execution_progress.current_pass` reports the current stage during a running job. +While a CLI invocation runs, progress updates every five seconds with `heartbeat_at`, +`pass_elapsed_seconds`, `job_elapsed_seconds`, `job_remaining_seconds`, +`completed_model_passes`, `maximum_model_passes`, `cli_running`, `cli_output_bytes`, +and `last_cli_activity_seconds_ago`. The heartbeat means the worker is alive; a +change in output bytes means CLI activity. Neither is a percentage complete or +proof that a particular case has been reasoned about. No event bodies or reasoning +text are exposed. Poll the existing status URL every five seconds to display it. ## Shared bounds and failure behavior -The original 900-second maximum job deadline and USD 5 CLI estimated-cost allowance -apply to **all calls combined**, including routing time. Each invocation receives +The user-approved 1200-second maximum job deadline and USD 10 CLI estimated-cost +guard apply to **all calls combined**, including routing time. These are also the +defaults when omitted. Explicit lower client limits remain effective. This guard +uses Claude CLI estimated-cost accounting; it is not subscription billing. Each invocation receives only the remaining time and estimated-cost allowance. Available usage/cost is accumulated after every call; unknown cost accounting stops further hosted calls. No partial partition substitutes for unfinished passes. The CLI's estimated-cost @@ -122,7 +131,8 @@ claim to know model-generated proposal sizes in advance. The existing 8K-context local route cannot admit the new reconciliation schema and instructions. Local-only requests fail with capacity errors and never start hosted jobs. The separate `/local-model` endpoint and its limits are unchanged. No bounds -were increased, and no provider was switched after an error. +other than the explicitly approved time/cost guards were increased, and no provider +was switched after an error. Failures include existing CLI diagnostics and a fixed `review_pass`, completed-pass count, and available aggregate usage. New explicit codes include `pass_capacity`, @@ -148,3 +158,10 @@ Deploy or roll back through Git and Flux. Revert only the multi-pass commits to restore the preceding single-pass policy while preserving the earlier CLI diagnostic fix. Wait for active jobs before a worker restart because completed results are held in memory. No Vault credential or ingress/routing changes are required. + +The first live 363-case multi-pass run at the previous 900-second / USD 5 +guards stopped after 747.513 seconds during large-family review. Its actual CLI +final subtype was `error_max_budget_usd`; three full passes had completed. The CLI +reported an aggregate estimate of 5.62765, showing the documented within-generation +overshoot. It was not a timeout, turn-limit failure, or accepted partial answer. +The adapter now reports this condition as `job_cost_budget_exhausted`. diff --git a/scripts/ops/hermes_suite_multipass_transport_probe.py b/scripts/ops/hermes_suite_multipass_transport_probe.py index 8bce9ba0..c989d81d 100755 --- a/scripts/ops/hermes_suite_multipass_transport_probe.py +++ b/scripts/ops/hermes_suite_multipass_transport_probe.py @@ -105,10 +105,10 @@ def main(): env['ANTHROPIC_BASE_URL'] = 'http://127.0.0.1:' + str(server.server_port) return env - def backend(request, cancel, *, invocation): + def backend(request, cancel, *, invocation, progress=None): active.clear() active.update(invocation) - return original_generate(request, cancel, invocation=invocation) + return original_generate(request, cancel, invocation=invocation, progress=progress) suite_backends.claude_environment = environment suite_backends.claude_generate = backend diff --git a/services/hermes/scripts/suite_api.py b/services/hermes/scripts/suite_api.py index cf5520ce..3239bf1c 100644 --- a/services/hermes/scripts/suite_api.py +++ b/services/hermes/scripts/suite_api.py @@ -12,7 +12,7 @@ import re import subprocess import threading -from suite_contract import (EXECUTION_REVISION, MAX_BODY, MAX_CASES, MAX_RESULT, MODELS, PROMPT_REVISION, +from suite_contract import (COST_LIMIT, EXECUTION_REVISION, MAX_BODY, MAX_CASES, MAX_RESULT, MODELS, PROMPT_REVISION, PROMPT_SHA256, REVISION, TIMEOUT, Problem, encoded, preflight, validate_request) from suite_jobs import Jobs @@ -137,6 +137,8 @@ class Handler(BaseHTTPRequestHandler): "strategy": ["whole_suite"], "max_request_bytes": MAX_BODY, "max_result_bytes": MAX_RESULT, "max_cases": MAX_CASES, "max_seconds": TIMEOUT, "concurrency": 1, "queue": False, + "max_cost_usd": COST_LIMIT, "cost_basis": "CLI estimate guard; not subscription billing", + "progress_interval_seconds": 5, "result_retention_seconds": 3600, "idempotency_retention_seconds": 604800, "tokenizer": None, "provider_retention_verified": False}) match = re.fullmatch(r"/v1/synthetic/(14|75|363)", self.path) diff --git a/services/hermes/scripts/suite_backends.py b/services/hermes/scripts/suite_backends.py index b668db01..ff5f4d5d 100644 --- a/services/hermes/scripts/suite_backends.py +++ b/services/hermes/scripts/suite_backends.py @@ -167,6 +167,8 @@ def parse_claude(raw, expected_model, **process_info): if not initialized or not final: fail("incomplete_generation", "missing_init_or_final_event") if final.get("is_error") or final.get("subtype") != "success": + if final.get("subtype") == "error_max_budget_usd": + fail("job_cost_budget_exhausted", "cli_final_result") status = final.get("api_error_status") code = "rate_limit" if status == 429 else "incomplete_generation" if status in (401, 403): @@ -204,7 +206,7 @@ def parse_claude(raw, expected_model, **process_info): "turns": diagnostics["turns"], "cli_diagnostics": diagnostics} -def claude_generate(request, cancel, *, invocation=None): +def claude_generate(request, cancel, *, invocation=None, progress=None): """Run one fresh job in tmpfs; input, output, configuration and caches expire together.""" token = Path("/vault/secrets/claude-token").read_text().strip() if not token: @@ -229,13 +231,21 @@ def claude_generate(request, cancel, *, invocation=None): raise Problem("backend_unavailable", 503, failure_stage="process_start", **snapshot("", subprocess_timeout_seconds=seconds)) from None deadline = time.monotonic() + seconds + next_progress = 0.0 try: while process.poll() is None: if cancel.wait(0.1): raise Problem("cancelled", 409) if time.monotonic() > deadline: raise Problem("timeout", 504) - if (root / "output").stat().st_size > OUTPUT_BYTES: + activity = (root / "output").stat() + if progress and time.monotonic() >= next_progress: + # File activity proves CLI activity, not semantic completion. + # Do not read event bodies or emit model reasoning as progress. + progress({"cli_running": True, "cli_output_bytes": activity.st_size, + "last_cli_activity_seconds_ago": round(max(0, time.time() - activity.st_mtime), 1)}) + next_progress = time.monotonic() + 5 + if activity.st_size > OUTPUT_BYTES: raise Problem("response_too_large", 502) except Problem as exc: failure = exc diff --git a/services/hermes/scripts/suite_contract.py b/services/hermes/scripts/suite_contract.py index 0a9f0635..00ce8a21 100644 --- a/services/hermes/scripts/suite_contract.py +++ b/services/hermes/scripts/suite_contract.py @@ -8,12 +8,13 @@ from collections import Counter REVISION = "suite-v6-20260929" PROMPT_REVISION = "implementation-proximity-multipass-v3-20260929" -EXECUTION_REVISION = "suite-multipass-v1-20260929" +EXECUTION_REVISION = "suite-multipass-v2-20260929" CLAUDE_MAX_TURNS = 6 MAX_BODY = 1 << 20 MAX_RESULT = 1 << 20 MAX_CASES = 400 -TIMEOUT = 900 +TIMEOUT = 1200 +COST_LIMIT = 10.0 RESULT_TTL = 3600 FIELDS = {"description", "success_criteria", "preconditions", "operating_condition", "case_type", "verification_method", "target", "swci", "verifies", @@ -142,10 +143,10 @@ def validate_request(raw, permissions): if execution.get("strategy", "whole_suite") != "whole_suite": raise Problem("unsupported_strategy", 422) seconds = execution.get("max_seconds", TIMEOUT) - cost = execution.get("max_cost_usd", 5.0) + cost = execution.get("max_cost_usd", COST_LIMIT) if type(seconds) is not int or not 10 <= seconds <= TIMEOUT: raise Problem("invalid_timeout") - if type(cost) not in (int, float) or not 0 < cost <= 5: + if type(cost) not in (int, float) or not 0 < cost <= COST_LIMIT: raise Problem("invalid_cost_limit") for key in ("campaign", "suite"): if type(raw[key]) is not str or not 1 <= len(raw[key]) <= 128: diff --git a/services/hermes/scripts/suite_jobs.py b/services/hermes/scripts/suite_jobs.py index 0bd61728..2a23feb0 100644 --- a/services/hermes/scripts/suite_jobs.py +++ b/services/hermes/scripts/suite_jobs.py @@ -174,6 +174,9 @@ class Jobs: document.update(status="failed", error={"code": "internal_worker_error"}) finally: document["wall_seconds"] = round(time.monotonic() - started, 3) + if "execution_progress" in document: + document["execution_progress"].update(cli_running=False, heartbeat_at=time.time(), + job_elapsed_seconds=document["wall_seconds"]) with self.lock: if event.is_set() and document["status"] == "completed": document.update(status="cancelled", error={"code": "cancelled"}) diff --git a/services/hermes/scripts/suite_multipass.py b/services/hermes/scripts/suite_multipass.py index 69e56aa2..18b6c098 100644 --- a/services/hermes/scripts/suite_multipass.py +++ b/services/hermes/scripts/suite_multipass.py @@ -100,7 +100,8 @@ class Workflow: def __init__(self, request, selected, cancel, client_ip, progress=None): self.request, self.provider, self.cancel = request, selected["provider"], cancel self.client_ip, self.progress = client_ip, progress or (lambda _: None) - self.deadline = time.monotonic() + request["execution"]["max_seconds"] + self.started = time.monotonic() + self.deadline = self.started + request["execution"]["max_seconds"] self.cost_limit, self.spent = request["execution"]["max_cost_usd"], 0.0 self.records = [] self.last_metadata = {} @@ -129,10 +130,19 @@ class Workflow: raise Problem("job_cost_budget_exhausted", 502) effective = {**source, "execution": {**source["execution"], "max_seconds": remaining_seconds, "max_cost_usd": remaining_cost}} - self.progress({"current_pass": stage, "completed_model_passes": len(self.records), "passes": self.records}) started = time.monotonic() + def report(activity=None): + now = time.monotonic() + self.progress({"current_pass": stage, "completed_model_passes": len(self.records), + "maximum_model_passes": MAX_PASSES, "passes": self.records, + "heartbeat_at": time.time(), "pass_elapsed_seconds": round(now - started, 1), + "job_elapsed_seconds": round(now - self.started, 1), + "job_remaining_seconds": round(max(0, self.deadline - now), 1), + "cost_used_usd_estimate": round(self.spent, 8), + "cost_limit_usd_estimate": self.cost_limit, **(activity or {})}) + report() if self.provider == "claude": - value, metadata = suite_backends.claude_generate(effective, self.cancel, invocation=call) + value, metadata = suite_backends.claude_generate(effective, self.cancel, invocation=call, progress=report) elif self.provider == "local": value, metadata = suite_backends.local_generate(effective, self.cancel, self.client_ip, invocation=call) else: @@ -153,7 +163,7 @@ class Workflow: raise Problem("budget_accounting_unavailable", 502) self.spent += cost self.checkpoint() - self.progress({"current_pass": stage, "completed_model_passes": len(self.records), "passes": self.records}) + report({"cli_running": False}) return value def execute(self): diff --git a/services/hermes/suite-planner-deployment.yaml b/services/hermes/suite-planner-deployment.yaml index 38cff05d..62762780 100644 --- a/services/hermes/suite-planner-deployment.yaml +++ b/services/hermes/suite-planner-deployment.yaml @@ -36,7 +36,7 @@ spec: app: hermes-suite-planner annotations: fluentbit.io/exclude: "true" - ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v1-20260929 + ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v2-20260929 vault.hashicorp.com/agent-inject: "true" vault.hashicorp.com/agent-pre-populate-only: "true" vault.hashicorp.com/agent-init-first: "true" diff --git a/testing/fixtures/suite_balanced_families.json b/testing/fixtures/suite_balanced_families.json new file mode 100644 index 00000000..850c1b11 --- /dev/null +++ b/testing/fixtures/suite_balanced_families.json @@ -0,0 +1,331 @@ +{ + "request": { + "campaign": "SYNTHETIC", + "suite": "BALANCED-IMPLEMENTATION", + "cases": [ + { + "alias": "CASE-0001", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.", + "success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0002", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.", + "success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0003", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.", + "success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0004", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0005", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.", + "success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0006", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.", + "success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0007", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.", + "success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0008", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0009", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.", + "success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0010", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.", + "success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0011", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.", + "success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0012", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0013", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.", + "success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0014", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.", + "success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0015", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.", + "success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0016", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0017", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.", + "success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0018", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.", + "success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0019", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.", + "success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0020", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0021", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.", + "success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0022", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.", + "success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0023", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.", + "success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0024", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0025", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.", + "success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0026", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.", + "success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0027", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0028", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.", + "success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0029", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0030", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.", + "success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0031", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0032", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.", + "success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0033", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0034", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.", + "success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0035", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0036", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "nominal" + }, + { + "alias": "CASE-0037", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + }, + { + "alias": "CASE-0038", + "description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.", + "preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.", + "success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.", + "case_type": "fault injection" + } + ], + "routing": { + "allow_external": true, + "allowed_external_providers": [ + "claude" + ] + }, + "execution": { + "strategy": "whole_suite", + "max_seconds": 900, + "max_cost_usd": 5 + } + }, + "expected_implementation_family": { + "CASE-0001": "command", + "CASE-0002": "timing", + "CASE-0003": "report", + "CASE-0004": "queue", + "CASE-0005": "command", + "CASE-0006": "timing", + "CASE-0007": "report", + "CASE-0008": "queue", + "CASE-0009": "command", + "CASE-0010": "timing", + "CASE-0011": "report", + "CASE-0012": "queue", + "CASE-0013": "command", + "CASE-0014": "timing", + "CASE-0015": "report", + "CASE-0016": "queue", + "CASE-0017": "command", + "CASE-0018": "timing", + "CASE-0019": "report", + "CASE-0020": "queue", + "CASE-0021": "command", + "CASE-0022": "timing", + "CASE-0023": "report", + "CASE-0024": "queue", + "CASE-0025": "timing", + "CASE-0026": "report", + "CASE-0027": "queue", + "CASE-0028": "report", + "CASE-0029": "queue", + "CASE-0030": "report", + "CASE-0031": "queue", + "CASE-0032": "report", + "CASE-0033": "queue", + "CASE-0034": "report", + "CASE-0035": "queue", + "CASE-0036": "queue", + "CASE-0037": "queue", + "CASE-0038": "queue" + }, + "expected_natural_sizes": [ + 6, + 7, + 11, + 14 + ] +} diff --git a/testing/tests/test_suite_cli_diagnostics.py b/testing/tests/test_suite_cli_diagnostics.py index 1a84a83f..39a720fd 100644 --- a/testing/tests/test_suite_cli_diagnostics.py +++ b/testing/tests/test_suite_cli_diagnostics.py @@ -43,7 +43,7 @@ def request(): @pytest.mark.parametrize("subtype,code", [ ("error_max_turns", "incomplete_generation"), ("error_max_structured_output_retries", "incomplete_generation"), - ("error_max_budget_usd", "incomplete_generation"), + ("error_max_budget_usd", "job_cost_budget_exhausted"), ("error_during_execution", "incomplete_generation"), (CANARY, "incomplete_generation"), ]) @@ -179,6 +179,8 @@ def test_actual_subprocess_paths_cleanup_and_safe_metadata(tmp_path, monkeypatch script = "raise SystemExit(1)" elif mode == "oversize": monkeypatch.setattr(suite_backends, "OUTPUT_BYTES", 64) + elif mode == "success": + script += "time.sleep(0.25)" command = ["/not-installed-synthetic-cli"] if mode == "start" else [sys.executable, "-c", script] monkeypatch.setattr(suite_backends, "claude_command", lambda *_: command) value, cancel = request(), threading.Event() @@ -196,9 +198,13 @@ def test_actual_subprocess_paths_cleanup_and_safe_metadata(tmp_path, monkeypatch assert details["termination_reason"] == mode.replace("cancel", "cancelled") assert CANARY not in json.dumps(details) else: - _, metadata = suite_backends.claude_generate(value, cancel) + updates = [] + _, metadata = suite_backends.claude_generate(value, cancel, progress=updates.append) assert metadata["cli_diagnostics"]["exit_code"] == 0 assert metadata["temporary_files_deleted"] is True + assert updates and updates[-1]["cli_output_bytes"] > 0 + assert updates[-1]["cli_running"] is True + assert CANARY not in json.dumps(updates) assert list(tmp_path.iterdir()) == [] diff --git a/testing/tests/test_suite_multipass.py b/testing/tests/test_suite_multipass.py index 929f3a59..7dfe5ad8 100644 --- a/testing/tests/test_suite_multipass.py +++ b/testing/tests/test_suite_multipass.py @@ -71,12 +71,14 @@ def install_backend(monkeypatch, source, reconciled, *, a=None, b=None, reviewed audited = review(audited or reviewed, reconciled) answers = {'proposal_a': a or public(reconciled), 'proposal_b': b or public(reconciled), 'reconciliation': reconciled, 'large_family_review': reviewed, 'decision_audit': audited} - def backend(value, cancel, *, invocation): + def backend(value, cancel, *, invocation, progress=None): payload = json.loads(invocation['input']) assert {c['alias']: c for c in payload['suite']['cases']} == {c['alias']: c for c in source['cases']} assert payload['suite']['campaign'] == source['campaign'] assert payload['suite']['suite'] == source['suite'] calls.append((copy.deepcopy(value), copy.deepcopy(invocation))) + if progress: + progress({'cli_running': True, 'cli_output_bytes': 120, 'last_cli_activity_seconds_ago': 0}) cost = costs[len(calls)-1] if costs else 0.1 return copy.deepcopy(answers[invocation['stage']]), { 'model': MODELS['claude']['model'], 'usage': {'input_tokens': 100, 'output_tokens': 40}, @@ -100,7 +102,7 @@ def test_coherent_families_are_balanced_not_semantically_fragmented(monkeypatch, assert [g['name'] for g in result['groups']] == [f'Reset recovery ({i}/{len(sizes)})' for i in range(1,len(sizes)+1)] assert all('Work-size part' in g['description'] for g in result['groups']) assert metadata['review_summary']['decision_audit'][0]['decision'] == 'keep' - assert [call[0]['execution']['max_cost_usd'] for call in calls] == pytest.approx([5,4.9,4.8,4.7,4.6]) + assert [call[0]['execution']['max_cost_usd'] for call in calls] == pytest.approx([10,9.9,9.8,9.7,9.6]) assert all(calls[i+1][0]['execution']['max_seconds'] < calls[i][0]['execution']['max_seconds'] for i in range(4)) validate_result(result, source) @@ -119,6 +121,42 @@ def test_independent_orders_and_content_are_reproducible(monkeypatch): assert 'proposal_a' in json.loads(calls[2][1]['input'])['review_material'] +def test_progress_reports_activity_without_content_or_false_percentage(monkeypatch): + source = request(7) + family = natural(source, [[c['alias'] for c in source['cases']]]) + install_backend(monkeypatch, source, family) + updates = [] + workflow.generate(source, preflight(source), threading.Event(), '192.168.22.8', updates.append) + active = [value for value in updates if value.get('cli_running')] + assert len(active) == 5 + assert [v['completed_model_passes'] for v in active] == [0,1,2,3,4] + assert all(v['maximum_model_passes'] == 5 and v['cli_output_bytes'] == 120 for v in active) + assert all(v['heartbeat_at'] > 0 and v['job_remaining_seconds'] <= 1200 for v in active) + assert all(v['last_cli_activity_seconds_ago'] == 0 for v in active) + assert updates[-1]['cli_running'] is False + assert CANARY not in json.dumps(updates) + + +def test_twenty_minute_job_limit_retains_smaller_client_deadlines(): + source = request(7) + assert source['execution']['max_seconds'] == 1200 + source['execution']['max_seconds'] = 900 + assert validate_request(source, ['claude'])['execution']['max_seconds'] == 900 + source['execution']['max_seconds'] = 1201 + with pytest.raises(Problem, match='invalid_timeout'): + validate_request(source, ['claude']) + + +def test_cli_estimate_guard_default_and_client_override(): + source = request(7) + assert source['execution']['max_cost_usd'] == 10 + source['execution']['max_cost_usd'] = 5 + assert validate_request(source, ['claude'])['execution']['max_cost_usd'] == 5 + source['execution']['max_cost_usd'] = 10.1 + with pytest.raises(Problem, match='invalid_cost_limit'): + validate_request(source, ['claude']) + + def test_real_semantic_subdivisions_precede_work_sizing(monkeypatch): source = request(11) aliases = [c['alias'] for c in source['cases']] @@ -230,6 +268,7 @@ def test_existing_suite_sizes_have_separate_natural_and_task_counts(monkeypatch, def test_whole_job_cost_budget_and_failure_no_partial_answer(monkeypatch): source = request(14) + source['execution']['max_cost_usd'] = 5 family = natural(source,[[c['alias'] for c in source['cases']]]) calls = install_backend(monkeypatch,source,family,costs=[1,2,3]) with pytest.raises(Problem,match='job_cost_budget_exhausted') as raised: