hermes: report suite progress and extend approved job bounds

This commit is contained in:
jenkins 2026-09-29 13:53:46 -05:00
parent 7af4f4f206
commit 15c05cb7fb
12 changed files with 446 additions and 26 deletions

View File

@ -206,14 +206,15 @@
"max_seconds": {
"type": "integer",
"minimum": 10,
"maximum": 900,
"default": 900
"maximum": 1200,
"default": 1200
},
"max_cost_usd": {
"type": "number",
"exclusiveMinimum": 0,
"maximum": 5,
"default": 5
"maximum": 10,
"default": 10,
"description": "CLI estimated-cost guard, not subscription billing. Shared across all passes."
}
}
}

View File

@ -11,7 +11,7 @@ most **64 characters**, unique after whitespace and case normalization.
- Configuration: `suite-v6-20260929` (HTTP compatibility identifier).
- Policy: `implementation-five-v1-20260929`.
- Prompt: `implementation-proximity-multipass-v3-20260929`.
- Execution: `suite-multipass-v1-20260929`.
- Execution: `suite-multipass-v2-20260929`.
A single server-side job performs:
@ -98,11 +98,20 @@ capacity check, remaining budgets, usage, timing, and CLI diagnostics. The top-l
`cli_diagnostics` concerns the last invocation; `turns` is the sum of reported CLI
turns, while `model_pass_count` counts complete model review invocations.
`execution_progress.current_pass` reports the current stage during a running job.
While a CLI invocation runs, progress updates every five seconds with `heartbeat_at`,
`pass_elapsed_seconds`, `job_elapsed_seconds`, `job_remaining_seconds`,
`completed_model_passes`, `maximum_model_passes`, `cli_running`, `cli_output_bytes`,
and `last_cli_activity_seconds_ago`. The heartbeat means the worker is alive; a
change in output bytes means CLI activity. Neither is a percentage complete or
proof that a particular case has been reasoned about. No event bodies or reasoning
text are exposed. Poll the existing status URL every five seconds to display it.
## Shared bounds and failure behavior
The original 900-second maximum job deadline and USD 5 CLI estimated-cost allowance
apply to **all calls combined**, including routing time. Each invocation receives
The user-approved 1200-second maximum job deadline and USD 10 CLI estimated-cost
guard apply to **all calls combined**, including routing time. These are also the
defaults when omitted. Explicit lower client limits remain effective. This guard
uses Claude CLI estimated-cost accounting; it is not subscription billing. Each invocation receives
only the remaining time and estimated-cost allowance. Available usage/cost is
accumulated after every call; unknown cost accounting stops further hosted calls.
No partial partition substitutes for unfinished passes. The CLI's estimated-cost
@ -122,7 +131,8 @@ claim to know model-generated proposal sizes in advance.
The existing 8K-context local route cannot admit the new reconciliation schema and
instructions. Local-only requests fail with capacity errors and never start hosted
jobs. The separate `/local-model` endpoint and its limits are unchanged. No bounds
were increased, and no provider was switched after an error.
other than the explicitly approved time/cost guards were increased, and no provider
was switched after an error.
Failures include existing CLI diagnostics and a fixed `review_pass`, completed-pass
count, and available aggregate usage. New explicit codes include `pass_capacity`,
@ -148,3 +158,10 @@ Deploy or roll back through Git and Flux. Revert only the multi-pass commits to
restore the preceding single-pass policy while preserving the earlier CLI diagnostic
fix. Wait for active jobs before a worker restart because completed results are
held in memory. No Vault credential or ingress/routing changes are required.
The first live 363-case multi-pass run at the previous 900-second / USD 5
guards stopped after 747.513 seconds during large-family review. Its actual CLI
final subtype was `error_max_budget_usd`; three full passes had completed. The CLI
reported an aggregate estimate of 5.62765, showing the documented within-generation
overshoot. It was not a timeout, turn-limit failure, or accepted partial answer.
The adapter now reports this condition as `job_cost_budget_exhausted`.

View File

@ -105,10 +105,10 @@ def main():
env['ANTHROPIC_BASE_URL'] = 'http://127.0.0.1:' + str(server.server_port)
return env
def backend(request, cancel, *, invocation):
def backend(request, cancel, *, invocation, progress=None):
active.clear()
active.update(invocation)
return original_generate(request, cancel, invocation=invocation)
return original_generate(request, cancel, invocation=invocation, progress=progress)
suite_backends.claude_environment = environment
suite_backends.claude_generate = backend

View File

@ -12,7 +12,7 @@ import re
import subprocess
import threading
from suite_contract import (EXECUTION_REVISION, MAX_BODY, MAX_CASES, MAX_RESULT, MODELS, PROMPT_REVISION,
from suite_contract import (COST_LIMIT, EXECUTION_REVISION, MAX_BODY, MAX_CASES, MAX_RESULT, MODELS, PROMPT_REVISION,
PROMPT_SHA256, REVISION, TIMEOUT, Problem, encoded,
preflight, validate_request)
from suite_jobs import Jobs
@ -137,6 +137,8 @@ class Handler(BaseHTTPRequestHandler):
"strategy": ["whole_suite"], "max_request_bytes": MAX_BODY,
"max_result_bytes": MAX_RESULT, "max_cases": MAX_CASES,
"max_seconds": TIMEOUT, "concurrency": 1, "queue": False,
"max_cost_usd": COST_LIMIT, "cost_basis": "CLI estimate guard; not subscription billing",
"progress_interval_seconds": 5,
"result_retention_seconds": 3600, "idempotency_retention_seconds": 604800,
"tokenizer": None, "provider_retention_verified": False})
match = re.fullmatch(r"/v1/synthetic/(14|75|363)", self.path)

View File

@ -167,6 +167,8 @@ def parse_claude(raw, expected_model, **process_info):
if not initialized or not final:
fail("incomplete_generation", "missing_init_or_final_event")
if final.get("is_error") or final.get("subtype") != "success":
if final.get("subtype") == "error_max_budget_usd":
fail("job_cost_budget_exhausted", "cli_final_result")
status = final.get("api_error_status")
code = "rate_limit" if status == 429 else "incomplete_generation"
if status in (401, 403):
@ -204,7 +206,7 @@ def parse_claude(raw, expected_model, **process_info):
"turns": diagnostics["turns"], "cli_diagnostics": diagnostics}
def claude_generate(request, cancel, *, invocation=None):
def claude_generate(request, cancel, *, invocation=None, progress=None):
"""Run one fresh job in tmpfs; input, output, configuration and caches expire together."""
token = Path("/vault/secrets/claude-token").read_text().strip()
if not token:
@ -229,13 +231,21 @@ def claude_generate(request, cancel, *, invocation=None):
raise Problem("backend_unavailable", 503, failure_stage="process_start",
**snapshot("", subprocess_timeout_seconds=seconds)) from None
deadline = time.monotonic() + seconds
next_progress = 0.0
try:
while process.poll() is None:
if cancel.wait(0.1):
raise Problem("cancelled", 409)
if time.monotonic() > deadline:
raise Problem("timeout", 504)
if (root / "output").stat().st_size > OUTPUT_BYTES:
activity = (root / "output").stat()
if progress and time.monotonic() >= next_progress:
# File activity proves CLI activity, not semantic completion.
# Do not read event bodies or emit model reasoning as progress.
progress({"cli_running": True, "cli_output_bytes": activity.st_size,
"last_cli_activity_seconds_ago": round(max(0, time.time() - activity.st_mtime), 1)})
next_progress = time.monotonic() + 5
if activity.st_size > OUTPUT_BYTES:
raise Problem("response_too_large", 502)
except Problem as exc:
failure = exc

View File

@ -8,12 +8,13 @@ from collections import Counter
REVISION = "suite-v6-20260929"
PROMPT_REVISION = "implementation-proximity-multipass-v3-20260929"
EXECUTION_REVISION = "suite-multipass-v1-20260929"
EXECUTION_REVISION = "suite-multipass-v2-20260929"
CLAUDE_MAX_TURNS = 6
MAX_BODY = 1 << 20
MAX_RESULT = 1 << 20
MAX_CASES = 400
TIMEOUT = 900
TIMEOUT = 1200
COST_LIMIT = 10.0
RESULT_TTL = 3600
FIELDS = {"description", "success_criteria", "preconditions", "operating_condition",
"case_type", "verification_method", "target", "swci", "verifies",
@ -142,10 +143,10 @@ def validate_request(raw, permissions):
if execution.get("strategy", "whole_suite") != "whole_suite":
raise Problem("unsupported_strategy", 422)
seconds = execution.get("max_seconds", TIMEOUT)
cost = execution.get("max_cost_usd", 5.0)
cost = execution.get("max_cost_usd", COST_LIMIT)
if type(seconds) is not int or not 10 <= seconds <= TIMEOUT:
raise Problem("invalid_timeout")
if type(cost) not in (int, float) or not 0 < cost <= 5:
if type(cost) not in (int, float) or not 0 < cost <= COST_LIMIT:
raise Problem("invalid_cost_limit")
for key in ("campaign", "suite"):
if type(raw[key]) is not str or not 1 <= len(raw[key]) <= 128:

View File

@ -174,6 +174,9 @@ class Jobs:
document.update(status="failed", error={"code": "internal_worker_error"})
finally:
document["wall_seconds"] = round(time.monotonic() - started, 3)
if "execution_progress" in document:
document["execution_progress"].update(cli_running=False, heartbeat_at=time.time(),
job_elapsed_seconds=document["wall_seconds"])
with self.lock:
if event.is_set() and document["status"] == "completed":
document.update(status="cancelled", error={"code": "cancelled"})

View File

@ -100,7 +100,8 @@ class Workflow:
def __init__(self, request, selected, cancel, client_ip, progress=None):
self.request, self.provider, self.cancel = request, selected["provider"], cancel
self.client_ip, self.progress = client_ip, progress or (lambda _: None)
self.deadline = time.monotonic() + request["execution"]["max_seconds"]
self.started = time.monotonic()
self.deadline = self.started + request["execution"]["max_seconds"]
self.cost_limit, self.spent = request["execution"]["max_cost_usd"], 0.0
self.records = []
self.last_metadata = {}
@ -129,10 +130,19 @@ class Workflow:
raise Problem("job_cost_budget_exhausted", 502)
effective = {**source, "execution": {**source["execution"],
"max_seconds": remaining_seconds, "max_cost_usd": remaining_cost}}
self.progress({"current_pass": stage, "completed_model_passes": len(self.records), "passes": self.records})
started = time.monotonic()
def report(activity=None):
now = time.monotonic()
self.progress({"current_pass": stage, "completed_model_passes": len(self.records),
"maximum_model_passes": MAX_PASSES, "passes": self.records,
"heartbeat_at": time.time(), "pass_elapsed_seconds": round(now - started, 1),
"job_elapsed_seconds": round(now - self.started, 1),
"job_remaining_seconds": round(max(0, self.deadline - now), 1),
"cost_used_usd_estimate": round(self.spent, 8),
"cost_limit_usd_estimate": self.cost_limit, **(activity or {})})
report()
if self.provider == "claude":
value, metadata = suite_backends.claude_generate(effective, self.cancel, invocation=call)
value, metadata = suite_backends.claude_generate(effective, self.cancel, invocation=call, progress=report)
elif self.provider == "local":
value, metadata = suite_backends.local_generate(effective, self.cancel, self.client_ip, invocation=call)
else:
@ -153,7 +163,7 @@ class Workflow:
raise Problem("budget_accounting_unavailable", 502)
self.spent += cost
self.checkpoint()
self.progress({"current_pass": stage, "completed_model_passes": len(self.records), "passes": self.records})
report({"cli_running": False})
return value
def execute(self):

View File

@ -36,7 +36,7 @@ spec:
app: hermes-suite-planner
annotations:
fluentbit.io/exclude: "true"
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v1-20260929
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v2-20260929
vault.hashicorp.com/agent-inject: "true"
vault.hashicorp.com/agent-pre-populate-only: "true"
vault.hashicorp.com/agent-init-first: "true"

View File

@ -0,0 +1,331 @@
{
"request": {
"campaign": "SYNTHETIC",
"suite": "BALANCED-IMPLEMENTATION",
"cases": [
{
"alias": "CASE-0001",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.",
"success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0002",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0003",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0004",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0005",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.",
"success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0006",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0007",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0008",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0009",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.",
"success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0010",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0011",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0012",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0013",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.",
"success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0014",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0015",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0016",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0017",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.",
"success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0018",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0019",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0020",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0021",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.",
"success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0022",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0023",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0024",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0025",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0026",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0027",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0028",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0029",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0030",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0031",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0032",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0033",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0034",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0035",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0036",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "nominal"
},
{
"alias": "CASE-0037",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
},
{
"alias": "CASE-0038",
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
"case_type": "fault injection"
}
],
"routing": {
"allow_external": true,
"allowed_external_providers": [
"claude"
]
},
"execution": {
"strategy": "whole_suite",
"max_seconds": 900,
"max_cost_usd": 5
}
},
"expected_implementation_family": {
"CASE-0001": "command",
"CASE-0002": "timing",
"CASE-0003": "report",
"CASE-0004": "queue",
"CASE-0005": "command",
"CASE-0006": "timing",
"CASE-0007": "report",
"CASE-0008": "queue",
"CASE-0009": "command",
"CASE-0010": "timing",
"CASE-0011": "report",
"CASE-0012": "queue",
"CASE-0013": "command",
"CASE-0014": "timing",
"CASE-0015": "report",
"CASE-0016": "queue",
"CASE-0017": "command",
"CASE-0018": "timing",
"CASE-0019": "report",
"CASE-0020": "queue",
"CASE-0021": "command",
"CASE-0022": "timing",
"CASE-0023": "report",
"CASE-0024": "queue",
"CASE-0025": "timing",
"CASE-0026": "report",
"CASE-0027": "queue",
"CASE-0028": "report",
"CASE-0029": "queue",
"CASE-0030": "report",
"CASE-0031": "queue",
"CASE-0032": "report",
"CASE-0033": "queue",
"CASE-0034": "report",
"CASE-0035": "queue",
"CASE-0036": "queue",
"CASE-0037": "queue",
"CASE-0038": "queue"
},
"expected_natural_sizes": [
6,
7,
11,
14
]
}

View File

@ -43,7 +43,7 @@ def request():
@pytest.mark.parametrize("subtype,code", [
("error_max_turns", "incomplete_generation"),
("error_max_structured_output_retries", "incomplete_generation"),
("error_max_budget_usd", "incomplete_generation"),
("error_max_budget_usd", "job_cost_budget_exhausted"),
("error_during_execution", "incomplete_generation"),
(CANARY, "incomplete_generation"),
])
@ -179,6 +179,8 @@ def test_actual_subprocess_paths_cleanup_and_safe_metadata(tmp_path, monkeypatch
script = "raise SystemExit(1)"
elif mode == "oversize":
monkeypatch.setattr(suite_backends, "OUTPUT_BYTES", 64)
elif mode == "success":
script += "time.sleep(0.25)"
command = ["/not-installed-synthetic-cli"] if mode == "start" else [sys.executable, "-c", script]
monkeypatch.setattr(suite_backends, "claude_command", lambda *_: command)
value, cancel = request(), threading.Event()
@ -196,9 +198,13 @@ def test_actual_subprocess_paths_cleanup_and_safe_metadata(tmp_path, monkeypatch
assert details["termination_reason"] == mode.replace("cancel", "cancelled")
assert CANARY not in json.dumps(details)
else:
_, metadata = suite_backends.claude_generate(value, cancel)
updates = []
_, metadata = suite_backends.claude_generate(value, cancel, progress=updates.append)
assert metadata["cli_diagnostics"]["exit_code"] == 0
assert metadata["temporary_files_deleted"] is True
assert updates and updates[-1]["cli_output_bytes"] > 0
assert updates[-1]["cli_running"] is True
assert CANARY not in json.dumps(updates)
assert list(tmp_path.iterdir()) == []

View File

@ -71,12 +71,14 @@ def install_backend(monkeypatch, source, reconciled, *, a=None, b=None, reviewed
audited = review(audited or reviewed, reconciled)
answers = {'proposal_a': a or public(reconciled), 'proposal_b': b or public(reconciled),
'reconciliation': reconciled, 'large_family_review': reviewed, 'decision_audit': audited}
def backend(value, cancel, *, invocation):
def backend(value, cancel, *, invocation, progress=None):
payload = json.loads(invocation['input'])
assert {c['alias']: c for c in payload['suite']['cases']} == {c['alias']: c for c in source['cases']}
assert payload['suite']['campaign'] == source['campaign']
assert payload['suite']['suite'] == source['suite']
calls.append((copy.deepcopy(value), copy.deepcopy(invocation)))
if progress:
progress({'cli_running': True, 'cli_output_bytes': 120, 'last_cli_activity_seconds_ago': 0})
cost = costs[len(calls)-1] if costs else 0.1
return copy.deepcopy(answers[invocation['stage']]), {
'model': MODELS['claude']['model'], 'usage': {'input_tokens': 100, 'output_tokens': 40},
@ -100,7 +102,7 @@ def test_coherent_families_are_balanced_not_semantically_fragmented(monkeypatch,
assert [g['name'] for g in result['groups']] == [f'Reset recovery ({i}/{len(sizes)})' for i in range(1,len(sizes)+1)]
assert all('Work-size part' in g['description'] for g in result['groups'])
assert metadata['review_summary']['decision_audit'][0]['decision'] == 'keep'
assert [call[0]['execution']['max_cost_usd'] for call in calls] == pytest.approx([5,4.9,4.8,4.7,4.6])
assert [call[0]['execution']['max_cost_usd'] for call in calls] == pytest.approx([10,9.9,9.8,9.7,9.6])
assert all(calls[i+1][0]['execution']['max_seconds'] < calls[i][0]['execution']['max_seconds'] for i in range(4))
validate_result(result, source)
@ -119,6 +121,42 @@ def test_independent_orders_and_content_are_reproducible(monkeypatch):
assert 'proposal_a' in json.loads(calls[2][1]['input'])['review_material']
def test_progress_reports_activity_without_content_or_false_percentage(monkeypatch):
source = request(7)
family = natural(source, [[c['alias'] for c in source['cases']]])
install_backend(monkeypatch, source, family)
updates = []
workflow.generate(source, preflight(source), threading.Event(), '192.168.22.8', updates.append)
active = [value for value in updates if value.get('cli_running')]
assert len(active) == 5
assert [v['completed_model_passes'] for v in active] == [0,1,2,3,4]
assert all(v['maximum_model_passes'] == 5 and v['cli_output_bytes'] == 120 for v in active)
assert all(v['heartbeat_at'] > 0 and v['job_remaining_seconds'] <= 1200 for v in active)
assert all(v['last_cli_activity_seconds_ago'] == 0 for v in active)
assert updates[-1]['cli_running'] is False
assert CANARY not in json.dumps(updates)
def test_twenty_minute_job_limit_retains_smaller_client_deadlines():
source = request(7)
assert source['execution']['max_seconds'] == 1200
source['execution']['max_seconds'] = 900
assert validate_request(source, ['claude'])['execution']['max_seconds'] == 900
source['execution']['max_seconds'] = 1201
with pytest.raises(Problem, match='invalid_timeout'):
validate_request(source, ['claude'])
def test_cli_estimate_guard_default_and_client_override():
source = request(7)
assert source['execution']['max_cost_usd'] == 10
source['execution']['max_cost_usd'] = 5
assert validate_request(source, ['claude'])['execution']['max_cost_usd'] == 5
source['execution']['max_cost_usd'] = 10.1
with pytest.raises(Problem, match='invalid_cost_limit'):
validate_request(source, ['claude'])
def test_real_semantic_subdivisions_precede_work_sizing(monkeypatch):
source = request(11)
aliases = [c['alias'] for c in source['cases']]
@ -230,6 +268,7 @@ def test_existing_suite_sizes_have_separate_natural_and_task_counts(monkeypatch,
def test_whole_job_cost_budget_and_failure_no_partial_answer(monkeypatch):
source = request(14)
source['execution']['max_cost_usd'] = 5
family = natural(source,[[c['alias'] for c in source['cases']]])
calls = install_backend(monkeypatch,source,family,costs=[1,2,3])
with pytest.raises(Problem,match='job_cost_budget_exhausted') as raised: