hermes: report suite progress and extend approved job bounds
This commit is contained in:
parent
7af4f4f206
commit
15c05cb7fb
@ -206,14 +206,15 @@
|
||||
"max_seconds": {
|
||||
"type": "integer",
|
||||
"minimum": 10,
|
||||
"maximum": 900,
|
||||
"default": 900
|
||||
"maximum": 1200,
|
||||
"default": 1200
|
||||
},
|
||||
"max_cost_usd": {
|
||||
"type": "number",
|
||||
"exclusiveMinimum": 0,
|
||||
"maximum": 5,
|
||||
"default": 5
|
||||
"maximum": 10,
|
||||
"default": 10,
|
||||
"description": "CLI estimated-cost guard, not subscription billing. Shared across all passes."
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@ -11,7 +11,7 @@ most **64 characters**, unique after whitespace and case normalization.
|
||||
- Configuration: `suite-v6-20260929` (HTTP compatibility identifier).
|
||||
- Policy: `implementation-five-v1-20260929`.
|
||||
- Prompt: `implementation-proximity-multipass-v3-20260929`.
|
||||
- Execution: `suite-multipass-v1-20260929`.
|
||||
- Execution: `suite-multipass-v2-20260929`.
|
||||
|
||||
A single server-side job performs:
|
||||
|
||||
@ -98,11 +98,20 @@ capacity check, remaining budgets, usage, timing, and CLI diagnostics. The top-l
|
||||
`cli_diagnostics` concerns the last invocation; `turns` is the sum of reported CLI
|
||||
turns, while `model_pass_count` counts complete model review invocations.
|
||||
`execution_progress.current_pass` reports the current stage during a running job.
|
||||
While a CLI invocation runs, progress updates every five seconds with `heartbeat_at`,
|
||||
`pass_elapsed_seconds`, `job_elapsed_seconds`, `job_remaining_seconds`,
|
||||
`completed_model_passes`, `maximum_model_passes`, `cli_running`, `cli_output_bytes`,
|
||||
and `last_cli_activity_seconds_ago`. The heartbeat means the worker is alive; a
|
||||
change in output bytes means CLI activity. Neither is a percentage complete or
|
||||
proof that a particular case has been reasoned about. No event bodies or reasoning
|
||||
text are exposed. Poll the existing status URL every five seconds to display it.
|
||||
|
||||
## Shared bounds and failure behavior
|
||||
|
||||
The original 900-second maximum job deadline and USD 5 CLI estimated-cost allowance
|
||||
apply to **all calls combined**, including routing time. Each invocation receives
|
||||
The user-approved 1200-second maximum job deadline and USD 10 CLI estimated-cost
|
||||
guard apply to **all calls combined**, including routing time. These are also the
|
||||
defaults when omitted. Explicit lower client limits remain effective. This guard
|
||||
uses Claude CLI estimated-cost accounting; it is not subscription billing. Each invocation receives
|
||||
only the remaining time and estimated-cost allowance. Available usage/cost is
|
||||
accumulated after every call; unknown cost accounting stops further hosted calls.
|
||||
No partial partition substitutes for unfinished passes. The CLI's estimated-cost
|
||||
@ -122,7 +131,8 @@ claim to know model-generated proposal sizes in advance.
|
||||
The existing 8K-context local route cannot admit the new reconciliation schema and
|
||||
instructions. Local-only requests fail with capacity errors and never start hosted
|
||||
jobs. The separate `/local-model` endpoint and its limits are unchanged. No bounds
|
||||
were increased, and no provider was switched after an error.
|
||||
other than the explicitly approved time/cost guards were increased, and no provider
|
||||
was switched after an error.
|
||||
|
||||
Failures include existing CLI diagnostics and a fixed `review_pass`, completed-pass
|
||||
count, and available aggregate usage. New explicit codes include `pass_capacity`,
|
||||
@ -148,3 +158,10 @@ Deploy or roll back through Git and Flux. Revert only the multi-pass commits to
|
||||
restore the preceding single-pass policy while preserving the earlier CLI diagnostic
|
||||
fix. Wait for active jobs before a worker restart because completed results are
|
||||
held in memory. No Vault credential or ingress/routing changes are required.
|
||||
|
||||
The first live 363-case multi-pass run at the previous 900-second / USD 5
|
||||
guards stopped after 747.513 seconds during large-family review. Its actual CLI
|
||||
final subtype was `error_max_budget_usd`; three full passes had completed. The CLI
|
||||
reported an aggregate estimate of 5.62765, showing the documented within-generation
|
||||
overshoot. It was not a timeout, turn-limit failure, or accepted partial answer.
|
||||
The adapter now reports this condition as `job_cost_budget_exhausted`.
|
||||
|
||||
@ -105,10 +105,10 @@ def main():
|
||||
env['ANTHROPIC_BASE_URL'] = 'http://127.0.0.1:' + str(server.server_port)
|
||||
return env
|
||||
|
||||
def backend(request, cancel, *, invocation):
|
||||
def backend(request, cancel, *, invocation, progress=None):
|
||||
active.clear()
|
||||
active.update(invocation)
|
||||
return original_generate(request, cancel, invocation=invocation)
|
||||
return original_generate(request, cancel, invocation=invocation, progress=progress)
|
||||
|
||||
suite_backends.claude_environment = environment
|
||||
suite_backends.claude_generate = backend
|
||||
|
||||
@ -12,7 +12,7 @@ import re
|
||||
import subprocess
|
||||
import threading
|
||||
|
||||
from suite_contract import (EXECUTION_REVISION, MAX_BODY, MAX_CASES, MAX_RESULT, MODELS, PROMPT_REVISION,
|
||||
from suite_contract import (COST_LIMIT, EXECUTION_REVISION, MAX_BODY, MAX_CASES, MAX_RESULT, MODELS, PROMPT_REVISION,
|
||||
PROMPT_SHA256, REVISION, TIMEOUT, Problem, encoded,
|
||||
preflight, validate_request)
|
||||
from suite_jobs import Jobs
|
||||
@ -137,6 +137,8 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"strategy": ["whole_suite"], "max_request_bytes": MAX_BODY,
|
||||
"max_result_bytes": MAX_RESULT, "max_cases": MAX_CASES,
|
||||
"max_seconds": TIMEOUT, "concurrency": 1, "queue": False,
|
||||
"max_cost_usd": COST_LIMIT, "cost_basis": "CLI estimate guard; not subscription billing",
|
||||
"progress_interval_seconds": 5,
|
||||
"result_retention_seconds": 3600, "idempotency_retention_seconds": 604800,
|
||||
"tokenizer": None, "provider_retention_verified": False})
|
||||
match = re.fullmatch(r"/v1/synthetic/(14|75|363)", self.path)
|
||||
|
||||
@ -167,6 +167,8 @@ def parse_claude(raw, expected_model, **process_info):
|
||||
if not initialized or not final:
|
||||
fail("incomplete_generation", "missing_init_or_final_event")
|
||||
if final.get("is_error") or final.get("subtype") != "success":
|
||||
if final.get("subtype") == "error_max_budget_usd":
|
||||
fail("job_cost_budget_exhausted", "cli_final_result")
|
||||
status = final.get("api_error_status")
|
||||
code = "rate_limit" if status == 429 else "incomplete_generation"
|
||||
if status in (401, 403):
|
||||
@ -204,7 +206,7 @@ def parse_claude(raw, expected_model, **process_info):
|
||||
"turns": diagnostics["turns"], "cli_diagnostics": diagnostics}
|
||||
|
||||
|
||||
def claude_generate(request, cancel, *, invocation=None):
|
||||
def claude_generate(request, cancel, *, invocation=None, progress=None):
|
||||
"""Run one fresh job in tmpfs; input, output, configuration and caches expire together."""
|
||||
token = Path("/vault/secrets/claude-token").read_text().strip()
|
||||
if not token:
|
||||
@ -229,13 +231,21 @@ def claude_generate(request, cancel, *, invocation=None):
|
||||
raise Problem("backend_unavailable", 503, failure_stage="process_start",
|
||||
**snapshot("", subprocess_timeout_seconds=seconds)) from None
|
||||
deadline = time.monotonic() + seconds
|
||||
next_progress = 0.0
|
||||
try:
|
||||
while process.poll() is None:
|
||||
if cancel.wait(0.1):
|
||||
raise Problem("cancelled", 409)
|
||||
if time.monotonic() > deadline:
|
||||
raise Problem("timeout", 504)
|
||||
if (root / "output").stat().st_size > OUTPUT_BYTES:
|
||||
activity = (root / "output").stat()
|
||||
if progress and time.monotonic() >= next_progress:
|
||||
# File activity proves CLI activity, not semantic completion.
|
||||
# Do not read event bodies or emit model reasoning as progress.
|
||||
progress({"cli_running": True, "cli_output_bytes": activity.st_size,
|
||||
"last_cli_activity_seconds_ago": round(max(0, time.time() - activity.st_mtime), 1)})
|
||||
next_progress = time.monotonic() + 5
|
||||
if activity.st_size > OUTPUT_BYTES:
|
||||
raise Problem("response_too_large", 502)
|
||||
except Problem as exc:
|
||||
failure = exc
|
||||
|
||||
@ -8,12 +8,13 @@ from collections import Counter
|
||||
|
||||
REVISION = "suite-v6-20260929"
|
||||
PROMPT_REVISION = "implementation-proximity-multipass-v3-20260929"
|
||||
EXECUTION_REVISION = "suite-multipass-v1-20260929"
|
||||
EXECUTION_REVISION = "suite-multipass-v2-20260929"
|
||||
CLAUDE_MAX_TURNS = 6
|
||||
MAX_BODY = 1 << 20
|
||||
MAX_RESULT = 1 << 20
|
||||
MAX_CASES = 400
|
||||
TIMEOUT = 900
|
||||
TIMEOUT = 1200
|
||||
COST_LIMIT = 10.0
|
||||
RESULT_TTL = 3600
|
||||
FIELDS = {"description", "success_criteria", "preconditions", "operating_condition",
|
||||
"case_type", "verification_method", "target", "swci", "verifies",
|
||||
@ -142,10 +143,10 @@ def validate_request(raw, permissions):
|
||||
if execution.get("strategy", "whole_suite") != "whole_suite":
|
||||
raise Problem("unsupported_strategy", 422)
|
||||
seconds = execution.get("max_seconds", TIMEOUT)
|
||||
cost = execution.get("max_cost_usd", 5.0)
|
||||
cost = execution.get("max_cost_usd", COST_LIMIT)
|
||||
if type(seconds) is not int or not 10 <= seconds <= TIMEOUT:
|
||||
raise Problem("invalid_timeout")
|
||||
if type(cost) not in (int, float) or not 0 < cost <= 5:
|
||||
if type(cost) not in (int, float) or not 0 < cost <= COST_LIMIT:
|
||||
raise Problem("invalid_cost_limit")
|
||||
for key in ("campaign", "suite"):
|
||||
if type(raw[key]) is not str or not 1 <= len(raw[key]) <= 128:
|
||||
|
||||
@ -174,6 +174,9 @@ class Jobs:
|
||||
document.update(status="failed", error={"code": "internal_worker_error"})
|
||||
finally:
|
||||
document["wall_seconds"] = round(time.monotonic() - started, 3)
|
||||
if "execution_progress" in document:
|
||||
document["execution_progress"].update(cli_running=False, heartbeat_at=time.time(),
|
||||
job_elapsed_seconds=document["wall_seconds"])
|
||||
with self.lock:
|
||||
if event.is_set() and document["status"] == "completed":
|
||||
document.update(status="cancelled", error={"code": "cancelled"})
|
||||
|
||||
@ -100,7 +100,8 @@ class Workflow:
|
||||
def __init__(self, request, selected, cancel, client_ip, progress=None):
|
||||
self.request, self.provider, self.cancel = request, selected["provider"], cancel
|
||||
self.client_ip, self.progress = client_ip, progress or (lambda _: None)
|
||||
self.deadline = time.monotonic() + request["execution"]["max_seconds"]
|
||||
self.started = time.monotonic()
|
||||
self.deadline = self.started + request["execution"]["max_seconds"]
|
||||
self.cost_limit, self.spent = request["execution"]["max_cost_usd"], 0.0
|
||||
self.records = []
|
||||
self.last_metadata = {}
|
||||
@ -129,10 +130,19 @@ class Workflow:
|
||||
raise Problem("job_cost_budget_exhausted", 502)
|
||||
effective = {**source, "execution": {**source["execution"],
|
||||
"max_seconds": remaining_seconds, "max_cost_usd": remaining_cost}}
|
||||
self.progress({"current_pass": stage, "completed_model_passes": len(self.records), "passes": self.records})
|
||||
started = time.monotonic()
|
||||
def report(activity=None):
|
||||
now = time.monotonic()
|
||||
self.progress({"current_pass": stage, "completed_model_passes": len(self.records),
|
||||
"maximum_model_passes": MAX_PASSES, "passes": self.records,
|
||||
"heartbeat_at": time.time(), "pass_elapsed_seconds": round(now - started, 1),
|
||||
"job_elapsed_seconds": round(now - self.started, 1),
|
||||
"job_remaining_seconds": round(max(0, self.deadline - now), 1),
|
||||
"cost_used_usd_estimate": round(self.spent, 8),
|
||||
"cost_limit_usd_estimate": self.cost_limit, **(activity or {})})
|
||||
report()
|
||||
if self.provider == "claude":
|
||||
value, metadata = suite_backends.claude_generate(effective, self.cancel, invocation=call)
|
||||
value, metadata = suite_backends.claude_generate(effective, self.cancel, invocation=call, progress=report)
|
||||
elif self.provider == "local":
|
||||
value, metadata = suite_backends.local_generate(effective, self.cancel, self.client_ip, invocation=call)
|
||||
else:
|
||||
@ -153,7 +163,7 @@ class Workflow:
|
||||
raise Problem("budget_accounting_unavailable", 502)
|
||||
self.spent += cost
|
||||
self.checkpoint()
|
||||
self.progress({"current_pass": stage, "completed_model_passes": len(self.records), "passes": self.records})
|
||||
report({"cli_running": False})
|
||||
return value
|
||||
|
||||
def execute(self):
|
||||
|
||||
@ -36,7 +36,7 @@ spec:
|
||||
app: hermes-suite-planner
|
||||
annotations:
|
||||
fluentbit.io/exclude: "true"
|
||||
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v1-20260929
|
||||
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v2-20260929
|
||||
vault.hashicorp.com/agent-inject: "true"
|
||||
vault.hashicorp.com/agent-pre-populate-only: "true"
|
||||
vault.hashicorp.com/agent-init-first: "true"
|
||||
|
||||
331
testing/fixtures/suite_balanced_families.json
Normal file
331
testing/fixtures/suite_balanced_families.json
Normal file
@ -0,0 +1,331 @@
|
||||
{
|
||||
"request": {
|
||||
"campaign": "SYNTHETIC",
|
||||
"suite": "BALANCED-IMPLEMENTATION",
|
||||
"cases": [
|
||||
{
|
||||
"alias": "CASE-0001",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.",
|
||||
"success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0002",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
|
||||
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0003",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
|
||||
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0004",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0005",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.",
|
||||
"success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0006",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
|
||||
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0007",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
|
||||
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0008",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0009",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.",
|
||||
"success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0010",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
|
||||
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0011",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
|
||||
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0012",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0013",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.",
|
||||
"success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0014",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
|
||||
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0015",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
|
||||
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0016",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0017",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.",
|
||||
"success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0018",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
|
||||
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0019",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
|
||||
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0020",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0021",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use the reset RPC adapter with an in-memory transport stub and deterministic response fixtures.",
|
||||
"success_criteria": "Send a reset RPC, observe the returned status object and reset counter, and assert acceptance or rejection plus the expected counter change. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0022",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
|
||||
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0023",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
|
||||
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0024",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0025",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use a pulse generator and an oscilloscope adapter connected to the reset and ready lines; clear captured samples between cases.",
|
||||
"success_criteria": "Apply a reset pulse, capture both electrical traces, measure their relative transition times, and assert the supplied timing bound and recovery sequence. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0026",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
|
||||
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0027",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0028",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
|
||||
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0029",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0030",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
|
||||
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0031",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0032",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
|
||||
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0033",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0034",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Load offline reset-analysis report files through the same report parser and a rule-to-severity policy fixture; do not execute the target.",
|
||||
"success_criteria": "Parse report entries, count reset-related findings for the requested rule and severity, and assert the count and report location against the policy fixture. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0035",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0036",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "nominal"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0037",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
},
|
||||
{
|
||||
"alias": "CASE-0038",
|
||||
"description": "Verify reset recovery under a selected operating condition. Preserve each objective independently; similar reset terminology does not establish common test machinery.",
|
||||
"preconditions": "Use concurrent producer and consumer task drivers, queue-depth instrumentation, and sequence-number recording around a resettable bounded queue.",
|
||||
"success_criteria": "Drive the bounded queue through reset while producers run, observe depth and ordered delivery, and assert rejected writes and recovery after draining. Vary the supplied expected values while reusing these observations.",
|
||||
"case_type": "fault injection"
|
||||
}
|
||||
],
|
||||
"routing": {
|
||||
"allow_external": true,
|
||||
"allowed_external_providers": [
|
||||
"claude"
|
||||
]
|
||||
},
|
||||
"execution": {
|
||||
"strategy": "whole_suite",
|
||||
"max_seconds": 900,
|
||||
"max_cost_usd": 5
|
||||
}
|
||||
},
|
||||
"expected_implementation_family": {
|
||||
"CASE-0001": "command",
|
||||
"CASE-0002": "timing",
|
||||
"CASE-0003": "report",
|
||||
"CASE-0004": "queue",
|
||||
"CASE-0005": "command",
|
||||
"CASE-0006": "timing",
|
||||
"CASE-0007": "report",
|
||||
"CASE-0008": "queue",
|
||||
"CASE-0009": "command",
|
||||
"CASE-0010": "timing",
|
||||
"CASE-0011": "report",
|
||||
"CASE-0012": "queue",
|
||||
"CASE-0013": "command",
|
||||
"CASE-0014": "timing",
|
||||
"CASE-0015": "report",
|
||||
"CASE-0016": "queue",
|
||||
"CASE-0017": "command",
|
||||
"CASE-0018": "timing",
|
||||
"CASE-0019": "report",
|
||||
"CASE-0020": "queue",
|
||||
"CASE-0021": "command",
|
||||
"CASE-0022": "timing",
|
||||
"CASE-0023": "report",
|
||||
"CASE-0024": "queue",
|
||||
"CASE-0025": "timing",
|
||||
"CASE-0026": "report",
|
||||
"CASE-0027": "queue",
|
||||
"CASE-0028": "report",
|
||||
"CASE-0029": "queue",
|
||||
"CASE-0030": "report",
|
||||
"CASE-0031": "queue",
|
||||
"CASE-0032": "report",
|
||||
"CASE-0033": "queue",
|
||||
"CASE-0034": "report",
|
||||
"CASE-0035": "queue",
|
||||
"CASE-0036": "queue",
|
||||
"CASE-0037": "queue",
|
||||
"CASE-0038": "queue"
|
||||
},
|
||||
"expected_natural_sizes": [
|
||||
6,
|
||||
7,
|
||||
11,
|
||||
14
|
||||
]
|
||||
}
|
||||
@ -43,7 +43,7 @@ def request():
|
||||
@pytest.mark.parametrize("subtype,code", [
|
||||
("error_max_turns", "incomplete_generation"),
|
||||
("error_max_structured_output_retries", "incomplete_generation"),
|
||||
("error_max_budget_usd", "incomplete_generation"),
|
||||
("error_max_budget_usd", "job_cost_budget_exhausted"),
|
||||
("error_during_execution", "incomplete_generation"),
|
||||
(CANARY, "incomplete_generation"),
|
||||
])
|
||||
@ -179,6 +179,8 @@ def test_actual_subprocess_paths_cleanup_and_safe_metadata(tmp_path, monkeypatch
|
||||
script = "raise SystemExit(1)"
|
||||
elif mode == "oversize":
|
||||
monkeypatch.setattr(suite_backends, "OUTPUT_BYTES", 64)
|
||||
elif mode == "success":
|
||||
script += "time.sleep(0.25)"
|
||||
command = ["/not-installed-synthetic-cli"] if mode == "start" else [sys.executable, "-c", script]
|
||||
monkeypatch.setattr(suite_backends, "claude_command", lambda *_: command)
|
||||
value, cancel = request(), threading.Event()
|
||||
@ -196,9 +198,13 @@ def test_actual_subprocess_paths_cleanup_and_safe_metadata(tmp_path, monkeypatch
|
||||
assert details["termination_reason"] == mode.replace("cancel", "cancelled")
|
||||
assert CANARY not in json.dumps(details)
|
||||
else:
|
||||
_, metadata = suite_backends.claude_generate(value, cancel)
|
||||
updates = []
|
||||
_, metadata = suite_backends.claude_generate(value, cancel, progress=updates.append)
|
||||
assert metadata["cli_diagnostics"]["exit_code"] == 0
|
||||
assert metadata["temporary_files_deleted"] is True
|
||||
assert updates and updates[-1]["cli_output_bytes"] > 0
|
||||
assert updates[-1]["cli_running"] is True
|
||||
assert CANARY not in json.dumps(updates)
|
||||
assert list(tmp_path.iterdir()) == []
|
||||
|
||||
|
||||
|
||||
@ -71,12 +71,14 @@ def install_backend(monkeypatch, source, reconciled, *, a=None, b=None, reviewed
|
||||
audited = review(audited or reviewed, reconciled)
|
||||
answers = {'proposal_a': a or public(reconciled), 'proposal_b': b or public(reconciled),
|
||||
'reconciliation': reconciled, 'large_family_review': reviewed, 'decision_audit': audited}
|
||||
def backend(value, cancel, *, invocation):
|
||||
def backend(value, cancel, *, invocation, progress=None):
|
||||
payload = json.loads(invocation['input'])
|
||||
assert {c['alias']: c for c in payload['suite']['cases']} == {c['alias']: c for c in source['cases']}
|
||||
assert payload['suite']['campaign'] == source['campaign']
|
||||
assert payload['suite']['suite'] == source['suite']
|
||||
calls.append((copy.deepcopy(value), copy.deepcopy(invocation)))
|
||||
if progress:
|
||||
progress({'cli_running': True, 'cli_output_bytes': 120, 'last_cli_activity_seconds_ago': 0})
|
||||
cost = costs[len(calls)-1] if costs else 0.1
|
||||
return copy.deepcopy(answers[invocation['stage']]), {
|
||||
'model': MODELS['claude']['model'], 'usage': {'input_tokens': 100, 'output_tokens': 40},
|
||||
@ -100,7 +102,7 @@ def test_coherent_families_are_balanced_not_semantically_fragmented(monkeypatch,
|
||||
assert [g['name'] for g in result['groups']] == [f'Reset recovery ({i}/{len(sizes)})' for i in range(1,len(sizes)+1)]
|
||||
assert all('Work-size part' in g['description'] for g in result['groups'])
|
||||
assert metadata['review_summary']['decision_audit'][0]['decision'] == 'keep'
|
||||
assert [call[0]['execution']['max_cost_usd'] for call in calls] == pytest.approx([5,4.9,4.8,4.7,4.6])
|
||||
assert [call[0]['execution']['max_cost_usd'] for call in calls] == pytest.approx([10,9.9,9.8,9.7,9.6])
|
||||
assert all(calls[i+1][0]['execution']['max_seconds'] < calls[i][0]['execution']['max_seconds'] for i in range(4))
|
||||
validate_result(result, source)
|
||||
|
||||
@ -119,6 +121,42 @@ def test_independent_orders_and_content_are_reproducible(monkeypatch):
|
||||
assert 'proposal_a' in json.loads(calls[2][1]['input'])['review_material']
|
||||
|
||||
|
||||
def test_progress_reports_activity_without_content_or_false_percentage(monkeypatch):
|
||||
source = request(7)
|
||||
family = natural(source, [[c['alias'] for c in source['cases']]])
|
||||
install_backend(monkeypatch, source, family)
|
||||
updates = []
|
||||
workflow.generate(source, preflight(source), threading.Event(), '192.168.22.8', updates.append)
|
||||
active = [value for value in updates if value.get('cli_running')]
|
||||
assert len(active) == 5
|
||||
assert [v['completed_model_passes'] for v in active] == [0,1,2,3,4]
|
||||
assert all(v['maximum_model_passes'] == 5 and v['cli_output_bytes'] == 120 for v in active)
|
||||
assert all(v['heartbeat_at'] > 0 and v['job_remaining_seconds'] <= 1200 for v in active)
|
||||
assert all(v['last_cli_activity_seconds_ago'] == 0 for v in active)
|
||||
assert updates[-1]['cli_running'] is False
|
||||
assert CANARY not in json.dumps(updates)
|
||||
|
||||
|
||||
def test_twenty_minute_job_limit_retains_smaller_client_deadlines():
|
||||
source = request(7)
|
||||
assert source['execution']['max_seconds'] == 1200
|
||||
source['execution']['max_seconds'] = 900
|
||||
assert validate_request(source, ['claude'])['execution']['max_seconds'] == 900
|
||||
source['execution']['max_seconds'] = 1201
|
||||
with pytest.raises(Problem, match='invalid_timeout'):
|
||||
validate_request(source, ['claude'])
|
||||
|
||||
|
||||
def test_cli_estimate_guard_default_and_client_override():
|
||||
source = request(7)
|
||||
assert source['execution']['max_cost_usd'] == 10
|
||||
source['execution']['max_cost_usd'] = 5
|
||||
assert validate_request(source, ['claude'])['execution']['max_cost_usd'] == 5
|
||||
source['execution']['max_cost_usd'] = 10.1
|
||||
with pytest.raises(Problem, match='invalid_cost_limit'):
|
||||
validate_request(source, ['claude'])
|
||||
|
||||
|
||||
def test_real_semantic_subdivisions_precede_work_sizing(monkeypatch):
|
||||
source = request(11)
|
||||
aliases = [c['alias'] for c in source['cases']]
|
||||
@ -230,6 +268,7 @@ def test_existing_suite_sizes_have_separate_natural_and_task_counts(monkeypatch,
|
||||
|
||||
def test_whole_job_cost_budget_and_failure_no_partial_answer(monkeypatch):
|
||||
source = request(14)
|
||||
source['execution']['max_cost_usd'] = 5
|
||||
family = natural(source,[[c['alias'] for c in source['cases']]])
|
||||
calls = install_backend(monkeypatch,source,family,costs=[1,2,3])
|
||||
with pytest.raises(Problem,match='job_cost_budget_exhausted') as raised:
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user