hermes: raise suite job limits to thirty minutes and thirty dollars
This commit is contained in:
parent
0382ac463d
commit
3036064bf7
@ -206,14 +206,14 @@
|
||||
"max_seconds": {
|
||||
"type": "integer",
|
||||
"minimum": 10,
|
||||
"maximum": 1200,
|
||||
"default": 1200
|
||||
"maximum": 1800,
|
||||
"default": 1800
|
||||
},
|
||||
"max_cost_usd": {
|
||||
"type": "number",
|
||||
"exclusiveMinimum": 0,
|
||||
"maximum": 10,
|
||||
"default": 10,
|
||||
"maximum": 30,
|
||||
"default": 30,
|
||||
"description": "CLI estimated-cost guard, not subscription billing. Shared across all passes."
|
||||
}
|
||||
}
|
||||
|
||||
@ -11,7 +11,7 @@ most **64 characters**, unique after whitespace and case normalization.
|
||||
- Configuration: `suite-v6-20260929` (HTTP compatibility identifier).
|
||||
- Policy: `implementation-five-v1-20260929`.
|
||||
- Prompt: `implementation-proximity-multipass-v4-20260929`.
|
||||
- Execution: `suite-multipass-v3-20260929`.
|
||||
- Execution: `suite-multipass-v4-20260929`.
|
||||
|
||||
A single server-side job performs:
|
||||
|
||||
@ -136,7 +136,7 @@ invented percentage, because different passes have very different runtimes.
|
||||
|
||||
## Shared bounds and failure behavior
|
||||
|
||||
The user-approved 1200-second maximum job deadline and USD 10 CLI estimated-cost
|
||||
The user-approved 1800-second maximum job deadline and USD 30 CLI estimated-cost
|
||||
guard apply to **all calls combined**, including routing time. These are also the
|
||||
defaults when omitted. Explicit lower client limits remain effective. This guard
|
||||
uses Claude CLI estimated-cost accounting; it is not subscription billing. Each invocation receives
|
||||
@ -312,3 +312,9 @@ measurements are in [the acceptance evidence](evidence/hermes_suite_multipass_20
|
||||
Routine planner logs contained only allowed operational fields, the completed
|
||||
job's SQLite metadata contained no result/review/case content, and temporary CLI
|
||||
job directories were empty after completion.
|
||||
|
||||
Execution revision `suite-multipass-v4-20260929` raises only the shared job defaults
|
||||
and maximums to 1800 seconds and USD 30 in CLI estimate accounting. Explicit lower
|
||||
client limits remain effective. The existing request fields are unchanged. No
|
||||
inference or regression tests were rerun for this limit-only update, as requested;
|
||||
the acceptance measurements above retain their original revisions and bounds.
|
||||
|
||||
@ -13,8 +13,8 @@ The endpoint and request fields are unchanged. Optional top-level `review_summar
|
||||
is returned only with the authorized result. Configuration remains `suite-v6-20260929`;
|
||||
policy, prompt, and execution revisions identify the changed grouping behavior.
|
||||
The prior single-pass acceptance results below are historical and do not measure
|
||||
the new workflow. The new process shares the user-approved 1200-second deadline
|
||||
and USD 10 CLI estimate guard across all invocations. This is not subscription
|
||||
the new workflow. The new process shares the user-approved 1800-second deadline
|
||||
and USD 30 CLI estimate guard across all invocations. This is not subscription
|
||||
billing. Progress updates every five seconds during a CLI pass. The separate local-model endpoint is unchanged;
|
||||
the 8K local route cannot admit the new multi-pass reconciliation schema.
|
||||
|
||||
@ -200,7 +200,7 @@ LOCAL_ONLY_FIELD: token
|
||||
SYNTHETIC_EXTERNAL_FIELD: synthetic_token
|
||||
APPROVED_OPERATIONAL_FIELD: operational_token
|
||||
INITIAL_CONCURRENCY: 1; busy submissions return 429; no waiting queue
|
||||
JOB_TIMEOUT: up to 1200 seconds
|
||||
JOB_TIMEOUT: up to 1800 seconds
|
||||
HTTP_TIMEOUT: client 45 seconds; submit/status do not wait for inference
|
||||
TLS: existing worker.bstein.dev certificate; normal trusted CA verification
|
||||
```
|
||||
@ -253,8 +253,8 @@ UTF-8 byte limits, unique aliases, ownership, permissions, and exact result cove
|
||||
},
|
||||
"execution": {
|
||||
"strategy": "whole_suite",
|
||||
"max_seconds": 1200,
|
||||
"max_cost_usd": 10
|
||||
"max_seconds": 1800,
|
||||
"max_cost_usd": 30
|
||||
}
|
||||
}
|
||||
```
|
||||
@ -273,8 +273,8 @@ Missing routing policy means local-only. External provider names are `claude` an
|
||||
unknown providers, malformed booleans, duplicate names, and permission escalation
|
||||
are rejected before routing. `allow_external=false` cannot include providers.
|
||||
`whole_suite` is the only strategy. No batching, summarization, or reconciliation
|
||||
is silently substituted. `max_seconds` is 10-1200. The Claude CLI guard is at most
|
||||
USD 10 in its estimated usage accounting; this is not a verified subscription
|
||||
is silently substituted. `max_seconds` is 10-1800. The Claude CLI guard is at most
|
||||
USD 30 in its estimated usage accounting; this is not a verified subscription
|
||||
billing ceiling and can overshoot within a single provider call.
|
||||
|
||||
The gateway deterministically selects a fitting permitted backend, preferring
|
||||
@ -464,7 +464,7 @@ Small local requests reserve 1,024 overhead tokens and 2,048 output tokens withi
|
||||
8,192 context. The realistic 14/75/363 fixtures exceed this conservative local
|
||||
whole-suite budget and must use the approved hosted path or fail preflight.
|
||||
|
||||
All CLI invocations share at most 1200 seconds per suite job; cancellation and timeouts terminate its
|
||||
All CLI invocations share at most 1800 seconds per suite job; cancellation and timeouts terminate its
|
||||
process group. Async HTTP calls finish promptly, with a 30-second body-read timeout,
|
||||
40-second ingress response-header timeout, and recommended 45-second client timeout.
|
||||
Switchyard's ten-minute internal request limit carries only a tiny immediate routing
|
||||
@ -515,7 +515,7 @@ import json
|
||||
with open('synthetic-suite.json') as stream:
|
||||
request = json.load(stream)
|
||||
request['routing'] = {'allow_external': True, 'allowed_external_providers': ['claude']}
|
||||
request['execution'] = {'strategy': 'whole_suite', 'max_seconds': 1200, 'max_cost_usd': 10}
|
||||
request['execution'] = {'strategy': 'whole_suite', 'max_seconds': 1800, 'max_cost_usd': 30}
|
||||
with open('synthetic-request.json', 'w') as stream:
|
||||
json.dump(request, stream)
|
||||
PY
|
||||
|
||||
@ -8,13 +8,13 @@ from collections import Counter
|
||||
|
||||
REVISION = "suite-v6-20260929"
|
||||
PROMPT_REVISION = "implementation-proximity-multipass-v4-20260929"
|
||||
EXECUTION_REVISION = "suite-multipass-v3-20260929"
|
||||
EXECUTION_REVISION = "suite-multipass-v4-20260929"
|
||||
CLAUDE_MAX_TURNS = 6
|
||||
MAX_BODY = 1 << 20
|
||||
MAX_RESULT = 1 << 20
|
||||
MAX_CASES = 400
|
||||
TIMEOUT = 1200
|
||||
COST_LIMIT = 10.0
|
||||
TIMEOUT = 1800
|
||||
COST_LIMIT = 30.0
|
||||
RESULT_TTL = 3600
|
||||
FIELDS = {"description", "success_criteria", "preconditions", "operating_condition",
|
||||
"case_type", "verification_method", "target", "swci", "verifies",
|
||||
|
||||
@ -36,7 +36,7 @@ spec:
|
||||
app: hermes-suite-planner
|
||||
annotations:
|
||||
fluentbit.io/exclude: "true"
|
||||
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v3-20260929
|
||||
ai.bstein.dev/config-rev: suite-v6-multipass-cap5-v4-20260929
|
||||
vault.hashicorp.com/agent-inject: "true"
|
||||
vault.hashicorp.com/agent-pre-populate-only: "true"
|
||||
vault.hashicorp.com/agent-init-first: "true"
|
||||
|
||||
@ -119,7 +119,7 @@ def test_coherent_families_are_balanced_not_semantically_fragmented(monkeypatch,
|
||||
assert [g['name'] for g in result['groups']] == [f'Reset recovery ({i}/{len(sizes)})' for i in range(1,len(sizes)+1)]
|
||||
assert all('Work-size part' in g['description'] for g in result['groups'])
|
||||
assert metadata['review_summary']['decision_audit'][0]['decision'] == 'keep'
|
||||
assert [call[0]['execution']['max_cost_usd'] for call in calls] == pytest.approx([10,9.9,9.8,9.7,9.6])
|
||||
assert [call[0]['execution']['max_cost_usd'] for call in calls] == pytest.approx([30,29.9,29.8,29.7,29.6])
|
||||
assert all(calls[i+1][0]['execution']['max_seconds'] < calls[i][0]['execution']['max_seconds'] for i in range(4))
|
||||
validate_result(result, source)
|
||||
|
||||
@ -149,28 +149,28 @@ def test_progress_reports_activity_without_content_or_false_percentage(monkeypat
|
||||
assert len(active) == 5
|
||||
assert [v['completed_model_passes'] for v in active] == [0,1,2,3,4]
|
||||
assert all(v['maximum_model_passes'] == 5 and v['cli_output_bytes'] == 120 for v in active)
|
||||
assert all(v['heartbeat_at'] > 0 and v['job_remaining_seconds'] <= 1200 for v in active)
|
||||
assert all(v['heartbeat_at'] > 0 and v['job_remaining_seconds'] <= 1800 for v in active)
|
||||
assert all(v['last_cli_activity_seconds_ago'] == 0 for v in active)
|
||||
assert updates[-1]['cli_running'] is False
|
||||
assert CANARY not in json.dumps(updates)
|
||||
|
||||
|
||||
def test_twenty_minute_job_limit_retains_smaller_client_deadlines():
|
||||
def test_thirty_minute_job_limit_retains_smaller_client_deadlines():
|
||||
source = request(7)
|
||||
assert source['execution']['max_seconds'] == 1200
|
||||
assert source['execution']['max_seconds'] == 1800
|
||||
source['execution']['max_seconds'] = 900
|
||||
assert validate_request(source, ['claude'])['execution']['max_seconds'] == 900
|
||||
source['execution']['max_seconds'] = 1201
|
||||
source['execution']['max_seconds'] = 1801
|
||||
with pytest.raises(Problem, match='invalid_timeout'):
|
||||
validate_request(source, ['claude'])
|
||||
|
||||
|
||||
def test_cli_estimate_guard_default_and_client_override():
|
||||
source = request(7)
|
||||
assert source['execution']['max_cost_usd'] == 10
|
||||
assert source['execution']['max_cost_usd'] == 30
|
||||
source['execution']['max_cost_usd'] = 5
|
||||
assert validate_request(source, ['claude'])['execution']['max_cost_usd'] == 5
|
||||
source['execution']['max_cost_usd'] = 10.1
|
||||
source['execution']['max_cost_usd'] = 30.1
|
||||
with pytest.raises(Problem, match='invalid_cost_limit'):
|
||||
validate_request(source, ['claude'])
|
||||
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user