docs: record suite CLI failure diagnosis and synthetic regression

This commit is contained in:
jenkins 2026-09-29 11:47:21 -05:00
parent 2d762776c1
commit edb83b78d9
2 changed files with 401 additions and 2 deletions

View File

@ -0,0 +1,365 @@
{
"synthetic_only": true,
"request_bytes": 9607,
"client_wall_seconds": 23.186,
"health": {
"configuration_revision": "suite-v6-20260929",
"execution_revision": "claude-diagnostics-turns-v1-20260929",
"prompt_revision": "implementation-proximity-v2-20260929",
"prompt_sha256": "5002baf3f38436cbf0cdbfc769b5d16f9c13bf86d6f1e32c59d5e3498e16b552",
"status": "ready"
},
"preflight": {
"availability": "checked_at_dispatch",
"routing": {
"allow_external": true,
"allowed_external_providers": [
"claude"
]
},
"selection": {
"backend": "claude-code-2.1.226",
"case_count": 14,
"cli_model": "claude-opus-4-8[1m]",
"configuration_revision": "suite-v6-20260929",
"context": 1000000,
"enabled": true,
"execution_revision": "claude-diagnostics-turns-v1-20260929",
"input_bytes": 13568,
"input_count_method": "UTF-8 byte upper bound plus reserved harness overhead; not a tokenizer",
"input_token_bound": 21760,
"input_token_count": null,
"max_turns": 6,
"model": "claude-opus-4-8",
"output": 64000,
"output_reservation_tokens": 10910,
"output_reservation_verified": false,
"overhead": 8192,
"prompt_revision": "implementation-proximity-v2-20260929",
"prompt_sha256": "5002baf3f38436cbf0cdbfc769b5d16f9c13bf86d6f1e32c59d5e3498e16b552",
"provider": "claude",
"reasoning": "medium",
"source_sha256": "37c6ca89c6c8fb170b104b5aec4e5aae65463e5d74fe284ab7ef1ecb55a0aac3"
},
"status": "eligible"
},
"idempotent_replay": true,
"invalid_credential_rejected": true,
"job": {
"attempted_destinations": [
"switchyard:atlas/planning/claude",
"claude:claude-opus-4-8"
],
"cli_diagnostics": {
"api_error_status": null,
"assistant_json_text_present": false,
"assistant_structured_tool_input_present": true,
"compaction_event_seen": false,
"configured_max_output_tokens": 64000,
"cost_usd_estimate": 0.10299,
"duration_api_ms": 21970,
"execution_revision": "claude-diagnostics-turns-v1-20260929",
"exit_code": 0,
"final_event_seen": true,
"final_event_subtype": "success",
"final_event_type": "result",
"final_is_error": false,
"final_json_text_present": true,
"init_event_seen": true,
"invalid_event_count": 0,
"last_assistant_stop_reason": null,
"max_turns": 6,
"observed_model_limits": [
{
"contextWindow": 1000000,
"maxOutputTokens": 64000
}
],
"provider_stop_reason": "tool_use",
"provider_timeout_seconds": null,
"reasoning_effort": "medium",
"reasoning_token_limit": null,
"structured_output_is_object": true,
"structured_output_location": "result.structured_output",
"structured_output_present": true,
"structured_retry_limit_reached": false,
"subprocess_timeout_seconds": 899.9719768599607,
"termination_reason": "exited",
"termination_signal": null,
"turn_limit_reached": false,
"turns": 3,
"usage": {
"cache_creation": {
"ephemeral_1h_input_tokens": 0,
"ephemeral_5m_input_tokens": 0
},
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"input_tokens": 11073,
"output_tokens": 1905,
"server_tool_use": {
"web_fetch_requests": 0,
"web_search_requests": 0
}
}
},
"compaction": false,
"compaction_signal": "CLI events and disabled compaction",
"configuration_revision": "suite-v6-20260929",
"cost_usd_estimate": 0.10299,
"created_at": 1790700329.0014415,
"duration_api_ms": 21970,
"execution_revision": "claude-diagnostics-turns-v1-20260929",
"job_id": "6ad9067bfbc1441890a1fe49b8f3de70",
"model": "claude-opus-4-8",
"model_usage": {
"claude-opus-4-8[1m]": {
"canonicalModel": "claude-opus-4-8",
"contextWindow": 1000000,
"maxOutputTokens": 64000,
"provider": "firstParty"
}
},
"prompt_revision": "implementation-proximity-v2-20260929",
"prompt_sha256": "5002baf3f38436cbf0cdbfc769b5d16f9c13bf86d6f1e32c59d5e3498e16b552",
"result": {
"groups": [
{
"description": "Shared JSON command client and reset-state fixture: issue mode, capture returned JSON, assert acceptance flag and reported state. Members vary inputs/outcomes (true/false flags, error_code field) reusing same response observation.",
"members": [
"CASE-0001",
"CASE-0005",
"CASE-0009",
"CASE-0013"
],
"name": "Command/readback JSON client mode-acceptance tests"
},
{
"description": "Shared bench controller and oscilloscope adapter, same probe/capture: apply mode transition, capture waveform, assert amplitude and ordering. CASE-0006 adds transition-duration from captured samples. CASE-0002 and CASE-0014 identical.",
"members": [
"CASE-0002",
"CASE-0006",
"CASE-0010",
"CASE-0014"
],
"name": "Bench controller oscilloscope waveform tests"
},
{
"description": "Shared report parser and rule-policy fixture, no target execution: parse report, count findings for selected rule/severity, compare against qualitative bound. Members differ only in variant inputs/bounds.",
"members": [
"CASE-0003",
"CASE-0007",
"CASE-0011"
],
"name": "Offline static-analysis report parser count tests"
},
{
"description": "Distinct machinery: thermal chamber temperature cycling plus calibrated dimensional gauge measuring enclosure expansion against a generalized bound. Independent equipment; no shared observation with other families.",
"members": [
"CASE-0004"
],
"name": "Thermal chamber enclosure expansion measurement"
},
{
"description": "Distinct anechoic enclosure, microphone capture adapter, and spectral-analysis fixture: record acoustic output and assert each spectral peak below a frequency-dependent bound. No shared oscilloscope or chamber.",
"members": [
"CASE-0008"
],
"name": "Anechoic acoustic spectral-peak measurement"
},
{
"description": "Interface, observation mechanism, and fixture omitted; requires verifying a latent-state relationship with no defined stimulus or observation. Insufficient evidence to merge with any implemented mechanism.",
"members": [
"CASE-0012"
],
"name": "Unspecified latent-state verification (insufficient evidence)"
}
]
},
"result_retention_seconds": 3600,
"routing": {
"allow_external": true,
"allowed_external_providers": [
"claude"
]
},
"selection": {
"backend": "claude-code-2.1.226",
"case_count": 14,
"cli_model": "claude-opus-4-8[1m]",
"configuration_revision": "suite-v6-20260929",
"context": 1000000,
"enabled": true,
"execution_revision": "claude-diagnostics-turns-v1-20260929",
"input_bytes": 13568,
"input_count_method": "UTF-8 byte upper bound plus reserved harness overhead; not a tokenizer",
"input_token_bound": 21760,
"input_token_count": null,
"max_turns": 6,
"model": "claude-opus-4-8",
"output": 64000,
"output_reservation_tokens": 10910,
"output_reservation_verified": false,
"overhead": 8192,
"prompt_revision": "implementation-proximity-v2-20260929",
"prompt_sha256": "5002baf3f38436cbf0cdbfc769b5d16f9c13bf86d6f1e32c59d5e3498e16b552",
"provider": "claude",
"reasoning": "medium",
"source_sha256": "37c6ca89c6c8fb170b104b5aec4e5aae65463e5d74fe284ab7ef1ecb55a0aac3"
},
"status": "completed",
"temporary_files_deleted": true,
"truncation": false,
"turns": 3,
"usage": {
"cache_creation": {
"ephemeral_1h_input_tokens": 0,
"ephemeral_5m_input_tokens": 0
},
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"input_tokens": 11073,
"output_tokens": 1905,
"server_tool_use": {
"web_fetch_requests": 0,
"web_search_requests": 0
}
},
"wall_seconds": 22.694
},
"quality": {
"coverage": true,
"families": 6,
"pair_precision": 1.0,
"pair_recall": 1.0,
"false_merge_pairs": 0,
"missed_merge_pairs": 0
},
"singleton_families": 3,
"deployed_commit": "2d762776c100c280d491cfbd11119fb57c627ab0",
"native_cli_reproduction": [
{
"max_turns": 3,
"exit_code": 1,
"wall_seconds": 0.391,
"final": {
"type": "result",
"subtype": "error_max_turns",
"is_error": true,
"num_turns": 4,
"stop_reason": "tool_use",
"usage": {
"input_tokens": 30,
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"output_tokens": 15,
"server_tool_use": {
"web_search_requests": 0,
"web_fetch_requests": 0
},
"service_tier": "standard",
"cache_creation": {
"ephemeral_1h_input_tokens": 0,
"ephemeral_5m_input_tokens": 0
},
"inference_geo": "",
"iterations": [],
"speed": "standard"
}
},
"structured_output_present": false,
"requests": [
{
"stream": true,
"max_tokens": 64000,
"http_timeout_header": "600",
"whole_input": true,
"system_preserved": true
},
{
"stream": true,
"max_tokens": 64000,
"http_timeout_header": "600",
"whole_input": true,
"system_preserved": true
},
{
"stream": true,
"max_tokens": 64000,
"http_timeout_header": "600",
"whole_input": true,
"system_preserved": true
}
]
},
{
"max_turns": 6,
"exit_code": 0,
"wall_seconds": 0.411,
"final": {
"type": "result",
"subtype": "success",
"is_error": false,
"num_turns": 5,
"stop_reason": "tool_use",
"usage": {
"input_tokens": 40,
"cache_creation_input_tokens": 0,
"cache_read_input_tokens": 0,
"output_tokens": 20,
"server_tool_use": {
"web_search_requests": 0,
"web_fetch_requests": 0
},
"service_tier": "standard",
"cache_creation": {
"ephemeral_1h_input_tokens": 0,
"ephemeral_5m_input_tokens": 0
},
"inference_geo": "",
"iterations": [],
"speed": "standard"
}
},
"structured_output_present": true,
"requests": [
{
"stream": true,
"max_tokens": 64000,
"http_timeout_header": "600",
"whole_input": true,
"system_preserved": true
},
{
"stream": true,
"max_tokens": 64000,
"http_timeout_header": "600",
"whole_input": true,
"system_preserved": true
},
{
"stream": true,
"max_tokens": 64000,
"http_timeout_header": "600",
"whole_input": true,
"system_preserved": true
},
{
"stream": true,
"max_tokens": 64000,
"http_timeout_header": "600",
"whole_input": true,
"system_preserved": true
}
]
}
],
"unit_tests_passed": 65,
"content_logging_check": {
"routine_records": 2,
"unexpected_fields": 0,
"non_json_records": 0
},
"temporary_suite_directories_after_tests": 0,
"description_review": "No observed incorrect attribution of another family objective. Mechanisms, variations, and missing information match assigned synthetic cases."
}

View File

@ -28,7 +28,7 @@ In `suite_backends.parse_claude`, `incomplete_generation` could mean:
`model_context_window_exceeded`.
In `suite_backends.claude_generate`, it also meant a nonzero subprocess exit
**after** parsing a otherwise acceptable structured result. Thus a valid payload
**after** parsing an otherwise acceptable structured result. Thus a valid payload
could have existed on that branch, but the historical artifact needed to prove
or validate it no longer exists.
@ -122,14 +122,48 @@ cleanup, privacy canaries, exact membership, and idempotency. Native loopback
reproduction also passes. Kustomize rendering and client dry-run passed; Flux diff
shows only the planner ConfigMap and planner restart annotation.
## Deployed synthetic verification
Flux deployed code commit `2d762776c100c280d491cfbd11119fb57c627ab0`; the new
planner pod is Ready. LAN HTTPS testing ran from `titan-jh` at 192.168.22.8 to
192.168.22.50 with hostname TLS verification, no proxies, and no redirects.
The user's laptop was not used. No active job was present before rollout.
Fresh synthetic job `6ad9067bfbc1441890a1fe49b8f3de70` completed:
- 14 cases; 9,607 serialized request bytes.
- 22.694 server wall seconds, 23.186 client wall seconds; no job retry or rate-limit event.
- CLI exit 0, final `result/success`, `is_error=false`, three reported turns.
- Stop reason `tool_use`; structured object at `result.structured_output`.
- 11,073 input / 1,905 output tokens; 21,970 API milliseconds; USD 0.10299 CLI estimate.
- Canonical model `claude-opus-4-8`, firstParty, reported 1,000,000 context / 64,000 output.
- Six families, three singletons, exact alias coverage, pair precision/recall 1.0,
zero incorrect or missed merge pairs. No observed cross-family objective attribution.
- Same-key replay returned the same job; invalid credentials returned 401.
- Two routine log records contained only approved operational metadata; zero remaining
temporary suite directories after both live and loopback tests.
The native loopback three-versus-six-turn regression also passed on the deployed
binary. These checks establish the change's behavior on synthetic material, not
the deleted historical terminal condition or guaranteed success on future real jobs.
See [the synthetic verification evidence](evidence/hermes_suite_diagnostics_20260929.json).
## Recovery and the laptop's next attempt
The old answer cannot be recovered or validated after cleanup. No assignments
will be fabricated. No real suite has been rerun. The laptop must intentionally
create a **new attempt with a new Idempotency-Key** after deployment verification;
create a **new attempt with a new Idempotency-Key** after reviewing this verification;
reusing the old key correctly returns the same failed job. The input/request
schema and Python parsing contract do not require a change for this fix.
Rollback only this diagnostic/turn-limit commit through Git and Flux. Keep the
prior implementation-proximity prompt commit. Rollouts should occur with no active
jobs; in-memory completed results expire on a worker restart.
Exact rollback for this change (from a clean checkout, after active jobs finish):
```bash
git revert 2d762776c100c280d491cfbd11119fb57c627ab0
git push origin HEAD:main
flux reconcile kustomization hermes --namespace flux-system --with-source
```