From 1c8c7cf836a1afdc96c0745490dfc1dffc065406 Mon Sep 17 00:00:00 2001 From: jenkins Date: Tue, 29 Sep 2026 08:44:20 -0500 Subject: [PATCH] docs: publish suite planning acceptance results and client contract --- .../suite_planning_request.schema.json | 222 +++++ .../suite_planning_result.schema.json | 42 + .../hermes_suite_acceptance_20260929.json | 301 ++++++ docs/evidence/hermes_suite_local_success.json | 86 ++ .../hermes_suite_results_synthetic.json | 883 ++++++++++++++++++ docs/evidence/hermes_suite_success_14.json | 150 +++ docs/hermes_suite_planning.md | 207 +++- 7 files changed, 1881 insertions(+), 10 deletions(-) create mode 100644 docs/contracts/suite_planning_request.schema.json create mode 100644 docs/contracts/suite_planning_result.schema.json create mode 100644 docs/evidence/hermes_suite_acceptance_20260929.json create mode 100644 docs/evidence/hermes_suite_local_success.json create mode 100644 docs/evidence/hermes_suite_results_synthetic.json create mode 100644 docs/evidence/hermes_suite_success_14.json diff --git a/docs/contracts/suite_planning_request.schema.json b/docs/contracts/suite_planning_request.schema.json new file mode 100644 index 00000000..d3719f8c --- /dev/null +++ b/docs/contracts/suite_planning_request.schema.json @@ -0,0 +1,222 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "$id": "urn:atlas:suite-planning:request:v6", + "title": "Suite planning request", + "type": "object", + "additionalProperties": false, + "required": [ + "campaign", + "suite", + "cases" + ], + "properties": { + "campaign": { + "type": "string", + "minLength": 1, + "maxLength": 128 + }, + "suite": { + "type": "string", + "minLength": 1, + "maxLength": 128 + }, + "cases": { + "type": "array", + "minItems": 1, + "maxItems": 400, + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "alias", + "description" + ], + "properties": { + "case_type": { + "type": [ + "string", + "null" + ], + "maxLength": 32768, + "description": "At most 32768 UTF-8 bytes; byte limit is enforced by the server." + }, + "description": { + "type": [ + "string", + "null" + ], + "maxLength": 32768, + "description": "At most 32768 UTF-8 bytes; byte limit is enforced by the server." + }, + "functional_area": { + "type": [ + "string", + "null" + ], + "maxLength": 32768, + "description": "At most 32768 UTF-8 bytes; byte limit is enforced by the server." + }, + "functional_group": { + "type": [ + "string", + "null" + ], + "maxLength": 32768, + "description": "At most 32768 UTF-8 bytes; byte limit is enforced by the server." + }, + "functional_group_name": { + "type": [ + "string", + "null" + ], + "maxLength": 32768, + "description": "At most 32768 UTF-8 bytes; byte limit is enforced by the server." + }, + "operating_condition": { + "type": [ + "string", + "null" + ], + "maxLength": 32768, + "description": "At most 32768 UTF-8 bytes; byte limit is enforced by the server." + }, + "preconditions": { + "type": [ + "string", + "null" + ], + "maxLength": 32768, + "description": "At most 32768 UTF-8 bytes; byte limit is enforced by the server." + }, + "success_criteria": { + "type": [ + "string", + "null" + ], + "maxLength": 32768, + "description": "At most 32768 UTF-8 bytes; byte limit is enforced by the server." + }, + "swci": { + "type": [ + "string", + "null" + ], + "maxLength": 32768, + "description": "At most 32768 UTF-8 bytes; byte limit is enforced by the server." + }, + "target": { + "type": [ + "string", + "null" + ], + "maxLength": 32768, + "description": "At most 32768 UTF-8 bytes; byte limit is enforced by the server." + }, + "verification_method": { + "type": [ + "string", + "null" + ], + "maxLength": 32768, + "description": "At most 32768 UTF-8 bytes; byte limit is enforced by the server." + }, + "verifies": { + "type": [ + "string", + "null" + ], + "maxLength": 32768, + "description": "At most 32768 UTF-8 bytes; byte limit is enforced by the server." + }, + "alias": { + "type": "string", + "pattern": "^CASE-[A-Za-z0-9_-]{1,48}$" + }, + "campaign": { + "type": "string", + "description": "If supplied, must exactly match the top-level campaign." + }, + "suite": { + "type": "string", + "description": "If supplied, must exactly match the top-level suite." + } + } + } + }, + "routing": { + "type": "object", + "additionalProperties": false, + "properties": { + "allow_external": { + "type": "boolean", + "default": false + }, + "allowed_external_providers": { + "type": "array", + "uniqueItems": true, + "items": { + "enum": [ + "claude", + "codex" + ] + }, + "default": [] + } + }, + "allOf": [ + { + "if": { + "required": [ + "allow_external" + ], + "properties": { + "allow_external": { + "const": true + } + } + }, + "then": { + "required": [ + "allowed_external_providers" + ], + "properties": { + "allowed_external_providers": { + "minItems": 1 + } + } + }, + "else": { + "properties": { + "allowed_external_providers": { + "maxItems": 0 + } + } + } + } + ] + }, + "execution": { + "type": "object", + "additionalProperties": false, + "properties": { + "strategy": { + "const": "whole_suite", + "default": "whole_suite" + }, + "max_seconds": { + "type": "integer", + "minimum": 10, + "maximum": 900, + "default": 900 + }, + "max_cost_usd": { + "type": "number", + "exclusiveMinimum": 0, + "maximum": 5, + "default": 5 + } + } + } + }, + "description": "Additional server checks enforce unique aliases, fixed campaign/suite ownership, credential and data-scope permissions, 1 MiB request bytes, capacity, and idempotency. JSON Schema validation alone does not authorize or preflight a job." +} diff --git a/docs/contracts/suite_planning_result.schema.json b/docs/contracts/suite_planning_result.schema.json new file mode 100644 index 00000000..3b48d3e7 --- /dev/null +++ b/docs/contracts/suite_planning_result.schema.json @@ -0,0 +1,42 @@ +{ + "$schema": "https://json-schema.org/draft/2020-12/schema", + "type": "object", + "additionalProperties": false, + "required": [ + "groups" + ], + "properties": { + "groups": { + "type": "array", + "minItems": 1, + "items": { + "type": "object", + "additionalProperties": false, + "required": [ + "name", + "description", + "members" + ], + "properties": { + "name": { + "type": "string", + "minLength": 1, + "maxLength": 80 + }, + "description": { + "type": "string", + "minLength": 1, + "maxLength": 240 + }, + "members": { + "type": "array", + "minItems": 1, + "items": { + "type": "string" + } + } + } + } + } + } +} diff --git a/docs/evidence/hermes_suite_acceptance_20260929.json b/docs/evidence/hermes_suite_acceptance_20260929.json new file mode 100644 index 00000000..66046de0 --- /dev/null +++ b/docs/evidence/hermes_suite_acceptance_20260929.json @@ -0,0 +1,301 @@ +{ + "configuration_revision": "suite-v6-20260929", + "deployed_commit": "299b9ed16d79aeefb3dc5590a19dc07dc54167e7", + "date": "2026-09-29", + "client_host": "titan-jh", + "client_lan_ip": "192.168.22.8", + "destination": "192.168.22.50:443", + "tls_hostname": "worker.bstein.dev", + "proxy_bypass": true, + "laptop_test_performed": false, + "synthetic_only": true, + "results": [ + { + "size": 14, + "http_status": 200, + "request_bytes": 9847, + "client_wall_seconds": 16.947, + "exact_alias_coverage": true, + "server_wall_seconds": 16.706, + "provider": "claude", + "model": "claude-opus-4-8", + "context_limit": 1000000, + "output_limit": 64000, + "input_tokens": 4185, + "output_tokens": 1250, + "cli_turns": 2, + "cost_usd_estimate": 0.052175, + "singletons": 4, + "quality": { + "coverage": true, + "families": 9, + "pair_precision": 1.0, + "pair_recall": 1.0, + "false_merge_pairs": 0, + "missed_merge_pairs": 0 + }, + "description_review": "No observed cross-family objective attribution; manual review against synthetic source patterns.", + "compaction": false, + "truncation": false, + "attempted_destinations": [ + "switchyard:atlas/planning/claude", + "claude:claude-opus-4-8" + ], + "job_id": "efa5151c8dfa4f35a52dc83244d64762", + "provider_job_attempts": 1, + "automatic_retries_observed": 0, + "rate_limit_events_observed": 0 + }, + { + "size": 75, + "http_status": 200, + "request_bytes": 55984, + "client_wall_seconds": 25.755, + "exact_alias_coverage": true, + "server_wall_seconds": 24.922, + "provider": "claude", + "model": "claude-opus-4-8", + "context_limit": 1000000, + "output_limit": 64000, + "input_tokens": 19145, + "output_tokens": 2336, + "cli_turns": 2, + "cost_usd_estimate": 0.15412499999999998, + "singletons": 3, + "quality": { + "coverage": true, + "families": 9, + "pair_precision": 1.0, + "pair_recall": 1.0, + "false_merge_pairs": 0, + "missed_merge_pairs": 0 + }, + "description_review": "No observed cross-family objective attribution; manual review against synthetic source patterns.", + "compaction": false, + "truncation": false, + "attempted_destinations": [ + "switchyard:atlas/planning/claude", + "claude:claude-opus-4-8" + ], + "job_id": "f0e076ffe90643779593d778920cffc7", + "provider_job_attempts": 1, + "automatic_retries_observed": 0, + "rate_limit_events_observed": 0 + }, + { + "size": 363, + "http_status": 200, + "request_bytes": 273761, + "client_wall_seconds": 80.071, + "exact_alias_coverage": true, + "server_wall_seconds": 78.828, + "provider": "claude", + "model": "claude-opus-4-8", + "context_limit": 1000000, + "output_limit": 64000, + "input_tokens": 89753, + "output_tokens": 8568, + "cli_turns": 2, + "cost_usd_estimate": 0.6629649999999999, + "singletons": 3, + "quality": { + "coverage": true, + "families": 9, + "pair_precision": 1.0, + "pair_recall": 1.0, + "false_merge_pairs": 0, + "missed_merge_pairs": 0 + }, + "description_review": "No observed cross-family objective attribution; manual review against synthetic source patterns.", + "compaction": false, + "truncation": false, + "attempted_destinations": [ + "switchyard:atlas/planning/claude", + "claude:claude-opus-4-8" + ], + "job_id": "682e87d11aa344c3b4bd33bcab7ea431", + "provider_job_attempts": 1, + "automatic_retries_observed": 0, + "rate_limit_events_observed": 0 + } + ], + "field_lengths": { + "description": { + "min_bytes": 48, + "median_bytes": 234, + "max_bytes": 245 + }, + "preconditions": { + "min_bytes": 67, + "median_bytes": 136, + "max_bytes": 143 + }, + "case_type": { + "min_bytes": 7, + "median_bytes": 8, + "max_bytes": 15 + }, + "success_criteria": { + "min_bytes": 73, + "median_bytes": 132, + "max_bytes": 135 + } + }, + "whole_input_mock_transport": [ + { + "cases": 14, + "requested": "claude-opus-4-8[1m]", + "captured": [ + { + "path": "/v1/messages?beta=true", + "model": "claude-opus-4-8", + "max_tokens": 64000, + "bytes": 12832, + "complete_source": true + }, + { + "path": "/v1/messages?beta=true", + "model": "claude-opus-4-8", + "max_tokens": 64000, + "bytes": 12818, + "complete_source": true + } + ], + "returncode": 0 + }, + { + "cases": 75, + "requested": "claude-opus-4-8[1m]", + "captured": [ + { + "path": "/v1/messages?beta=true", + "model": "claude-opus-4-8", + "max_tokens": 64000, + "bytes": 60921, + "complete_source": true + }, + { + "path": "/v1/messages?beta=true", + "model": "claude-opus-4-8", + "max_tokens": 64000, + "bytes": 60907, + "complete_source": true + } + ], + "returncode": 0 + }, + { + "cases": 363, + "requested": "claude-opus-4-8[1m]", + "captured": [ + { + "path": "/v1/messages?beta=true", + "model": "claude-opus-4-8", + "max_tokens": 64000, + "bytes": 287914, + "complete_source": true + }, + { + "path": "/v1/messages?beta=true", + "model": "claude-opus-4-8", + "max_tokens": 64000, + "bytes": 287900, + "complete_source": true + } + ], + "returncode": 0 + } + ], + "policy_checks": { + "synthetic_scope": { + "configuration_revision": "suite-v6-20260929", + "allowed_external_providers": [ + "claude", + "codex" + ], + "generalized_external_providers": [], + "external_scope": "exact_synthetic_fixtures" + }, + "local_scope": { + "configuration_revision": "suite-v6-20260929", + "allowed_external_providers": [], + "generalized_external_providers": [], + "external_scope": "exact_synthetic_fixtures" + }, + "operational_scope": { + "configuration_revision": "suite-v6-20260929", + "allowed_external_providers": [ + "claude" + ], + "generalized_external_providers": [ + "claude" + ], + "external_scope": "generalized_claude_and_exact_synthetic_fixtures" + }, + "synthetic_modified_synthetic_preflight": { + "status": 403, + "error": "external_data_not_approved", + "provider": null + }, + "local_modified_synthetic_preflight": { + "status": 403, + "error": "provider_forbidden", + "provider": null + }, + "operational_modified_synthetic_preflight": { + "status": 200, + "error": null, + "provider": "claude" + }, + "operational_codex": { + "status": 403, + "error": "provider_forbidden" + }, + "operational_claude_codex": { + "status": 403, + "error": "provider_forbidden" + }, + "operational_unknown": { + "status": 400, + "error": "unknown_provider" + }, + "local_only_True": { + "status": 422, + "error": { + "code": "capacity_or_unsupported_backend", + "details": { + "candidates": { + "local": "capacity" + }, + "input_bytes": 11069, + "output_estimate": 2718 + } + } + }, + "local_only_False": { + "status": 422, + "error": { + "code": "capacity_or_unsupported_backend", + "details": { + "candidates": { + "local": "capacity" + }, + "input_bytes": 11069, + "output_estimate": 2718 + } + } + } + }, + "oversize_curl": { + "http_status": 413, + "expect_continue": true + }, + "unknowns": [ + "Provider training/data-sharing/retention settings", + "Physical provider serving hardware and region", + "Remaining account quota", + "Accurate tokenizer and full 1M-context benchmark", + "Real-case grouping quality", + "Model internal attention to every input field" + ] +} diff --git a/docs/evidence/hermes_suite_local_success.json b/docs/evidence/hermes_suite_local_success.json new file mode 100644 index 00000000..0fceba72 --- /dev/null +++ b/docs/evidence/hermes_suite_local_success.json @@ -0,0 +1,86 @@ +{ + "attempted_destinations": [ + "switchyard:atlas/planning/local", + "local:qwen2.5:14b-instruct-q4_0" + ], + "compaction": false, + "configuration_revision": "suite-v6-20260929", + "created_at": 1790689284.778549, + "job_id": "c5df215f12374d5e91680fce28eadfe2", + "model": "qwen2.5:14b-instruct-q4_0", + "provenance": { + "backend_api": "/api/generate", + "concurrency": 1, + "context_tokens": 8192, + "fallback": null, + "max_output_tokens": 2048, + "model": "qwen2.5:14b-instruct-q4_0", + "model_digest": "5449194ff8035ccb13a6409a5814de6c8f9c39f555f429e383ae0fb7137001bd", + "options": { + "num_ctx": 8192, + "num_predict": 2048, + "seed": 0, + "temperature": 0, + "top_k": 40, + "top_p": 1 + }, + "placement": "titan-24/RTX-3080-10GB", + "protocol_version": 1, + "runtime": "0.13.5", + "serving_configuration": { + "flash_attention": true, + "kv_cache_type": "q8_0", + "max_queue": 1, + "parallel_requests": 1 + }, + "timeout_seconds": 1200, + "wall_seconds": 3.48 + }, + "result": { + "groups": [ + { + "description": "Verifies parsing of configuration texts, both valid and invalid.", + "members": [ + "CASE-1", + "CASE-2" + ], + "name": "ConfigurationParsing" + } + ] + }, + "result_retention_seconds": 3600, + "routing": { + "allow_external": false, + "allowed_external_providers": [] + }, + "selection": { + "backend": "ollama-model-gate", + "case_count": 2, + "configuration_revision": "suite-v6-20260929", + "context": 8192, + "enabled": true, + "input_bytes": 1972, + "input_count_method": "UTF-8 byte upper bound plus reserved harness overhead; not a tokenizer", + "input_token_bound": 2996, + "input_token_count": null, + "model": "qwen2.5:14b-instruct-q4_0", + "output": 2048, + "output_reservation_tokens": 1260, + "output_reservation_verified": false, + "overhead": 1024, + "provider": "local", + "reasoning": "none", + "source_sha256": "f59c6b7985060fec965800330d751896fe57300cc5dc22b45844e00dbf840e7b" + }, + "status": "completed", + "truncation": false, + "usage": { + "eval_count": 58, + "eval_duration": 2149823584, + "load_duration": 110578986, + "prompt_eval_count": 307, + "prompt_eval_duration": 95273264, + "total_duration": 3451656008 + }, + "wall_seconds": 3.535 +} diff --git a/docs/evidence/hermes_suite_results_synthetic.json b/docs/evidence/hermes_suite_results_synthetic.json new file mode 100644 index 00000000..61e145a5 --- /dev/null +++ b/docs/evidence/hermes_suite_results_synthetic.json @@ -0,0 +1,883 @@ +[ + { + "size": 14, + "http_status": 200, + "request_bytes": 9847, + "client_wall_seconds": 16.947, + "exact_alias_coverage": true, + "job": { + "attempted_destinations": [ + "switchyard:atlas/planning/claude", + "claude:claude-opus-4-8" + ], + "compaction": false, + "compaction_signal": "CLI events and disabled compaction", + "configuration_revision": "suite-v6-20260929", + "cost_usd_estimate": 0.052175, + "created_at": 1790689083.5601273, + "duration_api_ms": 15944, + "job_id": "efa5151c8dfa4f35a52dc83244d64762", + "model": "claude-opus-4-8", + "model_usage": { + "claude-opus-4-8[1m]": { + "cacheCreationInputTokens": 0, + "cacheReadInputTokens": 0, + "canonicalModel": "claude-opus-4-8", + "contextWindow": 1000000, + "costUSD": 0.052175, + "inputTokens": 4185, + "maxOutputTokens": 64000, + "outputTokens": 1250, + "provider": "firstParty", + "webSearchRequests": 0 + } + }, + "result": { + "groups": [ + { + "description": "Cycle thermal chamber and measure enclosure expansion with a calibrated dimensional gauge. Unique physical fixture; procedure details unknown, so no reuse with other families.", + "members": [ + "CASE-0001" + ], + "name": "Thermal Chamber Expansion Measurement" + }, + { + "description": "Drive controllable physical watchdog pulses and capture reset-line timing with a digital capture fixture, checking deadlines across nominal/boundary/malformed inputs. Shared pulse-source and recorder machinery.", + "members": [ + "CASE-0002", + "CASE-0013" + ], + "name": "Watchdog Pulse Reset-Line Timing" + }, + { + "description": "Parse offline watchdog static-analysis reports, count findings by severity, and compare rule IDs against a policy table. Shared report-parser and severity-policy fixtures; analysis method.", + "members": [ + "CASE-0003", + "CASE-0009" + ], + "name": "Static-Analysis Findings Parser" + }, + { + "description": "Load configuration text fixtures through the parser and assert accepted values or diagnostic positions. Shared config-text builders and parser-call harness.", + "members": [ + "CASE-0004", + "CASE-0010" + ], + "name": "Configuration Parser Validation" + }, + { + "description": "Drive concurrent producer/consumer tasks against a bounded queue fixture and assert depth, rejected writes, ordering, and recovery. Shared task drivers and sequence assertions.", + "members": [ + "CASE-0005", + "CASE-0011" + ], + "name": "Bounded Message Queue Producer/Consumer" + }, + { + "description": "Submit role-scoped operations to the authorization decision function using identity fixtures and in-memory policy store; compare allow/deny decisions and audit fields. Shared policy machinery.", + "members": [ + "CASE-0006", + "CASE-0012" + ], + "name": "Authorization Decision Matrix" + }, + { + "description": "Build encoded watchdog status frames via byte-array builder and check decoded fields, checksum status, and rejection codes. Distinct decoder-call fixture; single case.", + "members": [ + "CASE-0007" + ], + "name": "Watchdog Status Frame Decoder" + }, + { + "description": "Record acoustic output in an anechoic fixture and verify spectral peak magnitude against a frequency-dependent threshold. Unique measurement setup; no shared machinery.", + "members": [ + "CASE-0008" + ], + "name": "Anechoic Acoustic Output Measurement" + }, + { + "description": "Rebuild the same source twice in clean build containers and compare artifact digests after removing allowed timestamp metadata. Unique build-container machinery.", + "members": [ + "CASE-0014" + ], + "name": "Reproducible Build Digest Comparison" + } + ] + }, + "result_retention_seconds": 3600, + "routing": { + "allow_external": true, + "allowed_external_providers": [ + "claude" + ] + }, + "selection": { + "backend": "claude-code-2.1.226", + "case_count": 14, + "cli_model": "claude-opus-4-8[1m]", + "configuration_revision": "suite-v6-20260929", + "context": 1000000, + "enabled": true, + "input_bytes": 11035, + "input_count_method": "UTF-8 byte upper bound plus reserved harness overhead; not a tokenizer", + "input_token_bound": 19227, + "input_token_count": null, + "model": "claude-opus-4-8", + "output": 64000, + "output_reservation_tokens": 10910, + "output_reservation_verified": false, + "overhead": 8192, + "provider": "claude", + "reasoning": "medium", + "source_sha256": "99fd35cb089e57285670f6def4339a86754316e67ce3e53ec7efa95968534ca9" + }, + "status": "completed", + "temporary_files_deleted": true, + "truncation": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "inference_geo": "not_available", + "input_tokens": 4185, + "iterations": [], + "output_tokens": 1250, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + }, + "service_tier": "standard", + "speed": "standard" + }, + "wall_seconds": 16.706 + } + }, + { + "size": 75, + "http_status": 200, + "request_bytes": 55984, + "client_wall_seconds": 25.755, + "exact_alias_coverage": true, + "job": { + "attempted_destinations": [ + "switchyard:atlas/planning/claude", + "claude:claude-opus-4-8" + ], + "compaction": false, + "compaction_signal": "CLI events and disabled compaction", + "configuration_revision": "suite-v6-20260929", + "cost_usd_estimate": 0.15412499999999998, + "created_at": 1790689100.8911846, + "duration_api_ms": 24235, + "job_id": "f0e076ffe90643779593d778920cffc7", + "model": "claude-opus-4-8", + "model_usage": { + "claude-opus-4-8[1m]": { + "cacheCreationInputTokens": 0, + "cacheReadInputTokens": 0, + "canonicalModel": "claude-opus-4-8", + "contextWindow": 1000000, + "costUSD": 0.15412499999999998, + "inputTokens": 19145, + "maxOutputTokens": 64000, + "outputTokens": 2336, + "provider": "firstParty", + "webSearchRequests": 0 + } + }, + "result": { + "groups": [ + { + "description": "Shared byte-array builder and decoder-call fixture; assert decoded fields, checksum status, and rejection code. Same machinery across nominal, boundary, and fault-injection variants; parameterize inputs and expected results.", + "members": [ + "CASE-0007", + "CASE-0013", + "CASE-0019", + "CASE-0025", + "CASE-0031", + "CASE-0037", + "CASE-0043", + "CASE-0049", + "CASE-0055", + "CASE-0061", + "CASE-0067", + "CASE-0073" + ], + "name": "Watchdog Status Frame Decoder" + }, + { + "description": "Shared controllable pulse source and reset-line recorder with digital capture; measure reset-line timing against a deadline. One driver/measurement harness parameterized over input variations.", + "members": [ + "CASE-0002", + "CASE-0008", + "CASE-0014", + "CASE-0020", + "CASE-0026", + "CASE-0032", + "CASE-0044", + "CASE-0050", + "CASE-0056", + "CASE-0062", + "CASE-0068", + "CASE-0074" + ], + "name": "Physical Watchdog Pulse Timing" + }, + { + "description": "Analysis-method family sharing a static-analysis report parser and severity policy fixture; count findings by severity and compare rule IDs to a policy table. Firmware not executed.", + "members": [ + "CASE-0003", + "CASE-0009", + "CASE-0015", + "CASE-0021", + "CASE-0027", + "CASE-0033", + "CASE-0039", + "CASE-0045", + "CASE-0051", + "CASE-0057", + "CASE-0063", + "CASE-0069" + ], + "name": "Static-Analysis Report Parser" + }, + { + "description": "Shared configuration text fixtures and parser call without network stack; assert accepted values or diagnostic positions from returned parser objects. Parameterize input documents.", + "members": [ + "CASE-0004", + "CASE-0010", + "CASE-0016", + "CASE-0022", + "CASE-0028", + "CASE-0034", + "CASE-0040", + "CASE-0046", + "CASE-0052", + "CASE-0058", + "CASE-0064", + "CASE-0070" + ], + "name": "Configuration Document Parser" + }, + { + "description": "Shared concurrent task drivers, bounded queue fixture, and sequence-number assertions; observe queue depth, rejected writes, ordering, and recovery. Same harness parameterized over load variants.", + "members": [ + "CASE-0005", + "CASE-0011", + "CASE-0017", + "CASE-0023", + "CASE-0029", + "CASE-0035", + "CASE-0041", + "CASE-0047", + "CASE-0053", + "CASE-0059", + "CASE-0065", + "CASE-0071" + ], + "name": "Bounded Message Queue Producer/Consumer" + }, + { + "description": "Shared identity fixtures and in-memory policy store; compare allow/deny decisions and audit-event fields to a permissions matrix. No interactive login. Parameterize role-scoped operations.", + "members": [ + "CASE-0006", + "CASE-0012", + "CASE-0018", + "CASE-0024", + "CASE-0030", + "CASE-0036", + "CASE-0042", + "CASE-0048", + "CASE-0054", + "CASE-0060", + "CASE-0066", + "CASE-0072" + ], + "name": "Authorization Decision Function" + }, + { + "description": "Standalone thermal-cycle test measuring enclosure expansion with a calibrated gauge against dimensional tolerance. Unique fixture; procedure details unknown, so no reuse with other families.", + "members": [ + "CASE-0001" + ], + "name": "Thermal Chamber Expansion" + }, + { + "description": "Standalone anechoic-fixture recording comparing spectral peak magnitude to a frequency-dependent threshold. Unique fixture; procedure details unknown, so kept separate.", + "members": [ + "CASE-0038" + ], + "name": "Acoustic Output Measurement" + }, + { + "description": "Standalone analysis rebuilding source twice in clean containers and comparing artifact digests after removing allowed timestamp metadata. Unique machinery; kept separate.", + "members": [ + "CASE-0075" + ], + "name": "Reproducible Build Digest" + } + ] + }, + "result_retention_seconds": 3600, + "routing": { + "allow_external": true, + "allowed_external_providers": [ + "claude" + ] + }, + "selection": { + "backend": "claude-code-2.1.226", + "case_count": 75, + "cli_model": "claude-opus-4-8[1m]", + "configuration_revision": "suite-v6-20260929", + "context": 1000000, + "enabled": true, + "input_bytes": 57172, + "input_count_method": "UTF-8 byte upper bound plus reserved harness overhead; not a tokenizer", + "input_token_bound": 65364, + "input_token_count": null, + "model": "claude-opus-4-8", + "output": 64000, + "output_reservation_tokens": 18291, + "output_reservation_verified": false, + "overhead": 8192, + "provider": "claude", + "reasoning": "medium", + "source_sha256": "3cca99c228210deee67c4f7a4674bad55283440c5662f8add1c5182409447d32" + }, + "status": "completed", + "temporary_files_deleted": true, + "truncation": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "inference_geo": "not_available", + "input_tokens": 19145, + "iterations": [], + "output_tokens": 2336, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + }, + "service_tier": "standard", + "speed": "standard" + }, + "wall_seconds": 24.922 + } + }, + { + "size": 363, + "http_status": 200, + "request_bytes": 273761, + "client_wall_seconds": 80.071, + "exact_alias_coverage": true, + "job": { + "attempted_destinations": [ + "switchyard:atlas/planning/claude", + "claude:claude-opus-4-8" + ], + "compaction": false, + "compaction_signal": "CLI events and disabled compaction", + "configuration_revision": "suite-v6-20260929", + "cost_usd_estimate": 0.6629649999999999, + "created_at": 1790689126.669432, + "duration_api_ms": 78221, + "job_id": "682e87d11aa344c3b4bd33bcab7ea431", + "model": "claude-opus-4-8", + "model_usage": { + "claude-opus-4-8[1m]": { + "cacheCreationInputTokens": 0, + "cacheReadInputTokens": 0, + "canonicalModel": "claude-opus-4-8", + "contextWindow": 1000000, + "costUSD": 0.6629649999999999, + "inputTokens": 89753, + "maxOutputTokens": 64000, + "outputTokens": 8568, + "provider": "firstParty", + "webSearchRequests": 0 + } + }, + "result": { + "groups": [ + { + "description": "Byte-array builder feeds encoded status frames to the message decoder; assert decoded fields, checksum status, and rejection code. Case-type variants are parameters over the same fixture.", + "members": [ + "CASE-0007", + "CASE-0013", + "CASE-0019", + "CASE-0025", + "CASE-0031", + "CASE-0037", + "CASE-0043", + "CASE-0049", + "CASE-0055", + "CASE-0061", + "CASE-0067", + "CASE-0073", + "CASE-0079", + "CASE-0085", + "CASE-0091", + "CASE-0097", + "CASE-0103", + "CASE-0109", + "CASE-0115", + "CASE-0121", + "CASE-0127", + "CASE-0133", + "CASE-0139", + "CASE-0145", + "CASE-0151", + "CASE-0157", + "CASE-0163", + "CASE-0169", + "CASE-0175", + "CASE-0181", + "CASE-0187", + "CASE-0193", + "CASE-0199", + "CASE-0205", + "CASE-0211", + "CASE-0217", + "CASE-0223", + "CASE-0229", + "CASE-0235", + "CASE-0241", + "CASE-0247", + "CASE-0253", + "CASE-0259", + "CASE-0265", + "CASE-0271", + "CASE-0277", + "CASE-0283", + "CASE-0289", + "CASE-0295", + "CASE-0301", + "CASE-0307", + "CASE-0313", + "CASE-0319", + "CASE-0325", + "CASE-0331", + "CASE-0337", + "CASE-0343", + "CASE-0349", + "CASE-0355", + "CASE-0361" + ], + "name": "Watchdog status frame decoder harness" + }, + { + "description": "Controllable pulse source stops/resumes watchdog pulses; reset-line recorder and digital capture fixture measure timing against a deadline. Shared stimulus and timing assertions across variants.", + "members": [ + "CASE-0002", + "CASE-0008", + "CASE-0014", + "CASE-0020", + "CASE-0026", + "CASE-0032", + "CASE-0038", + "CASE-0044", + "CASE-0050", + "CASE-0056", + "CASE-0062", + "CASE-0068", + "CASE-0074", + "CASE-0080", + "CASE-0086", + "CASE-0092", + "CASE-0098", + "CASE-0104", + "CASE-0110", + "CASE-0116", + "CASE-0122", + "CASE-0128", + "CASE-0134", + "CASE-0140", + "CASE-0146", + "CASE-0152", + "CASE-0158", + "CASE-0164", + "CASE-0170", + "CASE-0176", + "CASE-0188", + "CASE-0194", + "CASE-0200", + "CASE-0206", + "CASE-0212", + "CASE-0218", + "CASE-0224", + "CASE-0230", + "CASE-0236", + "CASE-0242", + "CASE-0248", + "CASE-0254", + "CASE-0260", + "CASE-0266", + "CASE-0272", + "CASE-0278", + "CASE-0284", + "CASE-0290", + "CASE-0296", + "CASE-0302", + "CASE-0308", + "CASE-0314", + "CASE-0320", + "CASE-0326", + "CASE-0332", + "CASE-0338", + "CASE-0344", + "CASE-0350", + "CASE-0356", + "CASE-0362" + ], + "name": "Watchdog reset-line timing rig" + }, + { + "description": "Loads offline static-analysis reports and a severity policy fixture; counts findings by severity and checks each rule ID against the policy table. No firmware execution; shared parser and policy comparison.", + "members": [ + "CASE-0003", + "CASE-0009", + "CASE-0015", + "CASE-0021", + "CASE-0027", + "CASE-0033", + "CASE-0039", + "CASE-0045", + "CASE-0051", + "CASE-0057", + "CASE-0063", + "CASE-0069", + "CASE-0075", + "CASE-0081", + "CASE-0087", + "CASE-0093", + "CASE-0099", + "CASE-0105", + "CASE-0111", + "CASE-0117", + "CASE-0123", + "CASE-0129", + "CASE-0135", + "CASE-0141", + "CASE-0147", + "CASE-0153", + "CASE-0159", + "CASE-0165", + "CASE-0171", + "CASE-0177", + "CASE-0183", + "CASE-0189", + "CASE-0195", + "CASE-0201", + "CASE-0207", + "CASE-0213", + "CASE-0219", + "CASE-0225", + "CASE-0231", + "CASE-0237", + "CASE-0243", + "CASE-0249", + "CASE-0255", + "CASE-0261", + "CASE-0267", + "CASE-0273", + "CASE-0279", + "CASE-0285", + "CASE-0291", + "CASE-0297", + "CASE-0303", + "CASE-0309", + "CASE-0315", + "CASE-0321", + "CASE-0327", + "CASE-0333", + "CASE-0339", + "CASE-0345", + "CASE-0351", + "CASE-0357" + ], + "name": "Static-analysis report parser" + }, + { + "description": "Builds configuration text fixtures and calls the parser without the network stack; asserts accepted values or diagnostic positions from returned parser objects. Shared text builders and assertions.", + "members": [ + "CASE-0004", + "CASE-0010", + "CASE-0016", + "CASE-0022", + "CASE-0028", + "CASE-0034", + "CASE-0040", + "CASE-0046", + "CASE-0052", + "CASE-0058", + "CASE-0064", + "CASE-0070", + "CASE-0076", + "CASE-0082", + "CASE-0088", + "CASE-0094", + "CASE-0100", + "CASE-0106", + "CASE-0112", + "CASE-0118", + "CASE-0124", + "CASE-0130", + "CASE-0136", + "CASE-0142", + "CASE-0148", + "CASE-0154", + "CASE-0160", + "CASE-0166", + "CASE-0172", + "CASE-0178", + "CASE-0184", + "CASE-0190", + "CASE-0196", + "CASE-0202", + "CASE-0208", + "CASE-0214", + "CASE-0220", + "CASE-0226", + "CASE-0232", + "CASE-0238", + "CASE-0244", + "CASE-0250", + "CASE-0256", + "CASE-0262", + "CASE-0268", + "CASE-0274", + "CASE-0280", + "CASE-0286", + "CASE-0292", + "CASE-0298", + "CASE-0304", + "CASE-0310", + "CASE-0316", + "CASE-0322", + "CASE-0328", + "CASE-0334", + "CASE-0340", + "CASE-0346", + "CASE-0352", + "CASE-0358" + ], + "name": "Configuration parser harness" + }, + { + "description": "Concurrent producer/consumer task drivers push a bounded queue to its limit; assert depth, rejected writes, delivery ordering via sequence numbers, and drain recovery. Shared drivers and assertions.", + "members": [ + "CASE-0005", + "CASE-0011", + "CASE-0017", + "CASE-0023", + "CASE-0029", + "CASE-0035", + "CASE-0041", + "CASE-0047", + "CASE-0053", + "CASE-0059", + "CASE-0065", + "CASE-0071", + "CASE-0077", + "CASE-0083", + "CASE-0089", + "CASE-0095", + "CASE-0101", + "CASE-0107", + "CASE-0113", + "CASE-0119", + "CASE-0125", + "CASE-0131", + "CASE-0137", + "CASE-0143", + "CASE-0149", + "CASE-0155", + "CASE-0161", + "CASE-0167", + "CASE-0173", + "CASE-0179", + "CASE-0185", + "CASE-0191", + "CASE-0197", + "CASE-0203", + "CASE-0209", + "CASE-0215", + "CASE-0221", + "CASE-0227", + "CASE-0233", + "CASE-0239", + "CASE-0245", + "CASE-0251", + "CASE-0257", + "CASE-0263", + "CASE-0269", + "CASE-0275", + "CASE-0281", + "CASE-0287", + "CASE-0293", + "CASE-0299", + "CASE-0305", + "CASE-0311", + "CASE-0317", + "CASE-0323", + "CASE-0329", + "CASE-0335", + "CASE-0341", + "CASE-0347", + "CASE-0353", + "CASE-0359" + ], + "name": "Bounded message queue concurrency driver" + }, + { + "description": "Identity fixtures and in-memory policy store submit role-scoped operations; compare allow/deny decisions and audit-event fields against the permissions matrix. Shared fixtures and assertions.", + "members": [ + "CASE-0006", + "CASE-0012", + "CASE-0018", + "CASE-0024", + "CASE-0030", + "CASE-0036", + "CASE-0042", + "CASE-0048", + "CASE-0054", + "CASE-0060", + "CASE-0066", + "CASE-0072", + "CASE-0078", + "CASE-0084", + "CASE-0090", + "CASE-0096", + "CASE-0102", + "CASE-0108", + "CASE-0114", + "CASE-0120", + "CASE-0126", + "CASE-0132", + "CASE-0138", + "CASE-0144", + "CASE-0150", + "CASE-0156", + "CASE-0162", + "CASE-0168", + "CASE-0174", + "CASE-0180", + "CASE-0186", + "CASE-0192", + "CASE-0198", + "CASE-0204", + "CASE-0210", + "CASE-0216", + "CASE-0222", + "CASE-0228", + "CASE-0234", + "CASE-0240", + "CASE-0246", + "CASE-0252", + "CASE-0258", + "CASE-0264", + "CASE-0270", + "CASE-0276", + "CASE-0282", + "CASE-0288", + "CASE-0294", + "CASE-0300", + "CASE-0306", + "CASE-0312", + "CASE-0318", + "CASE-0324", + "CASE-0330", + "CASE-0336", + "CASE-0342", + "CASE-0348", + "CASE-0354", + "CASE-0360" + ], + "name": "Authorization decision function harness" + }, + { + "description": "Cycles a thermal chamber and measures enclosure expansion with a calibrated dimensional gauge. Unique physical fixture; procedure details unknown, so shares no machinery with other families.", + "members": [ + "CASE-0001" + ], + "name": "Thermal chamber expansion measurement" + }, + { + "description": "Records acoustic output in an anechoic fixture and checks spectral peak magnitude against a frequency-dependent threshold. Unique acoustic equipment; procedure details unknown.", + "members": [ + "CASE-0182" + ], + "name": "Anechoic acoustic emission measurement" + }, + { + "description": "Rebuilds identical source twice in clean containers and compares artifact digests after stripping allowed timestamp metadata. Distinct build-analysis machinery from all runtime harnesses.", + "members": [ + "CASE-0363" + ], + "name": "Reproducible build digest comparison" + } + ] + }, + "result_retention_seconds": 3600, + "routing": { + "allow_external": true, + "allowed_external_providers": [ + "claude" + ] + }, + "selection": { + "backend": "claude-code-2.1.226", + "case_count": 363, + "cli_model": "claude-opus-4-8[1m]", + "configuration_revision": "suite-v6-20260929", + "context": 1000000, + "enabled": true, + "input_bytes": 274949, + "input_count_method": "UTF-8 byte upper bound plus reserved harness overhead; not a tokenizer", + "input_token_bound": 283141, + "input_token_count": null, + "model": "claude-opus-4-8", + "output": 64000, + "output_reservation_tokens": 53139, + "output_reservation_verified": false, + "overhead": 8192, + "provider": "claude", + "reasoning": "medium", + "source_sha256": "311c96701a3be823afd2bf2c2d682da56e5f383108001f498a8a672ab41a9037" + }, + "status": "completed", + "temporary_files_deleted": true, + "truncation": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "inference_geo": "not_available", + "input_tokens": 89753, + "iterations": [], + "output_tokens": 8568, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + }, + "service_tier": "standard", + "speed": "standard" + }, + "wall_seconds": 78.828 + } + } +] diff --git a/docs/evidence/hermes_suite_success_14.json b/docs/evidence/hermes_suite_success_14.json new file mode 100644 index 00000000..b3e089f3 --- /dev/null +++ b/docs/evidence/hermes_suite_success_14.json @@ -0,0 +1,150 @@ +{ + "attempted_destinations": [ + "switchyard:atlas/planning/claude", + "claude:claude-opus-4-8" + ], + "compaction": false, + "compaction_signal": "CLI events and disabled compaction", + "configuration_revision": "suite-v6-20260929", + "cost_usd_estimate": 0.052175, + "created_at": 1790689083.5601273, + "duration_api_ms": 15944, + "job_id": "efa5151c8dfa4f35a52dc83244d64762", + "model": "claude-opus-4-8", + "model_usage": { + "claude-opus-4-8[1m]": { + "cacheCreationInputTokens": 0, + "cacheReadInputTokens": 0, + "canonicalModel": "claude-opus-4-8", + "contextWindow": 1000000, + "costUSD": 0.052175, + "inputTokens": 4185, + "maxOutputTokens": 64000, + "outputTokens": 1250, + "provider": "firstParty", + "webSearchRequests": 0 + } + }, + "result": { + "groups": [ + { + "description": "Cycle thermal chamber and measure enclosure expansion with a calibrated dimensional gauge. Unique physical fixture; procedure details unknown, so no reuse with other families.", + "members": [ + "CASE-0001" + ], + "name": "Thermal Chamber Expansion Measurement" + }, + { + "description": "Drive controllable physical watchdog pulses and capture reset-line timing with a digital capture fixture, checking deadlines across nominal/boundary/malformed inputs. Shared pulse-source and recorder machinery.", + "members": [ + "CASE-0002", + "CASE-0013" + ], + "name": "Watchdog Pulse Reset-Line Timing" + }, + { + "description": "Parse offline watchdog static-analysis reports, count findings by severity, and compare rule IDs against a policy table. Shared report-parser and severity-policy fixtures; analysis method.", + "members": [ + "CASE-0003", + "CASE-0009" + ], + "name": "Static-Analysis Findings Parser" + }, + { + "description": "Load configuration text fixtures through the parser and assert accepted values or diagnostic positions. Shared config-text builders and parser-call harness.", + "members": [ + "CASE-0004", + "CASE-0010" + ], + "name": "Configuration Parser Validation" + }, + { + "description": "Drive concurrent producer/consumer tasks against a bounded queue fixture and assert depth, rejected writes, ordering, and recovery. Shared task drivers and sequence assertions.", + "members": [ + "CASE-0005", + "CASE-0011" + ], + "name": "Bounded Message Queue Producer/Consumer" + }, + { + "description": "Submit role-scoped operations to the authorization decision function using identity fixtures and in-memory policy store; compare allow/deny decisions and audit fields. Shared policy machinery.", + "members": [ + "CASE-0006", + "CASE-0012" + ], + "name": "Authorization Decision Matrix" + }, + { + "description": "Build encoded watchdog status frames via byte-array builder and check decoded fields, checksum status, and rejection codes. Distinct decoder-call fixture; single case.", + "members": [ + "CASE-0007" + ], + "name": "Watchdog Status Frame Decoder" + }, + { + "description": "Record acoustic output in an anechoic fixture and verify spectral peak magnitude against a frequency-dependent threshold. Unique measurement setup; no shared machinery.", + "members": [ + "CASE-0008" + ], + "name": "Anechoic Acoustic Output Measurement" + }, + { + "description": "Rebuild the same source twice in clean build containers and compare artifact digests after removing allowed timestamp metadata. Unique build-container machinery.", + "members": [ + "CASE-0014" + ], + "name": "Reproducible Build Digest Comparison" + } + ] + }, + "result_retention_seconds": 3600, + "routing": { + "allow_external": true, + "allowed_external_providers": [ + "claude" + ] + }, + "selection": { + "backend": "claude-code-2.1.226", + "case_count": 14, + "cli_model": "claude-opus-4-8[1m]", + "configuration_revision": "suite-v6-20260929", + "context": 1000000, + "enabled": true, + "input_bytes": 11035, + "input_count_method": "UTF-8 byte upper bound plus reserved harness overhead; not a tokenizer", + "input_token_bound": 19227, + "input_token_count": null, + "model": "claude-opus-4-8", + "output": 64000, + "output_reservation_tokens": 10910, + "output_reservation_verified": false, + "overhead": 8192, + "provider": "claude", + "reasoning": "medium", + "source_sha256": "99fd35cb089e57285670f6def4339a86754316e67ce3e53ec7efa95968534ca9" + }, + "status": "completed", + "temporary_files_deleted": true, + "truncation": false, + "turns": 2, + "usage": { + "cache_creation": { + "ephemeral_1h_input_tokens": 0, + "ephemeral_5m_input_tokens": 0 + }, + "cache_creation_input_tokens": 0, + "cache_read_input_tokens": 0, + "inference_geo": "not_available", + "input_tokens": 4185, + "iterations": [], + "output_tokens": 1250, + "server_tool_use": { + "web_fetch_requests": 0, + "web_search_requests": 0 + }, + "service_tier": "standard", + "speed": "standard" + }, + "wall_seconds": 16.706 +} diff --git a/docs/hermes_suite_planning.md b/docs/hermes_suite_planning.md index 4ee474a2..f5a2a08d 100644 --- a/docs/hermes_suite_planning.md +++ b/docs/hermes_suite_planning.md @@ -4,6 +4,102 @@ This optional API groups one complete campaign/suite into implementation familie The existing `/local-model` API, GPU allocation, serving model, and limits are unchanged. Deployment uses Flux; the application and workbook stay on the laptop. +## Acceptance results, 2026-09-29 + +The final runs use `suite-v6-20260929`, deployed commit `299b9ed1`, including +case-type fields. These supersede the preliminary runs without that field. +All tests used synthetic content. HTTPS tests ran from `titan-jh`, source +`192.168.22.8`, directly on its LAN interface to `192.168.22.50:443`, with hostname +certificate verification and proxies disabled. The user's WSL route has not been +tested; run the health command below from the laptop before uploading anything. + +| Cases | Actual request bytes | CLI input/output tokens | Client wall seconds | Families / singletons | Exact coverage | Pattern match | +| --- | ---: | --- | ---: | --- | --- | --- | +| 14 | 9,847 | 4,185 / 1,250 | 16.947 | 9 / 4 | Yes | Exact | +| 75 | 55,984 | 19,145 / 2,336 | 25.755 | 9 / 3 | Yes | Exact | +| 363 | 273,761 | 89,753 / 8,568 | 80.071 | 9 / 3 | Yes | Exact | + +Request sizes are the UTF-8 compact JSON bytes actually submitted, excluding HTTP +headers. Token counts are CLI-reported aggregate usage across each fresh job's +two structured-output turns; they are not an exact standalone input-token count. +All three final jobs completed on their first attempt with no observed retries or +rate-limit errors. Two CLI turns are part of producing the structured result, not +automatic job retries. CLI cost estimates were USD 0.052175, 0.154125, and 0.662965; +these are estimates, not verified subscription charges. Remaining account quota +is not available through this endpoint and may be consumed by other account users. + +The actual terminal `modelUsage` reports `claude-opus-4-8[1m]`, with +`canonicalModel=claude-opus-4-8`, `provider=firstParty`, context 1,000,000 and maximum +output 64,000. The invocation requests `claude-opus-4-8[1m]`; these results also +independently verify the returned model metadata, rather than trusting the request +flag or the internal `atlas/planning/claude` route label. Provider physical hardware +and inference region are unknown. This verifies the tested suite sizes, not the +entire advertised million-token context. + +The 363-case fixture has descriptions of 48-245 bytes (median 234), preconditions +67-143 (median 136), success criteria 73-135 (median 132), and case types 7-15 bytes. +Most cases have multi-sentence fields plus operating conditions, verification +method, and a generic reference. Six machinery patterns are interleaved throughout +the input. Watchdog terminology occurs in three different machinery patterns. A +distant pair has identical complete case text but different aliases. Distinctive +thermal, acoustic, and build cases occur at the beginning, middle, and end. + +Against the independent synthetic implementation map, all sizes had pair precision +and recall 1.0, zero incorrect merge pairs, and zero missed merge pairs. Manual +review of all 27 family descriptions found no objectives attributed to another +family. Four singletons in the 14-case fixture and three in the larger fixtures +are expected. These are clear synthetic patterns, not evidence of quality on the +user's more ambiguous real suite. No real case quality is claimed. + +Evidence for complete input handling goes beyond IDs: + +- Unit checks compare every serialized case and field with the original fixture. +- The deployed native CLI was also run against a loopback mock provider, with the + exact production command and full 14/75/363 inputs. The captured requests contained + the complete source string, byte for byte, including every field. The largest + first provider request was 287,914 bytes. This transport check used no hosted model. +- Live jobs returned the expected distinct beginning/middle/end patterns and + reunited interleaved cases. Initialization allowed only `StructuredOutput`, with + no MCP, plugins, or file-reading tools. Each job used fresh HOME/config and stdin. +- No CLI compaction event or incomplete-output signal was observed. Compaction was + disabled, and the gateway performs no summarization or selective file reading. + +Those observations support complete transmission and the tested grouping behavior. +They cannot inspect provider internals or prove that a model attended equally to +every word. The response's false compaction/truncation values describe gateway/CLI +observations, not a provider-side attestation. An exact tokenizer is unavailable; +preflight's byte bound and output reservation are documented below. + +The local-only two-case parser example also completed in 3.535 seconds on the RTX +backend with exact coverage and no hosted destination. An earlier local result +incorrectly appended descriptions to aliases and was rejected; the new adapter +constrains the local output schema to supplied aliases. The existing local-model +endpoint itself is unchanged. + +Verification also covered authentication failures, permission escalation, unknown +providers, synthetic-versus-operational scope, local-only capacity rejection, +management-path rejection, cross-client ownership, idempotent replay, oversized +HTTP 413, and content-free routine worker logs. Mocked backend failures verified +that no local failure launches Claude. No shared backend was disrupted. The focused +unit suite passes 35 tests. Service renders, client dry-runs, and Flux diffs passed. +The installed kubectl rejects the repository's combined `--server-side --dry-run=client` +syntax; the supported client dry-run was used instead. + +Full evidence and successful normalized envelopes: + +- [Acceptance measurements and policy checks](evidence/hermes_suite_acceptance_20260929.json) +- [Actual complete 14-case Claude response](evidence/hermes_suite_success_14.json) +- [Actual complete two-case local response](evidence/hermes_suite_local_success.json) +- [All final synthetic family results](evidence/hermes_suite_results_synthetic.json) + +Ready: synthetic client testing, direct complete-suite execution at all three +tested sizes, and the separately approved Claude credential. Remaining client work: +WSL connectivity, explicit client-envelope adaptation, and a controlled real pilot +after reviewing this handoff. Codex capacity verification and provider/account +retention or training controls remain unresolved. The user's account-specific +approval is recorded; broader organizational approval or release classification +has not been established by this infrastructure work. + ## Connection and credentials ``` @@ -19,6 +115,7 @@ AUTHENTICATION_HEADER: Authorization: Bearer VAULT_PATH: kv/atlas/hermes/suite-planning-api LOCAL_ONLY_FIELD: token SYNTHETIC_EXTERNAL_FIELD: synthetic_token +APPROVED_OPERATIONAL_FIELD: operational_token INITIAL_CONCURRENCY: 1; busy submissions return 429; no waiting queue JOB_TIMEOUT: up to 900 seconds HTTP_TIMEOUT: client 45 seconds; submit/status do not wait for inference @@ -31,16 +128,34 @@ Neither token grants provider, Vault, Kubernetes, shell, or ClickUp access. The old local-model token remains at `kv/atlas/hermes/model-gate-lan-api`. Tokens are distinct; an old local-model token does not authenticate this API. -The synthetic token allows external inference only for the exact public synthetic -fixtures supplied by this API. The server verifies the entire fixture content hash. -Changing fixture text or sending real generalized records returns -`external_data_not_approved`, even with `allow_external=true`. -Real data needs separate approval of the existing Hermes Claude account and its -data controls, followed by an explicit server permission change. A request flag -cannot grant that permission. Local-only records do not have this synthetic restriction. +The credentials have different server-enforced scopes: + +| Vault field | Local inference | Hosted inference | +| --- | --- | --- | +| `token` | Allowed within local capacity | Prohibited | +| `synthetic_token` | Allowed within local capacity | Exact API fixtures only; full content hash checked | +| `operational_token` | Allowed within local capacity | Generalized CASE records, Claude only, under the explicit account approval | + +`synthetic_token` has not been broadened. Changing a fixture or sending real +generalized records with that token returns `external_data_not_approved` even +with `allow_external=true`. The operational credential is separate, grants only +`claude`, and requires the server's `PLANNING_GENERALIZED_CLAUDE_APPROVED=true`. +The user explicitly approved the existing Hermes first-party Claude OAuth account +for generalized CASE records on 2026-09-29. No real records were used in acceptance. +Original identifiers, uncensored text, and alias mappings remain on the laptop. +The server cannot determine whether arbitrary prose is adequately generalized; +that remains the application's and organization's responsibility. Account-specific +training and retention settings remain unverified. A request flag cannot grant +credential permission or override data scope. ## Request contract +Submit with `POST https://worker.bstein.dev/suite-planning/v1/jobs`. +The [request JSON Schema](contracts/suite_planning_request.schema.json) describes +the accepted wire object. The [result JSON Schema](contracts/suite_planning_result.schema.json) +describes `result`, not the whole job envelope. Additional server checks enforce +UTF-8 byte limits, unique aliases, ownership, permissions, and exact result coverage. + ``` { "campaign": "SYNTHETIC", @@ -120,7 +235,7 @@ Example completed envelope (illustrative synthetic values): { "job_id": "0123456789abcdef0123456789abcdef", "status": "completed", - "configuration_revision": "suite-v1-20260929", + "configuration_revision": "suite-v6-20260929", "routing": {"allow_external": true, "allowed_external_providers": ["claude"]}, "selection": {"provider": "claude", "backend": "claude-code-2.1.226", "model": "claude-opus-4-8"}, "model": "claude-opus-4-8", @@ -140,6 +255,12 @@ model usage, timing, and estimated cost where available. Unknown values are null Failed jobs retain status and fixed error codes but no provider stderr. Status remains HTTP 200 for an existing failed job; inspect `status` and `error.code`. Invalid submissions use 400/401/403/409/413/415/422/429 as appropriate. +Application errors have the shape `{"error":{"code":"provider_forbidden","details":{}}}`. +Errors raised by Traefik, including its 413 body rejection, can be plain text. +Check HTTP status before parsing JSON. With a large upload, a client can see a +connection reset while still writing after ingress rejects the request; check +the advertised byte cap locally before upload. `Expect: 100-continue` produced +an explicit HTTP 413 in the LAN curl acceptance check. Failures distinguish routing, backend availability, provider authentication, timeouts, invalid JSON, incomplete generation, changed model/capabilities, compaction, and invalid assignments. Failed output is never a completed result. @@ -150,6 +271,35 @@ Every alias must occur exactly once. Duplicate, missing, invented, or foreign aliases reject the entire result. Names must be unique within the returned suite. Coverage validation is independent of the model and does not prove semantic quality. +## Mapping the prepared client job + +The prepared `job.json` is not the wire request. Unknown top-level and case fields +are rejected; this service does not silently unpack the client envelope. + +| Prepared client field | Service mapping | +| --- | --- | +| `job_id` | Use a stable `Idempotency-Key` header, 8-128 ASCII letters/digits/underscore/hyphen. The service returns its own `job_id`; retain both locally. Hash an incompatible client ID once rather than generating a new retry key. | +| `client_spec_version` | Keep locally and validate compatibility with `/v1/capabilities`; not a submitted field. | +| `task` | This endpoint implements only whole-suite implementation-family planning. Validate that task locally; not a submitted field. | +| `routing` | Send only `allow_external` and `allowed_external_providers`. | +| `execution` | Send only `strategy`, `max_seconds`, `max_cost_usd`. | +| `instructions` | Not supported as custom instructions. The server uses the documented fixed implementation-family prompt. If custom instructions are required, this contract does not yet support that request; do not silently discard them. | +| `input` | Lift its one complete suite into top-level `campaign`, `suite`, and `cases`. Keep these identity strings generalized if needed. | +| `output_schema` | Not accepted. The service pins groups with `name`, `description`, `members`; compare the client schema and explicitly adapt or reject incompatible expectations. | +| case `case_id` | Rename to `alias`; must already be the unique CASE alias, never the original identifier. | +| case `name` | No separate wire field. Preserve it explicitly in `description`, for example `Name: ...\nDescription: ...`, if the model needs it. Do not silently lose meaningful name content. | +| case `description` | `description`, optionally combined with the name as above. | +| case `preconditions`, `case_type`, `success_criteria` | Same-named fields, strings or null. | + +The application must select one exact campaign/suite before constructing this +object. It must retain the complete original case records and alias mapping +locally. Group `members` contain aliases only. Server-generated family names and +descriptions do not become automatically permitted ClickUp export content. + +Only small client functions are needed: build the wire projection, validate the +fixed task/schema contract, preflight, submit with a stable key, poll, save the +result, and independently audit aliases. No importer replacement is needed. + ## Installed route capabilities | Property | Codex | Claude | Existing local API | @@ -256,14 +406,17 @@ curl --noproxy '*' --resolve worker.bstein.dev:443:192.168.22.50 \ https://worker.bstein.dev/suite-planning/healthz ``` -Local-only synthetic grouping (either scoped credential): +Local-only synthetic grouping (any scoped credential): ``` +cat > local-synthetic.json <<'JSON' +{"campaign":"SYNTHETIC","suite":"PARSER","cases":[{"alias":"CASE-1","description":"Parse valid configuration text and check returned fields.","preconditions":"An isolated parser fixture is reset before each case.","case_type":"nominal","success_criteria":"Returned fields match the supplied configuration values."},{"alias":"CASE-2","description":"Parse invalid configuration text and check returned diagnostics.","preconditions":"An isolated parser fixture is reset before each case.","case_type":"fault injection","success_criteria":"The malformed input produces the specified diagnostic fields."}],"routing":{"allow_external":false,"allowed_external_providers":[]},"execution":{"strategy":"whole_suite"}} +JSON curl --noproxy '*' --resolve worker.bstein.dev:443:192.168.22.50 \ --connect-timeout 10 --max-time 45 --fail-with-body \ -H "Authorization: Bearer $SUITE_PLANNING_TOKEN" \ -H 'Content-Type: application/json' -H 'Idempotency-Key: local-parser-pilot-001' \ - --data-binary '{"campaign":"SYNTHETIC","suite":"PARSER","cases":[{"alias":"CASE-1","description":"Parse valid configuration text and check returned fields."},{"alias":"CASE-2","description":"Parse invalid configuration text and check returned diagnostics."}],"routing":{"allow_external":false,"allowed_external_providers":[]},"execution":{"strategy":"whole_suite"}}' \ + --data-binary @local-synthetic.json \ https://worker.bstein.dev/suite-planning/v1/jobs ``` @@ -283,6 +436,11 @@ request['execution'] = {'strategy': 'whole_suite', 'max_seconds': 900, 'max_cost with open('synthetic-request.json', 'w') as stream: json.dump(request, stream) PY +curl --noproxy '*' --resolve worker.bstein.dev:443:192.168.22.50 \ + --connect-timeout 10 --max-time 45 --fail-with-body \ + -H "Authorization: Bearer $SUITE_PLANNING_TOKEN" \ + -H 'Content-Type: application/json' --data-binary @synthetic-request.json \ + https://worker.bstein.dev/suite-planning/v1/preflight curl --noproxy '*' --resolve worker.bstein.dev:443:192.168.22.50 \ --connect-timeout 10 --max-time 45 --fail-with-body \ -H "Authorization: Bearer $SUITE_PLANNING_TOKEN" \ @@ -300,6 +458,29 @@ Repeat the last GET until `status` is `completed`, `failed`, or `cancelled`. Save the result locally before retention expires. Keep the original key for network retries. Use a new key only when intentionally starting another provider job. +Capability, status-only, and cancellation requests: + +``` +curl --noproxy '*' --resolve worker.bstein.dev:443:192.168.22.50 \ + --connect-timeout 10 --max-time 45 --fail-with-body \ + -H "Authorization: Bearer $SUITE_PLANNING_TOKEN" \ + https://worker.bstein.dev/suite-planning/v1/capabilities +curl --noproxy '*' --resolve worker.bstein.dev:443:192.168.22.50 \ + --connect-timeout 10 --max-time 45 --fail-with-body \ + -H "Authorization: Bearer $SUITE_PLANNING_TOKEN" \ + "https://worker.bstein.dev/suite-planning/v1/jobs/$JOB_ID" +curl --noproxy '*' --resolve worker.bstein.dev:443:192.168.22.50 \ + --connect-timeout 10 --max-time 45 --fail-with-body -X DELETE \ + -H "Authorization: Bearer $SUITE_PLANNING_TOKEN" \ + "https://worker.bstein.dev/suite-planning/v1/jobs/$JOB_ID" +``` + +For an approved generalized job, obtain `operational_token`, construct the same +wire object locally, and request only `allowed_external_providers:["claude"]`. +The credential name does not go in the HTTP request; only its secret value is +used as the bearer. Neither a provider credential nor a Vault token belongs in +this header. + The standalone `scripts/ops/hermes_suite_probe.py` uses only Python urllib and the same API credential. Its HTTPS handler implements the equivalent of curl `--resolve`, retaining certificate verification and bypassing proxies. It retrieves @@ -329,6 +510,12 @@ the new path does not send content there; account-side retention remains separat Rollback through Git/Flux: +To revoke only generalized-data permission, set +`PLANNING_GENERALIZED_CLAUDE_APPROVED=false` and reconcile `hermes`. +Revoke/rotate `operational_token` through Vault if the credential itself must be +invalidated, then restart through a Flux-tracked deployment revision so the +pre-populated credential mount refreshes. Keep both older token fields intact. + 1. Remove the three suite-planner resource entries and its ConfigMap generator from `services/hermes/kustomization.yaml`; reconcile `hermes` to remove the endpoint. 2. Remove `suite_decision`, the three suite targets/routes, and the corresponding