From 16b30e3a37b3a652b11c22e02652052188c0d7a1 Mon Sep 17 00:00:00 2001 From: jenkins Date: Tue, 29 Sep 2026 08:16:13 -0500 Subject: [PATCH] hermes: pin observed Opus runtime for suite planning --- docs/hermes_suite_planning.md | 342 ++++++++++++++++++ services/hermes/scripts/suite_contract.py | 6 +- services/hermes/suite-planner-deployment.yaml | 2 +- 3 files changed, 346 insertions(+), 4 deletions(-) create mode 100644 docs/hermes_suite_planning.md diff --git a/docs/hermes_suite_planning.md b/docs/hermes_suite_planning.md new file mode 100644 index 00000000..4ee474a2 --- /dev/null +++ b/docs/hermes_suite_planning.md @@ -0,0 +1,342 @@ +# Suite planning API + +This optional API groups one complete campaign/suite into implementation families. +The existing `/local-model` API, GPU allocation, serving model, and limits are unchanged. +Deployment uses Flux; the application and workbook stay on the laptop. + +## Connection and credentials + +``` +LAN_HOST: worker.bstein.dev +LAN_IP: 192.168.22.50 +PORT: 443 +BASE_URL: https://worker.bstein.dev/suite-planning +HEALTH_URL: https://worker.bstein.dev/suite-planning/healthz +CAPABILITIES_URL: https://worker.bstein.dev/suite-planning/v1/capabilities +PREFLIGHT_URL: https://worker.bstein.dev/suite-planning/v1/preflight +SUBMIT_URL: https://worker.bstein.dev/suite-planning/v1/jobs +AUTHENTICATION_HEADER: Authorization: Bearer +VAULT_PATH: kv/atlas/hermes/suite-planning-api +LOCAL_ONLY_FIELD: token +SYNTHETIC_EXTERNAL_FIELD: synthetic_token +INITIAL_CONCURRENCY: 1; busy submissions return 429; no waiting queue +JOB_TIMEOUT: up to 900 seconds +HTTP_TIMEOUT: client 45 seconds; submit/status do not wait for inference +TLS: existing worker.bstein.dev certificate; normal trusted CA verification +``` + +Retrieve the appropriate field privately from +[Vault](https://vault.bstein.dev/ui/vault/secrets/kv/show/atlas/hermes/suite-planning-api). +Neither token grants provider, Vault, Kubernetes, shell, or ClickUp access. +The old local-model token remains at `kv/atlas/hermes/model-gate-lan-api`. +Tokens are distinct; an old local-model token does not authenticate this API. + +The synthetic token allows external inference only for the exact public synthetic +fixtures supplied by this API. The server verifies the entire fixture content hash. +Changing fixture text or sending real generalized records returns +`external_data_not_approved`, even with `allow_external=true`. +Real data needs separate approval of the existing Hermes Claude account and its +data controls, followed by an explicit server permission change. A request flag +cannot grant that permission. Local-only records do not have this synthetic restriction. + +## Request contract + +``` +{ + "campaign": "SYNTHETIC", + "suite": "PARSER", + "cases": [ + {"alias": "CASE-1", "description": "Parse a valid configuration and assert accepted fields."}, + {"alias": "CASE-2", "description": "Parse an invalid configuration and assert diagnostic fields."} + ], + "routing": { + "allow_external": false, + "allowed_external_providers": [] + }, + "execution": { + "strategy": "whole_suite", + "max_seconds": 900, + "max_cost_usd": 5 + } +} +``` + +`campaign`, `suite`, and nonempty `cases` are required. Each case requires a +unique `CASE-*` alias and a description. Optional case fields are +`success_criteria`, `preconditions`, `operating_condition`, `case_type`, +`verification_method`, `target`, `swci`, `verifies`, `functional_area`, +`functional_group`, and `functional_group_name`. Values are strings or null. +Optional per-case campaign/suite values must exactly match the job's ownership. +Unknown fields are rejected. `[reference]` is ordinary source text, never an alias. +Identical descriptions with different aliases remain distinct cases. + +Missing routing policy means local-only. External provider names are `claude` and +`codex`. The requested list must be contained in the credential's permissions; +unknown providers, malformed booleans, duplicate names, and permission escalation +are rejected before routing. `allow_external=false` cannot include providers. +`whole_suite` is the only strategy. No batching, summarization, or reconciliation +is silently substituted. `max_seconds` is 10-900. The Claude CLI budget is at most +USD 5 in its estimated usage accounting; this is not a verified subscription +billing ceiling and can overshoot within a single provider call. + +The gateway deterministically selects a fitting permitted backend, preferring +local capacity. It asks the fixed Switchyard route to confirm that binding, then +performs one inference attempt. Switchyard receives only `select`, never case text. +Its new routes each have exactly one target, zero client retries, and no classifier: +`atlas/planning/local`, `atlas/planning/claude`, `atlas/planning/codex`. +These are decision routes; clients use the HTTPS job API to obtain inference results. +Existing general Hermes routes retain their old behavior and are not this API. + +## Job lifecycle and normalized response + +- `GET /healthz`: authenticated service readiness. +- `GET /v1/capabilities`: permissions, models, configuration revision, and limits. +- `POST /v1/preflight`: validate the identical proposed job without generation. +- `POST /v1/jobs`: require `Idempotency-Key`, return 202 and a job ID. +- `GET /v1/jobs/`: status and metadata. +- `GET /v1/jobs//result`: status plus the normalized result when complete. +- `DELETE /v1/jobs/`: request cancellation. +- `GET /v1/synthetic/14`, `/75`, `/363`: immutable synthetic fixtures. + +Retry a submission with the SAME key and SAME body. It returns the original job, +not a second provider attempt. A changed body with that key returns 409. Keys are +scoped to the authenticated credential and retained seven days. The maximum is +10,000 metadata records; excess submissions fail rather than evicting live keys. +No provider retry or fallback occurs after an error. An operator may intentionally +start a new attempt with a new key after diagnosing the previous failure. + +Results remain in memory for at most one hour (cleanup every 30 seconds), with a +128-result bound. Durable local SQLite holds only operational metadata and hashes, +not prompts or results. Restarts mark unfinished jobs `interrupted_no_retry` and +never resubmit them. Previously completed results are lost on restart; retrieval +returns 410. A cancellation holds the slot until the underlying attempt ends. +Claude cancellation kills the isolated process group. Local generation cannot be +cancelled at the existing API; its eventual result is discarded. Provider-side +cancellation and exact billing after disconnection are not guaranteed. + +Example completed envelope (illustrative synthetic values): + +``` +{ + "job_id": "0123456789abcdef0123456789abcdef", + "status": "completed", + "configuration_revision": "suite-v1-20260929", + "routing": {"allow_external": true, "allowed_external_providers": ["claude"]}, + "selection": {"provider": "claude", "backend": "claude-code-2.1.226", "model": "claude-opus-4-8"}, + "model": "claude-opus-4-8", + "attempted_destinations": ["switchyard:atlas/planning/claude", "claude:claude-opus-4-8"], + "compaction": false, + "truncation": false, + "usage": {"input_tokens": 1000, "output_tokens": 100}, + "wall_seconds": 12.3, + "result": { + "groups": [{"name": "Configuration parser fixtures", "description": "Parameterize configuration text and assert parsed fields or diagnostics.", "members": ["CASE-1", "CASE-2"]}] + } +} +``` + +The real envelope also includes input bytes, input hash, reservation method, +model usage, timing, and estimated cost where available. Unknown values are null. +Failed jobs retain status and fixed error codes but no provider stderr. Status +remains HTTP 200 for an existing failed job; inspect `status` and `error.code`. +Invalid submissions use 400/401/403/409/413/415/422/429 as appropriate. +Failures distinguish routing, backend availability, provider authentication, +timeouts, invalid JSON, incomplete generation, changed model/capabilities, +compaction, and invalid assignments. Failed output is never a completed result. + +The result is a JSON object. The underlying Claude stream/result envelope and +Ollama's JSON-in-a-string representation are normalized on the server. +Every alias must occur exactly once. Duplicate, missing, invented, or foreign +aliases reject the entire result. Names must be unique within the returned suite. +Coverage validation is independent of the model and does not prove semantic quality. + +## Installed route capabilities + +| Property | Codex | Claude | Existing local API | +| --- | --- | --- | --- | +| Installed client | Codex CLI 0.154.0 | Claude Code 2.1.226, pinned binary SHA-256 | Ollama 0.13.5 | +| Existing broker mode | Direct subscription Responses transport, not `codex exec` | Native CLI print mode behind a wrapper | Native `/api/generate` | +| New planner backend | Disabled pending effective output-budget verification | Fresh native CLI process, wrapper bypass avoided | Unchanged model-gate LAN listener | +| Auth | ChatGPT Pro claim verified locally | Existing first-party OAuth setup token; associated credential metadata says Max 20x | Scoped local bearer | +| Visible catalog | `gpt-6-astra`, `gpt-5.6-sol`, `gpt-5.6-terra`, `gpt-5.6-luna`, `gpt-5.5` | Fable 5, Opus 5 with 1M option, Sonnet 5, Haiku 4.5 | Pinned Qwen 2.5 14B Q4 | +| Exact selected ID | `gpt-6-astra` reserved, not enabled | `claude-opus-4-8` | `qwen2.5:14b-instruct-q4_0` | +| Context evidence | Local account model cache 272,000, 95% effective = 258,400 | Actual CLI response: 1,000,000 | Verified serving configuration: 8,192 | +| Output control | Existing broker removes public token-limit fields; effective maximum unverified | Actual CLI reports 64,000; environment pins that ceiling | 2,048 | +| New planner concurrency | None | One across the whole planning service | Shares the existing serialized GPU backend | +| Hardware | Provider hosted; broker on titan-22 | Provider hosted; CLI on titan-22 | RTX 3080 10GB on titan-24 | + +The Claude live catalog resolves `claude-fable-5[1m]` to `claude-fable-5`, +`default`/`opus[1m]` to `claude-opus-5[1m]`, `sonnet` to `claude-sonnet-5`, +and `haiku` to `claude-haiku-4-5-20251001`. Catalog availability is not proof +that every model has been exercised. Fable 5 suite requests reported `claude-opus-4-8` at runtime and were rejected. +The initial working route therefore explicitly pins the observed Opus 4.8 model; +the advertised Fable name is not verified for full-suite execution. +Only the tiny Opus/Fable probes and recorded +planner acceptance jobs were executed. No Claude agent wrote infrastructure code. +Subscription quotas and account retention settings are not verified by model discovery. + +The legacy Codex and Claude wrapper scripts enable permission bypass; the Claude +wrapper also enables automatic compaction and shared settings. Neither is used by +the new worker. Codex supports `exec --ephemeral --output-schema --json` and stdin, +but that separate CLI execution mode has not been approved as a whole-suite backend. +The direct Codex broker uses `store=false`, streaming Responses, and a 900-second +read timeout. That flag does not establish provider Zero Data Retention. + +Claude invocation, with the schema supplied by the server: + +``` +/opt/cli/claude -p --output-format stream-json --verbose \ + --no-session-persistence --safe-mode --tools '' \ + --strict-mcp-config --mcp-config '{"mcpServers":{}}' \ + --setting-sources '' --disable-slash-commands --permission-mode dontAsk \ + --no-chrome --model 'claude-opus-4-8[1m]' --effort medium \ + --max-budget-usd 5 --max-turns 3 \ + --system-prompt '' --json-schema '' +``` + +The complete suite is an isolated input file connected to stdin, never an argv +string or interpolated shell command. The process receives only a short allowlisted +environment, its OAuth token, and a fresh temporary HOME/config directory. +`DISABLE_COMPACT`, `DISABLE_AUTO_COMPACT`, `DISABLE_TELEMETRY`, +`DISABLE_ERROR_REPORTING`, `DISABLE_PROMPT_CACHING`, +`CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC`, `CLAUDE_CODE_DISABLE_AUTO_MEMORY`, +and `CLAUDE_CODE_SKIP_PROMPT_HISTORY` are enabled; retries and updates are disabled. +The actual initialization event must report only `StructuredOutput`, no MCP servers +or plugins. The worker rejects compaction events, unexpected models, incomplete +output, changed limits, and missing terminal results. + +These are client controls, not provider ZDR. Account-specific training/retention +settings remain unverified. See [Claude data usage](https://code.claude.com/docs/en/data-usage), +[Claude CLI flags](https://code.claude.com/docs/en/cli-reference), +[Claude environment controls](https://code.claude.com/docs/en/env-vars), and +[Codex noninteractive behavior](https://learn.chatgpt.com/docs/non-interactive-mode). + +## Capacity and transport + +The new API accepts at most 1 MiB and 400 cases per request, with at most 32 KiB +per field. The existing local endpoint still accepts only 128 KiB and its original +context/output limits. Claude subprocess stdout is bounded to 4 MiB; normalized +results to 1 MiB. Names are at most 80 characters, descriptions at most 240. + +Preflight includes the entire serialized suite, fixed instructions, schema, reserved +harness overhead, and output space. There is no accurate account-specific tokenizer. +The input check uses UTF-8 byte count plus 8,192 reserved harness tokens, with the +64,000 output ceiling reserved for each of at most three turns against context. This is a conservative bound, +not a measured token count. The output reservation is an explicit estimate allowing +a family per case and 8,192 reasoning tokens; it is not a verified bound on arbitrary +generated wording. Medium adaptive reasoning can consume output budget. Incomplete +output fails explicitly; the server never drops cases to turn it into success. + +Small local requests reserve 1,024 overhead tokens and 2,048 output tokens within +8,192 context. The realistic 14/75/363 fixtures exceed this conservative local +whole-suite budget and must use the approved hosted path or fail preflight. + +CLI execution lasts at most 900 seconds; cancellation and timeouts terminate its +process group. Async HTTP calls finish promptly, with a 30-second body-read timeout, +40-second ingress response-header timeout, and recommended 45-second client timeout. +Switchyard's ten-minute internal request limit carries only a tiny immediate routing +decision, not the long inference job or its large body. + +If a real complete suite cannot fit, the current API returns capacity/unsupported. +A future strategy would extract profiles with preserved aliases, generate overlapping +implementation candidates across all batches, compare candidates across batch +boundaries, then reconcile and audit the entire suite. Independent batch grouping +followed by concatenation is not supported. Transport chunking would not add context. + +## WSL examples + +Supply only the scoped credential privately. Do not add `-k` or `-L`. + +``` +read -rsp 'Suite API token: ' SUITE_PLANNING_TOKEN +export SUITE_PLANNING_TOKEN +curl --noproxy '*' --resolve worker.bstein.dev:443:192.168.22.50 \ + --connect-timeout 10 --max-time 45 --fail-with-body \ + -H "Authorization: Bearer $SUITE_PLANNING_TOKEN" \ + https://worker.bstein.dev/suite-planning/healthz +``` + +Local-only synthetic grouping (either scoped credential): + +``` +curl --noproxy '*' --resolve worker.bstein.dev:443:192.168.22.50 \ + --connect-timeout 10 --max-time 45 --fail-with-body \ + -H "Authorization: Bearer $SUITE_PLANNING_TOKEN" \ + -H 'Content-Type: application/json' -H 'Idempotency-Key: local-parser-pilot-001' \ + --data-binary '{"campaign":"SYNTHETIC","suite":"PARSER","cases":[{"alias":"CASE-1","description":"Parse valid configuration text and check returned fields."},{"alias":"CASE-2","description":"Parse invalid configuration text and check returned diagnostics."}],"routing":{"allow_external":false,"allowed_external_providers":[]},"execution":{"strategy":"whole_suite"}}' \ + https://worker.bstein.dev/suite-planning/v1/jobs +``` + +Externally allowed, complete 363-case synthetic grouping (use `synthetic_token`): + +``` +curl --noproxy '*' --resolve worker.bstein.dev:443:192.168.22.50 \ + --connect-timeout 10 --max-time 45 --fail-with-body \ + -H "Authorization: Bearer $SUITE_PLANNING_TOKEN" \ + https://worker.bstein.dev/suite-planning/v1/synthetic/363 > synthetic-suite.json +python3 - <<'PY' +import json +with open('synthetic-suite.json') as stream: + request = json.load(stream) +request['routing'] = {'allow_external': True, 'allowed_external_providers': ['claude']} +request['execution'] = {'strategy': 'whole_suite', 'max_seconds': 900, 'max_cost_usd': 5} +with open('synthetic-request.json', 'w') as stream: + json.dump(request, stream) +PY +curl --noproxy '*' --resolve worker.bstein.dev:443:192.168.22.50 \ + --connect-timeout 10 --max-time 45 --fail-with-body \ + -H "Authorization: Bearer $SUITE_PLANNING_TOKEN" \ + -H 'Content-Type: application/json' -H 'Idempotency-Key: synthetic-363-pilot-001' \ + --data-binary @synthetic-request.json \ + https://worker.bstein.dev/suite-planning/v1/jobs > submitted-job.json +JOB_ID=$(python3 -c 'import json; print(json.load(open("submitted-job.json"))["job_id"])') +curl --noproxy '*' --resolve worker.bstein.dev:443:192.168.22.50 \ + --connect-timeout 10 --max-time 45 --fail-with-body \ + -H "Authorization: Bearer $SUITE_PLANNING_TOKEN" \ + "https://worker.bstein.dev/suite-planning/v1/jobs/$JOB_ID/result" +``` + +Repeat the last GET until `status` is `completed`, `failed`, or `cancelled`. +Save the result locally before retention expires. Keep the original key for network +retries. Use a new key only when intentionally starting another provider job. + +The standalone `scripts/ops/hermes_suite_probe.py` uses only Python urllib and +the same API credential. Its HTTPS handler implements the equivalent of curl +`--resolve`, retaining certificate verification and bypassing proxies. It retrieves +synthetic fixtures, submits, checks idempotency, polls, and validates exact coverage. +It does not integrate with or replace the importer. Proposed application additions +are `get_capabilities()`, `preflight_suite()`, `submit_suite(idempotency_key)`, +`get_job_result()`, and `cancel_job()`. Preserve the application's local selection, +alias mapping, ownership and coverage audits, and export policy. + +## Deployment, logging, and rollback + +The planner runs on titan-22 alongside the existing RWO CLI tools volume. Only an +init container reads its tools subdirectory and copies the hash-pinned native +binary. The serving container has no agent home, cluster token, repository, ClickUp +credentials, provider API keys, or arbitrary command endpoint. Its requests are +100m CPU/512 MiB and limits are 2 CPU/2 GiB. No shared GPU workload is changed. + +Input/output/config/debug storage is per-job tmpfs, removed on success, failure, +timeout, and cancellation. Pod termination also clears it. The observed CLI writes +configuration and backup files despite no session persistence; these share that +temporary directory. The durable metadata PVC uses local-path and is outside the +Longhorn backup path. Ordinary worker logs contain job ID, selected provider/model, +status and duration only. Provider stderr is discarded. The pod is excluded from +Fluent Bit; no prompt is sent to the existing Switchyard routing logs. TLS ingress +does not log bodies. Central archival deletion is not claimed or required because +the new path does not send content there; account-side retention remains separate. + +Rollback through Git/Flux: + +1. Remove the three suite-planner resource entries and its ConfigMap generator from + `services/hermes/kustomization.yaml`; reconcile `hermes` to remove the endpoint. +2. Remove `suite_decision`, the three suite targets/routes, and the corresponding + Switchyard revision update. Reconcile; keep unrelated routes unchanged. +3. Remove the suite Vault seed/bootstrap resources and the suite role/policy additions + if no longer needed. Remove the created Vault role/policy and secret through the + normal Vault administrative workflow; removing a completed Job does not revoke them. +4. Deleting the metadata PVC removes retry protection, so do it only after all jobs + are terminal and no client can submit. No provider job is replayed automatically. + +The existing `/local-model` endpoint and token do not require rollback changes. diff --git a/services/hermes/scripts/suite_contract.py b/services/hermes/scripts/suite_contract.py index f2d28163..a26e54d9 100644 --- a/services/hermes/scripts/suite_contract.py +++ b/services/hermes/scripts/suite_contract.py @@ -19,8 +19,8 @@ MODELS = { "local": {"model": "qwen2.5:14b-instruct-q4_0", "context": 8192, "output": 2048, "overhead": 1024, "backend": "ollama-model-gate", "enabled": True, "reasoning": "none"}, - "claude": {"model": "claude-fable-5", "context": 1000000, - "cli_model": "claude-fable-5[1m]", + "claude": {"model": "claude-opus-4-8", "context": 1000000, + "cli_model": "claude-opus-4-8[1m]", "output": 64000, "overhead": 8192, "backend": "claude-code-2.1.226", "enabled": True, "reasoning": "medium"}, "codex": {"model": "gpt-6-astra", "context": 258400, @@ -156,7 +156,7 @@ def preflight(request): reasons[provider] = "unverified_output_capacity" continue reserve = output_estimate + (8192 if provider == "claude" else 0) - if reserve > model["output"] or input_bytes + model["overhead"] + model["output"] > model["context"]: + if reserve > model["output"] or input_bytes + model["overhead"] + model["output"] * (3 if provider == "claude" else 1) > model["context"]: reasons[provider] = "capacity" continue return {"provider": provider, **model, "configuration_revision": REVISION, diff --git a/services/hermes/suite-planner-deployment.yaml b/services/hermes/suite-planner-deployment.yaml index c176967f..20b642fa 100644 --- a/services/hermes/suite-planner-deployment.yaml +++ b/services/hermes/suite-planner-deployment.yaml @@ -36,7 +36,7 @@ spec: app: hermes-suite-planner annotations: fluentbit.io/exclude: "true" - ai.bstein.dev/config-rev: suite-v4-20260929 + ai.bstein.dev/config-rev: suite-v5-20260929 vault.hashicorp.com/agent-inject: "true" vault.hashicorp.com/agent-pre-populate-only: "true" vault.hashicorp.com/agent-init-first: "true"