From 616865ba5c446cccf8a5e287bc73a439e9d5580e Mon Sep 17 00:00:00 2001 From: jenkins Date: Mon, 28 Sep 2026 19:24:24 -0500 Subject: [PATCH] docs: record verified RTX LAN endpoint and pilot limitations --- .../LAN_API.md => docs/hermes_lan_api.md | 94 ++++++++++++++++++- services/hermes/NOTES.md | 2 +- 2 files changed, 94 insertions(+), 2 deletions(-) rename services/hermes/LAN_API.md => docs/hermes_lan_api.md (76%) diff --git a/services/hermes/LAN_API.md b/docs/hermes_lan_api.md similarity index 76% rename from services/hermes/LAN_API.md rename to docs/hermes_lan_api.md index a9567ab6..de4e3740 100644 --- a/services/hermes/LAN_API.md +++ b/docs/hermes_lan_api.md @@ -79,7 +79,7 @@ curl --noproxy '*' --resolve worker.bstein.dev:443:192.168.22.50 \ "format": { "type": "object", "properties": { - "case_id": {"type": "string"}, + "case_id": {"type": "string", "enum": ["SYN-LAN-001"]}, "target": {"type": "string"}, "setup": {"type": "string"}, "stimulus": {"type": "string"}, @@ -264,3 +264,95 @@ Use deliberate Git/Flux changes to restore the GPU: The model API can be moved back to the Jetson only as an explicit, documented configuration change with its actual placement recorded. It will not do this silently if the RTX server is down. + +## Verification evidence (2026-09-28 local time) + +- Git/Flux applied the GPU reservation, gateway cutover, and quiet runtime. + `hermes`, `ai-llm`, and `game-stream` Kustomizations reported Ready. +- Sixty focused unit/HTTP tests passed across the LAN handler and native adapter. + Isolated HTTP fixtures failed generation after successful model preflight and + attempted redirects; no alternate backend was contacted and no error body was + reflected. Shared inference services were not interrupted for failure testing. +- Actual LAN jump host `192.168.22.8` connected directly over `end0` to + `192.168.22.50:443`, with system CA verification. The certificate is Let's Encrypt + YR1, contains `worker.bstein.dev`, and expires 2026-11-19. +- The literal curl health, `LAN_POC_OK`, and structured-profile examples passed. + The example preserved target/setup, `17 + 20 = 37`, and unspecified authentication. + The same profile took 25.672 seconds on the Jetson and 4.633 seconds on the warm + RTX backend (gateway wall time). A cold GPU profile took 11.377 seconds including + a 6.800-second model load. These are small synthetic examples, not batch estimates. +- A separate urllib probe verified 401 for missing/invalid credentials, 404 for + management/agent paths, 413 for context overflow, 400 for an unapproved model, + and 429 during an active generation. It retained hostname verification while + connecting to the explicit LAN address with environment proxies disabled. +- GPU runtime external TCP connections were blocked. An unapproved cluster pod + could not connect to its Service. Only the gateway's approved path succeeded. +- Runtime stdout/stderr produced zero lines after suppression. Gateway log checks + found zero occurrences of the synthetic source marker, case ID, or generated + success text. Pod annotations exclude both components from central collection. + A central OpenSearch content query could not run because its pod was unscheduled; + this verification therefore checks content prevention at the producers, not an + end-to-end search of the log store. Its existing availability issue is separate + from the inference endpoint. The attempted global collector change was reverted; + the final logging change is scoped to the two inference-path pods. + +An additional noisier prompt prefixed a diagnostic marker to the synthetic case. +With an unconstrained string ID, Qwen concatenated that marker into `case_id` while +preserving the other facts. This is a measured fidelity failure, not a transport +failure. The example now constrains known IDs with a JSON-schema `enum` and still +requires independent application validation of identities and facts. Do not infer +FA01 planning quality, best-model status, or a 1,661-case runtime from these tests. + +The Windows/WSL client route has **not** been tested from the laptop. Run the +commands above there and check the Windows/VPN route as well as WSL routing. +No actual roster, manually modified importer, FA01 profile extraction, or complete +FA01 suite evaluation was accessed or tested. No ClickUp operation was performed. + +## Actual synthetic response envelope + +Generated JSON is the text value of `response`; durations are nanoseconds except +`inference_provenance.wall_seconds`. This example contains synthetic content only. + +```json +{ + "model": "qwen2.5:14b-instruct-q4_0", + "created_at": "2026-09-29T00:23:58.197589469Z", + "response": "{\n \"case_id\": \"SYN-LAN-001\",\n \"target\": \"calculator HTTP API\",\n \"setup\": \"test instance running\",\n \"stimulus\": \"POST /add with a=17, b=20\",\n \"expected_sum\": 37,\n \"missing_information\": [\n \"Authentication details are unspecified\"\n ]\n}", + "done": true, + "done_reason": "stop", + "total_duration": 4806324917, + "load_duration": 103122083, + "prompt_eval_count": 107, + "prompt_eval_duration": 85072076, + "eval_count": 87, + "eval_duration": 3109012062, + "inference_provenance": { + "model": "qwen2.5:14b-instruct-q4_0", + "model_digest": "5449194ff8035ccb13a6409a5814de6c8f9c39f555f429e383ae0fb7137001bd", + "runtime": "0.13.5", + "placement": "titan-24/RTX-3080-10GB", + "context_tokens": 8192, + "max_output_tokens": 2048, + "timeout_seconds": 1200, + "concurrency": 1, + "fallback": null, + "protocol_version": 1, + "serving_configuration": { + "parallel_requests": 1, + "max_queue": 1, + "flash_attention": true, + "kv_cache_type": "q8_0" + }, + "options": { + "num_ctx": 8192, + "num_predict": 512, + "temperature": 0, + "top_p": 1, + "top_k": 40, + "seed": 0 + }, + "backend_api": "/api/generate", + "wall_seconds": 4.838 + } +} +``` diff --git a/services/hermes/NOTES.md b/services/hermes/NOTES.md index 963848c8..921e6292 100644 --- a/services/hermes/NOTES.md +++ b/services/hermes/NOTES.md @@ -9,7 +9,7 @@ allow only `192.168.22.0/24`, and the gateway also requires a bearer token from Vault at `kv/atlas/hermes/model-gate-lan-api`, field `token`. The endpoint contract and literal WSL curl commands are in -[LAN_API.md](LAN_API.md). The laptop needs only curl or Python urllib and the +[the LAN API guide](../../docs/hermes_lan_api.md). The laptop needs only curl or Python urllib and the scoped bearer credential. Public DNS still selects the worker dashboard; use `--resolve worker.bstein.dev:443:192.168.22.50` for the private connection.