hermes: serve authenticated local inference on the LAN

This commit is contained in:
jenkins 2026-09-28 17:10:12 -05:00
parent 4bd487b8b3
commit 713cb8f80f
9 changed files with 475 additions and 2 deletions

View File

@ -0,0 +1,40 @@
#!/usr/bin/env bash
# Call the private gateway with normal TLS verification and no public DNS hop.
set -euo pipefail
if [[ ${1:-} == --help ]]; then
cat <<'USAGE'
Usage: HERMES_LAN_TOKEN_FILE=/path/to/token hermes_lan_generate.sh < prompt.txt
Without a token file, reads kv/atlas/hermes/model-gate-lan-api through Vault CLI.
Optional: HERMES_LAN_ADDRESS (192.168.22.50), HERMES_LAN_MAX_TOKENS (256).
USAGE
exit 0
fi
umask 077
scratch=$(mktemp -d)
trap 'rm -rf -- "$scratch"' EXIT
if [[ -n ${HERMES_LAN_TOKEN_FILE:-} ]]; then
token=$(cat -- "$HERMES_LAN_TOKEN_FILE")
else
token=$(vault kv get -field=token kv/atlas/hermes/model-gate-lan-api)
fi
if [[ ! $token =~ ^[0-9a-f]{64}$ ]]; then
printf 'Invalid LAN gateway token\n' >&2
exit 1
fi
# Keep the bearer out of process arguments and shell tracing in the client.
printf 'header = "Authorization: Bearer %s"\n' "$token" > "$scratch/curl.conf"
unset token
python3 -c '
import json, os, sys
json.dump({"model": "qwen2.5:14b-instruct-q4_0", "prompt": sys.stdin.read(),
"stream": False, "options": {"num_predict": int(os.getenv("HERMES_LAN_MAX_TOKENS", "256"))}}, sys.stdout)
' > "$scratch/request.json"
curl --fail-with-body --silent --show-error --connect-timeout 10 --max-time 310 \
--noproxy worker.bstein.dev \
--resolve "worker.bstein.dev:443:${HERMES_LAN_ADDRESS:-192.168.22.50}" \
--config "$scratch/curl.conf" \
--header 'Content-Type: application/json' \
--data-binary "@$scratch/request.json" \
https://worker.bstein.dev/local-model/api/generate

View File

@ -1,5 +1,32 @@
# Hermes on Atlas: operator guide
## LAN model gateway
The private base URL is `https://worker.bstein.dev/local-model`, reached through
`192.168.22.50:443` on the Atlas LAN. This separate Traefik LoadBalancer preserves
client IPs; the shared public LoadBalancer masks them. The service and middleware
allow only `192.168.22.0/24`, and the gateway also requires a bearer token from
Vault at `kv/atlas/hermes/model-gate-lan-api`, field `token`.
Use `scripts/ops/hermes_lan_generate.sh < prompt.txt` from the LAN with a logged-in
Vault CLI, or set `HERMES_LAN_TOKEN_FILE` to a private file containing that token.
The helper connects directly to the LAN IP while verifying the existing
`worker.bstein.dev` TLS certificate. It needs no DNS override. Public DNS continues
to serve the worker dashboard; it does not select the LAN gateway automatically.
`GET /healthz` and `POST /api/generate` are the only gateway operations. Generate
accepts `model: qwen2.5:14b-instruct-q4_0`, a string `prompt`, `stream: false`, and
optional `num_predict`, `temperature`, `top_p`, and `seed` inside `options`.
Requests are capped at 128 KiB and 2048 output tokens. Concurrent LAN generations
receive 429 and can retry. Requests use only the local Ollama service; no hosted
fallback, model management, tools, sessions, image, or voice APIs are exposed.
A missing/wrong token returns 401, an unavailable credential or local model
returns 503, and a non-LAN source returns 403. Flux bootstraps the Vault roles
and creates the token only if absent. After deliberate token rotation, roll the
model-gate through its Flux-tracked deployment revision to reload the injected
credential.
This is the mental model and demonstration script for the operator instance at
`triage.bstein.dev`. Read it once, then prove each section in the live UI. The
consumer instance at `chat.bstein.dev` is intentionally separate and is not the

View File

@ -42,6 +42,7 @@ resources:
- oauth2-proxy.yaml
- agent-certificate.yaml
- agent-ingress.yaml
- model-gate-lan-ingress.yaml
- execution-worker-rbac.yaml
- hux-evidence-rbac.yaml
- chat-cluster-read-rbac.yaml

View File

@ -10,6 +10,8 @@ data:
"""Normalize Jetson text requests and coordinate the titan-24 image handoff."""
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
import hmac
import ipaddress
import json
import os
from pathlib import Path
@ -22,12 +24,18 @@ data:
LISTEN_HOST = os.environ.get("LISTEN_HOST", "0.0.0.0")
LISTEN_PORT = int(os.environ.get("LISTEN_PORT", "8080"))
LAN_LISTEN_PORT = int(os.environ.get("LAN_LISTEN_PORT", "8082"))
HANDOFF_PORT = int(os.environ.get("HANDOFF_PORT", "8081"))
UPSTREAM_URL = os.environ.get("UPSTREAM_URL", "http://ollama.ai.svc.cluster.local:11434").rstrip("/")
LOCAL_IMAGE_URL = os.environ.get("LOCAL_IMAGE_URL", "http://hermes-local-image.hermes.svc.cluster.local:9004").rstrip("/")
HANDOFF_TIMEOUT_SEC = float(os.environ.get("HANDOFF_TIMEOUT_SEC", "1200"))
IMAGE_NAMESPACE = os.environ.get("IMAGE_NAMESPACE", "hermes")
IMAGE_DEPLOYMENT = os.environ.get("IMAGE_DEPLOYMENT", "hermes-local-image")
LAN_API_TOKEN_FILE = Path(os.environ.get("LAN_API_TOKEN_FILE", "/vault/secrets/lan-api-token"))
LAN_MODEL = "qwen2.5:14b-instruct-q4_0"
LAN_NETWORK = ipaddress.ip_network("192.168.22.0/24")
LAN_MAX_BODY = 131072
_lan_inference = threading.BoundedSemaphore(1)
KUBE_HOST = os.environ.get("KUBERNETES_SERVICE_HOST", "kubernetes.default.svc")
KUBE_PORT = os.environ.get("KUBERNETES_SERVICE_PORT_HTTPS", "443")
KUBE_TOKEN = Path("/var/run/secrets/kubernetes.io/serviceaccount/token")
@ -109,6 +117,138 @@ data:
return False, last_error
def _normalize_lan_generate(body: bytes | None) -> bytes:
"""Accept only the one supported stateless local generation operation."""
if not body:
raise ValueError("request body is required")
try:
payload = json.loads(body)
except (TypeError, ValueError, json.JSONDecodeError) as exc:
raise ValueError("invalid JSON") from exc
if not isinstance(payload, dict):
raise ValueError("JSON object is required")
if payload.get("model") != LAN_MODEL:
raise ValueError(f"model must be {LAN_MODEL}")
if payload.get("stream") is not False:
raise ValueError("stream must be false")
if not isinstance(payload.get("prompt"), str):
raise ValueError("prompt must be a string")
# Only stateless text and bounded sampling settings cross this boundary.
invalid = sorted(set(payload) - {"model", "prompt", "stream", "options"})
if invalid:
raise ValueError(f"unsupported fields: {', '.join(invalid)}")
options = payload.get("options", {})
if not isinstance(options, dict) or set(options) - {"num_predict", "temperature", "top_p", "seed"}:
raise ValueError("unsupported options")
count = options.get("num_predict", 256)
if type(count) is not int or not 1 <= count <= 2048:
raise ValueError("num_predict must be between 1 and 2048")
for key, upper in (("temperature", 2), ("top_p", 1)):
value = options.get(key, 1)
if type(value) not in (int, float) or not 0 <= value <= upper:
raise ValueError(f"invalid {key}")
if "seed" in options and (type(options["seed"]) is not int or not -1 <= options["seed"] <= 2147483647):
raise ValueError("invalid seed")
payload["options"] = {**options, "num_predict": count}
return json.dumps(payload, separators=(",", ":")).encode("utf-8")
def _is_lan_client(client_ip: str) -> bool:
"""Validate the address appended by the trusted ingress proxy."""
try:
return ipaddress.ip_address(client_ip) in LAN_NETWORK
except ValueError:
return False
class LanHandler(BaseHTTPRequestHandler):
"""Narrow authenticated LAN API: health plus one pinned Ollama operation."""
protocol_version = "HTTP/1.1"
def setup(self) -> None:
"""Bound the time a client can hold a request body open."""
super().setup()
self.connection.settimeout(10)
def _json(self, status: int, payload: dict) -> None:
"""End each request so rejected bodies cannot become another request."""
body = json.dumps(payload, separators=(",", ":")).encode("utf-8")
self.send_response(status)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(body)))
self.send_header("Cache-Control", "no-store")
self.send_header("Connection", "close")
self.close_connection = True
self.end_headers()
self.wfile.write(body)
def _authorized(self) -> bool:
"""Require both the ingress-observed LAN address and the Vault token."""
forwarded_for = self.headers.get("X-Forwarded-For", "").split(",")[-1].strip()
if not _is_lan_client(forwarded_for):
self._json(403, {"error": "LAN source required"})
return False
try:
expected = LAN_API_TOKEN_FILE.read_text(encoding="utf-8").strip()
except OSError:
self._json(503, {"error": "LAN API credential unavailable"})
return False
supplied = self.headers.get("Authorization", "")
if not expected or not supplied.startswith("Bearer ") or not hmac.compare_digest(supplied[7:].encode(), expected.encode()):
self._json(401, {"error": "Bearer authentication required"})
return False
return True
def do_GET(self) -> None:
"""Expose authenticated readiness without model or session metadata."""
if self.path != "/healthz":
self._json(404, {"error": "not found"})
return
if self._authorized():
self._json(200, {"status": "ok", "model": LAN_MODEL})
def do_POST(self) -> None:
"""Validate and serialize bounded inference against local Ollama only."""
if self.path != "/api/generate":
self._json(404, {"error": "not found"})
return
if not self._authorized():
return
try:
if self.headers.get("Transfer-Encoding"):
raise ValueError("Transfer-Encoding is unsupported")
length = int(self.headers.get("Content-Length", "0"))
if not 0 < length <= LAN_MAX_BODY:
self._json(413, {"error": "body must be between 1 and 131072 bytes"})
return
body = _normalize_lan_generate(self.rfile.read(length))
except ValueError as exc:
self._json(400, {"error": str(exc)})
return
if not _lan_inference.acquire(blocking=False):
self._json(429, {"error": "local model is busy; retry later"})
return
request = Request(f"{UPSTREAM_URL}/api/generate", data=body, headers={"Content-Type": "application/json"}, method="POST")
try:
try:
with urlopen(request, timeout=300) as response:
payload = json.load(response)
self._json(response.status, payload)
except HTTPError:
self._json(502, {"error": "local Ollama rejected the request"})
except (TimeoutError, URLError, ValueError):
# This endpoint has one upstream only; never route externally.
self._json(503, {"error": "local Ollama unavailable"})
finally:
_lan_inference.release()
def log_message(self, format_string: str, *args) -> None:
# Deliberately log only transport metadata. Never log prompts or responses.
print(f"model-gate-lan {self.address_string()} {format_string % args}", flush=True)
class Handler(BaseHTTPRequestHandler):
"""Proxy the non-preemptible titan-20 text fallback."""
@ -234,4 +374,6 @@ data:
if __name__ == "__main__":
handoff = ThreadingHTTPServer((LISTEN_HOST, HANDOFF_PORT), HandoffHandler)
threading.Thread(target=handoff.serve_forever, name="gpu-handoff", daemon=True).start()
lan_api = ThreadingHTTPServer((LISTEN_HOST, LAN_LISTEN_PORT), LanHandler)
threading.Thread(target=lan_api.serve_forever, name="lan-model-api", daemon=True).start()
ThreadingHTTPServer((LISTEN_HOST, LISTEN_PORT), Handler).serve_forever()

View File

@ -15,7 +15,19 @@ spec:
template:
metadata:
annotations:
ai.bstein.dev/config-rev: "20260811-switchyard-model-fields"
ai.bstein.dev/config-rev: "20260928-lan-model-api-v1"
vault.hashicorp.com/agent-inject: "true"
vault.hashicorp.com/agent-pre-populate-only: "true"
vault.hashicorp.com/agent-init-first: "true"
vault.hashicorp.com/agent-run-as-user: "65532"
vault.hashicorp.com/agent-run-as-group: "65532"
vault.hashicorp.com/role: hermes-model-gate
vault.hashicorp.com/agent-inject-secret-lan-api-token: kv/data/atlas/hermes/model-gate-lan-api
vault.hashicorp.com/agent-inject-template-lan-api-token: |
{{- with secret "kv/data/atlas/hermes/model-gate-lan-api" -}}
{{ .Data.data.token }}
{{- end -}}
vault.hashicorp.com/agent-inject-perms-lan-api-token: "0400"
labels:
app: hermes-model-gate
spec:
@ -64,6 +76,8 @@ spec:
containerPort: 8080
- name: handoff
containerPort: 8081
- name: lan-api
containerPort: 8082
env:
- name: UPSTREAM_URL
value: http://ollama.ai.svc.cluster.local:11434
@ -134,6 +148,22 @@ spec:
---
apiVersion: v1
kind: Service
metadata:
name: hermes-model-gate-lan-api
namespace: hermes
labels:
app: hermes-model-gate
spec:
type: ClusterIP
selector:
app: hermes-model-gate
ports:
- name: http
port: 8082
targetPort: lan-api
---
apiVersion: v1
kind: Service
metadata:
name: hermes-gpu-handoff
namespace: hermes

View File

@ -0,0 +1,47 @@
# services/hermes/model-gate-lan-ingress.yaml
apiVersion: traefik.io/v1alpha1
kind: Middleware
metadata:
name: hermes-model-gate-lan-allowlist
namespace: hermes
spec:
ipAllowList:
sourceRange:
- 192.168.22.0/24
---
apiVersion: traefik.io/v1alpha1
kind: Middleware
metadata:
name: hermes-model-gate-lan-prefix
namespace: hermes
spec:
stripPrefix:
prefixes:
- /local-model
---
apiVersion: networking.k8s.io/v1
kind: Ingress
metadata:
name: hermes-model-gate-lan
namespace: hermes
annotations:
traefik.ingress.kubernetes.io/router.entrypoints: websecure
traefik.ingress.kubernetes.io/router.middlewares: hermes-hermes-model-gate-lan-allowlist@kubernetescrd,hermes-hermes-model-gate-lan-prefix@kubernetescrd
traefik.ingress.kubernetes.io/router.tls: "true"
spec:
ingressClassName: traefik
tls:
- hosts:
- worker.bstein.dev
secretName: hermes-sites-tls
rules:
- host: worker.bstein.dev
http:
paths:
- path: /local-model
pathType: Prefix
backend:
service:
name: hermes-model-gate-lan-api
port:
number: 8082

View File

@ -29,7 +29,7 @@ spec:
podSelector:
matchLabels:
app: hermes-model-gate
policyTypes: [Ingress]
policyTypes: [Ingress, Egress]
ingress:
- from:
- podSelector:
@ -48,6 +48,68 @@ spec:
app: ariadne
ports:
- {protocol: TCP, port: 8081}
- from:
- namespaceSelector:
matchLabels:
kubernetes.io/metadata.name: traefik
podSelector:
matchLabels:
app.kubernetes.io/name: traefik
ports:
- {protocol: TCP, port: 8082}
egress:
# The LAN listener and the original internal model-gate listeners share a
# pod. Constrain the entire pod so no prompt/response can leave the cluster:
# the only inference destination is the local Ollama Service.
- to:
- namespaceSelector:
matchLabels:
kubernetes.io/metadata.name: kube-system
podSelector:
matchLabels:
k8s-app: kube-dns
ports:
- {protocol: UDP, port: 53}
- {protocol: TCP, port: 53}
- to:
- namespaceSelector:
matchLabels:
kubernetes.io/metadata.name: vault
podSelector:
matchLabels:
app: vault
ports:
- {protocol: TCP, port: 8200}
- to:
- namespaceSelector:
matchLabels:
kubernetes.io/metadata.name: ai
podSelector:
matchLabels:
app: ollama
ports:
- {protocol: TCP, port: 11434}
- to:
- podSelector:
matchLabels:
app: hermes-local-image
ports:
- {protocol: TCP, port: 9004}
# Kubernetes API for the existing image-handoff reads: the ClusterIP
# (matched pre-DNAT on some CNIs) plus the real apiserver endpoints on
# 6443, mirroring hermes-chat-tenant-isolation.
- to:
- ipBlock:
cidr: 10.43.0.1/32
- ipBlock:
cidr: 192.168.22.11/32
- ipBlock:
cidr: 192.168.22.12/32
- ipBlock:
cidr: 192.168.22.13/32
ports:
- {protocol: TCP, port: 443}
- {protocol: TCP, port: 6443}
---
apiVersion: networking.k8s.io/v1
kind: NetworkPolicy

View File

View File

@ -0,0 +1,124 @@
"""Exercise the LAN gateway's authentication and local-only HTTP boundary."""
import http.client
import io
import json
from pathlib import Path
import threading
from urllib.error import URLError
import pytest
import yaml
@pytest.fixture
def gateway(tmp_path):
"""Run the shipped handler with an isolated token and intercepted upstream."""
path = Path(__file__).parents[2] / "services/hermes/model-gate-configmap.yaml"
namespace = {"__name__": "test_gateway"}
exec(compile(yaml.safe_load(path.read_text())["data"]["model_gate.py"], str(path), "exec"), namespace)
token = tmp_path / "token"
token.write_text("a" * 64)
namespace["LAN_API_TOKEN_FILE"] = token
calls = []
def upstream(request, timeout):
calls.append(request)
response = io.BytesIO(b'{"response":"LAN_OK","done":true}')
response.status = 200
return response
namespace["urlopen"] = upstream
server = namespace["ThreadingHTTPServer"](("127.0.0.1", 0), namespace["LanHandler"])
thread = threading.Thread(target=server.serve_forever, daemon=True)
thread.start()
def request(method="GET", path="/healthz", body=None, **headers):
defaults = {"Authorization": "Bearer " + "a" * 64, "X-Forwarded-For": "192.168.22.8"}
defaults.update(headers)
connection = http.client.HTTPConnection(*server.server_address, timeout=3)
connection.request(method, path, body=body, headers=defaults)
response = connection.getresponse()
result = response.status, json.loads(response.read())
connection.close()
return result
yield namespace, calls, request
server.shutdown()
server.server_close()
thread.join(timeout=2)
@pytest.mark.parametrize("address", ["", "203.0.113.1", "192.168.22.evil", "192.168.22.8, 203.0.113.1"])
def test_non_lan_and_spoofed_addresses_are_rejected(gateway, address):
_, calls, request = gateway
assert request(**{"X-Forwarded-For": address})[0] == 403
assert not calls
@pytest.mark.parametrize("token", ["", "Bearer wrong", "Basic abc", "Bearer é"])
def test_token_is_required(gateway, token):
_, calls, request = gateway
assert request(Authorization=token)[0] == 401
assert not calls
def test_health_requires_a_readable_credential(gateway):
namespace, calls, request = gateway
assert request()[0] == 200
namespace["LAN_API_TOKEN_FILE"].unlink()
assert request()[0] == 503
assert not calls
def test_generation_uses_only_local_upstream_without_forwarding_token(gateway):
namespace, calls, request = gateway
body = json.dumps({"model": namespace["LAN_MODEL"], "prompt": "Say LAN_OK", "stream": False})
assert request("POST", "/api/generate", body) == (200, {"response": "LAN_OK", "done": True})
assert calls[0].full_url == "http://ollama.ai.svc.cluster.local:11434/api/generate"
assert "Authorization" not in calls[0].headers
assert json.loads(calls[0].data)["options"]["num_predict"] == 256
@pytest.mark.parametrize("changes", [
{"model": "external/model"}, {"stream": True}, {"tools": []}, {"images": []},
{"context": []}, {"keep_alive": 0}, {"options": {"num_predict": -1}},
{"options": {"num_predict": 2049}}, {"options": {"num_gpu": 100}},
{"options": {"temperature": float("nan")}}, {"prompt": []},
])
def test_unsupported_payloads_never_reach_ollama(gateway, changes):
namespace, calls, request = gateway
payload = {"model": namespace["LAN_MODEL"], "prompt": "hello", "stream": False, **changes}
assert request("POST", "/api/generate", json.dumps(payload))[0] == 400
assert not calls
@pytest.mark.parametrize("length,status", [("-1", 413), ("131073", 413), ("invalid", 400)])
def test_invalid_request_lengths_are_rejected_before_read(gateway, length, status):
_, calls, request = gateway
assert request("POST", "/api/generate", "{}", **{"Content-Length": length})[0] == status
assert not calls
def test_outage_and_concurrent_request_fail_without_fallback(gateway):
namespace, calls, request = gateway
body = json.dumps({"model": namespace["LAN_MODEL"], "prompt": "hello", "stream": False})
namespace["_lan_inference"].acquire()
assert request("POST", "/api/generate", body)[0] == 429
namespace["_lan_inference"].release()
def unavailable(*args, **kwargs):
raise URLError("local service unavailable")
namespace["urlopen"] = unavailable
assert request("POST", "/api/generate", body)[0] == 503
assert namespace["_lan_inference"].acquire(blocking=False)
namespace["_lan_inference"].release()
assert not calls
@pytest.mark.parametrize("path", ["/api/pull", "/api/delete", "/v1/chat/completions", "/api/chat"])
def test_model_management_and_agent_routes_are_not_exposed(gateway, path):
_, calls, request = gateway
assert request("POST", path, "{}")[0] == 404
assert not calls