hermes: serve authenticated local inference on the LAN
This commit is contained in:
parent
4bd487b8b3
commit
713cb8f80f
40
scripts/ops/hermes_lan_generate.sh
Executable file
40
scripts/ops/hermes_lan_generate.sh
Executable file
@ -0,0 +1,40 @@
|
||||
#!/usr/bin/env bash
|
||||
# Call the private gateway with normal TLS verification and no public DNS hop.
|
||||
set -euo pipefail
|
||||
|
||||
if [[ ${1:-} == --help ]]; then
|
||||
cat <<'USAGE'
|
||||
Usage: HERMES_LAN_TOKEN_FILE=/path/to/token hermes_lan_generate.sh < prompt.txt
|
||||
Without a token file, reads kv/atlas/hermes/model-gate-lan-api through Vault CLI.
|
||||
Optional: HERMES_LAN_ADDRESS (192.168.22.50), HERMES_LAN_MAX_TOKENS (256).
|
||||
USAGE
|
||||
exit 0
|
||||
fi
|
||||
|
||||
umask 077
|
||||
scratch=$(mktemp -d)
|
||||
trap 'rm -rf -- "$scratch"' EXIT
|
||||
if [[ -n ${HERMES_LAN_TOKEN_FILE:-} ]]; then
|
||||
token=$(cat -- "$HERMES_LAN_TOKEN_FILE")
|
||||
else
|
||||
token=$(vault kv get -field=token kv/atlas/hermes/model-gate-lan-api)
|
||||
fi
|
||||
if [[ ! $token =~ ^[0-9a-f]{64}$ ]]; then
|
||||
printf 'Invalid LAN gateway token\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
# Keep the bearer out of process arguments and shell tracing in the client.
|
||||
printf 'header = "Authorization: Bearer %s"\n' "$token" > "$scratch/curl.conf"
|
||||
unset token
|
||||
python3 -c '
|
||||
import json, os, sys
|
||||
json.dump({"model": "qwen2.5:14b-instruct-q4_0", "prompt": sys.stdin.read(),
|
||||
"stream": False, "options": {"num_predict": int(os.getenv("HERMES_LAN_MAX_TOKENS", "256"))}}, sys.stdout)
|
||||
' > "$scratch/request.json"
|
||||
curl --fail-with-body --silent --show-error --connect-timeout 10 --max-time 310 \
|
||||
--noproxy worker.bstein.dev \
|
||||
--resolve "worker.bstein.dev:443:${HERMES_LAN_ADDRESS:-192.168.22.50}" \
|
||||
--config "$scratch/curl.conf" \
|
||||
--header 'Content-Type: application/json' \
|
||||
--data-binary "@$scratch/request.json" \
|
||||
https://worker.bstein.dev/local-model/api/generate
|
||||
@ -1,5 +1,32 @@
|
||||
# Hermes on Atlas: operator guide
|
||||
|
||||
## LAN model gateway
|
||||
|
||||
The private base URL is `https://worker.bstein.dev/local-model`, reached through
|
||||
`192.168.22.50:443` on the Atlas LAN. This separate Traefik LoadBalancer preserves
|
||||
client IPs; the shared public LoadBalancer masks them. The service and middleware
|
||||
allow only `192.168.22.0/24`, and the gateway also requires a bearer token from
|
||||
Vault at `kv/atlas/hermes/model-gate-lan-api`, field `token`.
|
||||
|
||||
Use `scripts/ops/hermes_lan_generate.sh < prompt.txt` from the LAN with a logged-in
|
||||
Vault CLI, or set `HERMES_LAN_TOKEN_FILE` to a private file containing that token.
|
||||
The helper connects directly to the LAN IP while verifying the existing
|
||||
`worker.bstein.dev` TLS certificate. It needs no DNS override. Public DNS continues
|
||||
to serve the worker dashboard; it does not select the LAN gateway automatically.
|
||||
|
||||
`GET /healthz` and `POST /api/generate` are the only gateway operations. Generate
|
||||
accepts `model: qwen2.5:14b-instruct-q4_0`, a string `prompt`, `stream: false`, and
|
||||
optional `num_predict`, `temperature`, `top_p`, and `seed` inside `options`.
|
||||
Requests are capped at 128 KiB and 2048 output tokens. Concurrent LAN generations
|
||||
receive 429 and can retry. Requests use only the local Ollama service; no hosted
|
||||
fallback, model management, tools, sessions, image, or voice APIs are exposed.
|
||||
|
||||
A missing/wrong token returns 401, an unavailable credential or local model
|
||||
returns 503, and a non-LAN source returns 403. Flux bootstraps the Vault roles
|
||||
and creates the token only if absent. After deliberate token rotation, roll the
|
||||
model-gate through its Flux-tracked deployment revision to reload the injected
|
||||
credential.
|
||||
|
||||
This is the mental model and demonstration script for the operator instance at
|
||||
`triage.bstein.dev`. Read it once, then prove each section in the live UI. The
|
||||
consumer instance at `chat.bstein.dev` is intentionally separate and is not the
|
||||
|
||||
@ -42,6 +42,7 @@ resources:
|
||||
- oauth2-proxy.yaml
|
||||
- agent-certificate.yaml
|
||||
- agent-ingress.yaml
|
||||
- model-gate-lan-ingress.yaml
|
||||
- execution-worker-rbac.yaml
|
||||
- hux-evidence-rbac.yaml
|
||||
- chat-cluster-read-rbac.yaml
|
||||
|
||||
@ -10,6 +10,8 @@ data:
|
||||
"""Normalize Jetson text requests and coordinate the titan-24 image handoff."""
|
||||
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
import hmac
|
||||
import ipaddress
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
@ -22,12 +24,18 @@ data:
|
||||
|
||||
LISTEN_HOST = os.environ.get("LISTEN_HOST", "0.0.0.0")
|
||||
LISTEN_PORT = int(os.environ.get("LISTEN_PORT", "8080"))
|
||||
LAN_LISTEN_PORT = int(os.environ.get("LAN_LISTEN_PORT", "8082"))
|
||||
HANDOFF_PORT = int(os.environ.get("HANDOFF_PORT", "8081"))
|
||||
UPSTREAM_URL = os.environ.get("UPSTREAM_URL", "http://ollama.ai.svc.cluster.local:11434").rstrip("/")
|
||||
LOCAL_IMAGE_URL = os.environ.get("LOCAL_IMAGE_URL", "http://hermes-local-image.hermes.svc.cluster.local:9004").rstrip("/")
|
||||
HANDOFF_TIMEOUT_SEC = float(os.environ.get("HANDOFF_TIMEOUT_SEC", "1200"))
|
||||
IMAGE_NAMESPACE = os.environ.get("IMAGE_NAMESPACE", "hermes")
|
||||
IMAGE_DEPLOYMENT = os.environ.get("IMAGE_DEPLOYMENT", "hermes-local-image")
|
||||
LAN_API_TOKEN_FILE = Path(os.environ.get("LAN_API_TOKEN_FILE", "/vault/secrets/lan-api-token"))
|
||||
LAN_MODEL = "qwen2.5:14b-instruct-q4_0"
|
||||
LAN_NETWORK = ipaddress.ip_network("192.168.22.0/24")
|
||||
LAN_MAX_BODY = 131072
|
||||
_lan_inference = threading.BoundedSemaphore(1)
|
||||
KUBE_HOST = os.environ.get("KUBERNETES_SERVICE_HOST", "kubernetes.default.svc")
|
||||
KUBE_PORT = os.environ.get("KUBERNETES_SERVICE_PORT_HTTPS", "443")
|
||||
KUBE_TOKEN = Path("/var/run/secrets/kubernetes.io/serviceaccount/token")
|
||||
@ -109,6 +117,138 @@ data:
|
||||
return False, last_error
|
||||
|
||||
|
||||
def _normalize_lan_generate(body: bytes | None) -> bytes:
|
||||
"""Accept only the one supported stateless local generation operation."""
|
||||
|
||||
if not body:
|
||||
raise ValueError("request body is required")
|
||||
try:
|
||||
payload = json.loads(body)
|
||||
except (TypeError, ValueError, json.JSONDecodeError) as exc:
|
||||
raise ValueError("invalid JSON") from exc
|
||||
if not isinstance(payload, dict):
|
||||
raise ValueError("JSON object is required")
|
||||
if payload.get("model") != LAN_MODEL:
|
||||
raise ValueError(f"model must be {LAN_MODEL}")
|
||||
if payload.get("stream") is not False:
|
||||
raise ValueError("stream must be false")
|
||||
if not isinstance(payload.get("prompt"), str):
|
||||
raise ValueError("prompt must be a string")
|
||||
# Only stateless text and bounded sampling settings cross this boundary.
|
||||
invalid = sorted(set(payload) - {"model", "prompt", "stream", "options"})
|
||||
if invalid:
|
||||
raise ValueError(f"unsupported fields: {', '.join(invalid)}")
|
||||
options = payload.get("options", {})
|
||||
if not isinstance(options, dict) or set(options) - {"num_predict", "temperature", "top_p", "seed"}:
|
||||
raise ValueError("unsupported options")
|
||||
count = options.get("num_predict", 256)
|
||||
if type(count) is not int or not 1 <= count <= 2048:
|
||||
raise ValueError("num_predict must be between 1 and 2048")
|
||||
for key, upper in (("temperature", 2), ("top_p", 1)):
|
||||
value = options.get(key, 1)
|
||||
if type(value) not in (int, float) or not 0 <= value <= upper:
|
||||
raise ValueError(f"invalid {key}")
|
||||
if "seed" in options and (type(options["seed"]) is not int or not -1 <= options["seed"] <= 2147483647):
|
||||
raise ValueError("invalid seed")
|
||||
payload["options"] = {**options, "num_predict": count}
|
||||
return json.dumps(payload, separators=(",", ":")).encode("utf-8")
|
||||
|
||||
|
||||
def _is_lan_client(client_ip: str) -> bool:
|
||||
"""Validate the address appended by the trusted ingress proxy."""
|
||||
try:
|
||||
return ipaddress.ip_address(client_ip) in LAN_NETWORK
|
||||
except ValueError:
|
||||
return False
|
||||
|
||||
|
||||
class LanHandler(BaseHTTPRequestHandler):
|
||||
"""Narrow authenticated LAN API: health plus one pinned Ollama operation."""
|
||||
|
||||
protocol_version = "HTTP/1.1"
|
||||
|
||||
def setup(self) -> None:
|
||||
"""Bound the time a client can hold a request body open."""
|
||||
super().setup()
|
||||
self.connection.settimeout(10)
|
||||
|
||||
def _json(self, status: int, payload: dict) -> None:
|
||||
"""End each request so rejected bodies cannot become another request."""
|
||||
body = json.dumps(payload, separators=(",", ":")).encode("utf-8")
|
||||
self.send_response(status)
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.send_header("Content-Length", str(len(body)))
|
||||
self.send_header("Cache-Control", "no-store")
|
||||
self.send_header("Connection", "close")
|
||||
self.close_connection = True
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
|
||||
def _authorized(self) -> bool:
|
||||
"""Require both the ingress-observed LAN address and the Vault token."""
|
||||
forwarded_for = self.headers.get("X-Forwarded-For", "").split(",")[-1].strip()
|
||||
if not _is_lan_client(forwarded_for):
|
||||
self._json(403, {"error": "LAN source required"})
|
||||
return False
|
||||
try:
|
||||
expected = LAN_API_TOKEN_FILE.read_text(encoding="utf-8").strip()
|
||||
except OSError:
|
||||
self._json(503, {"error": "LAN API credential unavailable"})
|
||||
return False
|
||||
supplied = self.headers.get("Authorization", "")
|
||||
if not expected or not supplied.startswith("Bearer ") or not hmac.compare_digest(supplied[7:].encode(), expected.encode()):
|
||||
self._json(401, {"error": "Bearer authentication required"})
|
||||
return False
|
||||
return True
|
||||
|
||||
def do_GET(self) -> None:
|
||||
"""Expose authenticated readiness without model or session metadata."""
|
||||
if self.path != "/healthz":
|
||||
self._json(404, {"error": "not found"})
|
||||
return
|
||||
if self._authorized():
|
||||
self._json(200, {"status": "ok", "model": LAN_MODEL})
|
||||
|
||||
def do_POST(self) -> None:
|
||||
"""Validate and serialize bounded inference against local Ollama only."""
|
||||
if self.path != "/api/generate":
|
||||
self._json(404, {"error": "not found"})
|
||||
return
|
||||
if not self._authorized():
|
||||
return
|
||||
try:
|
||||
if self.headers.get("Transfer-Encoding"):
|
||||
raise ValueError("Transfer-Encoding is unsupported")
|
||||
length = int(self.headers.get("Content-Length", "0"))
|
||||
if not 0 < length <= LAN_MAX_BODY:
|
||||
self._json(413, {"error": "body must be between 1 and 131072 bytes"})
|
||||
return
|
||||
body = _normalize_lan_generate(self.rfile.read(length))
|
||||
except ValueError as exc:
|
||||
self._json(400, {"error": str(exc)})
|
||||
return
|
||||
if not _lan_inference.acquire(blocking=False):
|
||||
self._json(429, {"error": "local model is busy; retry later"})
|
||||
return
|
||||
request = Request(f"{UPSTREAM_URL}/api/generate", data=body, headers={"Content-Type": "application/json"}, method="POST")
|
||||
try:
|
||||
try:
|
||||
with urlopen(request, timeout=300) as response:
|
||||
payload = json.load(response)
|
||||
self._json(response.status, payload)
|
||||
except HTTPError:
|
||||
self._json(502, {"error": "local Ollama rejected the request"})
|
||||
except (TimeoutError, URLError, ValueError):
|
||||
# This endpoint has one upstream only; never route externally.
|
||||
self._json(503, {"error": "local Ollama unavailable"})
|
||||
finally:
|
||||
_lan_inference.release()
|
||||
|
||||
def log_message(self, format_string: str, *args) -> None:
|
||||
# Deliberately log only transport metadata. Never log prompts or responses.
|
||||
print(f"model-gate-lan {self.address_string()} {format_string % args}", flush=True)
|
||||
|
||||
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
"""Proxy the non-preemptible titan-20 text fallback."""
|
||||
|
||||
@ -234,4 +374,6 @@ data:
|
||||
if __name__ == "__main__":
|
||||
handoff = ThreadingHTTPServer((LISTEN_HOST, HANDOFF_PORT), HandoffHandler)
|
||||
threading.Thread(target=handoff.serve_forever, name="gpu-handoff", daemon=True).start()
|
||||
lan_api = ThreadingHTTPServer((LISTEN_HOST, LAN_LISTEN_PORT), LanHandler)
|
||||
threading.Thread(target=lan_api.serve_forever, name="lan-model-api", daemon=True).start()
|
||||
ThreadingHTTPServer((LISTEN_HOST, LISTEN_PORT), Handler).serve_forever()
|
||||
|
||||
@ -15,7 +15,19 @@ spec:
|
||||
template:
|
||||
metadata:
|
||||
annotations:
|
||||
ai.bstein.dev/config-rev: "20260811-switchyard-model-fields"
|
||||
ai.bstein.dev/config-rev: "20260928-lan-model-api-v1"
|
||||
vault.hashicorp.com/agent-inject: "true"
|
||||
vault.hashicorp.com/agent-pre-populate-only: "true"
|
||||
vault.hashicorp.com/agent-init-first: "true"
|
||||
vault.hashicorp.com/agent-run-as-user: "65532"
|
||||
vault.hashicorp.com/agent-run-as-group: "65532"
|
||||
vault.hashicorp.com/role: hermes-model-gate
|
||||
vault.hashicorp.com/agent-inject-secret-lan-api-token: kv/data/atlas/hermes/model-gate-lan-api
|
||||
vault.hashicorp.com/agent-inject-template-lan-api-token: |
|
||||
{{- with secret "kv/data/atlas/hermes/model-gate-lan-api" -}}
|
||||
{{ .Data.data.token }}
|
||||
{{- end -}}
|
||||
vault.hashicorp.com/agent-inject-perms-lan-api-token: "0400"
|
||||
labels:
|
||||
app: hermes-model-gate
|
||||
spec:
|
||||
@ -64,6 +76,8 @@ spec:
|
||||
containerPort: 8080
|
||||
- name: handoff
|
||||
containerPort: 8081
|
||||
- name: lan-api
|
||||
containerPort: 8082
|
||||
env:
|
||||
- name: UPSTREAM_URL
|
||||
value: http://ollama.ai.svc.cluster.local:11434
|
||||
@ -134,6 +148,22 @@ spec:
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: hermes-model-gate-lan-api
|
||||
namespace: hermes
|
||||
labels:
|
||||
app: hermes-model-gate
|
||||
spec:
|
||||
type: ClusterIP
|
||||
selector:
|
||||
app: hermes-model-gate
|
||||
ports:
|
||||
- name: http
|
||||
port: 8082
|
||||
targetPort: lan-api
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: hermes-gpu-handoff
|
||||
namespace: hermes
|
||||
|
||||
47
services/hermes/model-gate-lan-ingress.yaml
Normal file
47
services/hermes/model-gate-lan-ingress.yaml
Normal file
@ -0,0 +1,47 @@
|
||||
# services/hermes/model-gate-lan-ingress.yaml
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: Middleware
|
||||
metadata:
|
||||
name: hermes-model-gate-lan-allowlist
|
||||
namespace: hermes
|
||||
spec:
|
||||
ipAllowList:
|
||||
sourceRange:
|
||||
- 192.168.22.0/24
|
||||
---
|
||||
apiVersion: traefik.io/v1alpha1
|
||||
kind: Middleware
|
||||
metadata:
|
||||
name: hermes-model-gate-lan-prefix
|
||||
namespace: hermes
|
||||
spec:
|
||||
stripPrefix:
|
||||
prefixes:
|
||||
- /local-model
|
||||
---
|
||||
apiVersion: networking.k8s.io/v1
|
||||
kind: Ingress
|
||||
metadata:
|
||||
name: hermes-model-gate-lan
|
||||
namespace: hermes
|
||||
annotations:
|
||||
traefik.ingress.kubernetes.io/router.entrypoints: websecure
|
||||
traefik.ingress.kubernetes.io/router.middlewares: hermes-hermes-model-gate-lan-allowlist@kubernetescrd,hermes-hermes-model-gate-lan-prefix@kubernetescrd
|
||||
traefik.ingress.kubernetes.io/router.tls: "true"
|
||||
spec:
|
||||
ingressClassName: traefik
|
||||
tls:
|
||||
- hosts:
|
||||
- worker.bstein.dev
|
||||
secretName: hermes-sites-tls
|
||||
rules:
|
||||
- host: worker.bstein.dev
|
||||
http:
|
||||
paths:
|
||||
- path: /local-model
|
||||
pathType: Prefix
|
||||
backend:
|
||||
service:
|
||||
name: hermes-model-gate-lan-api
|
||||
port:
|
||||
number: 8082
|
||||
@ -29,7 +29,7 @@ spec:
|
||||
podSelector:
|
||||
matchLabels:
|
||||
app: hermes-model-gate
|
||||
policyTypes: [Ingress]
|
||||
policyTypes: [Ingress, Egress]
|
||||
ingress:
|
||||
- from:
|
||||
- podSelector:
|
||||
@ -48,6 +48,68 @@ spec:
|
||||
app: ariadne
|
||||
ports:
|
||||
- {protocol: TCP, port: 8081}
|
||||
- from:
|
||||
- namespaceSelector:
|
||||
matchLabels:
|
||||
kubernetes.io/metadata.name: traefik
|
||||
podSelector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: traefik
|
||||
ports:
|
||||
- {protocol: TCP, port: 8082}
|
||||
egress:
|
||||
# The LAN listener and the original internal model-gate listeners share a
|
||||
# pod. Constrain the entire pod so no prompt/response can leave the cluster:
|
||||
# the only inference destination is the local Ollama Service.
|
||||
- to:
|
||||
- namespaceSelector:
|
||||
matchLabels:
|
||||
kubernetes.io/metadata.name: kube-system
|
||||
podSelector:
|
||||
matchLabels:
|
||||
k8s-app: kube-dns
|
||||
ports:
|
||||
- {protocol: UDP, port: 53}
|
||||
- {protocol: TCP, port: 53}
|
||||
- to:
|
||||
- namespaceSelector:
|
||||
matchLabels:
|
||||
kubernetes.io/metadata.name: vault
|
||||
podSelector:
|
||||
matchLabels:
|
||||
app: vault
|
||||
ports:
|
||||
- {protocol: TCP, port: 8200}
|
||||
- to:
|
||||
- namespaceSelector:
|
||||
matchLabels:
|
||||
kubernetes.io/metadata.name: ai
|
||||
podSelector:
|
||||
matchLabels:
|
||||
app: ollama
|
||||
ports:
|
||||
- {protocol: TCP, port: 11434}
|
||||
- to:
|
||||
- podSelector:
|
||||
matchLabels:
|
||||
app: hermes-local-image
|
||||
ports:
|
||||
- {protocol: TCP, port: 9004}
|
||||
# Kubernetes API for the existing image-handoff reads: the ClusterIP
|
||||
# (matched pre-DNAT on some CNIs) plus the real apiserver endpoints on
|
||||
# 6443, mirroring hermes-chat-tenant-isolation.
|
||||
- to:
|
||||
- ipBlock:
|
||||
cidr: 10.43.0.1/32
|
||||
- ipBlock:
|
||||
cidr: 192.168.22.11/32
|
||||
- ipBlock:
|
||||
cidr: 192.168.22.12/32
|
||||
- ipBlock:
|
||||
cidr: 192.168.22.13/32
|
||||
ports:
|
||||
- {protocol: TCP, port: 443}
|
||||
- {protocol: TCP, port: 6443}
|
||||
---
|
||||
apiVersion: networking.k8s.io/v1
|
||||
kind: NetworkPolicy
|
||||
|
||||
0
services/vault/scripts/vault_hermes_model_gate_lan_token_ensure.sh
Normal file → Executable file
0
services/vault/scripts/vault_hermes_model_gate_lan_token_ensure.sh
Normal file → Executable file
124
testing/tests/test_hermes_model_gate_lan.py
Normal file
124
testing/tests/test_hermes_model_gate_lan.py
Normal file
@ -0,0 +1,124 @@
|
||||
"""Exercise the LAN gateway's authentication and local-only HTTP boundary."""
|
||||
|
||||
import http.client
|
||||
import io
|
||||
import json
|
||||
from pathlib import Path
|
||||
import threading
|
||||
from urllib.error import URLError
|
||||
|
||||
import pytest
|
||||
import yaml
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def gateway(tmp_path):
|
||||
"""Run the shipped handler with an isolated token and intercepted upstream."""
|
||||
path = Path(__file__).parents[2] / "services/hermes/model-gate-configmap.yaml"
|
||||
namespace = {"__name__": "test_gateway"}
|
||||
exec(compile(yaml.safe_load(path.read_text())["data"]["model_gate.py"], str(path), "exec"), namespace)
|
||||
token = tmp_path / "token"
|
||||
token.write_text("a" * 64)
|
||||
namespace["LAN_API_TOKEN_FILE"] = token
|
||||
calls = []
|
||||
|
||||
def upstream(request, timeout):
|
||||
calls.append(request)
|
||||
response = io.BytesIO(b'{"response":"LAN_OK","done":true}')
|
||||
response.status = 200
|
||||
return response
|
||||
|
||||
namespace["urlopen"] = upstream
|
||||
server = namespace["ThreadingHTTPServer"](("127.0.0.1", 0), namespace["LanHandler"])
|
||||
thread = threading.Thread(target=server.serve_forever, daemon=True)
|
||||
thread.start()
|
||||
|
||||
def request(method="GET", path="/healthz", body=None, **headers):
|
||||
defaults = {"Authorization": "Bearer " + "a" * 64, "X-Forwarded-For": "192.168.22.8"}
|
||||
defaults.update(headers)
|
||||
connection = http.client.HTTPConnection(*server.server_address, timeout=3)
|
||||
connection.request(method, path, body=body, headers=defaults)
|
||||
response = connection.getresponse()
|
||||
result = response.status, json.loads(response.read())
|
||||
connection.close()
|
||||
return result
|
||||
|
||||
yield namespace, calls, request
|
||||
server.shutdown()
|
||||
server.server_close()
|
||||
thread.join(timeout=2)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("address", ["", "203.0.113.1", "192.168.22.evil", "192.168.22.8, 203.0.113.1"])
|
||||
def test_non_lan_and_spoofed_addresses_are_rejected(gateway, address):
|
||||
_, calls, request = gateway
|
||||
assert request(**{"X-Forwarded-For": address})[0] == 403
|
||||
assert not calls
|
||||
|
||||
|
||||
@pytest.mark.parametrize("token", ["", "Bearer wrong", "Basic abc", "Bearer é"])
|
||||
def test_token_is_required(gateway, token):
|
||||
_, calls, request = gateway
|
||||
assert request(Authorization=token)[0] == 401
|
||||
assert not calls
|
||||
|
||||
|
||||
def test_health_requires_a_readable_credential(gateway):
|
||||
namespace, calls, request = gateway
|
||||
assert request()[0] == 200
|
||||
namespace["LAN_API_TOKEN_FILE"].unlink()
|
||||
assert request()[0] == 503
|
||||
assert not calls
|
||||
|
||||
|
||||
def test_generation_uses_only_local_upstream_without_forwarding_token(gateway):
|
||||
namespace, calls, request = gateway
|
||||
body = json.dumps({"model": namespace["LAN_MODEL"], "prompt": "Say LAN_OK", "stream": False})
|
||||
assert request("POST", "/api/generate", body) == (200, {"response": "LAN_OK", "done": True})
|
||||
assert calls[0].full_url == "http://ollama.ai.svc.cluster.local:11434/api/generate"
|
||||
assert "Authorization" not in calls[0].headers
|
||||
assert json.loads(calls[0].data)["options"]["num_predict"] == 256
|
||||
|
||||
|
||||
@pytest.mark.parametrize("changes", [
|
||||
{"model": "external/model"}, {"stream": True}, {"tools": []}, {"images": []},
|
||||
{"context": []}, {"keep_alive": 0}, {"options": {"num_predict": -1}},
|
||||
{"options": {"num_predict": 2049}}, {"options": {"num_gpu": 100}},
|
||||
{"options": {"temperature": float("nan")}}, {"prompt": []},
|
||||
])
|
||||
def test_unsupported_payloads_never_reach_ollama(gateway, changes):
|
||||
namespace, calls, request = gateway
|
||||
payload = {"model": namespace["LAN_MODEL"], "prompt": "hello", "stream": False, **changes}
|
||||
assert request("POST", "/api/generate", json.dumps(payload))[0] == 400
|
||||
assert not calls
|
||||
|
||||
|
||||
@pytest.mark.parametrize("length,status", [("-1", 413), ("131073", 413), ("invalid", 400)])
|
||||
def test_invalid_request_lengths_are_rejected_before_read(gateway, length, status):
|
||||
_, calls, request = gateway
|
||||
assert request("POST", "/api/generate", "{}", **{"Content-Length": length})[0] == status
|
||||
assert not calls
|
||||
|
||||
|
||||
def test_outage_and_concurrent_request_fail_without_fallback(gateway):
|
||||
namespace, calls, request = gateway
|
||||
body = json.dumps({"model": namespace["LAN_MODEL"], "prompt": "hello", "stream": False})
|
||||
namespace["_lan_inference"].acquire()
|
||||
assert request("POST", "/api/generate", body)[0] == 429
|
||||
namespace["_lan_inference"].release()
|
||||
|
||||
def unavailable(*args, **kwargs):
|
||||
raise URLError("local service unavailable")
|
||||
|
||||
namespace["urlopen"] = unavailable
|
||||
assert request("POST", "/api/generate", body)[0] == 503
|
||||
assert namespace["_lan_inference"].acquire(blocking=False)
|
||||
namespace["_lan_inference"].release()
|
||||
assert not calls
|
||||
|
||||
|
||||
@pytest.mark.parametrize("path", ["/api/pull", "/api/delete", "/v1/chat/completions", "/api/chat"])
|
||||
def test_model_management_and_agent_routes_are_not_exposed(gateway, path):
|
||||
_, calls, request = gateway
|
||||
assert request("POST", path, "{}")[0] == 404
|
||||
assert not calls
|
||||
Loading…
x
Reference in New Issue
Block a user