diff --git a/services/hermes/agent-deployment.yaml b/services/hermes/agent-deployment.yaml index a221802cf..c54358ae1 100644 --- a/services/hermes/agent-deployment.yaml +++ b/services/hermes/agent-deployment.yaml @@ -24,7 +24,7 @@ spec: ai.bstein.dev/execution: Hermes Kanban with durable direct Codex and Claude Code CLI workers ai.bstein.dev/model-policy: Jetson-assisted AUTO routing, low through xhigh, cross-provider fallback ai.bstein.dev/placement: rpi5 preferred; Jetson deferred until state storage is available - ai.bstein.dev/config-rev: "20260811-codex-cli-auxiliary" + ai.bstein.dev/config-rev: "20260811-codex-image-broker" vault.hashicorp.com/agent-inject: "true" vault.hashicorp.com/role: hermes-agent vault.hashicorp.com/agent-inject-secret-anthropic-token: kv/data/atlas/hermes/agent-tokens @@ -39,6 +39,11 @@ spec: {{ . }} {{- end -}} {{- end }} + vault.hashicorp.com/agent-inject-secret-chat-relay-key: kv/data/atlas/hermes/chat-telegram + vault.hashicorp.com/agent-inject-template-chat-relay-key: | + {{- with secret "kv/data/atlas/hermes/chat-telegram" -}} + {{ .Data.data.relay_key }} + {{- end }} vault.hashicorp.com/agent-inject-secret-oidc-config: kv/data/atlas/hermes/agent-oidc vault.hashicorp.com/agent-inject-template-oidc-config: | {{- with secret "kv/data/atlas/hermes/agent-oidc" -}} @@ -127,6 +132,10 @@ spec: token="$(tr -d '\r\n' < /vault/secrets/gitea-token)" case "${token}" in ""|""|"") ;; *) upsert_env GITEA_TOKEN "${token}" ;; esac fi + if [ -s /vault/secrets/chat-relay-key ]; then + token="$(tr -d '\r\n' < /vault/secrets/chat-relay-key)" + [ -z "${token}" ] || upsert_env HERMES_IMAGE_BROKER_KEY "${token}" + fi upsert_env GITEA_USERNAME bstein upsert_env GIT_ASKPASS /opt/coordinator/gitea_askpass.sh upsert_env GIT_TERMINAL_PROMPT 0 @@ -677,6 +686,52 @@ spec: resources: requests: {cpu: 25m, memory: 64Mi} limits: {cpu: 250m, memory: 512Mi} + - name: image-broker + image: registry.bstein.dev/bstein/hermes-agent@sha256:cf07be056feea8e4f2d5512899f990732f3d26b4915257de8f90105d96cec67d + imagePullPolicy: IfNotPresent + command: [/bin/sh, -ec] + args: + - | + set -a + . /opt/data/.env + set +a + exec /opt/hermes/.venv/bin/python /opt/coordinator/image_broker.py + ports: + - {name: image-broker, containerPort: 9002, protocol: TCP} + env: + - {name: HERMES_HOME, value: /opt/data} + - {name: HERMES_AUTH_FILE, value: /shared-auth/auth.json} + - {name: HOME, value: /opt/data/home} + - {name: CODEX_HOME, value: /opt/data/home/.codex} + - {name: PYTHONPATH, value: /opt/hermes} + - {name: HERMES_IMAGE_BROKER_DEFAULT_MODEL, value: gpt-image-2-high} + readinessProbe: + tcpSocket: {port: image-broker} + initialDelaySeconds: 5 + periodSeconds: 10 + livenessProbe: + tcpSocket: {port: image-broker} + initialDelaySeconds: 30 + periodSeconds: 30 + securityContext: + allowPrivilegeEscalation: false + capabilities: + drop: [ALL] + readOnlyRootFilesystem: true + runAsNonRoot: true + runAsUser: 10000 + runAsGroup: 10000 + seccompProfile: + type: RuntimeDefault + volumeMounts: + - {name: home, mountPath: /opt/data} + - {name: provider-auth, mountPath: /shared-auth} + - {name: coordinator, mountPath: /opt/coordinator, readOnly: true} + - {name: codex-runtime-patch, mountPath: /opt/hermes/agent/auxiliary_client.py, subPath: auxiliary_client.py} + - {name: tmp, mountPath: /tmp} + resources: + requests: {cpu: 50m, memory: 128Mi} + limits: {cpu: "1", memory: 1Gi} volumes: - name: home persistentVolumeClaim: diff --git a/services/hermes/chat-configmap.yaml b/services/hermes/chat-configmap.yaml index c1c4fc8e2..5b01341a7 100644 --- a/services/hermes/chat-configmap.yaml +++ b/services/hermes/chat-configmap.yaml @@ -41,12 +41,15 @@ data: local: model: small language: auto + image_gen: + provider: atlas-broker + model: gpt-image-2-high model_catalog: enabled: true ttl_hours: 1 platform_toolsets: - cli: [browser, clarify, delegation, file, memory, python_sandbox, session_search, skills, todo, vision, web] - api_server: [browser, clarify, delegation, file, memory, python_sandbox, session_search, skills, todo, vision, web] + cli: [browser, clarify, delegation, file, image_gen, memory, python_sandbox, session_search, skills, todo, vision, web] + api_server: [browser, clarify, delegation, file, image_gen, memory, python_sandbox, session_search, skills, todo, vision, web] dashboard: public_url: https://chat.hermes.bstein.dev display: @@ -97,6 +100,13 @@ data: Voice conversations use the same assistant, session, AUTO route, tools, and private workspace as typed conversations. Whisper and speech synthesis are transport services only; they do not select or replace the answering model. + + When a user asks to create or edit an image, use the image generation tool. + Default to its highest-quality GPT Image tier, honor requested aspect ratio, + and use uploaded images as references when provided. Return the generated + image inline so the WebUI offers its normal preview and download controls. + Do not substitute Python drawing, SVG, diagrams, or placeholder artwork for + a requested generative image. AGENTS.md: | # Private Hermes chat @@ -108,5 +118,7 @@ data: for bounded temporary work. Use delegation selectively for independent research or verification, then present a single final answer. + Use the image generation tool for natural-language image creation and + editing requests; generated images remain in this tenant's private cache. Do not claim access to Kubernetes, Vault, Gitea, Brad's projects, other users, the agent coordinator, or automated triage. diff --git a/services/hermes/chat-statefulset.yaml b/services/hermes/chat-statefulset.yaml index bf4e6570a..8d09aa5ca 100644 --- a/services/hermes/chat-statefulset.yaml +++ b/services/hermes/chat-statefulset.yaml @@ -28,7 +28,7 @@ spec: ai.bstein.dev/role: isolated-user-chat ai.bstein.dev/isolation: one Hermes process and PVC per Keycloak subject ai.bstein.dev/model-policy: uniform automatic policy with per-user overrides - ai.bstein.dev/config-rev: "20260810-noise-gated-voice" + ai.bstein.dev/config-rev: "20260811-private-image-studio" vault.hashicorp.com/agent-inject: "true" vault.hashicorp.com/role: hermes-chat vault.hashicorp.com/agent-inject-secret-anthropic-token: kv/data/atlas/hermes/agent-tokens @@ -120,6 +120,8 @@ spec: fi { grep -v '^API_SERVER_KEY=' /opt/data/.env || true; printf 'API_SERVER_KEY=%s\n' "${relay_key}"; } > /opt/data/.env.tmp mv /opt/data/.env.tmp /opt/data/.env + { grep -v '^HERMES_IMAGE_BROKER_KEY=' /opt/data/.env || true; printf 'HERMES_IMAGE_BROKER_KEY=%s\n' "${relay_key}"; } > /opt/data/.env.tmp + mv /opt/data/.env.tmp /opt/data/.env if [ -s /vault/secrets/anthropic-token ]; then token="$(tr -d '\r\n' < /vault/secrets/anthropic-token)" if [ -n "${token}" ]; then @@ -183,6 +185,9 @@ spec: - | ordinal="${HOSTNAME##*-}" export HERMES_CODE_SANDBOX_URL="http://hermes-chat-sandbox-${ordinal}.hermes-chat-sandbox.hermes.svc.cluster.local:9080/v1/execute" + set -a + . /opt/data/.env + set +a exec /opt/hermes/.venv/bin/hermes gateway run ports: - {name: api, containerPort: 8642, protocol: TCP} @@ -201,11 +206,16 @@ spec: - {name: API_SERVER_HOST, value: 0.0.0.0} - {name: API_SERVER_PORT, value: "8642"} - {name: API_SERVER_CORS_ORIGINS, value: https://chat.hermes.bstein.dev} + - {name: HERMES_IMAGE_BROKER_URL, value: http://hermes-image-broker.hermes.svc.cluster.local:9002} volumeMounts: - {name: home, mountPath: /opt/data} - {name: workspace, mountPath: /opt/data/workspace} - - {name: provider-auth, mountPath: /shared-auth, readOnly: true} + # The shared provider pool uses auth.lock to serialize token refresh + # across tenant gateways. Tenant files and conversations remain on + # their own PVCs; only provider credentials are shared here. + - {name: provider-auth, mountPath: /shared-auth} - {name: auth-patch, mountPath: /opt/hermes/hermes_cli/auth.py, subPath: auth.py} + - {name: image-plugin, mountPath: /opt/hermes/plugins/image_gen/atlas-broker, readOnly: true} readinessProbe: tcpSocket: {port: api} initialDelaySeconds: 30 @@ -299,6 +309,9 @@ spec: defaultMode: 0555 - name: auth-patch emptyDir: {} + - name: image-plugin + configMap: + name: hermes-chat-image-plugin - name: tmp emptyDir: sizeLimit: 256Mi diff --git a/services/hermes/kustomization.yaml b/services/hermes/kustomization.yaml index aec81c35c..778ebec0e 100644 --- a/services/hermes/kustomization.yaml +++ b/services/hermes/kustomization.yaml @@ -55,6 +55,7 @@ configMapGenerator: - hermes_coordinator.py=scripts/hermes_coordinator.py - hermes_model_routing.py=scripts/hermes_model_routing.py - hermes_stt_client.py=scripts/hermes_stt_client.py + - image_broker.py=scripts/image_broker.py - install_agent_tools.sh=scripts/install_agent_tools.sh - migrate_herdr_state.py=scripts/migrate_herdr_state.py - patch_hermes_auth.py=scripts/patch_hermes_auth.py @@ -76,6 +77,13 @@ configMapGenerator: - plugin.yaml=plugins/auto-router/plugin.yaml options: disableNameSuffixHash: true + - name: hermes-chat-image-plugin + namespace: hermes + files: + - __init__.py=plugins/image-gen-broker/__init__.py + - plugin.yaml=plugins/image-gen-broker/plugin.yaml + options: + disableNameSuffixHash: true - name: hermes-triage-skill namespace: hermes files: diff --git a/services/hermes/model-gate-deployment.yaml b/services/hermes/model-gate-deployment.yaml index a8c9511cd..69f79a153 100644 --- a/services/hermes/model-gate-deployment.yaml +++ b/services/hermes/model-gate-deployment.yaml @@ -14,6 +14,8 @@ spec: app: hermes-model-gate template: metadata: + annotations: + ai.bstein.dev/config-rev: "20260811-reasoning-clamp" labels: app: hermes-model-gate spec: diff --git a/services/hermes/networkpolicy.yaml b/services/hermes/networkpolicy.yaml index 059a6ee0d..c7ee4a942 100644 --- a/services/hermes/networkpolicy.yaml +++ b/services/hermes/networkpolicy.yaml @@ -76,6 +76,12 @@ spec: app.kubernetes.io/name: traefik ports: - {protocol: TCP, port: 4180} + - from: + - podSelector: + matchLabels: + app: hermes-chat-tenant + ports: + - {protocol: TCP, port: 9002} # agent.hermes.bstein.dev is an owner-only engineering workstation. The # browser boundary remains OAuth-protected, while its workers need to reach # every cluster namespace, Atlas LAN service, and hosted provider endpoint. @@ -224,6 +230,12 @@ spec: app: hermes-model-gate ports: - {protocol: TCP, port: 8080} + - to: + - podSelector: + matchLabels: + app: hermes-agent + ports: + - {protocol: TCP, port: 9002} - to: - podSelector: matchLabels: diff --git a/services/hermes/plugins/image-gen-broker/__init__.py b/services/hermes/plugins/image-gen-broker/__init__.py new file mode 100644 index 000000000..d33e9e41f --- /dev/null +++ b/services/hermes/plugins/image-gen-broker/__init__.py @@ -0,0 +1,223 @@ +"""Hermes image backend that keeps owner OAuth outside family chat pods.""" + +from __future__ import annotations + +import base64 +import json +import os +from pathlib import Path +from typing import Any +from urllib.error import HTTPError, URLError +from urllib.request import Request, urlopen + +from agent.image_gen_provider import ( + DEFAULT_ASPECT_RATIO, + ImageGenProvider, + error_response, + normalize_reference_images, + resolve_aspect_ratio, + save_b64_image, + success_response, +) + + +BROKER_URL = os.environ.get( + "HERMES_IMAGE_BROKER_URL", + "http://hermes-image-broker.hermes.svc.cluster.local:9002", +).rstrip("/") +DEFAULT_MODEL = "gpt-image-2-high" +MODELS = { + "gpt-image-2-low": ("GPT Image 2 (Fast)", "low"), + "gpt-image-2-medium": ("GPT Image 2 (Balanced)", "medium"), + "gpt-image-2-high": ("GPT Image 2 (Highest quality)", "high"), +} +MAX_INPUT_BYTES = 25 << 20 + + +def _broker_key() -> str: + """Load the internal relay key from process env or the private .env.""" + value = os.environ.get("HERMES_IMAGE_BROKER_KEY", "").strip() + if value: + return value + home = Path(os.environ.get("HERMES_HOME", "/opt/data")) + try: + for line in (home / ".env").read_text(encoding="utf-8").splitlines(): + if line.startswith("HERMES_IMAGE_BROKER_KEY="): + return line.split("=", 1)[1].strip() + except OSError: + pass + return "" + + +def _request(path: str, payload: dict[str, Any] | None, timeout: float) -> dict[str, Any]: + """Call the internal broker with a bounded authenticated JSON request.""" + key = _broker_key() + if not key: + raise RuntimeError("private image broker key is unavailable") + data = None if payload is None else json.dumps(payload).encode("utf-8") + request = Request( + f"{BROKER_URL}{path}", + data=data, + method="GET" if data is None else "POST", + headers={ + "Authorization": f"Bearer {key}", + "Content-Type": "application/json", + "Accept": "application/json", + }, + ) + try: + with urlopen(request, timeout=timeout) as response: # noqa: S310 - fixed internal URL + return json.loads(response.read()) + except HTTPError as exc: + body = exc.read().decode("utf-8", errors="replace")[:1000] + raise RuntimeError(f"image broker returned HTTP {exc.code}: {body}") from exc + except URLError as exc: + raise RuntimeError(f"image broker is unavailable: {exc.reason}") from exc + + +def _local_image_data_url(value: str) -> str: + """Convert a private uploaded image to a data URL for broker transport.""" + candidate = value.strip() + lowered = candidate.lower() + if lowered.startswith(("http://", "https://", "data:image/")): + return candidate + try: + from agent.file_safety import get_read_block_error + + blocked = get_read_block_error(candidate) + if blocked: + raise ValueError(blocked) + except ImportError: + pass + path = Path(os.path.expanduser(candidate)).resolve() + raw = path.read_bytes() + if not raw or len(raw) > MAX_INPUT_BYTES: + raise ValueError("reference image must be between 1 byte and 25 MiB") + if raw.startswith(b"\x89PNG\r\n\x1a\n"): + mime = "image/png" + elif raw.startswith(b"\xff\xd8\xff"): + mime = "image/jpeg" + elif raw.startswith((b"GIF87a", b"GIF89a")): + mime = "image/gif" + elif raw.startswith(b"RIFF") and raw[8:12] == b"WEBP": + mime = "image/webp" + else: + raise ValueError("reference image must be PNG, JPEG, GIF, or WebP") + return f"data:{mime};base64,{base64.b64encode(raw).decode('ascii')}" + + +class AtlasBrokerImageProvider(ImageGenProvider): + """High-quality GPT Image generation through the owner-only broker.""" + + @property + def name(self) -> str: + return "atlas-broker" + + @property + def display_name(self) -> str: + return "Atlas private image studio" + + def is_available(self) -> bool: + try: + result = _request("/health", None, 5.0) + return bool(result.get("success") and result.get("available")) + except Exception: + return False + + def list_models(self) -> list[dict[str, Any]]: + return [ + { + "id": model, + "display": display, + "speed": "~30s–3min", + "strengths": "OpenAI GPT Image 2 quality", + "price": "included account capacity", + } + for model, (display, _quality) in MODELS.items() + ] + + def default_model(self) -> str: + return DEFAULT_MODEL + + def capabilities(self) -> dict[str, Any]: + return {"modalities": ["text", "image"], "max_reference_images": 16} + + def generate( + self, + prompt: str, + aspect_ratio: str = DEFAULT_ASPECT_RATIO, + *, + image_url: str | None = None, + reference_image_urls: list[str] | None = None, + **kwargs: Any, + ) -> dict[str, Any]: + prompt = (prompt or "").strip() + aspect = resolve_aspect_ratio(aspect_ratio) + if not prompt: + return error_response( + error="Prompt is required", + error_type="invalid_argument", + provider=self.name, + aspect_ratio=aspect, + ) + model = str(kwargs.get("model") or DEFAULT_MODEL) + if model not in MODELS: + model = DEFAULT_MODEL + try: + primary = _local_image_data_url(image_url) if image_url else None + references = [ + _local_image_data_url(item) + for item in (normalize_reference_images(reference_image_urls) or [])[:16] + ] + response = _request( + "/v1/images/generations", + { + "prompt": prompt, + "aspect_ratio": aspect, + "model": model, + "image_url": primary, + "reference_image_urls": references, + }, + 1200.0, + ) + if not response.get("success"): + return error_response( + error=str(response.get("error") or "image generation failed"), + error_type=str(response.get("error_type") or "provider_error"), + provider=self.name, + model=model, + prompt=prompt, + aspect_ratio=aspect, + ) + image = save_b64_image( + str(response["image_b64"]), prefix=f"atlas_{model}" + ) + return success_response( + image=str(image), + model=str(response.get("model") or model), + prompt=prompt, + aspect_ratio=aspect, + provider=self.name, + modality="image" if primary or references else "text", + extra={ + "quality": response.get("quality", MODELS[model][1]), + "size": response.get("size"), + "input_image_count": response.get( + "input_image_count", len(references) + bool(primary) + ), + }, + ) + except Exception as exc: + return error_response( + error=f"Private image studio failed: {exc}", + error_type="broker_error", + provider=self.name, + model=model, + prompt=prompt, + aspect_ratio=aspect, + ) + + +def register(ctx: Any) -> None: + """Register the broker as a normal Hermes image backend.""" + ctx.register_image_gen_provider(AtlasBrokerImageProvider()) diff --git a/services/hermes/plugins/image-gen-broker/plugin.yaml b/services/hermes/plugins/image-gen-broker/plugin.yaml new file mode 100644 index 000000000..fae5bdf6c --- /dev/null +++ b/services/hermes/plugins/image-gen-broker/plugin.yaml @@ -0,0 +1,5 @@ +name: atlas-broker +version: 1.0.0 +description: "Private GPT Image broker for isolated family chat tenants." +author: bstein +kind: backend diff --git a/services/hermes/scripts/hermes_model_routing.py b/services/hermes/scripts/hermes_model_routing.py index 9b56a511a..47e3c2e46 100644 --- a/services/hermes/scripts/hermes_model_routing.py +++ b/services/hermes/scripts/hermes_model_routing.py @@ -368,10 +368,12 @@ def _profile_config( "model": primary_model, "openai_runtime": "codex_app_server", } - config["fallback_providers"] = [ - fallback, - copy.deepcopy(ATLAS_FALLBACK), - ] + config["fallback_providers"] = [fallback] + # Local OpenAI-compatible runtimes top out at the equivalent of high. + # Never silently downgrade an explicitly xhigh task after both hosted + # provider lanes are exhausted. + if effort != "xhigh": + config["fallback_providers"].append(copy.deepcopy(ATLAS_FALLBACK)) agent = config.setdefault("agent", {}) if isinstance(agent, dict): agent["reasoning_effort"] = effort @@ -481,7 +483,7 @@ def configure_routes(root: Path, codex: Catalog, claude: Catalog) -> dict[str, A ), env_values, ) - local = ["custom/gpt-oss:20b"] + local = [] if effort == "xhigh" else ["custom/gpt-oss:20b"] routes[codex_name] = [ f"openai-codex/{codex_model}", f"anthropic/{claude_model}", @@ -510,7 +512,6 @@ def configure_routes(root: Path, codex: Catalog, claude: Catalog) -> dict[str, A routes["synthesis-xhigh"] = [ f"anthropic/{claude_models['xhigh']}", f"openai-codex/{codex_models['xhigh']}", - "custom/gpt-oss:20b", ] _write_yaml( root / "profile.yaml", diff --git a/services/hermes/scripts/image_broker.py b/services/hermes/scripts/image_broker.py new file mode 100644 index 000000000..4c31fa9f9 --- /dev/null +++ b/services/hermes/scripts/image_broker.py @@ -0,0 +1,211 @@ +#!/usr/bin/env python3 +"""Private GPT Image broker for isolated Hermes chat tenants.""" + +from __future__ import annotations + +import base64 +import hmac +import importlib.util +import json +import os +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path +from typing import Any + + +HOST = os.environ.get("HERMES_IMAGE_BROKER_HOST", "0.0.0.0") +PORT = int(os.environ.get("HERMES_IMAGE_BROKER_PORT", "9002")) +TOKEN = os.environ.get("HERMES_IMAGE_BROKER_KEY", "").strip() +PROVIDER_PATH = Path( + os.environ.get( + "HERMES_IMAGE_PROVIDER_PATH", + "/opt/hermes/plugins/image_gen/openai-codex/__init__.py", + ) +) +DEFAULT_MODEL = os.environ.get( + "HERMES_IMAGE_BROKER_DEFAULT_MODEL", "gpt-image-2-high" +) +MAX_BODY_BYTES = int(os.environ.get("HERMES_IMAGE_BROKER_MAX_BODY", str(96 << 20))) +QUEUE_TIMEOUT_SECONDS = float( + os.environ.get("HERMES_IMAGE_BROKER_QUEUE_TIMEOUT", "900") +) +ALLOWED_MODELS = { + "gpt-image-2-low", + "gpt-image-2-medium", + "gpt-image-2-high", +} +ALLOWED_ASPECTS = {"landscape", "square", "portrait"} +_generation_lock = threading.Lock() + + +def _load_provider() -> Any: + """Load the pinned Hermes provider without duplicating its API client.""" + spec = importlib.util.spec_from_file_location( + "atlas_openai_codex_image_provider", PROVIDER_PATH + ) + if spec is None or spec.loader is None: + raise RuntimeError(f"cannot load image provider from {PROVIDER_PATH}") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module.OpenAICodexImageGenProvider() + + +_PROVIDER: Any | None = None + + +def _provider() -> Any: + """Load the provider once after the sidecar environment is ready.""" + global _PROVIDER + if _PROVIDER is None: + _PROVIDER = _load_provider() + return _PROVIDER + + +def _authorized(header: str | None) -> bool: + """Compare the bearer token without leaking timing information.""" + if not TOKEN or not header or not header.startswith("Bearer "): + return False + return hmac.compare_digest(header[7:].strip(), TOKEN) + + +def _generate(payload: dict[str, Any]) -> dict[str, Any]: + """Generate once, return bytes, and remove the broker-side cache copy.""" + prompt = str(payload.get("prompt") or "").strip() + if not prompt: + raise ValueError("prompt is required") + aspect = str(payload.get("aspect_ratio") or "landscape").lower() + if aspect not in ALLOWED_ASPECTS: + raise ValueError("aspect_ratio must be landscape, square, or portrait") + model = str(payload.get("model") or DEFAULT_MODEL) + if model not in ALLOWED_MODELS: + raise ValueError("unsupported image quality tier") + image_url = payload.get("image_url") + references = payload.get("reference_image_urls") + if image_url is not None and not isinstance(image_url, str): + raise ValueError("image_url must be a string") + if references is not None and not ( + isinstance(references, list) + and all(isinstance(item, str) for item in references) + ): + raise ValueError("reference_image_urls must be a list of strings") + + acquired = _generation_lock.acquire(timeout=QUEUE_TIMEOUT_SECONDS) + if not acquired: + raise TimeoutError("image generation queue is busy; retry shortly") + old_model = os.environ.get("OPENAI_IMAGE_MODEL") + try: + os.environ["OPENAI_IMAGE_MODEL"] = model + result = _provider().generate( + prompt, + aspect, + image_url=image_url, + reference_image_urls=references, + ) + finally: + if old_model is None: + os.environ.pop("OPENAI_IMAGE_MODEL", None) + else: + os.environ["OPENAI_IMAGE_MODEL"] = old_model + _generation_lock.release() + + if not result.get("success"): + return result + image_path = Path(str(result["image"])) + try: + raw = image_path.read_bytes() + finally: + image_path.unlink(missing_ok=True) + response = {key: value for key, value in result.items() if key != "image"} + response.update( + { + "image_b64": base64.b64encode(raw).decode("ascii"), + "mime_type": "image/png", + } + ) + return response + + +class Handler(BaseHTTPRequestHandler): + """Small authenticated JSON API; prompts and images are never logged.""" + + server_version = "HermesImageBroker/1" + + def _json(self, status: int, value: dict[str, Any]) -> None: + body = json.dumps(value, separators=(",", ":")).encode("utf-8") + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.send_header("Cache-Control", "no-store") + self.end_headers() + self.wfile.write(body) + + def _check_auth(self) -> bool: + if _authorized(self.headers.get("Authorization")): + return True + self._json(401, {"success": False, "error": "unauthorized"}) + return False + + def do_GET(self) -> None: # noqa: N802 - BaseHTTPRequestHandler API + if self.path != "/health": + self._json(404, {"success": False, "error": "not found"}) + return + if not self._check_auth(): + return + self._json( + 200, + { + "success": True, + "provider": "openai-codex", + "available": bool(_provider().is_available()), + "busy": _generation_lock.locked(), + }, + ) + + def do_POST(self) -> None: # noqa: N802 - BaseHTTPRequestHandler API + if self.path != "/v1/images/generations": + self._json(404, {"success": False, "error": "not found"}) + return + if not self._check_auth(): + return + try: + length = int(self.headers.get("Content-Length", "0")) + except ValueError: + length = 0 + if length <= 0 or length > MAX_BODY_BYTES: + self._json(413, {"success": False, "error": "invalid request size"}) + return + try: + payload = json.loads(self.rfile.read(length)) + if not isinstance(payload, dict): + raise ValueError("JSON object required") + result = _generate(payload) + self._json(200 if result.get("success") else 502, result) + except ValueError as exc: + self._json(400, {"success": False, "error": str(exc)}) + except TimeoutError as exc: + self._json(503, {"success": False, "error": str(exc)}) + except Exception as exc: # keep provider details useful but omit prompts + self._json( + 502, + { + "success": False, + "error": f"image provider failed: {type(exc).__name__}: {exc}", + }, + ) + + def log_message(self, format: str, *args: Any) -> None: + """Log only method/path/status metadata, never request bodies.""" + print(f"image-broker {self.address_string()} {format % args}", flush=True) + + +def main() -> None: + """Serve until Kubernetes terminates the sidecar.""" + if not TOKEN: + raise SystemExit("HERMES_IMAGE_BROKER_KEY is required") + server = ThreadingHTTPServer((HOST, PORT), Handler) + server.serve_forever() + + +if __name__ == "__main__": + main() diff --git a/services/hermes/scripts/patch_codex_runtime.py b/services/hermes/scripts/patch_codex_runtime.py index 6f8e9c979..527f36fb3 100644 --- a/services/hermes/scripts/patch_codex_runtime.py +++ b/services/hermes/scripts/patch_codex_runtime.py @@ -173,6 +173,35 @@ FALLBACK_DISPATCH_AFTER = ''' while (api_call_count < agent.max_iterations an # Reset per-turn checkpoint dedup so each iteration can take one snapshot ''' +RETRY_FALLBACK_DISPATCH_BEFORE = ''' while retry_count < max_retries: + # ── Nous Portal rate limit guard ────────────────────── +''' +RETRY_FALLBACK_DISPATCH_AFTER = ''' while retry_count < max_retries: + # Fallback activation happens inside this retry loop. Dispatch a + # newly selected Codex app-server route before the next iteration + # tries to build OpenAI-compatible kwargs from the intentionally + # absent bearer-token client. + if agent.api_mode == "codex_app_server": + fallback_user_message = user_message + if getattr(agent, "_codex_cross_provider_fallback", False): + from agent.codex_runtime import build_cross_provider_codex_prompt + + fallback_user_message = build_cross_provider_codex_prompt( + messages, + user_message, + ) + agent._codex_cross_provider_fallback = False + return agent._run_codex_app_server_turn( + user_message=fallback_user_message, + original_user_message=original_user_message, + messages=messages, + effective_task_id=effective_task_id, + should_review_memory=_should_review_memory, + ) + + # ── Nous Portal rate limit guard ────────────────────── +''' + FALLBACK_CONTEXT_BEFORE = '''def run_codex_app_server_turn( ''' FALLBACK_CONTEXT_AFTER = '''def build_cross_provider_codex_prompt( @@ -343,16 +372,20 @@ def patch_fallback(source: Path, destination: Path) -> None: def patch_loop(source: Path, destination: Path) -> None: """Dispatch a mid-turn Codex fallback through app-server.""" content = source.read_text(encoding="utf-8") - destination.parent.mkdir(parents=True, exist_ok=True) - destination.write_text( - _replace_once( - content, - FALLBACK_DISPATCH_BEFORE, - FALLBACK_DISPATCH_AFTER, - "Codex fallback dispatch", - ), - encoding="utf-8", + content = _replace_once( + content, + FALLBACK_DISPATCH_BEFORE, + FALLBACK_DISPATCH_AFTER, + "Codex outer-loop fallback dispatch", ) + content = _replace_once( + content, + RETRY_FALLBACK_DISPATCH_BEFORE, + RETRY_FALLBACK_DISPATCH_AFTER, + "Codex retry-loop fallback dispatch", + ) + destination.parent.mkdir(parents=True, exist_ok=True) + destination.write_text(content, encoding="utf-8") def patch_auxiliary(source: Path, destination: Path) -> None: diff --git a/services/hermes/service.yaml b/services/hermes/service.yaml index c5974182c..bf1a078da 100644 --- a/services/hermes/service.yaml +++ b/services/hermes/service.yaml @@ -77,6 +77,23 @@ spec: --- apiVersion: v1 kind: Service +metadata: + name: hermes-image-broker + namespace: hermes + labels: + app: hermes-agent +spec: + type: ClusterIP + selector: + app: hermes-agent + ports: + - name: http + port: 9002 + targetPort: image-broker + protocol: TCP +--- +apiVersion: v1 +kind: Service metadata: name: hermes-ollama namespace: hermes diff --git a/services/hermes/voice-deployment.yaml b/services/hermes/voice-deployment.yaml index db35a68b6..721b8df62 100644 --- a/services/hermes/voice-deployment.yaml +++ b/services/hermes/voice-deployment.yaml @@ -115,12 +115,12 @@ spec: annotations: ai.bstein.dev/role: private-chat-text-to-speech ai.bstein.dev/model: piper-en-us-lessac-medium - ai.bstein.dev/gpu: CPU-only on routing node + ai.bstein.dev/gpu: CPU-only beside Whisper on the voice node spec: automountServiceAccountToken: false enableServiceLinks: false nodeSelector: - kubernetes.io/hostname: titan-20 + kubernetes.io/hostname: titan-21 containers: - name: tts image: registry.bstein.dev/bstein/hermes-jetson-tts@sha256:5cb9e57faab46365bff606c559af57b9505892be2909aea1ac523568d478b2cc diff --git a/services/vault/hermes-auth-role-bootstrap-job.yaml b/services/vault/hermes-auth-role-bootstrap-job.yaml index dc930330a..765b1b354 100644 --- a/services/vault/hermes-auth-role-bootstrap-job.yaml +++ b/services/vault/hermes-auth-role-bootstrap-job.yaml @@ -3,7 +3,7 @@ apiVersion: batch/v1 kind: Job metadata: - name: vault-k8s-auth-hermes-4 + name: vault-k8s-auth-hermes-5 namespace: vault spec: backoffLimit: 2 diff --git a/services/vault/scripts/vault_k8s_auth_configure.sh b/services/vault/scripts/vault_k8s_auth_configure.sh index 084de6f46..356088017 100644 --- a/services/vault/scripts/vault_k8s_auth_configure.sh +++ b/services/vault/scripts/vault_k8s_auth_configure.sh @@ -256,7 +256,7 @@ write_policy_and_role "game-stream" "game-stream" "game-stream-vault" \ write_policy_and_role "hermes" "hermes" "hermes-vault,hermes-triage" \ "hermes/triage-oidc hermes/agent-tokens" "" write_policy_and_role "hermes-agent" "hermes" "hermes-agent" \ - "hermes/agent-oidc hermes/agent-tokens" "" + "hermes/agent-oidc hermes/agent-tokens hermes/chat-telegram" "" write_policy_and_role "hermes-chat" "hermes" "hermes-chat" \ "hermes/chat-oidc hermes/chat-telegram hermes/agent-tokens" "" write_policy_and_role "veles" "veles" "veles-backend,veles-generator,veles-postgres,veles-vault-sync" \ diff --git a/testing/tests/test_hermes_chat_quality.py b/testing/tests/test_hermes_chat_quality.py index d2da21a4f..bad07b9a9 100644 --- a/testing/tests/test_hermes_chat_quality.py +++ b/testing/tests/test_hermes_chat_quality.py @@ -12,6 +12,7 @@ import yaml ROOT = Path(__file__).parents[2] HERMES = ROOT / "services" / "hermes" +VAULT = ROOT / "services" / "vault" def _documents(path: Path) -> list[dict]: @@ -39,6 +40,7 @@ def test_chat_config_enables_real_research_compute_and_delegation(): assert "python_sandbox" in toolsets assert "vision" in toolsets assert "web" in toolsets + assert "image_gen" in toolsets assert "terminal" not in toolsets assert "code_execution" not in toolsets @@ -277,7 +279,7 @@ def test_voice_workloads_have_deliberate_xavier_placement(): assert "@sha256:" in stt["containers"][0]["image"] assert "@sha256:" in tts["containers"][0]["image"] assert stt["nodeSelector"] == {"kubernetes.io/hostname": "titan-21"} - assert tts["nodeSelector"] == {"kubernetes.io/hostname": "titan-20"} + assert tts["nodeSelector"] == {"kubernetes.io/hostname": "titan-21"} assert stt["automountServiceAccountToken"] is False assert tts["automountServiceAccountToken"] is False assert stt["enableServiceLinks"] is False @@ -302,6 +304,97 @@ def test_voice_workloads_have_deliberate_xavier_placement(): assert all("hostPath" not in volume for volume in tts["volumes"]) +def test_chat_image_generation_uses_private_owner_broker(): + """Family pods get image bytes without receiving the owner's OAuth file.""" + configmap = _documents(HERMES / "chat-configmap.yaml")[0] + config = yaml.safe_load(configmap["data"]["config.yaml"]) + assert config["image_gen"] == { + "provider": "atlas-broker", + "model": "gpt-image-2-high", + } + + statefulset = _documents(HERMES / "chat-statefulset.yaml")[0] + pod = statefulset["spec"]["template"]["spec"] + hermes = next(item for item in pod["containers"] if item["name"] == "hermes") + mounts = {item["name"]: item for item in hermes["volumeMounts"]} + assert mounts["image-plugin"]["mountPath"] == ( + "/opt/hermes/plugins/image_gen/atlas-broker" + ) + assert mounts["provider-auth"].get("readOnly") is not True + assert not any(mount["name"] == "home" and "agent" in str(mount) for mount in hermes["volumeMounts"]) + env = {item["name"]: item["value"] for item in hermes["env"]} + assert env["HERMES_IMAGE_BROKER_URL"].startswith("http://hermes-image-broker.") + + agent = _documents(HERMES / "agent-deployment.yaml")[0] + containers = agent["spec"]["template"]["spec"]["containers"] + broker = next(item for item in containers if item["name"] == "image-broker") + assert broker["ports"] == [ + {"name": "image-broker", "containerPort": 9002, "protocol": "TCP"} + ] + assert broker["securityContext"]["readOnlyRootFilesystem"] is True + assert broker["securityContext"]["runAsNonRoot"] is True + + services = _documents(HERMES / "service.yaml") + service = next( + item for item in services if item["metadata"]["name"] == "hermes-image-broker" + ) + assert service["spec"]["selector"] == {"app": "hermes-agent"} + + policies = _documents(HERMES / "networkpolicy.yaml") + agent_policy = next( + item for item in policies if item["metadata"]["name"] == "hermes-agent-isolation" + ) + image_ingress = next( + rule + for rule in agent_policy["spec"]["ingress"] + if rule["ports"] == [{"protocol": "TCP", "port": 9002}] + ) + assert image_ingress["from"][0]["podSelector"]["matchLabels"] == { + "app": "hermes-chat-tenant" + } + vault_policy = (VAULT / "scripts" / "vault_k8s_auth_configure.sh").read_text() + assert ( + '"hermes/agent-oidc hermes/agent-tokens hermes/chat-telegram"' + in vault_policy + ) + + +def test_image_broker_returns_bytes_and_removes_owner_cache(tmp_path: Path, monkeypatch): + """The broker must not retain a family user's generated image.""" + broker_path = HERMES / "scripts" / "image_broker.py" + spec = importlib.util.spec_from_file_location("hermes_image_broker", broker_path) + assert spec and spec.loader + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + generated = tmp_path / "generated.png" + generated.write_bytes(b"\x89PNG\r\n\x1a\nprivate-image") + + class Provider: + def generate(self, prompt, aspect, **kwargs): + assert prompt == "paint a blue sphere" + assert aspect == "square" + return { + "success": True, + "image": str(generated), + "model": "gpt-image-2-high", + "quality": "high", + } + + monkeypatch.setattr(module, "_PROVIDER", Provider()) + result = module._generate( + { + "prompt": "paint a blue sphere", + "aspect_ratio": "square", + "model": "gpt-image-2-high", + } + ) + + assert result["success"] is True + assert result["image_b64"] + assert "image" not in result + assert not generated.exists() + + def test_chat_auth_file_mount_survives_atomic_provider_refresh(): statefulset = _documents(HERMES / "chat-statefulset.yaml")[0] containers = statefulset["spec"]["template"]["spec"]["containers"] @@ -320,6 +413,22 @@ def test_chat_auth_file_mount_survives_atomic_provider_refresh(): item["name"]: item["value"] for item in next(item for item in containers if item["name"] == "hermes")["env"] } + hermes_mount = next( + item + for item in next( + item for item in containers if item["name"] == "hermes" + )["volumeMounts"] + if item["name"] == "provider-auth" + ) + assert hermes_mount.get("readOnly") is not True + webui_mount = next( + item + for item in next( + item for item in containers if item["name"] == "webui" + )["volumeMounts"] + if item["name"] == "provider-auth" + ) + assert webui_mount["readOnly"] is True assert hermes_env["AGENT_BROWSER_EXECUTABLE_PATH"].endswith("/chrome-linux/headless_shell") assert "--no-sandbox" in hermes_env["AGENT_BROWSER_ARGS"] diff --git a/testing/tests/test_hermes_cli_lanes.py b/testing/tests/test_hermes_cli_lanes.py index 5d91cd007..ad8778f36 100644 --- a/testing/tests/test_hermes_cli_lanes.py +++ b/testing/tests/test_hermes_cli_lanes.py @@ -789,7 +789,7 @@ def test_agent_auth_is_bstein_group_and_email_bounded(): assert '"full.path":"true"' in script -def test_agent_network_boundary_allows_only_authenticated_web_surfaces(): +def test_agent_network_boundary_allows_only_authenticated_web_and_image_broker_surfaces(): documents = [ item for item in yaml.safe_load_all((HERMES / "networkpolicy.yaml").read_text()) @@ -811,7 +811,17 @@ def test_agent_network_boundary_allows_only_authenticated_web_surfaces(): } ], "ports": [{"protocol": "TCP", "port": 4180}], - } + }, + { + "from": [ + { + "podSelector": { + "matchLabels": {"app": "hermes-chat-tenant"} + } + } + ], + "ports": [{"protocol": "TCP", "port": 9002}], + }, ] assert isolation["spec"]["egress"] == [{}] @@ -950,12 +960,21 @@ def test_codex_runtime_patch_uses_cli_and_forwards_route(tmp_path: Path): assert "agent._codex_cross_provider_fallback = True" in fallback_content loop = tmp_path / "conversation_loop.py" - loop.write_text(codex_runtime_patch.FALLBACK_DISPATCH_BEFORE, encoding="utf-8") + loop.write_text( + codex_runtime_patch.FALLBACK_DISPATCH_BEFORE + + codex_runtime_patch.RETRY_FALLBACK_DISPATCH_BEFORE, + encoding="utf-8", + ) loop_out = tmp_path / "patched/conversation_loop.py" codex_runtime_patch.patch_loop(loop, loop_out) loop_content = loop_out.read_text() - assert 'if agent.api_mode == "codex_app_server"' in loop_content + assert loop_content.count('if agent.api_mode == "codex_app_server"') == 2 assert "build_cross_provider_codex_prompt" in loop_content + retry_dispatch = loop_content.index( + "Fallback activation happens inside this retry loop" + ) + api_kwargs = loop_content.find("agent._build_api_kwargs", retry_dispatch) + assert api_kwargs == -1 or retry_dispatch < api_kwargs auxiliary = tmp_path / "auxiliary_client.py" auxiliary.write_text( diff --git a/testing/tests/test_hermes_coordinator.py b/testing/tests/test_hermes_coordinator.py index 6bf9c0886..1d0bd3917 100644 --- a/testing/tests/test_hermes_coordinator.py +++ b/testing/tests/test_hermes_coordinator.py @@ -106,6 +106,9 @@ def test_configure_routes_builds_cross_provider_fallback_profiles(tmp_path: Path codex_profile = yaml.safe_load( (tmp_path / "profiles/codex-high/config.yaml").read_text(encoding="utf-8") ) + codex_xhigh_profile = yaml.safe_load( + (tmp_path / "profiles/codex-xhigh/config.yaml").read_text(encoding="utf-8") + ) claude_profile = yaml.safe_load( (tmp_path / "profiles/claude-high/config.yaml").read_text(encoding="utf-8") ) @@ -123,6 +126,18 @@ def test_configure_routes_builds_cross_provider_fallback_profiles(tmp_path: Path assert codex_profile["agent"]["reasoning_effort"] == "high" assert codex_profile["fallback_providers"][1] == routing.ATLAS_FALLBACK assert len(codex_profile["fallback_providers"]) == 2 + assert codex_xhigh_profile["fallback_providers"] == [ + {"provider": "anthropic", "model": "claude-opus-5"} + ] + assert routes["codex-xhigh"] == [ + "openai-codex/gpt-5.6-sol", + "anthropic/claude-opus-5", + ] + assert routes["claude-xhigh"] == [ + "anthropic/claude-opus-5", + "openai-codex/gpt-5.6-sol", + ] + assert all("custom/" not in route for route in routes["synthesis-xhigh"]) assert routes["coordinator"][0] == "openai-codex/gpt-5.6-terra" assert "custom/qwen2.5:14b-instruct-q4_0" not in routes["coordinator"] assert "max" not in json.dumps(routes)