diff --git a/dockerfiles/Dockerfile.hermes-jetson-stt b/dockerfiles/Dockerfile.hermes-jetson-stt
new file mode 100644
index 000000000..0e4c6db96
--- /dev/null
+++ b/dockerfiles/Dockerfile.hermes-jetson-stt
@@ -0,0 +1,37 @@
+# syntax=docker/dockerfile:1
+# dockerfiles/Dockerfile.hermes-jetson-stt
+FROM dustynv/whisper@sha256:725c23a68ace3ee667b6465468c38d90e6a0cda9891923a7dd98b851d557569e
+
+USER root
+
+# The JetPack 5 base supplies the CUDA-enabled PyTorch build. Keep Whisper new
+# enough for the multilingual turbo model while preserving that Jetson stack.
+RUN python3 -m pip install --no-cache-dir --force-reinstall --no-deps \
+ openai-whisper==20250625
+
+# Keep model acquisition in the audited image build. Runtime pods never need
+# public egress and Whisper verifies the same digest when loading this cache.
+ADD --checksum=sha256:aff26ae408abcba5fbf8813c21e62b0941638c5f6eebfb145be0c9839262a19a --chmod=0444 \
+ https://openaipublic.azureedge.net/main/whisper/models/aff26ae408abcba5fbf8813c21e62b0941638c5f6eebfb145be0c9839262a19a/large-v3-turbo.pt \
+ /opt/models/whisper/large-v3-turbo.pt
+
+COPY dockerfiles/hermes-jetson-stt-server.py /opt/atlas/hermes-jetson-stt-server.py
+RUN chmod 0555 /opt/atlas/hermes-jetson-stt-server.py
+
+# The vendor image defaults to /opt/whisper, whose 2023 source checkout would
+# shadow the pinned Python package. Run Atlas code from its own directory.
+WORKDIR /opt/atlas
+
+# Import the Xavier CUDA stack and confirm Whisper resolves the baked artifact.
+# Full GPU warm-up is covered by the Kubernetes startup probe on titan-21.
+RUN python3 -c "from pathlib import Path; import torch, whisper; print(whisper.__file__, whisper.__version__, whisper.available_models()); assert 'large-v3-turbo' in whisper.available_models(); assert Path('/opt/models/whisper/large-v3-turbo.pt').is_file(); print(torch.__version__)"
+
+ENV HERMES_STT_HOST=0.0.0.0 \
+ HERMES_STT_PORT=9000 \
+ HERMES_STT_MODEL=large-v3-turbo \
+ HERMES_STT_CACHE=/opt/models/whisper \
+ PYTHONDONTWRITEBYTECODE=1 \
+ PYTHONUNBUFFERED=1
+
+EXPOSE 9000
+ENTRYPOINT ["python3", "/opt/atlas/hermes-jetson-stt-server.py"]
diff --git a/dockerfiles/Dockerfile.hermes-jetson-tts b/dockerfiles/Dockerfile.hermes-jetson-tts
new file mode 100644
index 000000000..abd29cbec
--- /dev/null
+++ b/dockerfiles/Dockerfile.hermes-jetson-tts
@@ -0,0 +1,32 @@
+# syntax=docker/dockerfile:1
+# dockerfiles/Dockerfile.hermes-jetson-tts
+FROM python:3.11-slim-bookworm@sha256:d29f48a31a8b408ed19272ca1e7b10ebae13b240a27e862d3d4217c528e2e0c3
+
+RUN python -m pip install --no-cache-dir piper-tts==1.5.0
+
+# Pin the voice data in the image so synthesis has no runtime dependency on
+# Hugging Face or mutable model metadata.
+ADD --checksum=sha256:4cabf7c3a638017137f34a1516522032d4fe3f38228a843cc9b764ddcbcd9e09 --chmod=0444 \
+ https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/en/en_US/lessac/high/en_US-lessac-high.onnx?download=true \
+ /opt/models/piper/en_US-lessac-high.onnx
+ADD --checksum=sha256:db42b97d9859f257bc1561b8ed980e7fb2398402050a74ddd6cbec931a92412f --chmod=0444 \
+ https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/en/en_US/lessac/high/en_US-lessac-high.onnx.json?download=true \
+ /opt/models/piper/en_US-lessac-high.onnx.json
+
+COPY dockerfiles/hermes-jetson-tts-server.py /opt/atlas/hermes-jetson-tts-server.py
+RUN chmod 0555 /opt/atlas/hermes-jetson-tts-server.py
+
+# Load the actual pinned voice during the ARM64 build. This catches package or
+# model-format drift before the image can reach Flux.
+RUN python -c "from pathlib import Path; from piper import PiperVoice; p=Path('/opt/models/piper'); v=PiperVoice.load(p/'en_US-lessac-high.onnx', p/'en_US-lessac-high.onnx.json', use_cuda=False, download_dir=p); assert v.config.sample_rate > 0"
+
+ENV HERMES_TTS_HOST=0.0.0.0 \
+ HERMES_TTS_PORT=9001 \
+ HERMES_TTS_VOICE=en_US-lessac-high \
+ HERMES_TTS_CACHE=/opt/models/piper \
+ OMP_NUM_THREADS=2 \
+ PYTHONDONTWRITEBYTECODE=1 \
+ PYTHONUNBUFFERED=1
+
+EXPOSE 9001
+ENTRYPOINT ["python", "/opt/atlas/hermes-jetson-tts-server.py"]
diff --git a/dockerfiles/Dockerfile.hermes-webui b/dockerfiles/Dockerfile.hermes-webui
index 338ed40de..eae112213 100644
--- a/dockerfiles/Dockerfile.hermes-webui
+++ b/dockerfiles/Dockerfile.hermes-webui
@@ -2,7 +2,7 @@
# dockerfiles/Dockerfile.hermes-webui
FROM ghcr.io/nesquena/hermes-webui@sha256:a83a3893111dcb250e7aa7aa657d3d6f4570b0e2fd00d9b7569246fc5e7339b2 AS webui
-FROM registry.bstein.dev/bstein/hermes-agent@sha256:15c5c538c0b58686af2e54e10bc870b23284789d485a609349df24ed3053622f
+FROM registry.bstein.dev/bstein/hermes-agent@sha256:a09d36b7467d5810bd814b05d72004629695850d188a2c2a6041af8e2539ba08
USER root
@@ -29,12 +29,71 @@ before = '
Max
\
if before not in source:
raise SystemExit("Hermes WebUI xhigh UI patch context changed")
index.write_text(source.replace(before, "", 1), encoding="utf-8")
+
+# oauth2-proxy returns 401 for browser API and health probes when the secure
+# session expires. Re-enter OIDC with the complete return path instead of
+# presenting an endless, inaccurate "connection lost" loop.
+ui = Path("/opt/hermes-webui/static/ui.js")
+source = ui.read_text(encoding="utf-8")
+before = ''' const res=await fetcher(_offlineHealthUrl(),opts);
+ return !!(res&&res.ok);
+'''
+after = ''' const res=await fetcher(_offlineHealthUrl(),opts);
+ if(res&&(res.status===401||res.status===403)){
+ const rd=window.location.pathname+window.location.search+window.location.hash;
+ window.location.assign('/oauth2/start?rd='+encodeURIComponent(rd));
+ return false;
+ }
+ return !!(res&&res.ok);
+'''
+if source.count(before) != 1:
+ raise SystemExit("Hermes WebUI auth-recovery patch context changed")
+ui.write_text(source.replace(before, after, 1), encoding="utf-8")
+
+# Make delegated session hierarchy obvious and collapsible in the sidebar.
+sessions = Path("/opt/hermes-webui/static/sessions.js")
+source = sessions.read_text(encoding="utf-8")
+before = ''' const childLabel=t('session_meta_children', childCount);
+ childCountEl.textContent=childLabel;
+ childCountEl.title=_sessionChildBadgeTooltip(childLabel);
+'''
+after = ''' const childLabel=t('session_meta_children', childCount);
+ const childrenExpanded=_expandedChildSessionKeys.has(lineageKey)||!!searchQueryRaw;
+ childCountEl.textContent=(childrenExpanded?'▾ ':'▸ ')+childLabel;
+ childCountEl.setAttribute('aria-expanded',childrenExpanded?'true':'false');
+ childCountEl.title=_sessionChildBadgeTooltip(childLabel);
+'''
+if source.count(before) != 1:
+ raise SystemExit("Hermes WebUI child-session toggle patch context changed")
+sessions.write_text(source.replace(before, after, 1), encoding="utf-8")
+
+# A profile's model is only its default; a session-level selector can override
+# it. Label the scope so the dropdown does not contradict the effective model.
+panels = Path("/opt/hermes-webui/static/panels.js")
+source = panels.read_text(encoding="utf-8")
+before = " if (typeof p.model === 'string' && p.model) meta.push(p.model.split('/').pop());\n"
+after = " if (typeof p.model === 'string' && p.model) meta.push('profile default: ' + p.model.split('/').pop());\n"
+if source.count(before) != 2:
+ raise SystemExit("Hermes WebUI profile-model label patch context changed")
+panels.write_text(source.replace(before, after, 2), encoding="utf-8")
PY
+# Add the Atlas voice bridge as a narrow integration layer. It activates only
+# when a tenant's server-side STT capability reports the private Jetson route.
+COPY dockerfiles/hermes-webui-atlas-patch.py /tmp/hermes-webui-atlas-patch.py
+COPY dockerfiles/hermes-webui-atlas-voice.js /opt/hermes-webui/static/atlas-voice.js
+RUN /opt/hermes/.venv/bin/python /tmp/hermes-webui-atlas-patch.py
+
RUN /opt/hermes/.venv/bin/python -c 'import cryptography, yaml' \
&& grep -Fq 'VALID_REASONING_EFFORTS = ("minimal", "low", "medium", "high", "xhigh")' \
/opt/hermes-webui/api/config.py \
- && ! grep -Fq 'data-effort="max"' /opt/hermes-webui/static/index.html
+ && ! grep -Fq 'data-effort="max"' /opt/hermes-webui/static/index.html \
+ && grep -Fq "window.location.assign('/oauth2/start?rd='" /opt/hermes-webui/static/ui.js \
+ && grep -Fq "childrenExpanded?'▾ ':'▸ '" /opt/hermes-webui/static/sessions.js \
+ && grep -Fq "profile default: ' + p.model" /opt/hermes-webui/static/panels.js \
+ && grep -Fq 'Atlas Jetson (private)' /opt/hermes-webui/static/index.html \
+ && grep -Fq 'HERMES_WEBUI_ATLAS_TTS_URL' /opt/hermes-webui/api/routes.py \
+ && grep -Fq "capability.provider!=='local_command'" /opt/hermes-webui/static/atlas-voice.js
# Exercise the real server process in the target architecture before publish.
RUN set -eu; \
diff --git a/dockerfiles/hermes-jetson-stt-server.py b/dockerfiles/hermes-jetson-stt-server.py
new file mode 100644
index 000000000..91550be88
--- /dev/null
+++ b/dockerfiles/hermes-jetson-stt-server.py
@@ -0,0 +1,136 @@
+#!/usr/bin/env python3
+"""Small OpenAI-compatible Whisper service for the dedicated Jetson."""
+
+from __future__ import annotations
+
+import cgi
+import json
+import os
+import tempfile
+import threading
+from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
+from pathlib import Path
+
+import torch
+import whisper
+
+
+HOST = os.getenv("HERMES_STT_HOST", "0.0.0.0")
+PORT = int(os.getenv("HERMES_STT_PORT", "9000"))
+MODEL_NAME = os.getenv("HERMES_STT_MODEL", "large-v3-turbo")
+CACHE_DIR = Path(os.getenv("HERMES_STT_CACHE", "/cache/whisper"))
+MAX_AUDIO_BYTES = 30 * 1024 * 1024
+MODEL_LOCK = threading.Lock()
+
+
+def _json(handler: BaseHTTPRequestHandler, status: int, payload: dict) -> None:
+ body = json.dumps(payload).encode("utf-8")
+ handler.send_response(status)
+ handler.send_header("Content-Type", "application/json")
+ handler.send_header("Content-Length", str(len(body)))
+ handler.send_header("Cache-Control", "no-store")
+ handler.end_headers()
+ handler.wfile.write(body)
+
+
+class SpeechHandler(BaseHTTPRequestHandler):
+ """Serve health and transcription without exposing a general runtime."""
+
+ server_version = "AtlasWhisper/1"
+
+ def log_message(self, message: str, *args: object) -> None:
+ print(f"[stt] {self.address_string()} {message % args}", flush=True)
+
+ def do_GET(self) -> None:
+ if self.path != "/health":
+ _json(self, 404, {"error": "not found"})
+ return
+ _json(
+ self,
+ 200,
+ {
+ "ok": True,
+ "model": MODEL_NAME,
+ "device": "cuda" if torch.cuda.is_available() else "cpu",
+ },
+ )
+
+ def do_POST(self) -> None:
+ if self.path != "/v1/audio/transcriptions":
+ _json(self, 404, {"error": "not found"})
+ return
+ content_length = int(self.headers.get("Content-Length", "0") or "0")
+ if content_length <= 0 or content_length > MAX_AUDIO_BYTES:
+ _json(self, 413, {"error": "audio payload is missing or too large"})
+ return
+
+ content_type = self.headers.get("Content-Type", "")
+ if not content_type.lower().startswith("multipart/form-data"):
+ _json(self, 400, {"error": "multipart/form-data is required"})
+ return
+
+ form = cgi.FieldStorage(
+ fp=self.rfile,
+ headers=self.headers,
+ environ={
+ "REQUEST_METHOD": "POST",
+ "CONTENT_TYPE": content_type,
+ "CONTENT_LENGTH": str(content_length),
+ },
+ )
+ audio = form["file"] if "file" in form else None
+ if audio is None or not getattr(audio, "file", None):
+ _json(self, 400, {"error": "file is required"})
+ return
+
+ suffix = Path(getattr(audio, "filename", "audio.wav") or "audio.wav").suffix
+ suffix = suffix if suffix in {".wav", ".webm", ".ogg", ".mp3", ".m4a"} else ".wav"
+ language = str(form.getfirst("language", "auto") or "auto").strip().lower()
+ temp_path = ""
+ try:
+ with tempfile.NamedTemporaryFile(prefix="atlas-stt-", suffix=suffix, delete=False) as temp:
+ temp_path = temp.name
+ while True:
+ chunk = audio.file.read(1024 * 1024)
+ if not chunk:
+ break
+ temp.write(chunk)
+
+ with MODEL_LOCK:
+ result = self.server.model.transcribe( # type: ignore[attr-defined]
+ temp_path,
+ language=None if language in {"", "auto"} else language,
+ task="transcribe",
+ fp16=torch.cuda.is_available(),
+ condition_on_previous_text=False,
+ temperature=0,
+ verbose=False,
+ )
+ transcript = str(result.get("text") or "").strip()
+ _json(self, 200, {"text": transcript, "model": MODEL_NAME})
+ except Exception as exc:
+ print(f"[stt] transcription failed: {exc}", flush=True)
+ _json(self, 500, {"error": "transcription failed"})
+ finally:
+ if temp_path:
+ try:
+ os.unlink(temp_path)
+ except OSError:
+ pass
+
+
+def main() -> None:
+ """Warm Whisper once, then serve concurrent clients through one GPU lock."""
+ CACHE_DIR.mkdir(parents=True, exist_ok=True)
+ if not torch.cuda.is_available():
+ raise RuntimeError("CUDA is required for the Atlas Whisper service")
+ print(f"[stt] loading Whisper {MODEL_NAME} into CUDA", flush=True)
+ model = whisper.load_model(MODEL_NAME, device="cuda", download_root=str(CACHE_DIR))
+ server = ThreadingHTTPServer((HOST, PORT), SpeechHandler)
+ server.model = model # type: ignore[attr-defined]
+ print(f"[stt] ready on {HOST}:{PORT}", flush=True)
+ server.serve_forever(poll_interval=0.25)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/dockerfiles/hermes-jetson-tts-server.py b/dockerfiles/hermes-jetson-tts-server.py
new file mode 100644
index 000000000..c5ed37799
--- /dev/null
+++ b/dockerfiles/hermes-jetson-tts-server.py
@@ -0,0 +1,108 @@
+#!/usr/bin/env python3
+"""CPU-only Piper service shared with the routing Jetson."""
+
+from __future__ import annotations
+
+import io
+import json
+import os
+import threading
+import wave
+from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
+from pathlib import Path
+
+from piper import PiperVoice, SynthesisConfig
+
+
+HOST = os.getenv("HERMES_TTS_HOST", "0.0.0.0")
+PORT = int(os.getenv("HERMES_TTS_PORT", "9001"))
+VOICE_NAME = os.getenv("HERMES_TTS_VOICE", "en_US-lessac-high")
+CACHE_DIR = Path(os.getenv("HERMES_TTS_CACHE", "/cache/piper"))
+MAX_TEXT_CHARS = 5000
+VOICE_LOCK = threading.Lock()
+
+
+def _json(handler: BaseHTTPRequestHandler, status: int, payload: dict) -> None:
+ body = json.dumps(payload).encode("utf-8")
+ handler.send_response(status)
+ handler.send_header("Content-Type", "application/json")
+ handler.send_header("Content-Length", str(len(body)))
+ handler.send_header("Cache-Control", "no-store")
+ handler.end_headers()
+ handler.wfile.write(body)
+
+
+class SpeechHandler(BaseHTTPRequestHandler):
+ """Serve health and bounded local speech synthesis."""
+
+ server_version = "AtlasPiper/1"
+
+ def log_message(self, message: str, *args: object) -> None:
+ print(f"[tts] {self.address_string()} {message % args}", flush=True)
+
+ def do_GET(self) -> None:
+ if self.path != "/health":
+ _json(self, 404, {"error": "not found"})
+ return
+ _json(self, 200, {"ok": True, "voice": VOICE_NAME, "device": "cpu"})
+
+ def do_POST(self) -> None:
+ if self.path != "/v1/audio/speech":
+ _json(self, 404, {"error": "not found"})
+ return
+ content_length = int(self.headers.get("Content-Length", "0") or "0")
+ if content_length <= 0 or content_length > 64 * 1024:
+ _json(self, 413, {"error": "request is missing or too large"})
+ return
+ try:
+ payload = json.loads(self.rfile.read(content_length).decode("utf-8"))
+ text = str(payload.get("input") or payload.get("text") or "").strip()
+ speed = float(payload.get("speed") or 1.0)
+ except (TypeError, ValueError, json.JSONDecodeError):
+ _json(self, 400, {"error": "invalid JSON request"})
+ return
+ if not text:
+ _json(self, 400, {"error": "input is required"})
+ return
+ if len(text) > MAX_TEXT_CHARS:
+ _json(self, 400, {"error": "input is too long"})
+ return
+ speed = min(2.0, max(0.5, speed))
+
+ output = io.BytesIO()
+ try:
+ with VOICE_LOCK, wave.open(output, "wb") as wav_file:
+ self.server.voice.synthesize_wav( # type: ignore[attr-defined]
+ text,
+ wav_file,
+ SynthesisConfig(length_scale=1.0 / speed),
+ )
+ audio = output.getvalue()
+ self.send_response(200)
+ self.send_header("Content-Type", "audio/wav")
+ self.send_header("Content-Length", str(len(audio)))
+ self.send_header("Cache-Control", "no-store")
+ self.end_headers()
+ self.wfile.write(audio)
+ except Exception as exc:
+ print(f"[tts] synthesis failed: {exc}", flush=True)
+ _json(self, 500, {"error": "speech synthesis failed"})
+
+
+def main() -> None:
+ """Load the checksum-pinned voice from the image and serve it on CPU."""
+ CACHE_DIR.mkdir(parents=True, exist_ok=True)
+ model_path = CACHE_DIR / f"{VOICE_NAME}.onnx"
+ config_path = CACHE_DIR / f"{VOICE_NAME}.onnx.json"
+ if not model_path.exists() or not config_path.exists():
+ raise RuntimeError(f"baked Piper voice is missing: {VOICE_NAME}")
+ print(f"[tts] loading Piper voice {VOICE_NAME} on CPU", flush=True)
+ voice = PiperVoice.load(model_path, config_path, use_cuda=False, download_dir=CACHE_DIR)
+ server = ThreadingHTTPServer((HOST, PORT), SpeechHandler)
+ server.voice = voice # type: ignore[attr-defined]
+ print(f"[tts] ready on {HOST}:{PORT}", flush=True)
+ server.serve_forever(poll_interval=0.25)
+
+
+if __name__ == "__main__":
+ main()
diff --git a/dockerfiles/hermes-python-sandbox-tool.py b/dockerfiles/hermes-python-sandbox-tool.py
index 5689727fc..725481b49 100644
--- a/dockerfiles/hermes-python-sandbox-tool.py
+++ b/dockerfiles/hermes-python-sandbox-tool.py
@@ -50,12 +50,13 @@ registry.register(
schema={
"name": "python_sandbox",
"description": (
- "Run Python in this user's separate credential-free computation "
- "sandbox. Use it for statistics, probability, Monte Carlo simulation, "
- "data transforms, and calculations. The sandbox has no Kubernetes, "
- "Vault, model-provider credentials, or access to other users. Public "
- "research belongs in web_search; pass only the data needed for the "
- "calculation and print the result."
+ "Run Python in this user's credential-free computation sandbox. "
+ "Its /workspace and /opt/data/workspace paths expose the same private "
+ "workspace, and /tmp is writable for temporary verifiers. "
+ "Use it for file verification, statistics, probability, Monte Carlo "
+ "simulation, data transforms, and calculations. The sandbox has no "
+ "Kubernetes, Vault, model-provider credentials, or access to other "
+ "users. Public research belongs in web_search."
),
"parameters": {
"type": "object",
diff --git a/dockerfiles/hermes-webui-atlas-patch.py b/dockerfiles/hermes-webui-atlas-patch.py
new file mode 100644
index 000000000..20d043992
--- /dev/null
+++ b/dockerfiles/hermes-webui-atlas-patch.py
@@ -0,0 +1,96 @@
+#!/usr/bin/env python3
+"""Apply fail-closed Atlas voice integration patches to pinned Hermes WebUI."""
+
+from pathlib import Path
+
+
+ROOT = Path("/opt/hermes-webui")
+
+
+def replace_exact(path: Path, before: str, after: str, count: int = 1) -> None:
+ """Replace an exact upstream fragment and fail when the pin has drifted."""
+ source = path.read_text(encoding="utf-8")
+ if source.count(before) != count:
+ raise SystemExit(f"Atlas voice patch context changed in {path}: {before[:80]!r}")
+ path.write_text(source.replace(before, after, count), encoding="utf-8")
+
+
+index = ROOT / "static/index.html"
+replace_exact(
+ index,
+ '',
+ '',
+)
+replace_exact(
+ index,
+ '',
+ '\n',
+)
+
+ui = ROOT / "static/ui.js"
+replace_exact(ui, "function _playEdgeTtsChunked(text, btn){", "function _playEdgeTtsChunked(text, btn, engineOverride){")
+replace_exact(
+ ui,
+ "body:JSON.stringify({text:chunk, voice:voice, rate:rate, pitch:pitch})",
+ "body:JSON.stringify({text:chunk, voice:voice, rate:rate, pitch:pitch, engine:engineOverride||'edge'})",
+)
+replace_exact(
+ ui,
+ "if(engine==='edge'){\n _playEdgeTtsChunked(clean, btn);",
+ "if(engine==='edge'||engine==='atlas'){\n _playEdgeTtsChunked(clean, btn, engine);",
+)
+replace_exact(
+ ui,
+ "if(engine==='edge'){\n _playEdgeTtsChunked(clean, null);",
+ "if(engine==='edge'||engine==='atlas'){\n _playEdgeTtsChunked(clean, null, engine);",
+)
+
+routes = ROOT / "api/routes.py"
+marker = " # ── ElevenLabs TTS ──────────────────────────────────────────────────\n"
+atlas = ''' # ── Atlas private Jetson TTS ─────────────────────────────────────────
+ if engine == "atlas":
+ atlas_url = os.getenv("HERMES_WEBUI_ATLAS_TTS_URL", "").strip()
+ expected_url = "http://hermes-tts.hermes.svc.cluster.local:9001/v1/audio/speech"
+ if atlas_url != expected_url:
+ from api.helpers import bad as _bad
+ return _bad(handler, "Atlas private TTS is not configured", 503)
+ speed = 1.0
+ if rate_str:
+ try:
+ speed = max(0.5, min(2.0, 1.0 + (float(rate_str.rstrip("%")) / 100.0)))
+ except ValueError:
+ speed = 1.0
+ request_body = json.dumps({
+ "model": "piper",
+ "input": text,
+ "voice": "en_US-lessac-high",
+ "speed": speed,
+ }).encode("utf-8")
+ request = Request(atlas_url, data=request_body, headers={
+ "Content-Type": "application/json",
+ "Accept": "audio/wav",
+ })
+ try:
+ with _tts_open(
+ request,
+ timeout=45,
+ opener_factory=lambda: build_opener(ProxyHandler({}), _NoRedirectTtsHandler()),
+ ) as response:
+ audio_data = _buffer_tts_audio_response(response)
+ except Exception:
+ logger.exception("Atlas private TTS generation failed")
+ from api.helpers import bad as _bad
+ return _bad(handler, "Atlas private TTS generation failed", 502)
+ handler.send_response(200)
+ handler.send_header("Content-Type", "audio/wav")
+ handler.send_header("Cache-Control", "no-store")
+ handler.send_header("Content-Length", str(len(audio_data)))
+ handler.end_headers()
+ try:
+ handler.wfile.write(audio_data)
+ except (BrokenPipeError, ConnectionResetError):
+ pass
+ return True
+
+'''
+replace_exact(routes, marker, atlas + marker)
diff --git a/dockerfiles/hermes-webui-atlas-voice.js b/dockerfiles/hermes-webui-atlas-voice.js
new file mode 100644
index 000000000..a59007a27
--- /dev/null
+++ b/dockerfiles/hermes-webui-atlas-voice.js
@@ -0,0 +1,287 @@
+// Natural turn-taking for chat.hermes.bstein.dev using the private Jetsons.
+(function(){
+ 'use strict';
+
+ const modeBtn=document.getElementById('btnVoiceMode');
+ const bar=document.getElementById('voiceModeBar');
+ const indicator=document.getElementById('voiceModeIndicator');
+ const label=document.getElementById('voiceModeLabel');
+ const composer=document.getElementById('msg');
+ if(!modeBtn||!bar||!indicator||!label||!composer||!navigator.mediaDevices||!window.MediaRecorder) return;
+
+ let ready=false;
+ let active=false;
+ let state='idle';
+ let generation=0;
+ let recorder=null;
+ let stream=null;
+ let audioContext=null;
+ let vadTimer=null;
+ let currentAudio=null;
+ let thinkingSession=null;
+ const originalAutoRead=window.autoReadLastAssistant;
+ const originalApplyPreference=window._applyVoiceModePref;
+
+ function toast(message){
+ if(typeof window.showToast==='function') window.showToast(message,3000);
+ }
+
+ function setState(next, customLabel){
+ state=next;
+ indicator.className='voice-mode-indicator '+next;
+ label.textContent=customLabel||(next==='listening'?'Listening…':next==='speaking'?'Speaking…':next==='thinking'?'Thinking…':'');
+ bar.style.display=active&&next!=='idle'?'':'none';
+ }
+
+ function stopCapture(){
+ if(vadTimer){clearInterval(vadTimer);vadTimer=null;}
+ if(recorder&&recorder.state!=='inactive'){
+ try{recorder.stop();}catch(_){ }
+ }
+ recorder=null;
+ if(stream){stream.getTracks().forEach(function(track){track.stop();});stream=null;}
+ if(audioContext){try{audioContext.close();}catch(_){ }audioContext=null;}
+ }
+
+ function stopPlayback(){
+ if(!currentAudio) return;
+ try{currentAudio.pause();currentAudio.currentTime=0;}catch(_){ }
+ currentAudio=null;
+ }
+
+ function deactivate(showMessage){
+ generation+=1;
+ active=false;
+ state='idle';
+ thinkingSession=null;
+ stopCapture();
+ stopPlayback();
+ modeBtn.classList.remove('active');
+ bar.style.display='none';
+ if(showMessage) toast('Hands-free voice mode off');
+ }
+
+ function restartSoon(token, delay){
+ window.setTimeout(function(){
+ if(active&&token===generation) startListening(token);
+ },delay||500);
+ }
+
+ function sendTranscript(transcript, token){
+ if(!active||token!==generation) return;
+ const text=String(transcript||'').trim();
+ if(!text){restartSoon(token,350);return;}
+ composer.value=text;
+ if(typeof window.autoResize==='function') window.autoResize();
+ thinkingSession=(typeof S!=='undefined'&&S.session)?S.session.session_id:null;
+ setState('thinking');
+ if(typeof window.send==='function') window.send();
+ }
+
+ async function transcribe(blob, token){
+ if(!active||token!==generation) return;
+ setState('thinking','Transcribing…');
+ const ext=(blob.type||'').indexOf('ogg')>=0?'ogg':'webm';
+ const form=new FormData();
+ form.append('file',new File([blob],'voice-input.'+ext,{type:blob.type||'audio/'+ext}));
+ try{
+ const response=await fetch('/api/transcribe',{method:'POST',body:form});
+ const payload=await response.json().catch(function(){return {};});
+ if(!response.ok) throw new Error(payload.error||('Whisper request failed: '+response.status));
+ sendTranscript(payload.transcript,token);
+ }catch(error){
+ if(!active||token!==generation) return;
+ deactivate(false);
+ toast((error&&error.message)||'Private Whisper is unavailable');
+ // If the browser supplies its own recognizer, hand control back to the
+ // upstream voice implementation until the Jetson becomes healthy again.
+ if(window.SpeechRecognition||window.webkitSpeechRecognition){
+ modeBtn.removeEventListener('click',onVoiceClick,true);
+ window.setTimeout(function(){modeBtn.click();},50);
+ }
+ }
+ }
+
+ async function startListening(token){
+ if(!active||token!==generation) return;
+ stopCapture();
+ setState('listening');
+ try{
+ const capture=await navigator.mediaDevices.getUserMedia({
+ audio:{echoCancellation:true,noiseSuppression:true,autoGainControl:true},
+ });
+ if(!active||token!==generation){capture.getTracks().forEach(function(track){track.stop();});return;}
+ stream=capture;
+ const Context=window.AudioContext||window.webkitAudioContext;
+ audioContext=new Context();
+ const analyser=audioContext.createAnalyser();
+ analyser.fftSize=1024;
+ audioContext.createMediaStreamSource(stream).connect(analyser);
+ const samples=new Uint8Array(analyser.fftSize);
+ const mimeTypes=['audio/webm;codecs=opus','audio/ogg;codecs=opus','audio/webm'];
+ const mime=mimeTypes.find(function(value){return MediaRecorder.isTypeSupported(value);})||'';
+ const chunks=[];
+ let heardSpeech=false;
+ let lastSpeech=Date.now();
+ const started=Date.now();
+ recorder=new MediaRecorder(stream,mime?{mimeType:mime}:undefined);
+ recorder.ondataavailable=function(event){if(event.data&&event.data.size) chunks.push(event.data);};
+ recorder.onstop=function(){
+ if(vadTimer){clearInterval(vadTimer);vadTimer=null;}
+ const recordedStream=stream;
+ stream=null;
+ if(recordedStream) recordedStream.getTracks().forEach(function(track){track.stop();});
+ if(audioContext){try{audioContext.close();}catch(_){ }audioContext=null;}
+ recorder=null;
+ if(!active||token!==generation) return;
+ if(!heardSpeech||!chunks.length){restartSoon(token,300);return;}
+ transcribe(new Blob(chunks,{type:mime||'audio/webm'}),token);
+ };
+ recorder.start(250);
+ const silenceMs=Math.max(900,parseInt(localStorage.getItem('hermes-voice-silence-ms')||'1600',10)||1600);
+ vadTimer=window.setInterval(function(){
+ if(!active||token!==generation||!recorder||recorder.state==='inactive') return;
+ analyser.getByteTimeDomainData(samples);
+ let energy=0;
+ for(let i=0;i0.025){heardSpeech=true;lastSpeech=now;}
+ const finished=heardSpeech&&(now-lastSpeech)>=silenceMs;
+ const timedOut=now-started>=90000;
+ const idle=(!heardSpeech)&&(now-started)>=20000;
+ if(finished||timedOut||idle){
+ if(vadTimer){clearInterval(vadTimer);vadTimer=null;}
+ try{recorder.stop();}catch(_){ }
+ }
+ },100);
+ }catch(error){
+ if(!active||token!==generation) return;
+ deactivate(false);
+ toast((error&&error.message)||'Microphone permission is required');
+ }
+ }
+
+ function cleanForSpeech(text){
+ if(typeof window._stripForTTS==='function') return window._stripForTTS(text);
+ return String(text||'').replace(/```[\s\S]*?```/g,' code block ').replace(/\s+/g,' ').trim();
+ }
+
+ function playBlob(blob, token){
+ return new Promise(function(resolve,reject){
+ if(!active||token!==generation){resolve();return;}
+ const url=URL.createObjectURL(blob);
+ const audio=new Audio(url);
+ currentAudio=audio;
+ function cleanup(){
+ if(currentAudio===audio) currentAudio=null;
+ URL.revokeObjectURL(url);
+ }
+ audio.onended=function(){cleanup();resolve();};
+ audio.onerror=function(){cleanup();reject(new Error('Local speech playback failed'));};
+ audio.play().catch(function(error){cleanup();reject(error);});
+ });
+ }
+
+ async function speakResponse(token){
+ if(!active||token!==generation) return;
+ const currentSession=(typeof S!=='undefined'&&S.session)?S.session.session_id:null;
+ if(thinkingSession&¤tSession&&thinkingSession!==currentSession){
+ thinkingSession=null;
+ restartSoon(token,250);
+ return;
+ }
+ thinkingSession=null;
+ const rows=document.querySelectorAll('.msg-row[data-role="assistant"], .assistant-segment[data-raw-text]');
+ if(!rows.length){restartSoon(token,250);return;}
+ const text=cleanForSpeech(rows[rows.length-1].dataset.rawText||'');
+ if(!text){restartSoon(token,250);return;}
+ setState('speaking');
+ const chunks=typeof window._splitForTTS==='function'?window._splitForTTS(text,900):[text];
+ try{
+ for(const chunk of chunks){
+ if(!active||token!==generation) return;
+ const response=await fetch('/api/tts',{
+ method:'POST',
+ headers:{'Content-Type':'application/json'},
+ body:JSON.stringify({text:chunk,engine:'atlas'}),
+ });
+ if(!response.ok){
+ const payload=await response.json().catch(function(){return {};});
+ throw new Error(payload.error||('Local speech request failed: '+response.status));
+ }
+ await playBlob(await response.blob(),token);
+ }
+ }catch(error){
+ if(active&&token===generation) toast((error&&error.message)||'Local speech is unavailable');
+ }
+ restartSoon(token,450);
+ }
+
+ function activate(){
+ generation+=1;
+ const token=generation;
+ active=true;
+ modeBtn.classList.add('active');
+ toast('Hands-free private voice mode on');
+ if(typeof window.stopTTS==='function') window.stopTTS();
+ if(typeof S!=='undefined'&&S.busy){setState('thinking');return;}
+ startListening(token);
+ }
+
+ function onVoiceClick(event){
+ if(!ready) return;
+ event.preventDefault();
+ event.stopImmediatePropagation();
+ if(active) deactivate(true); else activate();
+ }
+
+ async function initialize(){
+ try{
+ const response=await fetch('/api/transcribe/capability',{cache:'no-store'});
+ const capability=await response.json().catch(function(){return {};});
+ if(!response.ok||!capability.available||capability.provider!=='local_command') return;
+ ready=true;
+ if(localStorage.getItem('hermes-atlas-voice-initialized')!=='1'){
+ localStorage.setItem('hermes-atlas-voice-initialized','1');
+ localStorage.setItem('hermes-voice-mode-button','true');
+ localStorage.setItem('hermes-tts-engine','atlas');
+ localStorage.setItem('hermes-tts-enabled','true');
+ localStorage.setItem('hermes-voice-silence-ms','1600');
+ }
+ const selector=document.getElementById('settingsTtsEngine');
+ if(selector&&!selector.querySelector('option[value="atlas"]')){
+ const option=document.createElement('option');
+ option.value='atlas';
+ option.textContent='Atlas Jetson (private)';
+ selector.insertBefore(option,selector.firstChild);
+ }
+ modeBtn.style.display=localStorage.getItem('hermes-voice-mode-button')==='false'?'none':'';
+ modeBtn.addEventListener('click',onVoiceClick,true);
+ window._applyVoiceModePref=function(){
+ if(typeof originalApplyPreference==='function') originalApplyPreference();
+ if(ready){
+ const enabled=localStorage.getItem('hermes-voice-mode-button')!=='false';
+ modeBtn.style.display=enabled?'':'none';
+ if(!enabled&&active) deactivate(false);
+ }
+ };
+ window.autoReadLastAssistant=function(){
+ if(active&&state==='thinking'){speakResponse(generation);return;}
+ if(typeof originalAutoRead==='function') originalAutoRead.apply(this,arguments);
+ };
+ window._voiceModeActive=function(){return active;};
+ window._voiceModeDeactivate=function(){deactivate(false);};
+ window._voiceModeImmediateSend=function(){
+ if(active&&recorder&&recorder.state!=='inactive') recorder.stop();
+ };
+ }catch(_){
+ // The upstream browser voice implementation remains available as fallback.
+ }
+ }
+
+ initialize();
+})();
diff --git a/services/ai-llm/deployment.yaml b/services/ai-llm/deployment.yaml
index 75c0f9c14..06db8282b 100644
--- a/services/ai-llm/deployment.yaml
+++ b/services/ai-llm/deployment.yaml
@@ -21,7 +21,7 @@ spec:
app: ollama
annotations:
ai.bstein.dev/model: qwen2.5:3b-instruct-q4_0,qwen2.5:14b-instruct-q4_0
- ai.bstein.dev/gpu: GPU pool (titan-20/21)
+ ai.bstein.dev/gpu: titan-20 shared routing GPU
ai.bstein.dev/restartedAt: "2026-01-26T12:00:00Z"
spec:
affinity:
@@ -33,7 +33,6 @@ spec:
operator: In
values:
- titan-20
- - titan-21
runtimeClassName: nvidia
volumes:
- name: models
diff --git a/services/hermes/agent-configmap.yaml b/services/hermes/agent-configmap.yaml
index 138da181c..63e4a2e66 100644
--- a/services/hermes/agent-configmap.yaml
+++ b/services/hermes/agent-configmap.yaml
@@ -99,7 +99,10 @@ data:
home_mode: auto
approvals:
- mode: smart
+ # This owner-only workspace is already bounded by non-root execution,
+ # repository scope, and read-only cluster RBAC. Keep routine engineering
+ # work non-interactive while retaining explicit hard denies below.
+ mode: "off"
deny:
- "*kubectl apply*"
- "*kubectl delete*"
@@ -111,6 +114,10 @@ data:
- "*flux suspend*"
- "*flux resume*"
- "*vault kv*"
+ - "*git push --force*"
+ - "*git push -f*"
+ - "*git reset --hard*"
+ - "*git clean -f*"
dashboard:
public_url: https://agent.hermes.bstein.dev
@@ -204,9 +211,15 @@ data:
`herdr-dispatch --shape --effort [--provider codex|claude]`
Add `--start --project --task --prompt ` to
- create a Herdr workspace and launch the selected CLI. Use `herdr agent list`,
- `herdr agent wait`, `herdr agent read`, and `herdr agent prompt` to supervise
- it. If Codex reports its first-use login requirement, run
+ create a separate Herdr worker space and launch the selected CLI. The visible
+ project tab remains a single Hermes coordinator pane. Never split Codex,
+ Claude, or a second Hermes process into that tab. Worker-space labels include
+ the parent project, provider, and task (for example
+ `cassandra-claude-review`) so the left rail preserves ownership and purpose.
+ Keep completed workers available as labeled spaces until their evidence has
+ been synthesized; do not tile them over the coordinator. Use `herdr agent
+ list`, `herdr agent wait`, `herdr agent read`, and `herdr agent prompt` to
+ supervise them. If Codex reports its first-use login requirement, run
`codex login --device-auth` once and ask Brad to complete the displayed code.
A hosted capacity failure should fall across providers at the same effort
diff --git a/services/hermes/agent-deployment.yaml b/services/hermes/agent-deployment.yaml
index be5cb1f25..8fe3cc48f 100644
--- a/services/hermes/agent-deployment.yaml
+++ b/services/hermes/agent-deployment.yaml
@@ -24,7 +24,7 @@ spec:
ai.bstein.dev/execution: Herdr-supervised Codex and Claude Code
ai.bstein.dev/model-policy: Jetson-assisted AUTO routing, low through xhigh, cross-provider fallback
ai.bstein.dev/placement: rpi5 preferred; Jetson deferred until state storage is available
- ai.bstein.dev/config-rev: "20260809-terminal-recovery-v2"
+ ai.bstein.dev/config-rev: "20260810-worker-spaces-permissions"
vault.hashicorp.com/agent-inject: "true"
vault.hashicorp.com/role: hermes-agent
vault.hashicorp.com/agent-inject-secret-anthropic-token: kv/data/atlas/hermes/agent-tokens
@@ -350,7 +350,7 @@ spec:
requests: {cpu: 250m, memory: 512Mi}
limits: {cpu: "2", memory: 4Gi}
- name: webui
- image: registry.bstein.dev/bstein/hermes-webui@sha256:a771858bd668d25e19c74864baea5425101c8cd5215d1ba3a312f3312ce6c5e1
+ image: registry.bstein.dev/bstein/hermes-webui@sha256:06f195df381abc60e97f31c0676044faffd1a51cb365cda77a919eee04827c71
imagePullPolicy: IfNotPresent
command: [/bin/sh, -ec]
args:
diff --git a/services/hermes/chat-configmap.yaml b/services/hermes/chat-configmap.yaml
index 57fe1da21..afecbb0f1 100644
--- a/services/hermes/chat-configmap.yaml
+++ b/services/hermes/chat-configmap.yaml
@@ -35,6 +35,12 @@ data:
backend: ddgs
search_backend: ddgs
extract_backend: public-extract
+ stt:
+ enabled: true
+ provider: local_command
+ local:
+ model: large-v3-turbo
+ language: auto
model_catalog:
enabled: true
ttl_hours: 1
@@ -81,16 +87,26 @@ data:
user's private workspace and may use this user's private memory, skills,
profiles, and task list. Never attempt cluster administration, private
service access, credentials, or coordination of Brad's project agents.
- Python may run only through the credential-free sandbox tool. The user's
- conversations and files must never be mixed with another Keycloak user's
- state.
+ Python may run only through the credential-free sandbox tool. Its
+ `/workspace` and `/opt/data/workspace` both expose the same private 10 GiB
+ workspace, and its writable `/tmp` is available for temporary
+ verification scripts. Read and verify generated artifacts directly there;
+ do not claim the sandbox is disconnected from the user's files. The user's
+ conversations and files must never be mixed with another Keycloak user's state.
+
+ Voice conversations use the same assistant, session, AUTO route, tools, and
+ private workspace as typed conversations. Whisper and speech synthesis are
+ transport services only; they do not select or replace the answering model.
AGENTS.md: |
# Private Hermes chat
This runtime belongs to one authenticated Keycloak identity and one private
- persistent volume. Provide conversational help with the private workspace,
+ persistent workspace. Provide conversational help with the private workspace,
memory, skills, profiles, task list, session search, public web tools, and
- the separate per-tenant browser and Python sandbox. Use delegation selectively for
- independent research or verification, then present a single final answer.
+ the per-tenant browser and Python sandbox. The Python sandbox sees the same
+ workspace at both `/workspace` and `/opt/data/workspace`, and may use `/tmp`
+ for bounded temporary work. Use
+ delegation selectively for independent research or verification, then
+ present a single final answer.
Do not claim access to Kubernetes, Vault, Gitea, Brad's projects, other
users, the agent coordinator, or automated triage.
diff --git a/services/hermes/chat-sandbox.yaml b/services/hermes/chat-sandbox.yaml
index 4c7a148e3..e73335a36 100644
--- a/services/hermes/chat-sandbox.yaml
+++ b/services/hermes/chat-sandbox.yaml
@@ -12,39 +12,147 @@ spec:
selector:
app: hermes-chat-sandbox
ports:
- - name: http
- port: 9080
- targetPort: http
+ - {name: http, port: 9080, targetPort: http, protocol: TCP}
---
apiVersion: apps/v1
-kind: StatefulSet
+kind: Deployment
metadata:
- name: hermes-chat-sandbox
+ name: hermes-chat-sandbox-0
namespace: hermes
labels:
app: hermes-chat-sandbox
+ ai.bstein.dev/tenant-ordinal: "0"
spec:
- serviceName: hermes-chat-sandbox
- replicas: 4
- podManagementPolicy: Parallel
- persistentVolumeClaimRetentionPolicy:
- whenDeleted: Retain
- whenScaled: Retain
+ replicas: 1
selector:
matchLabels:
app: hermes-chat-sandbox
+ ai.bstein.dev/tenant-ordinal: "0"
template:
metadata:
labels:
app: hermes-chat-sandbox
+ ai.bstein.dev/tenant-ordinal: "0"
annotations:
ai.bstein.dev/role: isolated-user-computation
- ai.bstein.dev/isolation: one credential-free sandbox per chat tenant
+ ai.bstein.dev/isolation: credential-free and default-deny network
spec:
+ hostname: hermes-chat-sandbox-0
+ subdomain: hermes-chat-sandbox
automountServiceAccountToken: false
enableServiceLinks: false
securityContext:
- fsGroup: 20000
+ fsGroup: 10000
+ fsGroupChangePolicy: OnRootMismatch
+ seccompProfile:
+ type: RuntimeDefault
+ affinity: &sandbox-affinity
+ nodeAffinity:
+ requiredDuringSchedulingIgnoredDuringExecution:
+ nodeSelectorTerms:
+ - matchExpressions:
+ - key: kubernetes.io/arch
+ operator: In
+ values: [arm64]
+ - key: node-role.kubernetes.io/worker
+ operator: In
+ values: ["true"]
+ - key: kubernetes.io/hostname
+ operator: NotIn
+ values: [titan-05, titan-08, titan-13, titan-14, titan-17, titan-18, titan-19]
+ preferredDuringSchedulingIgnoredDuringExecution:
+ - weight: 100
+ preference:
+ matchExpressions:
+ - key: hardware
+ operator: In
+ values: [rpi5]
+ - weight: 40
+ preference:
+ matchExpressions:
+ - key: hardware
+ operator: In
+ values: [rpi4]
+ podAntiAffinity:
+ preferredDuringSchedulingIgnoredDuringExecution:
+ - weight: 100
+ podAffinityTerm:
+ labelSelector:
+ matchLabels:
+ app: hermes-chat-sandbox
+ topologyKey: kubernetes.io/hostname
+ containers:
+ - &sandbox-container
+ name: sandbox
+ image: registry.bstein.dev/bstein/hermes-chat-sandbox@sha256:17ee62b8e61c08573a3a8cca903b38ec43800cb44ec29340e1bc095176544bca
+ imagePullPolicy: IfNotPresent
+ ports:
+ - {name: http, containerPort: 9080, protocol: TCP}
+ volumeMounts:
+ - {name: workspace, mountPath: /workspace}
+ - {name: workspace, mountPath: /opt/data/workspace}
+ - {name: tmp, mountPath: /tmp}
+ readinessProbe:
+ httpGet: {path: /health, port: http}
+ initialDelaySeconds: 3
+ periodSeconds: 10
+ timeoutSeconds: 3
+ livenessProbe:
+ httpGet: {path: /health, port: http}
+ initialDelaySeconds: 10
+ periodSeconds: 30
+ timeoutSeconds: 3
+ securityContext:
+ allowPrivilegeEscalation: false
+ capabilities:
+ drop: [ALL]
+ readOnlyRootFilesystem: true
+ runAsNonRoot: true
+ runAsUser: 20000
+ runAsGroup: 10000
+ seccompProfile:
+ type: RuntimeDefault
+ resources:
+ requests: {cpu: 50m, memory: 128Mi}
+ limits: {cpu: "1", memory: 1Gi}
+ volumes:
+ - name: workspace
+ persistentVolumeClaim:
+ claimName: workspace-hermes-chat-tenant-0
+ - &sandbox-tmp
+ name: tmp
+ emptyDir:
+ sizeLimit: 256Mi
+---
+apiVersion: apps/v1
+kind: Deployment
+metadata:
+ name: hermes-chat-sandbox-1
+ namespace: hermes
+ labels:
+ app: hermes-chat-sandbox
+ ai.bstein.dev/tenant-ordinal: "1"
+spec:
+ replicas: 1
+ selector:
+ matchLabels:
+ app: hermes-chat-sandbox
+ ai.bstein.dev/tenant-ordinal: "1"
+ template:
+ metadata:
+ labels:
+ app: hermes-chat-sandbox
+ ai.bstein.dev/tenant-ordinal: "1"
+ annotations:
+ ai.bstein.dev/role: isolated-user-computation
+ ai.bstein.dev/isolation: credential-free and default-deny network
+ spec:
+ hostname: hermes-chat-sandbox-1
+ subdomain: hermes-chat-sandbox
+ automountServiceAccountToken: false
+ enableServiceLinks: false
+ securityContext:
+ fsGroup: 10000
fsGroupChangePolicy: OnRootMismatch
seccompProfile:
type: RuntimeDefault
@@ -61,7 +169,7 @@ spec:
values: ["true"]
- key: kubernetes.io/hostname
operator: NotIn
- values: [titan-05, titan-08, titan-13, titan-14, titan-17, titan-18]
+ values: [titan-05, titan-08, titan-13, titan-14, titan-17, titan-18, titan-19]
preferredDuringSchedulingIgnoredDuringExecution:
- weight: 100
preference:
@@ -91,6 +199,7 @@ spec:
- {name: http, containerPort: 9080, protocol: TCP}
volumeMounts:
- {name: workspace, mountPath: /workspace}
+ - {name: workspace, mountPath: /opt/data/workspace}
- {name: tmp, mountPath: /tmp}
readinessProbe:
httpGet: {path: /health, port: http}
@@ -109,24 +218,232 @@ spec:
readOnlyRootFilesystem: true
runAsNonRoot: true
runAsUser: 20000
- runAsGroup: 20000
+ runAsGroup: 10000
seccompProfile:
type: RuntimeDefault
resources:
requests: {cpu: 50m, memory: 128Mi}
limits: {cpu: "1", memory: 1Gi}
volumes:
+ - name: workspace
+ persistentVolumeClaim:
+ claimName: workspace-hermes-chat-tenant-1
+ - name: tmp
+ emptyDir:
+ sizeLimit: 256Mi
+---
+apiVersion: apps/v1
+kind: Deployment
+metadata:
+ name: hermes-chat-sandbox-2
+ namespace: hermes
+ labels:
+ app: hermes-chat-sandbox
+ ai.bstein.dev/tenant-ordinal: "2"
+spec:
+ replicas: 1
+ selector:
+ matchLabels:
+ app: hermes-chat-sandbox
+ ai.bstein.dev/tenant-ordinal: "2"
+ template:
+ metadata:
+ labels:
+ app: hermes-chat-sandbox
+ ai.bstein.dev/tenant-ordinal: "2"
+ annotations:
+ ai.bstein.dev/role: isolated-user-computation
+ ai.bstein.dev/isolation: credential-free and default-deny network
+ spec:
+ hostname: hermes-chat-sandbox-2
+ subdomain: hermes-chat-sandbox
+ automountServiceAccountToken: false
+ enableServiceLinks: false
+ securityContext:
+ fsGroup: 10000
+ fsGroupChangePolicy: OnRootMismatch
+ seccompProfile:
+ type: RuntimeDefault
+ affinity:
+ nodeAffinity:
+ requiredDuringSchedulingIgnoredDuringExecution:
+ nodeSelectorTerms:
+ - matchExpressions:
+ - key: kubernetes.io/arch
+ operator: In
+ values: [arm64]
+ - key: node-role.kubernetes.io/worker
+ operator: In
+ values: ["true"]
+ - key: kubernetes.io/hostname
+ operator: NotIn
+ values: [titan-05, titan-08, titan-13, titan-14, titan-17, titan-18, titan-19]
+ preferredDuringSchedulingIgnoredDuringExecution:
+ - weight: 100
+ preference:
+ matchExpressions:
+ - key: hardware
+ operator: In
+ values: [rpi5]
+ - weight: 40
+ preference:
+ matchExpressions:
+ - key: hardware
+ operator: In
+ values: [rpi4]
+ podAntiAffinity:
+ preferredDuringSchedulingIgnoredDuringExecution:
+ - weight: 100
+ podAffinityTerm:
+ labelSelector:
+ matchLabels:
+ app: hermes-chat-sandbox
+ topologyKey: kubernetes.io/hostname
+ containers:
+ - name: sandbox
+ image: registry.bstein.dev/bstein/hermes-chat-sandbox@sha256:17ee62b8e61c08573a3a8cca903b38ec43800cb44ec29340e1bc095176544bca
+ imagePullPolicy: IfNotPresent
+ ports:
+ - {name: http, containerPort: 9080, protocol: TCP}
+ volumeMounts:
+ - {name: workspace, mountPath: /workspace}
+ - {name: workspace, mountPath: /opt/data/workspace}
+ - {name: tmp, mountPath: /tmp}
+ readinessProbe:
+ httpGet: {path: /health, port: http}
+ initialDelaySeconds: 3
+ periodSeconds: 10
+ timeoutSeconds: 3
+ livenessProbe:
+ httpGet: {path: /health, port: http}
+ initialDelaySeconds: 10
+ periodSeconds: 30
+ timeoutSeconds: 3
+ securityContext:
+ allowPrivilegeEscalation: false
+ capabilities:
+ drop: [ALL]
+ readOnlyRootFilesystem: true
+ runAsNonRoot: true
+ runAsUser: 20000
+ runAsGroup: 10000
+ seccompProfile:
+ type: RuntimeDefault
+ resources:
+ requests: {cpu: 50m, memory: 128Mi}
+ limits: {cpu: "1", memory: 1Gi}
+ volumes:
+ - name: workspace
+ persistentVolumeClaim:
+ claimName: workspace-hermes-chat-tenant-2
+ - name: tmp
+ emptyDir:
+ sizeLimit: 256Mi
+---
+apiVersion: apps/v1
+kind: Deployment
+metadata:
+ name: hermes-chat-sandbox-3
+ namespace: hermes
+ labels:
+ app: hermes-chat-sandbox
+ ai.bstein.dev/tenant-ordinal: "3"
+spec:
+ replicas: 1
+ selector:
+ matchLabels:
+ app: hermes-chat-sandbox
+ ai.bstein.dev/tenant-ordinal: "3"
+ template:
+ metadata:
+ labels:
+ app: hermes-chat-sandbox
+ ai.bstein.dev/tenant-ordinal: "3"
+ annotations:
+ ai.bstein.dev/role: isolated-user-computation
+ ai.bstein.dev/isolation: credential-free and default-deny network
+ spec:
+ hostname: hermes-chat-sandbox-3
+ subdomain: hermes-chat-sandbox
+ automountServiceAccountToken: false
+ enableServiceLinks: false
+ securityContext:
+ fsGroup: 10000
+ fsGroupChangePolicy: OnRootMismatch
+ seccompProfile:
+ type: RuntimeDefault
+ affinity:
+ nodeAffinity:
+ requiredDuringSchedulingIgnoredDuringExecution:
+ nodeSelectorTerms:
+ - matchExpressions:
+ - key: kubernetes.io/arch
+ operator: In
+ values: [arm64]
+ - key: node-role.kubernetes.io/worker
+ operator: In
+ values: ["true"]
+ - key: kubernetes.io/hostname
+ operator: NotIn
+ values: [titan-05, titan-08, titan-13, titan-14, titan-17, titan-18, titan-19]
+ preferredDuringSchedulingIgnoredDuringExecution:
+ - weight: 100
+ preference:
+ matchExpressions:
+ - key: hardware
+ operator: In
+ values: [rpi5]
+ - weight: 40
+ preference:
+ matchExpressions:
+ - key: hardware
+ operator: In
+ values: [rpi4]
+ podAntiAffinity:
+ preferredDuringSchedulingIgnoredDuringExecution:
+ - weight: 100
+ podAffinityTerm:
+ labelSelector:
+ matchLabels:
+ app: hermes-chat-sandbox
+ topologyKey: kubernetes.io/hostname
+ containers:
+ - name: sandbox
+ image: registry.bstein.dev/bstein/hermes-chat-sandbox@sha256:17ee62b8e61c08573a3a8cca903b38ec43800cb44ec29340e1bc095176544bca
+ imagePullPolicy: IfNotPresent
+ ports:
+ - {name: http, containerPort: 9080, protocol: TCP}
+ volumeMounts:
+ - {name: workspace, mountPath: /workspace}
+ - {name: workspace, mountPath: /opt/data/workspace}
+ - {name: tmp, mountPath: /tmp}
+ readinessProbe:
+ httpGet: {path: /health, port: http}
+ initialDelaySeconds: 3
+ periodSeconds: 10
+ timeoutSeconds: 3
+ livenessProbe:
+ httpGet: {path: /health, port: http}
+ initialDelaySeconds: 10
+ periodSeconds: 30
+ timeoutSeconds: 3
+ securityContext:
+ allowPrivilegeEscalation: false
+ capabilities:
+ drop: [ALL]
+ readOnlyRootFilesystem: true
+ runAsNonRoot: true
+ runAsUser: 20000
+ runAsGroup: 10000
+ seccompProfile:
+ type: RuntimeDefault
+ resources:
+ requests: {cpu: 50m, memory: 128Mi}
+ limits: {cpu: "1", memory: 1Gi}
+ volumes:
+ - name: workspace
+ persistentVolumeClaim:
+ claimName: workspace-hermes-chat-tenant-3
- name: tmp
emptyDir:
sizeLimit: 256Mi
- volumeClaimTemplates:
- - metadata:
- name: workspace
- labels:
- app: hermes-chat-sandbox
- spec:
- accessModes: [ReadWriteOnce]
- storageClassName: astreae
- resources:
- requests:
- storage: 2Gi
diff --git a/services/hermes/chat-statefulset.yaml b/services/hermes/chat-statefulset.yaml
index d5788cb24..e7eaba884 100644
--- a/services/hermes/chat-statefulset.yaml
+++ b/services/hermes/chat-statefulset.yaml
@@ -28,7 +28,7 @@ spec:
ai.bstein.dev/role: isolated-user-chat
ai.bstein.dev/isolation: one Hermes process and PVC per Keycloak subject
ai.bstein.dev/model-policy: uniform automatic policy with per-user overrides
- ai.bstein.dev/config-rev: "20260809-browser-runtime-path"
+ ai.bstein.dev/config-rev: "20260810-natural-jetson-voice"
vault.hashicorp.com/agent-inject: "true"
vault.hashicorp.com/role: hermes-chat
vault.hashicorp.com/agent-inject-secret-anthropic-token: kv/data/atlas/hermes/agent-tokens
@@ -100,6 +100,12 @@ spec:
- |
set -eu
mkdir -p /opt/data/home/.local/bin /opt/data/logs /opt/data/workspace
+ if [ ! -e /opt/data/workspace/.hermes-workspace-v1 ]; then
+ if [ -d /legacy-home/workspace ]; then
+ cp -a /legacy-home/workspace/. /opt/data/workspace/
+ fi
+ touch /opt/data/workspace/.hermes-workspace-v1
+ fi
cp /config/config.yaml /opt/data/config.yaml
cp /config/SOUL.md /opt/data/SOUL.md
cp /config/AGENTS.md /opt/data/workspace/AGENTS.md
@@ -142,6 +148,8 @@ spec:
type: RuntimeDefault
volumeMounts:
- {name: home, mountPath: /opt/data}
+ - {name: home, mountPath: /legacy-home, readOnly: true}
+ - {name: workspace, mountPath: /opt/data/workspace}
- {name: config, mountPath: /config, readOnly: true}
resources:
requests: {cpu: 25m, memory: 32Mi}
@@ -195,6 +203,7 @@ spec:
- {name: API_SERVER_CORS_ORIGINS, value: https://chat.hermes.bstein.dev}
volumeMounts:
- {name: home, mountPath: /opt/data}
+ - {name: workspace, mountPath: /opt/data/workspace}
- {name: provider-auth, mountPath: /shared-auth, readOnly: true}
- {name: auth-patch, mountPath: /opt/hermes/hermes_cli/auth.py, subPath: auth.py}
readinessProbe:
@@ -217,7 +226,7 @@ spec:
requests: {cpu: 250m, memory: 512Mi}
limits: {cpu: "1", memory: 2Gi}
- name: webui
- image: registry.bstein.dev/bstein/hermes-webui@sha256:a771858bd668d25e19c74864baea5425101c8cd5215d1ba3a312f3312ce6c5e1
+ image: registry.bstein.dev/bstein/hermes-webui@sha256:06f195df381abc60e97f31c0676044faffd1a51cb365cda77a919eee04827c71
imagePullPolicy: IfNotPresent
command: [/bin/sh, -ec]
args:
@@ -246,9 +255,14 @@ spec:
- {name: HERMES_WEBUI_ALLOWED_ORIGINS, value: https://chat.hermes.bstein.dev}
- {name: HERMES_WEBUI_TRUST_FORWARDED_HOST, value: "1"}
- {name: HERMES_WEBUI_TRUST_FORWARDED_PROTO, value: "1"}
+ - {name: HERMES_STT_URL, value: http://hermes-stt.hermes.svc.cluster.local:9000/v1/audio/transcriptions}
+ - {name: HERMES_LOCAL_STT_COMMAND, value: "/opt/hermes/.venv/bin/python /opt/coordinator/hermes_stt_client.py {input_path} --output-dir {output_dir} --language {language} --model {model}"}
+ - {name: HERMES_WEBUI_ATLAS_TTS_URL, value: http://hermes-tts.hermes.svc.cluster.local:9001/v1/audio/speech}
volumeMounts:
- {name: home, mountPath: /opt/data}
+ - {name: workspace, mountPath: /opt/data/workspace}
- {name: provider-auth, mountPath: /shared-auth, readOnly: true}
+ - {name: coordinator, mountPath: /opt/coordinator, readOnly: true}
- {name: tmp, mountPath: /tmp}
readinessProbe:
httpGet: {path: /health, port: webui}
@@ -299,3 +313,14 @@ spec:
resources:
requests:
storage: 10Gi
+ - metadata:
+ name: workspace
+ labels:
+ app: hermes-chat-tenant
+ ai.bstein.dev/data: user-workspace
+ spec:
+ accessModes: [ReadWriteMany]
+ storageClassName: astreae
+ resources:
+ requests:
+ storage: 10Gi
diff --git a/services/hermes/deployment.yaml b/services/hermes/deployment.yaml
index d3ec899bd..6185329be 100644
--- a/services/hermes/deployment.yaml
+++ b/services/hermes/deployment.yaml
@@ -345,7 +345,7 @@ spec:
cpu: "2"
memory: 4Gi
- name: webui
- image: registry.bstein.dev/bstein/hermes-webui@sha256:a771858bd668d25e19c74864baea5425101c8cd5215d1ba3a312f3312ce6c5e1
+ image: registry.bstein.dev/bstein/hermes-webui@sha256:06f195df381abc60e97f31c0676044faffd1a51cb365cda77a919eee04827c71
imagePullPolicy: IfNotPresent
command: [/bin/sh, -ec]
args:
diff --git a/services/hermes/kustomization.yaml b/services/hermes/kustomization.yaml
index 49251205c..8bd1448ef 100644
--- a/services/hermes/kustomization.yaml
+++ b/services/hermes/kustomization.yaml
@@ -20,6 +20,7 @@ resources:
- ollama-deployment.yaml
- deployment.yaml
- agent-deployment.yaml
+ - voice-deployment.yaml
- chat-statefulset.yaml
- chat-sandbox.yaml
- chat-router.yaml
@@ -49,6 +50,7 @@ configMapGenerator:
- herdr_tab_router.py=scripts/herdr_tab_router.py
- hermes_coordinator.py=scripts/hermes_coordinator.py
- hermes_model_routing.py=scripts/hermes_model_routing.py
+ - hermes_stt_client.py=scripts/hermes_stt_client.py
- patch_hermes_auth.py=scripts/patch_hermes_auth.py
- patch_ttyd_index.py=scripts/patch_ttyd_index.py
options:
diff --git a/services/hermes/networkpolicy.yaml b/services/hermes/networkpolicy.yaml
index d1bb2c106..d287d3027 100644
--- a/services/hermes/networkpolicy.yaml
+++ b/services/hermes/networkpolicy.yaml
@@ -164,14 +164,13 @@ spec:
podSelector:
matchLabels:
app: hermes-chat-sandbox
- apps.kubernetes.io/pod-index: "0"
+ ai.bstein.dev/tenant-ordinal: "0"
policyTypes: [Ingress]
ingress:
- from:
- podSelector:
matchLabels:
- app: hermes-chat-tenant
- apps.kubernetes.io/pod-index: "0"
+ statefulset.kubernetes.io/pod-name: hermes-chat-tenant-0
ports:
- {protocol: TCP, port: 9080}
---
@@ -184,14 +183,13 @@ spec:
podSelector:
matchLabels:
app: hermes-chat-sandbox
- apps.kubernetes.io/pod-index: "1"
+ ai.bstein.dev/tenant-ordinal: "1"
policyTypes: [Ingress]
ingress:
- from:
- podSelector:
matchLabels:
- app: hermes-chat-tenant
- apps.kubernetes.io/pod-index: "1"
+ statefulset.kubernetes.io/pod-name: hermes-chat-tenant-1
ports:
- {protocol: TCP, port: 9080}
---
@@ -204,14 +202,13 @@ spec:
podSelector:
matchLabels:
app: hermes-chat-sandbox
- apps.kubernetes.io/pod-index: "2"
+ ai.bstein.dev/tenant-ordinal: "2"
policyTypes: [Ingress]
ingress:
- from:
- podSelector:
matchLabels:
- app: hermes-chat-tenant
- apps.kubernetes.io/pod-index: "2"
+ statefulset.kubernetes.io/pod-name: hermes-chat-tenant-2
ports:
- {protocol: TCP, port: 9080}
---
@@ -224,14 +221,13 @@ spec:
podSelector:
matchLabels:
app: hermes-chat-sandbox
- apps.kubernetes.io/pod-index: "3"
+ ai.bstein.dev/tenant-ordinal: "3"
policyTypes: [Ingress]
ingress:
- from:
- podSelector:
matchLabels:
- app: hermes-chat-tenant
- apps.kubernetes.io/pod-index: "3"
+ statefulset.kubernetes.io/pod-name: hermes-chat-tenant-3
ports:
- {protocol: TCP, port: 9080}
---
@@ -294,6 +290,15 @@ spec:
app: hermes-chat-sandbox
ports:
- {protocol: TCP, port: 9080}
+ - to:
+ - podSelector:
+ matchExpressions:
+ - key: app
+ operator: In
+ values: [hermes-stt, hermes-tts]
+ ports:
+ - {protocol: TCP, port: 9000}
+ - {protocol: TCP, port: 9001}
- to:
- ipBlock:
cidr: 0.0.0.0/0
@@ -426,3 +431,35 @@ spec:
- {protocol: TCP, port: 7681}
- {protocol: TCP, port: 8787}
- {protocol: TCP, port: 8080}
+---
+apiVersion: networking.k8s.io/v1
+kind: NetworkPolicy
+metadata:
+ name: hermes-private-voice
+ namespace: hermes
+spec:
+ podSelector:
+ matchExpressions:
+ - key: app
+ operator: In
+ values: [hermes-stt, hermes-tts]
+ policyTypes: [Ingress, Egress]
+ ingress:
+ - from:
+ - podSelector:
+ matchLabels:
+ app: hermes-chat-tenant
+ ports:
+ - {protocol: TCP, port: 9000}
+ - {protocol: TCP, port: 9001}
+ egress:
+ - to:
+ - namespaceSelector:
+ matchLabels:
+ kubernetes.io/metadata.name: kube-system
+ podSelector:
+ matchLabels:
+ k8s-app: kube-dns
+ ports:
+ - {protocol: UDP, port: 53}
+ - {protocol: TCP, port: 53}
diff --git a/services/hermes/oauth2-proxy.yaml b/services/hermes/oauth2-proxy.yaml
index 907da2b75..379662f22 100644
--- a/services/hermes/oauth2-proxy.yaml
+++ b/services/hermes/oauth2-proxy.yaml
@@ -192,7 +192,9 @@ spec:
- --cookie-csrf-per-request=true
- --cookie-csrf-per-request-limit=8
- --cookie-refresh=1h
- - --cookie-expire=8h
+ - --cookie-expire=168h
+ - --api-route=^/api/
+ - --api-route=^/health$
- --upstream=http://hermes-triage.hermes.svc.cluster.local:8787
- --http-address=0.0.0.0:4180
- --skip-provider-button=true
@@ -286,9 +288,11 @@ spec:
- --cookie-csrf-per-request=true
- --cookie-csrf-per-request-limit=8
- --cookie-refresh=1h
- - --cookie-expire=8h
+ - --cookie-expire=168h
- --custom-templates-dir=/etc/oauth2-proxy/templates
- '--skip-auth-route=GET=^/sw[.]js([?].*)?$'
+ - --api-route=^/api/
+ - --api-route=^/health$
- --upstream=http://hermes-chat-router.hermes.svc.cluster.local:8080
- --http-address=0.0.0.0:4180
- --skip-provider-button=true
diff --git a/services/hermes/scripts/herdr_dispatch.py b/services/hermes/scripts/herdr_dispatch.py
index 7143f6ca6..e8e9730ca 100644
--- a/services/hermes/scripts/herdr_dispatch.py
+++ b/services/hermes/scripts/herdr_dispatch.py
@@ -18,7 +18,7 @@ HERDR_BIN = Path("/opt/data/tools/bin/herdr")
CODEX_AUTH = Path("/opt/data/home/.codex/auth.json")
PROMPT_READY_MARKERS = {
"codex": "OpenAI Codex",
- "claude": "accept edits on",
+ "claude": "Claude Code",
}
@@ -122,7 +122,9 @@ def launch_worker(
"PATH": "/opt/data/tools/bin:" + env.get("PATH", ""),
}
)
- label = _slug(task)
+ project_label = _slug(project.name, 12)
+ task_label = _slug(task, 14)
+ label = _slug(f"{project_label}-{plan['worker']}-{task_label}")
created = _run(
[
str(HERDR_BIN),
@@ -141,7 +143,7 @@ def launch_worker(
except (KeyError, TypeError) as error:
raise RuntimeError("Herdr workspace response omitted the root pane") from error
- agent_name = _slug(f"{plan['worker']}-{task}")
+ agent_name = label
command = [
str(HERDR_BIN),
"agent",
@@ -162,10 +164,9 @@ def launch_worker(
plan["model"],
"-c",
f'model_reasoning_effort="{plan["effort"]}"',
- "-c",
- 'approval_policy="on-request"',
- "-c",
- 'sandbox_mode="workspace-write"',
+ "--approve-for-me",
+ "--sandbox",
+ "workspace-write",
]
)
else:
@@ -176,7 +177,7 @@ def launch_worker(
"--effort",
plan["effort"],
"--permission-mode",
- "acceptEdits",
+ "auto",
]
)
started = _run(command, env)
diff --git a/services/hermes/scripts/hermes_stt_client.py b/services/hermes/scripts/hermes_stt_client.py
new file mode 100644
index 000000000..da9228ad2
--- /dev/null
+++ b/services/hermes/scripts/hermes_stt_client.py
@@ -0,0 +1,76 @@
+#!/usr/bin/env python3
+"""Send one Hermes local-command STT request to the private Whisper service."""
+
+from __future__ import annotations
+
+import argparse
+import json
+import mimetypes
+import os
+import secrets
+from pathlib import Path
+from urllib.request import Request, urlopen
+
+
+def _multipart(audio: Path, language: str, model: str) -> tuple[bytes, str]:
+ boundary = f"atlas-hermes-{secrets.token_hex(12)}"
+ mime = mimetypes.guess_type(audio.name)[0] or "application/octet-stream"
+ chunks: list[bytes] = []
+
+ def field(name: str, value: str) -> None:
+ chunks.extend(
+ [
+ f"--{boundary}\r\n".encode(),
+ f'Content-Disposition: form-data; name="{name}"\r\n\r\n'.encode(),
+ value.encode(),
+ b"\r\n",
+ ]
+ )
+
+ field("language", language)
+ field("model", model)
+ chunks.extend(
+ [
+ f"--{boundary}\r\n".encode(),
+ f'Content-Disposition: form-data; name="file"; filename="{audio.name}"\r\n'.encode(),
+ f"Content-Type: {mime}\r\n\r\n".encode(),
+ audio.read_bytes(),
+ b"\r\n",
+ f"--{boundary}--\r\n".encode(),
+ ]
+ )
+ return b"".join(chunks), boundary
+
+
+def main() -> None:
+ """Transcribe one file and emit the .txt contract Hermes expects."""
+ parser = argparse.ArgumentParser()
+ parser.add_argument("input_path", type=Path)
+ parser.add_argument("--output-dir", required=True, type=Path)
+ parser.add_argument("--language", default="auto")
+ parser.add_argument("--model", default="large-v3-turbo")
+ args = parser.parse_args()
+
+ body, boundary = _multipart(args.input_path, args.language, args.model)
+ request = Request(
+ os.getenv(
+ "HERMES_STT_URL",
+ "http://hermes-stt.hermes.svc.cluster.local:9000/v1/audio/transcriptions",
+ ),
+ data=body,
+ headers={
+ "Content-Type": f"multipart/form-data; boundary={boundary}",
+ "Accept": "application/json",
+ },
+ method="POST",
+ )
+ with urlopen(request, timeout=120) as response:
+ result = json.loads(response.read().decode("utf-8"))
+ transcript = str(result.get("text") or "").strip()
+ args.output_dir.mkdir(parents=True, exist_ok=True)
+ output = args.output_dir / f"{args.input_path.stem}.txt"
+ output.write_text(transcript, encoding="utf-8")
+
+
+if __name__ == "__main__":
+ main()
diff --git a/services/hermes/voice-deployment.yaml b/services/hermes/voice-deployment.yaml
new file mode 100644
index 000000000..63f087c2f
--- /dev/null
+++ b/services/hermes/voice-deployment.yaml
@@ -0,0 +1,170 @@
+# services/hermes/voice-deployment.yaml
+apiVersion: apps/v1
+kind: Deployment
+metadata:
+ name: hermes-stt
+ namespace: hermes
+ labels:
+ app: hermes-stt
+spec:
+ replicas: 1
+ strategy:
+ type: Recreate
+ selector:
+ matchLabels:
+ app: hermes-stt
+ template:
+ metadata:
+ labels:
+ app: hermes-stt
+ annotations:
+ ai.bstein.dev/role: private-chat-speech-to-text
+ ai.bstein.dev/model: whisper-turbo
+ ai.bstein.dev/gpu: titan-21 dedicated speech time-slice
+ spec:
+ runtimeClassName: nvidia
+ nodeSelector:
+ kubernetes.io/hostname: titan-21
+ containers:
+ - name: stt
+ image: registry.bstein.dev/bstein/hermes-jetson-stt@sha256:a67494014716a65152b7595122b1717a742db206c6f556c740fecc110452f65b
+ imagePullPolicy: IfNotPresent
+ ports:
+ - {name: http, containerPort: 9000, protocol: TCP}
+ env:
+ - {name: HOME, value: /tmp}
+ - {name: XDG_CACHE_HOME, value: /tmp/cache}
+ - {name: HERMES_STT_MODEL, value: large-v3-turbo}
+ - {name: HERMES_STT_CACHE, value: /opt/models/whisper}
+ - {name: NVIDIA_VISIBLE_DEVICES, value: all}
+ - {name: NVIDIA_DRIVER_CAPABILITIES, value: compute,utility}
+ - {name: JETSON_JETPACK, value: "5"}
+ startupProbe:
+ httpGet: {path: /health, port: http}
+ periodSeconds: 10
+ timeoutSeconds: 5
+ failureThreshold: 90
+ readinessProbe:
+ httpGet: {path: /health, port: http}
+ periodSeconds: 10
+ timeoutSeconds: 5
+ livenessProbe:
+ httpGet: {path: /health, port: http}
+ periodSeconds: 30
+ timeoutSeconds: 10
+ failureThreshold: 3
+ securityContext:
+ allowPrivilegeEscalation: false
+ capabilities:
+ drop: [ALL]
+ readOnlyRootFilesystem: true
+ runAsUser: 10000
+ runAsGroup: 10000
+ seccompProfile:
+ type: RuntimeDefault
+ volumeMounts:
+ - {name: tmp, mountPath: /tmp}
+ resources:
+ requests:
+ cpu: "2"
+ memory: 4Gi
+ nvidia.com/gpu.shared: 1
+ limits:
+ cpu: "6"
+ memory: 10Gi
+ nvidia.com/gpu.shared: 1
+ volumes:
+ - name: tmp
+ emptyDir:
+ sizeLimit: 1Gi
+---
+apiVersion: v1
+kind: Service
+metadata:
+ name: hermes-stt
+ namespace: hermes
+spec:
+ selector:
+ app: hermes-stt
+ ports:
+ - {name: http, port: 9000, targetPort: http, protocol: TCP}
+---
+apiVersion: apps/v1
+kind: Deployment
+metadata:
+ name: hermes-tts
+ namespace: hermes
+ labels:
+ app: hermes-tts
+spec:
+ replicas: 1
+ strategy:
+ type: Recreate
+ selector:
+ matchLabels:
+ app: hermes-tts
+ template:
+ metadata:
+ labels:
+ app: hermes-tts
+ annotations:
+ ai.bstein.dev/role: private-chat-text-to-speech
+ ai.bstein.dev/model: piper-en-us-lessac-high
+ ai.bstein.dev/gpu: CPU-only on routing node
+ spec:
+ nodeSelector:
+ kubernetes.io/hostname: titan-20
+ containers:
+ - name: tts
+ image: registry.bstein.dev/bstein/hermes-jetson-tts@sha256:6483ca4e89d0663c770b7b3e3c9c9ff85d457ae877f2e83c55765cadc901d9b7
+ imagePullPolicy: IfNotPresent
+ ports:
+ - {name: http, containerPort: 9001, protocol: TCP}
+ env:
+ - {name: HOME, value: /tmp}
+ - {name: XDG_CACHE_HOME, value: /tmp/cache}
+ - {name: HERMES_TTS_VOICE, value: en_US-lessac-high}
+ - {name: HERMES_TTS_CACHE, value: /opt/models/piper}
+ startupProbe:
+ httpGet: {path: /health, port: http}
+ periodSeconds: 5
+ timeoutSeconds: 3
+ failureThreshold: 60
+ readinessProbe:
+ httpGet: {path: /health, port: http}
+ periodSeconds: 10
+ timeoutSeconds: 5
+ livenessProbe:
+ httpGet: {path: /health, port: http}
+ periodSeconds: 30
+ timeoutSeconds: 10
+ failureThreshold: 3
+ securityContext:
+ allowPrivilegeEscalation: false
+ capabilities:
+ drop: [ALL]
+ readOnlyRootFilesystem: true
+ runAsUser: 10000
+ runAsGroup: 10000
+ seccompProfile:
+ type: RuntimeDefault
+ volumeMounts:
+ - {name: tmp, mountPath: /tmp}
+ resources:
+ requests: {cpu: 500m, memory: 512Mi}
+ limits: {cpu: "2", memory: 2Gi}
+ volumes:
+ - name: tmp
+ emptyDir:
+ sizeLimit: 512Mi
+---
+apiVersion: v1
+kind: Service
+metadata:
+ name: hermes-tts
+ namespace: hermes
+spec:
+ selector:
+ app: hermes-tts
+ ports:
+ - {name: http, port: 9001, targetPort: http, protocol: TCP}
diff --git a/testing/tests/test_hermes_chat_quality.py b/testing/tests/test_hermes_chat_quality.py
index f7850f871..71fc1819c 100644
--- a/testing/tests/test_hermes_chat_quality.py
+++ b/testing/tests/test_hermes_chat_quality.py
@@ -41,44 +41,68 @@ def test_chat_config_enables_real_research_compute_and_delegation():
assert "code_execution" not in toolsets
-def test_sandbox_has_no_credentials_token_or_egress():
+def test_sandbox_shares_only_the_tenant_workspace_without_credentials():
sandbox_docs = _documents(HERMES / "chat-sandbox.yaml")
- statefulset = next(doc for doc in sandbox_docs if doc["kind"] == "StatefulSet")
- pod_spec = statefulset["spec"]["template"]["spec"]
- container = pod_spec["containers"][0]
+ deployments = [doc for doc in sandbox_docs if doc["kind"] == "Deployment"]
+ assert len(deployments) == 4
+ for ordinal, deployment in enumerate(deployments):
+ pod_spec = deployment["spec"]["template"]["spec"]
+ container = pod_spec["containers"][0]
+ assert pod_spec["automountServiceAccountToken"] is False
+ assert container["securityContext"]["readOnlyRootFilesystem"] is True
+ assert container["securityContext"]["runAsNonRoot"] is True
+ assert container["securityContext"]["runAsGroup"] == 10000
+ assert not container.get("env")
+ assert {mount["mountPath"] for mount in container["volumeMounts"]} == {
+ "/tmp",
+ "/workspace",
+ "/opt/data/workspace",
+ }
+ workspace_volume = next(
+ item for item in pod_spec["volumes"] if item["name"] == "workspace"
+ )
+ assert workspace_volume["persistentVolumeClaim"]["claimName"] == (
+ f"workspace-hermes-chat-tenant-{ordinal}"
+ )
- assert pod_spec["automountServiceAccountToken"] is False
- assert container["securityContext"]["readOnlyRootFilesystem"] is True
- assert container["securityContext"]["runAsNonRoot"] is True
- assert not container.get("env")
- assert {mount["mountPath"] for mount in container["volumeMounts"]} == {
- "/tmp",
- "/workspace",
- }
+ statefulset = _documents(HERMES / "chat-statefulset.yaml")[0]
+ templates = statefulset["spec"]["volumeClaimTemplates"]
+ workspace = next(item for item in templates if item["metadata"]["name"] == "workspace")
+ assert workspace["spec"]["resources"]["requests"]["storage"] == "10Gi"
+ assert workspace["spec"]["accessModes"] == ["ReadWriteMany"]
+
+ pod_spec = statefulset["spec"]["template"]["spec"]
+ hermes = next(item for item in pod_spec["containers"] if item["name"] == "hermes")
+ startup = hermes["args"][0]
+ assert "hermes-chat-sandbox-${ordinal}.hermes-chat-sandbox" in startup
+ assert any(
+ mount["name"] == "workspace" and mount["mountPath"] == "/opt/data/workspace"
+ for mount in hermes["volumeMounts"]
+ )
policies = _documents(HERMES / "networkpolicy.yaml")
deny = next(
- doc
- for doc in policies
- if doc["kind"] == "NetworkPolicy"
- and doc["metadata"]["name"] == "hermes-chat-sandbox-deny"
+ item
+ for item in policies
+ if item["metadata"]["name"] == "hermes-chat-sandbox-deny"
)
- assert deny["spec"]["policyTypes"] == ["Ingress", "Egress"]
+ assert deny["spec"]["ingress"] == []
assert deny["spec"]["egress"] == []
for ordinal in range(4):
policy = next(
- doc
- for doc in policies
- if doc["kind"] == "NetworkPolicy"
- and doc["metadata"]["name"] == f"hermes-chat-sandbox-tenant-{ordinal}"
+ item
+ for item in policies
+ if item["metadata"]["name"] == f"hermes-chat-sandbox-tenant-{ordinal}"
)
assert policy["spec"]["podSelector"]["matchLabels"][
- "apps.kubernetes.io/pod-index"
+ "ai.bstein.dev/tenant-ordinal"
] == str(ordinal)
source = policy["spec"]["ingress"][0]["from"][0]["podSelector"][
"matchLabels"
]
- assert source["apps.kubernetes.io/pod-index"] == str(ordinal)
+ assert source["statefulset.kubernetes.io/pod-name"] == (
+ f"hermes-chat-tenant-{ordinal}"
+ )
def test_gateway_image_honors_ui_model_and_caps_reasoning():
@@ -114,6 +138,100 @@ def test_chat_oauth_allows_stale_service_worker_retirement():
assert "--cookie-csrf-per-request=true" in args
assert "--cookie-csrf-per-request-limit=8" in args
assert "--trusted-proxy-ip=10.42.0.0/16" in args
+ assert "--api-route=^/api/" in args
+ assert "--api-route=^/health$" in args
+ assert "--cookie-expire=168h" in args
+
+
+def test_webui_recovers_auth_and_labels_session_scoped_controls():
+ dockerfile = (ROOT / "dockerfiles" / "Dockerfile.hermes-webui").read_text()
+
+ assert "res.status===401||res.status===403" in dockerfile
+ assert "window.location.assign('/oauth2/start?rd='" in dockerfile
+ assert "childrenExpanded?'▾ ':'▸ '" in dockerfile
+ assert "profile default: ' + p.model" in dockerfile
+
+
+def test_chat_voice_uses_private_jetson_services_and_shared_auto_route():
+ statefulset = _documents(HERMES / "chat-statefulset.yaml")[0]
+ containers = statefulset["spec"]["template"]["spec"]["containers"]
+ webui = next(item for item in containers if item["name"] == "webui")
+ env = {item["name"]: item["value"] for item in webui["env"]}
+
+ assert env["HERMES_STT_URL"] == (
+ "http://hermes-stt.hermes.svc.cluster.local:9000/v1/audio/transcriptions"
+ )
+ assert env["HERMES_WEBUI_ATLAS_TTS_URL"] == (
+ "http://hermes-tts.hermes.svc.cluster.local:9001/v1/audio/speech"
+ )
+ assert "hermes_stt_client.py" in env["HERMES_LOCAL_STT_COMMAND"]
+
+ configmap = _documents(HERMES / "chat-configmap.yaml")[0]
+ config = yaml.safe_load(configmap["data"]["config.yaml"])
+ assert config["stt"] == {
+ "enabled": True,
+ "provider": "local_command",
+ "local": {"model": "large-v3-turbo", "language": "auto"},
+ }
+
+ dockerfile = (ROOT / "dockerfiles" / "Dockerfile.hermes-webui").read_text()
+ assert "hermes-webui-atlas-patch.py" in dockerfile
+ assert "hermes-webui-atlas-voice.js" in dockerfile
+ voice_script = (ROOT / "dockerfiles" / "hermes-webui-atlas-voice.js").read_text()
+ assert "/api/transcribe/capability" in voice_script
+ assert "/api/transcribe" in voice_script
+ assert "/api/tts" in voice_script
+ assert "speakResponse(generation)" in voice_script
+ assert "restartSoon(token,450)" in voice_script
+
+
+def test_voice_models_are_baked_and_runtime_has_no_public_egress():
+ stt_dockerfile = (ROOT / "dockerfiles" / "Dockerfile.hermes-jetson-stt").read_text()
+ tts_dockerfile = (ROOT / "dockerfiles" / "Dockerfile.hermes-jetson-tts").read_text()
+ assert "ADD --checksum=sha256:aff26ae4" in stt_dockerfile
+ assert "--chmod=0444" in stt_dockerfile
+ assert "HERMES_STT_CACHE=/opt/models/whisper" in stt_dockerfile
+ assert "ADD --checksum=sha256:4cabf7c3" in tts_dockerfile
+ assert "ADD --checksum=sha256:db42b97d" in tts_dockerfile
+ assert tts_dockerfile.count("--chmod=0444") == 2
+ assert "HERMES_TTS_CACHE=/opt/models/piper" in tts_dockerfile
+ tts_server = (ROOT / "dockerfiles" / "hermes-jetson-tts-server.py").read_text()
+ assert "download_voice" not in tts_server
+ assert "baked Piper voice is missing" in tts_server
+
+ policies = _documents(HERMES / "networkpolicy.yaml")
+ voice_policy = next(
+ item for item in policies if item["metadata"]["name"] == "hermes-private-voice"
+ )
+ assert voice_policy["spec"]["policyTypes"] == ["Ingress", "Egress"]
+ assert not any(
+ "ipBlock" in destination
+ for rule in voice_policy["spec"]["egress"]
+ for destination in rule.get("to", [])
+ )
+
+
+def test_voice_workloads_have_deliberate_xavier_placement():
+ documents = _documents(HERMES / "voice-deployment.yaml")
+ deployments = {
+ item["metadata"]["name"]: item
+ for item in documents
+ if item["kind"] == "Deployment"
+ }
+ stt = deployments["hermes-stt"]["spec"]["template"]["spec"]
+ tts = deployments["hermes-tts"]["spec"]["template"]["spec"]
+
+ assert "@sha256:" in stt["containers"][0]["image"]
+ assert "@sha256:" in tts["containers"][0]["image"]
+ assert stt["nodeSelector"] == {"kubernetes.io/hostname": "titan-21"}
+ assert tts["nodeSelector"] == {"kubernetes.io/hostname": "titan-20"}
+ assert stt["runtimeClassName"] == "nvidia"
+ stt_resources = stt["containers"][0]["resources"]
+ assert stt_resources["requests"]["nvidia.com/gpu.shared"] == 1
+ assert stt_resources["limits"]["nvidia.com/gpu.shared"] == 1
+ assert "nvidia.com/gpu.shared" not in tts["containers"][0]["resources"]["requests"]
+ assert all("hostPath" not in volume for volume in stt["volumes"])
+ assert all("hostPath" not in volume for volume in tts["volumes"])
def test_chat_auth_file_mount_survives_atomic_provider_refresh():
diff --git a/testing/tests/test_hermes_herdr.py b/testing/tests/test_hermes_herdr.py
index d64e7d7c4..fa0baa659 100644
--- a/testing/tests/test_hermes_herdr.py
+++ b/testing/tests/test_hermes_herdr.py
@@ -82,7 +82,7 @@ def test_claude_worker_waits_for_prompt_readiness(tmp_path: Path, monkeypatch):
"wait-output",
"w2:p1",
"--match",
- "accept edits on",
+ "Claude Code",
"--source",
"recent",
"--lines",
@@ -94,7 +94,7 @@ def test_claude_worker_waits_for_prompt_readiness(tmp_path: Path, monkeypatch):
assert prompt[1:5] == [
"agent",
"prompt",
- "claude-review",
+ "project-claude-review",
"Check the implementation.",
]
assert prompt[-9:] == [
@@ -109,6 +109,53 @@ def test_claude_worker_waits_for_prompt_readiness(tmp_path: Path, monkeypatch):
"15000",
]
+ workspace_create = calls[0]
+ assert workspace_create[-3:] == ["--label", "project-claude-review", "--no-focus"]
+ start = calls[1]
+ assert start[1:4] == ["agent", "start", "project-claude-review"]
+ assert start[-2:] == ["--permission-mode", "auto"]
+
+
+def test_codex_worker_uses_automatic_review_without_prompts(tmp_path: Path, monkeypatch):
+ herdr = tmp_path / "herdr"
+ herdr.touch()
+ project = tmp_path / "cassandra"
+ project.mkdir()
+ auth = tmp_path / "auth.json"
+ auth.write_text("{}", encoding="utf-8")
+ calls = []
+
+ def fake_run(command, env):
+ calls.append(command)
+ if command[1:3] == ["workspace", "create"]:
+ return {"result": {"root_pane": {"pane_id": "w3:p1"}}}
+ return {"result": {"ok": True}}
+
+ monkeypatch.setattr(dispatch, "HERDR_BIN", herdr)
+ monkeypatch.setattr(dispatch, "CODEX_AUTH", auth)
+ monkeypatch.setattr(dispatch, "_run", fake_run)
+ plan = {"worker": "codex", "model": "gpt-5.6-sol", "effort": "high"}
+
+ dispatch.launch_worker(plan, project, "api-fix", None)
+
+ assert calls[0][-3:] == ["--label", "cassandra-codex-api-fix", "--no-focus"]
+ start = calls[1]
+ assert start[1:4] == ["agent", "start", "cassandra-codex-api-fix"]
+ assert "--approve-for-me" in start
+ assert start[-2:] == ["--sandbox", "workspace-write"]
+ assert not any("on-request" in value for value in start)
+
+
+def test_agent_config_uses_bounded_noninteractive_approvals():
+ configmap = yaml.safe_load((HERMES / "agent-configmap.yaml").read_text())
+ config = yaml.safe_load(configmap["data"]["config.yaml"])
+
+ assert config["approvals"]["mode"] == "off"
+ denied = config["approvals"]["deny"]
+ assert "*kubectl apply*" in denied
+ assert "*git push --force*" in denied
+ assert "*git reset --hard*" in denied
+
def test_auth_patch_honors_explicit_shared_store(tmp_path: Path):
source = tmp_path / "auth.py"