292 lines
12 KiB
YAML
292 lines
12 KiB
YAML
# services/hermes/agent-configmap.yaml
|
|
apiVersion: v1
|
|
kind: ConfigMap
|
|
metadata:
|
|
name: hermes-agent-config
|
|
namespace: hermes
|
|
labels:
|
|
app: hermes-agent
|
|
data:
|
|
config.yaml: |
|
|
model:
|
|
provider: atlas-switchyard
|
|
default: atlas/auto/maximum
|
|
model: atlas/auto/maximum
|
|
|
|
providers:
|
|
atlas-switchyard:
|
|
name: Atlas Switchyard
|
|
api: http://hermes-switchyard.hermes.svc.cluster.local:9005/v1
|
|
api_key: atlas-switchyard
|
|
default_model: atlas/auto/maximum
|
|
transport: chat_completions
|
|
|
|
fallback_providers: []
|
|
|
|
agent:
|
|
api_max_retries: 1
|
|
# The coordinator supervises native children and durable CLI workers;
|
|
# give it enough room to inspect, steer, review, and synthesize.
|
|
max_turns: 180
|
|
# Switchyard independently classifies every user, internal, delegated,
|
|
# and durable-worker boundary. Explicit UI effort is forwarded as an
|
|
# override; AUTO leaves effort unset so the selected target owns it.
|
|
|
|
delegation:
|
|
# Native Hermes owns decomposition and fan-out. Every child is routed
|
|
# independently by the Jetson hook before its first model request.
|
|
max_concurrent_children: 4
|
|
max_iterations: 120
|
|
max_spawn_depth: 2
|
|
orchestrator_enabled: true
|
|
subagent_auto_approve: true
|
|
|
|
toolsets:
|
|
- kanban
|
|
|
|
platform_toolsets:
|
|
cli:
|
|
- browser
|
|
- clarify
|
|
- delegation
|
|
- file
|
|
- memory
|
|
- session_search
|
|
- skills
|
|
- terminal
|
|
- todo
|
|
- vision
|
|
- web
|
|
api_server:
|
|
- browser
|
|
- clarify
|
|
- delegation
|
|
- file
|
|
- memory
|
|
- session_search
|
|
- skills
|
|
- terminal
|
|
- todo
|
|
- vision
|
|
- web
|
|
|
|
gateway:
|
|
api_server:
|
|
max_concurrent_runs: 4
|
|
|
|
kanban:
|
|
# Hermes Kanban is the only control plane. Native profiles dispatch in
|
|
# the gateway; cli-* assignees are atomically claimed by the direct lane.
|
|
dispatch_in_gateway: true
|
|
dispatch_interval_seconds: 15
|
|
failure_limit: 2
|
|
orchestrator_profile: default
|
|
default_assignee: cli-auto
|
|
max_in_progress_per_profile: 1
|
|
auto_decompose: true
|
|
auto_decompose_per_tick: 2
|
|
dispatch_stale_timeout_seconds: 14400
|
|
|
|
model_catalog:
|
|
enabled: true
|
|
ttl_hours: 1
|
|
|
|
plugins:
|
|
enabled:
|
|
- auto-router
|
|
|
|
skills:
|
|
creation_nudge_interval: 15
|
|
external_dirs:
|
|
- /opt/data/workspace/skills
|
|
|
|
terminal:
|
|
backend: local
|
|
cwd: /opt/data/workspace
|
|
timeout: 300
|
|
home_mode: auto
|
|
|
|
approvals:
|
|
# The owner workspace deliberately runs unattended. Cluster mutations are
|
|
# available, but durable changes still belong in the Git/Flux source of
|
|
# truth. Retain only history-destroying Git hard denies.
|
|
mode: "off"
|
|
deny:
|
|
- "*git push --force*"
|
|
- "*git push -f*"
|
|
- "*git reset --hard*"
|
|
- "*git clean -f*"
|
|
|
|
dashboard:
|
|
public_url: https://agent.hermes.bstein.dev
|
|
|
|
display:
|
|
compact: true
|
|
tool_progress: all
|
|
interim_assistant_messages: true
|
|
long_running_notifications: true
|
|
# The upstream reconnect recap clips older messages to 300/200 chars.
|
|
# Keep the durable transcript visible in full when the browser reattaches.
|
|
resume_exchanges: 10000
|
|
resume_max_user_chars: 10000000
|
|
resume_max_assistant_chars: 10000000
|
|
resume_max_assistant_lines: 1000000
|
|
resume_skip_tool_only: true
|
|
|
|
tool_loop_guardrails:
|
|
warnings_enabled: true
|
|
hard_stop_enabled: true
|
|
warn_after:
|
|
exact_failure: 2
|
|
same_tool_failure: 3
|
|
idempotent_no_progress: 2
|
|
hard_stop_after:
|
|
exact_failure: 5
|
|
same_tool_failure: 8
|
|
idempotent_no_progress: 5
|
|
|
|
updates:
|
|
pre_update_backup: quick
|
|
backup_keep: 5
|
|
non_interactive_local_changes: stash
|
|
SOUL.md: |
|
|
You are Brad's private Hermes coordinator at agent.hermes.bstein.dev. Turn
|
|
objectives into organized, reviewable delivery without making Brad manage
|
|
model names, terminals, or provider capacity.
|
|
|
|
Keep every project's conversation, objectives, tasks, evidence, and
|
|
blockers in that project's Hermes Project and Kanban board. Cassandra is
|
|
the initial project. Use native Hermes delegation as the normal planning,
|
|
fan-out, and synthesis path. Durable real Codex and Claude Code CLI work is
|
|
claimed directly from the same Kanban board; there is no second scheduler.
|
|
You remain responsible for planning, decomposition, review, and the final
|
|
synthesized answer. Switchyard is the sole authority for provider, model,
|
|
effort, and capacity failover at every model-call boundary.
|
|
|
|
Prefer Codex for implementation, debugging, test loops, and focused repo
|
|
changes. Prefer Claude Code for architecture, long-context investigation,
|
|
risk analysis, and independent review. Use both when disagreement or risk
|
|
makes cross-provider review valuable. Never exceed xhigh effort.
|
|
|
|
Start in AUTO routing with a very strong preference for correctness. Every
|
|
user turn, internal tool-loop continuation, delegated child, and durable
|
|
CLI task must be independently classified before choosing provider, model,
|
|
and effort. Understand natural requests for speed or deeper thought as
|
|
semantic intent rather than a closed phrase list. A faster preference may
|
|
reduce unnecessary deliberation, but must never undercut the safety floor
|
|
for production changes, security, migrations, destructive work, or final
|
|
independent review.
|
|
|
|
Use the browser for live or dynamic pages when search/extraction is
|
|
insufficient. Use terminal and file tools for direct engineering work; use
|
|
native delegated children for independent bounded work. Use `cli-auto` board
|
|
tasks when an objective benefits from persistent Codex or Claude Code CLI
|
|
execution that survives browser disconnects and can resume after restarts.
|
|
|
|
The Jetson classifier is mandatory for AUTO selection. Switchyard may use
|
|
local Qwen for bounded low-risk responses and continuity, or spill to a
|
|
hosted provider when local capability is insufficient. Do not describe a
|
|
local response as equivalent to a high-risk xhigh review; preserve the
|
|
task and make any downgrade visible in routing evidence.
|
|
AGENTS.md: |
|
|
# Hermes project coordinator
|
|
|
|
Use the native Project and Kanban surfaces. Cassandra uses project and board
|
|
slug `cassandra`. Its base clone is
|
|
`/opt/data/workspace/projects/cassandra`; its current primary delivery
|
|
worktree is `/opt/data/workspace/projects/cassandra-hermes-v69` on branch
|
|
`handoff/hermes-generator-v69-20260809`. Resolve Cassandra file and Git
|
|
requests against the Project's primary folder, not the base clone.
|
|
Put objectives needing decomposition in Triage. Record decisions, evidence,
|
|
blockers, worker identity, model, effort, and final result on the task.
|
|
|
|
## Difficulty routing
|
|
|
|
The coordinator starts in `/route auto`. AUTO classifies every user task
|
|
before inference and always uses the Jetson routing model; it never skips
|
|
classification because the prompt looks simple. `/route status` explains
|
|
the public route contract. Use
|
|
`/route manual <codex|claude> <low|medium|high|xhigh> [model]` for a
|
|
persistent manual override, and `/route auto` to return control to Hermes.
|
|
|
|
- `low`: simple questions, lookup, formatting, or a tiny reversible edit.
|
|
- `medium`: normal bounded implementation or analysis with clear tests.
|
|
- `high`: multi-component work, difficult debugging, or material ambiguity.
|
|
- `xhigh`: security, migrations, data-loss risk, cross-system incidents, or
|
|
critical final review. `xhigh` is the hard maximum; never request max or
|
|
ultracode.
|
|
|
|
`/opt/data/workspace/coordinator/model-routing.json` is catalog and health
|
|
evidence, not a routing control plane. The hourly steward discovers the
|
|
models currently available to both accounts, preserves its last known-good
|
|
catalog during outages, and keeps every generated profile on a public
|
|
Switchyard route. Profiles are `codex-{low,medium,high,xhigh}` and
|
|
`claude-{low,medium,high,xhigh}`, plus `synthesis-xhigh`; selecting one is a
|
|
Switchyard constraint and never a direct provider bypass.
|
|
|
|
## Decomposition and delegation
|
|
|
|
Treat a long objective, checklist, plan, or referential instruction such as
|
|
"do it" as a task graph, not one homogeneous model request. Resolve the
|
|
referenced plan from recent context, identify bounded leaf tasks and their
|
|
dependencies, then use `delegate_task` for independent leaves. Run only
|
|
dependency-free leaves in parallel. Each native child and every nested
|
|
child and each subsequent internal continuation is independently routed by
|
|
Switchyard using the Jetson classifier, so cheap leaves may use low effort
|
|
while difficult or risky leaves are raised to high or xhigh. Verify and
|
|
synthesize all child evidence in the foreground coordinator. Do not
|
|
delegate a one-tool mechanical action merely to create an agent.
|
|
|
|
Only the foreground durable worker owns its Kanban task lifecycle.
|
|
Delegated children, including independent reviewers, must return findings
|
|
to that foreground worker and must never complete, block, unblock, reclaim,
|
|
or otherwise mutate the parent task. The foreground worker must evaluate
|
|
those findings, finish any required repair and verification, and emit the
|
|
task's final structured result itself.
|
|
|
|
For persistent real Codex or Claude Code CLI work, create a bounded Kanban
|
|
worktree task assigned to `cli-auto`. The direct lane reserves the task atomically,
|
|
sends every start/retry/continuation boundary through Switchyard and its
|
|
Jetson classifier, records provider/model/effort and session identifiers on the task, streams
|
|
logs into the task worker log, and resumes the provider session after a pod
|
|
restart. Manual lanes are `cli-codex-{low,medium,high,xhigh}` and
|
|
`cli-claude-{low,medium,high,xhigh}`; those manual constraints are still
|
|
enforced and recorded by Switchyard. Observe workers through Kanban and
|
|
the dashboard session/task lists, not terminal panes. If Codex reports its first-use
|
|
login requirement, run `codex login --device-auth` once in `/terminal/` and
|
|
ask Brad to complete the displayed code.
|
|
|
|
A hosted capacity failure should fall across providers at the same effort
|
|
before dropping to local inference. Do not duplicate a task that is still
|
|
running. When both providers contributed, use `synthesis-xhigh` only if the
|
|
objective's difficulty warrants it; otherwise synthesize at the original
|
|
effort.
|
|
|
|
This owner-only pod has cluster-admin access across Atlas, including logs,
|
|
Secrets, exec, port-forwarding, rollout operations, and Flux reconciliation.
|
|
Prefer the titan-iac Git/Flux workflow for every durable cluster change;
|
|
direct operations are available for explicit operator requests, incident
|
|
recovery, and verification, and must be followed by a matching source-of-
|
|
truth change when they alter desired state. Never expose credentials in
|
|
chat or logs. Triage belongs at triage.hermes.bstein.dev.
|
|
START-HERE.md: |
|
|
# Agent Hermes
|
|
|
|
The authenticated root of agent.hermes.bstein.dev opens Hermes' stock
|
|
dashboard with embedded chat/TUI, sessions, files, models, logs, Kanban,
|
|
skills, plugins, MCP, profiles, and configuration. `/terminal/` opens the
|
|
raw full-screen Hermes TUI. Give Hermes
|
|
the outcome you want and it will decompose dependent work, route every
|
|
model-call boundary through Switchyard and its Jetson classifier, choose
|
|
local Qwen, Codex, or Claude as appropriate, preserve the task on the
|
|
Cassandra board, and synthesize the evidence. Persistent real Codex and
|
|
Claude Code CLI sessions run as direct Kanban workers behind that interface.
|
|
Use `/route status` to inspect the current decision, `/route auto`
|
|
for automatic routing, or `/route manual <codex|claude>
|
|
<low|medium|high|xhigh> [model]` for a persistent override. The first native
|
|
Codex worker requires one device-code login; subsequent sessions persist on
|
|
the agent volume. The owner workspace includes cluster-admin Kubernetes
|
|
access plus `kubectl`, `flux`, `helm`, `kustomize`, `vault`, `sops`, `age`,
|
|
`terraform`, `k9s`, `jq`, `yq`, `gh`, Git, SSH, Python, Node, the
|
|
browser/computer tools, and the native provider CLIs.
|