diff --git a/knowledge/catalog/atlas-summary.json b/knowledge/catalog/atlas-summary.json index 9acf7417..05ba796d 100644 --- a/knowledge/catalog/atlas-summary.json +++ b/knowledge/catalog/atlas-summary.json @@ -1,8 +1,8 @@ { "counts": { "helmrelease_host_hints": 23, - "http_endpoints": 61, - "services": 91, - "workloads": 125 + "http_endpoints": 63, + "services": 98, + "workloads": 130 } } diff --git a/knowledge/catalog/atlas.json b/knowledge/catalog/atlas.json index d7247bfd..d6227f4c 100644 --- a/knowledge/catalog/atlas.json +++ b/knowledge/catalog/atlas.json @@ -106,6 +106,31 @@ "path": "services/hermes-chat", "targetNamespace": "hermes-chat" }, + { + "name": "hermes-observer-bindings", + "path": "services/hermes-observer-bindings", + "targetNamespace": null + }, + { + "name": "hermes-observer-rbac", + "path": "services/hermes-observer-rbac", + "targetNamespace": null + }, + { + "name": "hermes-scm-broker", + "path": "services/hermes-scm-broker", + "targetNamespace": "hermes-scm" + }, + { + "name": "hermes-scm-broker-code", + "path": "services/hermes/scm-common", + "targetNamespace": "hermes-scm" + }, + { + "name": "hermes-scm-namespace", + "path": "services/hermes-scm-namespace", + "targetNamespace": null + }, { "name": "hermes-triage-demo", "path": "services/hermes-triage-demo", @@ -251,6 +276,11 @@ "path": "infrastructure/vault-csi", "targetNamespace": "kube-system" }, + { + "name": "vault-hermes-jenkins-token-seed", + "path": "services/vault-hermes-jenkins-token-seed", + "targetNamespace": "vault" + }, { "name": "vault-injector", "path": "infrastructure/vault-injector", @@ -304,7 +334,7 @@ "node-role.kubernetes.io/worker": "true" }, "images": [ - "registry.bstein.dev/bstein/bstein-dev-home-backend:0.1.1-464" + "registry.bstein.dev/bstein/bstein-dev-home-backend:0.1.1-479" ] }, { @@ -320,7 +350,7 @@ "node-role.kubernetes.io/worker": "true" }, "images": [ - "registry.bstein.dev/bstein/bstein-dev-home-frontend:0.1.1-464" + "registry.bstein.dev/bstein/bstein-dev-home-frontend:0.1.1-479" ] }, { @@ -902,7 +932,7 @@ "serviceAccountName": "hermes-node-ssh-access", "nodeSelector": {}, "images": [ - "busybox:1.37" + "python@sha256:6d43704baacd1bfbe7c295d7f13079d5d8104ed33568873133f8fc69980419df" ] }, { @@ -915,8 +945,8 @@ "serviceAccountName": "hermes-triage", "nodeSelector": {}, "images": [ - "registry.bstein.dev/bstein/hermes-agent@sha256:cce1f65dc7d30fdce4b1748cc7d3434b9761ae9946a8c8be6049204c503d60a8", - "registry.bstein.dev/bstein/hermes-webui@sha256:9c2fe8341c7b650e08d10acead3151b19e2af737863268bafb39b3d9517575b1" + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37", + "registry.bstein.dev/bstein/hermes-webui@sha256:ac6ba7bfd8a86227f31a9a96ebea41227ccf70391f7dcf34206f4d58e835e50e" ] }, { @@ -930,7 +960,7 @@ "nodeSelector": {}, "images": [ "quay.io/oauth2-proxy/oauth2-proxy:v7.15.3@sha256:10a1165743a192e1940b4708fb9647027185ce11a681a1c5519b442ff7f1f561", - "registry.bstein.dev/bstein/hermes-agent@sha256:cce1f65dc7d30fdce4b1748cc7d3434b9761ae9946a8c8be6049204c503d60a8" + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37" ] }, { @@ -943,7 +973,7 @@ "serviceAccountName": "hermes-chat", "nodeSelector": {}, "images": [ - "registry.bstein.dev/bstein/hermes-chat-router@sha256:a4010fdc5b6dce2696e6fec06eb38a3992817faa01ff3472122c870038b22bd2" + "registry.bstein.dev/bstein/hermes-chat-router@sha256:6744cb7b87c6050f1b97c0675cd280b8295b3b5826ba1d6d92e37aee6fd0b8c4" ] }, { @@ -1058,6 +1088,48 @@ "registry.bstein.dev/bstein/hermes-chat-sandbox@sha256:17ee62b8e61c08573a3a8cca903b38ec43800cb44ec29340e1bc095176544bca" ] }, + { + "kind": "Deployment", + "namespace": "hermes", + "name": "hermes-execution-mediator-0", + "labels": { + "app": "hermes-execution-mediator", + "pool-ordinal": "0" + }, + "serviceAccountName": "hermes-execution-worker", + "nodeSelector": {}, + "images": [ + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37" + ] + }, + { + "kind": "Deployment", + "namespace": "hermes", + "name": "hermes-execution-mediator-1", + "labels": { + "app": "hermes-execution-mediator", + "pool-ordinal": "1" + }, + "serviceAccountName": "hermes-execution-worker", + "nodeSelector": {}, + "images": [ + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37" + ] + }, + { + "kind": "Deployment", + "namespace": "hermes", + "name": "hermes-execution-mediator-2", + "labels": { + "app": "hermes-execution-mediator", + "pool-ordinal": "2" + }, + "serviceAccountName": "hermes-execution-worker", + "nodeSelector": {}, + "images": [ + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37" + ] + }, { "kind": "Deployment", "namespace": "hermes", @@ -1177,8 +1249,36 @@ "serviceAccountName": "hermes-chat", "nodeSelector": {}, "images": [ - "registry.bstein.dev/bstein/hermes-agent@sha256:cce1f65dc7d30fdce4b1748cc7d3434b9761ae9946a8c8be6049204c503d60a8", - "registry.bstein.dev/bstein/hermes-webui@sha256:9c2fe8341c7b650e08d10acead3151b19e2af737863268bafb39b3d9517575b1" + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37", + "registry.bstein.dev/bstein/hermes-webui@sha256:c276a9e17c9057237472f39640c9ebac9d4359d9f62a34df52a759eef43167bf" + ] + }, + { + "kind": "StatefulSet", + "namespace": "hermes", + "name": "hermes-execution-worker", + "labels": { + "app": "hermes-execution-worker", + "app.kubernetes.io/name": "hermes-execution-worker", + "app.kubernetes.io/part-of": "hermes" + }, + "serviceAccountName": "hermes-execution-worker", + "nodeSelector": {}, + "images": [ + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37" + ] + }, + { + "kind": "Deployment", + "namespace": "hermes-scm", + "name": "hermes-scm-broker", + "labels": { + "app": "hermes-scm-broker" + }, + "serviceAccountName": "hermes-scm-broker", + "nodeSelector": {}, + "images": [ + "registry.bstein.dev/bstein/hermes-agent@sha256:37ebf720c783ae908a602916ffccf88d43d205a157957f5dc4b487867aee45e7" ] }, { @@ -1528,7 +1628,7 @@ "kubernetes.io/os": "linux" }, "images": [ - "registry.bstein.dev/bstein/metis-sentinel:0.1.0-293-amd64" + "registry.bstein.dev/bstein/metis-sentinel:0.1.0-304-amd64" ] }, { @@ -1544,7 +1644,7 @@ "kubernetes.io/os": "linux" }, "images": [ - "registry.bstein.dev/bstein/metis-sentinel:0.1.0-293-arm64" + "registry.bstein.dev/bstein/metis-sentinel:0.1.0-304-arm64" ] }, { @@ -1633,7 +1733,7 @@ "node-role.kubernetes.io/worker": "true" }, "images": [ - "registry.bstein.dev/bstein/ariadne:0.1.0-454" + "registry.bstein.dev/bstein/ariadne:0.1.0-464" ] }, { @@ -1661,7 +1761,7 @@ "serviceAccountName": "metis", "nodeSelector": {}, "images": [ - "registry.bstein.dev/bstein/metis:0.1.0-293-arm64" + "registry.bstein.dev/bstein/metis:0.1.0-304-arm64" ] }, { @@ -3179,6 +3279,22 @@ } ] }, + { + "namespace": "hermes", + "name": "hermes-cli-lane-metrics", + "type": "ClusterIP", + "selector": { + "app": "hermes-agent" + }, + "ports": [ + { + "name": "lane-metrics", + "port": 9011, + "targetPort": "lane-metrics", + "protocol": "TCP" + } + ] + }, { "namespace": "hermes", "name": "hermes-codex-broker", @@ -3195,6 +3311,82 @@ } ] }, + { + "namespace": "hermes", + "name": "hermes-execution-mediator-0", + "type": "ClusterIP", + "selector": { + "app": "hermes-execution-mediator", + "pool-ordinal": "0" + }, + "ports": [ + { + "name": "mediator", + "port": 9009, + "targetPort": "mediator", + "protocol": "TCP" + } + ] + }, + { + "namespace": "hermes", + "name": "hermes-execution-mediator-1", + "type": "ClusterIP", + "selector": { + "app": "hermes-execution-mediator", + "pool-ordinal": "1" + }, + "ports": [ + { + "name": "mediator", + "port": 9009, + "targetPort": "mediator", + "protocol": "TCP" + } + ] + }, + { + "namespace": "hermes", + "name": "hermes-execution-mediator-2", + "type": "ClusterIP", + "selector": { + "app": "hermes-execution-mediator", + "pool-ordinal": "2" + }, + "ports": [ + { + "name": "mediator", + "port": 9009, + "targetPort": "mediator", + "protocol": "TCP" + } + ] + }, + { + "namespace": "hermes", + "name": "hermes-execution-pool", + "type": "ClusterIP", + "selector": { + "app": "hermes-agent" + }, + "ports": [ + { + "name": "http", + "port": 9007, + "targetPort": "execution-pool", + "protocol": "TCP" + } + ] + }, + { + "namespace": "hermes", + "name": "hermes-execution-worker", + "type": "ClusterIP", + "selector": { + "app": "hermes-execution-worker" + }, + "ports": [] + }, { "namespace": "hermes", "name": "hermes-gpu-handoff", @@ -3393,6 +3585,22 @@ } ] }, + { + "namespace": "hermes-scm", + "name": "hermes-scm-broker", + "type": "ClusterIP", + "selector": { + "app": "hermes-scm-broker" + }, + "ports": [ + { + "name": "http", + "port": 9081, + "targetPort": "http", + "protocol": "TCP" + } + ] + }, { "namespace": "jellyfin", "name": "jellyfin", @@ -4423,6 +4631,26 @@ "source": "hermes" } }, + { + "host": "chat.hermes.bstein.dev", + "path": "/", + "backend": { + "namespace": "hermes", + "service": "oauth2-proxy-hermes-chat", + "port": "http", + "workloads": [ + { + "kind": "Deployment", + "name": "oauth2-proxy-hermes-chat" + } + ] + }, + "via": { + "kind": "Ingress", + "name": "hermes-sites", + "source": "hermes" + } + }, { "host": "ci.bstein.dev", "path": "/", @@ -5183,6 +5411,26 @@ "source": "hermes" } }, + { + "host": "triage.hermes.bstein.dev", + "path": "/", + "backend": { + "namespace": "hermes", + "service": "oauth2-proxy-hermes-triage", + "port": "http", + "workloads": [ + { + "kind": "Deployment", + "name": "oauth2-proxy-hermes-triage" + } + ] + }, + "via": { + "kind": "Ingress", + "name": "hermes-sites", + "source": "hermes" + } + }, { "host": "vault.bstein.dev", "path": "/", diff --git a/knowledge/catalog/atlas.yaml b/knowledge/catalog/atlas.yaml index 8ae1b49b..8e9ec24c 100644 --- a/knowledge/catalog/atlas.yaml +++ b/knowledge/catalog/atlas.yaml @@ -65,6 +65,21 @@ sources: - name: hermes-chat path: services/hermes-chat targetNamespace: hermes-chat +- name: hermes-observer-bindings + path: services/hermes-observer-bindings + targetNamespace: null +- name: hermes-observer-rbac + path: services/hermes-observer-rbac + targetNamespace: null +- name: hermes-scm-broker + path: services/hermes-scm-broker + targetNamespace: hermes-scm +- name: hermes-scm-broker-code + path: services/hermes/scm-common + targetNamespace: hermes-scm +- name: hermes-scm-namespace + path: services/hermes-scm-namespace + targetNamespace: null - name: hermes-triage-demo path: services/hermes-triage-demo targetNamespace: null @@ -152,6 +167,9 @@ sources: - name: vault-csi path: infrastructure/vault-csi targetNamespace: kube-system +- name: vault-hermes-jenkins-token-seed + path: services/vault-hermes-jenkins-token-seed + targetNamespace: vault - name: vault-injector path: infrastructure/vault-injector targetNamespace: vault @@ -187,7 +205,7 @@ workloads: kubernetes.io/arch: arm64 node-role.kubernetes.io/worker: 'true' images: - - registry.bstein.dev/bstein/bstein-dev-home-backend:0.1.1-464 + - registry.bstein.dev/bstein/bstein-dev-home-backend:0.1.1-479 - kind: Deployment namespace: bstein-dev-home name: bstein-dev-home-frontend @@ -198,7 +216,7 @@ workloads: kubernetes.io/arch: arm64 node-role.kubernetes.io/worker: 'true' images: - - registry.bstein.dev/bstein/bstein-dev-home-frontend:0.1.1-464 + - registry.bstein.dev/bstein/bstein-dev-home-frontend:0.1.1-479 - kind: Deployment namespace: bstein-dev-home name: bstein-dev-home-vault-sync @@ -604,7 +622,7 @@ workloads: serviceAccountName: hermes-node-ssh-access nodeSelector: {} images: - - busybox:1.37 + - python@sha256:6d43704baacd1bfbe7c295d7f13079d5d8104ed33568873133f8fc69980419df - kind: Deployment namespace: hermes name: hermes @@ -613,8 +631,8 @@ workloads: serviceAccountName: hermes-triage nodeSelector: {} images: - - registry.bstein.dev/bstein/hermes-agent@sha256:cce1f65dc7d30fdce4b1748cc7d3434b9761ae9946a8c8be6049204c503d60a8 - - registry.bstein.dev/bstein/hermes-webui@sha256:9c2fe8341c7b650e08d10acead3151b19e2af737863268bafb39b3d9517575b1 + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 + - registry.bstein.dev/bstein/hermes-webui@sha256:ac6ba7bfd8a86227f31a9a96ebea41227ccf70391f7dcf34206f4d58e835e50e - kind: Deployment namespace: hermes name: hermes-agent @@ -624,7 +642,7 @@ workloads: nodeSelector: {} images: - quay.io/oauth2-proxy/oauth2-proxy:v7.15.3@sha256:10a1165743a192e1940b4708fb9647027185ce11a681a1c5519b442ff7f1f561 - - registry.bstein.dev/bstein/hermes-agent@sha256:cce1f65dc7d30fdce4b1748cc7d3434b9761ae9946a8c8be6049204c503d60a8 + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 - kind: Deployment namespace: hermes name: hermes-chat-router @@ -633,7 +651,7 @@ workloads: serviceAccountName: hermes-chat nodeSelector: {} images: - - registry.bstein.dev/bstein/hermes-chat-router@sha256:a4010fdc5b6dce2696e6fec06eb38a3992817faa01ff3472122c870038b22bd2 + - registry.bstein.dev/bstein/hermes-chat-router@sha256:6744cb7b87c6050f1b97c0675cd280b8295b3b5826ba1d6d92e37aee6fd0b8c4 - kind: Deployment namespace: hermes name: hermes-chat-sandbox-0 @@ -714,6 +732,36 @@ workloads: nodeSelector: {} images: - registry.bstein.dev/bstein/hermes-chat-sandbox@sha256:17ee62b8e61c08573a3a8cca903b38ec43800cb44ec29340e1bc095176544bca +- kind: Deployment + namespace: hermes + name: hermes-execution-mediator-0 + labels: + app: hermes-execution-mediator + pool-ordinal: '0' + serviceAccountName: hermes-execution-worker + nodeSelector: {} + images: + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 +- kind: Deployment + namespace: hermes + name: hermes-execution-mediator-1 + labels: + app: hermes-execution-mediator + pool-ordinal: '1' + serviceAccountName: hermes-execution-worker + nodeSelector: {} + images: + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 +- kind: Deployment + namespace: hermes + name: hermes-execution-mediator-2 + labels: + app: hermes-execution-mediator + pool-ordinal: '2' + serviceAccountName: hermes-execution-worker + nodeSelector: {} + images: + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 - kind: Deployment namespace: hermes name: hermes-local-image @@ -797,8 +845,28 @@ workloads: serviceAccountName: hermes-chat nodeSelector: {} images: - - registry.bstein.dev/bstein/hermes-agent@sha256:cce1f65dc7d30fdce4b1748cc7d3434b9761ae9946a8c8be6049204c503d60a8 - - registry.bstein.dev/bstein/hermes-webui@sha256:9c2fe8341c7b650e08d10acead3151b19e2af737863268bafb39b3d9517575b1 + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 + - registry.bstein.dev/bstein/hermes-webui@sha256:c276a9e17c9057237472f39640c9ebac9d4359d9f62a34df52a759eef43167bf +- kind: StatefulSet + namespace: hermes + name: hermes-execution-worker + labels: + app: hermes-execution-worker + app.kubernetes.io/name: hermes-execution-worker + app.kubernetes.io/part-of: hermes + serviceAccountName: hermes-execution-worker + nodeSelector: {} + images: + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 +- kind: Deployment + namespace: hermes-scm + name: hermes-scm-broker + labels: + app: hermes-scm-broker + serviceAccountName: hermes-scm-broker + nodeSelector: {} + images: + - registry.bstein.dev/bstein/hermes-agent@sha256:37ebf720c783ae908a602916ffccf88d43d205a157957f5dc4b487867aee45e7 - kind: Deployment namespace: jellyfin name: jellyfin @@ -1037,7 +1105,7 @@ workloads: kubernetes.io/arch: amd64 kubernetes.io/os: linux images: - - registry.bstein.dev/bstein/metis-sentinel:0.1.0-293-amd64 + - registry.bstein.dev/bstein/metis-sentinel:0.1.0-304-amd64 - kind: DaemonSet namespace: maintenance name: metis-sentinel-arm64 @@ -1048,7 +1116,7 @@ workloads: kubernetes.io/arch: arm64 kubernetes.io/os: linux images: - - registry.bstein.dev/bstein/metis-sentinel:0.1.0-293-arm64 + - registry.bstein.dev/bstein/metis-sentinel:0.1.0-304-arm64 - kind: DaemonSet namespace: maintenance name: node-image-sweeper @@ -1108,7 +1176,7 @@ workloads: kubernetes.io/arch: arm64 node-role.kubernetes.io/worker: 'true' images: - - registry.bstein.dev/bstein/ariadne:0.1.0-454 + - registry.bstein.dev/bstein/ariadne:0.1.0-464 - kind: Deployment namespace: maintenance name: maintenance-vault-sync @@ -1127,7 +1195,7 @@ workloads: serviceAccountName: metis nodeSelector: {} images: - - registry.bstein.dev/bstein/metis:0.1.0-293-arm64 + - registry.bstein.dev/bstein/metis:0.1.0-304-arm64 - kind: Deployment namespace: maintenance name: oauth2-proxy-metis @@ -2120,6 +2188,16 @@ services: port: 9006 targetPort: claude-broker protocol: TCP +- namespace: hermes + name: hermes-cli-lane-metrics + type: ClusterIP + selector: + app: hermes-agent + ports: + - name: lane-metrics + port: 9011 + targetPort: lane-metrics + protocol: TCP - namespace: hermes name: hermes-codex-broker type: ClusterIP @@ -2130,6 +2208,55 @@ services: port: 9003 targetPort: codex-broker protocol: TCP +- namespace: hermes + name: hermes-execution-mediator-0 + type: ClusterIP + selector: + app: hermes-execution-mediator + pool-ordinal: '0' + ports: + - name: mediator + port: 9009 + targetPort: mediator + protocol: TCP +- namespace: hermes + name: hermes-execution-mediator-1 + type: ClusterIP + selector: + app: hermes-execution-mediator + pool-ordinal: '1' + ports: + - name: mediator + port: 9009 + targetPort: mediator + protocol: TCP +- namespace: hermes + name: hermes-execution-mediator-2 + type: ClusterIP + selector: + app: hermes-execution-mediator + pool-ordinal: '2' + ports: + - name: mediator + port: 9009 + targetPort: mediator + protocol: TCP +- namespace: hermes + name: hermes-execution-pool + type: ClusterIP + selector: + app: hermes-agent + ports: + - name: http + port: 9007 + targetPort: execution-pool + protocol: TCP +- namespace: hermes + name: hermes-execution-worker + type: ClusterIP + selector: + app: hermes-execution-worker + ports: [] - namespace: hermes name: hermes-gpu-handoff type: ClusterIP @@ -2254,6 +2381,16 @@ services: port: 80 targetPort: http protocol: TCP +- namespace: hermes-scm + name: hermes-scm-broker + type: ClusterIP + selector: + app: hermes-scm-broker + ports: + - name: http + port: 9081 + targetPort: http + protocol: TCP - namespace: jellyfin name: jellyfin type: ClusterIP @@ -2894,13 +3031,24 @@ http_endpoints: namespace: hermes service: oauth2-proxy-hermes-chat port: http - workloads: + workloads: &id004 - kind: Deployment name: oauth2-proxy-hermes-chat via: kind: Ingress name: hermes-sites source: hermes +- host: chat.hermes.bstein.dev + path: / + backend: + namespace: hermes + service: oauth2-proxy-hermes-chat + port: http + workloads: *id004 + via: + kind: Ingress + name: hermes-sites + source: hermes - host: ci.bstein.dev path: / backend: @@ -3005,7 +3153,7 @@ http_endpoints: namespace: comms service: matrix-guest-register port: 8080 - workloads: &id005 + workloads: &id006 - kind: Deployment name: matrix-guest-register via: @@ -3018,7 +3166,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: &id004 + workloads: &id005 - kind: Deployment name: matrix-authentication-service via: @@ -3031,7 +3179,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: *id004 + workloads: *id005 via: kind: Ingress name: matrix-routing @@ -3042,7 +3190,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: *id004 + workloads: *id005 via: kind: Ingress name: matrix-routing @@ -3053,7 +3201,7 @@ http_endpoints: namespace: comms service: matrix-guest-register port: 8080 - workloads: *id005 + workloads: *id006 via: kind: Ingress name: matrix-routing @@ -3101,7 +3249,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: *id004 + workloads: *id005 via: kind: Ingress name: matrix-routing @@ -3145,7 +3293,7 @@ http_endpoints: namespace: comms service: matrix-guest-register port: 8080 - workloads: *id005 + workloads: *id006 via: kind: Ingress name: matrix-routing @@ -3156,7 +3304,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: *id004 + workloads: *id005 via: kind: Ingress name: matrix-routing @@ -3167,7 +3315,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: *id004 + workloads: *id005 via: kind: Ingress name: matrix-routing @@ -3178,7 +3326,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: *id004 + workloads: *id005 via: kind: Ingress name: matrix-routing @@ -3189,7 +3337,7 @@ http_endpoints: namespace: comms service: matrix-guest-register port: 8080 - workloads: *id005 + workloads: *id006 via: kind: Ingress name: matrix-routing @@ -3367,13 +3515,24 @@ http_endpoints: namespace: hermes service: oauth2-proxy-hermes-triage port: http - workloads: + workloads: &id007 - kind: Deployment name: oauth2-proxy-hermes-triage via: kind: Ingress name: hermes-sites source: hermes +- host: triage.hermes.bstein.dev + path: / + backend: + namespace: hermes + service: oauth2-proxy-hermes-triage + port: http + workloads: *id007 + via: + kind: Ingress + name: hermes-sites + source: hermes - host: vault.bstein.dev path: / backend: @@ -3406,7 +3565,7 @@ http_endpoints: namespace: veles service: veles-backend port: 80 - workloads: &id006 + workloads: &id008 - kind: Deployment name: veles-backend via: @@ -3419,7 +3578,7 @@ http_endpoints: namespace: veles service: veles-backend port: 80 - workloads: *id006 + workloads: *id008 via: kind: Ingress name: veles @@ -3430,7 +3589,7 @@ http_endpoints: namespace: veles service: veles-backend port: 80 - workloads: *id006 + workloads: *id008 via: kind: Ingress name: veles diff --git a/knowledge/catalog/metrics.json b/knowledge/catalog/metrics.json index f26012fb..589033f1 100644 --- a/knowledge/catalog/metrics.json +++ b/knowledge/catalog/metrics.json @@ -1,4 +1,440 @@ [ + { + "dashboard": "Atlas AI Operations", + "panel_title": "Codex Weekly Remaining", + "panel_id": 1, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(atlas_ai_quota_remaining_percent{provider=\"openai\",limit=\"codex\",window=\"seven_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Codex Spark Weekly Remaining", + "panel_id": 2, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(atlas_ai_quota_remaining_percent{provider=\"openai\",limit=\"gpt-5-3-codex-spark\",window=\"seven_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Claude 5h Remaining", + "panel_id": 3, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(atlas_ai_quota_remaining_percent{provider=\"anthropic\",window=\"five_hour\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"anthropic\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Claude 7d Remaining", + "panel_id": 4, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(atlas_ai_quota_remaining_percent{provider=\"anthropic\",window=\"seven_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"anthropic\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Quota Collectors Healthy", + "panel_id": 5, + "panel_type": "stat", + "description": "Successful latest quota fetches. Providers are polled independently every five minutes.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum(atlas_ai_quota_fetch_success) or on() vector(0)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Oldest Quota Sample", + "panel_id": 6, + "panel_type": "stat", + "description": "Age of the stalest successful provider quota snapshot.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "max((time() - atlas_ai_quota_last_success_timestamp_seconds) and (atlas_ai_quota_last_success_timestamp_seconds > 0)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Codex Weekly Reset In", + "panel_id": 7, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(clamp_min(atlas_ai_quota_reset_timestamp_seconds{provider=\"openai\",limit=\"codex\",window=\"seven_day\"} - time(), 0) and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Claude 5h Reset In", + "panel_id": 8, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(clamp_min(atlas_ai_quota_reset_timestamp_seconds{provider=\"anthropic\",window=\"five_hour\"} - time(), 0) and on(provider) (atlas_ai_quota_fetch_success{provider=\"anthropic\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Claude 7d Reset In", + "panel_id": 9, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(clamp_min(atlas_ai_quota_reset_timestamp_seconds{provider=\"anthropic\",window=\"seven_day\"} - time(), 0) and on(provider) (atlas_ai_quota_fetch_success{provider=\"anthropic\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Codex Tokens (Latest Day)", + "panel_id": 10, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(atlas_ai_account_tokens{provider=\"openai\",period=\"latest_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Codex Tokens (7d)", + "panel_id": 11, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(atlas_ai_account_tokens{provider=\"openai\",period=\"seven_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Switchyard Requests", + "panel_id": 12, + "panel_type": "stat", + "description": "Hosted model requests observed by Switchyard in the selected dashboard range.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum(increase(switchyard_requests_total[$__range])) or on() vector(0)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Model Selection Rate", + "panel_id": 13, + "panel_type": "timeseries", + "description": "AUTO and fixed-route decisions by selected provider, model family, and reasoning effort.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum by (selected_model) (rate(switchyard_decisions_total[5m]))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Provider Selections (Range)", + "panel_id": 14, + "panel_type": "bargauge", + "description": "Switchyard selections grouped by provider over the selected dashboard range.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sort_desc(sum by (provider) (label_replace(increase(switchyard_decisions_total{selected_model=~\"(route|worker)/(codex|claude|local)/.*\"}[$__range]), \"provider\", \"$2\", \"selected_model\", \"^(route|worker)/(codex|claude|local)/.*\")))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Token Throughput", + "panel_id": 15, + "panel_type": "timeseries", + "description": "Prompt, cache, reasoning, and completion token rates reported by hosted Switchyard calls.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum(rate(switchyard_prompt_tokens_total[5m]))", + "sum(rate(switchyard_cached_tokens_total[5m]))", + "sum(rate(switchyard_cache_creation_tokens_total[5m]))", + "sum(rate(switchyard_reasoning_tokens_total[5m]))", + "sum(rate(switchyard_completion_tokens_total[5m]))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Model Call p95 Latency", + "panel_id": 16, + "panel_type": "timeseries", + "description": "95th percentile upstream latency for each selected model route.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "histogram_quantile(0.95, sum by (le, model) (rate(switchyard_model_call_latency_ms_bucket[5m])))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Prompt Cache Share", + "panel_id": 17, + "panel_type": "stat", + "description": "Cached tokens as a share of prompt plus cached tokens; higher generally means less repeated provider work.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "100 * sum(rate(switchyard_cached_tokens_total[5m])) / clamp_min(sum(rate(switchyard_prompt_tokens_total[5m])) + sum(rate(switchyard_cached_tokens_total[5m])), 1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Client Success Rate", + "panel_id": 18, + "panel_type": "stat", + "description": "Successful client-facing Switchyard responses in the selected range.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "100 * sum(increase(switchyard_client_responses_total{outcome=\"success\"}[$__range])) / clamp_min(sum(increase(switchyard_client_responses_total[$__range])), 1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Classifier Fail-Open (Range)", + "panel_id": 19, + "panel_type": "stat", + "description": "Local classifier failures that safely fell back to the conservative hosted route.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum(increase(switchyard_classifier_fail_open_total[$__range])) or on() vector(0)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Upstream Errors (Range)", + "panel_id": 20, + "panel_type": "stat", + "description": "Hosted model attempts that returned errors in the selected range.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum(increase(switchyard_errors_total[$__range])) or on() vector(0)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Local Classifier Calls", + "panel_id": 21, + "panel_type": "timeseries", + "description": "Local Qwen routing-classifier activity, split by successful and failed calls.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum by (outcome) (rate(switchyard_llm_calls_total{selected_model=~\"qwen.*\"}[5m]))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Routing Overhead p95", + "panel_id": 22, + "panel_type": "timeseries", + "description": "95th percentile time Switchyard spends selecting a model before the upstream call.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "histogram_quantile(0.95, sum by (le, algorithm) (rate(switchyard_routing_overhead_ms_bucket[5m])))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Traffic Lanes (Range)", + "panel_id": 23, + "panel_type": "bargauge", + "description": "Request volume split between interactive route traffic and durable worker traffic.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sort_desc(sum by (lane) (label_replace(increase(switchyard_requests_total{model=~\"(route|worker)/.*\"}[$__range]), \"lane\", \"$1\", \"model\", \"^(route|worker)/.*\")))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Hermes Workload CPU (Attribution Proxy)", + "panel_id": 25, + "panel_type": "timeseries", + "description": "Compute use by Hermes pod/container. Switchyard currently exposes model and worker-vs-route attribution, but not tenant-slot token labels; CPU is clearly marked as a proxy rather than token usage.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum by (pod, container) (rate(container_cpu_usage_seconds_total{namespace=\"hermes\",pod=~\"hermes-(agent|chat-tenant|switchyard|model-gate).*\",container!=\"\",image!=\"\"}[5m]))" + ] + }, { "dashboard": "Atlas GitOps", "panel_title": "Flux Source", diff --git a/knowledge/diagrams/atlas-http.mmd b/knowledge/diagrams/atlas-http.mmd index 5c0c9be7..534dbdf1 100644 --- a/knowledge/diagrams/atlas-http.mmd +++ b/knowledge/diagrams/atlas-http.mmd @@ -56,6 +56,8 @@ flowchart LR host_chat_bstein_dev --> svc_hermes_oauth2_proxy_hermes_chat wl_hermes_oauth2_proxy_hermes_chat["hermes/oauth2-proxy-hermes-chat (Deployment)"] svc_hermes_oauth2_proxy_hermes_chat --> wl_hermes_oauth2_proxy_hermes_chat + host_chat_hermes_bstein_dev["chat.hermes.bstein.dev"] + host_chat_hermes_bstein_dev --> svc_hermes_oauth2_proxy_hermes_chat host_ci_bstein_dev["ci.bstein.dev"] svc_jenkins_jenkins["jenkins/jenkins (Service)"] host_ci_bstein_dev --> svc_jenkins_jenkins @@ -175,6 +177,8 @@ flowchart LR host_triage_bstein_dev --> svc_hermes_oauth2_proxy_hermes_triage wl_hermes_oauth2_proxy_hermes_triage["hermes/oauth2-proxy-hermes-triage (Deployment)"] svc_hermes_oauth2_proxy_hermes_triage --> wl_hermes_oauth2_proxy_hermes_triage + host_triage_hermes_bstein_dev["triage.hermes.bstein.dev"] + host_triage_hermes_bstein_dev --> svc_hermes_oauth2_proxy_hermes_triage host_vault_bstein_dev["vault.bstein.dev"] svc_vaultwarden_vaultwarden_service["vaultwarden/vaultwarden-service (Service)"] host_vault_bstein_dev --> svc_vaultwarden_vaultwarden_service diff --git a/services/comms/knowledge/catalog/atlas-summary.json b/services/comms/knowledge/catalog/atlas-summary.json index 9acf7417..05ba796d 100644 --- a/services/comms/knowledge/catalog/atlas-summary.json +++ b/services/comms/knowledge/catalog/atlas-summary.json @@ -1,8 +1,8 @@ { "counts": { "helmrelease_host_hints": 23, - "http_endpoints": 61, - "services": 91, - "workloads": 125 + "http_endpoints": 63, + "services": 98, + "workloads": 130 } } diff --git a/services/comms/knowledge/catalog/atlas.json b/services/comms/knowledge/catalog/atlas.json index d7247bfd..d6227f4c 100644 --- a/services/comms/knowledge/catalog/atlas.json +++ b/services/comms/knowledge/catalog/atlas.json @@ -106,6 +106,31 @@ "path": "services/hermes-chat", "targetNamespace": "hermes-chat" }, + { + "name": "hermes-observer-bindings", + "path": "services/hermes-observer-bindings", + "targetNamespace": null + }, + { + "name": "hermes-observer-rbac", + "path": "services/hermes-observer-rbac", + "targetNamespace": null + }, + { + "name": "hermes-scm-broker", + "path": "services/hermes-scm-broker", + "targetNamespace": "hermes-scm" + }, + { + "name": "hermes-scm-broker-code", + "path": "services/hermes/scm-common", + "targetNamespace": "hermes-scm" + }, + { + "name": "hermes-scm-namespace", + "path": "services/hermes-scm-namespace", + "targetNamespace": null + }, { "name": "hermes-triage-demo", "path": "services/hermes-triage-demo", @@ -251,6 +276,11 @@ "path": "infrastructure/vault-csi", "targetNamespace": "kube-system" }, + { + "name": "vault-hermes-jenkins-token-seed", + "path": "services/vault-hermes-jenkins-token-seed", + "targetNamespace": "vault" + }, { "name": "vault-injector", "path": "infrastructure/vault-injector", @@ -304,7 +334,7 @@ "node-role.kubernetes.io/worker": "true" }, "images": [ - "registry.bstein.dev/bstein/bstein-dev-home-backend:0.1.1-464" + "registry.bstein.dev/bstein/bstein-dev-home-backend:0.1.1-479" ] }, { @@ -320,7 +350,7 @@ "node-role.kubernetes.io/worker": "true" }, "images": [ - "registry.bstein.dev/bstein/bstein-dev-home-frontend:0.1.1-464" + "registry.bstein.dev/bstein/bstein-dev-home-frontend:0.1.1-479" ] }, { @@ -902,7 +932,7 @@ "serviceAccountName": "hermes-node-ssh-access", "nodeSelector": {}, "images": [ - "busybox:1.37" + "python@sha256:6d43704baacd1bfbe7c295d7f13079d5d8104ed33568873133f8fc69980419df" ] }, { @@ -915,8 +945,8 @@ "serviceAccountName": "hermes-triage", "nodeSelector": {}, "images": [ - "registry.bstein.dev/bstein/hermes-agent@sha256:cce1f65dc7d30fdce4b1748cc7d3434b9761ae9946a8c8be6049204c503d60a8", - "registry.bstein.dev/bstein/hermes-webui@sha256:9c2fe8341c7b650e08d10acead3151b19e2af737863268bafb39b3d9517575b1" + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37", + "registry.bstein.dev/bstein/hermes-webui@sha256:ac6ba7bfd8a86227f31a9a96ebea41227ccf70391f7dcf34206f4d58e835e50e" ] }, { @@ -930,7 +960,7 @@ "nodeSelector": {}, "images": [ "quay.io/oauth2-proxy/oauth2-proxy:v7.15.3@sha256:10a1165743a192e1940b4708fb9647027185ce11a681a1c5519b442ff7f1f561", - "registry.bstein.dev/bstein/hermes-agent@sha256:cce1f65dc7d30fdce4b1748cc7d3434b9761ae9946a8c8be6049204c503d60a8" + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37" ] }, { @@ -943,7 +973,7 @@ "serviceAccountName": "hermes-chat", "nodeSelector": {}, "images": [ - "registry.bstein.dev/bstein/hermes-chat-router@sha256:a4010fdc5b6dce2696e6fec06eb38a3992817faa01ff3472122c870038b22bd2" + "registry.bstein.dev/bstein/hermes-chat-router@sha256:6744cb7b87c6050f1b97c0675cd280b8295b3b5826ba1d6d92e37aee6fd0b8c4" ] }, { @@ -1058,6 +1088,48 @@ "registry.bstein.dev/bstein/hermes-chat-sandbox@sha256:17ee62b8e61c08573a3a8cca903b38ec43800cb44ec29340e1bc095176544bca" ] }, + { + "kind": "Deployment", + "namespace": "hermes", + "name": "hermes-execution-mediator-0", + "labels": { + "app": "hermes-execution-mediator", + "pool-ordinal": "0" + }, + "serviceAccountName": "hermes-execution-worker", + "nodeSelector": {}, + "images": [ + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37" + ] + }, + { + "kind": "Deployment", + "namespace": "hermes", + "name": "hermes-execution-mediator-1", + "labels": { + "app": "hermes-execution-mediator", + "pool-ordinal": "1" + }, + "serviceAccountName": "hermes-execution-worker", + "nodeSelector": {}, + "images": [ + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37" + ] + }, + { + "kind": "Deployment", + "namespace": "hermes", + "name": "hermes-execution-mediator-2", + "labels": { + "app": "hermes-execution-mediator", + "pool-ordinal": "2" + }, + "serviceAccountName": "hermes-execution-worker", + "nodeSelector": {}, + "images": [ + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37" + ] + }, { "kind": "Deployment", "namespace": "hermes", @@ -1177,8 +1249,36 @@ "serviceAccountName": "hermes-chat", "nodeSelector": {}, "images": [ - "registry.bstein.dev/bstein/hermes-agent@sha256:cce1f65dc7d30fdce4b1748cc7d3434b9761ae9946a8c8be6049204c503d60a8", - "registry.bstein.dev/bstein/hermes-webui@sha256:9c2fe8341c7b650e08d10acead3151b19e2af737863268bafb39b3d9517575b1" + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37", + "registry.bstein.dev/bstein/hermes-webui@sha256:c276a9e17c9057237472f39640c9ebac9d4359d9f62a34df52a759eef43167bf" + ] + }, + { + "kind": "StatefulSet", + "namespace": "hermes", + "name": "hermes-execution-worker", + "labels": { + "app": "hermes-execution-worker", + "app.kubernetes.io/name": "hermes-execution-worker", + "app.kubernetes.io/part-of": "hermes" + }, + "serviceAccountName": "hermes-execution-worker", + "nodeSelector": {}, + "images": [ + "registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37" + ] + }, + { + "kind": "Deployment", + "namespace": "hermes-scm", + "name": "hermes-scm-broker", + "labels": { + "app": "hermes-scm-broker" + }, + "serviceAccountName": "hermes-scm-broker", + "nodeSelector": {}, + "images": [ + "registry.bstein.dev/bstein/hermes-agent@sha256:37ebf720c783ae908a602916ffccf88d43d205a157957f5dc4b487867aee45e7" ] }, { @@ -1528,7 +1628,7 @@ "kubernetes.io/os": "linux" }, "images": [ - "registry.bstein.dev/bstein/metis-sentinel:0.1.0-293-amd64" + "registry.bstein.dev/bstein/metis-sentinel:0.1.0-304-amd64" ] }, { @@ -1544,7 +1644,7 @@ "kubernetes.io/os": "linux" }, "images": [ - "registry.bstein.dev/bstein/metis-sentinel:0.1.0-293-arm64" + "registry.bstein.dev/bstein/metis-sentinel:0.1.0-304-arm64" ] }, { @@ -1633,7 +1733,7 @@ "node-role.kubernetes.io/worker": "true" }, "images": [ - "registry.bstein.dev/bstein/ariadne:0.1.0-454" + "registry.bstein.dev/bstein/ariadne:0.1.0-464" ] }, { @@ -1661,7 +1761,7 @@ "serviceAccountName": "metis", "nodeSelector": {}, "images": [ - "registry.bstein.dev/bstein/metis:0.1.0-293-arm64" + "registry.bstein.dev/bstein/metis:0.1.0-304-arm64" ] }, { @@ -3179,6 +3279,22 @@ } ] }, + { + "namespace": "hermes", + "name": "hermes-cli-lane-metrics", + "type": "ClusterIP", + "selector": { + "app": "hermes-agent" + }, + "ports": [ + { + "name": "lane-metrics", + "port": 9011, + "targetPort": "lane-metrics", + "protocol": "TCP" + } + ] + }, { "namespace": "hermes", "name": "hermes-codex-broker", @@ -3195,6 +3311,82 @@ } ] }, + { + "namespace": "hermes", + "name": "hermes-execution-mediator-0", + "type": "ClusterIP", + "selector": { + "app": "hermes-execution-mediator", + "pool-ordinal": "0" + }, + "ports": [ + { + "name": "mediator", + "port": 9009, + "targetPort": "mediator", + "protocol": "TCP" + } + ] + }, + { + "namespace": "hermes", + "name": "hermes-execution-mediator-1", + "type": "ClusterIP", + "selector": { + "app": "hermes-execution-mediator", + "pool-ordinal": "1" + }, + "ports": [ + { + "name": "mediator", + "port": 9009, + "targetPort": "mediator", + "protocol": "TCP" + } + ] + }, + { + "namespace": "hermes", + "name": "hermes-execution-mediator-2", + "type": "ClusterIP", + "selector": { + "app": "hermes-execution-mediator", + "pool-ordinal": "2" + }, + "ports": [ + { + "name": "mediator", + "port": 9009, + "targetPort": "mediator", + "protocol": "TCP" + } + ] + }, + { + "namespace": "hermes", + "name": "hermes-execution-pool", + "type": "ClusterIP", + "selector": { + "app": "hermes-agent" + }, + "ports": [ + { + "name": "http", + "port": 9007, + "targetPort": "execution-pool", + "protocol": "TCP" + } + ] + }, + { + "namespace": "hermes", + "name": "hermes-execution-worker", + "type": "ClusterIP", + "selector": { + "app": "hermes-execution-worker" + }, + "ports": [] + }, { "namespace": "hermes", "name": "hermes-gpu-handoff", @@ -3393,6 +3585,22 @@ } ] }, + { + "namespace": "hermes-scm", + "name": "hermes-scm-broker", + "type": "ClusterIP", + "selector": { + "app": "hermes-scm-broker" + }, + "ports": [ + { + "name": "http", + "port": 9081, + "targetPort": "http", + "protocol": "TCP" + } + ] + }, { "namespace": "jellyfin", "name": "jellyfin", @@ -4423,6 +4631,26 @@ "source": "hermes" } }, + { + "host": "chat.hermes.bstein.dev", + "path": "/", + "backend": { + "namespace": "hermes", + "service": "oauth2-proxy-hermes-chat", + "port": "http", + "workloads": [ + { + "kind": "Deployment", + "name": "oauth2-proxy-hermes-chat" + } + ] + }, + "via": { + "kind": "Ingress", + "name": "hermes-sites", + "source": "hermes" + } + }, { "host": "ci.bstein.dev", "path": "/", @@ -5183,6 +5411,26 @@ "source": "hermes" } }, + { + "host": "triage.hermes.bstein.dev", + "path": "/", + "backend": { + "namespace": "hermes", + "service": "oauth2-proxy-hermes-triage", + "port": "http", + "workloads": [ + { + "kind": "Deployment", + "name": "oauth2-proxy-hermes-triage" + } + ] + }, + "via": { + "kind": "Ingress", + "name": "hermes-sites", + "source": "hermes" + } + }, { "host": "vault.bstein.dev", "path": "/", diff --git a/services/comms/knowledge/catalog/atlas.yaml b/services/comms/knowledge/catalog/atlas.yaml index 8ae1b49b..8e9ec24c 100644 --- a/services/comms/knowledge/catalog/atlas.yaml +++ b/services/comms/knowledge/catalog/atlas.yaml @@ -65,6 +65,21 @@ sources: - name: hermes-chat path: services/hermes-chat targetNamespace: hermes-chat +- name: hermes-observer-bindings + path: services/hermes-observer-bindings + targetNamespace: null +- name: hermes-observer-rbac + path: services/hermes-observer-rbac + targetNamespace: null +- name: hermes-scm-broker + path: services/hermes-scm-broker + targetNamespace: hermes-scm +- name: hermes-scm-broker-code + path: services/hermes/scm-common + targetNamespace: hermes-scm +- name: hermes-scm-namespace + path: services/hermes-scm-namespace + targetNamespace: null - name: hermes-triage-demo path: services/hermes-triage-demo targetNamespace: null @@ -152,6 +167,9 @@ sources: - name: vault-csi path: infrastructure/vault-csi targetNamespace: kube-system +- name: vault-hermes-jenkins-token-seed + path: services/vault-hermes-jenkins-token-seed + targetNamespace: vault - name: vault-injector path: infrastructure/vault-injector targetNamespace: vault @@ -187,7 +205,7 @@ workloads: kubernetes.io/arch: arm64 node-role.kubernetes.io/worker: 'true' images: - - registry.bstein.dev/bstein/bstein-dev-home-backend:0.1.1-464 + - registry.bstein.dev/bstein/bstein-dev-home-backend:0.1.1-479 - kind: Deployment namespace: bstein-dev-home name: bstein-dev-home-frontend @@ -198,7 +216,7 @@ workloads: kubernetes.io/arch: arm64 node-role.kubernetes.io/worker: 'true' images: - - registry.bstein.dev/bstein/bstein-dev-home-frontend:0.1.1-464 + - registry.bstein.dev/bstein/bstein-dev-home-frontend:0.1.1-479 - kind: Deployment namespace: bstein-dev-home name: bstein-dev-home-vault-sync @@ -604,7 +622,7 @@ workloads: serviceAccountName: hermes-node-ssh-access nodeSelector: {} images: - - busybox:1.37 + - python@sha256:6d43704baacd1bfbe7c295d7f13079d5d8104ed33568873133f8fc69980419df - kind: Deployment namespace: hermes name: hermes @@ -613,8 +631,8 @@ workloads: serviceAccountName: hermes-triage nodeSelector: {} images: - - registry.bstein.dev/bstein/hermes-agent@sha256:cce1f65dc7d30fdce4b1748cc7d3434b9761ae9946a8c8be6049204c503d60a8 - - registry.bstein.dev/bstein/hermes-webui@sha256:9c2fe8341c7b650e08d10acead3151b19e2af737863268bafb39b3d9517575b1 + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 + - registry.bstein.dev/bstein/hermes-webui@sha256:ac6ba7bfd8a86227f31a9a96ebea41227ccf70391f7dcf34206f4d58e835e50e - kind: Deployment namespace: hermes name: hermes-agent @@ -624,7 +642,7 @@ workloads: nodeSelector: {} images: - quay.io/oauth2-proxy/oauth2-proxy:v7.15.3@sha256:10a1165743a192e1940b4708fb9647027185ce11a681a1c5519b442ff7f1f561 - - registry.bstein.dev/bstein/hermes-agent@sha256:cce1f65dc7d30fdce4b1748cc7d3434b9761ae9946a8c8be6049204c503d60a8 + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 - kind: Deployment namespace: hermes name: hermes-chat-router @@ -633,7 +651,7 @@ workloads: serviceAccountName: hermes-chat nodeSelector: {} images: - - registry.bstein.dev/bstein/hermes-chat-router@sha256:a4010fdc5b6dce2696e6fec06eb38a3992817faa01ff3472122c870038b22bd2 + - registry.bstein.dev/bstein/hermes-chat-router@sha256:6744cb7b87c6050f1b97c0675cd280b8295b3b5826ba1d6d92e37aee6fd0b8c4 - kind: Deployment namespace: hermes name: hermes-chat-sandbox-0 @@ -714,6 +732,36 @@ workloads: nodeSelector: {} images: - registry.bstein.dev/bstein/hermes-chat-sandbox@sha256:17ee62b8e61c08573a3a8cca903b38ec43800cb44ec29340e1bc095176544bca +- kind: Deployment + namespace: hermes + name: hermes-execution-mediator-0 + labels: + app: hermes-execution-mediator + pool-ordinal: '0' + serviceAccountName: hermes-execution-worker + nodeSelector: {} + images: + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 +- kind: Deployment + namespace: hermes + name: hermes-execution-mediator-1 + labels: + app: hermes-execution-mediator + pool-ordinal: '1' + serviceAccountName: hermes-execution-worker + nodeSelector: {} + images: + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 +- kind: Deployment + namespace: hermes + name: hermes-execution-mediator-2 + labels: + app: hermes-execution-mediator + pool-ordinal: '2' + serviceAccountName: hermes-execution-worker + nodeSelector: {} + images: + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 - kind: Deployment namespace: hermes name: hermes-local-image @@ -797,8 +845,28 @@ workloads: serviceAccountName: hermes-chat nodeSelector: {} images: - - registry.bstein.dev/bstein/hermes-agent@sha256:cce1f65dc7d30fdce4b1748cc7d3434b9761ae9946a8c8be6049204c503d60a8 - - registry.bstein.dev/bstein/hermes-webui@sha256:9c2fe8341c7b650e08d10acead3151b19e2af737863268bafb39b3d9517575b1 + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 + - registry.bstein.dev/bstein/hermes-webui@sha256:c276a9e17c9057237472f39640c9ebac9d4359d9f62a34df52a759eef43167bf +- kind: StatefulSet + namespace: hermes + name: hermes-execution-worker + labels: + app: hermes-execution-worker + app.kubernetes.io/name: hermes-execution-worker + app.kubernetes.io/part-of: hermes + serviceAccountName: hermes-execution-worker + nodeSelector: {} + images: + - registry.bstein.dev/bstein/hermes-agent@sha256:4a385fbd04ae1e3e8ca0237db98a318606a2a03786ef467757b8da5eee4e4b37 +- kind: Deployment + namespace: hermes-scm + name: hermes-scm-broker + labels: + app: hermes-scm-broker + serviceAccountName: hermes-scm-broker + nodeSelector: {} + images: + - registry.bstein.dev/bstein/hermes-agent@sha256:37ebf720c783ae908a602916ffccf88d43d205a157957f5dc4b487867aee45e7 - kind: Deployment namespace: jellyfin name: jellyfin @@ -1037,7 +1105,7 @@ workloads: kubernetes.io/arch: amd64 kubernetes.io/os: linux images: - - registry.bstein.dev/bstein/metis-sentinel:0.1.0-293-amd64 + - registry.bstein.dev/bstein/metis-sentinel:0.1.0-304-amd64 - kind: DaemonSet namespace: maintenance name: metis-sentinel-arm64 @@ -1048,7 +1116,7 @@ workloads: kubernetes.io/arch: arm64 kubernetes.io/os: linux images: - - registry.bstein.dev/bstein/metis-sentinel:0.1.0-293-arm64 + - registry.bstein.dev/bstein/metis-sentinel:0.1.0-304-arm64 - kind: DaemonSet namespace: maintenance name: node-image-sweeper @@ -1108,7 +1176,7 @@ workloads: kubernetes.io/arch: arm64 node-role.kubernetes.io/worker: 'true' images: - - registry.bstein.dev/bstein/ariadne:0.1.0-454 + - registry.bstein.dev/bstein/ariadne:0.1.0-464 - kind: Deployment namespace: maintenance name: maintenance-vault-sync @@ -1127,7 +1195,7 @@ workloads: serviceAccountName: metis nodeSelector: {} images: - - registry.bstein.dev/bstein/metis:0.1.0-293-arm64 + - registry.bstein.dev/bstein/metis:0.1.0-304-arm64 - kind: Deployment namespace: maintenance name: oauth2-proxy-metis @@ -2120,6 +2188,16 @@ services: port: 9006 targetPort: claude-broker protocol: TCP +- namespace: hermes + name: hermes-cli-lane-metrics + type: ClusterIP + selector: + app: hermes-agent + ports: + - name: lane-metrics + port: 9011 + targetPort: lane-metrics + protocol: TCP - namespace: hermes name: hermes-codex-broker type: ClusterIP @@ -2130,6 +2208,55 @@ services: port: 9003 targetPort: codex-broker protocol: TCP +- namespace: hermes + name: hermes-execution-mediator-0 + type: ClusterIP + selector: + app: hermes-execution-mediator + pool-ordinal: '0' + ports: + - name: mediator + port: 9009 + targetPort: mediator + protocol: TCP +- namespace: hermes + name: hermes-execution-mediator-1 + type: ClusterIP + selector: + app: hermes-execution-mediator + pool-ordinal: '1' + ports: + - name: mediator + port: 9009 + targetPort: mediator + protocol: TCP +- namespace: hermes + name: hermes-execution-mediator-2 + type: ClusterIP + selector: + app: hermes-execution-mediator + pool-ordinal: '2' + ports: + - name: mediator + port: 9009 + targetPort: mediator + protocol: TCP +- namespace: hermes + name: hermes-execution-pool + type: ClusterIP + selector: + app: hermes-agent + ports: + - name: http + port: 9007 + targetPort: execution-pool + protocol: TCP +- namespace: hermes + name: hermes-execution-worker + type: ClusterIP + selector: + app: hermes-execution-worker + ports: [] - namespace: hermes name: hermes-gpu-handoff type: ClusterIP @@ -2254,6 +2381,16 @@ services: port: 80 targetPort: http protocol: TCP +- namespace: hermes-scm + name: hermes-scm-broker + type: ClusterIP + selector: + app: hermes-scm-broker + ports: + - name: http + port: 9081 + targetPort: http + protocol: TCP - namespace: jellyfin name: jellyfin type: ClusterIP @@ -2894,13 +3031,24 @@ http_endpoints: namespace: hermes service: oauth2-proxy-hermes-chat port: http - workloads: + workloads: &id004 - kind: Deployment name: oauth2-proxy-hermes-chat via: kind: Ingress name: hermes-sites source: hermes +- host: chat.hermes.bstein.dev + path: / + backend: + namespace: hermes + service: oauth2-proxy-hermes-chat + port: http + workloads: *id004 + via: + kind: Ingress + name: hermes-sites + source: hermes - host: ci.bstein.dev path: / backend: @@ -3005,7 +3153,7 @@ http_endpoints: namespace: comms service: matrix-guest-register port: 8080 - workloads: &id005 + workloads: &id006 - kind: Deployment name: matrix-guest-register via: @@ -3018,7 +3166,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: &id004 + workloads: &id005 - kind: Deployment name: matrix-authentication-service via: @@ -3031,7 +3179,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: *id004 + workloads: *id005 via: kind: Ingress name: matrix-routing @@ -3042,7 +3190,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: *id004 + workloads: *id005 via: kind: Ingress name: matrix-routing @@ -3053,7 +3201,7 @@ http_endpoints: namespace: comms service: matrix-guest-register port: 8080 - workloads: *id005 + workloads: *id006 via: kind: Ingress name: matrix-routing @@ -3101,7 +3249,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: *id004 + workloads: *id005 via: kind: Ingress name: matrix-routing @@ -3145,7 +3293,7 @@ http_endpoints: namespace: comms service: matrix-guest-register port: 8080 - workloads: *id005 + workloads: *id006 via: kind: Ingress name: matrix-routing @@ -3156,7 +3304,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: *id004 + workloads: *id005 via: kind: Ingress name: matrix-routing @@ -3167,7 +3315,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: *id004 + workloads: *id005 via: kind: Ingress name: matrix-routing @@ -3178,7 +3326,7 @@ http_endpoints: namespace: comms service: matrix-authentication-service port: 8080 - workloads: *id004 + workloads: *id005 via: kind: Ingress name: matrix-routing @@ -3189,7 +3337,7 @@ http_endpoints: namespace: comms service: matrix-guest-register port: 8080 - workloads: *id005 + workloads: *id006 via: kind: Ingress name: matrix-routing @@ -3367,13 +3515,24 @@ http_endpoints: namespace: hermes service: oauth2-proxy-hermes-triage port: http - workloads: + workloads: &id007 - kind: Deployment name: oauth2-proxy-hermes-triage via: kind: Ingress name: hermes-sites source: hermes +- host: triage.hermes.bstein.dev + path: / + backend: + namespace: hermes + service: oauth2-proxy-hermes-triage + port: http + workloads: *id007 + via: + kind: Ingress + name: hermes-sites + source: hermes - host: vault.bstein.dev path: / backend: @@ -3406,7 +3565,7 @@ http_endpoints: namespace: veles service: veles-backend port: 80 - workloads: &id006 + workloads: &id008 - kind: Deployment name: veles-backend via: @@ -3419,7 +3578,7 @@ http_endpoints: namespace: veles service: veles-backend port: 80 - workloads: *id006 + workloads: *id008 via: kind: Ingress name: veles @@ -3430,7 +3589,7 @@ http_endpoints: namespace: veles service: veles-backend port: 80 - workloads: *id006 + workloads: *id008 via: kind: Ingress name: veles diff --git a/services/comms/knowledge/catalog/metrics.json b/services/comms/knowledge/catalog/metrics.json index f26012fb..589033f1 100644 --- a/services/comms/knowledge/catalog/metrics.json +++ b/services/comms/knowledge/catalog/metrics.json @@ -1,4 +1,440 @@ [ + { + "dashboard": "Atlas AI Operations", + "panel_title": "Codex Weekly Remaining", + "panel_id": 1, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(atlas_ai_quota_remaining_percent{provider=\"openai\",limit=\"codex\",window=\"seven_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Codex Spark Weekly Remaining", + "panel_id": 2, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(atlas_ai_quota_remaining_percent{provider=\"openai\",limit=\"gpt-5-3-codex-spark\",window=\"seven_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Claude 5h Remaining", + "panel_id": 3, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(atlas_ai_quota_remaining_percent{provider=\"anthropic\",window=\"five_hour\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"anthropic\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Claude 7d Remaining", + "panel_id": 4, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(atlas_ai_quota_remaining_percent{provider=\"anthropic\",window=\"seven_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"anthropic\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Quota Collectors Healthy", + "panel_id": 5, + "panel_type": "stat", + "description": "Successful latest quota fetches. Providers are polled independently every five minutes.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum(atlas_ai_quota_fetch_success) or on() vector(0)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Oldest Quota Sample", + "panel_id": 6, + "panel_type": "stat", + "description": "Age of the stalest successful provider quota snapshot.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "max((time() - atlas_ai_quota_last_success_timestamp_seconds) and (atlas_ai_quota_last_success_timestamp_seconds > 0)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Codex Weekly Reset In", + "panel_id": 7, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(clamp_min(atlas_ai_quota_reset_timestamp_seconds{provider=\"openai\",limit=\"codex\",window=\"seven_day\"} - time(), 0) and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Claude 5h Reset In", + "panel_id": 8, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(clamp_min(atlas_ai_quota_reset_timestamp_seconds{provider=\"anthropic\",window=\"five_hour\"} - time(), 0) and on(provider) (atlas_ai_quota_fetch_success{provider=\"anthropic\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Claude 7d Reset In", + "panel_id": 9, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(clamp_min(atlas_ai_quota_reset_timestamp_seconds{provider=\"anthropic\",window=\"seven_day\"} - time(), 0) and on(provider) (atlas_ai_quota_fetch_success{provider=\"anthropic\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Codex Tokens (Latest Day)", + "panel_id": 10, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(atlas_ai_account_tokens{provider=\"openai\",period=\"latest_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Codex Tokens (7d)", + "panel_id": 11, + "panel_type": "stat", + "description": "Live first-party CLI account telemetry. Unavailable means the provider did not return a fresh structured value.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "(atlas_ai_account_tokens{provider=\"openai\",period=\"seven_day\"} and on(provider) (atlas_ai_quota_fetch_success{provider=\"openai\"} == 1)) or on() vector(-1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Switchyard Requests", + "panel_id": 12, + "panel_type": "stat", + "description": "Hosted model requests observed by Switchyard in the selected dashboard range.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum(increase(switchyard_requests_total[$__range])) or on() vector(0)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Model Selection Rate", + "panel_id": 13, + "panel_type": "timeseries", + "description": "AUTO and fixed-route decisions by selected provider, model family, and reasoning effort.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum by (selected_model) (rate(switchyard_decisions_total[5m]))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Provider Selections (Range)", + "panel_id": 14, + "panel_type": "bargauge", + "description": "Switchyard selections grouped by provider over the selected dashboard range.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sort_desc(sum by (provider) (label_replace(increase(switchyard_decisions_total{selected_model=~\"(route|worker)/(codex|claude|local)/.*\"}[$__range]), \"provider\", \"$2\", \"selected_model\", \"^(route|worker)/(codex|claude|local)/.*\")))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Token Throughput", + "panel_id": 15, + "panel_type": "timeseries", + "description": "Prompt, cache, reasoning, and completion token rates reported by hosted Switchyard calls.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum(rate(switchyard_prompt_tokens_total[5m]))", + "sum(rate(switchyard_cached_tokens_total[5m]))", + "sum(rate(switchyard_cache_creation_tokens_total[5m]))", + "sum(rate(switchyard_reasoning_tokens_total[5m]))", + "sum(rate(switchyard_completion_tokens_total[5m]))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Model Call p95 Latency", + "panel_id": 16, + "panel_type": "timeseries", + "description": "95th percentile upstream latency for each selected model route.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "histogram_quantile(0.95, sum by (le, model) (rate(switchyard_model_call_latency_ms_bucket[5m])))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Prompt Cache Share", + "panel_id": 17, + "panel_type": "stat", + "description": "Cached tokens as a share of prompt plus cached tokens; higher generally means less repeated provider work.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "100 * sum(rate(switchyard_cached_tokens_total[5m])) / clamp_min(sum(rate(switchyard_prompt_tokens_total[5m])) + sum(rate(switchyard_cached_tokens_total[5m])), 1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Client Success Rate", + "panel_id": 18, + "panel_type": "stat", + "description": "Successful client-facing Switchyard responses in the selected range.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "100 * sum(increase(switchyard_client_responses_total{outcome=\"success\"}[$__range])) / clamp_min(sum(increase(switchyard_client_responses_total[$__range])), 1)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Classifier Fail-Open (Range)", + "panel_id": 19, + "panel_type": "stat", + "description": "Local classifier failures that safely fell back to the conservative hosted route.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum(increase(switchyard_classifier_fail_open_total[$__range])) or on() vector(0)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Upstream Errors (Range)", + "panel_id": 20, + "panel_type": "stat", + "description": "Hosted model attempts that returned errors in the selected range.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum(increase(switchyard_errors_total[$__range])) or on() vector(0)" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Local Classifier Calls", + "panel_id": 21, + "panel_type": "timeseries", + "description": "Local Qwen routing-classifier activity, split by successful and failed calls.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum by (outcome) (rate(switchyard_llm_calls_total{selected_model=~\"qwen.*\"}[5m]))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Routing Overhead p95", + "panel_id": 22, + "panel_type": "timeseries", + "description": "95th percentile time Switchyard spends selecting a model before the upstream call.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "histogram_quantile(0.95, sum by (le, algorithm) (rate(switchyard_routing_overhead_ms_bucket[5m])))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Traffic Lanes (Range)", + "panel_id": 23, + "panel_type": "bargauge", + "description": "Request volume split between interactive route traffic and durable worker traffic.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sort_desc(sum by (lane) (label_replace(increase(switchyard_requests_total{model=~\"(route|worker)/.*\"}[$__range]), \"lane\", \"$1\", \"model\", \"^(route|worker)/.*\")))" + ] + }, + { + "dashboard": "Atlas AI Operations", + "panel_title": "Hermes Workload CPU (Attribution Proxy)", + "panel_id": 25, + "panel_type": "timeseries", + "description": "Compute use by Hermes pod/container. Switchyard currently exposes model and worker-vs-route attribution, but not tenant-slot token labels; CPU is clearly marked as a proxy rather than token usage.", + "tags": [ + "atlas", + "ai", + "hermes", + "switchyard" + ], + "datasource_uid": "atlas-vm", + "datasource_type": "prometheus", + "exprs": [ + "sum by (pod, container) (rate(container_cpu_usage_seconds_total{namespace=\"hermes\",pod=~\"hermes-(agent|chat-tenant|switchyard|model-gate).*\",container!=\"\",image!=\"\"}[5m]))" + ] + }, { "dashboard": "Atlas GitOps", "panel_title": "Flux Source", diff --git a/services/comms/knowledge/diagrams/atlas-http.mmd b/services/comms/knowledge/diagrams/atlas-http.mmd index 5c0c9be7..534dbdf1 100644 --- a/services/comms/knowledge/diagrams/atlas-http.mmd +++ b/services/comms/knowledge/diagrams/atlas-http.mmd @@ -56,6 +56,8 @@ flowchart LR host_chat_bstein_dev --> svc_hermes_oauth2_proxy_hermes_chat wl_hermes_oauth2_proxy_hermes_chat["hermes/oauth2-proxy-hermes-chat (Deployment)"] svc_hermes_oauth2_proxy_hermes_chat --> wl_hermes_oauth2_proxy_hermes_chat + host_chat_hermes_bstein_dev["chat.hermes.bstein.dev"] + host_chat_hermes_bstein_dev --> svc_hermes_oauth2_proxy_hermes_chat host_ci_bstein_dev["ci.bstein.dev"] svc_jenkins_jenkins["jenkins/jenkins (Service)"] host_ci_bstein_dev --> svc_jenkins_jenkins @@ -175,6 +177,8 @@ flowchart LR host_triage_bstein_dev --> svc_hermes_oauth2_proxy_hermes_triage wl_hermes_oauth2_proxy_hermes_triage["hermes/oauth2-proxy-hermes-triage (Deployment)"] svc_hermes_oauth2_proxy_hermes_triage --> wl_hermes_oauth2_proxy_hermes_triage + host_triage_hermes_bstein_dev["triage.hermes.bstein.dev"] + host_triage_hermes_bstein_dev --> svc_hermes_oauth2_proxy_hermes_triage host_vault_bstein_dev["vault.bstein.dev"] svc_vaultwarden_vaultwarden_service["vaultwarden/vaultwarden-service (Service)"] host_vault_bstein_dev --> svc_vaultwarden_vaultwarden_service