Compare commits

...

837 Commits

Author SHA1 Message Date
jenkins
cb2342bbe7 docs: record stability repairs and settling evidence 2026-10-04 08:51:31 -05:00
jenkins
1aa556d2b1 logging: retire completed maintenance placement check 2026-10-04 08:51:31 -05:00
flux-bot
29609c170b chore(maintenance): automated image update 2026-10-04 13:38:47 +00:00
flux-bot
ba44f17e2c chore(maintenance): automated image update 2026-10-04 13:38:38 +00:00
flux-bot
b18b962b29 chore(maintenance): automated image update 2026-10-04 13:36:38 +00:00
flux-bot
af5a8f25be chore(maintenance): automated image update 2026-10-04 13:34:40 +00:00
jenkins
b12991cb9e ci(data-prepper): start Docker with one loopback listener 2026-10-04 08:32:03 -05:00
jenkins
4d9fe1456e logging: recheck maintenance after build I/O containment 2026-10-04 08:15:46 -05:00
jenkins
b51cbc2f3c ci(data-prepper): keep image layers on disposable build storage 2026-10-04 08:12:57 -05:00
jenkins
d85642adc1 ops: report preemption and terminating service pods 2026-10-04 08:07:44 -05:00
jenkins
eb1eb3698a logging: keep maintenance from preempting running work 2026-10-04 08:04:46 -05:00
flux-bot
1ee9f6be1d chore(maintenance): automated image update 2026-10-04 12:53:42 +00:00
flux-bot
f1a0596b89 chore(maintenance): automated image update 2026-10-04 12:53:32 +00:00
flux-bot
8f0eeaeef6 chore(maintenance): automated image update 2026-10-04 12:52:32 +00:00
flux-bot
da8653bdbb chore(maintenance): automated image update 2026-10-04 12:49:33 +00:00
jenkins
baf6c3db32 maintenance: leave runtime restarts to planned maintenance
Some checks failed
Tests / Declarative: Post Actions failed: 56, skipped: 89, passed: 4301
2026-10-04 07:46:47 -05:00
jenkins
42d8f62912 maintenance: right-size idle node limit helper 2026-10-04 07:44:29 -05:00
flux-bot
5499eeed2b chore(bstein-dev-home): automated image update 2026-10-04 12:30:28 +00:00
flux-bot
4ce24883bd chore(bstein-dev-home): automated image update 2026-10-04 12:30:03 +00:00
jenkins
fd76b043a6 maintenance: restore single-node cleanup rollout budget 2026-10-04 07:24:46 -05:00
jenkins
a223b9fa49 maintenance: leave image garbage collection to kubelet 2026-10-04 07:20:32 -05:00
jenkins
8ba6ffc471 maintenance: serialize sweeper startup around offline pods 2026-10-04 07:17:16 -05:00
jenkins
b4f0f6b48c maintenance: preserve active pod logs during cleanup 2026-10-04 07:14:23 -05:00
jenkins
171979eac7 core: quarantine Titan-12 runtime storage stalls 2026-10-04 07:14:23 -05:00
flux-bot
b5ba9287d8 chore(maintenance): automated image update 2026-10-04 12:06:35 +00:00
jenkins
0282f1af5f ops: add read-only cluster settling comparison 2026-10-04 06:55:56 -05:00
jenkins
b2a1761e77 keycloak: retain completed Metis key bootstrap 2026-10-04 06:51:20 -05:00
jenkins
59d206ba38 maintenance: retire repeating emergency root cleanup 2026-10-04 06:48:53 -05:00
flux-bot
0f640b4ea3 chore(maintenance): automated image update 2026-10-04 11:41:11 +00:00
flux-bot
f267a69856 chore(maintenance): automated image update 2026-10-04 11:23:10 +00:00
jenkins
7a81c3c41e observer: use native HTTP readiness without interpreter startup 2026-10-04 06:19:59 -05:00
jenkins
658ddaddd1 soteria: exclude disposable build volumes from backups 2026-10-04 06:09:07 -05:00
jenkins
d5fcbbfa69 core: preserve offline and power-fault node quarantine 2026-10-04 06:04:37 -05:00
jenkins
82a0930d9f maintenance: spread auth helpers off quarantined workers 2026-10-04 05:38:32 -05:00
jenkins
63eafdd3db metallb: remove unused BGP sidecars from the L2 cluster 2026-10-04 05:31:36 -05:00
jenkins
70e1da36f1 ci: move IaC and Data Prepper scratch onto disposable volumes 2026-10-04 05:27:06 -05:00
jenkins
2f158c6b0a ci: add disposable Longhorn scratch storage for builds 2026-10-04 05:25:09 -05:00
jenkins
785a01fcd0 network: retire the completed VIP pool transition 2026-10-04 05:22:45 -05:00
jenkins
4b087af18d network: move ingress VIP announcements to stable control planes 2026-10-04 05:21:13 -05:00
jenkins
689d686830 network: select shared ingress pool by its fixed VIP 2026-10-04 05:14:45 -05:00
jenkins
f5ff949ade traefik: isolate ingress on stable NVMe control planes 2026-10-04 05:13:13 -05:00
jenkins
21a4f677f2 ci: close placement gaps and fit measured helper reservations 2026-10-04 05:02:10 -05:00
jenkins
d6800b14da logging: retire the completed capacity recovery check 2026-10-04 04:51:10 -05:00
jenkins
53419f7611 logging: verify recovered capacity before clearing the index block 2026-10-04 04:45:27 -05:00
jenkins
cbf0c7f9f9 logging: expand existing OpenSearch storage below disk watermarks 2026-10-04 04:41:58 -05:00
flux-bot
012464d38a chore(maintenance): automated image update 2026-10-04 09:40:16 +00:00
flux-bot
ff76d2bd8d chore(maintenance): automated image update 2026-10-04 09:40:07 +00:00
jenkins
ca8e3abc71 logging: retain safe diagnostics for the failing maintenance job 2026-10-04 04:38:29 -05:00
flux-bot
7db50e0c80 chore(maintenance): automated image update 2026-10-04 09:38:07 +00:00
flux-bot
08723dbcf0 chore(maintenance): automated image update 2026-10-04 09:36:07 +00:00
jenkins
570ef69b80 mailu: gate scanner readiness on its native ping response 2026-10-04 04:26:48 -05:00
jenkins
7a61fff82d nextcloud: preserve installed apps during ordinary restarts 2026-10-04 04:25:46 -05:00
jenkins
5ba2211ad4 nextcloud: fit measured resources and serialize volume rollout 2026-10-04 04:19:11 -05:00
jenkins
64eb26014f mailu: reserve ClamAV working memory on titan-22 2026-10-04 04:19:11 -05:00
jenkins
2857ca40ae core: quarantine titan-08 runtime storage stalls 2026-10-04 04:19:11 -05:00
jenkins
03089836da longhorn: retire completed one-time recovery permissions 2026-10-04 03:56:15 -05:00
jenkins
56c959da95 longhorn: scope recovery access to the stranded engine pod 2026-10-04 03:52:08 -05:00
jenkins
4296912447 longhorn: recover the guarded stranded tenant engine once 2026-10-04 03:49:36 -05:00
jenkins
bb05ec9f7a maintenance: fit log guard and retain completed storage setup jobs 2026-10-04 03:48:56 -05:00
jenkins
1f5329b001 av-test: measure bounded signal continuity and reject cyclic pairing 2026-10-04 03:45:01 -05:00
jenkins
6ee8eb2e6c gitea: move to available CPU capacity and reserve measured memory 2026-10-04 03:33:53 -05:00
jenkins
dfcf75866d docs: prioritize steady service operation and record containment 2026-10-04 03:18:49 -05:00
jenkins
8a6d3db16a ci: reserve storage nodes and serialize build capacity 2026-10-04 03:05:53 -05:00
jenkins
d47e7b8e35 maintenance: stop elapsed-only CI build cancellation 2026-10-04 03:00:29 -05:00
jenkins
eb0060756e core: preserve node placement labels during quarantine 2026-10-04 02:56:31 -05:00
jenkins
cc238c0b6c docs: refresh cluster health and Metis rollout status 2026-10-04 02:44:26 -05:00
jenkins
52dffae88c docs: record log repair and active RAM-log migration boundary 2026-10-04 01:13:54 -05:00
jenkins
9522bcc5d8 nodes: restore verified Ananke peer SSH access 2026-10-04 01:10:16 -05:00
jenkins
1cdcbc7d6c docs: distinguish coordinator success from peer access failure 2026-10-04 00:55:40 -05:00
jenkins
d7dd48eee2 docs: record Ananke access verification and node repair limits 2026-10-04 00:54:45 -05:00
jenkins
886dbeee10 recovery: wire Vault passwords to existing Ananke host actions 2026-10-04 00:45:33 -05:00
jenkins
36c7e76e56 nodes: document verified administration and remaining recovery work 2026-10-04 00:37:36 -05:00
jenkins
c3f892fd0e nodes: retire stale log hooks and reserve Ariadne memory 2026-10-04 00:22:46 -05:00
jenkins
c2f418998c nodes: quarantine titan-11 after recurring undervoltage 2026-10-04 00:14:34 -05:00
jenkins
1bd8ae3197 nodes: repair credential audit and preserve Vault records 2026-10-04 00:09:20 -05:00
jenkins
feca767631 docs: update remaining cluster stabilization work 2026-10-03 23:40:43 -05:00
flux-bot
c0d128e67b chore(maintenance): automated image update 2026-10-04 02:31:02 +00:00
flux-bot
567f48f7d3 chore(maintenance): automated image update 2026-10-04 02:30:00 +00:00
flux-bot
4b0d907b96 chore(maintenance): automated image update 2026-10-04 02:22:56 +00:00
flux-bot
e98ec12549 chore(maintenance): automated image update 2026-10-04 02:11:56 +00:00
jenkins
b41dab9886 docs: record verified backup storage cap blocker
Some checks failed
Tests / Declarative: Post Actions failed: 45, skipped: 89, passed: 4270
2026-10-03 16:33:02 -05:00
jenkins
5d50d2f8c8 av-test: add bounded numeric paired-signal diagnostics 2026-10-03 16:21:28 -05:00
jenkins
0ad4ec866b docs: record storage recovery and remaining health checks 2026-10-03 16:20:11 -05:00
jenkins
b0be99b739 storage: bound replica balancing and rebuild pressure 2026-10-03 16:16:09 -05:00
jenkins
0984164ff6 crypto: restore Monerod after clean volume detach 2026-10-03 16:09:09 -05:00
jenkins
4659653692 backup: scope Soteria permissions to Longhorn mode 2026-10-03 15:58:22 -05:00
jenkins
b119076cca crypto: unstage failed Monerod storage mount 2026-10-03 15:54:31 -05:00
jenkins
df99cdacc1 storage: preserve Monerod blocks before recovery 2026-10-03 15:52:16 -05:00
jenkins
b48f1dbd1d docs: track six cluster health priorities 2026-10-03 15:46:13 -05:00
flux-bot
c345e962c0 chore(maintenance): automated image update 2026-10-03 15:17:57 +00:00
flux-bot
a620a139cd chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 46, skipped: 89, passed: 4269
2026-10-03 10:21:11 +00:00
flux-bot
1ccb0b5ce1 chore(bstein-dev-home): automated image update 2026-10-03 10:18:45 +00:00
jenkins
3d4d167413 docs: confirm peer recovery update 2026-10-03 02:51:51 -05:00
jenkins
6a0d0699d9 docs: record recovery guards and current health 2026-10-03 02:50:38 -05:00
jenkins
0180ba6698 maintenance: allow active Pi builds to finish 2026-10-03 02:28:39 -05:00
jenkins
f1fe01830a logging: restore normal rollback checks after recovery 2026-10-03 02:03:50 -05:00
jenkins
2f043d104c logging: unblock rollback to offline placement 2026-10-03 01:53:48 -05:00
jenkins
b0bfb6a948 backup: schedule application recovery copies and document verified repairs 2026-10-03 01:34:57 -05:00
jenkins
7e96995866 logging: leave image and runtime log rotation to kubelet 2026-10-03 01:25:25 -05:00
jenkins
103e20645e postgres: reserve database capacity and probe pinned runtime 2026-10-03 01:23:30 -05:00
jenkins
5f3f31849c vault: keep injector replicas on separate workers 2026-10-03 01:18:45 -05:00
jenkins
440244f3ab jenkins: queue builds behind a two-agent cluster budget 2026-10-03 01:13:41 -05:00
jenkins
3eaada682c backup: add native application database recovery bundles 2026-10-03 01:12:43 -05:00
jenkins
4907c3c898 finance: reschedule stalled startup after worker quarantine 2026-10-03 01:11:13 -05:00
jenkins
2f84de22ad core: bound node role reconciliation without overlapping jobs 2026-10-03 01:11:13 -05:00
jenkins
7fa3bc9c34 core: quarantine workers with sustained runtime storage stalls 2026-10-03 01:04:58 -05:00
jenkins
3355532e00 monitoring: allow exporter rollout past offline nodes 2026-10-03 00:54:08 -05:00
jenkins
fc2be38643 docs: publish cluster operating guide and tracked repair evidence 2026-10-03 00:51:55 -05:00
jenkins
b85850986b finance: give Firefly initialization a bounded startup window 2026-10-03 00:48:00 -05:00
jenkins
0780c93f69 monitoring: expose datastore backup freshness and missing copies 2026-10-03 00:46:38 -05:00
jenkins
0b7669f0f6 flux: finish ownership migration for legacy application definitions 2026-10-03 00:40:04 -05:00
jenkins
6c9398ea03 gitops: restore missing Helm resources through drift correction 2026-10-03 00:37:41 -05:00
jenkins
3a8afd50af flux: reconcile definitions from Git and adopt active service state 2026-10-03 00:37:41 -05:00
jenkins
fa9251ae48 scheduling: stop evicting pods for historical restart counts 2026-10-03 00:35:10 -05:00
jenkins
2ae4020eb8 monitoring: allow metrics storage on the Pi fallback pool 2026-10-03 00:35:10 -05:00
jenkins
480cd0df39 monitoring: prefer storage-ready worker for metrics volume 2026-10-03 00:29:38 -05:00
jenkins
3ebdaa813a backup: exclude local-only inference workspaces from cloud policy 2026-10-03 00:24:29 -05:00
jenkins
2ade2b5a4c monitoring: recover Pushgateway away from stale iSCSI session 2026-10-03 00:24:29 -05:00
jenkins
3159759e3a logging: use the OpenSearch chart nodeAffinity setting 2026-10-03 00:22:44 -05:00
jenkins
3519a54801 logging: schedule OpenSearch with realistic memory and fallback 2026-10-03 00:19:34 -05:00
jenkins
80aff49855 recovery: schedule native datastore backups with LAN replica 2026-10-03 00:19:34 -05:00
jenkins
f04c84ee7c maintenance: retire unconditional agent restart daemon 2026-10-03 00:16:57 -05:00
jenkins
99c7308b5f av-test: allow bounded ARM readiness interpreter startup 2026-10-02 21:50:55 -05:00
jenkins
dd394b3eb3 av-test: generic page and bounded numeric-only receipts 2026-10-02 21:47:13 -05:00
flux-bot
1e06701d69 chore(maintenance): automated image update 2026-10-03 01:59:00 +00:00
flux-bot
9e364d6894 chore(maintenance): automated image update 2026-10-03 01:57:44 +00:00
flux-bot
0c763dd983 chore(maintenance): automated image update 2026-10-03 01:54:50 +00:00
flux-bot
a7a2e18912 chore(maintenance): automated image update 2026-10-03 01:48:41 +00:00
jenkins
e035cadb4d lesavka: host no-upload receiver observer at dedicated HTTPS path
Some checks failed
Tests / Declarative: Post Actions failed: 42, skipped: 89, passed: 4273
2026-10-02 17:02:43 -05:00
flux-bot
d114ce71e8 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 43, skipped: 89, passed: 4272
2026-10-02 01:33:13 +00:00
jenkins
04a737e156 maintenance: restore titan-22 overlay after USB disconnect
Some checks failed
Tests / Declarative: Post Actions failed: 43, skipped: 89, passed: 4272
2026-09-30 11:53:55 -05:00
jenkins
3a1c5ea2dc hermes: record successful LAN recovery verification 2026-09-30 11:11:04 -05:00
jenkins
0574eb590c hermes: recover suite planner on healthy amd64 node 2026-09-30 11:05:59 -05:00
jenkins
9255a64b23 hermes: recover suite passes and adapt large-suite analysis 2026-09-30 01:52:19 -05:00
jenkins
cef01b0d72 hermes: adapt suite review capacity and report admission failures 2026-09-30 00:58:42 -05:00
jenkins
83446c23dc fix(hermes): allow explicit one-hour suite job budgets 2026-09-29 23:01:55 -05:00
flux-bot
190a6de9d2 chore(maintenance): automated image update 2026-09-30 01:56:25 +00:00
flux-bot
8ca5afbc18 chore(maintenance): automated image update 2026-09-30 01:55:21 +00:00
flux-bot
54f3e824d6 chore(maintenance): automated image update 2026-09-30 01:53:25 +00:00
flux-bot
9b4e3ecb8c chore(maintenance): automated image update 2026-09-30 01:48:16 +00:00
jenkins
1bc2f22bcc docs(hermes): record 61-case diagnostic acceptance
Some checks failed
Tests / Declarative: Post Actions failed: 42, skipped: 89, passed: 4218
2026-09-29 16:48:41 -05:00
jenkins
6178265f70 fix(hermes): retain safe CLI failure categories 2026-09-29 16:45:53 -05:00
jenkins
ce3025fb9c hermes: select xhigh reasoning for large suite jobs 2026-09-29 16:05:30 -05:00
jenkins
a281c99634 hermes: raise suite planning reasoning effort to high 2026-09-29 15:57:56 -05:00
jenkins
bb4c0b6951 docs: record Claude 5.5 deployment checks 2026-09-29 15:53:45 -05:00
jenkins
206c61993d hermes: upgrade suite planner for Claude 5.5 models 2026-09-29 15:50:36 -05:00
jenkins
3036064bf7 hermes: raise suite job limits to thirty minutes and thirty dollars 2026-09-29 14:56:18 -05:00
jenkins
0382ac463d docs: record multi-pass suite acceptance and progress results 2026-09-29 14:28:05 -05:00
jenkins
9c327c6fd0 hermes: require one structured assignment per suite case 2026-09-29 14:10:03 -05:00
jenkins
15c05cb7fb hermes: report suite progress and extend approved job bounds 2026-09-29 13:53:46 -05:00
jenkins
7af4f4f206 hermes: reconcile whole-suite proposals and cap tasks at five cases 2026-09-29 13:29:10 -05:00
jenkins
edb83b78d9 docs: record suite CLI failure diagnosis and synthetic regression 2026-09-29 11:47:21 -05:00
jenkins
2d762776c1 hermes: retain safe CLI diagnostics and allow format repair headroom 2026-09-29 11:43:02 -05:00
jenkins
5f58df0f03 docs: record implementation-proximity prompt acceptance 2026-09-29 11:04:04 -05:00
jenkins
a7319ea03b hermes: group suites by incremental test implementation effort 2026-09-29 10:59:42 -05:00
flux-bot
c1ad84d9a4 chore(maintenance): automated image update 2026-09-29 13:50:30 +00:00
jenkins
1c8c7cf836 docs: publish suite planning acceptance results and client contract 2026-09-29 08:44:20 -05:00
jenkins
299b9ed16d hermes: scope approved Claude access to an operational credential 2026-09-29 08:33:35 -05:00
jenkins
16b30e3a37 hermes: pin observed Opus runtime for suite planning 2026-09-29 08:16:13 -05:00
jenkins
5af516f9c2 hermes: pin account-listed Claude context variant and bound ingress 2026-09-29 08:09:48 -05:00
jenkins
089fe21384 hermes: validate canonical Claude model aliases 2026-09-29 08:04:27 -05:00
jenkins
d7d2d5430b hermes: verify native planning runtime before readiness 2026-09-29 07:56:36 -05:00
jenkins
9f0d74d01c hermes: add policy-controlled whole-suite planning API 2026-09-29 07:50:52 -05:00
flux-bot
6990c2dcb7 chore(maintenance): automated image update 2026-09-29 01:59:51 +00:00
flux-bot
0d805b61ad chore(maintenance): automated image update 2026-09-29 01:58:49 +00:00
flux-bot
7c75356ee6 chore(maintenance): automated image update 2026-09-29 01:54:45 +00:00
flux-bot
f4e838e594 chore(maintenance): automated image update 2026-09-29 01:49:47 +00:00
jenkins
616865ba5c docs: record verified RTX LAN endpoint and pilot limitations
Some checks failed
Tests / Declarative: Post Actions failed: 48, skipped: 89, passed: 4063
2026-09-28 19:24:24 -05:00
jenkins
1e084493a9 ai: suppress backend content logs and verify isolated failures 2026-09-28 19:18:27 -05:00
jenkins
45e9fa5635 hermes: serve the LAN pilot on the reserved RTX 3080 2026-09-28 19:12:28 -05:00
jenkins
39036a0f9e ai: reserve RTX 3080 for the local inference pilot 2026-09-28 19:04:14 -05:00
jenkins
20bf855159 hermes: harden native LAN inference and document its contract 2026-09-28 18:58:00 -05:00
jenkins
279c363835 ai: add scoped planning adapter and fix structured reasoning calls 2026-09-28 18:39:44 -05:00
jenkins
8a3f53082f ai: keep invalid planning results out of reusable cache 2026-09-28 18:10:16 -05:00
jenkins
e9c901f177 ai: add curl access for the private batch API 2026-09-28 18:07:53 -05:00
jenkins
ff64e27aec docs: record local planning pilot and export boundaries 2026-09-28 18:07:06 -05:00
jenkins
721e013576 ai: add private batch client and planning pilot contracts 2026-09-28 18:06:22 -05:00
jenkins
7dc36c89a0 hermes: expose pinned local batch inference behind LAN auth 2026-09-28 17:58:58 -05:00
jenkins
ee7b83b5db ai: allow batch pilot on the reserved titan-23 node 2026-09-28 17:54:47 -05:00
jenkins
360af7a421 ai: isolate CPU batch inference on titan-23 2026-09-28 17:52:23 -05:00
jenkins
713cb8f80f hermes: serve authenticated local inference on the LAN 2026-09-28 17:10:12 -05:00
jenkins
4bd487b8b3 traefik: add private ingress with preserved client addresses 2026-09-28 17:10:12 -05:00
jenkins
0274335dfe vault: provision Hermes LAN model gateway credentials 2026-09-28 17:06:07 -05:00
flux-bot
fc321d2fcc chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 40, skipped: 89, passed: 3967
2026-09-26 14:01:44 +00:00
flux-bot
725c642d67 chore(maintenance): automated image update 2026-09-26 14:01:29 +00:00
flux-bot
4cc46ea36f chore(maintenance): automated image update 2026-09-26 13:58:30 +00:00
flux-bot
8b20d93632 chore(maintenance): automated image update 2026-09-26 13:53:27 +00:00
flux-bot
1411983176 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 42, skipped: 89, passed: 3965
2026-09-25 13:57:34 +00:00
flux-bot
8e3d071f2c chore(maintenance): automated image update 2026-09-25 13:57:13 +00:00
flux-bot
344f300448 chore(maintenance): automated image update 2026-09-25 13:54:33 +00:00
flux-bot
52004a5cc9 chore(maintenance): automated image update 2026-09-25 13:49:09 +00:00
flux-bot
a5eed7687c chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-25 01:52:43 +00:00
flux-bot
16d35e76de chore(maintenance): automated image update 2026-09-25 01:48:19 +00:00
flux-bot
e37ad9baae chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 40, skipped: 89, passed: 3967
2026-09-24 13:56:57 +00:00
flux-bot
d27e4c964d chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-24 01:58:15 +00:00
flux-bot
801f462f10 chore(maintenance): automated image update 2026-09-24 01:57:58 +00:00
flux-bot
e0ee42b10b chore(maintenance): automated image update 2026-09-24 01:53:56 +00:00
flux-bot
ca29b93a50 chore(maintenance): automated image update 2026-09-24 01:48:55 +00:00
flux-bot
6378dc12b5 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-23 13:58:54 +00:00
flux-bot
78d2613ec2 chore(maintenance): automated image update 2026-09-23 13:57:47 +00:00
flux-bot
62c4780369 chore(maintenance): automated image update 2026-09-23 13:53:44 +00:00
flux-bot
cdd5f63e12 chore(maintenance): automated image update 2026-09-23 13:48:45 +00:00
flux-bot
771f21a188 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 45, skipped: 89, passed: 3962
2026-09-22 01:55:30 +00:00
flux-bot
091bb19097 chore(maintenance): automated image update 2026-09-22 01:54:28 +00:00
flux-bot
576b0e9c3b chore(maintenance): automated image update 2026-09-22 01:52:30 +00:00
flux-bot
64a5ad4efd chore(maintenance): automated image update 2026-09-22 01:47:27 +00:00
flux-bot
f4cfd8f09b chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-21 09:52:45 +00:00
flux-bot
e40272fa4c chore(bstein-dev-home): automated image update 2026-09-21 09:50:45 +00:00
flux-bot
cc553a4332 chore(maintenance): automated image update 2026-09-21 08:28:36 +00:00
flux-bot
796489ab24 chore(maintenance): automated image update 2026-09-21 08:27:33 +00:00
flux-bot
7bb53a3179 chore(maintenance): automated image update 2026-09-21 08:26:32 +00:00
flux-bot
5276f5fe83 chore(maintenance): automated image update 2026-09-21 08:24:32 +00:00
flux-bot
b79a689840 chore(maintenance): automated image update 2026-09-21 01:35:24 +00:00
flux-bot
e40ed083f6 chore(maintenance): automated image update 2026-09-21 01:35:10 +00:00
flux-bot
be67df15fc chore(maintenance): automated image update 2026-09-21 01:34:07 +00:00
flux-bot
736c344f3c chore(maintenance): automated image update 2026-09-21 01:32:12 +00:00
flux-bot
39d9709f90 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-20 13:47:06 +00:00
flux-bot
4d13a549a2 chore(maintenance): automated image update 2026-09-20 13:46:06 +00:00
flux-bot
cf36a1164b chore(maintenance): automated image update 2026-09-20 13:45:05 +00:00
flux-bot
33b5c06e34 chore(maintenance): automated image update 2026-09-20 13:43:05 +00:00
flux-bot
e17dc54a54 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-20 09:39:36 +00:00
flux-bot
32f8395021 chore(bstein-dev-home): automated image update 2026-09-20 09:38:35 +00:00
flux-bot
26ff3de357 chore(maintenance): automated image update 2026-09-20 05:25:51 +00:00
flux-bot
1a1ba5903d chore(maintenance): automated image update 2026-09-20 01:57:34 +00:00
flux-bot
f333127fb4 chore(maintenance): automated image update 2026-09-20 01:57:25 +00:00
flux-bot
bb9e8f2465 chore(maintenance): automated image update 2026-09-20 01:54:25 +00:00
flux-bot
40d0e8975c chore(maintenance): automated image update 2026-09-20 01:48:30 +00:00
flux-bot
fed2af5758 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-19 17:17:07 +00:00
flux-bot
9aa88b59db chore(maintenance): automated image update 2026-09-19 13:56:04 +00:00
flux-bot
e06adfb556 chore(maintenance): automated image update 2026-09-19 13:55:48 +00:00
flux-bot
29c8861fa7 chore(maintenance): automated image update 2026-09-19 13:52:48 +00:00
flux-bot
daca71b49f chore(maintenance): automated image update 2026-09-19 13:47:53 +00:00
flux-bot
646ad43e33 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-19 05:15:39 +00:00
flux-bot
f011700c5c chore(maintenance): automated image update 2026-09-19 01:49:26 +00:00
flux-bot
e78281b9f6 chore(maintenance): automated image update 2026-09-19 01:49:17 +00:00
flux-bot
b601e63186 chore(maintenance): automated image update 2026-09-19 01:46:17 +00:00
flux-bot
97e51d6397 chore(maintenance): automated image update 2026-09-19 01:42:24 +00:00
flux-bot
9e36bb9fdc chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-18 17:41:10 +00:00
flux-bot
5ab5a473b4 chore(maintenance): automated image update 2026-09-18 13:55:56 +00:00
flux-bot
608b062350 chore(maintenance): automated image update 2026-09-18 13:54:59 +00:00
flux-bot
236193c745 chore(maintenance): automated image update 2026-09-18 13:52:51 +00:00
flux-bot
2e7dbfe6ce chore(maintenance): automated image update 2026-09-18 13:46:58 +00:00
flux-bot
af8bf0c86c chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-18 05:17:42 +00:00
flux-bot
eb7576c58e chore(maintenance): automated image update 2026-09-18 01:55:23 +00:00
flux-bot
37142cd811 chore(maintenance): automated image update 2026-09-18 01:54:30 +00:00
flux-bot
1408e75dcb chore(maintenance): automated image update 2026-09-18 01:52:23 +00:00
flux-bot
5e22f5999d chore(maintenance): automated image update 2026-09-18 01:45:30 +00:00
flux-bot
bb7215d091 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-17 17:15:15 +00:00
flux-bot
a680cf87fb chore(maintenance): automated image update 2026-09-17 13:47:04 +00:00
flux-bot
8bbf9f0f8a chore(maintenance): automated image update 2026-09-17 13:46:55 +00:00
flux-bot
4cabb79784 chore(maintenance): automated image update 2026-09-17 13:45:55 +00:00
flux-bot
7b7ddd2761 chore(maintenance): automated image update 2026-09-17 13:44:01 +00:00
flux-bot
74e575af0b chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-17 05:16:48 +00:00
flux-bot
cd7c8da4c1 chore(maintenance): automated image update 2026-09-16 05:16:55 +00:00
flux-bot
cfde38735d chore(maintenance): automated image update 2026-09-16 02:00:36 +00:00
flux-bot
820ac8cfd2 chore(maintenance): automated image update 2026-09-16 01:58:41 +00:00
flux-bot
41b56c2ff4 chore(maintenance): automated image update 2026-09-16 01:56:36 +00:00
flux-bot
1951b82a21 chore(maintenance): automated image update 2026-09-16 01:49:39 +00:00
flux-bot
442f22a9b6 chore(bstein-dev-home): automated image update 2026-09-15 21:41:59 +00:00
flux-bot
b59bf0f33f chore(bstein-dev-home): automated image update 2026-09-15 21:40:15 +00:00
flux-bot
8374bb4b11 chore(maintenance): automated image update 2026-09-15 17:18:30 +00:00
flux-bot
b23ee363e3 chore(maintenance): automated image update 2026-09-15 13:34:19 +00:00
flux-bot
3f18dfc44d chore(maintenance): automated image update 2026-09-15 13:34:07 +00:00
flux-bot
071ae5d1c9 chore(maintenance): automated image update 2026-09-15 13:33:07 +00:00
flux-bot
20b5076c53 chore(maintenance): automated image update 2026-09-15 13:31:09 +00:00
flux-bot
910daadb92 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-15 10:00:35 +00:00
flux-bot
0543172783 chore(bstein-dev-home): automated image update 2026-09-15 09:59:51 +00:00
flux-bot
283faeec96 chore(maintenance): automated image update 2026-09-15 05:17:59 +00:00
flux-bot
25d5a8d9bf chore(maintenance): automated image update 2026-09-15 01:59:43 +00:00
flux-bot
815a0719f2 chore(maintenance): automated image update 2026-09-15 01:58:46 +00:00
flux-bot
7816439804 chore(maintenance): automated image update 2026-09-15 01:55:43 +00:00
flux-bot
4ac4b813b9 chore(maintenance): automated image update 2026-09-15 01:46:45 +00:00
flux-bot
122d707c50 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-14 21:38:08 +00:00
flux-bot
cda136682e chore(bstein-dev-home): automated image update 2026-09-14 21:37:23 +00:00
flux-bot
4f4cdd7e1a chore(maintenance): automated image update 2026-09-14 17:16:30 +00:00
flux-bot
7f46d638b9 chore(maintenance): automated image update 2026-09-14 13:34:20 +00:00
flux-bot
38060acc96 chore(maintenance): automated image update 2026-09-14 13:34:12 +00:00
flux-bot
6652f0d513 chore(maintenance): automated image update 2026-09-14 13:32:12 +00:00
flux-bot
a8d21531a9 chore(maintenance): automated image update 2026-09-14 13:30:14 +00:00
flux-bot
028e182d8a chore(hermes): promote validated image release
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 89, passed: 3968
2026-09-14 06:59:45 +00:00
jenkins
302d6bb132 ci(hermes): reserve disk space before image builds 2026-09-14 00:55:21 -05:00
jenkins
cfcd932065 fix(hermes): preserve paused editors during terminal reattachment 2026-09-14 00:36:31 -05:00
jenkins
15c79c6da4 Merge remote-tracking branch 'origin/main' 2026-09-14 00:28:23 -05:00
jenkins
7a69e9ea5c fix(hermes): relay history scrolling and repaint reattached terminals 2026-09-14 00:27:30 -05:00
flux-bot
a2ffedc5f9 chore(maintenance): automated image update 2026-09-14 05:14:57 +00:00
jenkins
990a83d34b fix(hermes): scroll dashboard conversation history in the TUI 2026-09-14 00:09:43 -05:00
flux-bot
1433271c76 chore(maintenance): automated image update 2026-09-14 01:50:54 +00:00
flux-bot
5d510f308e chore(maintenance): automated image update 2026-09-14 01:50:45 +00:00
flux-bot
d0e270cbe7 chore(maintenance): automated image update 2026-09-14 01:48:45 +00:00
jenkins
c7b7d622ba Merge remote-tracking branch 'origin/main' 2026-09-13 20:45:55 -05:00
jenkins
71d0b171e6 flux: allow degraded Vault provider fleets during workload rollouts 2026-09-13 20:45:09 -05:00
flux-bot
cc3914c9cd chore(maintenance): automated image update 2026-09-14 01:43:46 +00:00
jenkins
e07cbdfce8 hermes: recover confirmation for the published Soteria task 2026-09-13 20:41:00 -05:00
jenkins
73618ca8f8 hermes: accept verified Forgejo draft update acknowledgments 2026-09-13 20:32:12 -05:00
jenkins
33e6f84a64 hermes: publish self-contained Git packs through the SCM broker 2026-09-13 20:07:11 -05:00
jenkins
82cc28b383 hermes: enforce native provider availability for manual assignments
Some checks failed
Tests / Declarative: Post Actions failed: 42, skipped: 89, passed: 3938
2026-09-13 19:26:44 -05:00
jenkins
18f5cfdbb5 hermes: mount the Go installer into execution worker init 2026-09-13 19:19:34 -05:00
jenkins
0a51ffe4a7 hermes: resume failed PR publications with trusted bounded retries 2026-09-13 19:11:35 -05:00
jenkins
fe3e9ba16b hermes: let chat updates proceed while SCM is unavailable 2026-09-13 19:05:48 -05:00
jenkins
3cdcb2a2df hermes: equip workers with verified Go tools before accepting work 2026-09-13 18:48:04 -05:00
jenkins
d66af9dbd4 hermes: deploy verified release and roll broker code by content 2026-09-13 18:44:33 -05:00
jenkins
6cfeff74d6 hermes: classify mediator conflicts as publication rejections 2026-09-13 18:37:56 -05:00
jenkins
00b8780345 hermes: normalize preserved titles and report retry conflicts 2026-09-13 18:30:05 -05:00
jenkins
cac4c206cd hermes: bound publication titles by UTF-8 bytes 2026-09-13 18:17:39 -05:00
jenkins
3231e1af07 hermes: recover structured provider completion records 2026-09-13 18:09:47 -05:00
jenkins
9cc49dedeb hermes: sign recovery attestations for the target worker 2026-09-13 18:07:04 -05:00
jenkins
743014ca33 hermes: pin the legacy publication terminal outcome 2026-09-13 18:01:18 -05:00
jenkins
37bd860a47 hermes: recognize retained blocked publication runs 2026-09-13 17:59:26 -05:00
jenkins
bf2a1ab528 hermes: resume preserved PR publications without model retries 2026-09-13 17:54:54 -05:00
jenkins
4a94977dbc hermes: stage complete lane regression dependencies 2026-09-13 17:32:45 -05:00
jenkins
36209965b4 hermes: clarify routine implementation routing effort 2026-09-13 17:04:34 -05:00
jenkins
8f2359b635 hermes: preserve legacy provider session directories 2026-09-13 16:52:44 -05:00
jenkins
13c3891996 hermes: align classifier examples with capability routes 2026-09-13 16:52:44 -05:00
jenkins
bd5f261402 hermes: use native task links throughout PR follow-ups 2026-09-13 16:42:49 -05:00
jenkins
38a794c2a9 hermes: resolve PR follow-ups from their trusted root 2026-09-13 16:27:01 -05:00
jenkins
4d7087d394 hermes: move chat tenants off the failing node runtime 2026-09-13 16:23:55 -05:00
jenkins
13a7dbab46 hermes: reserve PR follow-ups for mediated workers 2026-09-13 16:15:50 -05:00
jenkins
74d8485f6a hermes: isolate PR state and recover the Soteria board 2026-09-13 15:46:39 -05:00
jenkins
106ee8ce14 hermes: select the fixed SQLite library on ARM 2026-09-13 15:44:44 -05:00
jenkins
ce3a68616a hermes: retain SQLite search and extension support 2026-09-13 15:38:39 -05:00
jenkins
d12145db01 hermes: use the portable SQLite install target 2026-09-13 15:34:13 -05:00
jenkins
e0c3f6591d hermes: fix the SQLite WAL reset runtime bug 2026-09-13 15:30:40 -05:00
jenkins
aaeb47eb40 hermes: replace the SCM broker without volume contention 2026-09-13 15:07:43 -05:00
jenkins
e040102121 hermes: route by capability and continue existing pull requests 2026-09-13 15:04:01 -05:00
jenkins
46f9995fe5 hermes: roll out the resumed transcript patch 2026-09-13 14:27:29 -05:00
jenkins
4dd5d1fcf7 vault: scope Hermes task grants to the SCM broker 2026-09-13 14:26:54 -05:00
jenkins
17d70759e0 hermes: show recorded tool activity in resumed chats 2026-09-13 14:25:25 -05:00
jenkins
364d1b3503 hermes: recover interrupted provider CLI installs 2026-09-13 14:11:40 -05:00
jenkins
0192ef8ee8 hermes: mount routing catalog for execution workers 2026-09-13 14:02:20 -05:00
flux-bot
06ef4bc6c1 chore(maintenance): automated image update 2026-09-13 13:41:31 +00:00
flux-bot
fb2b71acd1 chore(maintenance): automated image update 2026-09-13 13:41:21 +00:00
flux-bot
591db11835 chore(maintenance): automated image update 2026-09-13 13:40:16 +00:00
flux-bot
98cea72a1b chore(maintenance): automated image update 2026-09-13 13:38:16 +00:00
flux-bot
aacfec4de5 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 53, skipped: 89, passed: 3780
2026-09-13 09:38:35 +00:00
flux-bot
64b8f5ecca chore(bstein-dev-home): automated image update 2026-09-13 09:38:04 +00:00
jenkins
969febe7e4 hermes: keep chat on verified worker hosts 2026-09-13 02:29:44 -05:00
jenkins
69022366ff hermes: avoid stalled titan-04 for chat tenants 2026-09-13 02:06:08 -05:00
jenkins
d141b33a7d hermes: resolve difficulty routes from live model catalogs 2026-09-13 01:48:05 -05:00
jenkins
2ab737f8f8 fix(hermes): hide the worker overlay on dashboard pages 2026-09-12 22:37:23 -05:00
flux-bot
240426d153 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 50, skipped: 81, passed: 3765
2026-09-12 21:55:05 +00:00
flux-bot
301e345956 chore(bstein-dev-home): automated image update 2026-09-12 21:54:40 +00:00
flux-bot
6dcc52bb7d chore(maintenance): automated image update 2026-09-12 13:38:28 +00:00
flux-bot
747dc46938 chore(maintenance): automated image update 2026-09-12 13:38:19 +00:00
flux-bot
7f09e32619 chore(maintenance): automated image update 2026-09-12 13:37:19 +00:00
flux-bot
6da1a25605 chore(maintenance): automated image update 2026-09-12 13:35:18 +00:00
flux-bot
2e5d7fc8a2 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-12 09:37:36 +00:00
flux-bot
a0a2cea573 chore(bstein-dev-home): automated image update 2026-09-12 09:37:13 +00:00
flux-bot
7acfd02809 chore(maintenance): automated image update 2026-09-12 01:37:51 +00:00
flux-bot
e4572279dd chore(maintenance): automated image update 2026-09-12 01:36:49 +00:00
flux-bot
75bc741d9b chore(maintenance): automated image update 2026-09-12 01:34:48 +00:00
flux-bot
839b15c880 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 51, skipped: 81, passed: 3764
2026-09-11 21:39:07 +00:00
flux-bot
2d6a06eafd chore(bstein-dev-home): automated image update 2026-09-11 21:37:47 +00:00
flux-bot
b6d6068c9e chore(maintenance): automated image update 2026-09-11 13:37:25 +00:00
flux-bot
12347a000c chore(maintenance): automated image update 2026-09-11 13:36:21 +00:00
flux-bot
d8faedc62c chore(maintenance): automated image update 2026-09-11 13:35:20 +00:00
flux-bot
e0ede6d3ef chore(maintenance): automated image update 2026-09-11 13:33:19 +00:00
flux-bot
1f4881e795 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 50, skipped: 81, passed: 3765
2026-09-11 10:01:44 +00:00
flux-bot
743d07e2e0 chore(bstein-dev-home): automated image update 2026-09-11 09:59:24 +00:00
flux-bot
3aff553a81 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 50, skipped: 81, passed: 3765
2026-09-10 21:53:13 +00:00
flux-bot
eb22f3d60a chore(bstein-dev-home): automated image update 2026-09-10 21:50:56 +00:00
flux-bot
6f428fb1c0 chore(maintenance): automated image update 2026-09-10 13:52:36 +00:00
flux-bot
e55102479f chore(maintenance): automated image update 2026-09-10 13:52:23 +00:00
flux-bot
2945ffbfaf chore(maintenance): automated image update 2026-09-10 13:48:21 +00:00
flux-bot
eeca7dce6a chore(maintenance): automated image update 2026-09-10 13:42:21 +00:00
flux-bot
50c06ccc66 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-10 09:53:46 +00:00
flux-bot
664972321e chore(bstein-dev-home): automated image update 2026-09-10 09:51:31 +00:00
flux-bot
e228d013ae chore(maintenance): automated image update 2026-09-10 01:52:02 +00:00
flux-bot
6c0f27b2ab chore(maintenance): automated image update 2026-09-10 01:51:53 +00:00
flux-bot
6c13835303 chore(maintenance): automated image update 2026-09-10 01:47:52 +00:00
flux-bot
49bdfffa2e chore(maintenance): automated image update 2026-09-10 01:42:52 +00:00
flux-bot
016e1bad7f chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-09 13:35:23 +00:00
flux-bot
22c12a822f chore(maintenance): automated image update 2026-09-09 13:34:23 +00:00
flux-bot
8e8eb33c1d chore(maintenance): automated image update 2026-09-09 13:33:22 +00:00
flux-bot
bf995bfc0f chore(maintenance): automated image update 2026-09-09 13:31:22 +00:00
flux-bot
2b32395975 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-09 09:37:56 +00:00
flux-bot
0e36e64a00 chore(bstein-dev-home): automated image update 2026-09-09 09:37:37 +00:00
flux-bot
ee09cd03d0 chore(maintenance): automated image update 2026-09-09 01:34:54 +00:00
flux-bot
2f7b048153 chore(maintenance): automated image update 2026-09-09 01:33:54 +00:00
flux-bot
d56e7e9a5e chore(maintenance): automated image update 2026-09-09 01:32:53 +00:00
flux-bot
c53792f8e6 chore(maintenance): automated image update 2026-09-09 01:30:53 +00:00
flux-bot
f13dd7476e chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-08 21:49:21 +00:00
flux-bot
0a8d720e07 chore(bstein-dev-home): automated image update 2026-09-08 21:48:13 +00:00
flux-bot
c7538d911b chore(maintenance): automated image update 2026-09-08 13:52:25 +00:00
flux-bot
ed8b374d43 chore(maintenance): automated image update 2026-09-08 13:51:26 +00:00
flux-bot
3ae7801532 chore(maintenance): automated image update 2026-09-08 13:48:24 +00:00
flux-bot
3efe37302f chore(maintenance): automated image update 2026-09-08 13:43:23 +00:00
flux-bot
2b9d17be7d chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-08 09:50:51 +00:00
flux-bot
c267be8c16 chore(bstein-dev-home): automated image update 2026-09-08 09:48:40 +00:00
flux-bot
38ad593162 chore(maintenance): automated image update 2026-09-08 01:34:54 +00:00
flux-bot
c2df9f5877 chore(maintenance): automated image update 2026-09-08 01:33:57 +00:00
flux-bot
cfaa5b8c18 chore(maintenance): automated image update 2026-09-08 01:32:54 +00:00
flux-bot
33d8c01d6c chore(maintenance): automated image update 2026-09-08 01:30:53 +00:00
flux-bot
9daf82e8e7 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-07 21:51:27 +00:00
flux-bot
a0f141590f chore(bstein-dev-home): automated image update 2026-09-07 21:50:16 +00:00
flux-bot
b4e95c7eac chore(maintenance): automated image update 2026-09-07 13:35:26 +00:00
flux-bot
d19bf2f57c chore(maintenance): automated image update 2026-09-07 13:34:26 +00:00
flux-bot
38da5ebf04 chore(maintenance): automated image update 2026-09-07 13:33:26 +00:00
flux-bot
f71c38f2e9 chore(maintenance): automated image update 2026-09-07 13:31:25 +00:00
flux-bot
7725f71343 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-07 09:51:59 +00:00
flux-bot
b2d575cf12 chore(bstein-dev-home): automated image update 2026-09-07 09:49:51 +00:00
flux-bot
2cb5d40fa8 chore(maintenance): automated image update 2026-09-07 01:53:06 +00:00
flux-bot
9350d77a5e chore(maintenance): automated image update 2026-09-07 01:52:56 +00:00
flux-bot
1d78b99ef5 chore(maintenance): automated image update 2026-09-07 01:47:56 +00:00
flux-bot
2034717653 chore(maintenance): automated image update 2026-09-07 01:41:55 +00:00
flux-bot
c2ef798aa1 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-06 21:39:36 +00:00
flux-bot
463458655a chore(bstein-dev-home): automated image update 2026-09-06 21:39:24 +00:00
flux-bot
f0ad63235d chore(maintenance): automated image update 2026-09-06 13:53:32 +00:00
flux-bot
3fee4bc457 chore(maintenance): automated image update 2026-09-06 13:52:31 +00:00
flux-bot
284e27edf7 chore(maintenance): automated image update 2026-09-06 13:48:29 +00:00
flux-bot
f62f6db79e chore(maintenance): automated image update 2026-09-06 13:43:29 +00:00
flux-bot
575c8e0eec chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-06 09:51:59 +00:00
flux-bot
556da6f4fd chore(bstein-dev-home): automated image update 2026-09-06 09:50:00 +00:00
flux-bot
565d4783ca chore(maintenance): automated image update 2026-09-06 01:52:59 +00:00
flux-bot
5b78bd8275 chore(maintenance): automated image update 2026-09-06 01:52:01 +00:00
flux-bot
a88d40bd97 chore(maintenance): automated image update 2026-09-06 01:47:55 +00:00
flux-bot
0eab756578 chore(maintenance): automated image update 2026-09-06 01:43:00 +00:00
flux-bot
f103014d29 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-05 13:51:27 +00:00
flux-bot
2f19cec59b chore(maintenance): automated image update 2026-09-05 13:50:32 +00:00
flux-bot
395f168b38 chore(maintenance): automated image update 2026-09-05 13:47:27 +00:00
flux-bot
ad7f4bd122 chore(maintenance): automated image update 2026-09-05 13:42:32 +00:00
flux-bot
4016b38619 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-05 09:52:45 +00:00
flux-bot
9d311607a5 chore(bstein-dev-home): automated image update 2026-09-05 09:50:27 +00:00
flux-bot
cda120e475 chore(maintenance): automated image update 2026-09-05 01:53:01 +00:00
flux-bot
fe5bf21b52 chore(maintenance): automated image update 2026-09-05 01:52:05 +00:00
flux-bot
b97794be7b chore(maintenance): automated image update 2026-09-05 01:48:04 +00:00
flux-bot
838179637f chore(maintenance): automated image update 2026-09-05 01:43:05 +00:00
flux-bot
b7bf1eca3f chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-04 21:38:20 +00:00
flux-bot
e4a3d6a9be chore(bstein-dev-home): automated image update 2026-09-04 21:38:01 +00:00
flux-bot
dd735fa182 chore(maintenance): automated image update 2026-09-04 13:34:33 +00:00
flux-bot
1b4051b737 chore(maintenance): automated image update 2026-09-04 13:33:38 +00:00
flux-bot
d5e24aa0e8 chore(maintenance): automated image update 2026-09-04 13:32:32 +00:00
flux-bot
a8feb7328a chore(maintenance): automated image update 2026-09-04 13:30:36 +00:00
flux-bot
fd5c31c6e4 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-04 10:01:57 +00:00
flux-bot
7572776974 chore(bstein-dev-home): automated image update 2026-09-04 09:59:37 +00:00
flux-bot
66179e82ae chore(maintenance): automated image update 2026-09-04 01:54:15 +00:00
flux-bot
9d7f3e2a39 chore(maintenance): automated image update 2026-09-04 01:54:06 +00:00
flux-bot
48e1718814 chore(maintenance): automated image update 2026-09-04 01:49:05 +00:00
flux-bot
0f3140fd98 chore(maintenance): automated image update 2026-09-04 01:43:09 +00:00
flux-bot
5a479cf474 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 52, skipped: 81, passed: 3763
2026-09-03 21:52:16 +00:00
flux-bot
59783b912b chore(bstein-dev-home): automated image update 2026-09-03 21:50:10 +00:00
flux-bot
1112483510 chore(maintenance): automated image update 2026-09-03 13:34:53 +00:00
flux-bot
07931c4113 chore(maintenance): automated image update 2026-09-03 13:34:41 +00:00
flux-bot
999f3e0448 chore(maintenance): automated image update 2026-09-03 13:33:36 +00:00
flux-bot
db8bc715a6 chore(maintenance): automated image update 2026-09-03 13:31:41 +00:00
flux-bot
2b51d87a8b chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3766
2026-09-03 09:52:52 +00:00
flux-bot
00ad2edf2b chore(bstein-dev-home): automated image update 2026-09-03 09:50:44 +00:00
jenkins
87c0a8b201 fix(monitoring): bound Titan test metric labels 2026-09-02 23:50:39 -03:00
flux-bot
b34888abb3 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3765
2026-09-02 21:39:24 +00:00
flux-bot
84dd5d049b chore(bstein-dev-home): automated image update 2026-09-02 21:39:17 +00:00
flux-bot
1bfdffdd11 chore(maintenance): automated image update 2026-09-02 13:52:47 +00:00
flux-bot
c1efca420a chore(maintenance): automated image update 2026-09-02 13:51:52 +00:00
flux-bot
8a9e4204fe chore(maintenance): automated image update 2026-09-02 13:47:46 +00:00
flux-bot
fa7a81c235 chore(maintenance): automated image update 2026-09-02 13:42:50 +00:00
flux-bot
8bf8214192 chore(bstein-dev-home): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3765
2026-09-02 09:39:01 +00:00
flux-bot
01d7192630 chore(bstein-dev-home): automated image update 2026-09-02 09:37:52 +00:00
flux-bot
4ca04c6300 chore(maintenance): automated image update 2026-09-02 01:52:30 +00:00
flux-bot
bf1a54a091 chore(maintenance): automated image update 2026-09-02 01:52:20 +00:00
flux-bot
14fd7a9791 chore(maintenance): automated image update 2026-09-02 01:48:20 +00:00
flux-bot
d0a71dde5c chore(maintenance): automated image update 2026-09-02 01:42:24 +00:00
jenkins
0f32df9aee docs: follow local atlas-iac checkout path 2026-09-01 21:08:18 -03:00
jenkins
7b44db48c2 hermes: reload broker for Titan SCM paths 2026-09-01 21:02:44 -03:00
jenkins
bbfa2018e2 gitea: preserve completed identity bootstrap job 2026-09-01 20:51:49 -03:00
jenkins
8d1302765f gitea: migrate sources to titan/atlas-iac 2026-09-01 20:43:50 -03:00
flux-bot
3daa3019f9 chore(bstein-dev-home): automated image update 2026-09-01 21:52:37 +00:00
flux-bot
b35edc33b8 chore(bstein-dev-home): automated image update 2026-09-01 21:50:26 +00:00
flux-bot
e89dc06202 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 48, skipped: 81, passed: 3766
2026-09-01 13:33:50 +00:00
flux-bot
f0cbf1de70 chore(maintenance): automated image update 2026-09-01 13:32:56 +00:00
flux-bot
889988e656 chore(maintenance): automated image update 2026-09-01 13:31:50 +00:00
flux-bot
4b1d8d4a90 chore(maintenance): automated image update 2026-09-01 13:29:56 +00:00
flux-bot
a13a11e784 chore(bstein-dev-home): automated image update 2026-09-01 09:39:09 +00:00
flux-bot
8b2bfeabf8 chore(bstein-dev-home): automated image update 2026-09-01 09:38:00 +00:00
flux-bot
e31596a661 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 48, skipped: 81, passed: 3766
2026-09-01 01:36:23 +00:00
flux-bot
f10fbacbed chore(maintenance): automated image update 2026-09-01 01:34:21 +00:00
flux-bot
da30779443 chore(maintenance): automated image update 2026-09-01 01:33:22 +00:00
flux-bot
3b1f88b115 chore(bstein-dev-home): automated image update 2026-08-31 21:42:37 +00:00
flux-bot
debff1c88b chore(bstein-dev-home): automated image update 2026-08-31 21:41:34 +00:00
flux-bot
a02aa7361d chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 51, skipped: 81, passed: 3763
2026-08-31 13:35:07 +00:00
flux-bot
b3a42e7f5c chore(maintenance): automated image update 2026-08-31 13:34:50 +00:00
flux-bot
4f70ebffa7 chore(maintenance): automated image update 2026-08-31 13:32:49 +00:00
flux-bot
ecf421f949 chore(maintenance): automated image update 2026-08-31 13:30:49 +00:00
flux-bot
76a789f514 chore(bstein-dev-home): automated image update 2026-08-31 09:37:12 +00:00
flux-bot
d26bce5c8e chore(bstein-dev-home): automated image update 2026-08-31 09:36:09 +00:00
flux-bot
a95cae1c03 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3765
2026-08-31 01:35:21 +00:00
flux-bot
6f7391cab2 chore(maintenance): automated image update 2026-08-31 01:34:21 +00:00
flux-bot
a4a22f1559 chore(maintenance): automated image update 2026-08-31 01:33:21 +00:00
flux-bot
ba72b2376c chore(maintenance): automated image update 2026-08-31 01:31:21 +00:00
flux-bot
edd52154e1 chore(bstein-dev-home): automated image update 2026-08-30 21:37:51 +00:00
flux-bot
97afaa00e7 chore(bstein-dev-home): automated image update 2026-08-30 21:37:42 +00:00
flux-bot
862a1e2c3d chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 48, skipped: 81, passed: 3766
2026-08-30 13:34:45 +00:00
flux-bot
9a54d0995d chore(maintenance): automated image update 2026-08-30 13:33:43 +00:00
flux-bot
acf90fae0b chore(maintenance): automated image update 2026-08-30 13:31:43 +00:00
flux-bot
81ada2389d chore(bstein-dev-home): automated image update 2026-08-30 09:37:13 +00:00
flux-bot
5ce958be6c chore(bstein-dev-home): automated image update 2026-08-30 09:37:05 +00:00
flux-bot
bac9bfd047 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 48, skipped: 81, passed: 3766
2026-08-30 01:34:13 +00:00
flux-bot
87cf86fb84 chore(maintenance): automated image update 2026-08-30 01:33:14 +00:00
flux-bot
ffd93cdf93 chore(maintenance): automated image update 2026-08-30 01:32:13 +00:00
flux-bot
fa6dd0a134 chore(maintenance): automated image update 2026-08-30 01:30:12 +00:00
flux-bot
7397a5fbbe chore(bstein-dev-home): automated image update 2026-08-29 21:38:40 +00:00
flux-bot
e0492a2996 chore(bstein-dev-home): automated image update 2026-08-29 21:37:38 +00:00
flux-bot
81c96ee6ae chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 48, skipped: 81, passed: 3766
2026-08-29 13:34:50 +00:00
flux-bot
811dbb6fab chore(maintenance): automated image update 2026-08-29 13:34:41 +00:00
flux-bot
1d3fa82f35 chore(maintenance): automated image update 2026-08-29 13:33:41 +00:00
flux-bot
b4bbe2583a chore(maintenance): automated image update 2026-08-29 13:31:42 +00:00
flux-bot
0f15c6cbd3 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 48, skipped: 81, passed: 3766
2026-08-29 01:35:10 +00:00
flux-bot
ec349682de chore(maintenance): automated image update 2026-08-29 01:34:13 +00:00
flux-bot
57d506b146 chore(maintenance): automated image update 2026-08-29 01:33:10 +00:00
flux-bot
53b1d3f1a5 chore(maintenance): automated image update 2026-08-29 01:31:10 +00:00
flux-bot
2d62e61b47 chore(bstein-dev-home): automated image update 2026-08-28 21:39:42 +00:00
flux-bot
eba2745fbb chore(bstein-dev-home): automated image update 2026-08-28 21:38:41 +00:00
flux-bot
d492d5038f chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 48, skipped: 81, passed: 3766
2026-08-28 13:33:57 +00:00
flux-bot
1748dc7a88 chore(maintenance): automated image update 2026-08-28 13:33:45 +00:00
flux-bot
c7156a4574 chore(maintenance): automated image update 2026-08-28 13:32:42 +00:00
flux-bot
827c029c45 chore(maintenance): automated image update 2026-08-28 13:30:41 +00:00
flux-bot
1688066b13 chore(bstein-dev-home): automated image update 2026-08-28 09:51:13 +00:00
flux-bot
d0428fa1f7 chore(bstein-dev-home): automated image update 2026-08-28 09:49:14 +00:00
flux-bot
40c441640c chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 48, skipped: 81, passed: 3766
2026-08-28 01:34:04 +00:00
flux-bot
58399d345d chore(maintenance): automated image update 2026-08-28 01:33:05 +00:00
flux-bot
afd3aa2f85 chore(maintenance): automated image update 2026-08-28 01:32:03 +00:00
flux-bot
0243689d1a chore(maintenance): automated image update 2026-08-28 01:30:04 +00:00
flux-bot
4ce2daa934 chore(bstein-dev-home): automated image update 2026-08-27 22:06:34 +00:00
flux-bot
689715688d chore(bstein-dev-home): automated image update 2026-08-27 22:05:44 +00:00
flux-bot
6d202d7c8a chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 48, skipped: 81, passed: 3766
2026-08-27 13:35:38 +00:00
flux-bot
9757db13ab chore(maintenance): automated image update 2026-08-27 13:34:39 +00:00
flux-bot
464a0fe477 chore(maintenance): automated image update 2026-08-27 13:33:38 +00:00
flux-bot
a7379ea52a chore(maintenance): automated image update 2026-08-27 13:31:37 +00:00
flux-bot
36a8e6e4c6 chore(bstein-dev-home): automated image update 2026-08-27 09:51:08 +00:00
flux-bot
05f84f43ed chore(bstein-dev-home): automated image update 2026-08-27 09:48:09 +00:00
flux-bot
c038a10857 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 48, skipped: 81, passed: 3766
2026-08-27 01:52:12 +00:00
flux-bot
066b125c94 chore(maintenance): automated image update 2026-08-27 01:51:12 +00:00
flux-bot
da56cfbf71 chore(maintenance): automated image update 2026-08-27 01:47:11 +00:00
flux-bot
3b4ceddd8f chore(maintenance): automated image update 2026-08-27 01:42:10 +00:00
flux-bot
5a0e2089b6 chore(bstein-dev-home): automated image update 2026-08-26 21:50:41 +00:00
flux-bot
1a65c315fc chore(bstein-dev-home): automated image update 2026-08-26 21:47:42 +00:00
flux-bot
7477c304e2 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 48, skipped: 81, passed: 3766
2026-08-26 13:34:32 +00:00
flux-bot
88ad08f668 chore(maintenance): automated image update 2026-08-26 13:34:24 +00:00
flux-bot
9afe4eb6c6 chore(maintenance): automated image update 2026-08-26 13:32:28 +00:00
flux-bot
156b0440bf chore(maintenance): automated image update 2026-08-26 13:31:23 +00:00
flux-bot
371bd8087a chore(bstein-dev-home): automated image update 2026-08-26 09:51:49 +00:00
flux-bot
c61be05153 chore(bstein-dev-home): automated image update 2026-08-26 09:49:51 +00:00
flux-bot
2f07e89151 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 48, skipped: 81, passed: 3766
2026-08-26 01:35:04 +00:00
flux-bot
3c6b2bd381 chore(maintenance): automated image update 2026-08-26 01:34:56 +00:00
flux-bot
54e7caf4f0 chore(maintenance): automated image update 2026-08-26 01:32:58 +00:00
flux-bot
3ac9b9a570 chore(maintenance): automated image update 2026-08-26 01:31:55 +00:00
jenkins
391a7f2f1f hermes(agent): make titan-22 the strong primary home
Now that both the agent image (a68d1c4d, via the kustomize images: override)
and the hux sidecar (build-39) are multi-arch with amd64 leaves, move the worker
onto the amd64 accelerator titan-22:
- Add an OR'd nodeSelectorTerm for amd64 + node-role.kubernetes.io/accelerator +
  hostname titan-22, with NO worker=true requirement. Keep the arm64 pi-fleet
  term as an OR'd fallback so the worker is never stranded.
- Strong primary preference: hostname=titan-22 at weight 100 (scheduler max),
  pi-fleet rpi5 nudge lowered to 50, so hermes actually lives on titan-22.
- Tolerate node-role.kubernetes.io/accelerator=true:NoSchedule (harmless where
  absent) and the soft atlas.bstein.dev/media-primary:PreferNoSchedule that
  titan-22 currently carries, so the weight-100 preference is not offset and
  placement is deterministic.

Completes the titan-22 effort the flip branches missed; the earlier branches
never repointed to multi-arch images, which is why the worker never landed here.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 20:17:28 -03:00
jenkins
975399cbe0 test(monitoring): guard Claude renewal delivery path 2026-08-25 20:06:09 -03:00
flux-bot
6c35930f0b chore(hermes): promote validated image release 2026-08-25 23:05:30 +00:00
jenkins
0ad2ffefce monitoring: prefer worker nodes for Alertmanager 2026-08-25 20:00:05 -03:00
jenkins
1f636489f7 monitoring: keep Alertmanager on available rpi5 workers 2026-08-25 19:52:55 -03:00
jenkins
8d3a1193b6 monitoring: move Alertmanager recovery to available worker 2026-08-25 19:38:48 -03:00
jenkins
2c91aea01d fix(hermes-webui): verify OCI revision label on multi-arch index children
The webui release handoff verified org.opencontainers.image.revision on the
Harbor artifact's own extra_attrs.config.Labels. That works for a single-arch
image, but a multi-arch manifest list has no top-level config, so Harbor reports
the label on each per-arch child. build-38 built + published the index fine, then
failed post-publish with 'Harbor artifact omitted OCI image labels'.

verify_registry_digest now checks the top-level config labels when present
(single-arch, unchanged) and otherwise walks the index references, fetching each
child artifact by digest and asserting its revision label. Mirrors how the agent
image lane already tolerates a multi-arch index, without dropping the supply-chain
label check. Adds multi-arch pass/reject tests.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 19:37:27 -03:00
jenkins
35f650ef41 monitoring: recover Alertmanager placement drift 2026-08-25 19:34:42 -03:00
jenkins
db446c9244 monitoring(ai): alert before Claude quota auth expires 2026-08-25 19:16:01 -03:00
flux-bot
2a11c8d207 chore(bstein-dev-home): automated image update 2026-08-25 22:07:28 +00:00
flux-bot
c30a7c2d2b chore(bstein-dev-home): automated image update 2026-08-25 22:06:28 +00:00
jenkins
4908e1ea37 test(hermes): track agent base repoint to in-cluster multi-arch mirror
test_gateway_image_honors_ui_model_and_caps_reasoning still pinned the old
arm64-only docker.io base (nousresearch/hermes-agent@sha256:47d4bd4c...). The
agent image moved to the multi-arch mirror base
(harbor-core.harbor.svc.cluster.local/mirror/hermes-agent@sha256:9c841866...)
in 8a710845 for the two-leg build; the test wasn't updated, so it was a latent
red only the webui/quality lane runs. Point the assertion at the current base.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 19:04:58 -03:00
jenkins
342677dde7 monitoring(ai): preserve quota data across rollouts 2026-08-25 18:33:18 -03:00
jenkins
6f783b7778 build(hermes-webui): multi-arch image (arm64 + amd64)
Repoints both Dockerfile.hermes-webui FROM bases to in-cluster Harbor mirrors
(webui base OCI index + the a68d1c4d multi-arch hermes-agent manifest list),
adds the suspended webui base-mirror Job, and gives the WebUI image build an
amd64 leg on titan-24 plus a manifest-list combine — so the hux sidecar can
schedule onto the amd64 accelerator node titan-22.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 18:31:00 -03:00
jenkins
e7d6759140 fix(hermes-agent): reinstall codex when this arch's native dep is missing
The arch-specific CLI stamp assumed the shared node_modules keeps both arches'
codex native deps, but a sibling-arch npm install removes this arch's binary from
the shared volume. So after running on the other arch, the stamp exists yet the
native codex dep is gone -> configure-agent-clients fails -> churn. Gate the
install on the current arch's native codex package being present, so it self-heals.
2026-08-25 18:23:13 -03:00
jenkins
9c516b9808 build(hermes-webui): multi-arch image (arm64 + amd64)
Make registry.bstein.dev/bstein/hermes-webui a linux/amd64 + linux/arm64
manifest list so the agent pod's `hux` sidecar (which runs the webui image)
can schedule onto the amd64 node titan-22. Reuses the hermes-agent multi-arch
pattern already on main.

- Dockerfile.hermes-webui: repoint both FROMs to multi-arch, internal sources.
  The upstream WebUI base (ghcr sha256:a83a3893..., already a multi-arch OCI
  index) is now pulled from the in-cluster Harbor mirror; the agent base moves
  from the retired arm64-only leaf (81970563) to the multi-arch agent index
  (a68d1c4d). Kaniko selects the matching arch leaf per build node.
- services/harbor/hermes-webui-base-mirror-job.yaml: new suspended, operator-run
  skopeo `copy --all` Job mirroring the upstream WebUI base index into Harbor's
  `mirror` project (modeled on hermes-agent-base-mirror-job.yaml; reuses the
  generic ensure-project helper). Wired into the harbor kustomization.
- Jenkinsfile.hermes-webui-image: arm64 leg (titan-20) + amd64 leg (titan-24,
  hostname+arch pin, toleration Exists, resource-capped, own checkout scm) +
  Combine multi-arch index stage; per-arch evidence archived alongside the index.
- hermes_multiarch_combine.py: generalize the destination pattern/component to
  serve both hermes-agent and hermes-webui (fail-closed to just those two).
- Tests updated to the two-arch topology (two legs, combine, both FROM bases,
  the mirror Job, twelve archived evidence files).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 18:17:35 -03:00
jenkins
70002aeff7 Revert "Reapply "hermes(agent): make titan-22 the strong primary home (no worker label)""
This reverts commit 9e4fbf4e0df364230588749f80710a8369a3108f.
2026-08-25 17:49:44 -03:00
jenkins
a01eac5059 Revert "Reapply "hermes(agent): also tolerate titan-22's media-primary taint (flip applied pre-classification)""
This reverts commit 2e38422478b0d569441176431f104faae0dd1703.
2026-08-25 17:49:44 -03:00
jenkins
2e38422478 Reapply "hermes(agent): also tolerate titan-22's media-primary taint (flip applied pre-classification)"
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3753
This reverts commit 1c495a6e2c0a525b9e16608b16bc09c645e3c337.
2026-08-25 17:40:36 -03:00
jenkins
9e4fbf4e0d Reapply "hermes(agent): make titan-22 the strong primary home (no worker label)"
This reverts commit f8628e6ee0d2c33d287ec9086ff5328d982e88c3.
2026-08-25 17:40:36 -03:00
jenkins
fe40f68d5e merge: arch-aware agent runtime tooling install (feature/agent-tooling-multiarch) 2026-08-25 17:39:28 -03:00
jenkins
12a6d2c4f5 hermes(agent): make runtime tooling install architecture-aware
The hermes-agent installs its CLI toolchain at runtime into the shared
/opt/data/tools Longhorn volume, but every download hardcoded arm64. On
the amd64 node titan-22 that left configure-agent-clients failing with
"Missing optional dependency @openai/codex-linux-x64" and the operator
toolchain fetching arm64 binaries, so the pod churned.

Detect the running node's arch (uname -m; fail closed on anything but
aarch64/x86_64) and resolve every asset per-arch:

- install-agent-tools init script (agent-deployment.yaml): ttyd and
  kubectl download the arch-correct asset with the arch-correct sha256
  (real ttyd 1.7.7 x86_64 and kubectl v1.33.3 amd64 checksums added; the
  arm64 ones kept). The npm CLI stamp is now arch-specific
  (.cli-versions-<vers>-${arch}) so a fresh arch re-runs npm install and
  pulls its own native optional deps; npm keeps both arches' packages.

- install_agent_tools.sh: flux/helm/kustomize/jq/yq/gh/vault/sops/age/
  k9s/terraform/go URLs, tarball subdirs (helm linux-${arch}, gh dir),
  and checksums are all arch-resolved with both arches pinned. Stamps
  and the Go tree are arch-specific, and an active-arch marker forces a
  republish of the single-arch ${bin} binaries when the pod moves
  between arches on the shared volume. Single fetch/verify helper kept.

Tests updated to assert the arch-aware form (both arches' Go checksums,
${dl_arch} templating) instead of the arm64-only literal.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 17:37:57 -03:00
jenkins
f8628e6ee0 Revert "hermes(agent): make titan-22 the strong primary home (no worker label)"
This reverts commit 2de52ec3e2e42696ae411482cb16f27ff1f5d273.
2026-08-25 16:54:04 -03:00
jenkins
1c495a6e2c Revert "hermes(agent): also tolerate titan-22's media-primary taint (flip applied pre-classification)"
This reverts commit aa5898f9e33b371451c94af803cda1e0ed714591.
2026-08-25 16:54:04 -03:00
jenkins
aa5898f9e3 hermes(agent): also tolerate titan-22's media-primary taint (flip applied pre-classification)
Applying the titan-22 flip ahead of the full accelerator classification, so
titan-22 still carries its soft media-primary taint. Tolerate it too so the
strong titan-22 preference isn't penalised. Harmless once media-primary is gone.
2026-08-25 16:43:32 -03:00
jenkins
5631366c4e hermes(agent): make titan-22 the strong primary home (no worker label)
APPLY ONLY AFTER the multi-arch hermes-agent image is built + validated
(both arch leaves + promoted index). Supersedes the earlier titan-22
flip on feature/hermes-agent-multiarch (dfa50b75), which required
worker=true and tolerated the old media-primary taint.

Rewrites the runtime node affinity so hermes-agent runs on titan-22:
- Adds a second, OR'd nodeSelectorTerm matching amd64 + hostname
  titan-22 + node-role.kubernetes.io/accelerator=true. It does NOT
  require node-role.kubernetes.io/worker (titan-22 is no longer a
  generic worker).
- Keeps the arm64 pi-fleet term untouched as an OR'd fallback so the
  worker is never stranded if titan-22 is unavailable.
- Makes titan-22 the STRONG/primary preference: a hostname=titan-22
  preference at weight 100 (the scheduler maximum) outranks the pi-fleet
  rpi5 nudge, lowered to weight 50, so hermes actually lives on titan-22.
- Tolerates node-role.kubernetes.io/accelerator=true:NoSchedule so it
  can consider titan-22; this does not change jellyfin's media-core
  priority or preemption.

Updates test_hermes_agent_layout.py to the two-term topology, the
[100, 50] preference weights, the titan-22 primary preference, the
absence of a worker requirement on the titan-22 term, and the toleration.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 16:43:32 -03:00
flux-bot
aae22ea798 chore(hermes): promote validated image release 2026-08-25 19:28:55 +00:00
jenkins
a7fc9b20a0 fix(harbor): mirror to external Harbor endpoint (valid TLS)
harbor-core internally advertises an HTTPS token realm, so skopeo could not push
over HTTP. Push to registry.bstein.dev (valid cert, the path kaniko already
uses); the image lands in the same 'mirror' project and stays internally pullable.
2026-08-25 14:23:51 -03:00
jenkins
e7ac6351a6 fix(harbor): mark harbor-core insecure so skopeo pushes over HTTP
skopeo derived the token realm as HTTPS and got 'HTTP response to HTTPS client'.
A registries.conf with insecure=true for harbor-core:80 makes the registry AND
its token request use HTTP.
2026-08-25 14:21:15 -03:00
jenkins
4f5fc44013 fix(harbor): push the mirror over Harbor's HTTP port 80
harbor-core serves http on :80 only; the skopeo dest omitted the port so it
dialed :443 and timed out. Pin the dest to :80 (with --dest-tls-verify=false).
2026-08-25 14:18:35 -03:00
jenkins
6a25a7681a fix(harbor): run Vault init first in the base-image mirror Job
The Job's ensure-project init container reads /vault/secrets/harbor-admin-password,
but Vault appended its init container AFTER ensure-project, so the secret file
was absent and the init failed. Force vault-agent-init to run first.
2026-08-25 14:10:57 -03:00
jenkins
8a71084585 build(hermes-agent): source base image + test deps from in-cluster mirrors
The hermes-agent-image pipeline failed intermittently on external network:
Kaniko's docker.io fallback for the base image is IPv6-broken from build
pods, and the "Validate reviewed release source" stage pip-installed pytest
from files.pythonhosted.org (DNS failures). Neither should touch the public
internet.

Base image: repoint the Dockerfile FROM from docker.io to the in-cluster
Harbor "mirror" project, keeping the exact content-addressed index digest
(9c841866...) and both arch leaves. A Flux-managed one-shot Job
(services/harbor/hermes-agent-base-mirror-job.yaml, suspend: true like the
cassandra bootstrap job) runs `skopeo copy --all` from docker.io into Harbor
using the same Vault-injected admin credential as the existing Harbor
immutability jobs; a tiny fail-closed helper ensures the public target
project first. Digest pinning and multi-arch are preserved; Kaniko pulls it
over the internal insecure registry with no docker.io fallback.

Test deps: install pytest/PyYAML fully offline (`pip --no-index
--find-links`) from a reviewed in-repo wheelhouse
(ci/vendor/hermes-agent-test-wheels) matching the arm64 python:3.12 build
container, so the validate stage never resolves a public index.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 13:53:22 -03:00
jenkins
2675241739 fix(hermes-agent): checkout scm in the amd64 build leg (fix exit 128)
The amd64 leg runs on its own fresh titan-24 pod but never checked out the SCM,
so its independent reviewed-revision boundary check hit 'git rev-parse
origin/main -> fatal: not a git repository' and the build failed with exit 128
(the arm64 leg built and pushed fine). Add checkout scm to the amd64 stage.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 13:03:13 -03:00
jenkins
2bfdee6169 fix(hermes-agent): do NOT make titan-24 a general worker for the amd64 build
titan-24 is an accelerator node (co-hosts the out-of-cluster Sui validator), not
a general worker. The amd64 build leg was requiring node-role worker=true, which
forced labeling titan-24 as a worker and opened it to unrelated cluster
scheduling. It already pins by hostname+arch, so drop the worker requirement and
remove the titan-24 worker-join from the node-prefer CronJob entirely. The build
targets titan-24 specifically (hostname) and tolerates its taint; nothing else
in the cluster gets scheduled there.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 11:50:52 -03:00
jenkins
2b7d140d9e docs(hermes-agent): multi-arch rollout runbook
Operator steps in order: prepare/uncordon titan-24, merge, run one validation
build (both arch leaves + promoted index), apply the titan-22 affinity flip,
verify placement. Documents the correction that worker membership is reconciled
by the Flux node-prefer-noschedule CronJob, not Ansible, and lists what was
validated locally vs what only a real Jenkins build can prove.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 11:07:16 -03:00
jenkins
8ddff9b626 infra(core): join titan-24 as an amd64 worker for the image build leg
The native amd64 hermes-agent image leg builds on titan-24. Worker membership
in this cluster is reconciled by the node-prefer-noschedule CronJob (kubectl
label), not Ansible, so add titan-24 there:

- clear_worker titan-24 amd64  -> node-role.kubernetes.io/worker=true + hardware=amd64
- a soft PreferNoSchedule guard taint (atlas.bstein.dev/sui-validator=true)
  mirroring titan-22's media guard, so routine pods do not crowd the
  out-of-cluster Sui validator that co-hosts titan-24. GPU workloads pinned to
  titan-24 by hostname are unaffected (PreferNoSchedule never blocks a pinned
  pod), and the amd64 build pod tolerates this taint explicitly.

Operator note: this reconciler does not manage cordons (owned by Ananke
recovery). titan-24 is on the recovery uncordon denylist, so the operator must
ensure titan-24 is uncordoned/schedulable before the first amd64 build.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 11:05:42 -03:00
jenkins
ff2c003b4b build(hermes-agent): multi-arch image via two native kaniko legs
Repoint the hermes-agent base FROM at the upstream multi-arch OCI INDEX
digest (tag v2026.7.7.2, revision 9de9c25f) whose arm64 leaf is byte-for-byte
the previously pinned single-arch base, so the arm64 build is unchanged while
the same reviewed version now also resolves an amd64 leaf. Kaniko selects the
matching leaf per build platform.

Rework the release pipeline to build both arches natively and promote a
multi-arch image without switching off kaniko or weakening any existing
security assertion:

- Keep the arm64 kaniko leg on the unchanged rpi5 coordinating pod; it now
  pushes an arch-suffixed candidate tag (...-build-<N>-arm64).
- Add a second native amd64 kaniko leg on a titan-24-pinned, tolerating,
  resource-capped pod (ceiling strictly below the arm64 leg) that
  independently re-verifies the reviewed revision and stashes its leaf
  evidence (...-build-<N>-amd64).
- Add ci/scripts/hermes_multiarch_combine.py: a pure-python, fail-closed
  combiner that re-reads each per-arch leaf from the registry, proves its
  digest AND its config architecture, assembles a Docker manifest LIST
  (already inside the promote allow-list), refuses to overwrite an existing
  final tag, publishes the arch-less ...-build-<N> tag, and re-verifies the
  registry resolved the exact index referencing exactly the two leaves. It
  emits the index digest in the SAME digest-file/image-file format the
  single-arch step produced, so render/verify-evidence/hermes_oci_promote.py
  promote the INDEX with no change to those scripts.

Tests: add test_hermes_multiarch_combine.py (full hash/verification chain);
strengthen the image-builder suites for the two-arch topology (both kaniko
legs carry the reviewed heredoc-compat build-arg; amd64 leg pinned+capped+
boundary-checked; combine stage wiring; expanded evidence archive) without
weakening the arm64-leg assertions.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 11:05:29 -03:00
jenkins
500741020a fix(hermes): revert hard rpi5 requirement (it stranded worker)
Requiring rpi5 while the historical hostname exclusion list removes the very rpi5
nodes that currently have headroom (titan-04/06) left only loaded rpi5s
(titan-05/07/11), so the agent could not schedule and worker went down. Revert to
rpi5-PREFERRED (soft) so it schedules again; proper rpi5 placement needs the
exclusion list refreshed against current node health/capacity, tracked separately.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 10:25:54 -03:00
jenkins
c44f16e4ac fix(hermes): require rpi5 for worker (keep the heavy agent off rpi4)
Un-pinning let the scheduler land worker on titan-12 (rpi4). The agent is heavy
enough that an rpi4 risks the /api/status slowness that trips its liveness probe
- the exact flap we are avoiding. Make hardware=rpi5 a hard requirement so it
runs only on rpi5 storage workers (excluding the saturated/known-flaky ones);
the scheduler places it on a roomy rpi5 (titan-05).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 10:10:24 -03:00
jenkins
e54d581ef5 fix(hermes): un-pin worker from titan-08; spread across arm64 storage workers
Worker (hermes-agent) was hard-pinned to titan-08 (a workaround after an earlier
attempt to place it on the amd64 titan-22 failed on architecture). That single-
node pin is exactly what makes it fragile: a titan-08 blip (as just happened when
the node's Longhorn CSI went down) strands worker, and the Recreate strategy then
deadlocks because the replacement can't schedule on the one tight node.

Restore the intended multi-node design: run on any arm64 storage worker except the
known-bad/weak ones (matching the repo's own affinity test, which was red). Its
Longhorn volumes have data-locality disabled with replicas on titan-15/17/19, so
there is no locality penalty to running on another node; the scheduler now places
it on a roomier Pi (e.g. titan-05) and a node blip simply reschedules it.

Also relax the gateway /api/status liveness probe (timeout 10s->15s,
failureThreshold 3->5) so a transient slowness (e.g. a brief storage hiccup) no
longer trips a kill-and-restart cascade.

Follow-up (separate): multi-arch agent image to enable the amd64 titan-22 target.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 09:30:06 -03:00
flux-bot
8d781ea808 chore(bstein-dev-home): automated image update 2026-08-25 09:48:59 +00:00
flux-bot
ced73bf30d chore(bstein-dev-home): automated image update 2026-08-25 09:48:08 +00:00
flux-bot
5888850eba chore(hermes): promote validated image release
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 81, passed: 3727
2026-08-25 04:43:23 +00:00
jenkins
5b1f832072 hermes(voice): keep one consistent ellipsis on conversation status labels
The overlay status caption already carries a static ellipsis (e.g. 'Thinking…'),
and after ~2.5s of silent thinking the 'working' affordance added an animated
dots ::after on top of it — so the label intermittently rendered as 'Thinking……'
only on longer turns. Drop the animated text dots and keep the orb-halo shimmer
as the liveness cue, so the ellipsis stays a single, consistent static one.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 01:18:24 -03:00
flux-bot
9f49a1ba2c chore(hermes): promote validated image release 2026-08-25 03:37:13 +00:00
jenkins
a071d091b0 hermes(voice): add a centered Start conversation button to the new-chat screen
Conversation mode was only reachable via a small icon by the composer, which is
easy to miss on a fresh session. Inject a prominent, centred 'Start
conversation' button into the empty new-chat state (#emptyState), below the
subtitle and above the suggestions, so it sits in the vertical centre of the
screen. It is created only when local voice is available, honours the same
show/hide preference as the composer toggle, and enters conversation mode
through the same activate() path. Fully guarded so it degrades to nothing if the
empty state is absent. New probe scenario verifies it is created, visible, and
opens the conversation overlay on press.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-25 00:13:11 -03:00
flux-bot
a59590be79 chore(hermes): promote validated image release 2026-08-25 02:56:09 +00:00
jenkins
ffa477585d hermes(webui): fix mobile composer hidden behind the gesture bar
On an Android standalone PWA the composer's bottom control row was rendering
about one line below the visible viewport, behind the system gesture bar, so
those controls were unreachable. The theme already pads the titlebar with
env(safe-area-inset-top/left/right), but the viewport meta never opted into
viewport-fit=cover, so every safe-area inset collapsed to 0 and the bottom edge
had no reservation.

Add viewport-fit=cover to the viewport meta (base patch) so the insets carry
real values, and reserve env(safe-area-inset-bottom) at the bottom of the
composer (brand.css), additive with the app's existing --keyboard-bottom-inset
and absorbed by the flex-1 message scroller so total height stays within the
viewport. Inert on desktop (env() resolves to 0). Dockerfile verifies the meta
patch landed; brand/dockerfile contracts covered by tests.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 23:32:27 -03:00
flux-bot
b59f754f5a chore(hermes): promote validated image release 2026-08-25 01:53:50 +00:00
flux-bot
29ba03105e chore(maintenance): automated image update 2026-08-25 01:50:53 +00:00
flux-bot
7826529273 chore(maintenance): automated image update 2026-08-25 01:50:07 +00:00
flux-bot
774d74c458 chore(maintenance): automated image update 2026-08-25 01:46:53 +00:00
flux-bot
770b812ae5 chore(maintenance): automated image update 2026-08-25 01:41:05 +00:00
flux-bot
63bd63c283 chore(hermes): promote validated image release 2026-08-25 01:35:01 +00:00
jenkins
fd0a4b23f9 hermes(voice): auto-retry transient provider errors; prime STT acronyms
Diagnosed from a live voice session (e1b9fccb90ef): three turns failed with
raw '**Error:** HTTP 502 ... hermes-{claude,codex}-broker' because the agent
pod hosting the model brokers rolled mid-conversation. The voice client
correctly refused to speak the error envelope, but it then dropped the user's
utterance and forced them to repeat it three times.

Voice: on a TRANSIENT provider error (5xx/502/'error sending request'/timeout),
conversation mode now re-runs the errored turn in place through the app's own
regenerate action (which truncates the errored turn — no duplicate user
message) and stays in Thinking so its cues cover the reconnect gap. Bounded to
MAX_TRANSIENT_RETRIES (2); a non-transient error or an exhausted budget still
drops cleanly to 'let's try that again — listening'. The raw error is never
spoken. New probe scenarios cover retry-then-recover and the bounded-then-drop
path; source contract updated.

STT: the same session mis-transcribed 'CUI' as 'cue'. Prime the default
initial_prompt with the domain acronyms the user uses (CUI, FOUO, DoD, NIST,
CMMC, FIPS, RMF, POA&M, ATO, SBU) so they bias to uppercase forms.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 22:10:21 -03:00
flux-bot
700301d554 chore(hermes): promote validated image release 2026-08-25 00:35:53 +00:00
flux-bot
77380bfc25 chore(hermes): promote validated image release 2026-08-25 00:11:52 +00:00
flux-bot
f96070f6d9 chore(hermes): promote validated image release 2026-08-24 23:48:49 +00:00
jenkins
12ec6e5a70 hermes(voice): crop Hermes character into orb as a feathered watermark
The orb watermark was showing the character art as a low-opacity square
box (visible edges) — the earlier screen-blend pass washed it into a glow
that lost the artwork. Reprocess the source into a feathered circle so the
orb crops it (no box) while keeping the character's facial features, paint
it with a normal blend at 0.5 opacity, and size it to fill the orb (inset
7%). Verified by compositing over a simulated orb before shipping.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 20:36:51 -03:00
jenkins
1baab8e014 hermes(voice): restore always-opening overlay, character watermark, speaker default
- REGRESSION FIX: the full-screen conversation overlay stopped opening
  (fell back to the inline bar) because building the language/output
  selectors inside the overlay's single try/catch could throw on a
  phone (navigator.mediaDevices/setSinkId). Overlay build is now two
  phases: the essential orb+captions+controls attach first; the
  selectors attach after, each guarded, so a selector failure omits only
  that control and never the visualization. Control builders can no
  longer throw (inert hidden fallback); device enumeration is async
  after attach. Regression test covers mediaDevices-undefined and
  enumerate-rejects.
- Orb watermark is the processed Hermes character glyph (feathered,
  circle-cropped, screen-blended so the face glows on the dark orb), no
  more white/grey box.
- Output selector defaults to the loudspeaker (excludes the
  communications/earpiece endpoint that the open mic otherwise forces);
  explicit choice sticks; hidden gracefully where setSinkId is
  unsupported. 291 voice tests.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 20:24:31 -03:00
jenkins
b99952f16c hermes(stt): greedy single-temperature final decode for fast startup
Beam-2 with a temperature-fallback ladder made the final-model warmup
run all three temperature retries under beam search before the server
bound its port, so /health was refused for ~8 min and STT was down that
whole time on every roll (and hinted at slow per-utterance decodes).
Production now runs the accurate large-v3-turbo model greedily at a
single temperature, keeping the proper-noun priming prompt that fixes
names like Amy/Córdoba - fast startup, fast decodes, accuracy intact.
Beam stays env-tunable (HERMES_STT_FINAL_BEAM_SIZE) for a future pass;
serve-before-warmup is a recommended follow-up so cold start never
blocks readiness.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 20:20:28 -03:00
jenkins
546e960185 hermes(stt-client): retry a brief STT outage, never dump a traceback
A transcription that landed while the private Whisper service was
restarting (an image roll) crashed hermes_stt_client.py with a raw
urllib ConnectionRefused traceback that got dumped into the
conversation. The client now retries the request with backoff (up to 5
attempts, ~10s - long enough to ride an STT pod restart) and, on a
persistent outage, exits with one concise line ('speech transcription
unavailable...') instead of a stack trace. Delivered via the coordinator
ConfigMap; the next reconcile picks it up.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 20:16:57 -03:00
flux-bot
c54eeada1b chore(hermes): promote validated image release 2026-08-24 23:08:35 +00:00
jenkins
0722ab3f2c docs(hux): record final build-29 live evidence and STT/infra follow-ups
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 19:58:49 -03:00
flux-bot
88f8e9e629 chore(hermes): promote validated image release 2026-08-24 22:47:03 +00:00
jenkins
4d93ef5a0e hermes(voice): workspace nav home, character orb, conversation rename, voice-lang fix
Final conversation-mode polish from mobile testing:
- The Workspace toggle now sits with the chat/Telegram nav at every
  width: nav.rail on desktop, the top app titlebar on mobile. The
  floating pill that pushed the mobile composer's control row (and the
  conversation-mode button) off screen is gone - a fallback exists only
  for headless DOMs and is pinned to a top corner, never over the
  composer.
- The conversation orb watermark is now the Hermes character avatar
  (static/hermes-agent-192.png) instead of the caduceus staff.
- User-facing 'hands-free' copy renamed to 'Conversation mode'.
- Wrong-voice fix: strongReplyLanguage flagged Spanish on a single
  accented char, so an English reply naming European cities (Zürich,
  Málaga) overrode the correct English STT detection and was spoken by
  the Spanish voice. Detection now requires density (Cyrillic >=4 at
  >=50%, or inverted punctuation / >=2 accents corroborated by Spanish
  stopwords); plain English always speaks English, forced language wins,
  accent-free Spanish still routes via trusted STT. 294 voice tests.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 19:22:32 -03:00
flux-bot
4caf8942fd chore(bstein-dev-home): automated image update 2026-08-24 21:49:01 +00:00
flux-bot
a825d70d15 chore(bstein-dev-home): automated image update 2026-08-24 21:46:57 +00:00
jenkins
181a7517c2 hermes(voice): interim-ack tail, natural fillers, unified audio + output picker
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 72, passed: 3724
Three conversation-mode fixes in one pass:
- Interim-acknowledgement truncation: when an interim ack folds into the
  hidden worklog segment mid-speech, the retained unspoken tail is now
  flushed and spoken in full before the Thinking transition, and a
  distinct follow-up message is chunked from its own start and queued
  after the interim drains (no more 'stops after the first clause, rest
  resurfaces with the next message').
- Natural thinking fillers: brief per-language interjections (Umm/Hmm/
  One sec; Mmm/A ver; Хм/Секунду) on genuine >1.9s thinking gaps only,
  non-repeating, answer-preempting, mute-aware.
- One unified audio sink for every spoken output (reply, cues, fillers,
  WAV fallback) - fixes cues playing the loudspeaker while the reply
  used a different output - plus a tidy corner output-device selector
  (enumerateDevices + setSinkId, feature-detected, session-only) styled
  like the language selector. Also realigns two STT-server decode-param
  assertions to the dict form from the STT tuning commit. 284 voice tests.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 17:32:13 -03:00
jenkins
64b7bc55cf hermes(stt): tune final decode for fast-and-accurate (beam 2, name priming)
Default final beam 5 -> 2: the accuracy/speed knee - most of beam
search's benefit at ~2x greedy instead of ~5x, protecting commit
latency on the Jetson (still env-overridable via
HERMES_STT_FINAL_BEAM_SIZE). Prompt now also primes common names (Amy,
Claude, Hermes) so 'Amy' stops transcribing as 'aiming'.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 16:57:50 -03:00
jenkins
1fdfaff096 hermes(stt): accurate large-v3-turbo final decode, fast tiny partials
Proper nouns (Córdoba, Cancún) and dropped words came from decoding the
committed transcript with the small model. The image already ships
large-v3-turbo, so the final decode now uses it with beam_size=5, a
temperature fallback ladder, and a proper-noun/accents initial_prompt
that fixes first-pass capitalization and diacritics across EN/ES/RU;
the rolling previews stay on tiny at greedy so the on-the-fly feel is
unchanged. The accurate decode runs in the speculative predecode during
the end-of-speech silence and is cache-reused at commit, so perceived
latency stays low. All decode knobs are env-overridable for on-device
tuning (beam/temperature/prompt), with small as the guaranteed-present
rollback if turbo underperforms on the Jetson. 206 STT tests pass.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 16:39:00 -03:00
jenkins
f7b3506a88 hermes(chat): parse large cluster reads before capping the output
The size cap was applied to the raw wire body, so a nodes list (huge
because of status.images) truncated before the image-stripping ran and
came back as a truncation notice. Accept up to 6 MiB on the wire to
parse and clean, then enforce the 384 KiB model-facing cap on the
stripped result - nodes now returns real data. Delivered via the
cluster-read ConfigMap; picked up on the next pod roll.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 16:25:21 -03:00
jenkins
b40300efaf hermes(voice): a stray error segment must not discard a real reply
The 'Something went wrong - listening' state with no spoken answer was a
false positive: readAssistantTurn flagged the whole turn as an error if
ANY segment was error-stamped - including a recovered/transient tool
error or a cancellation notice from an earlier interim - and threw away
the real answer that the same turn produced. Error now surfaces only
when the turn yielded no spoken answer at all; a turn with real content
is spoken normally. Softened the genuine-error label to the friendlier
'Let's try that again - listening'. New probe scenarios lock both: an
error segment alongside an answer speaks the answer (error=false), and
an error-only turn still reports the error.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 16:18:48 -03:00
flux-bot
4de37f4f3a chore(hermes): promote validated image release 2026-08-24 19:08:58 +00:00
jenkins
3fd760de0e hermes(voice): fix HHermesProcessed + first-sentence-stop; orb mark, lang selector
Real root cause (confirmed against the live build-24 DOM): the caption
and TTS extraction fell back to turn.textContent whenever a settle-frame
race left no readable answer segment, scraping the avatar letter,
author name and 'Processed 13s' chip - and that truncated reply made
TTS speak only the first segment then drop to Listening even with the
mic muted (the muted-mic first-sentence-stop). Extraction now prefers
each answer segment's data-raw-text, else the answer .msg-body only
(excluding thinking/tool/worklog/role chrome), and the textContent
fallback is gone; a genuinely mid-flight reply retries briefly so the
whole thing is read before the overlay drains. Also: the app's own
caduceus mark embedded in the conversation orb as a subtle watermark; a
corner language selector (Auto + en/es/ru) that forces both the STT
hint and the reply voice; and a thinking affordance after 2.5s of dead
time. New response probe + extraction test prove caption==body-only and
that every sentence reaches the TTS queue. 278 voice-lane tests green.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 15:46:09 -03:00
jenkins
13a058a5c8 hermes(chat): teach the assistant cluster_read and voice acknowledgements
The chat prompt still forbade all cluster access, so Hermes told users
it had no visibility even though cluster_read is live - it now knows it
has a read-only cluster tool (no Secrets/Vault) and should use it rather
than deny. Adds tool/research-conditioned acknowledgement guidance: when
a turn needs a lookup, plan or calculation, open with one short 'on it,
~ETA' line then deliver the full answer; simple questions get no
preamble. In spoken mode that first line is read aloud. Fixes two
pre-existing exact-match test pins the HUX/cluster rollout had grown
(plugins.enabled list; a fieldRef env comprehension).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 15:20:42 -03:00
jenkins
834fec17c5 docs(hux): record build-24 fleet-live, cluster read, worker staging
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 14:41:17 -03:00
jenkins
ed0bc30d7c hermes(chat): make cluster_read robust to large node listings
Node .status.images (every cached image on the node) overflowed the
size cap and left json.loads parsing a truncated blob, so a nodes query
came back as a non-JSON error. The de-noise pass now summarizes that
list, and genuine truncation returns the readable prefix with a
narrow-your-query hint instead of an error.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 14:40:09 -03:00
jenkins
4ccf2066ca hermes(chat): allow tenant egress to the apiserver for cluster_read
The read-only cluster_read tool (and the HUX-12 producer) reach the
Kubernetes API through the kubernetes Service, which kube-proxy DNATs
from the 10.43.0.1 ClusterIP to a control-plane node on 192.168.22.11-13
:6443 - addresses the tenant egress except-block was dropping, so calls
failed with connection-refused. Egress now allows the ClusterIP and
those three apiserver endpoints on 443/6443. RBAC still bounds what is
readable (no Secrets); nothing else about the isolation changes.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 14:38:16 -03:00
flux-bot
04f060db3e chore(hermes): promote validated image release 2026-08-24 17:22:30 +00:00
jenkins
c4eb872690 monitoring(ai): show Claude Fable weekly quota 2026-08-24 14:13:35 -03:00
jenkins
d041f1d1ee hermes(voice): round-3 fixes, language routing, session-toast root cause
- Captions read message bodies only (the scraper was concatenating
  avatar, author and worklog chips); multi-segment interim turns are
  now speakable and drive clean speak-to-thinking-to-speak cycles when
  playback drains mid-turn.
- Dynamic endpointing: complete-looking partials (3+ words or terminal
  punctuation) endpoint at the base window; the long hold remains only
  for one-two-word fragments. A stale-busy 10s settle wait on every
  post-error send is gone.
- Both overlay captions are bounded, touch-scrollable regions with
  follow-tail; caps raised for long turns.
- Error envelopes are never spoken or captioned; errored turns run
  resyncCapture (fresh STT session on the hot mic).
- Workspace toggle now lives in the sidebar rail (floating button only
  below the rail breakpoint).
- False 'session unavailable' toast root-caused: the router continuity
  guard shows it on a 409 that fired when a transient profile-listing
  failure failed closed into a fake cross-profile mismatch; the patcher
  now answers from the alias cache and never claims a default-vs-named
  mismatch while aliases are unconfirmed.
- Barge-in sends carry a one-line cut-point marker with the last spoken
  sentence; visible-history truncation judged infeasible client-side.
- Language switching works end-to-end: sticky per-session STT language
  hint (restarting an unused next session on switch), reply voice from
  script evidence, STT detection, then stopword heuristic; cues and WAV
  fallback share the turn language.
- Legacy CI guards: node skip for the DOM probe, ffmpeg/codec skips for
  the container-fallback test. 272 voice-lane tests green.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 13:58:35 -03:00
jenkins
c0a9c92ee4 hermes(worker): stage inert HUX foundation on the worker instance
worker.bstein.dev (the hermes-agent Deployment) gains the same HUX
shape as chat, staged and inert: a foundation-only hux sidecar on the
reviewed WebUI image line (Flux setters bound, 5s probe budgets), an
init that provisions the HMAC identity as slot-100 on the durable home
subtree (create-once context key, O_EXCL subject binding, per-pod
worker key; no relay/router/evidence keys so those trusts fail closed),
and observe-only hook env in the agent container with the runtime
plugin mounted but deliberately NOT enabled - activation is a reviewed
one-line flip per docs/hux/WORKER-PLAN.md, which carries the rollout,
verification gates, canary/rollback ladder and open questions.
Cross-surface continuity remains unclaimed until the live gates pass.
7 new topology-adaptive delivery gates green.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 13:54:27 -03:00
jenkins
3bca7b7465 hermes(hux): inline approval prompts and ask-not-deny defaults
Re-enforcement prerequisites, code-side complete: the autonomy runtime
now docks pending-approval cards above the composer (newest first, cap
three, aria-live, allow-once / always / deny wired to the existing
decide route with idempotency; polling gated to active turns and
fail-tolerant), so parked tool calls are never silent. The default
capability matrix no longer denies by default: network and web_search
ask below autonomous (visible prompt) and nothing resolves to deny
except explicit grants or private mode; SO-39 stays intact - deploy and
external side effects always ask and external never auto-allows.
rules.py at 100% line+branch; 744 hux-lane tests green.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 13:53:00 -03:00
jenkins
d1225bcab4 hermes(chat): read-only Atlas cluster visibility for chat
RBAC: the built-in view ClusterRole (which never includes Secrets, so
Vault-managed material stays structurally invisible) plus a read-only
extra for nodes, namespaces, PVs, storage classes, CRDs, Flux resources
and metrics, bound to the chat service account. Tooling: a cluster-read
plugin registers a GET-only cluster_read tool against the in-cluster
API using the pod's projected token - secrets paths refused in the
handler as well, malformed segments rejected, responses bounded and
stripped of managedFields noise. Classified read_files/low in the HUX
capability map. RBAC applies on push; the tool activates when the pods
next roll (bundled with the round-3 voice build).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 13:33:48 -03:00
jenkins
aa6465afa8 docs(hux): record round-2 fleet-live evidence
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 12:05:04 -03:00
flux-bot
3beb478258 chore(hermes): promote validated image release 2026-08-24 14:44:10 +00:00
jenkins
043aa9ee89 release(hermes): bind block-style HUX build metadata
The env setters moved to block style so Flux can rewrite them; the
renderer belt now matches the same shape and the release test asserts
the bound value lines rather than the old flow mapping.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 11:21:35 -03:00
jenkins
09ba5f8ac0 hermes(voice): fix first-word clipping, always-stitch, panel glow, conversation mode
Round 2 from live testing:
- Clipping root cause: the endpointer committed on any 1.1s pause, so a
  thinking pause after a sentence opener sent one word; a speculative
  Whisper pass over that fragment then stalled the real commit ~3.5s on
  the Jetson. Young utterances now hold a 1.8s endpoint until 1.2s of
  speech accrues, speculative decode waits for 700ms of speech, resume
  is unconditional after any 650ms gap (server-frozen snapshots can
  never reach commit), and the noise floor is capped so playback echo
  cannot deafen onset. Deterministic capture-continuity probe added.
- Stitching now fires for any barge-cancelled send within 20s,
  regardless of partial assistant output.
- The workspace-drawer dark rectangle was our own HUX chrome resolving
  undefined theme tokens (--bg-primary) to a flat box; bootstrap.css
  bridges the real theme tokens and drops a compositor-hazard
  backdrop-filter.
- Conversation mode: full-viewport hands-free overlay with an energy-
  driven orb (mic RMS in, TTS activity out), state caption, live
  transcript/reply captions, mute and exit controls, focus trap,
  Escape, reduced-motion support. Voice lane 253 tests green.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 11:21:35 -03:00
jenkins
eebc5e6de6 docs(hux): record enforcement rollback and re-enable checklist
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 10:56:26 -03:00
jenkins
1793903f5e hermes(chat): observe-only enforcement and honest tool capabilities
Real traffic showed the first enforcement pass blocking core assistant
faculties: skills listing and the sandboxed Python classified as
unknown external side effects, browsing denied by default, with no
approval surface in the chat flow. Enforcement returns to observe-only
fleet-wide while the approvals UX and default grants are reworked, and
the capability map now tells the truth about the real toolset: the
Python sandbox is internal shell work, skills/todo/clarify/vision are
reads, browsing is network (medium), image generation writes an
artifact through the trusted broker. Unknown tools remain fail-closed.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 10:46:54 -03:00
jenkins
4765f8da2a docs(hux): record fleet-live evidence for HUX-01..11
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 09:44:24 -03:00
jenkins
5fc3e87060 hermes(chat): widen HUX to the fleet and unstick the roll
Removes the canary partition so all four tenants receive the reviewed
build-22 image with the HUX topology and enforcement (every lifecycle
gate passed on ordinal 3). Bumps the hux probe timeouts to 5s: the 2s
readiness exec budget timed out 405 times in 135 minutes on the loaded
ARM node — interpreter spawn cost, not service health — and the
resulting flapping stalled the StatefulSet roll at the canary. Converts
the HUX_IMAGE_TAG/DIGEST setters to block style with the marker on the
value scalar (Flux cannot attach setters inside flow mappings, which
left the metadata stale at build-21) and corrects them to build-22.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 09:31:45 -03:00
flux-bot
1dd8660f27 chore(hermes): promote validated image release 2026-08-24 10:07:40 +00:00
jenkins
7d070c4e1c test(hermes): skip node-dependent HUX suites on node-less runners
The titan-iac CI pod has no node binary; the gate helper and every
direct node invocation in the new HUX suites now skip with an explicit
reason instead of erroring, restoring the main-CI baseline. Runners
with node keep full enforcement.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:32:46 -03:00
jenkins
852d37c087 docs(hux): record enforcement-live canary evidence
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:29:07 -03:00
jenkins
91cfb901a2 hermes(voice): continuous mic, barge stitching, 1.15x speech
Three conversational fixes for hands-free chat:
- The microphone now stays hot for the whole session: capture runs on
  its own epoch, re-arms immediately after each utterance endpoints,
  and keeps recording through transcribing/thinking/speaking - speech
  is never lost to Hermes being busy. Speech onset during a response
  cancels it through the live capture path (echo-guarded exactly like
  the old monitor) without touching the running recorder.
- When the user talks over Hermes before any visible reply appeared,
  the interrupted utterance and the follow-up are stitched into one
  message (20s window), so the response addresses the whole thought.
- TTS speaks 15% faster by default (server-side length_scale, no pitch
  shift), user-tunable via hermes-voice-tts-speed (0.5-2.0), honored on
  streaming, WAV fallback and thinking-cue paths.

245 voice-lane tests pass; single getUserMedia site preserved;
Dockerfile grep guards verified.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:27:38 -03:00
jenkins
17037773bf test(hermes): delivery gate accepts the enforced HUX posture
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:22:26 -03:00
jenkins
4a6d4fdf0d hermes(chat): enforce HUX tool policy on the canary
Park/resume and gateway-owned stop receipts passed the live gates, so
HUX_TOOL_ENFORCEMENT=1 ships to ordinal 3 (partition unchanged). Every
canary tool call now requires a HUX release; the fleet stays observe-off
until the approvals UX proves out under real traffic.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:21:06 -03:00
jenkins
f2f54033a2 docs(hux): record Wave-B live gate evidence
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:13:57 -03:00
jenkins
ac1e282972 hermes(chat): enable autonomy wave on the HUX canary
hux.autonomy joins the canary flags (which also unlocks multimodal via
its dependency chain). Enforcement remains 0: approvals, gates, budgets
and stop receipts run observe-only until park/resume passes live.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:13:57 -03:00
jenkins
dc2b364354 docs(hux): record Wave-A live gate evidence
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:07:43 -03:00
jenkins
48120103a3 hermes(chat): enable Wave B cards on the HUX canary
Artifacts, research, friendly modes, multimodal and onboarding join the
canary flag set on ordinal 3 (partition unchanged, enforcement still 0).
Autonomy and release follow-through remain off pending their gates.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:07:43 -03:00
jenkins
ee33082242 docs(hux): record live foundation canary evidence
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:02:35 -03:00
jenkins
c05ec2386f hermes(chat): enable Wave A cards on the HUX canary
Ordinal 3 only (partition unchanged): activity timeline, projects,
privacy and memory control join the foundation flag, observe-only.
The roll also serves as the HUX-11 restart-persistence gate via the
recorded identity hashes.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:02:35 -03:00
jenkins
39a544828f hermes(chat): activate HUX foundation canary on ordinal 3
Some checks failed
Tests / Declarative: Post Actions failed: 71, skipped: 28, passed: 3680
Re-applies the staged HUX topology pinned to the reviewed build-21
image (git-2f535d3a...-build-21-release@sha256:e5b9b2fa...), with the
first-activation posture: HUX_FLAGS=hux.foundation only,
HUX_TOOL_ENFORCEMENT=0, and a RollingUpdate partition of 3 so only
hermes-chat-tenant-3 rolls. Adds the /healthz auth bypass on the chat
proxy so HUX-12 health receipts can observe a real 200, points the
evidence policy at it, and makes the delivery flag gate progressive
(foundation first, cards enabled per lifecycle acceptance; unknown
flags still never ship).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 05:47:46 -03:00
flux-bot
1d29c25c9b chore(hermes): promote validated image release 2026-08-24 08:44:32 +00:00
jenkins
2f535d3a30 test(hermes): portable node coverage gate for the HUX suites
Build 20 failed on the CI image's Node 20: --test-coverage-lines and
friends need Node >= 22.8 and --experimental-strip-types needs 22.6.
A shared helper now runs plain --experimental-test-coverage and
enforces the same per-source >=95 floors by parsing the coverage
table, so the gate is identical on Node 20 and newer local Nodes; the
TypeScript suites skip with an explicit reason on runtimes that cannot
strip types. Per-file gating also exposed pre-existing debt the old
aggregate thresholds hid (wave_b_projects_modes.js branches 90 / funcs
94.7) - recorded as explicit enforced floors, not waived.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 05:22:01 -03:00
jenkins
8f2abac3dc hold(hermes): keep chat activation topology out of the source release
Restores the five services/hermes manifests to the live state and
removes the evidence policy/RBAC resources, so pushing this source
chain applies nothing to the cluster beyond the inert hux_mode.py key
in the auto-router ConfigMap (old pods are protected by the plugin's
degrade-to-noop import shim). The activation topology returns as a
dedicated commit pinned to the newly built WebUI digest, per the staged
release sequence in docs/hux/HANDOFF.md.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:46:01 -03:00
jenkins
d50d014572 docs(hux): record completion increments and staged release runbook
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:45:38 -03:00
jenkins
ffa003d3fb test(hermes): tolerate fieldRef env entries in chat voice gate
The webui container now carries a POD_NAME fieldRef for HUX subPathExpr
mounts; the voice routing assertions only ever inspected literal
values.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:45:15 -03:00
jenkins
e2ad645733 hermes(router-plugin): deliver hux_mode.py through the auto-router ConfigMap
The generator now ships the HUX-06 adoption module alongside the
plugin, and the plugin degrades to its previous behaviour (no mode
adoption) if it starts against a ConfigMap rendered before the module
existed - the plugin can never fail to load mid-transition.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:44:28 -03:00
jenkins
dc034cb738 hermes(hux): bounded message-text search with privacy enforcement
HUX-03: GET /hux/v1/search now accepts include=message_text, an
explicit opt-in that scans the stored message events of the 100 most
recently active candidate conversations. Forgotten (tombstoned)
conversations, private-mode conversations, restricted events and fully
redacted events never match; the default indexed-fields search and its
response contract are unchanged (the shipped UI keeps requiring
message_text in not_indexed). Paginated mode uses one deterministic
total order (score desc, updated_at desc, id) with an offset cursor.
organization.py at 98% branch coverage.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:43:08 -03:00
jenkins
e40fc5ec5d hermes(chat): stage the HUX-12 evidence producer sidecar
Activation-layer staging, fail-closed until enablement: a per-tenant
hux-evidence-producer sidecar on the exact reviewed WebUI image runs
hux_producer.run_once on a 60s loop, inert until the Vault-staged
evidence key (tolerant init, tmpfs, 0400, staged only for the hux
service and producer containers - never hermes or webui), the policy
ConfigMap, and the scope ConfigMap exist. Adds least-privilege
read-only RBAC (pods+statefulset in hermes, the single named Flux
Kustomization), tenant egress to the Kubernetes API ClusterIP and the
traefik edge, the policy allowlist, hux_producer packaging in the WebUI
image, and a third expected WebUI consumer in the Flux release
renderer. Delivery and image-automation gates enforce the boundary.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:39:45 -03:00
jenkins
469e52fd20 hermes(hux): add the HUX-12 release evidence producer
A companion package (outside the network-free hux/ service package)
that independently verifies and binds the whole release chain before
any transition: reviewed proposal URL, Jenkins job/build/result and
revision, immutable Harbor tag/digest equality, Flux kustomization and
applied revision with pin containment, desired workload image, every
Ready pod imageID, bounded-age health receipt, and rollback target.
Pure injectable verifier core, HTTPS-only collectors (SA token for the
Kubernetes API), and an evidence-trust driver that posts exactly one
If-Match transition with deterministic idempotency. Rejects stale,
replayed, downgraded, incomplete, cross-workload, mismatched, and
self-asserted evidence. 100% line and branch coverage (71 tests).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:35:12 -03:00
jenkins
c089a5ec2a hermes(runtime): emit artifacts and web sources from real tool output
HUX-04/HUX-08: after a successful tool execution the hux-runtime plugin
now registers freshly written files as typed artifacts (create or
immutable version by conversation-scoped title, bounded to 1 MiB,
deterministic idempotency keys) and records up to three deduplicated
web sources from real network/web_search output. Emitters are fail-open
and can never break the tool result. Delivered through the plugin
ConfigMap; 95% branch coverage; quality contract updated.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:35:12 -03:00
jenkins
2eedcd2066 security(hux): private mode denies web, messaging, shell and delegation
HUX-06/HUX-10: a conversation whose stored friendly mode is private now
has network, web_search, send_message, shell and delegate refused by
both the approval resolver and the pre-side-effect gate, regardless of
autonomy policy — matching the mode catalog's tool contract. Memory
writes were already refused by the privacy state. Missing or unbound
conversations keep the existing matrix behaviour.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:28:23 -03:00
jenkins
0755e1bbd7 hermes(router-plugin): adopt HUX friendly modes at the route boundary
HUX-06: the auto-router now consults the loopback HUX service (worker
trust, HMAC-derived conversation id from the persisted session) for the
conversation's selected friendly mode and maps it onto the existing
provider-neutral pools: fast->auto/fast, thoughtful/research->auto/deep,
create->auto/balanced; private pins the local route and refuses hosted
overrides. Explicit UI picks keep precedence; every HUX absence or
failure falls through to the previous behaviour unchanged. 95% branch
coverage; quality contract registers the new module.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:27:02 -03:00
jenkins
b3d4538b4f docs(hux): describe the integrated staged topology and trust reality
HANDOFF.md and THREAT-MODEL.md now document the actual HMAC subject/
context identity, the shared RWX PVC with kubelet subPathExpr per-pod
isolation and its exact boundary, the uid-10000 no-store-mount agent
model, tmpfs transport key rotation, router X-Hux-* stripping, HUX-12
file-only evidence trust, staged-not-deployed status, and the verified
SO-46/48 (open) and SO-53 (shipped) states.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:23:24 -03:00
jenkins
b2cfbc94e8 hermes(webui): track the ignored HUX-04 artifacts card models
The artifacts/ directory matched a repo-wide ignore rule and was left
out of the UI card commit; its suites only passed locally because the
files existed on disk. Force-track the complete card so the source
push carries every module the tests import.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:19:03 -03:00
jenkins
13359769dc ci(hermes): make HUX topology gates all-or-nothing adaptive
The delivery and image-automation gates now enforce whichever state the
chat StatefulSet is actually in: with no hux sidecar they require zero
partial HUX wiring (no containers, volumes, PVC, or HUX_* env); with the
sidecar staged they enforce the full strict boundary. This lets the
reviewed source chain merge and build before the activation topology
lands, without ever waiving an activated assertion.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:17:02 -03:00
jenkins
659f70ee03 docs(hermes): record HUX WebUI foundation integration contract
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:12:42 -03:00
jenkins
71c05cf9c0 hermes(chat): stage HUX sidecar activation topology
Activation (not yet pushed): per-tenant HUX sidecar on the exact live
WebUI image, standalone Astreae RWX PVC with kubelet subPathExpr
per-pod isolation, root init container that provisions 0700 tenant
roots, a persistent HMAC context key, an immutable subject binding and
tmpfs relay/worker keys, the vendored hux-runtime plugin ConfigMap, and
observe-only defaults (HUX_TOOL_ENFORCEMENT=0). Delivery tests pin the
whole boundary; image-automation tests bind the HUX metadata setters.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:12:03 -03:00
jenkins
0b56d04e9d ci(hermes): gate HUX runtime plugin and new families in quality contract
Adds modes, multimodal, releases, release_security, retention_scheduler,
suggestions and the vendored hux-runtime plugin modules to lint, LOC and
coverage enforcement.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:12:03 -03:00
jenkins
9e99470fef release(hermes): render one or two WebUI consumers with HUX binding
The Flux release renderer now accepts an expected-consumer set per
workload (chat may carry the HUX sidecar as a second consumer of the
exact same WebUI image) and binds HUX_IMAGE_TAG/HUX_IMAGE_DIGEST env
metadata to the released tag and digest when those fields are present.
Rendering fails when the binding fields are incomplete, keeping the
image identity single-sourced. Tests adapt to both topologies.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:12:03 -03:00
jenkins
bd63b568e1 security(hermes): strip inbound X-Hux-* at the router boundary
Defense in depth: the tenant router deletes every browser-supplied
X-Hux-* header before asserting its own HUX identity headers, so no
client can forge subject, trust class, or relay key. Regression covers
X-Hux-Subject, X-Hux-Trust and X-Hux-Relay-Key forgeries.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:12:03 -03:00
jenkins
b66c762f5d hermes(webui): add HUX card UI models with contract-locked suites
Standalone per-card browser model/security/view modules for HUX-01..10
plus node+pytest suites that read the hux.v1 contract schemas directly.
Reconciled drift found on integration: the activity model now accepts
all 32 hux.event.v1 kinds (delegation.*, memory.suppressed,
memory.retrieval_removed, budget.exhausted, side_effect.*), the
autonomy model carries the external_side_effect capability, and the
foundation boundary test now asserts the shipped static HUX surface
exists on disk and that images never bake activated HUX_FLAGS.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 04:12:03 -03:00
jenkins
47165274f0 security(hermes): isolate HUX release evidence trust 2026-08-24 03:10:13 -03:00
jenkins
1eda927b71 hermes(webui): integrate trusted HUX workspace 2026-08-24 03:08:56 -03:00
jenkins
31c5eae5ae release(hermes): automate chat router image 2026-08-24 02:46:36 -03:00
jenkins
53d7c2c586 feat(hermes): complete HUX backend families 2026-08-24 02:28:45 -03:00
jenkins
a1070449a7 fix(hermes): vendor HUX runtime hooks 2026-08-24 02:23:04 -03:00
jenkins
438180a99f feat(hermes): gate tool runtime through HUX 2026-08-24 02:17:04 -03:00
jenkins
964103f02b merge(hermes): integrate hardened HUX foundation 2026-08-24 02:01:08 -03:00
jenkins
a260506983 security(hux): converge backend trust boundaries 2026-08-24 01:58:14 -03:00
flux-bot
5558c24f86 chore(hermes): promote validated image release 2026-08-24 04:48:04 +00:00
jenkins
f59f24a427 security(hux): harden backend trust boundaries 2026-08-24 01:31:12 -03:00
jenkins
be5566cabd fix(hermes): reuse exact speculative speech snapshot 2026-08-24 01:28:48 -03:00
flux-bot
a23cf838b2 chore(hermes): promote validated image release 2026-08-24 04:24:04 +00:00
jenkins
3c06ae867b fix(hermes): reject invalid release revisions early 2026-08-24 01:19:58 -03:00
flux-bot
33de8e38fe chore(hermes): promote validated image release 2026-08-24 04:05:59 +00:00
jenkins
98c7c6184f docs(hux): point the ledger at the rebased commit ids 2026-08-24 00:54:48 -03:00
jenkins
18b980d6fa hermes(hux): expose conversation privacy state for the agent hook
GET /hux/v1/conversations/{id}/privacy reports forgotten, memory_disabled,
topics, mode and memory_writes_allowed; the worker hook's memory gate reads it
and fails closed.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
2026-08-24 00:54:34 -03:00
jenkins
e1ef110c8a docs(hux): close the Wave A review in the handoff ledger 2026-08-24 00:48:20 -03:00
jenkins
a75ca29934 hermes(hux): force restricted sensitivity on secret-bearing artifact content
User artifacts are never rewritten, but a secret pattern in a version marks
the artifact restricted and audits the reason (review a2-C, SO-12).

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
2026-08-24 00:48:20 -03:00
jenkins
681b040885 hermes(hux): close Wave A review findings in events, memory, privacy and organization
F3 memory edits go through the same privacy shaping as proposals; F5 seq is
derived from the ledger tail so a crash between append and checkpoint never
duplicates; F7 transitions re-read under the lock and always write with the
loaded revision; F9 secret scrub on titles, passages, claims and notebooks
and forget blanks the conversation document; F13 no ghost conversations
from notices, idempotency under the lock, artifact titles searchable,
normalised paths.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
2026-08-24 00:48:20 -03:00
jenkins
6964a9d8c8 hermes(hux): close Wave A review findings in autonomy and the HTTP pipeline
F1 policy writes and allow grants are human-surface only; F2 worker trust is
confined to the hook allowlist and unexpected exceptions become audited 500
error records; F4 external side effects release only for the same run and
argument hash; F6 the gate honours budget exhaustion; F8 only the gateway
can vouch for an empty process registry and failed receipts can be
superseded; F11/F12 receipt revision and unshipped card routes.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
2026-08-24 00:48:20 -03:00
jenkins
74c6549965 docs(hux): record the Wave A review outcome in the threat model 2026-08-24 00:48:20 -03:00
jenkins
aeef0f8484 hermes(hux): worker-side hook library for approvals, gates, budgets, stop receipts and events
Stdlib client the agent runtime calls around its tool loop; fails closed for
side effects, fails open for telemetry, never carries raw arguments or outputs.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
2026-08-24 00:48:20 -03:00
jenkins
1b1a14e972 hermes(hux): contract 1.1.0 additive revision and Wave A consolidation
Adds receipt evidence kind, 422 unprocessable, optional revision on research
records, audit_stale on the privacy policy, per-route body caps (25 MiB
artifact uploads), promotion checks the project exists, memory rules skip
content-free statuses. Handoff ledger covers every Wave A card.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
2026-08-24 00:48:20 -03:00
jenkins
b8fb72fbcc hermes(hux): HUX-01 redacted activity events, HUX-02 memory ledger, HUX-10 privacy behaviour
Per-conversation monotonic event ledger with idempotent emit, SSE replay from
Last-Event-ID, per-kind detail allowlists and secret scrubbing, surface-aware
serve-time redaction; memory as an append-only ledger with no-store,
supersede, forget and retrieval tombstones so 'do not remember' blocks both
persistence and retrieval; sensitive-topic scoping, notices, conversation
forget and a retention job that never touches audit.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
2026-08-24 00:48:20 -03:00
jenkins
3ebee2cc15 hermes(hux): HUX-05 autonomy engine and minimal HUX-03 organization API
Scoped policies resolved through the single capability matrix, grants with
server-set expiry, approval queue with once/session/always/deny, pre-side-effect
gate that releases a once approval exactly once against the argument hash,
run budgets with exhaustion events, honest cancellation receipts; projects,
conversations, branch lineage and search over the indexed fields.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
2026-08-24 00:48:20 -03:00
jenkins
9f0ffe031c hermes(hux): HUX-04 artifact workspace and HUX-08 research backends
Immutable content-addressed artifact versions with If-Match, lineage that must
resolve under the caller, unified diffs, promotion; sources/passages/citations
with server-side hashing and dedupe, citation integrity checks and revisioned
research notebooks. Cross-tenant access is 404 and audited.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
2026-08-24 00:48:20 -03:00
jenkins
01abf89270 docs(hux): record draft PR #55 in the handoff ledger 2026-08-24 00:48:20 -03:00
jenkins
b9b7126ee1 docs(hux): start the hermes-next handoff ledger with HUX-11 2026-08-24 00:48:20 -03:00
jenkins
6a5e0d872b hermes(hux): HUX-11 foundation service core with threat and data models
Stdlib per-tenant service: trusted-header identity (router/relay/worker,
constant-time keys, slot pinned to the pod), fail-closed card flags with
capability negotiation, tenant-scoped store (atomic writes, revisions,
append-only ledgers, content-addressed blobs, manifest), audit outcome for
every request, and the /hux/v1 pipeline that maps errors to hux.error.v1.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
2026-08-24 00:48:20 -03:00
jenkins
b4145861d2 hermes(hux): freeze hux.v1 contract 1.0.0 and move reference modules into the foundation package
HUX-11 contract freeze: identity/capabilities/manifest/error records, event turn
and idempotency and delegation/side-effect kinds, memory no-store/supersedes/
retrieval removal, revisions for optimistic concurrency, artifact access,
budget scope/spend/subagents and external side-effect gating, notebook notes and
dedupe keys. ADR-0001 records the wire and compatibility rules.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
2026-08-24 00:48:20 -03:00
flux-bot
1a586d9523 chore(hermes): promote validated image release 2026-08-24 03:43:01 +00:00
jenkins
91eb4f92b7 perf(hermes): isolate rolling speech inference 2026-08-24 00:36:28 -03:00
flux-bot
c7aa7f462b chore(hermes): promote validated image release 2026-08-24 03:11:53 +00:00
jenkins
e9a6eba30b fix(hermes): stabilize chat canvas gradients 2026-08-24 00:01:52 -03:00
jenkins
af5ee4697b fix(hermes): warm STT before accepting speech 2026-08-23 23:52:30 -03:00
flux-bot
70d797deb3 chore(hermes): promote validated image release 2026-08-24 02:27:52 +00:00
jenkins
0dd6138d05 ci(hermes): permit WebUI test package setup 2026-08-23 23:10:26 -03:00
jenkins
817c7cebe0 ci(hermes): install WebUI voice test tools 2026-08-23 23:04:21 -03:00
flux-bot
9aed543dd3 chore(maintenance): automated image update 2026-08-24 01:50:51 +00:00
flux-bot
19eadc9556 chore(hermes): promote validated image release 2026-08-24 01:50:42 +00:00
flux-bot
30fff866d1 chore(maintenance): automated image update 2026-08-24 01:49:48 +00:00
flux-bot
66d09ec920 chore(maintenance): automated image update 2026-08-24 01:46:50 +00:00
flux-bot
807dd7c093 chore(maintenance): automated image update 2026-08-24 01:41:46 +00:00
flux-bot
fb0ae53b26 chore(hermes): promote validated image release 2026-08-24 01:38:43 +00:00
jenkins
d3cb6e9045 hermes(hux): add contract-only foundation for the chat UX program
Schemas, examples and a flag registry for the twelve HUX cards, a dependency-free
validator, the governance rules (memory ledger, autonomy matrix, friendly modes
mapped to real Switchyard routes, privacy defaults, suggestion gating, release
state machine) and the contract doc UI work codes against.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
2026-08-23 22:30:41 -03:00
jenkins
d254931a14 feat(hermes): ship full-duplex voice release 2026-08-23 22:13:52 -03:00
flux-bot
d3cbeb06c3 chore(hermes): promote validated image release 2026-08-23 23:39:07 +00:00
flux-bot
dbb0bf542e chore(hermes): promote validated image release 2026-08-23 23:24:33 +00:00
jenkins
1d8d466ccf ci(hermes): move WebUI builds off storage nodes 2026-08-23 20:20:29 -03:00
jenkins
80d728d4ff ci(hermes): bound voice image memory 2026-08-23 20:01:55 -03:00
flux-bot
ccbb459dc4 chore(hermes): promote validated image release 2026-08-23 22:40:52 +00:00
jenkins
a68568d0c9 fix(hermes): version chat release assets 2026-08-23 19:22:26 -03:00
flux-bot
55e21c9e08 chore(hermes): promote validated image release 2026-08-23 22:08:29 +00:00
jenkins
636f3fcf93 fix(hermes): harden release reconciliation 2026-08-23 18:53:11 -03:00
jenkins
708d611101 feat(hermes): stream hands-free voice turns 2026-08-23 18:29:12 -03:00
jenkins
45ab125692 fix(hermes): skip completed startup migration 2026-08-23 18:20:08 -03:00
flux-bot
0fab341263 chore(hermes): promote validated image release 2026-08-23 21:04:22 +00:00
jenkins
afaccbe65c ci(hermes-voice): build on roomy ARM accelerator 2026-08-23 17:53:05 -03:00
jenkins
33c6f187b3 fix(hermes): recognize containerd release images
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 19, passed: 2816
2026-08-23 17:23:51 -03:00
jenkins
b87f33530b feat(hermes): brand chat and prove image rollout 2026-08-23 17:13:14 -03:00
flux-bot
ddc46aa845 chore(hermes): promote validated image release 2026-08-23 20:12:17 +00:00
jenkins
6a55627866 ci(hermes-images): target spacious ARM builders 2026-08-23 17:03:16 -03:00
jenkins
21443895ff test(hermes): accept Flux release-tagged digests 2026-08-23 16:55:20 -03:00
jenkins
b98a7cbe9f ci(hermes-webui): reserve image build storage 2026-08-23 16:54:10 -03:00
jenkins
89b0c34854 build(hermes-webui): collapse patch snapshots 2026-08-23 16:52:27 -03:00
jenkins
0f9b43c8ba perf(hermes-voice): reduce conversational latency 2026-08-23 16:50:48 -03:00
flux-bot
cb5cf6ea12 chore(hermes): promote validated image release 2026-08-23 19:41:44 +00:00
jenkins
589a3133f2 ci(hermes-voice): reserve image build storage 2026-08-23 16:33:17 -03:00
jenkins
27763f297d ops(jenkins): resume corrected reconciliation 2026-08-23 16:30:14 -03:00
jenkins
73dd267c61 ops(jenkins): suspend failed reconciliation 2026-08-23 16:29:45 -03:00
jenkins
df03514d75 fix(flux): reconcile Jenkins desired state 2026-08-23 16:26:43 -03:00
jenkins
9d34b397dd fix(flux): bound Jenkins recovery timeout 2026-08-23 16:25:53 -03:00
jenkins
1a3e7f39f5 fix(jenkins): keep voice concurrency in pipeline 2026-08-23 16:21:19 -03:00
jenkins
5c986eabaa fix(hermes-voice): unpack Jetson image capabilities 2026-08-23 16:05:09 -03:00
jenkins
ef32843a76 fix(hermes-voice): finalize mobile audio and reduce latency 2026-08-23 15:59:43 -03:00
jenkins
79369c2357 release(hermes): automate private voice images 2026-08-23 15:54:56 -03:00
flux-bot
df18acbfa7 chore(hermes): promote validated image release 2026-08-23 18:35:22 +00:00
jenkins
225793facd fix(mailu): self-heal mailbox lock failures 2026-08-23 14:50:02 -03:00
jenkins
5b5df4ea59 fix(hermes): validate tagged WebUI releases 2026-08-23 14:27:55 -03:00
jenkins
8d9b75a28c fix(hermes): retain full automated image refs 2026-08-23 14:22:26 -03:00
flux-bot
0da2483dd3 chore(hermes): promote validated image release 2026-08-23 17:20:01 +00:00
jenkins
ad487b5e1c fix(hermes): reuse authenticated image checkout 2026-08-23 13:58:10 -03:00
jenkins
41233f94a9 fix(jenkins): persist Hermes release parameters 2026-08-23 13:53:21 -03:00
jenkins
c0b806e5d2 release(hermes): automate validated image promotion 2026-08-23 13:41:58 -03:00
jenkins
17c5f5093a placement(quality): use titan-22 spare capacity 2026-08-23 11:50:39 -03:00
jenkins
42795f3d61 cleanup(hermes): remove Veles observer binding 2026-08-23 11:41:33 -03:00
jenkins
0ebab9d412 placement(wger): use preemptible titan-22 capacity 2026-08-23 11:23:09 -03:00
jenkins
e22e4793bb health(wger): replace expensive application probes 2026-08-23 11:21:19 -03:00
jenkins
435e258e45 rollout(hermes): activate Claude health hysteresis 2026-08-23 11:19:48 -03:00
jenkins
6b378fd410 fix(hermes): preserve proven Claude authentication 2026-08-23 11:18:18 -03:00
jenkins
0682b2b813 resources(finance): right-size Actual Budget memory 2026-08-23 11:07:33 -03:00
jenkins
1e66a0bedf fix(hermes): tolerate worker lease handoff 2026-08-23 11:01:20 -03:00
jenkins
e9e355f25a resources: release Pi scheduler headroom 2026-08-23 10:53:57 -03:00
jenkins
6fabb61844 resources(comms): right-size idle Synapse requests 2026-08-23 10:50:53 -03:00
jenkins
b7be05427e fix(hermes): place arm64 agent on titan-08 2026-08-23 10:45:51 -03:00
jenkins
f1739e7108 fix(hermes): keep sessions off storage nodes 2026-08-23 10:43:42 -03:00
jenkins
1e6eda8633 fix: converge load-spread rollouts 2026-08-23 10:41:01 -03:00
jenkins
271f3e8c32 ops: spread saturated node workloads 2026-08-23 10:37:01 -03:00
jenkins
a019ecd556 monitoring(ai): compare provider quota windows 2026-08-23 09:48:25 -03:00
ced9a24cd9 Merge pull request 'Hermes WebUI release lane v2' (#48) from feature/t_8cbe6a55-hermes-webui-release-v2 into main
Reviewed-on: atlas/titan-iac#48
Reviewed-by: bstein <bstein@noreply.scm.bstein.dev>
2026-08-23 12:19:10 +00:00
58e1a29b2b Merge branch 'main' into feature/t_8cbe6a55-hermes-webui-release-v2
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 19, passed: 2782
2026-08-23 12:18:58 +00:00
flux-bot
51515932fc chore(bstein-dev-home): automated image update 2026-08-23 09:58:39 +00:00
flux-bot
1eaaf57abd chore(bstein-dev-home): automated image update 2026-08-23 09:57:02 +00:00
jenkins
fe7832efaf hermes: retain dual-provider quota health
Some checks failed
Tests / Declarative: Post Actions failed: 39, skipped: 19, passed: 2757
2026-08-23 02:52:42 -03:00
Hermes Agent
083005e44f fix: verify served Hermes PWA identity 2026-08-23 04:42:08 +00:00
Hermes Agent
99a9de936b fix: verify WebUI OCI source revision 2026-08-23 03:52:03 +00:00
Hermes Agent
735a2b240d Add reviewed Hermes WebUI release lane 2026-08-23 03:52:03 +00:00
jenkins
81b3e6b992 hermes: expose Claude remaining quota 2026-08-23 00:42:39 -03:00
jenkins
d036062519 hermes: stabilize AI quota collection 2026-08-23 00:14:58 -03:00
jenkins
3da6a10174 hermes: remove provider-biased auto fallbacks 2026-08-22 23:54:40 -03:00
jenkins
a7e09f9719 monitoring(ai): measure Claude routed usage 2026-08-22 23:38:40 -03:00
jenkins
69ad76bebc hermes: keep distributed workers on Claude 2026-08-22 23:02:15 -03:00
flux-bot
5136a0026e chore(maintenance): automated image update 2026-08-23 01:59:50 +00:00
flux-bot
42a86687c4 chore(maintenance): automated image update 2026-08-23 01:58:38 +00:00
flux-bot
049c39b339 chore(maintenance): automated image update 2026-08-23 01:54:50 +00:00
jenkins
8acb99d100 hermes: make Claude setup-token access durable 2026-08-22 22:52:28 -03:00
flux-bot
5e0b0d9cd6 chore(maintenance): automated image update 2026-08-23 01:49:36 +00:00
jenkins
b8294f8e90 longhorn: reduce Jellyfin policy churn 2026-08-22 22:18:03 -03:00
jenkins
eec373dd56 jellyfin: run media service on titan-22 2026-08-22 22:10:46 -03:00
jenkins
5430c9ac01 jellyfin: prepare titan-22 media host 2026-08-22 22:06:39 -03:00
jenkins
76fe550871 node(titan-22): remove completed canaries 2026-08-22 21:22:33 -03:00
jenkins
21264a71c8 node(titan-22): return worker to service 2026-08-22 21:21:03 -03:00
jenkins
091a787fc7 node(titan-22): persist Realtek firmware 2026-08-22 20:55:47 -03:00
jenkins
074f988731 node(titan-22): migrate k3s to USB ethernet 2026-08-22 20:42:51 -03:00
jenkins
db0cb86ca6 hermes: stabilize Claude access health probes 2026-08-22 18:07:58 -03:00
jenkins
9bffc077e8 hermes: avoid recursive home ownership rollout
Some checks failed
Tests / Declarative: Post Actions failed: 40, skipped: 19, passed: 2746
2026-08-22 17:38:42 -03:00
jenkins
bd50a7e4ad hermes: persist Claude subscription access 2026-08-22 17:32:30 -03:00
jenkins
9bf41dc0de hermes: decouple node maintenance health 2026-08-22 16:39:44 -03:00
jenkins
fd42e892f7 hermes: avoid flapping worker node 2026-08-22 16:17:33 -03:00
jenkins
fdf53b264b hermes: fit owner pod on healthy workers 2026-08-22 15:55:30 -03:00
jenkins
45a7a4dded platform: right-size Vault sidecar requests 2026-08-22 15:48:41 -03:00
jenkins
46a44241c1 hermes: reclaim healthy worker capacity 2026-08-22 15:42:56 -03:00
jenkins
8a8df5ee4f hermes: add stateful accelerator fallback 2026-08-22 15:32:21 -03:00
jenkins
9b7adc3826 mailu: allow Dovecot storage failover 2026-08-22 15:09:54 -03:00
jenkins
41658393a4 mailu: remount Dovecot storage after lock failure 2026-08-22 14:50:16 -03:00
flux-bot
a48322c5ee chore(maintenance): automated image update 2026-08-22 14:00:17 +00:00
flux-bot
821e27e845 chore(maintenance): automated image update 2026-08-22 13:54:16 +00:00
flux-bot
121daf9145 chore(maintenance): automated image update
Some checks failed
Tests / Declarative: Post Actions failed: 45, skipped: 19, passed: 2738
2026-08-22 01:54:06 +00:00
flux-bot
c19e9d77bf chore(maintenance): automated image update 2026-08-22 01:54:00 +00:00
flux-bot
5826c6054d chore(maintenance): automated image update 2026-08-22 01:49:59 +00:00
flux-bot
69f9ec99f2 chore(maintenance): automated image update 2026-08-22 01:43:57 +00:00
66c1c11d06 Merge pull request 'Rename the owner agent host to worker.bstein.dev (supersedes #41)' (#42) from feature/hermes-domain-rename-agent-worker-v2 into main
Reviewed-on: atlas/titan-iac#42
Reviewed-by: bstein <bstein@noreply.scm.bstein.dev>
2026-08-21 23:16:04 +00:00
ed98278981 Merge branch 'main' into feature/hermes-domain-rename-agent-worker-v2
Some checks failed
Tests / Declarative: Post Actions failed: 40, skipped: 19, passed: 2743
2026-08-21 23:15:16 +00:00
4c9f48cbbc Merge pull request 'docs(hermes): add multi-user chat capacity assessment' (#45) from hermes/t_65356568-multiuser-capacity-assessment into main
Reviewed-on: atlas/titan-iac#45
Reviewed-by: bstein <bstein@noreply.scm.bstein.dev>
2026-08-21 23:14:47 +00:00
8ce45159de Merge branch 'main' into hermes/t_65356568-multiuser-capacity-assessment
Some checks failed
Tests / Declarative: Post Actions failed: 41, skipped: 19, passed: 2740
2026-08-21 23:14:33 +00:00
0841681a06 Merge pull request 'fix(hermes): cli-auto capacity failover uses automatic Switchyard reclassification' (#32) from fix/cli-auto-failover-effort into main
Reviewed-on: atlas/titan-iac#32
Reviewed-by: bstein <bstein@noreply.scm.bstein.dev>
2026-08-21 23:13:08 +00:00
77905ff1db Merge branch 'main' into fix/cli-auto-failover-effort
Some checks failed
Tests / Declarative: Post Actions failed: 40, skipped: 19, passed: 2737
2026-08-21 23:12:12 +00:00
fcef884699 Merge pull request 'Hermes: combine multilingual Piper with Whisper language routing (after #43)' (#46) from feature/hermes-pr44-after-pr43 into main
Reviewed-on: atlas/titan-iac#46
Reviewed-by: bstein <bstein@noreply.scm.bstein.dev>
2026-08-21 21:30:06 +00:00
Hermes Agent
820872e117 feat(hermes-voice): route Whisper language to multilingual Piper
Port the original #27 detected-language pipeline onto the verified PR #39 prerequisite while preserving the current-main conversation instrument and host continuity changes.

Keep voice selection server-side with no user selector or client voice field. Reuse 207c16ab only for its stricter exact-code trust boundary, omitting malformed or absent language so Piper defaults to Amy.
2026-08-21 13:58:50 +00:00
Hermes Agent
724656d841 feat(hermes-tts): prepare fixed multilingual voice policy
Supersede draft PR #26 with a merge-safe prerequisite: bake and preload the amy, irina, and claude Piper models, route only validated server-side language to fixed voices, and leave the live voice deployment manifest unchanged.

Remove the pinned WebUI speaker selector and its persisted preference, omit client voice fields from every outbound TTS path, and keep hands-free Voice Mode and the conversation instrument intact. Hostile or legacy voice fields remain ignored by the Piper server.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-21 13:45:31 +00:00
431533dfce Merge pull request 'Fix hands-free STT WebM conversion' (#43) from fix/hermes-handsfree-stt-webm into main
Some checks failed
Tests / Declarative: Post Actions failed: 40, skipped: 7, passed: 2612
Reviewed-on: atlas/titan-iac#43
Reviewed-by: bstein <bstein@noreply.scm.bstein.dev>
2026-08-21 13:38:58 +00:00
Hermes Agent
d8ccf4a6fa docs(hermes): add multi-user chat capacity assessment
Read-only investigation of the chat request path, replica/resource
manifests, and live node metrics. Documents the confirmed 4-user
tenant-slot ceiling, a TENANT_SLOTS=8 vs replicas=4 configuration
drift, the shared Claude-broker concurrency=2 bottleneck, missing
HPA/PDB/staging environment, and a proposed SLO/load-test and staged
scaling plan pending human approval. No production manifests changed.

Co-Authored-By: Hermes Agent <hermes-automation@bstein.dev>
2026-08-21 12:38:02 +00:00
Hermes Agent
cfd8a75e95 fix(hermes-voice): preserve MediaRecorder container headers 2026-08-21 11:12:15 +00:00
Hermes Agent
94106bf252 refactor(hermes): rename the owner agent host to worker.bstein.dev
Introduce worker.bstein.dev as the canonical hostname for the owner-only
Hermes coordinator, previously agent.hermes.bstein.dev.

The rename is additive, matching the shape #38 restored for chat and triage.
CoreDNS, both agent Ingresses and the hermes-sites certificate now serve BOTH
names, so merging this cannot take away the endpoint the operator uses to
reach the coordinator. Retiring agent.hermes.bstein.dev is a separate,
separately scheduled change. No redirect middleware is added.

What switches to the new host:
- HERMES_DASHBOARD_PUBLIC_URL and the oauth2-proxy --redirect-url
- the Keycloak agent proxy rootUrl
- operator docs, skills, the ZAP baseline target and the triage monitor default

What stays dual-homed until retirement:
- CoreDNS hosts entry, both agent Ingress rules, certificate SANs
- API_SERVER_CORS_ORIGINS (now a comma-separated pair)
- the Keycloak redirect URIs, web origins and post-logout origins, so a
  rollback only needs the oauth2-proxy --redirect-url reverted and does not
  require re-running the ensure job

The agent client passes its legacy origin through the optional fourth argument
#38 added to ensure_proxy_client, so no second mechanism is introduced. The
immutable ensure Job goes -11 -> -12 because #38 already consumed -11 and that
run has completed; without a further bump this change would never be applied.
Login on the new host fails until the -12 Job completes.

Because the session and CSRF cookies use the __Host- prefix they are bound to
one origin, so a fresh login must start on worker.bstein.dev and existing
sessions do not carry over -- re-login is required after rollout.

#38's public-host continuity test now covers the agent proxy's dual origins
rather than asserting the agent surface was untouched by the rename.

Knowledge catalogs and diagrams regenerated with `make knowledge`.
2026-08-21 10:29:46 +00:00
Hermes Agent
004c41629e chore(knowledge): regenerate stale Atlas catalogs
`make knowledge` output on main no longer matched the manifests it renders
from. The legacy chat/triage hosts restored by #38 were missing from the
committed HTTP catalogs and diagrams, along with Flux kustomizations and
Grafana panels added since the last regeneration.

Pure `make knowledge` run against unmodified main, separated into its own
commit so the hostname rename that follows reviews as a hostname rename and
nothing else. No hand edits.
2026-08-21 10:24:46 +00:00
5f9c600f6e Merge pull request 'hermes: restore legacy chat/triage hosts alongside the renamed ones' (#38) from fix/hermes-restore-legacy-chat-triage-hosts-v2 into main
Reviewed-on: atlas/titan-iac#38
Reviewed-by: bstein <bstein@noreply.scm.bstein.dev>
2026-08-21 10:17:49 +00:00
4d072f32ce Merge branch 'main' into fix/hermes-restore-legacy-chat-triage-hosts-v2 2026-08-21 10:17:25 +00:00
flux-bot
ab33b582c8 chore(bstein-dev-home): automated image update 2026-08-21 10:06:01 +00:00
flux-bot
678698c44b chore(bstein-dev-home): automated image update 2026-08-21 10:04:01 +00:00
Hermes Agent
fd4bf69007 test(hermes): pin public chat/triage host continuity across every layer
The #34 rename dropped the legacy chat/triage names from the certificate
SANs, the hermes-sites Ingress, the CoreDNS overrides and the Keycloak
ensure script at the same time, so nothing failed loudly: DNS and TLS
still looked healthy while the legacy hosts served 404 and the renamed
hosts could not finish a login.

Pin the invariant that makes that silent: a public host is either served
by all four layers or by none. The table of hosts is the contract, so
retiring a name stays a deliberate edit rather than a side effect.

Verified to catch the regression: against the pre-fix tree these fail for
both legacy hosts on all four layers (9 failures); against this branch
the suite is green.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-21 08:57:50 +00:00
Hermes Agent
f4f51323f6 hermes: restore legacy chat/triage hosts alongside the renamed ones
PR #34 renamed the public chat/triage hosts in place rather than adding
the new names, so chat.hermes.bstein.dev and triage.hermes.bstein.dev
were dropped from the certificate SANs, the hermes-sites Ingress rules
and the CoreDNS overrides at once. Both legacy hosts now answer 404 with
Traefik's default self-signed certificate, and the renamed hosts cannot
complete a login because the Keycloak clients still carry the old
redirect URIs, so chat and triage are unreachable on every hostname.

Make the rename additive, which is the rollback path the post-merge
runbook asks for when the OIDC step fails:

- put the legacy names back on hermes-sites-tls and on the Ingress,
  pointing at the same oauth2-proxy backends
- restore both CoreDNS host overrides for in-cluster resolution
- teach ensure_proxy_client to register an optional legacy origin, so
  hermes-chat-proxy and hermes-triage-proxy accept the old and new
  redirect URIs, web origins and post-logout origins at the same time
  while rootUrl stays on the canonical new host
- bump the immutable ensure Job so Flux reruns the script

Serving both names is deliberate: oauth2-proxy cookies are host-bound,
so redirecting the legacy hosts would silently drop live sessions.
Retiring them stays a separate, explicit change.

Supersedes #36, which only bumped the Job and would have left the
legacy hosts dark.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-21 08:49:09 +00:00
e3de466ad1 Merge pull request 'Complete direct CLI lane concurrency omitted by #30 (supersedes #31)' (#35) from feature/hermes-direct-cli-lane-concurrency-2-replacement into main
Some checks failed
Tests / Declarative: Post Actions failed: 36, skipped: 7, passed: 2591
Reviewed-on: atlas/titan-iac#35
Reviewed-by: bstein <bstein@noreply.scm.bstein.dev>
2026-08-21 07:58:17 +00:00
66a537ddbe Merge branch 'main' into feature/hermes-direct-cli-lane-concurrency-2-replacement 2026-08-21 07:57:35 +00:00
8ace4d47e1 Merge pull request 'refactor(hermes): rename chat and triage public hostnames' (#34) from feature/hermes-domain-rename-chat-bstein-triage into main
Reviewed-on: atlas/titan-iac#34
Reviewed-by: bstein <bstein@noreply.scm.bstein.dev>
2026-08-21 07:48:55 +00:00
Hermes Agent
4277aa6a02 fix(hermes): allow two direct CLI lane workers 2026-08-21 07:11:55 +00:00
Hermes Agent
d579c08cbe fix(hermes): cli-auto capacity failover uses automatic Switchyard reclassification
Capacity/auth/quota failover for cli-auto previously called select_route
with a hardcoded manual lane (cli-{alternate}-{effort}), so the retry
boundary was classified as switchyard-manual instead of going through
Jetson automatic classification, making the routing evidence misleading.

Now the retry calls select_route(context, "cli-auto", exclude_provider=...)
so the boundary stays automatically classified with an explicit
failed-provider exclusion. If the classifier reclassifies to a lower
effort than the original route, the lane re-pins the chosen provider at
the original effort floor so a capacity failure never silently downgrades
a high/xhigh task. Manual lanes (assignee != cli-auto) remain unchanged
and still fail closed without switching providers.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-21 05:04:55 +00:00
Hermes Agent
2862594c62 fix(monitoring): measure Atlas availability honestly across telemetry gaps
The 2026-08-18 metrics-storage outage exposed two defects in the
availability pipeline that distorted the figure in opposite directions at
once.

The Overview panel fell back to a live one-hour Traefik ratio whenever the
yearly rollup sample went stale for 48h, and rendered it under the same
"365d" title. When the rollup stopped publishing on 2026-08-18 the panel
quietly swapped a 365-day measurement for a 60-minute one and read 99.74%
instead of the recorded 99.95%. The fallback is removed: a stale rollup now
renders no value, and a new atlas-availability-rollup-stale alert pages at
26h, well before the panel goes blank at 48h.

The yearly ratio also silently excluded the 34-hour telemetry gap, because
missing days contribute zero requests and zero failures. Absent data was
read as "nothing happened" — had Atlas genuinely been down in that window,
the figure would still have said 99.95%. Availability keeps its
measured-days-only definition, which is correct, but coverage is now
published alongside it and shown in a new panel, so a telemetry gap lowers
disclosed coverage instead of vanishing. The title reads "365d window" to
stop implying 365 days of data exist; request-v4 begins 2026-05-01.

The rollup job reported healthy runs across a day and a half of lost
publishes: a read-only VictoriaMetrics accepts an import and discards it.
It now reads each sample back and fails loudly when the write did not
survive.

Not addressed here: availability is still measured from inside the platform
via Traefik counters, so it cannot distinguish "Atlas down" from "telemetry
down", and misses failures that never reach Traefik (DNS, TLS, node dead).
An external synthetic prober is the real fix and needs a hosting decision.
2026-08-20 02:03:43 +00:00
955 changed files with 134452 additions and 4840 deletions

25
Jenkinsfile vendored
View File

@ -2,11 +2,16 @@
pipeline {
agent {
kubernetes {
// Keep build I/O off the node's runtime USB drive.
workspaceVolume dynamicPVC(accessModes: 'ReadWriteOnce', requestsSize: '20Gi', storageClassName: 'ci-scratch')
defaultContainer 'python'
yaml """
apiVersion: v1
kind: Pod
spec:
securityContext:
fsGroup: 1000
fsGroupChangePolicy: OnRootMismatch
nodeSelector:
kubernetes.io/arch: arm64
node-role.kubernetes.io/worker: "true"
@ -15,12 +20,9 @@ spec:
requiredDuringSchedulingIgnoredDuringExecution:
nodeSelectorTerms:
- matchExpressions:
- key: kubernetes.io/hostname
operator: NotIn
values:
- titan-04
- titan-06
- titan-11
- {key: hardware, operator: In, values: [rpi5, rpi4]}
- {key: node-role.kubernetes.io/worker, operator: In, values: ["true"]}
- {key: kubernetes.io/hostname, operator: NotIn, values: [titan-04, titan-05, titan-06, titan-08, titan-11, titan-12, titan-13, titan-14, titan-15, titan-17, titan-18, titan-19]}
preferredDuringSchedulingIgnoredDuringExecution:
- weight: 100
preference:
@ -65,6 +67,10 @@ spec:
}
}
environment {
PIP_CACHE_DIR = '/home/jenkins/agent/.cache/pip'
NPM_CONFIG_CACHE = '/home/jenkins/agent/.cache/npm'
SONAR_USER_HOME = '/home/jenkins/agent/.cache/sonar'
TMPDIR = '/home/jenkins/agent/.cache/tmp'
PIP_DISABLE_PIP_VERSION_CHECK = '1'
PYTHONUNBUFFERED = '1'
SUITE_NAME = 'titan_iac'
@ -86,6 +92,11 @@ spec:
buildDiscarder(logRotator(daysToKeepStr: '30', numToKeepStr: '200', artifactDaysToKeepStr: '30', artifactNumToKeepStr: '120'))
}
stages {
stage('Prepare build scratch') {
steps {
sh 'mkdir -p /home/jenkins/agent/.cache/tmp /home/jenkins/agent/.cache/go-tmp'
}
}
stage('Checkout') {
steps {
checkout scm
@ -480,7 +491,7 @@ PY
set +x
git config user.email "jenkins@bstein.dev"
git config user.name "jenkins"
git remote set-url origin https://${GIT_USER}:${GIT_TOKEN}@scm.bstein.dev/atlas/titan-iac.git
git remote set-url origin https://${GIT_USER}:${GIT_TOKEN}@scm.bstein.dev/titan/atlas-iac.git
git push origin HEAD:${FLUX_BRANCH}
'''
}

View File

@ -1,9 +1,9 @@
# titan-iac
# atlas-iac
Flux-managed Kubernetes desired-state config for `bstein.dev`.
Canonical source URL:
- `ssh://git@scm.bstein.dev:2242/atlas/titan-iac.git`
- `ssh://git@scm.bstein.dev:2242/titan/atlas-iac.git`
## Scope
@ -15,3 +15,69 @@ This repo contains cluster configuration consumed by Flux:
## Apply model
I use Git + Flux as the source of truth and avoid manual in-cluster edits for durable changes.
## Finding the configuration
Start with the reference chain, not a search for every file mentioning an app:
1. `clusters/atlas/flux-system/kustomization.yaml` includes the platform and application Flux definitions.
2. A definition under `clusters/atlas/flux-system/platform/` or `applications/` names the folder Flux reconciles in `spec.path`.
3. That folder's `kustomization.yaml` lists the resources and patches actually included. A file elsewhere in the repository is not automatically deployed.
4. A `HelmRelease` selects a chart and its values; a `Deployment` or `StatefulSet` defines the workload directly.
For example, Grafana and metrics configuration starts at
`clusters/atlas/flux-system/platform/monitoring/kustomization.yaml`, which points
to `services/monitoring/`. Dashboard source is generated by
`scripts/render/dashboards_render_atlas.py`; edit the generator when changing
generated dashboards.
| Location | Purpose |
|---|---|
| `infrastructure/` | Shared platform components such as networking, storage and PostgreSQL |
| `services/<name>/` | Application resources, settings and any service-specific `NOTES.md` |
| `services/maintenance/` | In-cluster maintenance tool configuration, including Soteria, Metis and Ariadne |
| `scripts/ops/` | Operator commands; inspect the script and its documented options before use |
| `dockerfiles/` | Custom image definitions |
Ananke also runs outside Kubernetes on hosts. Its deployed host configuration
and version must be checked separately; this repository alone is not yet a
complete description of every host-side setting.
## First checks when something is wrong
Run these read-only commands on the existing administrative host:
```bash
kubectl get nodes -o wide
kubectl get deployments,statefulsets -A
kubectl get kustomizations.kustomize.toolkit.fluxcd.io -A
kubectl get helmreleases.helm.toolkit.fluxcd.io -A
```
For one affected service, inspect its namespace and warning events:
```bash
NS=monitoring
kubectl -n "$NS" get pods -o wide
kubectl -n "$NS" get events --field-selector type=Warning --sort-by=.lastTimestamp
kubectl -n "$NS" get pvc
```
`Pending` with scheduling errors usually points to capacity or placement;
`FailedMount` points to storage; `Running` without readiness points to the
application, its probe or a dependency. Use the specific evidence before
choosing a repair. Avoid starting with a cluster-wide restart or recovery script.
Start with [Cluster operations](docs/CLUSTER_OPERATIONS.md) for the daily checks,
change workflow and recovery ownership. The
[implementation record](docs/CLUSTER_STABILIZATION.md) distinguishes verified
repairs from remaining faults. A green Flux/Helm status alone is not proof of
application health or restore readiness.
## Operating principle
Ordinary Kubernetes and component configuration should handle normal operation.
Homegrown recovery tools should cover specific demonstrated gaps. Core cluster
operation and recovery must not require an AI assistant or a model-backed
decision. A procedure should explain what it changes, how to check success and
how to recover if it fails; conversation history is not operational documentation.

View File

@ -28,6 +28,7 @@ spec:
operator: NotIn
values:
- titan-04
- titan-08
- titan-14
- titan-18
- titan-19
@ -99,9 +100,14 @@ spec:
requests:
cpu: 250m
memory: 1Gi
# The agent image build expands the SQLite and web build layers in
# its workspace. Reserve enough node-local space to keep it off the
# 25 GiB workers, whose unrequested Kaniko workspace was evicted.
ephemeral-storage: 32Gi
limits:
cpu: "2"
memory: 4Gi
ephemeral-storage: 40Gi
"""
}
}
@ -114,7 +120,7 @@ spec:
string(
name: 'EXPECTED_SOURCE_REVISION',
defaultValue: '',
description: 'Exact 40-character commit on atlas/titan-iac main.'
description: 'Full reviewed commit already contained by titan/atlas-iac main.'
)
string(
name: 'CONFIRM_PUBLISH',
@ -152,9 +158,11 @@ spec:
;;
esac
test "${#EXPECTED_SOURCE_REVISION}" -eq 40
main_revision="$(git rev-parse origin/main)"
git merge-base --is-ancestor "${EXPECTED_SOURCE_REVISION}" "${main_revision}"
git checkout --detach "${EXPECTED_SOURCE_REVISION}"
actual_revision="$(git rev-parse HEAD)"
test "${actual_revision}" = "${EXPECTED_SOURCE_REVISION}"
test "${actual_revision}" = "$(git rev-parse origin/main)"
test -z "$(git status --porcelain)"
test -f dockerfiles/Dockerfile.hermes-agent
case "${BUILD_NUMBER}" in
@ -166,10 +174,40 @@ spec:
printf '%s\n' \
"${HERMES_IMAGE}:git-${actual_revision}-build-${BUILD_NUMBER}" \
> build/hermes-agent.destination
printf '%s\n' "${actual_revision}" > build/hermes-agent.source-revision
'''
}
}
}
stage('Validate reviewed release source') {
steps {
container('python') {
sh '''
set -eu
# Install the pinned test deps fully offline from the reviewed,
# in-repo wheelhouse (ci/vendor/hermes-agent-test-wheels). --no-index
# forbids any network index, so this stage never resolves
# pypi.org/files.pythonhosted.org and cannot fail on public-internet
# DNS. The wheels match this pipeline's arm64 python:3.12 container
# (PyYAML is the cp312 manylinux aarch64 build; the rest are
# py3-none-any). Bump the wheelhouse when these pins change.
python3 -m pip install --disable-pip-version-check --no-cache-dir \
--no-index --find-links="${WORKSPACE}/ci/vendor/hermes-agent-test-wheels" \
--target=/tmp/hermes-agent-release-test-deps \
pytest==8.3.4 PyYAML==6.0.2
PYTHONPATH=/tmp/hermes-agent-release-test-deps \
python3 -m pytest -q \
testing/tests/test_hermes_image_builder.py \
testing/tests/test_hermes_image_builder_adversarial.py \
testing/tests/test_hermes_image_builder_coverage.py \
testing/tests/test_hermes_image_builder_fresh_review.py \
testing/tests/test_hermes_oci_promote.py \
testing/tests/test_hermes_multiarch_combine.py \
testing/tests/test_hermes_image_automation.py
'''
}
}
}
stage('Reject replay before publish') {
steps {
withCredentials([usernamePassword(
@ -181,17 +219,33 @@ spec:
set -eu
set +x
destination="$(cat build/hermes-agent.destination)"
source_revision="$(cat build/hermes-agent.source-revision)"
python3 ci/scripts/hermes_image_release.py assert-absent \
--source-revision "${EXPECTED_SOURCE_REVISION}" \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}" \
--destination "${destination}"
'''
}
}
}
stage('Build and publish without a daemon') {
stage('Build arm64 leg without a daemon') {
steps {
container('kaniko') {
sh '''#!/busybox/sh
set -eu
minimum_available_kib=16777216
available_kib="$(/busybox/df -Pk / | /busybox/awk 'NR == 2 { print $4 }')"
case "${available_kib}" in
''|*[!0-9]*)
echo "cannot determine Kaniko ephemeral-storage availability" >&2
exit 2
;;
esac
if [ "${available_kib}" -lt "${minimum_available_kib}" ]; then
echo "Kaniko requires at least 16 GiB of free ephemeral storage" >&2
exit 1
fi
'''
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
@ -201,7 +255,8 @@ spec:
set -eu
set +x
config_path=/kaniko/.docker/config.json
destination="$(cat build/hermes-agent.destination)"
destination="$(cat build/hermes-agent.destination)-arm64"
source_revision="$(cat build/hermes-agent.source-revision)"
umask 077
auth="$(printf '%s:%s' "${HARBOR_USER}" "${HARBOR_PASSWORD}" | /busybox/base64 | /busybox/tr -d '\n')"
/busybox/mkdir -p /kaniko/.docker
@ -215,18 +270,200 @@ spec:
--context="dir://${WORKSPACE}" \
--dockerfile="${WORKSPACE}/dockerfiles/Dockerfile.hermes-agent" \
--destination="${destination}" \
--digest-file="${WORKSPACE}/build/hermes-agent.digest" \
--image-name-tag-with-digest-file="${WORKSPACE}/build/hermes-agent.image" \
--digest-file="${WORKSPACE}/build/hermes-agent-arm64.digest" \
--image-name-tag-with-digest-file="${WORKSPACE}/build/hermes-agent-arm64.image" \
--build-arg=HERMES_KANIKO_HEREDOC_COMPAT=1 \
--label="org.opencontainers.image.revision=${EXPECTED_SOURCE_REVISION}" \
--label="org.opencontainers.image.revision=${source_revision}" \
--cleanup \
--push-retry=3
/busybox/chmod 644 build/hermes-agent.digest build/hermes-agent.image
/busybox/chmod 644 build/hermes-agent-arm64.digest build/hermes-agent-arm64.image
'''
}
}
}
}
stage('Build amd64 leg without a daemon') {
agent {
kubernetes {
yaml """
apiVersion: v1
kind: Pod
metadata:
labels:
atlas.bstein.dev/workload: hermes-agent-image-builder-amd64
spec:
serviceAccountName: hermes-image-builder
automountServiceAccountToken: false
enableServiceLinks: false
restartPolicy: Never
securityContext:
fsGroup: 1000
fsGroupChangePolicy: OnRootMismatch
# titan-24 is an accelerator node (not a general worker) that co-hosts the
# out-of-cluster Sui validator. Pin the disposable amd64 build to it by
# hostname + arch ONLY — do NOT require node-role worker, so titan-24 is never
# opened to general cluster scheduling. The toleration + tight caps below keep
# this off the validator's back.
nodeSelector:
kubernetes.io/arch: amd64
kubernetes.io/hostname: titan-24
tolerations:
# titan-24 co-hosts the out-of-cluster Sui validator; tolerate whatever
# PreferNoSchedule/NoSchedule guard taint the node carries so the pinned
# build lands, and rely on the tight resource caps below (not scheduling
# priority) to keep the disposable build from starving the validator.
- operator: Exists
imagePullSecrets:
- name: harbor-bstein-robot
containers:
- name: jnlp
image: jenkins/inbound-agent@sha256:8eda4fe2a66bcf6a5e43436d9918fc14c306204dc8fcd75f4e15e0e6e5dc759a
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
runAsNonRoot: true
runAsUser: 1000
seccompProfile:
type: RuntimeDefault
resources:
requests:
cpu: 25m
memory: 128Mi
limits:
cpu: 250m
memory: 384Mi
- name: kaniko
image: gcr.io/kaniko-project/executor@sha256:c3109d5926a997b100c4343944e06c6b30a6804b2f9abe0994d3de6ef92b028e
command: ["/busybox/sh", "-c"]
args: ["/busybox/sleep 99d"]
tty: true
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
add: ["CHOWN", "FOWNER", "DAC_OVERRIDE", "SETGID", "SETUID"]
privileged: false
runAsUser: 0
seccompProfile:
type: RuntimeDefault
resources:
requests:
cpu: 100m
memory: 512Mi
limits:
cpu: "1500m"
memory: 3Gi
"""
}
}
steps {
// This amd64 leg runs on its own fresh pod (titan-24), so it must check
// out the SCM itself before the reviewed-revision git boundary check —
// otherwise `git rev-parse origin/main` fails with "not a git repository".
checkout scm
container('jnlp') {
sh '''
set -eu
mkdir -p build
test "${PUBLISH_IMAGE}" = "true"
test "${CONFIRM_PUBLISH}" = "PUBLISH HERMES AGENT"
case "${EXPECTED_SOURCE_REVISION}" in
*[!0-9a-f]*|'')
echo "EXPECTED_SOURCE_REVISION must be a lowercase full commit" >&2
exit 2
;;
esac
test "${#EXPECTED_SOURCE_REVISION}" -eq 40
main_revision="$(git rev-parse origin/main)"
git merge-base --is-ancestor "${EXPECTED_SOURCE_REVISION}" "${main_revision}"
git checkout --detach "${EXPECTED_SOURCE_REVISION}"
actual_revision="$(git rev-parse HEAD)"
test "${actual_revision}" = "${EXPECTED_SOURCE_REVISION}"
test -z "$(git status --porcelain)"
test -f dockerfiles/Dockerfile.hermes-agent
case "${BUILD_NUMBER}" in
''|0*|*[!0-9]*)
echo "BUILD_NUMBER must be a positive decimal integer" >&2
exit 2
;;
esac
printf '%s\n' \
"${HERMES_IMAGE}:git-${actual_revision}-build-${BUILD_NUMBER}" \
> build/hermes-agent.destination
printf '%s\n' "${actual_revision}" > build/hermes-agent.source-revision
'''
}
container('kaniko') {
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
passwordVariable: 'HARBOR_PASSWORD'
)]) {
sh '''#!/busybox/sh
set -eu
set +x
config_path=/kaniko/.docker/config.json
destination="$(cat build/hermes-agent.destination)-amd64"
source_revision="$(cat build/hermes-agent.source-revision)"
umask 077
auth="$(printf '%s:%s' "${HARBOR_USER}" "${HARBOR_PASSWORD}" | /busybox/base64 | /busybox/tr -d '\n')"
/busybox/mkdir -p /kaniko/.docker
/busybox/printf '{"auths":{"registry.bstein.dev":{"auth":"%s"}}}\n' "${auth}" > "${config_path}"
unset HARBOR_USER HARBOR_PASSWORD auth
trap '/busybox/rm -f "${config_path}"' EXIT HUP INT TERM
umask 022
/kaniko/executor \
--registry-mirror=harbor-core.harbor.svc.cluster.local \
--insecure-registry=harbor-core.harbor.svc.cluster.local \
--context="dir://${WORKSPACE}" \
--dockerfile="${WORKSPACE}/dockerfiles/Dockerfile.hermes-agent" \
--destination="${destination}" \
--digest-file="${WORKSPACE}/build/hermes-agent-amd64.digest" \
--image-name-tag-with-digest-file="${WORKSPACE}/build/hermes-agent-amd64.image" \
--build-arg=HERMES_KANIKO_HEREDOC_COMPAT=1 \
--label="org.opencontainers.image.revision=${source_revision}" \
--cleanup \
--push-retry=3
/busybox/chmod 644 build/hermes-agent-amd64.digest build/hermes-agent-amd64.image
'''
}
}
stash(
name: 'hermes-agent-amd64-evidence',
includes: 'build/hermes-agent-amd64.digest,build/hermes-agent-amd64.image'
)
}
}
stage('Combine multi-arch index') {
steps {
unstash 'hermes-agent-amd64-evidence'
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
passwordVariable: 'HARBOR_PASSWORD'
)]) {
sh '''
set -eu
set +x
destination="$(cat build/hermes-agent.destination)"
source_revision="$(cat build/hermes-agent.source-revision)"
python3 ci/scripts/hermes_multiarch_combine.py \
--destination "${destination}" \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}" \
--arm64-digest-file build/hermes-agent-arm64.digest \
--arm64-image-file build/hermes-agent-arm64.image \
--amd64-digest-file build/hermes-agent-amd64.digest \
--amd64-image-file build/hermes-agent-amd64.image \
--digest-file build/hermes-agent.digest \
--image-file build/hermes-agent.image
test -s build/hermes-agent.digest
test -s build/hermes-agent.image
'''
}
}
}
stage('Render reviewed Flux handoff') {
steps {
withCredentials([usernamePassword(
@ -238,10 +475,11 @@ spec:
set -eu
set +x
destination="$(cat build/hermes-agent.destination)"
source_revision="$(cat build/hermes-agent.source-revision)"
python3 ci/scripts/hermes_image_release.py render \
--digest-file build/hermes-agent.digest \
--image-file build/hermes-agent.image \
--source-revision "${EXPECTED_SOURCE_REVISION}" \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}" \
--destination "${destination}" \
--kustomization services/hermes/kustomization.yaml \
@ -252,15 +490,19 @@ spec:
}
}
}
}
post {
success {
stage('Verify and archive release evidence') {
steps {
sh '''
set -eu
expected_files="$(printf '%s\n' \
build/hermes-agent.destination \
build/hermes-agent.digest \
build/hermes-agent.image \
build/hermes-agent.source-revision \
build/hermes-agent-arm64.digest \
build/hermes-agent-arm64.image \
build/hermes-agent-amd64.digest \
build/hermes-agent-amd64.image \
build/hermes-agent-release/hermes-agent-image.json \
build/hermes-agent-release/hermes-image-update.patch \
build/hermes-agent-release/hermes-kustomization.yaml \
@ -268,26 +510,42 @@ spec:
actual_files="$(find build -type f -print | LC_ALL=C sort)"
test "${actual_files}" = "${expected_files}"
destination="$(cat build/hermes-agent.destination)"
source_revision="$(cat build/hermes-agent.source-revision)"
python3 ci/scripts/hermes_image_release.py verify-evidence \
--digest-file build/hermes-agent.digest \
--image-file build/hermes-agent.image \
--source-revision "${EXPECTED_SOURCE_REVISION}" \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}" \
--destination "${destination}" \
--kustomization services/hermes/kustomization.yaml \
--output-dir build/hermes-agent-release
'''
archiveArtifacts(
artifacts: 'build/hermes-agent.destination,build/hermes-agent.digest,build/hermes-agent.image,build/hermes-agent-release/hermes-agent-image.json,build/hermes-agent-release/hermes-image-update.patch,build/hermes-agent-release/hermes-kustomization.yaml',
artifacts: 'build/hermes-agent.destination,build/hermes-agent.digest,build/hermes-agent.image,build/hermes-agent.source-revision,build/hermes-agent-arm64.digest,build/hermes-agent-arm64.image,build/hermes-agent-amd64.digest,build/hermes-agent-amd64.image,build/hermes-agent-release/hermes-agent-image.json,build/hermes-agent-release/hermes-image-update.patch,build/hermes-agent-release/hermes-kustomization.yaml',
allowEmptyArchive: false,
fingerprint: true
)
}
}
cleanup {
container('kaniko') {
sh '''#!/busybox/sh
/busybox/rm -f /kaniko/.docker/config.json
'''
stage('Publish Flux release tag') {
steps {
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
passwordVariable: 'HARBOR_PASSWORD'
)]) {
sh '''
set -eu
set +x
destination="$(cat build/hermes-agent.destination)"
source_revision="$(cat build/hermes-agent.source-revision)"
python3 ci/scripts/hermes_oci_promote.py \
--destination "${destination}" \
--digest-file build/hermes-agent.digest \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}"
'''
}
}
}
}

View File

@ -0,0 +1,339 @@
pipeline {
agent {
kubernetes {
defaultContainer 'python'
yaml """
apiVersion: v1
kind: Pod
metadata:
labels:
atlas.bstein.dev/workload: hermes-chat-router-image-builder
spec:
serviceAccountName: hermes-image-builder
automountServiceAccountToken: false
enableServiceLinks: false
restartPolicy: Never
securityContext:
fsGroup: 1000
fsGroupChangePolicy: OnRootMismatch
nodeSelector:
kubernetes.io/arch: arm64
affinity:
nodeAffinity:
requiredDuringSchedulingIgnoredDuringExecution:
nodeSelectorTerms:
- matchExpressions:
- key: kubernetes.io/hostname
operator: In
values: [titan-20]
imagePullSecrets:
- name: harbor-bstein-robot
containers:
- name: jnlp
image: jenkins/inbound-agent@sha256:8eda4fe2a66bcf6a5e43436d9918fc14c306204dc8fcd75f4e15e0e6e5dc759a
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
runAsNonRoot: true
runAsUser: 1000
seccompProfile:
type: RuntimeDefault
resources:
requests: {cpu: 25m, memory: 128Mi}
limits: {cpu: 500m, memory: 512Mi}
- name: python
image: registry.bstein.dev/bstein/python@sha256:269541d3387baae008df4608ead893dba2b5cdaad1a5a380731a88992d34b808
command: ["sleep"]
args: ["99d"]
tty: true
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
runAsNonRoot: true
runAsUser: 1000
seccompProfile:
type: RuntimeDefault
resources:
requests: {cpu: 25m, memory: 64Mi}
limits: {cpu: 250m, memory: 256Mi}
- name: golang
image: golang:1.24-alpine@sha256:8bee1901f1e530bfb4a7850aa7a479d17ae3a18beb6e09064ed54cfd245b7191
command: ["sleep"]
args: ["99d"]
tty: true
env:
- {name: GOCACHE, value: /tmp/go-cache}
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
runAsNonRoot: true
runAsUser: 1000
seccompProfile:
type: RuntimeDefault
resources:
requests: {cpu: 100m, memory: 128Mi}
limits: {cpu: "1", memory: 1Gi}
- name: kaniko
image: gcr.io/kaniko-project/executor@sha256:c3109d5926a997b100c4343944e06c6b30a6804b2f9abe0994d3de6ef92b028e
command: ["/busybox/sh", "-c"]
args: ["/busybox/sleep 99d"]
tty: true
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
add: ["CHOWN", "FOWNER", "DAC_OVERRIDE", "SETGID", "SETUID"]
privileged: false
runAsUser: 0
seccompProfile:
type: RuntimeDefault
resources:
requests:
cpu: 100m
memory: 256Mi
ephemeral-storage: 2Gi
limits:
cpu: "1"
memory: 2Gi
ephemeral-storage: 5Gi
"""
}
}
parameters {
booleanParam(
name: 'PUBLISH_IMAGE',
defaultValue: false,
description: 'Publish the reviewed main revision to Harbor.'
)
string(
name: 'EXPECTED_SOURCE_REVISION',
defaultValue: '',
description: 'Full reviewed commit that must be contained by titan/atlas-iac main.'
)
string(
name: 'CONFIRM_PUBLISH',
defaultValue: '',
description: 'Enter PUBLISH HERMES CHAT ROUTER to confirm the release.'
)
}
environment {
HERMES_IMAGE = 'registry.bstein.dev/bstein/hermes-chat-router'
}
options {
disableConcurrentBuilds()
buildDiscarder(logRotator(daysToKeepStr: '30', numToKeepStr: '100', artifactDaysToKeepStr: '30', artifactNumToKeepStr: '100'))
skipDefaultCheckout(true)
timeout(time: 45, unit: 'MINUTES')
}
stages {
stage('Checkout reviewed source') {
steps {
checkout scm
}
}
stage('Enforce release boundary') {
steps {
container('jnlp') {
sh '''
set -eu
mkdir -p build
test "${PUBLISH_IMAGE}" = "true"
test "${CONFIRM_PUBLISH}" = "PUBLISH HERMES CHAT ROUTER"
case "${EXPECTED_SOURCE_REVISION}" in
*[!0-9a-f]*|'')
echo "EXPECTED_SOURCE_REVISION must be a lowercase full commit" >&2
exit 2
;;
esac
test "${#EXPECTED_SOURCE_REVISION}" -eq 40
main_revision="$(git rev-parse HEAD)"
test "${main_revision}" = "$(git rev-parse origin/main)"
git merge-base --is-ancestor "${EXPECTED_SOURCE_REVISION}" "${main_revision}"
git checkout --detach "${EXPECTED_SOURCE_REVISION}"
actual_revision="$(git rev-parse HEAD)"
test "${actual_revision}" = "${EXPECTED_SOURCE_REVISION}"
test -z "$(git status --porcelain)"
test -f dockerfiles/Dockerfile.hermes-chat-router
case "${BUILD_NUMBER}" in
''|0*|*[!0-9]*)
echo "BUILD_NUMBER must be a positive decimal integer" >&2
exit 2
;;
esac
printf '%s\n' \
"${HERMES_IMAGE}:git-${actual_revision}-build-${BUILD_NUMBER}" \
> build/hermes-chat-router.destination
printf '%s\n' "${actual_revision}" \
> build/hermes-chat-router.source-revision
'''
}
}
}
stage('Validate reviewed router source') {
steps {
container('python') {
sh '''
set -eu
python3 -m pip install --disable-pip-version-check --no-cache-dir \
--target=/tmp/hermes-chat-router-test-deps \
pytest==8.3.4 PyYAML==6.0.2
PYTHONPATH=/tmp/hermes-chat-router-test-deps \
python3 -m pytest -q \
testing/tests/test_hermes_chat_router_release.py \
testing/tests/test_hermes_oci_promote.py \
testing/tests/test_hermes_image_automation.py
'''
}
container('golang') {
sh '''
set -eu
cd services/hermes/router
GO111MODULE=off CGO_ENABLED=0 go test ./...
'''
}
}
}
stage('Reject replay before publish') {
steps {
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
passwordVariable: 'HARBOR_PASSWORD'
)]) {
sh '''
set -eu
set +x
destination="$(cat build/hermes-chat-router.destination)"
source_revision="$(cat build/hermes-chat-router.source-revision)"
python3 ci/scripts/hermes_chat_router_release.py assert-absent \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}" \
--destination "${destination}"
'''
}
}
}
stage('Build and publish without a daemon') {
steps {
container('kaniko') {
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
passwordVariable: 'HARBOR_PASSWORD'
)]) {
sh '''#!/busybox/sh
set -eu
set +x
config_path=/kaniko/.docker/config.json
destination="$(cat build/hermes-chat-router.destination)"
source_revision="$(cat build/hermes-chat-router.source-revision)"
umask 077
auth="$(printf '%s:%s' "${HARBOR_USER}" "${HARBOR_PASSWORD}" | /busybox/base64 | /busybox/tr -d '\n')"
/busybox/mkdir -p /kaniko/.docker
/busybox/printf '{"auths":{"registry.bstein.dev":{"auth":"%s"}}}\n' "${auth}" > "${config_path}"
unset HARBOR_USER HARBOR_PASSWORD auth
trap '/busybox/rm -f "${config_path}"' EXIT HUP INT TERM
umask 022
/kaniko/executor \
--registry-mirror=harbor-core.harbor.svc.cluster.local \
--insecure-registry=harbor-core.harbor.svc.cluster.local \
--context="dir://${WORKSPACE}" \
--dockerfile="${WORKSPACE}/dockerfiles/Dockerfile.hermes-chat-router" \
--destination="${destination}" \
--digest-file="${WORKSPACE}/build/hermes-chat-router.digest" \
--image-name-tag-with-digest-file="${WORKSPACE}/build/hermes-chat-router.image" \
--label="org.opencontainers.image.revision=${source_revision}" \
--label="org.opencontainers.image.source=https://scm.bstein.dev/titan/atlas-iac" \
--label="org.opencontainers.image.title=hermes-chat-router" \
--cleanup \
--push-retry=3
/busybox/chmod 644 \
build/hermes-chat-router.digest build/hermes-chat-router.image
'''
}
}
}
}
stage('Render reviewed Flux handoff') {
steps {
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
passwordVariable: 'HARBOR_PASSWORD'
)]) {
sh '''
set -eu
set +x
destination="$(cat build/hermes-chat-router.destination)"
source_revision="$(cat build/hermes-chat-router.source-revision)"
python3 ci/scripts/hermes_chat_router_release.py render \
--digest-file build/hermes-chat-router.digest \
--image-file build/hermes-chat-router.image \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}" \
--destination "${destination}" \
--manifest services/hermes/chat-router.yaml \
--output-dir build/hermes-chat-router-release
'''
}
}
}
stage('Verify and archive release evidence') {
steps {
sh '''
set -eu
expected_files="$(printf '%s\n' \
build/hermes-chat-router.destination \
build/hermes-chat-router.digest \
build/hermes-chat-router.image \
build/hermes-chat-router.source-revision \
build/hermes-chat-router-release/hermes-chat-router-deployment.yaml \
build/hermes-chat-router-release/hermes-chat-router-image-update.patch \
build/hermes-chat-router-release/hermes-chat-router-image.json \
| LC_ALL=C sort)"
actual_files="$(find build -type f -print | LC_ALL=C sort)"
test "${actual_files}" = "${expected_files}"
destination="$(cat build/hermes-chat-router.destination)"
source_revision="$(cat build/hermes-chat-router.source-revision)"
python3 ci/scripts/hermes_chat_router_release.py verify-evidence \
--digest-file build/hermes-chat-router.digest \
--image-file build/hermes-chat-router.image \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}" \
--destination "${destination}" \
--manifest services/hermes/chat-router.yaml \
--output-dir build/hermes-chat-router-release
'''
archiveArtifacts(
artifacts: 'build/hermes-chat-router.destination,build/hermes-chat-router.digest,build/hermes-chat-router.image,build/hermes-chat-router.source-revision,build/hermes-chat-router-release/hermes-chat-router-deployment.yaml,build/hermes-chat-router-release/hermes-chat-router-image-update.patch,build/hermes-chat-router-release/hermes-chat-router-image.json',
allowEmptyArchive: false,
fingerprint: true
)
}
}
stage('Publish Flux release tag') {
steps {
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
passwordVariable: 'HARBOR_PASSWORD'
)]) {
sh '''
set -eu
set +x
destination="$(cat build/hermes-chat-router.destination)"
source_revision="$(cat build/hermes-chat-router.source-revision)"
python3 ci/scripts/hermes_oci_promote.py \
--destination "${destination}" \
--digest-file build/hermes-chat-router.digest \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}"
'''
}
}
}
}
}

View File

@ -0,0 +1,236 @@
pipeline {
agent {
kubernetes {
defaultContainer 'python'
yaml """
apiVersion: v1
kind: Pod
metadata:
labels:
atlas.bstein.dev/workload: hermes-voice-image-builder
spec:
serviceAccountName: hermes-image-builder
automountServiceAccountToken: false
enableServiceLinks: false
restartPolicy: Never
securityContext:
fsGroup: 1000
fsGroupChangePolicy: OnRootMismatch
nodeSelector:
kubernetes.io/arch: arm64
tolerations:
# Jetson kubelet/network maintenance can briefly outlast Kubernetes' five
# minute default. Keep the disposable build workspace intact long enough
# for the node to recover instead of restarting a large image expansion.
- key: node.kubernetes.io/not-ready
operator: Exists
effect: NoExecute
tolerationSeconds: 600
- key: node.kubernetes.io/unreachable
operator: Exists
effect: NoExecute
tolerationSeconds: 600
affinity:
nodeAffinity:
requiredDuringSchedulingIgnoredDuringExecution:
nodeSelectorTerms:
- matchExpressions:
- key: kubernetes.io/hostname
operator: In
# Voice images expand the large Jetson base and two Whisper
# models. Keep that transient I/O on the roomy idle
# accelerator, away from Longhorn and the live speech node.
values: [titan-20]
imagePullSecrets:
- name: harbor-bstein-robot
containers:
- name: jnlp
image: jenkins/inbound-agent@sha256:8eda4fe2a66bcf6a5e43436d9918fc14c306204dc8fcd75f4e15e0e6e5dc759a
securityContext:
allowPrivilegeEscalation: false
capabilities: {drop: ["ALL"]}
runAsNonRoot: true
runAsUser: 1000
seccompProfile: {type: RuntimeDefault}
resources:
requests: {cpu: 25m, memory: 256Mi}
limits: {cpu: 500m, memory: 512Mi}
- name: python
image: registry.bstein.dev/bstein/python@sha256:269541d3387baae008df4608ead893dba2b5cdaad1a5a380731a88992d34b808
command: ["sleep"]
args: ["99d"]
tty: true
securityContext:
allowPrivilegeEscalation: false
capabilities: {drop: ["ALL"]}
runAsNonRoot: true
runAsUser: 1000
seccompProfile: {type: RuntimeDefault}
resources:
requests: {cpu: 25m, memory: 64Mi}
limits: {cpu: 500m, memory: 512Mi}
- name: kaniko
image: gcr.io/kaniko-project/executor@sha256:c3109d5926a997b100c4343944e06c6b30a6804b2f9abe0994d3de6ef92b028e
command: ["/busybox/sh", "-c"]
args: ["/busybox/sleep 99d"]
tty: true
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
# The pinned Jetson Whisper base contains gst-ptp-helper with a
# security.capability xattr. Kaniko needs SETFCAP only while
# unpacking that reviewed base; the pod remains non-privileged.
add: ["CHOWN", "FOWNER", "DAC_OVERRIDE", "SETGID", "SETUID", "SETFCAP"]
runAsUser: 0
seccompProfile: {type: RuntimeDefault}
resources:
# The STT image snapshots two checksum-pinned Whisper models. A 1 GiB
# request let the builder overpack titan-20 and a 4 GiB cgroup killed
# Kaniko while it emitted the large model layer.
requests: {cpu: 250m, memory: 2Gi, ephemeral-storage: 10Gi}
limits: {cpu: "2", memory: 6Gi, ephemeral-storage: 20Gi}
"""
}
}
parameters {
booleanParam(name: 'PUBLISH_IMAGE', defaultValue: false, description: 'Publish the reviewed voice image to Harbor.')
choice(name: 'IMAGE_COMPONENT', choices: ['stt', 'tts'], description: 'Private voice component to build.')
string(name: 'EXPECTED_SOURCE_REVISION', defaultValue: '', description: 'Full reviewed commit contained by main.')
string(name: 'CONFIRM_PUBLISH', defaultValue: '', description: 'Exact component-specific confirmation.')
}
options {
disableConcurrentBuilds()
buildDiscarder(logRotator(daysToKeepStr: '30', numToKeepStr: '100', artifactDaysToKeepStr: '30', artifactNumToKeepStr: '100'))
skipDefaultCheckout(true)
timeout(time: 150, unit: 'MINUTES')
}
stages {
stage('Checkout reviewed source') {
steps { checkout scm }
}
stage('Enforce release boundary') {
steps {
container('jnlp') {
sh '''
set -eu
mkdir -p build
test "${PUBLISH_IMAGE}" = "true"
case "${IMAGE_COMPONENT}" in
stt) expected_confirmation='PUBLISH HERMES STT' ;;
tts) expected_confirmation='PUBLISH HERMES TTS' ;;
*) echo 'IMAGE_COMPONENT must be stt or tts' >&2; exit 2 ;;
esac
test "${CONFIRM_PUBLISH}" = "${expected_confirmation}"
case "${EXPECTED_SOURCE_REVISION}" in
*[!0-9a-f]*|'') echo 'EXPECTED_SOURCE_REVISION must be a lowercase full commit' >&2; exit 2 ;;
esac
test "${#EXPECTED_SOURCE_REVISION}" -eq 40
main_revision="$(git rev-parse HEAD)"
test "${main_revision}" = "$(git rev-parse origin/main)"
git merge-base --is-ancestor "${EXPECTED_SOURCE_REVISION}" "${main_revision}"
git checkout --detach "${EXPECTED_SOURCE_REVISION}"
actual_revision="$(git rev-parse HEAD)"
test "${actual_revision}" = "${EXPECTED_SOURCE_REVISION}"
test -z "$(git status --porcelain)"
case "${BUILD_NUMBER}" in ''|0*|*[!0-9]*) exit 2 ;; esac
image="registry.bstein.dev/bstein/hermes-jetson-${IMAGE_COMPONENT}"
printf '%s\n' "${image}:git-${actual_revision}-build-${BUILD_NUMBER}" > build/hermes-voice.destination
printf '%s\n' "${actual_revision}" > build/hermes-voice.source-revision
printf '%s\n' "${IMAGE_COMPONENT}" > build/hermes-voice.component
test -f "dockerfiles/Dockerfile.hermes-jetson-${IMAGE_COMPONENT}"
'''
}
}
}
stage('Validate reviewed voice source') {
steps {
container('python') {
sh '''
set -eu
python3 -m pip install --disable-pip-version-check --no-cache-dir \
--target=/tmp/hermes-voice-test-deps pytest==8.3.4 PyYAML==6.0.2
python3 -m py_compile \
dockerfiles/hermes-jetson-stt-server.py \
dockerfiles/hermes-jetson-tts-server.py \
dockerfiles/hermes_jetson_tts_cues.py
PYTHONPATH=/tmp/hermes-voice-test-deps python3 -m pytest -q \
testing/tests/test_hermes_stt_streaming.py \
testing/tests/test_hermes_stt_rolling_model.py \
testing/tests/test_hermes_tts_language_routing.py \
testing/tests/test_hermes_voice_language_routing.py \
testing/tests/test_hermes_oci_promote.py \
testing/tests/test_hermes_image_automation.py \
testing/tests/test_hermes_voice_release.py
'''
}
}
}
stage('Build and publish without a daemon') {
steps {
container('kaniko') {
withCredentials([usernamePassword(credentialsId: 'harbor-robot', usernameVariable: 'HARBOR_USER', passwordVariable: 'HARBOR_PASSWORD')]) {
sh '''#!/busybox/sh
set -eu
set +x
component="$(cat build/hermes-voice.component)"
destination="$(cat build/hermes-voice.destination)"
source_revision="$(cat build/hermes-voice.source-revision)"
config_path=/kaniko/.docker/config.json
umask 077
auth="$(printf '%s:%s' "${HARBOR_USER}" "${HARBOR_PASSWORD}" | /busybox/base64 | /busybox/tr -d '\n')"
/busybox/mkdir -p /kaniko/.docker
/busybox/printf '{"auths":{"registry.bstein.dev":{"auth":"%s"}}}\n' "${auth}" > "${config_path}"
unset HARBOR_USER HARBOR_PASSWORD auth
trap '/busybox/rm -f "${config_path}"' EXIT HUP INT TERM
umask 022
/kaniko/executor \
--registry-mirror=harbor-core.harbor.svc.cluster.local \
--insecure-registry=harbor-core.harbor.svc.cluster.local \
--compressed-caching=false \
--snapshot-mode=redo \
--context="dir://${WORKSPACE}" \
--dockerfile="${WORKSPACE}/dockerfiles/Dockerfile.hermes-jetson-${component}" \
--destination="${destination}" \
--digest-file="${WORKSPACE}/build/hermes-voice.digest" \
--image-name-tag-with-digest-file="${WORKSPACE}/build/hermes-voice.image" \
--label="org.opencontainers.image.revision=${source_revision}" \
--label="org.opencontainers.image.source=https://scm.bstein.dev/titan/atlas-iac" \
--label="org.opencontainers.image.title=hermes-jetson-${component}" \
--cleanup --push-retry=3
/busybox/chmod 644 build/hermes-voice.digest build/hermes-voice.image
'''
}
}
}
}
stage('Verify, archive, and publish Flux release') {
steps {
withCredentials([usernamePassword(credentialsId: 'harbor-robot', usernameVariable: 'HARBOR_USER', passwordVariable: 'HARBOR_PASSWORD')]) {
sh '''
set -eu
set +x
destination="$(cat build/hermes-voice.destination)"
source_revision="$(cat build/hermes-voice.source-revision)"
component="$(cat build/hermes-voice.component)"
digest="$(cat build/hermes-voice.digest)"
image="$(cat build/hermes-voice.image)"
test "${image}" = "${destination}@${digest}"
python3 ci/scripts/hermes_oci_promote.py \
--destination "${destination}" \
--digest-file build/hermes-voice.digest \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}" \
> build/hermes-voice.promotion.json
python3 -c 'import json; data=json.load(open("build/hermes-voice.promotion.json")); assert data["result"] in {"published", "already-present"}'
'''
archiveArtifacts(
artifacts: 'build/hermes-voice.component,build/hermes-voice.destination,build/hermes-voice.digest,build/hermes-voice.image,build/hermes-voice.source-revision,build/hermes-voice.promotion.json',
allowEmptyArchive: false,
fingerprint: true
)
}
}
}
}
}

View File

@ -0,0 +1,559 @@
pipeline {
agent {
kubernetes {
defaultContainer 'python'
yaml """
apiVersion: v1
kind: Pod
metadata:
labels:
atlas.bstein.dev/workload: hermes-webui-image-builder
spec:
serviceAccountName: hermes-image-builder
automountServiceAccountToken: false
enableServiceLinks: false
restartPolicy: Never
securityContext:
fsGroup: 1000
fsGroupChangePolicy: OnRootMismatch
nodeSelector:
kubernetes.io/arch: arm64
affinity:
nodeAffinity:
requiredDuringSchedulingIgnoredDuringExecution:
nodeSelectorTerms:
- matchExpressions:
- key: kubernetes.io/hostname
operator: In
# Keep disposable Kaniko expansion off the astreae Longhorn
# replica nodes. titan-20 is the roomy ARM image builder and
# has local NVMe for this transient I/O.
values:
- titan-20
imagePullSecrets:
- name: harbor-bstein-robot
containers:
- name: jnlp
image: jenkins/inbound-agent@sha256:8eda4fe2a66bcf6a5e43436d9918fc14c306204dc8fcd75f4e15e0e6e5dc759a
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
runAsNonRoot: true
runAsUser: 1000
seccompProfile:
type: RuntimeDefault
resources:
requests:
cpu: 25m
memory: 256Mi
limits:
cpu: 500m
memory: 512Mi
- name: python
image: registry.bstein.dev/bstein/python@sha256:269541d3387baae008df4608ead893dba2b5cdaad1a5a380731a88992d34b808
command: ["sleep"]
args: ["99d"]
tty: true
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
add: ["CHOWN", "FOWNER", "DAC_OVERRIDE", "SETGID", "SETUID"]
runAsNonRoot: false
runAsUser: 0
seccompProfile:
type: RuntimeDefault
resources:
requests:
cpu: 25m
memory: 64Mi
limits:
cpu: 250m
memory: 256Mi
- name: kaniko
image: gcr.io/kaniko-project/executor@sha256:c3109d5926a997b100c4343944e06c6b30a6804b2f9abe0994d3de6ef92b028e
command: ["/busybox/sh", "-c"]
args: ["/busybox/sleep 99d"]
tty: true
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
add: ["CHOWN", "FOWNER", "DAC_OVERRIDE", "SETGID", "SETUID"]
privileged: false
runAsUser: 0
seccompProfile:
type: RuntimeDefault
resources:
requests:
cpu: 250m
memory: 1Gi
ephemeral-storage: 10Gi
limits:
cpu: "2"
memory: 4Gi
ephemeral-storage: 20Gi
"""
}
}
parameters {
booleanParam(
name: 'PUBLISH_IMAGE',
defaultValue: false,
description: 'Publish the reviewed main revision to Harbor.'
)
string(
name: 'EXPECTED_SOURCE_REVISION',
defaultValue: '',
description: 'Full reviewed commit that must be contained by titan/atlas-iac main.'
)
string(
name: 'CONFIRM_PUBLISH',
defaultValue: '',
description: 'Enter PUBLISH HERMES WEBUI to confirm the release.'
)
}
environment {
HERMES_IMAGE = 'registry.bstein.dev/bstein/hermes-webui'
}
options {
disableConcurrentBuilds()
buildDiscarder(logRotator(daysToKeepStr: '30', numToKeepStr: '100', artifactDaysToKeepStr: '30', artifactNumToKeepStr: '100'))
skipDefaultCheckout(true)
timeout(time: 150, unit: 'MINUTES')
}
stages {
stage('Checkout reviewed source') {
steps {
checkout scm
}
}
stage('Enforce release boundary') {
steps {
container('jnlp') {
sh '''
set -eu
mkdir -p build
test "${PUBLISH_IMAGE}" = "true"
test "${CONFIRM_PUBLISH}" = "PUBLISH HERMES WEBUI"
case "${EXPECTED_SOURCE_REVISION}" in
*[!0-9a-f]*|'')
echo "EXPECTED_SOURCE_REVISION must be a lowercase full commit" >&2
exit 2
;;
esac
test "${#EXPECTED_SOURCE_REVISION}" -eq 40
main_revision="$(git rev-parse HEAD)"
test "${main_revision}" = "$(git rev-parse origin/main)"
git merge-base --is-ancestor "${EXPECTED_SOURCE_REVISION}" "${main_revision}"
git checkout --detach "${EXPECTED_SOURCE_REVISION}"
actual_revision="$(git rev-parse HEAD)"
test "${actual_revision}" = "${EXPECTED_SOURCE_REVISION}"
test -z "$(git status --porcelain)"
test -f dockerfiles/Dockerfile.hermes-webui
case "${BUILD_NUMBER}" in
''|0*|*[!0-9]*)
echo "BUILD_NUMBER must be a positive decimal integer" >&2
exit 2
;;
esac
printf '%s\n' \
"${HERMES_IMAGE}:git-${actual_revision}-build-${BUILD_NUMBER}" \
> build/hermes-webui.destination
printf '%s\n' "${actual_revision}" > build/hermes-webui.source-revision
'''
}
}
}
stage('Validate reviewed WebUI source') {
steps {
container('python') {
sh '''
set -eu
export DEBIAN_FRONTEND=noninteractive
apt-get update
apt-get install -y --no-install-recommends ffmpeg nodejs
rm -rf /var/lib/apt/lists/*
command -v ffmpeg >/dev/null
command -v node >/dev/null
python3 -m pip install --disable-pip-version-check --no-cache-dir \
--target=/tmp/hermes-webui-release-test-deps \
pytest==8.3.4 PyYAML==6.0.2
PYTHONPATH=/tmp/hermes-webui-release-test-deps \
python3 -m pytest -q \
testing/tests/test_hermes_chat_quality.py \
testing/tests/test_hermes_handsfree_stt.py \
testing/tests/test_hermes_voice_instrument.py \
testing/tests/test_hermes_voice_full_duplex.py \
testing/tests/test_hermes_voice_route_preflight.py \
testing/tests/test_hermes_voice_preflight_delivery.py \
testing/tests/test_hermes_thinking_voice_cues.py \
testing/tests/test_hermes_voice_language_routing.py \
testing/tests/test_hermes_webui_brand.py \
testing/tests/test_hermes_webui_release.py \
testing/tests/test_hermes_webui_hux_bff.py \
testing/tests/test_hermes_webui_hux_context.py \
testing/tests/test_hermes_webui_hux_integration.py \
testing/tests/test_hermes_webui_hux_backend_e2e.py \
testing/tests/test_hermes_hux_ui_runtime_wave_a.py \
testing/tests/test_hermes_hux_runtime_autonomy_privacy.py \
testing/tests/test_hermes_hux_runtime_stop.py \
testing/tests/test_hermes_hux_ui_runtime_wave_b.py \
testing/tests/test_hermes_hux_runtime_wave_c.py \
testing/tests/test_hermes_hux_runtime_plugin.py \
testing/tests/test_hermes_hux_runtime_vendor_parity.py \
testing/tests/test_hermes_hux_delivery.py \
testing/tests/test_hermes_oci_promote.py \
testing/tests/test_hermes_multiarch_combine.py \
testing/tests/test_hermes_image_automation.py
HUX_BACKEND_TESTS="$(find testing/tests -maxdepth 1 -type f \
-name 'test_hermes_hux_*.py' \
! -name '*_ui_*' \
! -name '*runtime*' \
! -name '*delivery*' \
| sort)"
test -n "${HUX_BACKEND_TESTS}"
PYTHONPATH=/tmp/hermes-webui-release-test-deps \
python3 -m pytest -q ${HUX_BACKEND_TESTS}
'''
}
}
}
stage('Reject replay before publish') {
steps {
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
passwordVariable: 'HARBOR_PASSWORD'
)]) {
sh '''
set -eu
set +x
destination="$(cat build/hermes-webui.destination)"
source_revision="$(cat build/hermes-webui.source-revision)"
python3 ci/scripts/hermes_webui_release.py assert-absent \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}" \
--destination "${destination}"
'''
}
}
}
stage('Build arm64 leg without a daemon') {
steps {
container('kaniko') {
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
passwordVariable: 'HARBOR_PASSWORD'
)]) {
sh '''#!/busybox/sh
set -eu
set +x
config_path=/kaniko/.docker/config.json
destination="$(cat build/hermes-webui.destination)-arm64"
source_revision="$(cat build/hermes-webui.source-revision)"
umask 077
auth="$(printf '%s:%s' "${HARBOR_USER}" "${HARBOR_PASSWORD}" | /busybox/base64 | /busybox/tr -d '\n')"
/busybox/mkdir -p /kaniko/.docker
/busybox/printf '{"auths":{"registry.bstein.dev":{"auth":"%s"}}}\n' "${auth}" > "${config_path}"
unset HARBOR_USER HARBOR_PASSWORD auth
trap '/busybox/rm -f "${config_path}"' EXIT HUP INT TERM
umask 022
/kaniko/executor \
--registry-mirror=harbor-core.harbor.svc.cluster.local \
--insecure-registry=harbor-core.harbor.svc.cluster.local \
--context="dir://${WORKSPACE}" \
--dockerfile="${WORKSPACE}/dockerfiles/Dockerfile.hermes-webui" \
--destination="${destination}" \
--build-arg="HERMES_WEBUI_RELEASE_ID=git-${source_revision}-build-${BUILD_NUMBER}" \
--digest-file="${WORKSPACE}/build/hermes-webui-arm64.digest" \
--image-name-tag-with-digest-file="${WORKSPACE}/build/hermes-webui-arm64.image" \
--label="org.opencontainers.image.revision=${source_revision}" \
--label="org.opencontainers.image.source=https://scm.bstein.dev/titan/atlas-iac" \
--label="org.opencontainers.image.title=hermes-webui" \
--cleanup \
--push-retry=3
/busybox/chmod 644 build/hermes-webui-arm64.digest build/hermes-webui-arm64.image
'''
}
}
}
}
stage('Build amd64 leg without a daemon') {
agent {
kubernetes {
yaml """
apiVersion: v1
kind: Pod
metadata:
labels:
atlas.bstein.dev/workload: hermes-webui-image-builder-amd64
spec:
serviceAccountName: hermes-image-builder
automountServiceAccountToken: false
enableServiceLinks: false
restartPolicy: Never
securityContext:
fsGroup: 1000
fsGroupChangePolicy: OnRootMismatch
# titan-24 is an accelerator node (not a general worker) that co-hosts the
# out-of-cluster Sui validator. Pin the disposable amd64 build to it by
# hostname + arch ONLY — do NOT require node-role worker, so titan-24 is never
# opened to general cluster scheduling. The toleration + tight caps below keep
# this off the validator's back.
nodeSelector:
kubernetes.io/arch: amd64
kubernetes.io/hostname: titan-24
tolerations:
# titan-24 co-hosts the out-of-cluster Sui validator; tolerate whatever
# PreferNoSchedule/NoSchedule guard taint the node carries so the pinned
# build lands, and rely on the tight resource caps below (not scheduling
# priority) to keep the disposable build from starving the validator.
- operator: Exists
imagePullSecrets:
- name: harbor-bstein-robot
containers:
- name: jnlp
image: jenkins/inbound-agent@sha256:8eda4fe2a66bcf6a5e43436d9918fc14c306204dc8fcd75f4e15e0e6e5dc759a
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
runAsNonRoot: true
runAsUser: 1000
seccompProfile:
type: RuntimeDefault
resources:
requests:
cpu: 25m
memory: 128Mi
limits:
cpu: 250m
memory: 384Mi
- name: kaniko
image: gcr.io/kaniko-project/executor@sha256:c3109d5926a997b100c4343944e06c6b30a6804b2f9abe0994d3de6ef92b028e
command: ["/busybox/sh", "-c"]
args: ["/busybox/sleep 99d"]
tty: true
securityContext:
allowPrivilegeEscalation: false
capabilities:
drop: ["ALL"]
add: ["CHOWN", "FOWNER", "DAC_OVERRIDE", "SETGID", "SETUID"]
privileged: false
runAsUser: 0
seccompProfile:
type: RuntimeDefault
resources:
requests:
cpu: 100m
memory: 512Mi
ephemeral-storage: 10Gi
limits:
cpu: "1500m"
memory: 3Gi
ephemeral-storage: 20Gi
"""
}
}
steps {
// This amd64 leg runs on its own fresh pod (titan-24), so it must check
// out the SCM itself before the reviewed-revision git boundary check —
// otherwise `git rev-parse origin/main` fails with "not a git repository".
checkout scm
container('jnlp') {
sh '''
set -eu
mkdir -p build
test "${PUBLISH_IMAGE}" = "true"
test "${CONFIRM_PUBLISH}" = "PUBLISH HERMES WEBUI"
case "${EXPECTED_SOURCE_REVISION}" in
*[!0-9a-f]*|'')
echo "EXPECTED_SOURCE_REVISION must be a lowercase full commit" >&2
exit 2
;;
esac
test "${#EXPECTED_SOURCE_REVISION}" -eq 40
main_revision="$(git rev-parse HEAD)"
test "${main_revision}" = "$(git rev-parse origin/main)"
git merge-base --is-ancestor "${EXPECTED_SOURCE_REVISION}" "${main_revision}"
git checkout --detach "${EXPECTED_SOURCE_REVISION}"
actual_revision="$(git rev-parse HEAD)"
test "${actual_revision}" = "${EXPECTED_SOURCE_REVISION}"
test -z "$(git status --porcelain)"
test -f dockerfiles/Dockerfile.hermes-webui
case "${BUILD_NUMBER}" in
''|0*|*[!0-9]*)
echo "BUILD_NUMBER must be a positive decimal integer" >&2
exit 2
;;
esac
printf '%s\n' \
"${HERMES_IMAGE}:git-${actual_revision}-build-${BUILD_NUMBER}" \
> build/hermes-webui.destination
printf '%s\n' "${actual_revision}" > build/hermes-webui.source-revision
'''
}
container('kaniko') {
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
passwordVariable: 'HARBOR_PASSWORD'
)]) {
sh '''#!/busybox/sh
set -eu
set +x
config_path=/kaniko/.docker/config.json
destination="$(cat build/hermes-webui.destination)-amd64"
source_revision="$(cat build/hermes-webui.source-revision)"
umask 077
auth="$(printf '%s:%s' "${HARBOR_USER}" "${HARBOR_PASSWORD}" | /busybox/base64 | /busybox/tr -d '\n')"
/busybox/mkdir -p /kaniko/.docker
/busybox/printf '{"auths":{"registry.bstein.dev":{"auth":"%s"}}}\n' "${auth}" > "${config_path}"
unset HARBOR_USER HARBOR_PASSWORD auth
trap '/busybox/rm -f "${config_path}"' EXIT HUP INT TERM
umask 022
/kaniko/executor \
--registry-mirror=harbor-core.harbor.svc.cluster.local \
--insecure-registry=harbor-core.harbor.svc.cluster.local \
--context="dir://${WORKSPACE}" \
--dockerfile="${WORKSPACE}/dockerfiles/Dockerfile.hermes-webui" \
--destination="${destination}" \
--build-arg="HERMES_WEBUI_RELEASE_ID=git-${source_revision}-build-${BUILD_NUMBER}" \
--digest-file="${WORKSPACE}/build/hermes-webui-amd64.digest" \
--image-name-tag-with-digest-file="${WORKSPACE}/build/hermes-webui-amd64.image" \
--label="org.opencontainers.image.revision=${source_revision}" \
--label="org.opencontainers.image.source=https://scm.bstein.dev/titan/atlas-iac" \
--label="org.opencontainers.image.title=hermes-webui" \
--cleanup \
--push-retry=3
/busybox/chmod 644 build/hermes-webui-amd64.digest build/hermes-webui-amd64.image
'''
}
}
stash(
name: 'hermes-webui-amd64-evidence',
includes: 'build/hermes-webui-amd64.digest,build/hermes-webui-amd64.image'
)
}
}
stage('Combine multi-arch index') {
steps {
unstash 'hermes-webui-amd64-evidence'
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
passwordVariable: 'HARBOR_PASSWORD'
)]) {
sh '''
set -eu
set +x
destination="$(cat build/hermes-webui.destination)"
source_revision="$(cat build/hermes-webui.source-revision)"
python3 ci/scripts/hermes_multiarch_combine.py \
--destination "${destination}" \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}" \
--arm64-digest-file build/hermes-webui-arm64.digest \
--arm64-image-file build/hermes-webui-arm64.image \
--amd64-digest-file build/hermes-webui-amd64.digest \
--amd64-image-file build/hermes-webui-amd64.image \
--digest-file build/hermes-webui.digest \
--image-file build/hermes-webui.image
test -s build/hermes-webui.digest
test -s build/hermes-webui.image
'''
}
}
}
stage('Render reviewed Flux handoff') {
steps {
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
passwordVariable: 'HARBOR_PASSWORD'
)]) {
sh '''
set -eu
set +x
destination="$(cat build/hermes-webui.destination)"
source_revision="$(cat build/hermes-webui.source-revision)"
python3 ci/scripts/hermes_webui_release.py render \
--digest-file build/hermes-webui.digest \
--image-file build/hermes-webui.image \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}" \
--destination "${destination}" \
--chat-manifest services/hermes/chat-statefulset.yaml \
--dashboard-manifest services/hermes/deployment.yaml \
--output-dir build/hermes-webui-release
test -s build/hermes-webui-release/hermes-webui-image-update.patch
test -s build/hermes-webui-release/hermes-webui-image.json
'''
}
}
}
stage('Verify and archive release evidence') {
steps {
sh '''
set -eu
expected_files="$(printf '%s\n' \
build/hermes-webui.destination \
build/hermes-webui.digest \
build/hermes-webui.image \
build/hermes-webui.source-revision \
build/hermes-webui-arm64.digest \
build/hermes-webui-arm64.image \
build/hermes-webui-amd64.digest \
build/hermes-webui-amd64.image \
build/hermes-webui-release/hermes-chat-statefulset.yaml \
build/hermes-webui-release/hermes-dashboard-deployment.yaml \
build/hermes-webui-release/hermes-webui-image.json \
build/hermes-webui-release/hermes-webui-image-update.patch \
| LC_ALL=C sort)"
actual_files="$(find build -type f -print | LC_ALL=C sort)"
test "${actual_files}" = "${expected_files}"
destination="$(cat build/hermes-webui.destination)"
source_revision="$(cat build/hermes-webui.source-revision)"
python3 ci/scripts/hermes_webui_release.py verify-evidence \
--digest-file build/hermes-webui.digest \
--image-file build/hermes-webui.image \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}" \
--destination "${destination}" \
--chat-manifest services/hermes/chat-statefulset.yaml \
--dashboard-manifest services/hermes/deployment.yaml \
--output-dir build/hermes-webui-release
'''
archiveArtifacts(
artifacts: 'build/hermes-webui.destination,build/hermes-webui.digest,build/hermes-webui.image,build/hermes-webui.source-revision,build/hermes-webui-arm64.digest,build/hermes-webui-arm64.image,build/hermes-webui-amd64.digest,build/hermes-webui-amd64.image,build/hermes-webui-release/hermes-chat-statefulset.yaml,build/hermes-webui-release/hermes-dashboard-deployment.yaml,build/hermes-webui-release/hermes-webui-image.json,build/hermes-webui-release/hermes-webui-image-update.patch',
allowEmptyArchive: false,
fingerprint: true
)
}
}
stage('Publish Flux release tag') {
steps {
withCredentials([usernamePassword(
credentialsId: 'harbor-robot',
usernameVariable: 'HARBOR_USER',
passwordVariable: 'HARBOR_PASSWORD'
)]) {
sh '''
set -eu
set +x
destination="$(cat build/hermes-webui.destination)"
source_revision="$(cat build/hermes-webui.source-revision)"
python3 ci/scripts/hermes_oci_promote.py \
--destination "${destination}" \
--digest-file build/hermes-webui.digest \
--source-revision "${source_revision}" \
--build-number "${BUILD_NUMBER}"
'''
}
}
}
}
}

View File

@ -1,11 +1,16 @@
pipeline {
agent {
kubernetes {
// Keep build I/O off the node's runtime USB drive.
workspaceVolume dynamicPVC(accessModes: 'ReadWriteOnce', requestsSize: '20Gi', storageClassName: 'ci-scratch')
defaultContainer 'python'
yaml """
apiVersion: v1
kind: Pod
spec:
securityContext:
fsGroup: 1000
fsGroupChangePolicy: OnRootMismatch
nodeSelector:
kubernetes.io/arch: arm64
node-role.kubernetes.io/worker: "true"
@ -14,12 +19,9 @@ spec:
requiredDuringSchedulingIgnoredDuringExecution:
nodeSelectorTerms:
- matchExpressions:
- key: kubernetes.io/hostname
operator: NotIn
values:
- titan-04
- titan-06
- titan-11
- {key: hardware, operator: In, values: [rpi5, rpi4]}
- {key: node-role.kubernetes.io/worker, operator: In, values: ["true"]}
- {key: kubernetes.io/hostname, operator: NotIn, values: [titan-04, titan-05, titan-06, titan-08, titan-11, titan-12, titan-13, titan-14, titan-15, titan-17, titan-18, titan-19]}
preferredDuringSchedulingIgnoredDuringExecution:
- weight: 100
preference:
@ -64,6 +66,10 @@ spec:
}
}
environment {
PIP_CACHE_DIR = '/home/jenkins/agent/.cache/pip'
NPM_CONFIG_CACHE = '/home/jenkins/agent/.cache/npm'
SONAR_USER_HOME = '/home/jenkins/agent/.cache/sonar'
TMPDIR = '/home/jenkins/agent/.cache/tmp'
PIP_DISABLE_PIP_VERSION_CHECK = '1'
PYTHONUNBUFFERED = '1'
SUITE_NAME = 'titan_iac'
@ -85,6 +91,11 @@ spec:
buildDiscarder(logRotator(daysToKeepStr: '30', numToKeepStr: '200', artifactDaysToKeepStr: '30', artifactNumToKeepStr: '120'))
}
stages {
stage('Prepare build scratch') {
steps {
sh 'mkdir -p /home/jenkins/agent/.cache/tmp /home/jenkins/agent/.cache/go-tmp'
}
}
stage('Checkout') {
steps {
checkout scm
@ -479,7 +490,7 @@ PY
set +x
git config user.email "jenkins@bstein.dev"
git config user.name "jenkins"
git remote set-url origin https://${GIT_USER}:${GIT_TOKEN}@scm.bstein.dev/atlas/titan-iac.git
git remote set-url origin https://${GIT_USER}:${GIT_TOKEN}@scm.bstein.dev/titan/atlas-iac.git
git push origin HEAD:${FLUX_BRANCH}
'''
}

View File

@ -0,0 +1,479 @@
#!/usr/bin/env python3
"""Verify and render a reviewable Hermes chat-router image release."""
from __future__ import annotations
import argparse
import base64
import difflib
import json
import os
import re
import urllib.error
import urllib.parse
import urllib.request
from pathlib import Path
from typing import Any, Callable
DEFAULT_IMAGE = "registry.bstein.dev/bstein/hermes-chat-router"
HARBOR_API_ORIGIN = "https://registry.bstein.dev/api/v2.0"
HARBOR_PROJECT = "bstein"
HARBOR_REPOSITORY = "hermes-chat-router"
IMMUTABLE_REPOSITORY_PATTERN = "hermes-chat-router"
IMMUTABLE_TAG_PATTERN = "git-*-build-*"
DIGEST_PATTERN = re.compile(r"^sha256:[0-9a-f]{64}$")
REVISION_PATTERN = re.compile(r"^[0-9a-f]{40}$")
BUILD_PATTERN = re.compile(r"^[1-9][0-9]*$")
DESTINATION_PATTERN = re.compile(
r"^registry\.bstein\.dev/bstein/hermes-chat-router:"
r"git-([0-9a-f]{40})-build-([1-9][0-9]*)$"
)
class _NoRedirect(urllib.request.HTTPRedirectHandler):
"""Never forward registry credentials to a redirect target."""
def redirect_request(self, _request, _file, _code, _message, _headers, _url):
return None
def _validated(value: str, pattern: re.Pattern[str], label: str) -> str:
"""Return a normalized value only when it matches the release contract."""
normalized = value.strip()
if not pattern.fullmatch(normalized):
raise ValueError(f"invalid {label}: expected {pattern.pattern}")
return normalized
def validate_destination(
destination: str, source_revision: str, build_number: str
) -> tuple[str, str]:
"""Bind the unique build tag to one reviewed source and Jenkins build."""
revision = _validated(source_revision, REVISION_PATTERN, "source revision")
build = _validated(build_number, BUILD_PATTERN, "build number")
match = DESTINATION_PATTERN.fullmatch(destination.strip())
if not match or match.groups() != (revision, build):
raise ValueError("destination does not match the reviewed revision and build")
return revision, build
def validate_kaniko_evidence(
*, digest_text: str, image_text: str, destination: str
) -> str:
"""Cross-check Kaniko's two independent output files."""
digests = digest_text.splitlines()
images = image_text.splitlines()
if len(digests) != 1 or len(images) != 1:
raise ValueError("Kaniko evidence must contain exactly one line per file")
digest = _validated(digests[0], DIGEST_PATTERN, "image digest")
if images[0].strip() != f"{destination}@{digest}":
raise ValueError("Kaniko image evidence does not match destination and digest")
return digest
def _registry_request(request: urllib.request.Request, timeout: int) -> Any:
"""Return same-origin registry responses without following redirects."""
opener = urllib.request.build_opener(_NoRedirect())
try:
return opener.open(request, timeout=timeout)
except urllib.error.HTTPError as exc:
return exc
def _authorization(username: str, password: str) -> str:
"""Build a Basic header without placing credentials in a URL."""
if not username or not password:
raise RuntimeError("Harbor credentials are unavailable")
token = base64.b64encode(f"{username}:{password}".encode()).decode("ascii")
return f"Basic {token}"
def _artifact_response(
destination: str,
*,
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any] = _registry_request,
) -> tuple[int, bytes]:
"""Read one exact Harbor artifact by candidate tag."""
if not DESTINATION_PATTERN.fullmatch(destination):
raise ValueError("invalid destination")
tag = urllib.parse.quote(destination.rsplit(":", 1)[1], safe="")
request = urllib.request.Request(
f"{HARBOR_API_ORIGIN}/projects/{HARBOR_PROJECT}/repositories/"
f"{HARBOR_REPOSITORY}/artifacts/{tag}?with_immutable_status=true",
headers={
"Accept": "application/json",
"Authorization": _authorization(username, password),
},
method="GET",
)
with opener(request, 20) as response:
body = response.read(1_048_577)
if len(body) > 1_048_576:
raise RuntimeError("Harbor artifact response exceeded the size limit")
return int(response.status), body
def _rules_response(
*,
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any] = _registry_request,
) -> tuple[int, bytes, dict[str, str]]:
"""Read the complete bounded Harbor immutable-rule page."""
request = urllib.request.Request(
f"{HARBOR_API_ORIGIN}/projects/{HARBOR_PROJECT}/immutabletagrules"
"?page=1&page_size=100",
headers={
"Accept": "application/json",
"Authorization": _authorization(username, password),
},
method="GET",
)
with opener(request, 20) as response:
body = response.read(1_048_577)
if len(body) > 1_048_576:
raise RuntimeError("Harbor immutable rule response exceeded the size limit")
return int(response.status), body, dict(response.headers)
def _normalized_rule(rule: dict[str, Any]) -> dict[str, Any]:
"""Select only immutable-policy fields used by this lane."""
return {
"disabled": bool(rule.get("disabled", False)),
"action": rule.get("action"),
"template": rule.get("template"),
"tag_selectors": [
{key: item.get(key) for key in ("kind", "decoration", "pattern")}
for item in rule.get("tag_selectors") or []
if isinstance(item, dict)
],
"scope_selectors": {
"repository": [
{key: item.get(key) for key in ("kind", "decoration", "pattern")}
for item in (rule.get("scope_selectors") or {}).get("repository", [])
if isinstance(item, dict)
]
},
}
def verify_immutable_policy(
*,
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any] = _registry_request,
) -> None:
"""Fail closed unless one exact active router immutability rule exists."""
status, body, headers = _rules_response(
username=username, password=password, opener=opener
)
if status != 200:
raise RuntimeError(f"Harbor immutable policy preflight returned HTTP {status}")
try:
rules = json.loads(body.decode("utf-8"))
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
raise RuntimeError("Harbor returned invalid immutable rule JSON") from exc
if not isinstance(rules, list) or not all(isinstance(item, dict) for item in rules):
raise RuntimeError("Harbor immutable rule list has an invalid shape")
total = next(
(value for key, value in headers.items() if key.lower() == "x-total-count"),
None,
)
if total is None or not str(total).isdecimal() or int(total) != len(rules):
raise RuntimeError("Harbor immutable rule page is incomplete")
expected = {
"disabled": False,
"action": "immutable",
"template": "immutable_template",
"tag_selectors": [
{
"kind": "doublestar",
"decoration": "matches",
"pattern": IMMUTABLE_TAG_PATTERN,
}
],
"scope_selectors": {
"repository": [
{
"kind": "doublestar",
"decoration": "repoMatches",
"pattern": IMMUTABLE_REPOSITORY_PATTERN,
}
]
},
}
matches = [
value
for value in map(_normalized_rule, rules)
if value["tag_selectors"] == expected["tag_selectors"]
and value["scope_selectors"] == expected["scope_selectors"]
]
if matches != [expected]:
raise RuntimeError("Harbor router immutable build-tag policy is not exact")
def assert_tag_absent(
destination: str,
*,
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any] = _registry_request,
) -> None:
"""Reject replay before Kaniko can target an already-used build tag."""
status, _ = _artifact_response(
destination, username=username, password=password, opener=opener
)
if status == 404:
return
if status == 200:
raise RuntimeError("Harbor destination tag already exists")
raise RuntimeError(f"Harbor destination preflight returned HTTP {status}")
def verify_registry_digest(
destination: str,
digest: str,
source_revision: str,
*,
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any] = _registry_request,
) -> None:
"""Verify Harbor's digest, immutable tag, and persisted source label."""
digest = _validated(digest, DIGEST_PATTERN, "image digest")
revision = _validated(source_revision, REVISION_PATTERN, "source revision")
status, body = _artifact_response(
destination, username=username, password=password, opener=opener
)
if status != 200:
raise RuntimeError(f"Harbor manifest verification returned HTTP {status}")
try:
artifact = json.loads(body.decode("utf-8"))
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
raise RuntimeError("Harbor returned invalid artifact JSON") from exc
tag = destination.rsplit(":", 1)[1]
matching = [
item
for item in artifact.get("tags") or []
if isinstance(item, dict) and item.get("name") == tag
]
labels = ((artifact.get("extra_attrs") or {}).get("config") or {}).get("Labels")
if artifact.get("digest") != digest:
raise RuntimeError("Harbor digest does not match Kaniko evidence")
if len(matching) != 1 or matching[0].get("immutable") is not True:
raise RuntimeError("Harbor did not enforce the candidate tag as immutable")
if not isinstance(labels, dict) or labels.get(
"org.opencontainers.image.revision"
) != revision:
raise RuntimeError("Harbor OCI source-revision label does not match")
def render_workload(source: str, digest: str) -> str:
"""Replace the single exact router image while preserving the Flux marker."""
digest = _validated(digest, DIGEST_PATTERN, "image digest")
identity = "apiVersion: apps/v1\nkind: Deployment\nmetadata:\n name: hermes-chat-router\n"
if not source.startswith("# services/hermes/chat-router.yaml\n" + identity):
raise ValueError("Flux target identity changed")
lines = source.splitlines(keepends=True)
matches: list[int] = []
for index, line in enumerate(lines):
stripped = line.strip()
if not stripped.startswith("image: "):
continue
value = stripped.removeprefix("image: ").split(" #", 1)[0]
image, separator, current_digest = value.rpartition("@")
if separator and re.fullmatch(
rf"{re.escape(DEFAULT_IMAGE)}(?::[A-Za-z0-9_][A-Za-z0-9_.-]{{0,127}})?",
image,
):
_validated(current_digest, DIGEST_PATTERN, "current Flux image digest")
matches.append(index)
if len(matches) != 1:
raise ValueError(f"expected exactly one router image; found {len(matches)}")
index = matches[0]
indent = lines[index][: len(lines[index]) - len(lines[index].lstrip())]
comment = ""
if " #" in lines[index]:
comment = " #" + lines[index].split(" #", 1)[1].rstrip("\n")
newline = "\n" if lines[index].endswith("\n") else ""
lines[index] = f"{indent}image: {DEFAULT_IMAGE}@{digest}{comment}{newline}"
return "".join(lines)
def _metadata(
digest: str, revision: str, build: str, destination: str
) -> dict[str, Any]:
return {
"build_number": build,
"digest": digest,
"flux_image": f"{DEFAULT_IMAGE}@{digest}",
"flux_targets": ["apps/Deployment/hermes/hermes-chat-router"],
"image": DEFAULT_IMAGE,
"published_tag": destination,
"source_revision": revision,
}
def _expected_artifacts(
*,
digest: str,
source_revision: str,
build_number: str,
destination: str,
manifest: Path,
) -> dict[str, str]:
source = manifest.read_text(encoding="utf-8")
rendered = render_workload(source, digest)
patch = "".join(
difflib.unified_diff(
source.splitlines(keepends=True),
rendered.splitlines(keepends=True),
fromfile="a/services/hermes/chat-router.yaml",
tofile="b/services/hermes/chat-router.yaml",
)
)
if not patch:
raise ValueError("published digest already matches the Flux target")
metadata = _metadata(digest, source_revision, build_number, destination)
return {
"hermes-chat-router-deployment.yaml": rendered,
"hermes-chat-router-image-update.patch": patch,
"hermes-chat-router-image.json": json.dumps(
metadata, indent=2, sort_keys=True
)
+ "\n",
}
def write_release_artifacts(
*,
digest: str,
source_revision: str,
build_number: str,
destination: str,
manifest: Path,
output_dir: Path,
) -> None:
"""Write deterministic, credential-free Flux handoff evidence."""
digest = _validated(digest, DIGEST_PATTERN, "image digest")
revision, build = validate_destination(destination, source_revision, build_number)
expected = _expected_artifacts(
digest=digest,
source_revision=revision,
build_number=build,
destination=destination,
manifest=manifest,
)
output_dir.mkdir(parents=True, exist_ok=True)
for name, content in expected.items():
(output_dir / name).write_text(content, encoding="utf-8")
def validate_release_artifacts(
*,
digest: str,
source_revision: str,
build_number: str,
destination: str,
manifest: Path,
output_dir: Path,
) -> None:
"""Recompute and compare every archived handoff byte."""
digest = _validated(digest, DIGEST_PATTERN, "image digest")
revision, build = validate_destination(destination, source_revision, build_number)
expected = _expected_artifacts(
digest=digest,
source_revision=revision,
build_number=build,
destination=destination,
manifest=manifest,
)
entries = list(output_dir.iterdir())
if {entry.name for entry in entries} != set(expected) or not all(
entry.is_file() and not entry.is_symlink() for entry in entries
):
raise ValueError("release output must contain exactly three evidence files")
for name, content in expected.items():
if (output_dir / name).read_text(encoding="utf-8") != content:
raise ValueError(f"release evidence is incomplete or mismatched: {name}")
def _credentials() -> tuple[str, str]:
username = os.environ.get("HARBOR_USER", "")
password = os.environ.get("HARBOR_PASSWORD", "")
if not username or not password:
raise RuntimeError("Harbor credentials are unavailable")
return username, password
def _add_common(parser: argparse.ArgumentParser) -> None:
parser.add_argument("--source-revision", required=True)
parser.add_argument("--build-number", required=True)
parser.add_argument("--destination", required=True)
def main() -> int:
"""Run one fail-closed candidate or evidence operation."""
parser = argparse.ArgumentParser(description=__doc__)
commands = parser.add_subparsers(dest="command", required=True)
absent = commands.add_parser("assert-absent")
_add_common(absent)
for name in ("render", "verify-evidence"):
command = commands.add_parser(name)
_add_common(command)
command.add_argument("--digest-file", required=True, type=Path)
command.add_argument("--image-file", required=True, type=Path)
command.add_argument("--manifest", required=True, type=Path)
command.add_argument("--output-dir", required=True, type=Path)
args = parser.parse_args()
validate_destination(args.destination, args.source_revision, args.build_number)
if args.command == "verify-evidence":
digest = validate_kaniko_evidence(
digest_text=args.digest_file.read_text(encoding="utf-8"),
image_text=args.image_file.read_text(encoding="utf-8"),
destination=args.destination,
)
validate_release_artifacts(
digest=digest,
source_revision=args.source_revision,
build_number=args.build_number,
destination=args.destination,
manifest=args.manifest,
output_dir=args.output_dir,
)
return 0
username, password = _credentials()
if args.command == "assert-absent":
verify_immutable_policy(username=username, password=password)
assert_tag_absent(
args.destination, username=username, password=password
)
return 0
digest = validate_kaniko_evidence(
digest_text=args.digest_file.read_text(encoding="utf-8"),
image_text=args.image_file.read_text(encoding="utf-8"),
destination=args.destination,
)
verify_registry_digest(
args.destination,
digest,
args.source_revision,
username=username,
password=password,
)
write_release_artifacts(
digest=digest,
source_revision=args.source_revision,
build_number=args.build_number,
destination=args.destination,
manifest=args.manifest,
output_dir=args.output_dir,
)
return 0
if __name__ == "__main__": # pragma: no cover - exercised through main()
raise SystemExit(main())

View File

@ -174,9 +174,7 @@ def _normalized_immutable_rule(rule: dict[str, Any]) -> dict[str, Any]:
"decoration": item.get("decoration"),
"pattern": item.get("pattern"),
}
for item in (rule.get("scope_selectors") or {}).get(
"repository", []
)
for item in (rule.get("scope_selectors") or {}).get("repository", [])
if isinstance(item, dict)
]
},
@ -317,7 +315,10 @@ def render_kustomization(source: str, digest: str, image: str = DEFAULT_IMAGE) -
index = matches[0]
newline = "\n" if lines[index].endswith("\n") else ""
prefix = lines[index][: len(lines[index]) - len(lines[index].lstrip())]
lines[index] = f"{prefix}digest: {digest}{newline}"
value = lines[index].strip().removeprefix("digest:").strip()
_current_digest, separator, comment = value.partition(" #")
suffix = f" #{comment}" if separator else ""
lines[index] = f"{prefix}digest: {digest}{suffix}{newline}"
return "".join(lines)
@ -414,7 +415,8 @@ def validate_release_artifacts(
"source_revision": source_revision,
}
expected = {
"hermes-agent-image.json": json.dumps(metadata, indent=2, sort_keys=True) + "\n",
"hermes-agent-image.json": json.dumps(metadata, indent=2, sort_keys=True)
+ "\n",
"hermes-image-update.patch": patch,
"hermes-kustomization.yaml": rendered,
}

View File

@ -0,0 +1,393 @@
#!/usr/bin/env python3
"""Assemble a fail-closed multi-arch manifest list from two per-arch leaves.
Kaniko builds one native image per architecture (arm64 on an rpi5 pod, amd64 on
titan-24) and pushes each under an arch-suffixed candidate tag
``...-build-<N>-<arch>``. Kaniko cannot combine, so this step:
1. Independently re-reads each per-arch candidate manifest from the registry and
binds it to the exact Kaniko digest evidence.
2. Proves each leaf really is the architecture it claims by reading its image
config (a swapped or cross-built leaf fails closed here).
3. Builds a Docker manifest *list* (not an OCI index) from the two verified
leaves -- Docker manifest lists are already inside the promotion allow-list,
so this keeps the security surface of ``hermes_oci_promote.py`` unchanged.
4. Refuses to overwrite an existing final tag, PUTs the list to the final
``...-build-<N>`` tag, and re-reads it to confirm the registry resolved the
exact index digest referencing exactly the two expected leaves.
The output ``--digest-file``/``--image-file`` deliberately use the SAME format
the single-arch Kaniko step produced (``<digest>`` and ``<destination>@<digest>``
for the final, arch-less tag). The whole downstream evidence chain --
``hermes_image_release.py`` render/verify-evidence and ``hermes_oci_promote.py``
-- therefore promotes the multi-arch INDEX with no further change.
"""
from __future__ import annotations
import argparse
import base64
import hashlib
import json
import os
import re
import urllib.error
import urllib.parse
import urllib.request
from pathlib import Path
from typing import Any, Callable
REGISTRY_ORIGIN = "https://registry.bstein.dev"
# The final (arch-less) Flux-visible tag; identical contract to the promoter.
# Both Hermes images that the multi-arch pipelines publish share the exact same
# arch-less final-tag contract; the repository name is the only difference and is
# captured here so the combiner stays fail-closed to just these two components.
DESTINATION_PATTERN = re.compile(
r"^registry\.bstein\.dev/bstein/(?P<component>hermes-agent|hermes-webui):"
r"git-(?P<revision>[0-9a-f]{40})-build-(?P<build>[1-9][0-9]*)$"
)
DIGEST_PATTERN = re.compile(r"^sha256:[0-9a-f]{64}$")
DOCKER_MANIFEST_LIST = "application/vnd.docker.distribution.manifest.list.v2+json"
# A per-arch leaf must be a single-image manifest, never itself a list/index.
LEAF_MANIFEST_TYPES = {
"application/vnd.docker.distribution.manifest.v2+json",
"application/vnd.oci.image.manifest.v1+json",
}
IMAGE_CONFIG_TYPES = {
"application/vnd.docker.container.image.v1+json",
"application/vnd.oci.image.config.v1+json",
}
# Deterministic architecture order -> deterministic manifest-list bytes/digest.
ARCHITECTURES = ("amd64", "arm64")
MAX_MANIFEST_BYTES = 4 * 1024 * 1024
MAX_CONFIG_BYTES = 1024 * 1024
class _NoRedirect(urllib.request.HTTPRedirectHandler):
"""Never forward registry credentials to another origin."""
def redirect_request(self, _request, _file, _code, _message, _headers, _url):
return None
def _registry_request(request: urllib.request.Request, timeout: int) -> Any:
"""Return normal and HTTP error responses without following redirects."""
opener = urllib.request.build_opener(_NoRedirect())
try:
return opener.open(request, timeout=timeout)
except urllib.error.HTTPError as exc:
return exc
def _status(response: Any) -> int:
"""Normalize urllib response and HTTPError status fields."""
return int(getattr(response, "status", getattr(response, "code", 0)))
def _authorization(username: str, password: str) -> str:
"""Build a Basic authorization value without placing it in a URL."""
if not username or not password:
raise RuntimeError("Harbor credentials are unavailable")
encoded = base64.b64encode(f"{username}:{password}".encode()).decode("ascii")
return f"Basic {encoded}"
def _manifest_url(component: str, reference: str) -> str:
"""Return one same-origin, path-escaped Docker Registry manifest URL."""
encoded = urllib.parse.quote(reference, safe="")
return f"{REGISTRY_ORIGIN}/v2/bstein/{component}/manifests/{encoded}"
def _blob_url(component: str, digest: str) -> str:
"""Return one same-origin, path-escaped Docker Registry blob URL."""
encoded = urllib.parse.quote(digest, safe="")
return f"{REGISTRY_ORIGIN}/v2/bstein/{component}/blobs/{encoded}"
def _read_evidence_pair(
*, digest_text: str, image_text: str, per_arch_tag_ref: str
) -> str:
"""Cross-check both Kaniko output files for one arch against its tag."""
digest_lines = digest_text.splitlines()
image_lines = image_text.splitlines()
if len(digest_lines) != 1:
raise ValueError("Kaniko digest evidence must contain exactly one line")
if len(image_lines) != 1:
raise ValueError("Kaniko image evidence must contain exactly one line")
digest = digest_lines[0].strip()
if not DIGEST_PATTERN.fullmatch(digest):
raise ValueError("invalid per-arch image digest")
if image_lines[0].strip() != f"{per_arch_tag_ref}@{digest}":
raise ValueError("per-arch image evidence does not match tag and digest")
return digest
def _verified_leaf(
*,
component: str,
per_arch_tag: str,
architecture: str,
expected_digest: str,
authorization: str,
opener: Callable[[urllib.request.Request, int], Any],
) -> dict[str, Any]:
"""Re-read one per-arch leaf and prove its digest, type, and architecture."""
accept = ", ".join(sorted(LEAF_MANIFEST_TYPES))
request = urllib.request.Request(
_manifest_url(component, per_arch_tag),
headers={"Accept": accept, "Authorization": authorization},
method="GET",
)
with opener(request, 30) as response:
if _status(response) != 200:
raise RuntimeError(
f"{architecture} leaf manifest returned HTTP {_status(response)}"
)
body = response.read(MAX_MANIFEST_BYTES + 1)
if len(body) > MAX_MANIFEST_BYTES:
raise RuntimeError(f"{architecture} leaf manifest exceeded the size limit")
observed_digest = response.headers.get("Docker-Content-Digest", "")
content_type = response.headers.get("Content-Type", "").split(";", 1)[0].strip()
if observed_digest != expected_digest:
raise RuntimeError(f"{architecture} leaf digest does not match build evidence")
# Defence in depth: the digest header is registry-asserted; recompute it too.
if f"sha256:{hashlib.sha256(body).hexdigest()}" != expected_digest:
raise RuntimeError(f"{architecture} leaf bytes do not hash to its digest")
if content_type not in LEAF_MANIFEST_TYPES:
raise RuntimeError(f"{architecture} leaf is not a single-image manifest")
try:
manifest = json.loads(body.decode("utf-8"))
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
raise RuntimeError(f"{architecture} leaf manifest is not valid JSON") from exc
config = manifest.get("config")
if not isinstance(config, dict):
raise RuntimeError(f"{architecture} leaf manifest omits its config descriptor")
config_digest = str(config.get("digest") or "")
config_type = str(config.get("mediaType") or "")
if not DIGEST_PATTERN.fullmatch(config_digest):
raise RuntimeError(f"{architecture} leaf config digest is invalid")
if config_type not in IMAGE_CONFIG_TYPES:
raise RuntimeError(f"{architecture} leaf config media type is unsupported")
config_request = urllib.request.Request(
_blob_url(component, config_digest),
headers={"Accept": config_type, "Authorization": authorization},
method="GET",
)
with opener(config_request, 30) as response:
if _status(response) != 200:
raise RuntimeError(
f"{architecture} leaf config returned HTTP {_status(response)}"
)
config_body = response.read(MAX_CONFIG_BYTES + 1)
if len(config_body) > MAX_CONFIG_BYTES:
raise RuntimeError(f"{architecture} leaf config exceeded the size limit")
if f"sha256:{hashlib.sha256(config_body).hexdigest()}" != config_digest:
raise RuntimeError(f"{architecture} leaf config bytes do not hash to its digest")
try:
config_json = json.loads(config_body.decode("utf-8"))
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
raise RuntimeError(f"{architecture} leaf config is not valid JSON") from exc
if config_json.get("architecture") != architecture:
raise RuntimeError(
f"{architecture} leaf config reports architecture "
f"{config_json.get('architecture')!r}"
)
if config_json.get("os") != "linux":
raise RuntimeError(f"{architecture} leaf config reports a non-linux os")
return {
"mediaType": content_type,
"size": len(body),
"digest": expected_digest,
"platform": {"architecture": architecture, "os": "linux"},
}
def _manifest_list_bytes(descriptors: list[dict[str, Any]]) -> bytes:
"""Serialize the manifest list deterministically for a stable index digest."""
document = {
"schemaVersion": 2,
"mediaType": DOCKER_MANIFEST_LIST,
"manifests": descriptors,
}
return json.dumps(document, sort_keys=True, separators=(",", ":")).encode("utf-8")
def combine_multiarch_index(
*,
destination: str,
arch_digests: dict[str, str],
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any] = _registry_request,
) -> dict[str, str]:
"""Verify both leaves, publish, and re-verify one multi-arch index tag."""
match = DESTINATION_PATTERN.fullmatch(destination.strip())
if not match:
raise ValueError("invalid multi-arch destination")
if set(arch_digests) != set(ARCHITECTURES):
raise ValueError("expected exactly the arm64 and amd64 per-arch digests")
component = match.group("component")
index_tag = destination.rsplit(":", 1)[1]
authorization = _authorization(username, password)
descriptors = [
_verified_leaf(
component=component,
per_arch_tag=f"{index_tag}-{architecture}",
architecture=architecture,
expected_digest=arch_digests[architecture],
authorization=authorization,
opener=opener,
)
for architecture in ARCHITECTURES
]
manifest_list = _manifest_list_bytes(descriptors)
if len(manifest_list) > MAX_MANIFEST_BYTES:
raise RuntimeError("assembled manifest list exceeded the size limit")
index_digest = f"sha256:{hashlib.sha256(manifest_list).hexdigest()}"
index_url = _manifest_url(component, index_tag)
head_request = urllib.request.Request(
index_url,
headers={"Accept": DOCKER_MANIFEST_LIST, "Authorization": authorization},
method="HEAD",
)
with opener(head_request, 20) as response:
head_status = _status(response)
existing_digest = response.headers.get("Docker-Content-Digest", "")
if head_status == 200:
if existing_digest != index_digest:
raise RuntimeError("final tag already exists with another index digest")
result = "already-present"
elif head_status == 404:
put_request = urllib.request.Request(
index_url,
data=manifest_list,
headers={
"Authorization": authorization,
"Content-Type": DOCKER_MANIFEST_LIST,
},
method="PUT",
)
with opener(put_request, 30) as response:
put_status = _status(response)
put_digest = response.headers.get("Docker-Content-Digest", "")
if put_status not in {201, 202}:
raise RuntimeError(f"index manifest returned HTTP {put_status}")
if put_digest and put_digest != index_digest:
raise RuntimeError("index manifest digest changed during publish")
result = "published"
else:
raise RuntimeError(f"final tag preflight returned HTTP {head_status}")
verify_request = urllib.request.Request(
index_url,
headers={"Accept": DOCKER_MANIFEST_LIST, "Authorization": authorization},
method="GET",
)
with opener(verify_request, 30) as response:
if _status(response) != 200:
raise RuntimeError(f"index verification returned HTTP {_status(response)}")
verify_body = response.read(MAX_MANIFEST_BYTES + 1)
if len(verify_body) > MAX_MANIFEST_BYTES:
raise RuntimeError("index verification exceeded the size limit")
verify_digest = response.headers.get("Docker-Content-Digest", "")
verify_type = response.headers.get("Content-Type", "").split(";", 1)[0].strip()
if verify_digest != index_digest:
raise RuntimeError("registry resolved the final tag to another index digest")
if verify_type != DOCKER_MANIFEST_LIST:
raise RuntimeError("registry did not store a Docker manifest list")
try:
published = json.loads(verify_body.decode("utf-8"))
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
raise RuntimeError("registry returned invalid index JSON") from exc
published_leaves = {
(
str((item or {}).get("platform", {}).get("architecture")),
str((item or {}).get("digest")),
)
for item in published.get("manifests") or []
}
expected_leaves = {
(architecture, arch_digests[architecture]) for architecture in ARCHITECTURES
}
if published_leaves != expected_leaves:
raise RuntimeError("published index does not reference the exact two leaves")
return {
"component": component,
"index_digest": index_digest,
"index_tag": index_tag,
"result": result,
**{f"{architecture}_digest": arch_digests[architecture] for architecture in ARCHITECTURES},
}
def _load_arch_digest(
*, destination: str, architecture: str, digest_file: Path, image_file: Path
) -> str:
"""Bind one arch's two Kaniko evidence files to its arch-suffixed tag."""
per_arch_tag_ref = f"{destination}-{architecture}"
return _read_evidence_pair(
digest_text=digest_file.read_text(encoding="utf-8"),
image_text=image_file.read_text(encoding="utf-8"),
per_arch_tag_ref=per_arch_tag_ref,
)
def main() -> int:
"""Combine two verified per-arch leaves and emit index digest evidence."""
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--destination", required=True)
parser.add_argument("--source-revision", required=True)
parser.add_argument("--build-number", required=True)
parser.add_argument("--arm64-digest-file", required=True, type=Path)
parser.add_argument("--arm64-image-file", required=True, type=Path)
parser.add_argument("--amd64-digest-file", required=True, type=Path)
parser.add_argument("--amd64-image-file", required=True, type=Path)
parser.add_argument("--digest-file", required=True, type=Path)
parser.add_argument("--image-file", required=True, type=Path)
args = parser.parse_args()
try:
match = DESTINATION_PATTERN.fullmatch(args.destination.strip())
if not match:
raise ValueError("invalid multi-arch destination")
if match.group("revision") != args.source_revision.strip():
raise ValueError("destination revision does not match evidence")
if match.group("build") != args.build_number.strip():
raise ValueError("destination build number does not match evidence")
destination = args.destination.strip()
arch_digests = {
"arm64": _load_arch_digest(
destination=destination,
architecture="arm64",
digest_file=args.arm64_digest_file,
image_file=args.arm64_image_file,
),
"amd64": _load_arch_digest(
destination=destination,
architecture="amd64",
digest_file=args.amd64_digest_file,
image_file=args.amd64_image_file,
),
}
result = combine_multiarch_index(
destination=destination,
arch_digests=arch_digests,
username=os.environ.get("HARBOR_USER", ""),
password=os.environ.get("HARBOR_PASSWORD", ""),
)
args.digest_file.write_text(result["index_digest"] + "\n", encoding="utf-8")
args.image_file.write_text(
f"{destination}@{result['index_digest']}\n", encoding="utf-8"
)
except (OSError, ValueError, RuntimeError, urllib.error.URLError) as exc:
print(json.dumps({"error": str(exc)}, sort_keys=True))
return 1
print(json.dumps(result, indent=2, sort_keys=True))
return 0
if __name__ == "__main__": # pragma: no cover - exercised through main()
raise SystemExit(main())

View File

@ -0,0 +1,193 @@
#!/usr/bin/env python3
"""Publish a validated Hermes candidate manifest under its Flux release tag."""
from __future__ import annotations
import argparse
import base64
import json
import os
import re
import urllib.error
import urllib.parse
import urllib.request
from pathlib import Path
from typing import Any, Callable
REGISTRY_ORIGIN = "https://registry.bstein.dev"
DESTINATION_PATTERN = re.compile(
r"^registry\.bstein\.dev/bstein/"
r"(?P<component>hermes-(?:agent|webui|chat-router|jetson-(?:stt|tts))):"
r"git-(?P<revision>[0-9a-f]{40})-build-(?P<build>[1-9][0-9]*)$"
)
DIGEST_PATTERN = re.compile(r"^sha256:[0-9a-f]{64}$")
MANIFEST_TYPES = {
"application/vnd.docker.distribution.manifest.v2+json",
"application/vnd.docker.distribution.manifest.list.v2+json",
"application/vnd.oci.image.manifest.v1+json",
"application/vnd.oci.image.index.v1+json",
}
MAX_MANIFEST_BYTES = 4 * 1024 * 1024
class _NoRedirect(urllib.request.HTTPRedirectHandler):
"""Never forward registry credentials to another origin."""
def redirect_request(self, _request, _file, _code, _message, _headers, _url):
return None
def _registry_request(request: urllib.request.Request, timeout: int) -> Any:
"""Return normal and HTTP error responses without following redirects."""
opener = urllib.request.build_opener(_NoRedirect())
try:
return opener.open(request, timeout=timeout)
except urllib.error.HTTPError as exc:
return exc
def _status(response: Any) -> int:
"""Normalize urllib response and HTTPError status fields."""
return int(getattr(response, "status", getattr(response, "code", 0)))
def _authorization(username: str, password: str) -> str:
"""Build a Basic authorization value without placing it in a URL."""
if not username or not password:
raise RuntimeError("Harbor credentials are unavailable")
encoded = base64.b64encode(f"{username}:{password}".encode()).decode("ascii")
return f"Basic {encoded}"
def _validated_release(
destination: str,
digest: str,
source_revision: str,
build_number: str,
) -> tuple[str, str, str]:
"""Bind the candidate and release tag to one reviewed source and build."""
match = DESTINATION_PATTERN.fullmatch(destination.strip())
if not match:
raise ValueError("invalid Hermes candidate destination")
if match.group("revision") != source_revision.strip():
raise ValueError("candidate source revision does not match evidence")
if match.group("build") != build_number.strip():
raise ValueError("candidate build number does not match evidence")
normalized_digest = digest.strip()
if not DIGEST_PATTERN.fullmatch(normalized_digest):
raise ValueError("invalid candidate digest")
candidate_tag = destination.rsplit(":", 1)[1]
return match.group("component"), candidate_tag, normalized_digest
def _manifest_url(component: str, tag: str) -> str:
"""Return one same-origin, path-escaped Docker Registry manifest URL."""
encoded_tag = urllib.parse.quote(tag, safe="")
return f"{REGISTRY_ORIGIN}/v2/bstein/{component}/manifests/{encoded_tag}"
def promote_candidate(
*,
destination: str,
digest: str,
source_revision: str,
build_number: str,
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any] = _registry_request,
) -> dict[str, str]:
"""Copy an exact candidate manifest to the immutable ``-release`` tag."""
component, candidate_tag, normalized_digest = _validated_release(
destination, digest, source_revision, build_number
)
release_tag = f"{candidate_tag}-release"
authorization = _authorization(username, password)
accept = ", ".join(sorted(MANIFEST_TYPES))
candidate_request = urllib.request.Request(
_manifest_url(component, candidate_tag),
headers={"Accept": accept, "Authorization": authorization},
method="GET",
)
with opener(candidate_request, 30) as response:
if _status(response) != 200:
raise RuntimeError(f"candidate manifest returned HTTP {_status(response)}")
manifest = response.read(MAX_MANIFEST_BYTES + 1)
if len(manifest) > MAX_MANIFEST_BYTES:
raise RuntimeError("candidate manifest exceeded the size limit")
observed_digest = response.headers.get("Docker-Content-Digest", "")
content_type = response.headers.get("Content-Type", "").split(";", 1)[0]
if observed_digest != normalized_digest:
raise RuntimeError("candidate manifest digest does not match build evidence")
if content_type not in MANIFEST_TYPES:
raise RuntimeError("candidate manifest returned an unsupported content type")
release_url = _manifest_url(component, release_tag)
head_request = urllib.request.Request(
release_url,
headers={"Accept": accept, "Authorization": authorization},
method="HEAD",
)
with opener(head_request, 20) as response:
head_status = _status(response)
existing_digest = response.headers.get("Docker-Content-Digest", "")
if head_status == 200:
if existing_digest != normalized_digest:
raise RuntimeError("release tag already exists with another digest")
result = "already-present"
elif head_status == 404:
put_request = urllib.request.Request(
release_url,
data=manifest,
headers={
"Authorization": authorization,
"Content-Type": content_type,
},
method="PUT",
)
with opener(put_request, 30) as response:
put_status = _status(response)
promoted_digest = response.headers.get("Docker-Content-Digest", "")
if put_status not in {201, 202}:
raise RuntimeError(f"release manifest returned HTTP {put_status}")
if promoted_digest and promoted_digest != normalized_digest:
raise RuntimeError("release manifest digest changed during promotion")
result = "published"
else:
raise RuntimeError(f"release tag preflight returned HTTP {head_status}")
return {
"component": component,
"digest": normalized_digest,
"release_tag": release_tag,
"result": result,
"source_revision": source_revision,
}
def main() -> int:
"""Promote one evidence-bound candidate and print credential-free metadata."""
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--destination", required=True)
parser.add_argument("--digest-file", required=True, type=Path)
parser.add_argument("--source-revision", required=True)
parser.add_argument("--build-number", required=True)
args = parser.parse_args()
try:
result = promote_candidate(
destination=args.destination,
digest=args.digest_file.read_text(encoding="utf-8"),
source_revision=args.source_revision,
build_number=args.build_number,
username=os.environ.get("HARBOR_USER", ""),
password=os.environ.get("HARBOR_PASSWORD", ""),
)
except (OSError, ValueError, RuntimeError, urllib.error.URLError) as exc:
print(json.dumps({"error": str(exc)}, sort_keys=True))
return 1
print(json.dumps(result, indent=2, sort_keys=True))
return 0
if __name__ == "__main__": # pragma: no cover - exercised through main()
raise SystemExit(main())

View File

@ -0,0 +1,273 @@
#!/usr/bin/env python3
"""Render and revalidate the two-workload Hermes WebUI Flux handoff."""
from __future__ import annotations
import difflib
import json
import re
from pathlib import Path
from typing import Any
DEFAULT_IMAGE = "registry.bstein.dev/bstein/hermes-webui"
DIGEST_PATTERN = re.compile(r"^sha256:[0-9a-f]{64}$")
REVISION_PATTERN = re.compile(r"^[0-9a-f]{40}$")
BUILD_PATTERN = re.compile(r"^[1-9][0-9]*$")
DESTINATION_PATTERN = re.compile(
r"^registry\.bstein\.dev/bstein/hermes-webui:"
r"git-([0-9a-f]{40})-build-([1-9][0-9]*)$"
)
def validated(value: str, pattern: re.Pattern[str], label: str) -> str:
"""Return a normalized value when it matches the release contract."""
normalized = value.strip()
if not pattern.fullmatch(normalized):
raise ValueError(f"invalid {label}: expected {pattern.pattern}")
return normalized
def validate_destination(
destination: str, source_revision: str, build_number: str
) -> tuple[str, str]:
"""Bind one unique build tag to the reviewed revision and Jenkins build."""
revision = validated(source_revision, REVISION_PATTERN, "source revision")
build = validated(build_number, BUILD_PATTERN, "build number")
match = DESTINATION_PATTERN.fullmatch(destination.strip())
if not match or match.groups() != (revision, build):
raise ValueError(
"destination must bind the reviewed revision and unique Jenkins build"
)
return revision, build
def render_workload(
source: str,
digest: str,
*,
kind: str,
name: str,
image: str = DEFAULT_IMAGE,
expected_images: int | tuple[int, ...] = 1,
) -> str:
"""Replace every expected WebUI consumer in one exact Flux workload."""
digest = validated(digest, DIGEST_PATTERN, "image digest")
identity = re.compile(
rf"\A(?:#[^\n]*\n)*apiVersion: apps/v1\nkind: {re.escape(kind)}\n"
rf"metadata:\n name: {re.escape(name)}\n"
)
if not identity.search(source):
raise ValueError(f"Flux target identity changed: expected {kind}/{name}")
lines = source.splitlines(keepends=True)
matches: list[int] = []
suffixes: dict[int, str] = {}
for index, line in enumerate(lines):
stripped = line.strip()
if not stripped.startswith("image: "):
continue
value, separator, comment = stripped.removeprefix("image: ").partition(" #")
current_image, at, current_digest = value.rpartition("@")
if not at or not re.fullmatch(
rf"{re.escape(image)}(?::[A-Za-z0-9_][A-Za-z0-9_.-]{{0,127}})?",
current_image,
):
continue
validated(current_digest, DIGEST_PATTERN, "current Flux image digest")
matches.append(index)
suffixes[index] = f" #{comment}" if separator else ""
allowed = (expected_images,) if isinstance(expected_images, int) else expected_images
if not allowed or any(count < 1 for count in allowed) or len(matches) not in allowed:
raise ValueError(
f"expected {allowed} {image!r} image(s) in {kind}/{name}; "
f"found {len(matches)}"
)
for index in matches:
newline = "\n" if lines[index].endswith("\n") else ""
prefix = lines[index][: len(lines[index]) - len(lines[index].lstrip())]
lines[index] = f"{prefix}image: {image}@{digest}{suffixes[index]}{newline}"
return "".join(lines)
def render_hux_build_metadata(
source: str, digest: str, source_revision: str, build_number: str
) -> str:
"""Bind HUX metadata when the activated sidecar fields are present."""
digest = validated(digest, DIGEST_PATTERN, "image digest")
revision = validated(source_revision, REVISION_PATTERN, "source revision")
build = validated(build_number, BUILD_PATTERN, "build number")
replacements = {
"HUX_IMAGE_TAG": f"git-{revision}-build-{build}-release",
"HUX_IMAGE_DIGEST": digest,
}
rendered = source
# Block style only: Flux setters cannot attach to values inside flow
# mappings, so the manifest keeps these as two-line entries with the
# marker comment on the value scalar. The renderer stays a belt on top
# of the Flux :tag/:digest setters and binds the same values.
present = {
name: rendered.count(f"- name: {name}\n") for name in replacements
}
if set(present.values()) == {0}:
return rendered
if any(count != 1 for count in present.values()):
raise ValueError(f"incomplete HUX build binding fields: {present}")
for name, value in replacements.items():
pattern = re.compile(
rf"^(?P<head>(?P<indent>\s*)- name: {name}\n(?P=indent) value: )"
rf"[^#\n]+(?P<suffix> #[^\n]*)?$",
re.MULTILINE,
)
rendered, count = pattern.subn(
lambda match: (
f"{match.group('head')}{value}{match.group('suffix') or ''}"
),
rendered,
)
if count != 1:
raise ValueError(f"expected exactly one {name} HUX build binding; found {count}")
return rendered
def _targets(chat_manifest: Path, dashboard_manifest: Path):
return (
(
chat_manifest,
"StatefulSet",
"hermes-chat-tenant",
"hermes-chat-statefulset.yaml",
(1, 2, 3),
),
(
dashboard_manifest,
"Deployment",
"hermes",
"hermes-dashboard-deployment.yaml",
1,
),
)
def _metadata(
digest: str, source_revision: str, build_number: str, destination: str
) -> dict[str, Any]:
return {
"build_number": build_number,
"digest": digest,
"flux_image": f"{DEFAULT_IMAGE}@{digest}",
"flux_targets": [
"apps/StatefulSet/hermes/hermes-chat-tenant",
"apps/Deployment/hermes/hermes",
],
"image": DEFAULT_IMAGE,
"published_tag": destination,
"source_revision": source_revision,
}
def _rendered_and_patch(
digest: str,
source_revision: str,
build_number: str,
chat_manifest: Path,
dashboard_manifest: Path,
) -> tuple[dict[str, str], str]:
rendered_targets: dict[str, str] = {}
patch_parts: list[str] = []
for path, kind, name, artifact_name, expected_images in _targets(chat_manifest, dashboard_manifest):
source = path.read_text(encoding="utf-8")
rendered = render_workload(
source,
digest,
kind=kind,
name=name,
expected_images=expected_images,
)
if name == "hermes-chat-tenant":
rendered = render_hux_build_metadata(
rendered, digest, source_revision, build_number
)
rendered_targets[artifact_name] = rendered
patch_parts.append(
"".join(
difflib.unified_diff(
source.splitlines(keepends=True),
rendered.splitlines(keepends=True),
fromfile=f"a/services/hermes/{path.name}",
tofile=f"b/services/hermes/{path.name}",
)
)
)
return rendered_targets, "".join(patch_parts)
def write_release_artifacts(
*,
digest: str,
source_revision: str,
build_number: str,
destination: str,
chat_manifest: Path,
dashboard_manifest: Path,
output_dir: Path,
) -> dict[str, Any]:
"""Write two rendered Flux targets, one patch, and credential-free evidence."""
digest = validated(digest, DIGEST_PATTERN, "image digest")
source_revision, build_number = validate_destination(
destination, source_revision, build_number
)
rendered_targets, patch = _rendered_and_patch(
digest, source_revision, build_number, chat_manifest, dashboard_manifest
)
if not patch:
raise ValueError("published digest already matches every Flux target")
output_dir.mkdir(parents=True, exist_ok=True)
for artifact_name, rendered in rendered_targets.items():
(output_dir / artifact_name).write_text(rendered, encoding="utf-8")
(output_dir / "hermes-webui-image-update.patch").write_text(patch, encoding="utf-8")
metadata = _metadata(digest, source_revision, build_number, destination)
(output_dir / "hermes-webui-image.json").write_text(
json.dumps(metadata, indent=2, sort_keys=True) + "\n", encoding="utf-8"
)
return metadata
def validate_release_artifacts(
*,
digest: str,
source_revision: str,
build_number: str,
destination: str,
chat_manifest: Path,
dashboard_manifest: Path,
output_dir: Path,
) -> None:
"""Revalidate the exact successful-build evidence without rewriting it."""
digest = validated(digest, DIGEST_PATTERN, "image digest")
source_revision, build_number = validate_destination(
destination, source_revision, build_number
)
expected_names = {
"hermes-webui-image.json",
"hermes-webui-image-update.patch",
"hermes-chat-statefulset.yaml",
"hermes-dashboard-deployment.yaml",
}
entries = list(output_dir.iterdir())
if {entry.name for entry in entries} != expected_names or not all(
entry.is_file() and not entry.is_symlink() for entry in entries
):
raise ValueError("release output must contain exactly four evidence files")
rendered_targets, patch = _rendered_and_patch(
digest, source_revision, build_number, chat_manifest, dashboard_manifest
)
metadata = _metadata(digest, source_revision, build_number, destination)
expected = {
"hermes-webui-image.json": json.dumps(metadata, indent=2, sort_keys=True)
+ "\n",
"hermes-webui-image-update.patch": patch,
**rendered_targets,
}
for name, expected_text in expected.items():
if (output_dir / name).read_text(encoding="utf-8") != expected_text:
raise ValueError(f"release evidence is incomplete or mismatched: {name}")

View File

@ -0,0 +1,469 @@
#!/usr/bin/env python3
"""Verify and render a reviewable Hermes WebUI image release."""
from __future__ import annotations
import argparse
import base64
import json
import os
import urllib.error
import urllib.parse
import urllib.request
from pathlib import Path
from typing import Any, Callable
from hermes_webui_flux_release import (
DEFAULT_IMAGE as DEFAULT_IMAGE,
DESTINATION_PATTERN,
DIGEST_PATTERN,
REVISION_PATTERN,
render_hux_build_metadata as render_hux_build_metadata,
render_workload as render_workload,
validate_destination,
validate_release_artifacts as validate_flux_release_artifacts,
validated as _validated,
write_release_artifacts,
)
HARBOR_API_ORIGIN = "https://registry.bstein.dev/api/v2.0"
HARBOR_PROJECT = "bstein"
HARBOR_REPOSITORY = "hermes-webui"
IMMUTABLE_REPOSITORY_PATTERN = "hermes-webui"
IMMUTABLE_TAG_PATTERN = "git-*-build-*"
class _NoRedirect(urllib.request.HTTPRedirectHandler):
"""Never send registry credentials to a redirect target."""
def redirect_request(self, _request, _file, _code, _message, _headers, _url):
return None
def validate_kaniko_evidence(
*, digest_text: str, image_text: str, destination: str
) -> str:
"""Cross-check both independent Kaniko output files against the destination."""
digest_lines = digest_text.splitlines()
image_lines = image_text.splitlines()
if len(digest_lines) != 1:
raise ValueError("Kaniko digest evidence must contain exactly one line")
if len(image_lines) != 1:
raise ValueError("Kaniko image evidence must contain exactly one line")
digest = _validated(digest_lines[0], DIGEST_PATTERN, "image digest")
if image_lines[0].strip() != f"{destination}@{digest}":
raise ValueError("Kaniko image evidence does not match destination and digest")
return digest
def _registry_request(request: urllib.request.Request, timeout: int) -> Any:
"""Make a registry request without following redirects."""
opener = urllib.request.build_opener(_NoRedirect())
try:
return opener.open(request, timeout=timeout)
except urllib.error.HTTPError as exc:
return exc
def _artifact_response(
destination: str,
*,
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any] = _registry_request,
) -> tuple[int, bytes]:
"""Read one exact Harbor artifact by tag with bounded response size."""
match = DESTINATION_PATTERN.fullmatch(destination)
if not match:
raise ValueError("invalid destination")
if not username or not password:
raise RuntimeError("Harbor credentials are empty")
tag = destination.rsplit(":", 1)[1]
encoded_tag = urllib.parse.quote(tag, safe="")
auth = base64.b64encode(f"{username}:{password}".encode()).decode("ascii")
request = urllib.request.Request(
f"{HARBOR_API_ORIGIN}/projects/{HARBOR_PROJECT}/repositories/"
f"{HARBOR_REPOSITORY}/artifacts/{encoded_tag}"
"?with_immutable_status=true",
headers={"Accept": "application/json", "Authorization": f"Basic {auth}"},
method="GET",
)
with opener(request, 20) as response:
body = response.read(1_048_577)
if len(body) > 1_048_576:
raise RuntimeError("Harbor artifact response exceeded the size limit")
return int(response.status), body
def _immutable_rules_response(
*,
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any] = _registry_request,
) -> tuple[int, bytes, dict[str, str]]:
"""Read the project policy with the same least-privilege publish identity."""
if not username or not password:
raise RuntimeError("Harbor credentials are empty")
auth = base64.b64encode(f"{username}:{password}".encode()).decode("ascii")
request = urllib.request.Request(
f"{HARBOR_API_ORIGIN}/projects/{HARBOR_PROJECT}/immutabletagrules"
"?page=1&page_size=100",
headers={"Accept": "application/json", "Authorization": f"Basic {auth}"},
method="GET",
)
with opener(request, 20) as response:
body = response.read(1_048_577)
if len(body) > 1_048_576:
raise RuntimeError("Harbor immutable rule response exceeded the size limit")
return int(response.status), body, dict(response.headers)
def _require_complete_rule_page(
rules: list[dict[str, Any]], headers: dict[str, str]
) -> None:
"""Require proof that the bounded first page contains every rule."""
raw_total = next(
(value for key, value in headers.items() if key.lower() == "x-total-count"),
None,
)
if raw_total is None or not str(raw_total).isdecimal():
raise RuntimeError("Harbor immutable rule list omitted a valid total count")
if int(raw_total) != len(rules):
raise RuntimeError("Harbor immutable rule list was truncated")
def _normalized_immutable_rule(rule: dict[str, Any]) -> dict[str, Any]:
"""Select only fields that bind the server-side build-tag policy."""
return {
"disabled": bool(rule.get("disabled", False)),
"action": rule.get("action"),
"template": rule.get("template"),
"tag_selectors": [
{
"kind": item.get("kind"),
"decoration": item.get("decoration"),
"pattern": item.get("pattern"),
}
for item in rule.get("tag_selectors") or []
if isinstance(item, dict)
],
"scope_selectors": {
"repository": [
{
"kind": item.get("kind"),
"decoration": item.get("decoration"),
"pattern": item.get("pattern"),
}
for item in (rule.get("scope_selectors") or {}).get("repository", [])
if isinstance(item, dict)
]
},
}
def verify_immutable_policy(
*,
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any] = _registry_request,
) -> None:
"""Fail closed before build unless the exact Harbor rule is active."""
status, body, headers = _immutable_rules_response(
username=username, password=password, opener=opener
)
if status != 200:
raise RuntimeError(f"Harbor immutable policy preflight returned HTTP {status}")
try:
rules = json.loads(body.decode("utf-8"))
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
raise RuntimeError("Harbor returned invalid immutable rule JSON") from exc
if not isinstance(rules, list) or not all(isinstance(item, dict) for item in rules):
raise RuntimeError("Harbor immutable rule list has an invalid shape")
_require_complete_rule_page(rules, headers)
expected = {
"disabled": False,
"action": "immutable",
"template": "immutable_template",
"tag_selectors": [
{
"kind": "doublestar",
"decoration": "matches",
"pattern": IMMUTABLE_TAG_PATTERN,
}
],
"scope_selectors": {
"repository": [
{
"kind": "doublestar",
"decoration": "repoMatches",
"pattern": IMMUTABLE_REPOSITORY_PATTERN,
}
]
},
}
matches = [
_normalized_immutable_rule(item)
for item in rules
if _normalized_immutable_rule(item)["tag_selectors"]
== expected["tag_selectors"]
and _normalized_immutable_rule(item)["scope_selectors"]
== expected["scope_selectors"]
]
if matches != [expected]:
raise RuntimeError("Harbor immutable build-tag policy is absent or not exact")
def assert_tag_absent(
destination: str,
*,
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any] = _registry_request,
) -> None:
"""Reject replay before Kaniko can push an already-used immutable identity."""
status, _body = _artifact_response(
destination, username=username, password=password, opener=opener
)
if status == 404:
return
if status == 200:
raise RuntimeError("Harbor destination tag already exists; refusing overwrite")
raise RuntimeError(f"Harbor destination preflight returned HTTP {status}")
OCI_REVISION_LABEL = "org.opencontainers.image.revision"
def _child_artifact_response(
child_digest: str,
*,
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any] = _registry_request,
) -> tuple[int, bytes]:
"""Read one per-arch child artifact of a multi-arch index by its digest."""
if not username or not password:
raise RuntimeError("Harbor credentials are empty")
encoded = urllib.parse.quote(child_digest, safe="")
auth = base64.b64encode(f"{username}:{password}".encode()).decode("ascii")
request = urllib.request.Request(
f"{HARBOR_API_ORIGIN}/projects/{HARBOR_PROJECT}/repositories/"
f"{HARBOR_REPOSITORY}/artifacts/{encoded}"
"?with_immutable_status=true",
headers={"Accept": "application/json", "Authorization": f"Basic {auth}"},
method="GET",
)
with opener(request, 20) as response:
body = response.read(1_048_577)
if len(body) > 1_048_576:
raise RuntimeError("Harbor child artifact response exceeded the size limit")
return int(response.status), body
def _config_labels(artifact: dict[str, Any]) -> Any:
"""Extract the OCI config labels Harbor reports for one artifact, if any."""
return ((artifact.get("extra_attrs") or {}).get("config") or {}).get("Labels")
def _verify_source_revision_label(
artifact: dict[str, Any],
source_revision: str,
*,
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any],
) -> None:
"""Assert the published image carries the reviewed source revision label.
A single-arch image exposes ``org.opencontainers.image.revision`` on its own
config. A multi-arch manifest list has no top-level config, so Harbor reports
the label on each per-arch child instead; verify every child in that case.
"""
labels = _config_labels(artifact)
if isinstance(labels, dict):
if labels.get(OCI_REVISION_LABEL) != source_revision:
raise RuntimeError("Harbor OCI source-revision label does not match")
return
references = artifact.get("references")
if not isinstance(references, list) or not references:
raise RuntimeError("Harbor artifact omitted OCI image labels")
for reference in references:
child_digest = str((reference or {}).get("child_digest") or "").strip()
if not DIGEST_PATTERN.fullmatch(child_digest):
raise RuntimeError("Harbor index reference omitted a valid child digest")
status, body = _child_artifact_response(
child_digest, username=username, password=password, opener=opener
)
if status != 200:
raise RuntimeError(
f"Harbor child artifact verification returned HTTP {status}"
)
try:
child = json.loads(body.decode("utf-8"))
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
raise RuntimeError("Harbor returned invalid child artifact JSON") from exc
child_labels = _config_labels(child)
if not isinstance(child_labels, dict):
raise RuntimeError("Harbor artifact omitted OCI image labels")
if child_labels.get(OCI_REVISION_LABEL) != source_revision:
raise RuntimeError("Harbor OCI source-revision label does not match")
def verify_registry_digest(
destination: str,
digest: str,
source_revision: str,
*,
username: str,
password: str,
opener: Callable[[urllib.request.Request, int], Any] = _registry_request,
) -> None:
"""Verify Harbor resolves the tag, digest, and persisted source revision."""
digest = _validated(digest, DIGEST_PATTERN, "image digest")
source_revision = _validated(
source_revision, REVISION_PATTERN, "source revision"
)
status, body = _artifact_response(
destination, username=username, password=password, opener=opener
)
if status != 200:
raise RuntimeError(f"Harbor manifest verification returned HTTP {status}")
try:
artifact = json.loads(body.decode("utf-8"))
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
raise RuntimeError("Harbor returned invalid artifact JSON") from exc
harbor_digest = str(artifact.get("digest") or "").strip()
if not DIGEST_PATTERN.fullmatch(harbor_digest):
raise RuntimeError("Harbor response omitted a valid artifact digest")
if harbor_digest != digest:
raise RuntimeError("Harbor digest does not match Kaniko evidence")
expected_tag = destination.rsplit(":", 1)[1]
matching_tags = [
item
for item in artifact.get("tags") or []
if isinstance(item, dict) and item.get("name") == expected_tag
]
if len(matching_tags) != 1:
raise RuntimeError("Harbor artifact does not contain the expected tag")
if matching_tags[0].get("immutable") is not True:
raise RuntimeError("Harbor did not enforce the expected tag as immutable")
_verify_source_revision_label(
artifact,
source_revision,
username=username,
password=password,
opener=opener,
)
def validate_release_artifacts(
*,
digest_file: Path,
image_file: Path,
source_revision: str,
build_number: str,
destination: str,
chat_manifest: Path,
dashboard_manifest: Path,
output_dir: Path,
) -> None:
"""Revalidate the exact successful-build evidence without rewriting it."""
digest = validate_kaniko_evidence(
digest_text=digest_file.read_text(encoding="utf-8"),
image_text=image_file.read_text(encoding="utf-8"),
destination=destination,
)
validate_flux_release_artifacts(
digest=digest,
source_revision=source_revision,
build_number=build_number,
destination=destination,
chat_manifest=chat_manifest,
dashboard_manifest=dashboard_manifest,
output_dir=output_dir,
)
def _credentials() -> tuple[str, str]:
"""Read the masked, runtime-only Jenkins credential environment."""
username = os.environ.get("HARBOR_USER", "")
password = os.environ.get("HARBOR_PASSWORD", "")
if not username or not password:
raise RuntimeError("Harbor credentials are unavailable")
return username, password
def _common_arguments(parser: argparse.ArgumentParser) -> None:
parser.add_argument("--source-revision", required=True)
parser.add_argument("--build-number", required=True)
parser.add_argument("--destination", required=True)
def main() -> int:
"""Fail closed around the unique tag, then verify and render after push."""
parser = argparse.ArgumentParser()
commands = parser.add_subparsers(dest="command", required=True)
absent = commands.add_parser("assert-absent")
_common_arguments(absent)
render = commands.add_parser("render")
_common_arguments(render)
render.add_argument("--digest-file", required=True, type=Path)
render.add_argument("--image-file", required=True, type=Path)
render.add_argument("--chat-manifest", required=True, type=Path)
render.add_argument("--dashboard-manifest", required=True, type=Path)
render.add_argument("--output-dir", required=True, type=Path)
verify = commands.add_parser("verify-evidence")
_common_arguments(verify)
verify.add_argument("--digest-file", required=True, type=Path)
verify.add_argument("--image-file", required=True, type=Path)
verify.add_argument("--chat-manifest", required=True, type=Path)
verify.add_argument("--dashboard-manifest", required=True, type=Path)
verify.add_argument("--output-dir", required=True, type=Path)
args = parser.parse_args()
validate_destination(args.destination, args.source_revision, args.build_number)
if args.command == "verify-evidence":
validate_release_artifacts(
digest_file=args.digest_file,
image_file=args.image_file,
source_revision=args.source_revision,
build_number=args.build_number,
destination=args.destination,
chat_manifest=args.chat_manifest,
dashboard_manifest=args.dashboard_manifest,
output_dir=args.output_dir,
)
return 0
username, password = _credentials()
if args.command == "assert-absent":
verify_immutable_policy(username=username, password=password)
assert_tag_absent(args.destination, username=username, password=password)
return 0
digest = validate_kaniko_evidence(
digest_text=args.digest_file.read_text(encoding="utf-8"),
image_text=args.image_file.read_text(encoding="utf-8"),
destination=args.destination,
)
verify_registry_digest(
args.destination,
digest,
args.source_revision,
username=username,
password=password,
)
write_release_artifacts(
digest=digest,
source_revision=args.source_revision,
build_number=args.build_number,
destination=args.destination,
chat_manifest=args.chat_manifest,
dashboard_manifest=args.dashboard_manifest,
output_dir=args.output_dir,
)
return 0
if __name__ == "__main__": # pragma: no cover - exercised through main()
raise SystemExit(main())

View File

@ -3,6 +3,7 @@
from __future__ import annotations
import hashlib
import json
import os
from glob import glob
@ -27,6 +28,8 @@ _infer_workspace_coverage_percent = _quality_helpers._infer_workspace_coverage_p
_load_optional_json = _quality_helpers._load_optional_json
_normalize_result_status = _quality_helpers._normalize_result_status
TEST_CASE_LABEL_MAX_BYTES = 240
def _escape_label(value: str) -> str:
"""Escape a Prometheus label value without changing its content."""
@ -39,6 +42,17 @@ def _label_str(labels: dict[str, str]) -> str:
return "{" + ",".join(parts) + "}" if parts else ""
def _bounded_test_name(value: str) -> str:
"""Keep test labels readable, unique, and safely below scraper limits."""
encoded = value.encode("utf-8")
if len(encoded) <= TEST_CASE_LABEL_MAX_BYTES:
return value
digest = hashlib.sha256(encoded).hexdigest()[:16]
suffix = f"...[sha256:{digest}]"
prefix = encoded[: TEST_CASE_LABEL_MAX_BYTES - len(suffix)]
return prefix.decode("utf-8", errors="ignore") + suffix
def _read_text(url: str) -> str:
"""Fetch a plain-text response body from the given URL."""
with urllib.request.urlopen(url, timeout=10) as response:
@ -121,7 +135,7 @@ def _collect_junit_cases(pattern: str) -> list[tuple[str, str]]:
status = "failed"
elif test_case.find("skipped") is not None:
status = "skipped"
cases.append((full_name, status))
cases.append((_bounded_test_name(full_name), status))
return cases
@ -246,7 +260,7 @@ def _build_payload(
for test_name, test_status in test_cases:
labels = {
**test_case_base_labels,
"test": test_name,
"test": _bounded_test_name(test_name),
"status": test_status,
}
lines.append(

Binary file not shown.

Binary file not shown.

Binary file not shown.

Binary file not shown.

View File

@ -5,7 +5,6 @@ metadata:
name: bstein-dev-home-migrations
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Migration jobs run only during portal schema changes."
spec:
interval: 10m

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: bstein-dev-home
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./services/bstein-dev-home

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: cassandra-auth
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./services/cassandra-auth

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: cassandra
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./services/cassandra

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: comms
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
prune: true

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: crypto
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./services/crypto

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: finance
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Finance apps are parked until the next storage and SSO pass."
spec:
interval: 10m
suspend: true
suspend: false
path: ./services/finance
prune: true
sourceRef:

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: game-stream
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Game streaming is optional and resumes only for planned use."
spec:
interval: 10m
suspend: true
suspend: false
path: ./services/game-stream
targetNamespace: game-stream
prune: true

View File

@ -1,11 +1,9 @@
# clusters/atlas/flux-system/applications/gitea/kustomization.yaml
apiVersion: kustomize.toolkit.fluxcd.io/v1beta2
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: gitea
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./services/gitea

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: harbor
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./services/harbor
@ -15,12 +13,20 @@ spec:
kind: GitRepository
name: flux-system
namespace: flux-system
wait: true
wait: false
timeout: 10m
healthChecks:
- apiVersion: batch/v1
kind: Job
name: harbor-hermes-agent-immutability-ensure-1
- apiVersion: apps/v1
kind: Deployment
name: harbor-core
namespace: harbor
- apiVersion: apps/v1
kind: Deployment
name: harbor-registry
namespace: harbor
- apiVersion: apps/v1
kind: Deployment
name: harbor-jobservice
namespace: harbor
dependsOn:
- name: core

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: health
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Health stack is staged pending account and mail flows."
spec:
interval: 10m
suspend: true
suspend: false
path: ./services/health
prune: true
sourceRef:

View File

@ -1,19 +0,0 @@
# clusters/atlas/flux-system/applications/hermes-scm-broker-code/kustomization.yaml
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: hermes-scm-broker-code
namespace: flux-system
spec:
interval: 10m
path: ./services/hermes/scm-common
targetNamespace: hermes-scm
prune: true
sourceRef:
kind: GitRepository
name: flux-system
namespace: flux-system
wait: true
timeout: 5m
dependsOn:
- name: hermes-scm-namespace

View File

@ -24,4 +24,3 @@ spec:
- name: vault
- name: gitea
- name: hermes-scm-namespace
- name: hermes-scm-broker-code

View File

@ -0,0 +1,26 @@
# clusters/atlas/flux-system/applications/hermes/image-automation.yaml
apiVersion: image.toolkit.fluxcd.io/v1
kind: ImageUpdateAutomation
metadata:
name: hermes
namespace: hermes
spec:
interval: 1m0s
sourceRef:
kind: GitRepository
name: flux-system
namespace: flux-system
git:
checkout:
ref:
branch: main
commit:
author:
email: ops@bstein.dev
name: flux-bot
messageTemplate: "chore(hermes): promote validated image release"
push:
branch: main
update:
strategy: Setters
path: services/hermes

View File

@ -26,6 +26,10 @@ spec:
kind: Deployment
name: hermes-switchyard
namespace: hermes
- apiVersion: apps/v1
kind: Deployment
name: hermes-suite-planner
namespace: hermes
- apiVersion: apps/v1
kind: Deployment
name: hermes-agent
@ -36,10 +40,9 @@ spec:
# would stall hermes-chat and hermes-observer-bindings, which dependsOn
# hermes. The pool reports its own health through the worker readiness probe,
# the per-ordinal mediator /ready endpoint, and the coordinator on :9007.
- apiVersion: apps/v1
kind: DaemonSet
name: hermes-node-ssh-access
namespace: hermes
# The node SSH hardener is deliberately absent. It reconciles every node,
# including offline and maintenance hosts, so its availability must not
# turn a node-local account issue into a blocked owner-service rollout.
- apiVersion: apps/v1
kind: Deployment
name: hermes
@ -66,6 +69,5 @@ spec:
- name: keycloak
- name: longhorn
- name: vault
- name: jenkins
- name: hermes-scm-broker
# CI availability must not block recovery of the inference services.
- name: hermes-observer-rbac

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: jellyfin
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Media stack stays paused while auth and storage changes are staged."
spec:
interval: 10m
suspend: true
suspend: false
path: ./services/jellyfin
targetNamespace: jellyfin
prune: true

View File

@ -5,7 +5,7 @@ metadata:
name: jenkins
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
kustomize.toolkit.fluxcd.io/ssa: Merge
spec:
interval: 10m
suspend: false
@ -29,4 +29,4 @@ spec:
name: jenkins
namespace: jenkins
wait: false
timeout: 20m
timeout: 5m

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: keycloak
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
prune: true

View File

@ -29,8 +29,8 @@ resources:
- ai-llm/kustomization.yaml
- openclaw/kustomization.yaml
- hermes-scm-namespace/kustomization.yaml
- hermes-scm-broker-code/kustomization.yaml
- hermes/kustomization.yaml
- hermes/image-automation.yaml
- hermes-observer-rbac/kustomization.yaml
- hermes-observer-bindings/kustomization.yaml
- hermes-scm-broker/kustomization.yaml
@ -40,7 +40,6 @@ resources:
- cassandra-auth/kustomization.yaml
- cassandra/kustomization.yaml
- cassandra/image-automation.yaml
- veles/kustomization.yaml
- typhon/kustomization.yaml
- nextcloud/kustomization.yaml
- nextcloud-mail-sync/kustomization.yaml

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: mailu
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Mail stack is staged and resumes only during mail rollout work."
spec:
interval: 10m
suspend: true
suspend: false
sourceRef:
kind: GitRepository
name: flux-system

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: monerod
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./services/crypto/monerod

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: nextcloud-mail-sync
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Mail sync resumes after Nextcloud and Mailu are active."
spec:
interval: 10m
suspend: true
suspend: false
prune: true
sourceRef:
kind: GitRepository

View File

@ -1,15 +1,12 @@
# clusters/atlas/flux-system/applications/nextcloud/kustomization.yaml
apiVersion: kustomize.toolkit.fluxcd.io/v1beta2
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization
metadata:
name: nextcloud
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Nextcloud is staged pending storage, SSO, and mail validation."
spec:
interval: 10m
suspend: true
suspend: false
path: ./services/nextcloud
targetNamespace: nextcloud
prune: true

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: oauth2-proxy
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
prune: true

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: openldap
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
prune: true

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: outline
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Outline is staged until shared auth and mail are ready."
spec:
interval: 10m
suspend: true
suspend: false
path: ./services/outline
prune: true
sourceRef:

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: pegasus
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./services/pegasus

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: planka
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Planka is staged until shared auth and mail are ready."
spec:
interval: 10m
suspend: true
suspend: false
path: ./services/planka
prune: true
sourceRef:

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: quality
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Quality stack changes resume only during quality-gate rollout work."
spec:
interval: 10m
suspend: true
suspend: false
path: ./services/quality
prune: true
sourceRef:

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: sui-metrics
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./services/sui-metrics/overlays/atlas

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: typhon
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Climate automation is staged until sensor and control loops are verified."
spec:
interval: 10m
suspend: true
suspend: false
path: ./services/typhon
prune: true
sourceRef:

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: vault-hermes-jenkins-token-seed
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
sourceRef:

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: vault
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
sourceRef:
@ -15,11 +13,15 @@ spec:
path: ./services/vault
targetNamespace: vault
prune: true
wait: true
wait: false
healthChecks:
- apiVersion: batch/v1
kind: Job
name: vault-k8s-auth-hermes-10
- apiVersion: apps/v1
kind: StatefulSet
name: vault
namespace: vault
- apiVersion: apps/v1
kind: Deployment
name: vault-injector-agent-injector
namespace: vault
dependsOn:
- name: longhorn

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: vaultwarden
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Vaultwarden stays separate from SSO rollout and resumes by hand."
spec:
interval: 10m
suspend: true
suspend: false
sourceRef:
kind: GitRepository
name: flux-system

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: veles
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./services/veles

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: wallet-monero-temp
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Temporary wallet RPC stack is opt-in for maintenance windows."
spec:
interval: 10m
suspend: true
suspend: false
path: ./services/crypto/wallet-monero-temp
targetNamespace: crypto
prune: true

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: xmr-miner
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Mining workloads are disabled unless explicitly enabled for a short run."
spec:
interval: 10m
suspend: true
suspend: false
path: ./services/crypto/xmr-miner
targetNamespace: crypto
prune: true

View File

@ -12,7 +12,7 @@ spec:
branch: main
secretRef:
name: flux-system-gitea
url: ssh://git@scm.bstein.dev:2242/atlas/titan-iac.git
url: ssh://git@scm.bstein.dev:2242/titan/atlas-iac.git
---
apiVersion: kustomize.toolkit.fluxcd.io/v1
kind: Kustomization

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: cert-manager-cleanup
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 30m
path: ./infrastructure/cert-manager/cleanup

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: cert-manager
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 30m
path: ./infrastructure/cert-manager

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: core
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./infrastructure/core

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: descheduler
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Descheduler is paused during node recovery and placement stabilization."
spec:
interval: 30m
suspend: true
suspend: false
path: ./infrastructure/descheduler
prune: true
sourceRef:

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: gitops-ui
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "GitOps UI is optional and resumes only for operator access work."
spec:
interval: 10m
suspend: true
suspend: false
timeout: 10m
path: ./services/gitops-ui
prune: true

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: helm
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 30m
sourceRef:

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: logging
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./services/logging

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: longhorn-adopt
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 30m
path: ./infrastructure/longhorn/adopt

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: longhorn-ui
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Longhorn UI ingress is optional and opened only for storage work."
spec:
interval: 10m
suspend: true
suspend: false
path: ./infrastructure/longhorn/ui-ingress
targetNamespace: longhorn-system
prune: true

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: longhorn
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 30m
path: ./infrastructure/longhorn/core

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: maintenance
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./services/maintenance

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: metallb
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 30m
sourceRef:

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: monitoring
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./services/monitoring

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: postgres
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./infrastructure/postgres

View File

@ -4,12 +4,9 @@ kind: Kustomization
metadata:
name: resource-guardrails
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
atlas.bstein.dev/suspend-reason: "Guardrail rollout is paused until service resource requests are normalized."
spec:
interval: 10m
suspend: true
suspend: false
path: ./infrastructure/resource-guardrails
prune: true
sourceRef:

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: traefik
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 10m
path: ./infrastructure/traefik

View File

@ -5,7 +5,7 @@ metadata:
name: vault-csi
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
kustomize.toolkit.fluxcd.io/ssa: Override
spec:
interval: 30m
sourceRef:
@ -14,5 +14,23 @@ spec:
namespace: flux-system
path: ./infrastructure/vault-csi
prune: true
wait: true
# A cordoned, unavailable node must not block unrelated workload rollouts.
# Keep the CSI driver gate and require at least one observed healthy provider.
wait: false
healthChecks:
- apiVersion: helm.toolkit.fluxcd.io/v2
kind: HelmRelease
name: secrets-store-csi-driver
namespace: kube-system
- apiVersion: apps/v1
kind: DaemonSet
name: vault-csi-provider
namespace: kube-system
healthCheckExprs:
- apiVersion: apps/v1
kind: DaemonSet
current: >-
has(status.observedGeneration) && has(metadata.generation) &&
status.observedGeneration >= metadata.generation &&
has(status.numberAvailable) && status.numberAvailable > 0
targetNamespace: kube-system

View File

@ -4,8 +4,6 @@ kind: Kustomization
metadata:
name: vault-injector
namespace: flux-system
annotations:
kustomize.toolkit.fluxcd.io/ssa: IfNotPresent
spec:
interval: 30m
path: ./infrastructure/vault-injector

View File

@ -1,6 +1,68 @@
# syntax=docker/dockerfile:1
# dockerfiles/Dockerfile.hermes-agent
FROM nousresearch/hermes-agent@sha256:47d4bd4cc420b70e40ed75efdade373e45b86b7382d4013a054208982bb6ba08
#
# Multi-arch base: this digest is the upstream OCI image INDEX for tag
# v2026.7.7.2 (revision 9de9c25f620ff7f1ce0fd5457d596052d5159596). The index
# fans out to both native leaves of the same reviewed upstream version:
# linux/arm64 -> sha256:47d4bd4cc420b70e40ed75efdade373e45b86b7382d4013a054208982bb6ba08
# linux/amd64 -> sha256:3db34ce19adfa080736a2a3feb0316dbcccc588faa9afe7fd8ae1c03b4f1a53a
# The arm64 leaf is byte-for-byte the previously pinned single-arch base, so the
# arm64 build is unchanged; Kaniko/containerd auto-selects the matching leaf per
# build platform (arm64 rpi5 pod vs amd64 titan-24 pod). Do NOT replace this with
# a per-arch leaf digest -- that would break the amd64 build leg.
#
# The reference below points at the in-cluster Harbor "mirror" project, NOT
# docker.io. The identical upstream index (same content-addressed digest
# 9c841866..., both leaves) is mirrored into Harbor with
# skopeo copy --all docker://nousresearch/hermes-agent@sha256:9c841866... \
# docker://harbor-core.harbor.svc.cluster.local/mirror/hermes-agent@sha256:9c841866...
# by the Flux-managed one-shot Job services/harbor/hermes-agent-base-mirror-job.yaml.
# Because the digest is content-addressed, mirroring reproduces the exact index
# and both leaf digests, so this remains fully digest-pinned and multi-arch while
# Kaniko pulls it over the internal insecure registry (see the pipeline's
# --insecure-registry/--registry-mirror flags) with no docker.io fallback.
# Re-run that Job whenever this digest is bumped, before publishing the image.
ARG HERMES_AGENT_BASE=harbor-core.harbor.svc.cluster.local/mirror/hermes-agent@sha256:9c841866021c54c4596849f6135717e8a4d52ba510b7f52c50aef1de1a283973
# Python 3.13 dynamically links libsqlite3.so.0, so a checked native library
# replaces the affected Debian 3.46.1 runtime without rebuilding Python. This
# target is also an emergency, small multi-arch layer when built with
# ``--target sqlite-runtime`` and HERMES_AGENT_BASE set to the current agent.
FROM ${HERMES_AGENT_BASE} AS sqlite-build
USER root
ARG SQLITE_AUTOCONF_VERSION=3510300
ARG SQLITE_AUTOCONF_SHA256=81f5be397049b0cae1b167f2225af7646fc0f82e4a9b3c48c9ea3a533e21d77a
ARG SQLITE_AMALGAMATION_SHA3=32d5424f97e0a7fc5ed2f6335afbb58be4e0298bd7117a34e39d345ff13d859e
RUN apt-get update \
&& apt-get install -y --no-install-recommends build-essential ca-certificates curl \
&& curl --fail --location --proto '=https' --tlsv1.2 \
--output /tmp/sqlite.tar.gz \
"https://www.sqlite.org/2026/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}.tar.gz" \
&& echo "${SQLITE_AUTOCONF_SHA256} /tmp/sqlite.tar.gz" | sha256sum --check --status \
&& tar -xzf /tmp/sqlite.tar.gz -C /tmp \
&& /opt/hermes/.venv/bin/python -c 'import hashlib, pathlib, sys; expected=sys.argv[1]; actual=hashlib.sha3_256(pathlib.Path(sys.argv[2]).read_bytes()).hexdigest(); raise SystemExit(actual != expected)' "${SQLITE_AMALGAMATION_SHA3}" "/tmp/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}/sqlite3.c" \
&& cd "/tmp/sqlite-autoconf-${SQLITE_AUTOCONF_VERSION}" \
&& CFLAGS='-DSQLITE_ENABLE_COLUMN_METADATA -DSQLITE_ENABLE_PREUPDATE_HOOK -DSQLITE_ENABLE_UNLOCK_NOTIFY -DSQLITE_ENABLE_UPDATE_DELETE_LIMIT -DSQLITE_SOUNDEX' \
./configure --prefix=/opt/sqlite --disable-static --all --scanstatus \
&& make -j"$(nproc)" \
&& make install \
&& test -L /opt/sqlite/lib/libsqlite3.so.0 \
&& /opt/sqlite/bin/sqlite3 ':memory:' 'select sqlite_version();' | grep -Fx '3.51.3'
FROM ${HERMES_AGENT_BASE} AS sqlite-runtime
USER root
COPY --from=sqlite-build /opt/sqlite/lib/ /usr/local/lib/
# Debian's arm64 multiarch path sorts ahead of libc.conf, which otherwise picks
# the older system SQLite before /usr/local/lib. Keep the checked library first.
RUN printf '%s\n' /usr/local/lib > /etc/ld.so.conf.d/00-hermes-sqlite.conf \
&& ldconfig \
&& /opt/hermes/.venv/bin/python -c 'import sqlite3; connection=sqlite3.connect(":memory:"); connection.execute("CREATE VIRTUAL TABLE documents USING fts5(body)"); connection.execute("CREATE VIRTUAL TABLE ranges USING rtree(id, low, high)"); assert connection.execute("SELECT json_valid(?), sqrt(9)", ("[]",)).fetchone() == (1, 3.0); options={row[0] for row in connection.execute("PRAGMA compile_options")}; required={"ENABLE_COLUMN_METADATA", "ENABLE_FTS4", "ENABLE_FTS5", "ENABLE_MATH_FUNCTIONS", "ENABLE_PREUPDATE_HOOK", "ENABLE_RTREE", "ENABLE_SESSION", "ENABLE_UNLOCK_NOTIFY", "ENABLE_UPDATE_DELETE_LIMIT"}; assert required <= options, sorted(required - options); assert sqlite3.sqlite_version_info >= (3, 51, 3), sqlite3.sqlite_version'
FROM sqlite-runtime
USER root
@ -1557,6 +1619,24 @@ function RootRedirect() {
NODE
RUN case "${HERMES_KANIKO_HEREDOC_COMPAT}" in 0) ;; 1) python /tmp/hermes-kaniko-heredoc-runner.py --dockerfile /tmp/hermes-agent.Dockerfile --block-index 9 ;; *) exit 2 ;; esac
# Reattachment reuses a live PTY, so a dashboard-only SIGCONT restores its
# alternate screen and current viewport after a bounded replay.
COPY dockerfiles/patch-hermes-terminal-replay.py /tmp/patch-hermes-terminal-replay.py
COPY dockerfiles/hermes-terminal-replay-regression.py /tmp/hermes-terminal-replay-regression.py
COPY dockerfiles/hermes-terminal-resume-regression.js /tmp/hermes-terminal-resume-regression.js
RUN /opt/hermes/.venv/bin/python /tmp/patch-hermes-terminal-replay.py \
&& HERMES_SOURCE_ROOT=/opt/hermes /opt/hermes/.venv/bin/python \
/tmp/hermes-terminal-replay-regression.py \
&& node /tmp/hermes-terminal-resume-regression.js \
&& node --check /opt/hermes/ui-tui/dist/entry.js
# The dashboard keeps keyboard and paste handling unchanged, but forwards only
# bounded SGR wheel reports when the TUI has explicitly enabled mouse tracking.
COPY dockerfiles/patch-hermes-dashboard-wheel.js /tmp/patch-hermes-dashboard-wheel.js
COPY dockerfiles/hermes-dashboard-terminal-input.ts /opt/hermes/web/src/lib/dashboard-terminal-input.ts
COPY dockerfiles/hermes-dashboard-terminal-input.test.ts /opt/hermes/web/src/lib/dashboard-terminal-input.test.ts
RUN node /tmp/patch-hermes-dashboard-wheel.js
COPY dockerfiles/patch-hermes-execution-safety.py /tmp/patch-hermes-execution-safety.py
COPY dockerfiles/hermes-execution-safety-regression.py /tmp/hermes-execution-safety-regression.py
COPY dockerfiles/hermes_execution_patch_support.py /tmp/hermes_execution_patch_support.py
@ -1566,7 +1646,7 @@ COPY dockerfiles/hermes_execution_regression_support.py /tmp/hermes_execution_re
COPY dockerfiles/hermes_run_safety_regression.py /tmp/hermes_run_safety_regression.py
COPY dockerfiles/hermes_decomposition_safety_regression.py /tmp/hermes_decomposition_safety_regression.py
COPY dockerfiles/hermes_lane_compatibility_regression.py /tmp/hermes_lane_compatibility_regression.py
COPY services/hermes/scripts/cli_lane_*.py /tmp/hermes-lane-regression/
COPY services/hermes/scripts/*.py /tmp/hermes-lane-regression/
RUN HERMES_CLI_LANE_SOURCE=/tmp/hermes-lane-regression \
HERMES_COMPATIBILITY_MODE=legacy \
/opt/hermes/.venv/bin/python /tmp/hermes_lane_compatibility_regression.py \
@ -1592,6 +1672,7 @@ COPY dockerfiles/hermes-session-migrate.py /opt/hermes/bin/hermes-session-migrat
RUN rm -f /tmp/hermes-agent.Dockerfile /tmp/hermes-kaniko-heredoc-runner.py
RUN cd /opt/hermes/web \
&& npm test -- src/lib/dashboard-terminal-input.test.ts \
&& npm run build \
&& grep -Fq 'await api.getSessions(1, 0' src/pages/ChatPage.tsx \
&& grep -Fq 'api.getSessions(1, 0' src/components/ChatSidebar.tsx \
@ -1623,6 +1704,8 @@ RUN cd /opt/hermes/web \
/opt/hermes/hermes_cli/web_server.py \
&& ! grep -Fq 'WebglAddon' src/pages/ChatPage.tsx \
&& grep -Fq 'scheduleTerminalPaint' src/pages/ChatPage.tsx \
&& ! grep -Fq 'attachCustomWheelEventHandler' src/pages/ChatPage.tsx \
&& grep -Fq 'isSgrWheelReport(data)' src/pages/ChatPage.tsx \
&& grep -Fq 'sessionTree.childrenByParent' \
src/components/ChatSessionList.tsx \
&& grep -Fq 'getSessions(SESSION_LIMIT, 0, scopeKey, "recent", true)' \

View File

@ -5,6 +5,12 @@
!dockerfiles/hermes-public-extract/**
!dockerfiles/hermes-session-activity-panel.tsx
!dockerfiles/hermes-session-migrate.py
!dockerfiles/patch-hermes-terminal-replay.py
!dockerfiles/hermes-terminal-replay-regression.py
!dockerfiles/hermes-terminal-resume-regression.js
!dockerfiles/patch-hermes-dashboard-wheel.js
!dockerfiles/hermes-dashboard-terminal-input.ts
!dockerfiles/hermes-dashboard-terminal-input.test.ts
!dockerfiles/Dockerfile.hermes-agent
!dockerfiles/hermes-kaniko-heredoc-runner.py
!dockerfiles/patch-hermes-execution-safety.py
@ -19,4 +25,4 @@
!services/
!services/hermes/
!services/hermes/scripts/
!services/hermes/scripts/cli_lane_*.py
!services/hermes/scripts/*.py

View File

@ -17,6 +17,9 @@ ADD --checksum=sha256:aff26ae408abcba5fbf8813c21e62b0941638c5f6eebfb145be0c98392
ADD --checksum=sha256:9ecf779972d90ba49c06d968637d720dd632c55bbf19d441fb42bf17a411e794 --chmod=0444 \
https://openaipublic.azureedge.net/main/whisper/models/9ecf779972d90ba49c06d968637d720dd632c55bbf19d441fb42bf17a411e794/small.pt \
/opt/models/whisper/small.pt
ADD --checksum=sha256:65147644a518d12f04e32d6f3b26facc3f8dd46e5390956a9424a650c0ce22b9 --chmod=0444 \
https://openaipublic.azureedge.net/main/whisper/models/65147644a518d12f04e32d6f3b26facc3f8dd46e5390956a9424a650c0ce22b9/tiny.pt \
/opt/models/whisper/tiny.pt
RUN chmod 0555 /opt/models /opt/models/whisper
COPY dockerfiles/hermes-jetson-stt-server.py /opt/atlas/hermes-jetson-stt-server.py
@ -28,11 +31,12 @@ WORKDIR /opt/atlas
# Import the Xavier CUDA stack and confirm Whisper resolves the baked artifact.
# Full GPU warm-up is covered by the Kubernetes startup probe on titan-21.
RUN python3 -c "import stat; from pathlib import Path; import torch, whisper; p=Path('/opt/models/whisper'); print(whisper.__file__, whisper.__version__, whisper.available_models()); assert {'large-v3-turbo','small'} <= set(whisper.available_models()); assert stat.S_IMODE(p.stat().st_mode)==0o555; assert all(stat.S_IMODE((p/name).stat().st_mode)==0o444 for name in ('large-v3-turbo.pt','small.pt')); print(torch.__version__)"
RUN python3 -c "import stat; from pathlib import Path; import torch, whisper; p=Path('/opt/models/whisper'); print(whisper.__file__, whisper.__version__, whisper.available_models()); assert {'large-v3-turbo','small','tiny'} <= set(whisper.available_models()); assert stat.S_IMODE(p.stat().st_mode)==0o555; assert all(stat.S_IMODE((p/name).stat().st_mode)==0o444 for name in ('large-v3-turbo.pt','small.pt','tiny.pt')); print(torch.__version__)"
ENV HERMES_STT_HOST=0.0.0.0 \
HERMES_STT_PORT=9000 \
HERMES_STT_MODEL=small \
HERMES_STT_MODEL=large-v3-turbo \
HERMES_STT_ROLLING_MODEL=tiny \
HERMES_STT_CACHE=/opt/models/whisper \
PYTHONDONTWRITEBYTECODE=1 \
PYTHONUNBUFFERED=1

View File

@ -24,18 +24,42 @@ ADD --checksum=sha256:f7d01dde371555732c4c314111ac79672b1a5ce2fc19266ab42178fd8d
ADD --checksum=sha256:45754dfdebb3b8661c3fc564713772deec6e064feeb5b4e9594857dc7305193a --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/en/en_US/lessac/low/en_US-lessac-low.onnx.json?download=true \
/opt/models/piper/en_US-lessac-low.onnx.json
# Multilingual chat voice policy: English -> amy, Russian -> irina, Spanish ->
# claude (Mexican Spanish, the only "claude" voice rhasspy/piper-voices
# publishes; there is no es_ES-claude).
ADD --checksum=sha256:b3a6e47b57b8c7fbe6a0ce2518161a50f59a9cdd8a50835c02cb02bdd6206c18 --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/en/en_US/amy/medium/en_US-amy-medium.onnx?download=true \
/opt/models/piper/en_US-amy-medium.onnx
ADD --checksum=sha256:95a23eb4d42909d38df73bb9ac7f45f597dbfcde2d1bf9526fdeaf5466977d77 --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/en/en_US/amy/medium/en_US-amy-medium.onnx.json?download=true \
/opt/models/piper/en_US-amy-medium.onnx.json
ADD --checksum=sha256:8ff38212d23da300bbe3705c645e6e5b9475f0bfde01558eb17813e22acaaaaa --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/ru/ru_RU/irina/medium/ru_RU-irina-medium.onnx?download=true \
/opt/models/piper/ru_RU-irina-medium.onnx
ADD --checksum=sha256:c2ec28bb38e2b59e93b959b3e40348c1afebbd272f30fed5d41205d08e98a9d7 --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/ru/ru_RU/irina/medium/ru_RU-irina-medium.onnx.json?download=true \
/opt/models/piper/ru_RU-irina-medium.onnx.json
ADD --checksum=sha256:3ef40a71ea63852cd8ab7e6fa7d2ecdcfa67a0b47c9c48e3f10e02ee02083ea0 --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/es/es_MX/claude/high/es_MX-claude-high.onnx?download=true \
/opt/models/piper/es_MX-claude-high.onnx
ADD --checksum=sha256:1afc81f703c0e4cb3b4d7c0dca096b8b54a98806807f0170cf5eb5557723c12d --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/es/es_MX/claude/high/es_MX-claude-high.onnx.json?download=true \
/opt/models/piper/es_MX-claude-high.onnx.json
RUN chmod 0555 /opt/models /opt/models/piper
COPY dockerfiles/hermes-jetson-tts-server.py /opt/atlas/hermes-jetson-tts-server.py
RUN chmod 0555 /opt/atlas/hermes-jetson-tts-server.py
COPY dockerfiles/hermes_jetson_tts_cues.py /opt/atlas/hermes_jetson_tts_cues.py
RUN chmod 0555 /opt/atlas/hermes-jetson-tts-server.py /opt/atlas/hermes_jetson_tts_cues.py
# Load the actual pinned voice during the ARM64 build. This catches package or
# model-format drift before the image can reach Flux.
RUN python -c "import stat; from pathlib import Path; from piper import PiperVoice; p=Path('/opt/models/piper'); models=[p/'en_US-lessac-high.onnx',p/'en_US-lessac-medium.onnx',p/'en_US-lessac-low.onnx']; assert stat.S_IMODE(p.stat().st_mode)==0o555; assert all(stat.S_IMODE(model.stat().st_mode)==0o444 for model in models); voices=[PiperVoice.load(model,Path(str(model)+'.json'),use_cuda=False,download_dir=p) for model in models]; assert all(voice.config.sample_rate>0 for voice in voices)"
# Load every pinned voice during the ARM64 build, including the three baked
# for the multilingual chat policy. This catches package or model-format
# drift before the image can reach Flux.
RUN python -c "import stat; from pathlib import Path; from piper import PiperVoice; p=Path('/opt/models/piper'); models=[p/'en_US-lessac-high.onnx',p/'en_US-lessac-medium.onnx',p/'en_US-lessac-low.onnx',p/'en_US-amy-medium.onnx',p/'ru_RU-irina-medium.onnx',p/'es_MX-claude-high.onnx']; assert stat.S_IMODE(p.stat().st_mode)==0o555; assert all(stat.S_IMODE(model.stat().st_mode)==0o444 for model in models); voices=[PiperVoice.load(model,Path(str(model)+'.json'),use_cuda=False,download_dir=p) for model in models]; assert all(voice.config.sample_rate>0 for voice in voices)"
ENV HERMES_TTS_HOST=0.0.0.0 \
HERMES_TTS_PORT=9001 \
HERMES_TTS_VOICE=en_US-lessac-medium \
HERMES_TTS_VOICE=en_US-amy-medium \
HERMES_TTS_CACHE=/opt/models/piper \
OMP_NUM_THREADS=2 \
PYTHONDONTWRITEBYTECODE=1 \

View File

@ -1,21 +1,57 @@
# syntax=docker/dockerfile:1.7
# dockerfiles/Dockerfile.hermes-switchyard
FROM rust:1.96.1-slim-bookworm@sha256:e18a79fc84dfcfc3ab5ba72290398a644c135c97eaa881447fddc354ee4701a3 AS build
FROM --platform=$BUILDPLATFORM rust:1.96.1-slim-bookworm@sha256:e18a79fc84dfcfc3ab5ba72290398a644c135c97eaa881447fddc354ee4701a3 AS build
ARG SWITCHYARD_LIBSY_SHA256=85f12e1ffefa168604044ccf78535a34dcde38eb59653ac27465e44113cd848c
ARG SWITCHYARD_SERVER_SHA256=95946e3df637143dcdd2a34c153bcaa57b7d4bc3dee53ed6c5497ddd996e16bc
ARG TARGETARCH
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
build-essential \
ca-certificates \
cmake \
curl \
gcc-aarch64-linux-gnu \
libc6-dev-arm64-cross \
linux-libc-dev-arm64-cross \
patch \
pkg-config \
&& rm -rf /var/lib/apt/lists/*
RUN cargo install \
--locked \
--version 0.2.0 \
--root /opt/switchyard \
switchyard-server
COPY dockerfiles/switchyard-capability-fallback.patch /tmp/switchyard-capability-fallback.patch
FROM debian:bookworm-slim@sha256:abd67ffcfa541b485a3dff59865ab629aa048a6c613e639d36e7456b0b229241
RUN rustup target add aarch64-unknown-linux-gnu
ENV CC_aarch64_unknown_linux_gnu=aarch64-linux-gnu-gcc \
CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_LINKER=aarch64-linux-gnu-gcc
RUN --mount=type=cache,id=switchyard-cargo-registry,target=/usr/local/cargo/registry,sharing=locked \
--mount=type=cache,id=switchyard-cargo-target-aarch64,target=/opt/switchyard-target,sharing=locked \
set -eux; \
test "$TARGETARCH" = arm64; \
mkdir -p /opt/switchyard-source; \
curl --fail --location --retry 3 --output /tmp/libsy.crate \
https://static.crates.io/crates/switchyard-libsy/switchyard-libsy-0.2.0.crate; \
echo "${SWITCHYARD_LIBSY_SHA256} /tmp/libsy.crate" | sha256sum --check --status -; \
tar --extract --gzip --file /tmp/libsy.crate --directory /opt/switchyard-source; \
patch --strip=1 --directory /opt/switchyard-source/switchyard-libsy-0.2.0 \
--input /tmp/switchyard-capability-fallback.patch; \
curl --fail --location --retry 3 --output /tmp/server.crate \
https://static.crates.io/crates/switchyard-server/switchyard-server-0.2.0.crate; \
echo "${SWITCHYARD_SERVER_SHA256} /tmp/server.crate" | sha256sum --check --status -; \
tar --extract --gzip --file /tmp/server.crate --directory /opt/switchyard-source; \
CARGO_TARGET_DIR=/opt/switchyard-target cargo test --locked \
--manifest-path /opt/switchyard-source/switchyard-libsy-0.2.0/Cargo.toml \
automatic_fallback --lib; \
printf '\n[patch.crates-io]\nswitchyard-libsy = { path = "/opt/switchyard-source/switchyard-libsy-0.2.0" }\n' \
>> /opt/switchyard-source/switchyard-server-0.2.0/Cargo.toml; \
CARGO_TARGET_DIR=/opt/switchyard-target cargo install --locked --target aarch64-unknown-linux-gnu \
--path /opt/switchyard-source/switchyard-server-0.2.0 \
--root /opt/switchyard; \
rm -rf /opt/switchyard-source /tmp/libsy.crate /tmp/server.crate /tmp/switchyard-capability-fallback.patch
FROM --platform=$TARGETPLATFORM debian:bookworm-slim@sha256:abd67ffcfa541b485a3dff59865ab629aa048a6c613e639d36e7456b0b229241
COPY --from=build /etc/ssl/certs/ca-certificates.crt /etc/ssl/certs/ca-certificates.crt
COPY --from=build /opt/switchyard/bin/switchyard-server /usr/local/bin/switchyard-server

View File

@ -1,114 +1,111 @@
# syntax=docker/dockerfile:1
# dockerfiles/Dockerfile.hermes-webui
FROM ghcr.io/nesquena/hermes-webui@sha256:a83a3893111dcb250e7aa7aa657d3d6f4570b0e2fd00d9b7569246fc5e7339b2 AS webui
#
# Both FROM bases below are multi-arch (linux/amd64 + linux/arm64). Kaniko builds
# one native leaf per node arch (arm64 on titan-20, amd64 on titan-24) and each
# leaf selects the matching arch from these indexes; ci/scripts/hermes_multiarch_combine.py
# then binds the two leaves into one manifest list. This is what lets the agent
# pod's `hux` sidecar (which runs this image) schedule onto the amd64 node titan-22.
#
# The upstream WebUI base is a multi-arch OCI index
# (sha256:a83a3893... -> amd64 sha256:54fd4990..., arm64 sha256:9094ae6a...). It is
# mirrored digest-for-digest into the in-cluster Harbor `mirror` project by
# services/harbor/hermes-webui-base-mirror-job.yaml so the build never depends on
# ghcr.io egress (flaky from build pods). Bump this digest and BOTH args in that
# Job together, then an operator re-runs the (suspended) Job once.
FROM harbor-core.harbor.svc.cluster.local/mirror/hermes-webui@sha256:a83a3893111dcb250e7aa7aa657d3d6f4570b0e2fd00d9b7569246fc5e7339b2 AS webui
FROM registry.bstein.dev/bstein/hermes-agent@sha256:81970563e542f0720773e72297810b3a844b83e381e278f25c0916c78d930107
# Layer each arch's WebUI on the matching leaf of the multi-arch Hermes agent index
# (sha256:a68d1c4d... -> amd64 sha256:c89ac4bc..., arm64 sha256:572854cb...), a Docker
# manifest list already in Harbor's bstein project. Kaniko's --registry-mirror pulls
# it internally through harbor-core.
FROM registry.bstein.dev/bstein/hermes-agent@sha256:a68d1c4d5517cc5e6719661e77be4f18d4964b07a85dcebaf6f62368646e4e6c
ARG HERMES_WEBUI_RELEASE_ID
USER root
# Keep WebUI and Hermes pinned together. The WebUI imports Hermes internals,
# while the gateway remains the only process that owns an agent conversation.
COPY --from=webui /apptoo /opt/hermes-webui
# The account policy caps user-selected reasoning at xhigh even when a provider
# advertises a newer, more expensive level.
RUN /opt/hermes/.venv/bin/python - <<'PY'
from pathlib import Path
config = Path("/opt/hermes-webui/api/config.py")
source = config.read_text(encoding="utf-8")
before = 'VALID_REASONING_EFFORTS = ("minimal", "low", "medium", "high", "xhigh", "max")'
after = 'VALID_REASONING_EFFORTS = ("minimal", "low", "medium", "high", "xhigh")'
if before not in source:
raise SystemExit("Hermes WebUI reasoning-effort patch context changed")
config.write_text(source.replace(before, after, 1), encoding="utf-8")
index = Path("/opt/hermes-webui/static/index.html")
source = index.read_text(encoding="utf-8")
before = ' <div class="reasoning-option" data-effort="max">Max</div>\n'
if before not in source:
raise SystemExit("Hermes WebUI xhigh UI patch context changed")
index.write_text(source.replace(before, "", 1), encoding="utf-8")
# oauth2-proxy returns 401 for browser API and health probes when the secure
# session expires. Re-enter OIDC with the complete return path instead of
# presenting an endless, inaccurate "connection lost" loop.
ui = Path("/opt/hermes-webui/static/ui.js")
source = ui.read_text(encoding="utf-8")
before = ''' const res=await fetcher(_offlineHealthUrl(),opts);
return !!(res&&res.ok);
'''
after = ''' const res=await fetcher(_offlineHealthUrl(),opts);
if(res&&(res.status===401||res.status===403)){
const rd=window.location.pathname+window.location.search+window.location.hash;
window.location.assign('/oauth2/start?rd='+encodeURIComponent(rd));
return false;
}
return !!(res&&res.ok);
'''
if source.count(before) != 1:
raise SystemExit("Hermes WebUI auth-recovery patch context changed")
ui.write_text(source.replace(before, after, 1), encoding="utf-8")
# Make delegated session hierarchy obvious and collapsible in the sidebar.
sessions = Path("/opt/hermes-webui/static/sessions.js")
source = sessions.read_text(encoding="utf-8")
before = ''' const childLabel=t('session_meta_children', childCount);
childCountEl.textContent=childLabel;
childCountEl.title=_sessionChildBadgeTooltip(childLabel);
'''
after = ''' const childLabel=t('session_meta_children', childCount);
const childrenExpanded=_expandedChildSessionKeys.has(lineageKey)||!!searchQueryRaw;
childCountEl.textContent=(childrenExpanded?'▾ ':'▸ ')+childLabel;
childCountEl.setAttribute('aria-expanded',childrenExpanded?'true':'false');
childCountEl.title=_sessionChildBadgeTooltip(childLabel);
'''
if source.count(before) != 1:
raise SystemExit("Hermes WebUI child-session toggle patch context changed")
sessions.write_text(source.replace(before, after, 1), encoding="utf-8")
# A profile's model is only its default; a session-level selector can override
# it. Label the scope so the dropdown does not contradict the effective model.
panels = Path("/opt/hermes-webui/static/panels.js")
source = panels.read_text(encoding="utf-8")
before = " if (typeof p.model === 'string' && p.model) meta.push(p.model.split('/').pop());\n"
after = ''' if (typeof p.model === 'string' && p.model) {
const routeLabels = {
'atlas/auto/fast': 'Automatic · Fast',
'atlas/auto/balanced': 'Automatic · Balanced',
'atlas/auto/deep': 'Automatic · Deep',
'atlas/auto/maximum': 'Automatic · Maximum',
};
meta.push('profile default: ' + (routeLabels[p.model] || p.model.split('/').pop()));
}
'''
if source.count(before) != 2:
raise SystemExit("Hermes WebUI profile-model label patch context changed")
panels.write_text(source.replace(before, after, 2), encoding="utf-8")
PY
COPY dockerfiles/hermes-hux-foundation/hux /opt/hermes-hux/hux
COPY dockerfiles/hermes-hux-foundation/hux_producer /opt/hermes-hux/hux_producer
COPY services/hermes/contracts/hux /opt/hermes-hux/contracts
ENV HUX_CONTRACT_DIR=/opt/hermes-hux/contracts
# Add the Atlas voice bridge as a narrow integration layer. It activates only
# when a tenant's server-side STT capability reports the private Jetson route.
COPY dockerfiles/hermes-webui-base-patch.py /tmp/hermes-webui-base-patch.py
COPY dockerfiles/hermes-webui-atlas-patch.py /tmp/hermes-webui-atlas-patch.py
COPY dockerfiles/hermes-webui-stt-patch.py /tmp/hermes-webui-stt-patch.py
COPY dockerfiles/hermes-webui-telegram-project-patch.py /tmp/hermes-webui-telegram-project-patch.py
COPY dockerfiles/hermes-webui-atlas-voice.js /opt/hermes-webui/static/atlas-voice.js
COPY dockerfiles/hermes-webui-atlas-voice-worklet.js /opt/hermes-webui/static/atlas-voice-worklet.js
COPY dockerfiles/hermes-webui-atlas-voice.css /opt/hermes-webui/static/atlas-voice.css
COPY dockerfiles/hermes-webui-router-patch.py /tmp/hermes-webui-router-patch.py
COPY dockerfiles/hermes-webui-router.js /opt/hermes-webui/static/atlas-router.js
RUN /opt/hermes/.venv/bin/python /tmp/hermes-webui-atlas-patch.py
RUN /opt/hermes/.venv/bin/python /tmp/hermes-webui-telegram-project-patch.py
RUN /opt/hermes/.venv/bin/python /tmp/hermes-webui-router-patch.py
COPY dockerfiles/hermes-webui-hux-bff-patch.py /tmp/hermes-webui-hux-bff-patch.py
COPY dockerfiles/hermes-webui-hux-context.py /tmp/hermes-webui-hux-context.py
COPY dockerfiles/hermes-webui-hux-context-patch.py /tmp/hermes-webui-hux-context-patch.py
COPY dockerfiles/hermes-webui-hux-patch.py /tmp/hermes-webui-hux-patch.py
COPY dockerfiles/hermes-webui-hux/foundation.css /opt/hermes-webui/static/hux/foundation.css
COPY dockerfiles/hermes-webui-hux/foundation.js /opt/hermes-webui/static/hux/foundation.js
COPY dockerfiles/hermes-webui-hux/shell.js /opt/hermes-webui/static/hux/shell.js
COPY dockerfiles/hermes-webui-hux/runtime/autonomy-privacy.css /opt/hermes-webui/static/hux/runtime/autonomy-privacy.css
COPY dockerfiles/hermes-webui-hux/runtime/autonomy-privacy.js /opt/hermes-webui/static/hux/runtime/autonomy-privacy.js
COPY dockerfiles/hermes-webui-hux/runtime/wave_a_activity_memory.css /opt/hermes-webui/static/hux/runtime/wave_a_activity_memory.css
COPY dockerfiles/hermes-webui-hux/runtime/wave_a_activity_memory.js /opt/hermes-webui/static/hux/runtime/wave_a_activity_memory.js
COPY dockerfiles/hermes-webui-hux/runtime/wave_a_contract.js /opt/hermes-webui/static/hux/runtime/wave_a_contract.js
COPY dockerfiles/hermes-webui-hux/runtime/wave_b.css /opt/hermes-webui/static/hux/runtime/wave_b.css
COPY dockerfiles/hermes-webui-hux/runtime/wave_b_artifacts_research.js /opt/hermes-webui/static/hux/runtime/wave_b_artifacts_research.js
COPY dockerfiles/hermes-webui-hux/runtime/wave_b_contract.js /opt/hermes-webui/static/hux/runtime/wave_b_contract.js
COPY dockerfiles/hermes-webui-hux/runtime/wave_b_projects_modes.js /opt/hermes-webui/static/hux/runtime/wave_b_projects_modes.js
COPY dockerfiles/hermes-webui-hux/runtime/wave_b_runtime.js /opt/hermes-webui/static/hux/runtime/wave_b_runtime.js
COPY dockerfiles/hermes-webui-hux/runtime/wave_c_multimodal_onboarding_release.css /opt/hermes-webui/static/hux/runtime/wave_c_multimodal_onboarding_release.css
COPY dockerfiles/hermes-webui-hux/runtime/wave_c_multimodal_onboarding_release.js /opt/hermes-webui/static/hux/runtime/wave_c_multimodal_onboarding_release.js
COPY dockerfiles/hermes-webui-hux/bootstrap.css /opt/hermes-webui/static/hux/bootstrap.css
COPY dockerfiles/hermes-webui-hux/bootstrap.js /opt/hermes-webui/static/hux/bootstrap.js
COPY dockerfiles/hermes-webui-brand-patch.py /tmp/hermes-webui-brand-patch.py
COPY dockerfiles/hermes-webui-manifest-patch.py /tmp/hermes-webui-manifest-patch.py
COPY dockerfiles/hermes-webui-release-patch.py /tmp/hermes-webui-release-patch.py
COPY dockerfiles/hermes-webui-smoke.py /tmp/hermes-webui-smoke.py
COPY dockerfiles/hermes-webui-brand.css /opt/hermes-webui/static/hermes-brand.css
COPY dockerfiles/hermes-webui-manifest.json /tmp/hermes-webui-manifest.json
COPY dockerfiles/hermes-webui-assets/hermes-agent.ico /opt/hermes-webui/static/hermes-agent.ico
COPY dockerfiles/hermes-webui-assets/hermes-agent-192.png /opt/hermes-webui/static/hermes-agent-192.png
COPY dockerfiles/hermes-webui-assets/hermes-agent-512.png /opt/hermes-webui/static/hermes-agent-512.png
RUN /opt/hermes/.venv/bin/python /tmp/hermes-webui-base-patch.py \
&& /opt/hermes/.venv/bin/python /tmp/hermes-webui-atlas-patch.py \
&& /opt/hermes/.venv/bin/python /tmp/hermes-webui-stt-patch.py \
&& /opt/hermes/.venv/bin/python /tmp/hermes-webui-telegram-project-patch.py \
&& /opt/hermes/.venv/bin/python /tmp/hermes-webui-router-patch.py \
&& /opt/hermes/.venv/bin/python /tmp/hermes-webui-hux-bff-patch.py \
&& /opt/hermes/.venv/bin/python /tmp/hermes-webui-hux-context-patch.py \
&& /opt/hermes/.venv/bin/python /tmp/hermes-webui-brand-patch.py \
&& /opt/hermes/.venv/bin/python /tmp/hermes-webui-manifest-patch.py \
&& /opt/hermes/.venv/bin/python /tmp/hermes-webui-hux-patch.py \
&& HERMES_WEBUI_RELEASE_ID="${HERMES_WEBUI_RELEASE_ID}" \
/opt/hermes/.venv/bin/python /tmp/hermes-webui-release-patch.py
RUN /opt/hermes/.venv/bin/python -c 'import cryptography, yaml' \
&& test -f /opt/hermes-hux/contracts/flags.json \
&& test -f /opt/hermes-hux/contracts/release-ledger.schema.json \
&& PYTHONPATH=/opt/hermes-hux /opt/hermes/.venv/bin/python -m compileall -q /opt/hermes-hux/hux /opt/hermes-hux/hux_producer \
&& PYTHONPATH=/opt/hermes-hux /opt/hermes/.venv/bin/python -c 'from pathlib import Path; from tempfile import TemporaryDirectory; from hux.server import build_router; root = TemporaryDirectory(); router = build_router(Path(root.name), {"HUX_FLAGS": ""}); assert router.routes; root.cleanup()' \
&& grep -Fq 'VALID_REASONING_EFFORTS = ("minimal", "low", "medium", "high", "xhigh")' \
/opt/hermes-webui/api/config.py \
&& ! grep -Fq 'data-effort="max"' /opt/hermes-webui/static/index.html \
&& grep -Fq 'res.status===401||res.status===403' /opt/hermes-webui/static/ui.js \
&& grep -Fq "window.location.assign('/oauth2/start?rd='" /opt/hermes-webui/static/ui.js \
&& grep -Fq "childrenExpanded?'▾ ':'▸ '" /opt/hermes-webui/static/sessions.js \
&& grep -Fq "TELEGRAM_PROJECT_NAME = 'Telegram'" /opt/hermes-webui/api/models.py \
&& grep -Fq "'atlas/auto/maximum': 'Automatic · Maximum'" /opt/hermes-webui/static/panels.js \
&& grep -Fq 'Atlas Jetson (private)' /opt/hermes-webui/static/index.html \
&& grep -Fq 'user-scalable=no, viewport-fit=cover' /opt/hermes-webui/static/index.html \
&& grep -Fq 'HERMES_WEBUI_ATLAS_TTS_URL' /opt/hermes-webui/api/routes.py \
&& grep -Fq 'settings["webui_bundle_version"]' /opt/hermes-webui/api/routes.py \
&& grep -Fq 'settings.webui_bundle_version||settings.webui_version' /opt/hermes-webui/static/panels.js \
&& grep -Fq 'Audio conversion failed: upload is invalid' /opt/hermes/tools/transcription_tools.py \
&& grep -Fq "capability.provider!=='local_command'" /opt/hermes-webui/static/atlas-voice.js \
&& grep -Fq 'prefers-reduced-motion: reduce' /opt/hermes-webui/static/atlas-voice.css \
&& grep -Fq 'id="voiceInstrumentStyles"' /opt/hermes-webui/static/index.html \
@ -116,11 +113,56 @@ RUN /opt/hermes/.venv/bin/python -c 'import cryptography, yaml' \
&& grep -Fq 'routing_priority:priority' /opt/hermes-webui/static/atlas-router.js \
&& grep -Fq "'atlas/auto/fast':'Automatic · Fast'" /opt/hermes-webui/static/atlas-router.js \
&& grep -Fq 'explicit_reasoning_effort' /opt/hermes-webui/api/gateway_chat.py \
&& grep -Fq 'from api.hux_bff import proxy_hux' /opt/hermes-webui/api/routes.py \
&& grep -Fq 'from api.hux_context import attach_hux_context' /opt/hermes-webui/api/routes.py \
&& grep -Fq '"schema": "hux.webui_context.v1"' /opt/hermes-webui/api/hux_context.py \
&& grep -Fq '"project_source": raw_project' /opt/hermes-webui/api/hux_context.py \
&& grep -Fq 'static/hux/bootstrap.js' /opt/hermes-webui/static/index.html \
&& grep -Fq "'./static/hux/bootstrap.js' + VQ" /opt/hermes-webui/static/sw.js \
&& grep -Fq 'value.hux_context' /opt/hermes-webui/static/hux/bootstrap.js \
&& grep -Fq "'/hux/v1/capabilities'" /opt/hermes-webui/static/hux/bootstrap.js \
&& grep -Fq "'/hux/v1/context/bootstrap'" /opt/hermes-webui/static/hux/bootstrap.js \
&& grep -Fq 'HermesHuxBootstrap.mergeTrustedContext(S.session, startData)' /opt/hermes-webui/static/messages.js \
&& grep -Fq 'respBody.cancelled===true && respBody.stream_id===streamId' /opt/hermes-webui/static/boot.js \
&& grep -Fq 'prefers-reduced-motion: no-preference' /opt/hermes-webui/static/hux/bootstrap.css \
&& grep -Fq '"language": detected_language' /opt/hermes/tools/transcription_tools.py \
&& grep -Fq "'language': detected" /opt/hermes-webui/api/upload.py \
&& grep -Fq 'def _atlas_tts_language(body):' /opt/hermes-webui/api/routes.py \
&& grep -Fq 'request_payload["language"] = _atlas_language' /opt/hermes-webui/api/routes.py \
&& grep -Fq 'takeSttLanguage(token)' /opt/hermes-webui/static/atlas-voice.js \
&& grep -Fq "registerProcessor('atlas-pcm-playback'" /opt/hermes-webui/static/atlas-voice-worklet.js \
&& grep -Fq "const TTS_STREAM_URL='/api/tts/stream'" /opt/hermes-webui/static/atlas-voice.js \
&& grep -Fq "const STT_STREAM_PATH='/api/transcribe/stream'" /opt/hermes-webui/static/atlas-voice.js \
&& grep -Fq '<title>Hermes Chat</title>' /opt/hermes-webui/static/index.html \
&& grep -Fq 'id="hermesBrandStyles"' /opt/hermes-webui/static/index.html \
&& grep -Fq 'static/hermes-agent-512.png' /opt/hermes-webui/static/index.html \
&& grep -Fq 'prefers-reduced-motion: reduce' /opt/hermes-webui/static/hermes-brand.css \
&& grep -Fq '"name": "Hermes Chat"' /opt/hermes-webui/static/manifest.json \
&& grep -Fq "'./static/hermes-agent-512.png'" /opt/hermes-webui/static/sw.js \
&& printf '%s %s\n' \
'aefe65e6574e6f46d3382588f46c508e1b3f3b3c9ce3dec6d335403a5374add9' \
'/opt/hermes-webui/static/hermes-agent.ico' \
| sha256sum -c - \
&& printf '%s %s\n' \
'0e4102cc715372058dd6ab55e9cde2567e46fee5ab6557b564cdd212ccd2616f' \
'/opt/hermes-webui/static/hermes-agent-192.png' \
| sha256sum -c - \
&& printf '%s %s\n' \
'6661e5ca0ecc690af213b9f85f961541e6d3d0e36946ede9d3c2c97dfbd3c23d' \
'/opt/hermes-webui/static/hermes-agent-512.png' \
| sha256sum -c - \
&& /opt/hermes/.venv/bin/python -m py_compile \
/opt/hermes-webui/api/routes.py \
/opt/hermes-webui/api/gateway_chat.py
/opt/hermes-webui/api/upload.py \
/opt/hermes-webui/api/gateway_chat.py \
/opt/hermes-webui/api/hux_bff.py \
/opt/hermes-webui/api/hux_context.py \
/opt/hermes/tools/transcription_tools.py
# Exercise the real server process in the target architecture before publish.
# Exercise branded responses from the real upstream server process in the
# target architecture before publish. The manifest checks resolve icon paths
# from the URL the server actually returns; no filesystem-only assertion can
# satisfy this gate.
RUN set -eu; \
mkdir -p /tmp/hermes-webui-smoke/home /tmp/hermes-webui-smoke/state /tmp/hermes-webui-smoke/workspace; \
HERMES_HOME=/tmp/hermes-webui-smoke/home \
@ -132,14 +174,24 @@ RUN set -eu; \
HERMES_WEBUI_SKIP_ONBOARDING=1 \
/opt/hermes/.venv/bin/python /opt/hermes-webui/server.py >/tmp/hermes-webui-smoke.log 2>&1 & \
server_pid=$!; \
cleanup() { kill "${server_pid}" 2>/dev/null || true; wait "${server_pid}" 2>/dev/null || true; }; \
trap cleanup EXIT HUP INT TERM; \
ready=0; \
for attempt in 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30; do \
if /opt/hermes/.venv/bin/python -c 'from urllib.request import urlopen; urlopen("http://127.0.0.1:18787/health", timeout=2).read()' >/dev/null 2>&1; then ready=1; break; fi; \
sleep 1; \
done; \
kill "${server_pid}" 2>/dev/null || true; \
wait "${server_pid}" 2>/dev/null || true; \
if [ "${ready}" != "1" ]; then cat /tmp/hermes-webui-smoke.log; exit 1; fi; \
smoke_passed=0; \
if [ "${ready}" = "1" ] \
&& /opt/hermes/.venv/bin/python /tmp/hermes-webui-smoke.py http://127.0.0.1:18787/; then \
smoke_passed=1; \
fi; \
cleanup; \
trap - EXIT HUP INT TERM; \
if [ "${ready}" != "1" ] || [ "${smoke_passed}" != "1" ]; then \
cat /tmp/hermes-webui-smoke.log; \
exit 1; \
fi; \
rm -rf /tmp/hermes-webui-smoke /tmp/hermes-webui-smoke.log
ENV HERMES_WEBUI_AGENT_DIR=/opt/hermes \
@ -150,6 +202,7 @@ ENV HERMES_WEBUI_AGENT_DIR=/opt/hermes \
HERMES_WEBUI_GATEWAY_USE_RUNS_API=true \
HERMES_WEBUI_SKIP_ONBOARDING=1 \
HERMES_WEBUI_SECURE=1 \
PYTHONPATH=/opt/hermes-hux \
PYTHONDONTWRITEBYTECODE=1 \
PYTHONUNBUFFERED=1

View File

@ -0,0 +1,20 @@
import { describe, expect, it } from "vitest";
import { isSgrWheelReport } from "./dashboard-terminal-input";
describe("isSgrWheelReport", () => {
it("accepts xterm SGR wheel reports, including modifiers", () => {
expect(isSgrWheelReport("\x1b[<64;80;25M")).toBe(true);
expect(isSgrWheelReport("\x1b[<65;80;25M")).toBe(true);
expect(isSgrWheelReport("\x1b[<84;80;25M")).toBe(true);
});
it("rejects clicks, releases, malformed reports, and arbitrary input", () => {
expect(isSgrWheelReport("\x1b[<0;80;25M")).toBe(false);
expect(isSgrWheelReport("\x1b[<64;80;25m")).toBe(false);
expect(isSgrWheelReport("\x1b[<66;80;25M")).toBe(false);
expect(isSgrWheelReport("\x1b[<64;10000;25M")).toBe(false);
expect(isSgrWheelReport("\x1b[<64;80;25M\rmalicious")).toBe(false);
expect(isSgrWheelReport("/submit\r")).toBe(false);
});
});

View File

@ -0,0 +1,27 @@
/** Validate the only mouse reports the dashboard may relay to its PTY. */
// SGR wheel reports use button codes 64/65 plus optional Shift/Alt/Ctrl bits.
// Keep coordinates bounded so arbitrary control text never reaches the PTY.
const SGR_WHEEL_REPORT = /^\x1b\[<(\d{1,2});(\d{1,4});(\d{1,4})M$/;
const MIN_WHEEL_BUTTON = 64;
const MAX_WHEEL_BUTTON = 95;
const MAX_TERMINAL_COORDINATE = 9_999;
/** Return whether ``data`` is one bounded SGR wheel report from xterm. */
export function isSgrWheelReport(data: string): boolean {
const match = SGR_WHEEL_REPORT.exec(data);
if (!match) return false;
const button = Number(match[1]);
const column = Number(match[2]);
const row = Number(match[3]);
return (
button >= MIN_WHEEL_BUTTON &&
button <= MAX_WHEEL_BUTTON &&
(button & 3) <= 1 &&
column >= 1 &&
column <= MAX_TERMINAL_COORDINATE &&
row >= 1 &&
row <= MAX_TERMINAL_COORDINATE
);
}

View File

@ -0,0 +1 @@
"""Hermes User Experience foundation service package."""

View File

@ -0,0 +1,424 @@
"""HUX-04 artifact workspace: typed, versioned, diffable outputs.
Every artifact is one ``hux.artifact.v1`` document under the caller's tenant
subtree. Content lives in the content-addressed blob store; the document only
carries ``content_ref`` hashes, so a version can never be rewritten once it
has been appended (SO-30). Blobs are served inside a JSON envelope with
``nosniff`` and never as an executable response type (SO-32). Every
referenced id is resolved under the caller's own subtree (SO-33); a foreign
or unknown id is ``404`` so ids cannot be probed. Sharing is out of scope for
this increment: ``access.mode`` is always ``owner`` (SO-34).
"""
from __future__ import annotations
import base64
import hashlib
from typing import Any
from hux import diffs
from hux.contracts import load_all, validate_record
from hux.errors import Conflict, Invalid, NotFound, TooLarge
from hux.http import Request, Response, Router, page
from hux.identity import Identity
from hux.store import TenantStore, check_id, new_id, now_iso
MAX_UPLOAD_BODY = 34 * 1024 * 1024 # 25 MiB of content plus base64 overhead (SO-31, SO-54)
def _project_exists(store, project_id: str) -> bool:
"""Ask the organization family when it is present; otherwise only the id shape is known."""
try:
from hux.organization import project_exists
except ModuleNotFoundError: # pragma: no cover - organization is always shipped with artifacts
return True
return project_exists(store, project_id)
FAMILY = "artifacts"
MAX_VERSION_BYTES = 25 * 1024 * 1024
MAX_VERSIONS = 200
MAX_ARTIFACTS = 2000
PAGE_SIZE = 50
ARTIFACT_TYPES = ("markdown", "code", "html", "svg", "image", "json", "csv", "document", "audio")
SENSITIVITIES = ("public", "personal", "sensitive", "restricted")
_SCHEMAS: dict[str, dict[str, Any]] = {}
def _schemas() -> dict[str, dict[str, Any]]:
if not _SCHEMAS:
_SCHEMAS.update(load_all())
return _SCHEMAS
# -- helpers shared with other lanes -------------------------------------------
def artifact_exists(store: TenantStore, artifact_id: str) -> bool:
"""True when ``artifact_id`` is a well-formed id owned by the store's tenant."""
try:
return store.exists(FAMILY, artifact_id)
except Invalid:
return False
def artifact_titles(store: TenantStore, ids: list[str]) -> dict[str, str]:
"""Map each resolvable artifact id to its title; unknown ids are omitted."""
titles: dict[str, str] = {}
for artifact_id in ids:
if artifact_exists(store, artifact_id):
titles[artifact_id] = store.get(FAMILY, artifact_id)["title"]
return titles
def emit_event(store: TenantStore, identity: Identity, artifact: dict[str, Any], kind: str, summary: str, detail: dict[str, Any]) -> None:
"""Record an activity event through the events lane when it is present.
Imported lazily so module import order never matters; a missing events
module (earlier build) is not an error because the artifact write itself
is the source of truth.
"""
conversation_id = artifact.get("conversation_id")
if conversation_id is None:
return
try:
from hux.events import emit
except ModuleNotFoundError:
return
version = artifact["current_version"]
evidence = [{"kind": "artifact_version", "id": f"{artifact['id']}@{version}", "hash": artifact["versions"][-1]["content_ref"]["hash"]}]
emit(store, identity, conversation_id, kind, summary, detail=detail, evidence=evidence, sensitivity=artifact["sensitivity"])
def replay(store: TenantStore, family: str, key: str) -> str | None:
"""Record id previously created under an Idempotency-Key, if any."""
if not key:
return None
for row in store.read(family, "idempotency"):
if row.get("key") == key:
return row["id"]
return None
def remember(store: TenantStore, family: str, key: str, record_id: str) -> None:
"""Persist an Idempotency-Key to record id mapping."""
if key:
store.append(family, "idempotency", {"key": key, "id": record_id, "at": now_iso()})
# -- record shaping ------------------------------------------------------------
def _actor(identity: Identity) -> dict[str, str]:
if identity.trust == "worker":
return {"type": "system", "id": "worker"}
return {"type": "user", "id": identity.subject}
def _body(request: Request) -> dict[str, Any]:
if not isinstance(request.body, dict):
raise Invalid("body must be a JSON object")
return request.body
def _string(body: dict[str, Any], key: str, limit: int, required: bool = True) -> str | None:
value = body.get(key)
if value is None:
if required:
raise Invalid(f"{key} is required")
return None
if not isinstance(value, str) or not value.strip() or len(value) > limit:
raise Invalid(f"{key} must be a non-empty string of at most {limit} characters")
return value
def secret_hits(data: bytes) -> list[str]:
"""Secret-pattern classes found in text content; binary content is not scanned (SO-12)."""
from hux.redaction import scrub_text
try:
text = data.decode("utf-8")
except UnicodeDecodeError:
return []
return scrub_text(text)[1]
def _content(body: dict[str, Any]) -> tuple[bytes, str]:
"""Decode the submitted content and return ``(bytes, mime)``."""
text, encoded = body.get("content"), body.get("content_base64")
if isinstance(text, str) and encoded is None:
data = text.encode("utf-8")
elif isinstance(encoded, str) and text is None:
try:
data = base64.b64decode(encoded, validate=True)
except (ValueError, TypeError) as error:
raise Invalid("content_base64 is not valid base64") from error
else:
raise Invalid("exactly one of content (utf-8 text) or content_base64 is required")
if len(data) > MAX_VERSION_BYTES:
raise TooLarge("version exceeds 25 MiB")
mime = _string(body, "mime", 120, required=False) or "application/octet-stream"
return data, mime
def _store_content(store: TenantStore, body: dict[str, Any]) -> dict[str, Any]:
"""Hash, verify against any client-supplied hash, store the blob, return ``content_ref``."""
data, mime = _content(body)
digest = hashlib.sha256(data).hexdigest()
claimed = body.get("hash")
if claimed is not None and claimed != f"sha256:{digest}":
raise Invalid("hash does not match the submitted content")
store.put_blob(digest, data)
return {"hash": f"sha256:{digest}", "bytes": len(data), "mime": mime}
def _optional_id(body: dict[str, Any], key: str) -> str | None:
value = body.get(key)
return None if value is None else check_id(value)
def _load_owned(request: Request) -> dict[str, Any]:
"""The artifact named in the path; unknown, malformed or foreign is 404."""
try:
artifact = request.store.get(FAMILY, request.params["id"])
except Invalid as error:
raise NotFound("artifact not found") from error
if artifact.get("owner") != request.identity.subject:
raise NotFound("artifact not found")
return artifact
def _version(artifact: dict[str, Any], number: Any) -> dict[str, Any]:
if not isinstance(number, int) or isinstance(number, bool):
raise Invalid("version must be an integer")
for entry in artifact["versions"]:
if entry["version"] == number:
return entry
raise NotFound(f"version {number} not found")
def _lineage(request: Request, body: dict[str, Any]) -> dict[str, Any] | None:
"""Resolve ``lineage`` under the caller's subtree; anything else is 404 (SO-33)."""
lineage = body.get("lineage")
if lineage is None:
return None
if not isinstance(lineage, dict):
raise Invalid("lineage must be an object")
try:
parent = request.store.get(FAMILY, check_id(lineage.get("artifact_id")))
except (Invalid, NotFound) as error:
raise NotFound("lineage artifact not found") from error
if parent.get("owner") != request.identity.subject:
raise NotFound("lineage artifact not found")
version = _version(parent, lineage.get("version"))
return {"artifact_id": parent["id"], "version": version["version"]}
def append_version(store: TenantStore, artifact: dict[str, Any], entry: dict[str, Any], expected_revision: int | None) -> dict[str, Any]:
"""Append an immutable version entry and persist the document.
The version number must be exactly ``current_version + 1``; any attempt to
write an existing number is a Conflict so history is never rewritten.
"""
if any(v["version"] == entry["version"] for v in artifact["versions"]):
raise Conflict(f"version {entry['version']} already exists")
if entry["version"] != artifact["current_version"] + 1:
raise Conflict("versions are appended in order")
if len(artifact["versions"]) >= MAX_VERSIONS:
raise TooLarge("artifact has reached 200 versions")
updated = {**artifact, "versions": [*artifact["versions"], entry], "current_version": entry["version"], "updated_at": now_iso()}
_check(updated)
return store.put(FAMILY, updated, expected_revision=expected_revision)
def _check(artifact: dict[str, Any]) -> None:
problems = validate_record(artifact, _schemas())
if problems:
raise Invalid("artifact does not satisfy hux.artifact.v1", problems)
# -- handlers ------------------------------------------------------------------
def create(request: Request) -> Response:
"""``POST /hux/v1/artifacts``: create an artifact with version 1."""
body = _body(request)
key = request.idempotency_key()
with request.store.lock(FAMILY):
existing = replay(request.store, FAMILY, key)
if existing is not None:
request.audit("artifacts.create", existing, reason="idempotent_replay")
return Response(200, request.store.get(FAMILY, existing), {"HUX-Replayed": "true"})
if request.store.count(FAMILY) >= MAX_ARTIFACTS:
raise TooLarge("tenant has reached 2000 artifacts")
artifact_type = _string(body, "type", 40)
if artifact_type not in ARTIFACT_TYPES:
raise Invalid("unknown artifact type")
sensitivity = body.get("sensitivity", "personal")
if sensitivity not in SENSITIVITIES:
raise Invalid("unknown sensitivity")
hits = secret_hits(_content(body)[0])
if hits:
sensitivity = "restricted"
stamp = now_iso()
version: dict[str, Any] = {"version": 1, "created_at": stamp, "created_by": _actor(request.identity), "content_ref": _store_content(request.store, body)}
for field, limit in (("message_id", 120), ("note", 200)):
if _string(body, field, limit, required=False) is not None:
version[field] = body[field]
lineage = _lineage(request, body)
if lineage is not None:
version["lineage"] = lineage
artifact: dict[str, Any] = {
"schema": "hux.artifact.v1",
"id": new_id("art"),
"owner": request.identity.subject,
"type": artifact_type,
"title": _string(body, "title", 200),
"current_version": 1,
"versions": [version],
"sensitivity": sensitivity,
"created_at": stamp,
"updated_at": stamp,
"revision": 1,
"access": {"mode": "owner"},
}
for field in ("conversation_id", "project_id"):
if _optional_id(body, field) is not None:
artifact[field] = body[field]
if _string(body, "language", 40, required=False) is not None:
artifact["language"] = body["language"]
_check(artifact)
stored = request.store.put(FAMILY, artifact)
remember(request.store, FAMILY, key, stored["id"])
request.audit("artifacts.create", stored["id"], reason="secret_pattern_restricted" if hits else "")
emit_event(request.store, request.identity, stored, "artifact.created", f"Created {stored['type']} artifact", {"artifact_id": stored["id"], "version": 1})
return Response(201, stored, {"ETag": str(stored["revision"])})
def list_artifacts(request: Request) -> Response:
"""``GET /hux/v1/artifacts?conversation_id=&project_id=&cursor=``: owned artifacts, oldest first."""
filters = {k: request.query[k] for k in ("conversation_id", "project_id") if request.query.get(k)}
cursor = request.query.get("cursor", "0")
if not cursor.isdigit():
raise Invalid("cursor must be a non-negative integer")
items = [
artifact
for artifact in request.store.scan(FAMILY)
if artifact.get("owner") == request.identity.subject and all(artifact.get(k) == v for k, v in filters.items())
]
start = int(cursor)
window = items[start : start + PAGE_SIZE]
request.audit("artifacts.list", "artifacts")
return page(window, str(start + PAGE_SIZE) if len(items) > start + PAGE_SIZE else None)
def get(request: Request) -> Response:
"""``GET /hux/v1/artifacts/{id}``: one artifact document."""
artifact = _load_owned(request)
request.audit("artifacts.get", artifact["id"])
return Response(200, artifact, {"ETag": str(artifact["revision"])})
def add_version(request: Request) -> Response:
"""``POST /hux/v1/artifacts/{id}/versions``: append a new immutable version (If-Match required)."""
body = _body(request)
expected = request.if_match()
if expected is None:
raise Invalid("If-Match is required to append a version")
key = request.idempotency_key()
with request.store.lock(FAMILY):
artifact = _load_owned(request)
existing = replay(request.store, FAMILY, key)
if existing is not None and existing.split("@")[0] == artifact["id"]:
request.audit("artifacts.version", existing, reason="idempotent_replay")
return Response(200, artifact, {"HUX-Replayed": "true", "ETag": str(artifact["revision"])})
if expected != artifact["revision"]:
raise Conflict(f"revision {expected} does not match current revision {artifact['revision']}", [str(artifact["revision"])])
number = artifact["current_version"] + 1
hits = secret_hits(_content(body)[0])
if hits:
artifact = {**artifact, "sensitivity": "restricted"}
entry: dict[str, Any] = {"version": number, "created_at": now_iso(), "created_by": _actor(request.identity), "content_ref": _store_content(request.store, body), "diff_from": artifact["current_version"]}
if body.get("diff_from") is not None:
entry["diff_from"] = _version(artifact, body["diff_from"])["version"]
for field, limit in (("message_id", 120), ("note", 200)):
if _string(body, field, limit, required=False) is not None:
entry[field] = body[field]
lineage = _lineage(request, body)
if lineage is not None:
entry["lineage"] = lineage
stored = append_version(request.store, artifact, entry, expected)
remember(request.store, FAMILY, key, f"{stored['id']}@{number}")
request.audit("artifacts.version", f"{stored['id']}@{number}", reason="secret_pattern_restricted" if hits else "")
emit_event(request.store, request.identity, stored, "artifact.version", f"New version {number}", {"artifact_id": stored["id"], "version": number})
return Response(201, stored, {"ETag": str(stored["revision"])})
def _version_number(request: Request) -> int:
raw = request.params["n"]
if not raw.isdigit() or int(raw) < 1:
raise Invalid("version must be a positive integer")
return int(raw)
def get_version(request: Request) -> Response:
"""``GET /hux/v1/artifacts/{id}/versions/{n}``: version metadata plus content."""
artifact = _load_owned(request)
entry = _version(artifact, _version_number(request))
data = request.store.get_blob(entry["content_ref"]["hash"].split(":", 1)[1])
text = diffs.decode_text(data) if diffs.is_text_type(artifact["type"]) else None
body: dict[str, Any] = {"artifact_id": artifact["id"], "type": artifact["type"], "version": entry}
if text is None:
body["content_base64"] = base64.b64encode(data).decode("ascii")
else:
body["content"] = text
request.audit("artifacts.get_version", f"{artifact['id']}@{entry['version']}")
headers = {"X-Content-Type-Options": "nosniff", "Content-Disposition": "attachment", "ETag": str(artifact["revision"])}
return Response(200, body, headers)
def diff(request: Request) -> Response:
"""``GET /hux/v1/artifacts/{id}/versions/{n}/diff?from=``: unified diff for text, sizes and hashes otherwise."""
artifact = _load_owned(request)
to_entry = _version(artifact, _version_number(request))
raw_from = request.query.get("from", "")
if raw_from and not raw_from.isdigit():
raise Invalid("from must be a version number")
from_number = int(raw_from) if raw_from else to_entry.get("diff_from", to_entry["version"])
from_entry = _version(artifact, from_number)
older = request.store.get_blob(from_entry["content_ref"]["hash"].split(":", 1)[1])
newer = request.store.get_blob(to_entry["content_ref"]["hash"].split(":", 1)[1])
request.audit("artifacts.diff", f"{artifact['id']}@{from_number}..{to_entry['version']}")
return Response(200, diffs.unified(artifact["type"], older, newer, from_number, to_entry["version"]), {"X-Content-Type-Options": "nosniff"})
def promote(request: Request) -> Response:
"""``POST /hux/v1/artifacts/{id}/promote``: mark the current version as the project's copy."""
body = _body(request)
project_id = check_id(body.get("project_id"))
if not _project_exists(request.store, project_id):
raise NotFound("project not found")
expected = request.if_match()
with request.store.lock(FAMILY):
artifact = _load_owned(request)
if expected is not None and expected != artifact["revision"]:
raise Conflict(f"revision {expected} does not match current revision {artifact['revision']}", [str(artifact["revision"])])
version = artifact["current_version"]
if body.get("version") is not None:
version = _version(artifact, body["version"])["version"]
stamp = now_iso()
updated = {**artifact, "project_id": project_id, "promotion": {"project_id": project_id, "version": version, "at": stamp}, "updated_at": stamp}
_check(updated)
stored = request.store.put(FAMILY, updated, expected_revision=artifact["revision"])
request.audit("artifacts.promote", f"{stored['id']}@{version}", reason="" if expected is not None else "unconditional_write")
emit_event(request.store, request.identity, stored, "artifact.promoted", f"Promoted version {version} to project", {"artifact_id": stored["id"], "version": version, "project_id": project_id})
return Response(200, stored, {"ETag": str(stored["revision"])})
def register(router: Router) -> None:
"""Attach HUX-04 routes."""
card = "HUX-04"
router.add("POST", "/hux/v1/artifacts", card, "artifacts.create", create, max_body=MAX_UPLOAD_BODY)
router.add("GET", "/hux/v1/artifacts", card, "artifacts.list", list_artifacts)
router.add("GET", "/hux/v1/artifacts/{id}", card, "artifacts.get", get)
router.add("POST", "/hux/v1/artifacts/{id}/versions", card, "artifacts.version", add_version, max_body=MAX_UPLOAD_BODY)
router.add("GET", "/hux/v1/artifacts/{id}/versions/{n}", card, "artifacts.get_version", get_version)
router.add("GET", "/hux/v1/artifacts/{id}/versions/{n}/diff", card, "artifacts.diff", diff)
router.add("POST", "/hux/v1/artifacts/{id}/promote", card, "artifacts.promote", promote)

View File

@ -0,0 +1,40 @@
"""Auditable outcome for every read and mutation.
One JSONL ledger per tenant per day. The record shape is
``common.schema.json#/$defs/audit_outcome``; it never carries request bodies,
only the action name, the resource id and the decision.
"""
from __future__ import annotations
from hux.identity import Identity
from hux.store import TenantStore, now_iso
OUTCOMES = ("allow", "deny", "not_found", "conflict", "flag_off")
def record(store: TenantStore, identity: Identity, action: str, resource: str, outcome: str, reason: str = "") -> dict:
"""Append an audit outcome and return it."""
if outcome not in OUTCOMES:
raise ValueError(f"unknown outcome {outcome!r}")
entry = {
"at": now_iso(),
"identity": identity.record(),
"action": action,
"resource": resource[:200],
"outcome": outcome,
}
if reason:
entry["reason"] = reason[:200]
store.append("audit", now_iso()[:10], entry)
return entry
def recent(store: TenantStore, limit: int = 200) -> list[dict]:
"""Newest audit rows across day ledgers, newest last."""
rows: list[dict] = []
for name in reversed(store.ledgers("audit")):
rows = store.read("audit", name) + rows
if len(rows) >= limit:
break
return rows[-limit:]

View File

@ -0,0 +1,258 @@
"""HUX-05 run controls: budgets, the pre-side-effect gate and cancellation receipts.
The agent hook reports spend per run and asks the gate before every side
effect. The gate releases only against an approved, unexpired approval for the
same run, capability and argument hash; a ``once`` approval is consumed by its
first release (SO-36, SO-37). A stop is not done until a receipt says what
actually happened to the run and its side effects (SO-41).
"""
from __future__ import annotations
import re
from typing import Any
from hux import policy, rules
from hux.errors import Invalid
from hux.http import Request, Response, Router
from hux.identity import Identity
from hux.store import TenantStore, check_id
BUDGETS = "budgets"
RECEIPTS = "receipts"
SPEND_KEYS = ("tokens", "tool_calls", "wall_clock_seconds", "delegations", "spend_units", "subagents")
LIMIT_OF = {
"tokens": "tokens_per_run", "tool_calls": "tool_calls_per_run", "wall_clock_seconds": "wall_clock_seconds",
"delegations": "delegations_per_run", "spend_units": "spend_units", "subagents": "subagents_per_run",
}
RUN_ID_RE = re.compile(r"^[A-Za-z0-9._:-]{1,120}$")
def checked_run_id(value: Any) -> str:
"""Return a canonical run id accepted by both body and path APIs."""
if not isinstance(value, str) or not RUN_ID_RE.fullmatch(value):
raise Invalid("run id is malformed or too long")
return value
def run_id_from(request: Request) -> str:
"""The run id in the path; the router already bounded its alphabet."""
return checked_run_id(request.params["id"])
# -- budgets -------------------------------------------------------------------
def exhausted(spent: dict[str, int], limits: dict[str, Any]) -> list[str]:
"""Limit names whose spend has reached them; a zero limit is exhausted immediately."""
return [LIMIT_OF[k] for k in SPEND_KEYS if LIMIT_OF[k] in limits and spent.get(k, 0) >= limits[LIMIT_OF[k]]]
def budget_state(store: TenantStore, identity: Identity, run_id: str, conversation_id: str | None = None) -> dict[str, Any]:
"""Current state for a run, enforced against its conversation's policy epoch aggregate."""
run_id = checked_run_id(run_id)
doc_id = f"bud_{policy.run_key(run_id)}"
stored = store.get(BUDGETS, doc_id) if store.exists(BUDGETS, doc_id) else {"spent": {}, "_conversation_id": None}
known_conversation = stored.get("_conversation_id") or run_conversation(store, run_id)
if known_conversation and conversation_id and known_conversation != conversation_id:
raise Invalid("run is already bound to another conversation")
conversation_id = known_conversation or conversation_id
level, scope_id = ("conversation", conversation_id) if conversation_id else ("global", None)
effective = policy.effective_policy(store, identity, level, scope_id)
limits = {k: v for k, v in effective["budgets"].items() if k != "scope"}
epoch = str(effective.get("_budget_epoch") or f"{effective['id']}:{effective['revision']}")
same_epoch = stored.get("_budget_epoch") == epoch
run_spent = {k: int(stored["spent"].get(k, 0)) if same_epoch else 0 for k in SPEND_KEYS}
aggregate = {k: 0 for k in SPEND_KEYS}
for record in store.scan(BUDGETS):
if record.get("_conversation_id") != conversation_id or record.get("_budget_epoch") != epoch:
continue
for key in SPEND_KEYS:
aggregate[key] += int(record.get("spent", {}).get(key, 0))
return policy.checked({
"schema": "hux.budget_state.v1", "run_id": run_id[:120], "spent": aggregate, "limits": limits,
"exhausted": exhausted(aggregate, limits), "_conversation_id": conversation_id, "_id": doc_id,
"_run_spent": run_spent, "_budget_epoch": epoch,
})
def get_budget(request: Request) -> Response:
"""``GET /hux/v1/runs/{id}/budget``: spend so far and what is exhausted."""
state = budget_state(request.store, request.identity, run_id_from(request))
request.audit("budgets.read", state["_id"])
return Response(200, policy.public(state))
def post_budget(request: Request) -> Response:
"""``POST /hux/v1/runs/{id}/budget``: add spend increments; emits budget.exhausted on the crossing."""
policy.require_worker(request.identity, "budget reports")
body = policy.body_dict(request)
run_id = run_id_from(request)
conversation_id = body.get("conversation_id")
if conversation_id is not None:
conversation_id = check_id(conversation_id)
increments = {}
for key in SPEND_KEYS:
value = body.get(key, 0)
if not isinstance(value, int) or isinstance(value, bool) or value < 0:
raise Invalid(f"{key} must be a non-negative integer")
increments[key] = value
with request.store.lock(BUDGETS):
before = budget_state(request.store, request.identity, run_id, conversation_id)
known = before["exhausted"] if request.store.exists(BUDGETS, before["_id"]) else []
spent = {k: before["_run_spent"][k] + increments[k] for k in SPEND_KEYS}
doc = {
"id": before["_id"], "run_id": run_id, "spent": spent,
"_conversation_id": before["_conversation_id"], "_budget_epoch": before["_budget_epoch"],
}
request.store.put(BUDGETS, doc)
after = budget_state(request.store, request.identity, run_id, conversation_id)
request.audit("budgets.write", after["_id"])
newly = [k for k in after["exhausted"] if k not in known]
if newly:
policy.emit(request.store, request.identity, after["_conversation_id"], "budget.exhausted",
f"Budget exhausted: {', '.join(newly)}", run_id=run_id)
return Response(200, policy.public(after))
# -- gate ----------------------------------------------------------------------
def hashes_of(approval: dict[str, Any]) -> set[str]:
"""Argument hashes the approval was requested for (tool_call evidence with a hash)."""
return {e["hash"] for e in approval["request"].get("evidence", []) if e.get("kind") == "tool_call" and e.get("hash")}
def run_conversation(store: TenantStore, run_id: str) -> str | None:
"""The conversation a trusted Worker bound to this run in its budget document (F4)."""
doc_id = f"bud_{policy.run_key(run_id)}"
if store.exists(BUDGETS, doc_id) and store.get(BUDGETS, doc_id).get("_conversation_id"):
return store.get(BUDGETS, doc_id)["_conversation_id"]
return None
def matching_approval(store: TenantStore, run_id: str, capability: str, argument_hash: str, external: bool, conversation_id: str | None) -> tuple[dict[str, Any] | None, str, str | None]:
"""The approval that releases this side effect, or why none does, plus the conversation the gate settled on.
Every release uses an approval record created for this exact run. A
session/always grant may let a later run create a new approved record,
but the old record itself never crosses run ids. External effects also
require the same argument hash; ``once`` names exactly one hash (SO-36,
SO-37, SO-39).
"""
reason = "no approval for this run and capability"
for record in store.scan(policy.APPROVALS):
record = policy.refresh(store, record)
if record["capability"] != capability:
continue
same_run = record["run_id"] == run_id
choice = record.get("decision", {}).get("choice")
if not same_run:
continue
if record["status"] != "approved":
reason = f"approval {record['id']} is {record['status']}"
continue
if policy.parse(record["expires_at"]) <= policy.now():
reason = f"approval {record['id']} has expired"
continue
if external and not record["request"]["external"]:
reason = f"approval {record['id']} was not requested as external"
continue
if (external or choice == "once") and argument_hash not in hashes_of(record):
reason = f"approval {record['id']} was for different arguments"
continue
if choice == "once" and record.get("_consumed_at"):
reason = f"approval {record['id']} was already consumed"
continue
return record, "released", conversation_id or record["conversation_id"]
return None, reason, conversation_id
def gate(request: Request) -> Response:
"""``POST /hux/v1/runs/{id}/gate``: may this side effect proceed right now?"""
policy.require_worker(request.identity, "gate checks")
body = policy.body_dict(request)
run_id = run_id_from(request)
capability = body.get("capability")
argument_hash = body.get("argument_hash")
if capability not in rules.CAPABILITIES:
raise Invalid("unknown capability")
if not isinstance(argument_hash, str) or not argument_hash.startswith("sha256:"):
raise Invalid("argument_hash must be sha256:<hex>")
external = bool(body.get("external", False))
conversation_id = run_conversation(request.store, run_id)
if policy.private_mode_denies(request.store, conversation_id, capability):
request.audit("gate.check", f"{run_id}:{capability}", outcome="deny", reason="private_mode")
policy.emit(request.store, request.identity, conversation_id, "side_effect.blocked", f"{capability} blocked: private mode", run_id=run_id)
return Response(200, {"proceed": False, "reason": "private_mode"})
state = budget_state(request.store, request.identity, run_id, conversation_id)
if state["exhausted"]: # F6: an exhausted run releases nothing, whatever was approved
request.audit("gate.check", f"{run_id}:{capability}", outcome="deny", reason="budget_exhausted")
policy.emit(request.store, request.identity, conversation_id, "budget.exhausted", f"Budget exhausted: {', '.join(state['exhausted'])}", run_id=run_id)
return Response(200, {"proceed": False, "reason": "budget_exhausted", "exhausted": state["exhausted"]})
with request.store.lock(policy.APPROVALS):
record, reason, conversation_id = matching_approval(request.store, run_id, capability, argument_hash, external, conversation_id)
if record is not None and record["decision"]["choice"] == "once":
request.store.put(policy.APPROVALS, {**record, "_consumed_at": policy.iso(policy.now())})
if record is None:
request.audit("gate.check", f"{run_id}:{capability}", outcome="deny", reason="policy_violation")
policy.emit(request.store, request.identity, conversation_id, "side_effect.blocked", f"{capability} blocked: {reason}"[:280], run_id=run_id)
return Response(200, {"proceed": False, "reason": reason})
request.audit("gate.check", record["id"])
policy.emit(request.store, request.identity, conversation_id, "side_effect.released", f"{capability} released by approval {record['id']}",
run_id=run_id, evidence=[{"kind": "approval", "id": record["id"]}])
return Response(200, {"proceed": True, "approval_id": record["id"], "reason": reason})
# -- stop ----------------------------------------------------------------------
def stop_outcome(body: dict[str, Any]) -> tuple[str, str]:
"""Return the cancellation outcome proven by the gateway's process registry (F8, SO-41)."""
if body.get("already_complete"):
return "already_complete", "already_complete"
if body.get("process_registry_empty") is not True:
return "failed_to_cancel", "process_registry_not_empty"
return "cancelled", "cancelled"
def stop(request: Request) -> Response:
"""``POST /hux/v1/runs/{id}/stop``: write the cancellation receipt; a repeat returns it.
A ``failed_to_cancel`` receipt is the one non-terminal outcome: a later
stop that really cancels or finds the run complete supersedes it with a
revision bump; an identical repeat still replays (F8).
"""
policy.require_worker(request.identity, "stop receipts")
body = policy.body_dict(request)
run_id = run_id_from(request)
receipt_id = f"rcpt_{policy.run_key(run_id)}"
side_effects = body.get("side_effects", [])
if not isinstance(side_effects, list):
raise Invalid("side_effects must be a list")
outcome, reason = stop_outcome(body)
with request.store.lock(RECEIPTS):
existing = request.store.get(RECEIPTS, receipt_id) if request.store.exists(RECEIPTS, receipt_id) else None
if existing is not None and (existing["outcome"] != "failed_to_cancel" or outcome == "failed_to_cancel"):
request.audit("runs.stop", receipt_id, reason="replayed")
return Response(200, policy.public(existing))
stamp = policy.iso(policy.now())
record: dict[str, Any] = {
"schema": "hux.cancel_receipt.v1", "id": receipt_id, "run_id": run_id, "requested_by": policy.actor_for(request.identity),
"requested_at": existing["requested_at"] if existing else stamp, "acknowledged_at": stamp, "outcome": outcome, "side_effects": side_effects,
}
if outcome != "failed_to_cancel":
record["completed_at"] = stamp
conversation_id = body.get("conversation_id", existing.get("conversation_id") if existing else None)
if conversation_id is not None:
record["conversation_id"] = check_id(conversation_id)
stored = request.store.put(RECEIPTS, policy.checked(record), expected_revision=existing["revision"] if existing else None)
request.audit("runs.stop", receipt_id, reason=reason if existing is None else f"superseded:{reason}")
policy.emit(request.store, request.identity, record.get("conversation_id"), "run.cancelled", f"Run stopped: {outcome} ({reason})",
run_id=run_id, evidence=[{"kind": "run", "id": receipt_id}])
return Response(201, policy.public(stored))
def register_routes(router: Router) -> None:
"""Attach the run-scoped HUX-05 routes; called by ``hux.policy.register``."""
router.add("GET", "/hux/v1/runs/{id}/budget", policy.CARD, "budgets.read", get_budget)
router.add("POST", "/hux/v1/runs/{id}/budget", policy.CARD, "budgets.write", post_budget)
router.add("POST", "/hux/v1/runs/{id}/gate", policy.CARD, "gate.check", gate)
router.add("POST", "/hux/v1/runs/{id}/stop", policy.CARD, "runs.stop", stop)

View File

@ -0,0 +1,244 @@
"""Load and validate the HUX (Hermes user-experience) contract schemas.
The schemas under ``services/hermes/contracts/hux`` are plain JSON Schema
2020-12 so browser and Go consumers can validate with their usual libraries.
CI has no ``jsonschema`` package, so this module carries a small validator for
the keyword subset the contracts actually use. Unsupported keywords fail
loudly rather than silently passing.
"""
from __future__ import annotations
import json
import os
import re
from pathlib import Path
from typing import Any
CONTRACT_DIR = Path(
os.environ.get("HUX_CONTRACT_DIR")
or Path(__file__).resolve().parents[3] / "services" / "hermes" / "contracts" / "hux"
)
SCHEMA_FILES = (
"common.schema.json",
"identity.schema.json",
"event.schema.json",
"memory.schema.json",
"project.schema.json",
"artifact.schema.json",
"permission.schema.json",
"mode.schema.json",
"multimodal.schema.json",
"citation.schema.json",
"suggestion.schema.json",
"privacy.schema.json",
"release.schema.json",
"release-ledger.schema.json",
)
SUPPORTED_KEYWORDS = frozenset(
{
"$schema", "$id", "$defs", "$ref", "title", "description",
"type", "const", "enum", "required", "properties",
"additionalProperties", "items", "minItems", "maxItems",
"uniqueItems", "minLength", "maxLength", "pattern",
"minimum", "maximum", "oneOf",
}
)
_TYPE_CHECKS = {
"object": lambda v: isinstance(v, dict),
"array": lambda v: isinstance(v, list),
"string": lambda v: isinstance(v, str),
"boolean": lambda v: isinstance(v, bool),
"integer": lambda v: isinstance(v, int) and not isinstance(v, bool),
"number": lambda v: isinstance(v, (int, float)) and not isinstance(v, bool),
"null": lambda v: v is None,
}
class ContractError(ValueError):
"""Raised when a schema uses something this validator does not support."""
def load_schema(name: str, directory: Path = CONTRACT_DIR) -> dict[str, Any]:
"""Read one schema file by name."""
return json.loads((directory / name).read_text(encoding="utf-8"))
def load_all(directory: Path = CONTRACT_DIR) -> dict[str, dict[str, Any]]:
"""Read every contract schema keyed by file name."""
return {name: load_schema(name, directory) for name in SCHEMA_FILES}
def load_flags(directory: Path = CONTRACT_DIR) -> dict[str, Any]:
"""Read the feature flag registry."""
return json.loads((directory / "flags.json").read_text(encoding="utf-8"))
def _walk(node: Any, path: str, problems: list[str]) -> None:
if isinstance(node, dict):
for key, value in node.items():
if path.endswith(("/properties", "/$defs")):
_walk(value, f"{path}/{key}", problems)
continue
if key not in SUPPORTED_KEYWORDS:
problems.append(f"{path}/{key}")
continue
_walk(value, f"{path}/{key}", problems)
elif isinstance(node, list):
for index, value in enumerate(node):
_walk(value, f"{path}/{index}", problems)
def unsupported_keywords(schema: dict[str, Any]) -> list[str]:
"""Return JSON-pointer style paths of keywords the validator ignores."""
problems: list[str] = []
_walk(schema, "#", problems)
return problems
def _resolve_ref(ref: str, current: str, schemas: dict[str, dict[str, Any]]) -> tuple[dict[str, Any], str]:
file_part, _, pointer = ref.partition("#")
file_name = file_part or current
if file_name not in schemas:
raise ContractError(f"unknown schema reference {ref!r}")
node: Any = schemas[file_name]
for token in [t for t in pointer.split("/") if t]:
if not isinstance(node, dict) or token not in node:
raise ContractError(f"unresolvable pointer {ref!r}")
node = node[token]
return node, file_name
def _check_type(schema: dict[str, Any], value: Any, path: str, errors: list[str]) -> bool:
expected = schema.get("type")
if expected is None:
return True
if expected not in _TYPE_CHECKS:
raise ContractError(f"unsupported type {expected!r} at {path}")
if not _TYPE_CHECKS[expected](value):
errors.append(f"{path}: expected {expected}")
return False
return True
def _check_scalars(schema: dict[str, Any], value: Any, path: str, errors: list[str]) -> None:
if "const" in schema and value != schema["const"]:
errors.append(f"{path}: expected constant {schema['const']!r}")
if "enum" in schema and value not in schema["enum"]:
errors.append(f"{path}: {value!r} not in enum")
if isinstance(value, str):
if "minLength" in schema and len(value) < schema["minLength"]:
errors.append(f"{path}: shorter than {schema['minLength']}")
if "maxLength" in schema and len(value) > schema["maxLength"]:
errors.append(f"{path}: longer than {schema['maxLength']}")
if "pattern" in schema and not re.search(schema["pattern"], value):
errors.append(f"{path}: does not match {schema['pattern']!r}")
if isinstance(value, (int, float)) and not isinstance(value, bool):
if "minimum" in schema and value < schema["minimum"]:
errors.append(f"{path}: below minimum {schema['minimum']}")
if "maximum" in schema and value > schema["maximum"]:
errors.append(f"{path}: above maximum {schema['maximum']}")
def _check_object(schema, value, path, errors, current, schemas) -> None:
properties = schema.get("properties", {})
for key in schema.get("required", []):
if key not in value:
errors.append(f"{path}: missing required {key!r}")
for key, item in value.items():
if key in properties:
_validate(properties[key], item, f"{path}/{key}", errors, current, schemas)
elif schema.get("additionalProperties") is False:
errors.append(f"{path}: unexpected property {key!r}")
def _check_array(schema, value, path, errors, current, schemas) -> None:
if "minItems" in schema and len(value) < schema["minItems"]:
errors.append(f"{path}: fewer than {schema['minItems']} items")
if "maxItems" in schema and len(value) > schema["maxItems"]:
errors.append(f"{path}: more than {schema['maxItems']} items")
if schema.get("uniqueItems"):
seen = [json.dumps(item, sort_keys=True) for item in value]
if len(set(seen)) != len(seen):
errors.append(f"{path}: items are not unique")
if "items" in schema:
for index, item in enumerate(value):
_validate(schema["items"], item, f"{path}/{index}", errors, current, schemas)
def _check_one_of(schema, value, path, errors, current, schemas) -> None:
attempts: list[list[str]] = []
for option in schema["oneOf"]:
sub: list[str] = []
_validate(option, value, path, sub, current, schemas)
attempts.append(sub)
matches = sum(not sub for sub in attempts)
if matches != 1:
errors.append(f"{path}: matched {matches} oneOf branches, expected exactly 1")
if matches == 0:
errors.extend(min(attempts, key=len))
def _validate(schema, value, path, errors, current, schemas) -> None:
if "$ref" in schema:
target, file_name = _resolve_ref(schema["$ref"], current, schemas)
_validate(target, value, path, errors, file_name, schemas)
return
if not _check_type(schema, value, path, errors):
return
_check_scalars(schema, value, path, errors)
if isinstance(value, dict):
_check_object(schema, value, path, errors, current, schemas)
if isinstance(value, list):
_check_array(schema, value, path, errors, current, schemas)
if "oneOf" in schema:
_check_one_of(schema, value, path, errors, current, schemas)
def validate(
schema_name: str,
value: Any,
schemas: dict[str, dict[str, Any]] | None = None,
pointer: str = "",
) -> list[str]:
"""Validate ``value`` against a schema file, or a ``#/$defs/...`` pointer inside it.
Returns a list of human-readable problems; an empty list means valid.
"""
schemas = schemas or load_all()
schema, file_name = _resolve_ref(f"{schema_name}#{pointer}", schema_name, schemas)
errors: list[str] = []
_validate(schema, value, "$", errors, file_name, schemas)
return errors
def record_schema_names(schemas: dict[str, dict[str, Any]] | None = None) -> dict[str, str]:
"""Map every ``schema`` constant (e.g. ``hux.event.v1``) to its file name."""
schemas = schemas or load_all()
found: dict[str, str] = {}
def visit(node: Any, file_name: str) -> None:
if isinstance(node, dict):
const = node.get("properties", {}).get("schema", {}).get("const")
if isinstance(const, str):
found[const] = file_name
for child in node.values():
visit(child, file_name)
elif isinstance(node, list):
for child in node:
visit(child, file_name)
for file_name, schema in schemas.items():
visit(schema, file_name)
return found
def validate_record(value: Any, schemas: dict[str, dict[str, Any]] | None = None) -> list[str]:
"""Validate a record by its own ``schema`` field."""
schemas = schemas or load_all()
if not isinstance(value, dict) or not isinstance(value.get("schema"), str):
return ["$: record has no string 'schema' field"]
file_name = record_schema_names(schemas).get(value["schema"])
if file_name is None:
return [f"$: unknown record schema {value['schema']!r}"]
return validate(file_name, value, schemas)

View File

@ -0,0 +1,50 @@
"""Version-to-version comparison for artifacts (HUX-04).
Text artifacts get a unified diff from ``difflib``; anything that is not
valid UTF-8, or whose artifact type is binary, is described by sizes and
hashes only so the UI never tries to render bytes as text.
"""
from __future__ import annotations
import difflib
import hashlib
from typing import Any
TEXT_TYPES = frozenset({"markdown", "code", "html", "svg", "json", "csv"})
def is_text_type(artifact_type: str) -> bool:
"""True for artifact types whose content is served as UTF-8 text."""
return artifact_type in TEXT_TYPES
def decode_text(data: bytes) -> str | None:
"""UTF-8 decode, or None when the bytes are not text."""
try:
return data.decode("utf-8")
except UnicodeDecodeError:
return None
def unified(artifact_type: str, older: bytes, newer: bytes, from_n: int, to_n: int) -> dict[str, Any]:
"""Diff record ``{"from", "to", "unified"|"binary"}`` for two versions."""
record: dict[str, Any] = {"from": from_n, "to": to_n}
old_text = decode_text(older) if is_text_type(artifact_type) else None
new_text = decode_text(newer) if is_text_type(artifact_type) else None
if old_text is None or new_text is None:
record["binary"] = {
"from_bytes": len(older),
"to_bytes": len(newer),
"from_hash": "sha256:" + hashlib.sha256(older).hexdigest(),
"to_hash": "sha256:" + hashlib.sha256(newer).hexdigest(),
}
return record
lines = difflib.unified_diff(
old_text.splitlines(keepends=True),
new_text.splitlines(keepends=True),
fromfile=f"v{from_n}",
tofile=f"v{to_n}",
)
record["unified"] = "".join(lines)
return record

View File

@ -0,0 +1,92 @@
"""Error types that map one-to-one onto ``hux.error.v1`` records."""
from __future__ import annotations
class HuxError(Exception):
"""Base class; ``status`` and ``code`` follow identity.schema.json#/$defs/error."""
status = 500
code = "invalid"
def __init__(self, message: str, details: list[str] | None = None) -> None:
super().__init__(message)
self.message = message
self.details = list(details or [])
def record(self) -> dict:
"""Serialise as a ``hux.error.v1`` record."""
body = {"schema": "hux.error.v1", "status": self.status, "code": self.code, "message": self.message[:280]}
if self.details:
body["details"] = [d[:280] for d in self.details[:32]]
return body
class Unauthorized(HuxError):
"""Identity headers missing, malformed or not trusted."""
status, code = 401, "unauthorized"
class Forbidden(HuxError):
"""Identity is valid but does not own the resource."""
status, code = 403, "forbidden"
class NotFound(HuxError):
"""Resource does not exist for this tenant."""
status, code = 404, "not_found"
class FlagOff(HuxError):
"""Card (or one of its dependencies) is disabled; indistinguishable from not found on purpose."""
status, code = 404, "flag_off"
class Conflict(HuxError):
"""If-Match revision mismatch or duplicate create."""
status, code = 409, "conflict"
class Invalid(HuxError):
"""Body failed contract validation."""
status, code = 400, "invalid"
class Unprocessable(HuxError):
"""Well-formed body that violates a rule (hash mismatch, policy violation)."""
status, code = 422, "unprocessable"
class TooLarge(HuxError):
"""Body, record or family exceeds its bound."""
status, code = 413, "too_large"
class RateLimited(HuxError):
"""The caller exceeded the bounded per-subject request rate."""
status, code = 429, "rate_limited"
def __init__(self, retry_after: int) -> None:
super().__init__("request rate limit exceeded")
self.retry_after = max(1, int(retry_after))
class ApprovalRequired(HuxError):
"""An action needs an approval record before it may proceed."""
status, code = 403, "approval_required"
class BudgetExhausted(HuxError):
"""A run budget has been spent."""
status, code = 429, "budget_exhausted"

View File

@ -0,0 +1,315 @@
"""HUX-01 activity events: the per-conversation append-only log and its readers.
``emit`` is the one write path every family uses. It runs the redaction
pipeline, allocates ``seq`` under the conversation lock, honours idempotency
keys and keeps a checkpoint document per conversation. Readers page by
``after_seq`` or follow an SSE stream whose ``id:`` is the seq, and every
served record passes serve-time redaction for the caller's surface.
"""
from __future__ import annotations
import json
import threading
import time
from pathlib import Path
from typing import Any
from collections.abc import Iterator
from hux import contracts, redaction
from hux.errors import Invalid, NotFound, TooLarge
from hux.http import Request, Response, Router, page
from hux.identity import Identity
from hux.store import TenantStore, new_id, now_iso
FAMILY = "events"
SEQ_FAMILY = "events_seq"
IDEM_FAMILY = "events_idem"
PAGE_MAX = 200
PAGE_DEFAULT = 100
STREAM_MAX_POLLS = 900
STREAM_POLL_SECONDS = 1.0
SCHEMAS = contracts.load_all()
KINDS = frozenset(contracts.load_schema("event.schema.json")["properties"]["kind"]["enum"])
USER_KINDS = frozenset({"message.user", "approval.resolved", "run.cancelled", "memory.forgotten", "suggestion.dismissed"})
_streams: dict[str, int] = {}
_streams_guard = threading.Lock()
def _seq_id(conversation_id: str) -> str:
return f"seq_{conversation_id}"
def is_private(store: TenantStore, conversation_id: str) -> bool:
"""True when the conversation document (HUX-03) says the mode is private (SO-28)."""
try:
return store.get("conversations", conversation_id).get("mode") == "private"
except NotFound:
return False
def _ledger_path(store: TenantStore, conversation_id: str) -> Path:
return store.root / FAMILY / f"{conversation_id}.jsonl"
def conversation_known(store: TenantStore, conversation_id: str) -> bool:
"""Ownership check (SO-18): the conversation document, its seq checkpoint or its event ledger exists in this subject's tree."""
return (
store.exists("conversations", conversation_id)
or store.exists(SEQ_FAMILY, _seq_id(conversation_id))
or _ledger_path(store, conversation_id).is_file()
)
def last_seq(store: TenantStore, conversation_id: str) -> int:
"""The seq of the newest complete ledger line, read from the file tail (0 when the ledger is empty or absent).
A crash between the ledger append and the checkpoint write leaves the
checkpoint behind the ledger; ``emit`` takes the larger of the two so a seq
is never handed out twice (F5). Only the last ``LINE_CAP_BYTES`` window is
read, and a torn trailing line is skipped.
"""
path = _ledger_path(store, conversation_id)
if not path.is_file():
return 0
with open(path, "rb") as handle:
handle.seek(0, 2)
size = handle.tell()
handle.seek(max(0, size - redaction.LINE_CAP_BYTES - 2))
tail = handle.read()
for line in reversed(tail.split(b"\n")):
if not line:
continue
try:
return int(json.loads(line).get("seq", 0))
except (json.JSONDecodeError, AttributeError, ValueError):
continue
return 0
def _actor(identity: Identity, kind: str) -> dict[str, str]:
if kind in USER_KINDS and identity.trust != "worker":
return {"type": "user", "id": identity.subject}
if identity.trust == "worker":
return {"type": "system", "id": "hux-worker"}
return {"type": "assistant", "id": "hermes"}
def _find_replay(store: TenantStore, conversation_id: str, key: str) -> dict[str, Any] | None:
for row in store.read(IDEM_FAMILY, conversation_id):
if row.get("idempotency_key") == key:
for event in store.read(FAMILY, conversation_id):
if event.get("id") == row.get("event_id"):
return event
return None
def build(identity: Identity, conversation_id: str, kind: str, summary: str, detail: dict | None, evidence: list | None,
sensitivity: str, run_id: str | None, turn: int | None, correlation_id: str | None, idempotency_key: str | None) -> dict[str, Any]:
"""Run the redaction pipeline and shape an unsequenced ``hux.event.v1`` record."""
if kind not in KINDS:
raise Invalid(f"unknown event kind {kind!r}")
if sensitivity not in ("public", "personal", "sensitive", "restricted"):
raise Invalid("unknown sensitivity")
hits: list[str] = []
summary_clean = redaction.scrub_value(str(summary or "").strip(), hits)[: redaction.SUMMARY_MAX] or "[empty]"
detail_clean = redaction.scrub_value(redaction.filter_detail(kind, detail), hits)
detail_clean, truncated = redaction.cap_detail(detail_clean)
evidence_clean = redaction.scrub_value(redaction.filter_evidence(evidence), hits)
stamp = now_iso()
record: dict[str, Any] = {
"schema": "hux.event.v1",
"id": new_id("evt"),
"seq": 0,
"ts": stamp,
"conversation_id": conversation_id,
"kind": kind,
"summary": summary_clean,
"provenance": {"surface": identity.surface, "actor": _actor(identity, kind), "recorded_at": stamp, "conversation_id": conversation_id},
"sensitivity": sensitivity,
"redaction": redaction.derive_level(sensitivity, hits, truncated),
"turn": max(0, int(turn or 0)),
"identity": identity.record(),
}
if detail_clean:
record["detail"] = detail_clean
if evidence_clean:
record["evidence"] = evidence_clean
if run_id:
record["run_id"] = str(run_id)[:120]
record["provenance"]["run_id"] = record["run_id"]
if correlation_id:
record["correlation_id"] = str(correlation_id)[:120]
if idempotency_key:
record["idempotency_key"] = idempotency_key
return record
def emit_with_status(store: TenantStore, identity: Identity, conversation_id: str, kind: str, summary: str, detail: dict | None = None,
evidence: list | None = None, sensitivity: str = "personal", run_id: str | None = None, turn: int | None = None,
correlation_id: str | None = None, idempotency_key: str | None = None) -> tuple[dict[str, Any] | None, bool]:
"""Like ``emit`` but also says whether the record was an idempotent replay. None means private mode (nothing written)."""
if is_private(store, conversation_id):
return None, False
record = build(identity, conversation_id, kind, summary, detail, evidence, sensitivity, run_id, turn, correlation_id, idempotency_key)
with store.lock(f"{FAMILY}:{conversation_id}"):
if idempotency_key:
existing = _find_replay(store, conversation_id, idempotency_key)
if existing is not None:
return existing, True
seq_id = _seq_id(conversation_id)
checkpoint = store.get(SEQ_FAMILY, seq_id) if store.exists(SEQ_FAMILY, seq_id) else {"id": seq_id, "next_seq": 1, "last_event_id": ""}
record["seq"] = max(int(checkpoint["next_seq"]), last_seq(store, conversation_id) + 1)
problems = contracts.validate_record(record, SCHEMAS)
if problems:
raise Invalid("event failed contract validation", problems)
line = json.dumps(record, sort_keys=True, separators=(",", ":")).encode()
if len(line) > redaction.LINE_CAP_BYTES:
raise TooLarge("event line exceeds 64 KiB")
store.append(FAMILY, conversation_id, record)
store.put(SEQ_FAMILY, {**checkpoint, "next_seq": record["seq"] + 1, "last_event_id": record["id"], "checkpointed_at": now_iso()})
if idempotency_key:
store.append(IDEM_FAMILY, conversation_id, {"idempotency_key": idempotency_key, "event_id": record["id"], "seq": record["seq"], "at": record["ts"]})
return record, False
def emit(store: TenantStore, identity: Identity, conversation_id: str, kind: str, summary: str, detail: dict | None = None,
evidence: list | None = None, sensitivity: str = "personal", run_id: str | None = None, turn: int | None = None,
correlation_id: str | None = None, idempotency_key: str | None = None) -> dict[str, Any] | None:
"""Append one event and return the stored record (None when the conversation is private)."""
record, _ = emit_with_status(store, identity, conversation_id, kind, summary, detail, evidence, sensitivity, run_id, turn, correlation_id, idempotency_key)
return record
def read_after(store: TenantStore, conversation_id: str, after_seq: int, limit: int) -> list[dict[str, Any]]:
"""Stored events with ``seq > after_seq``, oldest first, at most ``limit``."""
rows = [row for row in store.read(FAMILY, conversation_id) if int(row.get("seq", 0)) > after_seq]
return rows[:limit]
def rewrite_full(store: TenantStore, conversation_id: str, summary: str, reason: str, match=None) -> int:
"""Rewrite matching events of one conversation to ``redaction.level: full`` (SO-24, forget, decay). Returns the count."""
with store.lock(f"{FAMILY}:{conversation_id}"):
rows = store.read(FAMILY, conversation_id)
changed = 0
out: list[dict[str, Any]] = []
for row in rows:
if row.get("redaction", {}).get("level") != "full" and (match is None or match(row)):
row = redaction.full_redaction(row, summary, reason)
changed += 1
out.append(row)
if changed:
store.rewrite(FAMILY, conversation_id, out)
return changed
def redact_memory_references(store: TenantStore, memory_id: str) -> int:
"""Fully redact every event, in any conversation, that names a forgotten memory id (SO-24)."""
def references(row: dict[str, Any]) -> bool:
if row.get("detail", {}).get("memory_id") == memory_id:
return True
return any(ref.get("kind") == "memory" and ref.get("id") == memory_id for ref in row.get("evidence", []))
return sum(rewrite_full(store, name, "[forgotten memory]", "memory forgotten", references) for name in store.ledgers(FAMILY))
# -- routes ------------------------------------------------------------------
def _int_query(request: Request, name: str, default: int, floor: int = 0) -> int:
raw = request.query.get(name, "")
if raw == "":
return default
if not raw.lstrip("-").isdigit():
raise Invalid(f"{name} must be an integer")
return max(floor, int(raw))
def _require_conversation(request: Request) -> str:
conversation_id = request.params["id"]
if not conversation_known(request.store, conversation_id):
raise NotFound("conversation not found")
return conversation_id
def list_events(request: Request) -> Response:
"""``GET /hux/v1/conversations/{id}/events?after_seq=&limit=``: one page, ``next`` is the last seq served."""
conversation_id = _require_conversation(request)
after_seq = _int_query(request, "after_seq", 0)
limit = min(_int_query(request, "limit", PAGE_DEFAULT, 1), PAGE_MAX)
rows = read_after(request.store, conversation_id, after_seq, limit)
items = [redaction.redact_record(row, request.identity.surface) for row in rows]
request.audit("events.list", conversation_id)
return page(items, items[-1]["seq"] if len(items) == limit else None)
def append_event(request: Request) -> Response:
"""``POST /hux/v1/conversations/{id}/events``: append from the trusted hop; ids and seq are server-assigned (SO-15)."""
conversation_id = _require_conversation(request)
body = request.body if isinstance(request.body, dict) else None
if body is None:
raise Invalid("body must be an object")
if "id" in body or "seq" in body:
raise Invalid("id and seq are server-assigned")
if not isinstance(body.get("kind"), str) or not isinstance(body.get("summary"), str):
raise Invalid("kind and summary are required")
detail = body.get("detail") if isinstance(body.get("detail"), dict) else None
evidence = body.get("evidence") if isinstance(body.get("evidence"), list) else None
turn = body.get("turn") if isinstance(body.get("turn"), int) else None
record, replayed = emit_with_status(
request.store, request.identity, conversation_id, body["kind"], body["summary"], detail, evidence,
body.get("sensitivity", "personal"), body.get("run_id"), turn, body.get("correlation_id"), request.idempotency_key() or None,
)
if record is None:
request.audit("events.append", conversation_id, "allow", "private_mode")
return Response(204)
request.audit("events.append", f"{conversation_id}/{record['id']}", "allow", "replayed" if replayed else "")
served = redaction.redact_record(record, request.identity.surface)
return Response(200 if replayed else 201, served, {"HUX-Replayed": "true"} if replayed else {})
def _sse(record: dict[str, Any]) -> bytes:
return f"id: {record['seq']}\nevent: {record['kind']}\ndata: {json.dumps(record, sort_keys=True)}\n\n".encode()
def stream_events(request: Request) -> Response:
"""``GET /hux/v1/conversations/{id}/events/stream``: SSE replay from ``Last-Event-ID`` (or ``after_seq``), then a bounded live poll.
``max_polls`` and ``poll_ms`` (query) bound the live phase so a client, or a
test, decides how long to wait; the server caps them at the 15 minute idle
limit (SO-17). A second stream for the same (subject, conversation) closes
the first.
"""
conversation_id = _require_conversation(request)
last_id = request.header("Last-Event-ID")
cursor = int(last_id) if last_id.isdigit() else _int_query(request, "after_seq", 0)
max_polls = min(_int_query(request, "max_polls", STREAM_MAX_POLLS), STREAM_MAX_POLLS)
poll_seconds = min(_int_query(request, "poll_ms", int(STREAM_POLL_SECONDS * 1000)), 60000) / 1000
store, surface = request.store, request.identity.surface
key = f"{request.identity.subject}:{conversation_id}"
with _streams_guard:
token = _streams[key] = _streams.get(key, 0) + 1
request.audit("events.stream", conversation_id)
def generate() -> Iterator[bytes]:
position = cursor
yield b"retry: 2000\n\n"
polls = 0
while True:
for record in read_after(store, conversation_id, position, PAGE_MAX):
position = record["seq"]
yield _sse(redaction.redact_record(record, surface))
if polls >= max_polls or _streams.get(key) != token:
break
polls += 1
yield b": keepalive\n\n"
time.sleep(poll_seconds)
return Response(200, stream=generate)
def register(router: Router) -> None:
"""Attach HUX-01 routes."""
router.add("GET", "/hux/v1/conversations/{id}/events", "HUX-01", "events.list", list_events)
router.add("POST", "/hux/v1/conversations/{id}/events", "HUX-01", "events.append", append_event)
router.add("GET", "/hux/v1/conversations/{id}/events/stream", "HUX-01", "events.stream", stream_events)

View File

@ -0,0 +1,150 @@
"""Per-card feature flags and the capabilities record clients negotiate with.
Flags come from the ``HUX_FLAGS`` comma list. A card counts as enabled only
when it and every card it depends on are enabled, so a half-configured
deployment fails closed. Route ownership per card is declared here so the
capabilities record can tell a client exactly what it may call, and the
worker allowlist (SO-08) says which of those a ``trust: worker`` caller may
reach at all.
"""
from __future__ import annotations
import os
import re
from collections.abc import Mapping
from hux.errors import FlagOff
from hux.identity import Identity
from hux.rules import flag_enabled, flag_registry
CONTRACT_VERSION = "1.1.0"
RELEASE_TAG = re.compile(r"^git-([0-9a-f]{40})-build-[1-9][0-9]*-release$")
CARD_ROUTES: dict[str, list[str]] = {
"HUX-11": ["/hux/v1/capabilities", "/hux/v1/manifest", "/hux/v1/context/bootstrap"],
"HUX-01": ["/hux/v1/conversations/{id}/events", "/hux/v1/conversations/{id}/events/stream"],
"HUX-02": ["/hux/v1/memory", "/hux/v1/memory/{id}", "/hux/v1/memory/{id}/{action}", "/hux/v1/memory/export"],
"HUX-03": ["/hux/v1/projects", "/hux/v1/projects/{id}", "/hux/v1/conversations", "/hux/v1/conversations/{id}", "/hux/v1/conversations/{id}/branch", "/hux/v1/conversations/{id}/lineage", "/hux/v1/search"],
"HUX-04": ["/hux/v1/artifacts", "/hux/v1/artifacts/{id}", "/hux/v1/artifacts/{id}/versions", "/hux/v1/artifacts/{id}/versions/{n}", "/hux/v1/artifacts/{id}/versions/{n}/diff", "/hux/v1/artifacts/{id}/promote"],
"HUX-05": ["/hux/v1/policy", "/hux/v1/approvals", "/hux/v1/approvals/{id}", "/hux/v1/runs/{id}/stop", "/hux/v1/runs/{id}/budget", "/hux/v1/runs/{id}/gate"],
"HUX-06": ["/hux/v1/modes", "/hux/v1/projects/{project_id}/conversations/{id}/mode"],
"HUX-07": [
"/hux/v1/projects/{project_id}/conversations/{id}/multimodal/items",
"/hux/v1/projects/{project_id}/conversations/{id}/multimodal/items/{item_id}",
"/hux/v1/projects/{project_id}/conversations/{id}/multimodal/items/{item_id}/transcript-corrections",
"/hux/v1/projects/{project_id}/conversations/{id}/capture-intents",
],
"HUX-08": ["/hux/v1/sources", "/hux/v1/sources/{id}", "/hux/v1/passages", "/hux/v1/messages/{id}/citations", "/hux/v1/notebooks", "/hux/v1/notebooks/{id}"],
"HUX-09": [
"/hux/v1/projects/{project_id}/conversations/{id}/suggestions/evaluate",
"/hux/v1/projects/{project_id}/conversations/{id}/suggestions/{suggestion_id}/decisions",
"/hux/v1/projects/{project_id}/conversations/{id}/suggestions/states",
],
"HUX-10": ["/hux/v1/privacy/policy", "/hux/v1/privacy/notices", "/hux/v1/conversations/{id}/forget", "/hux/v1/privacy/audit", "/hux/v1/conversations/{id}/privacy"],
"HUX-12": [
"/hux/v1/projects/{project_id}/conversations/{id}/releases",
"/hux/v1/projects/{project_id}/conversations/{id}/releases/{release_id}",
"/hux/v1/projects/{project_id}/conversations/{id}/releases/{release_id}/transitions",
],
}
# The only (method, template) pairs a ``trust: worker`` caller may reach (SO-08).
# Everything else is 403 before the flag check. These are the agent-hook
# routes: negotiation, the approval/gate/budget/stop loop, activity events,
# the privacy policy, memory retrieval and proposals, research inputs and
# artifact writes. The conversation read is there so the hook can honour
# private mode (SO-28) before proposing a memory.
WORKER_ROUTES: frozenset[tuple[str, str]] = frozenset({
("GET", "/hux/v1/capabilities"), ("GET", "/hux/v1/manifest"),
("POST", "/hux/v1/context/bootstrap"),
("POST", "/hux/v1/approvals"),
("POST", "/hux/v1/runs/{id}/gate"), ("POST", "/hux/v1/runs/{id}/budget"), ("POST", "/hux/v1/runs/{id}/stop"),
("GET", "/hux/v1/runs/{id}/budget"),
("POST", "/hux/v1/conversations/{id}/events"), ("GET", "/hux/v1/conversations/{id}"),
("GET", "/hux/v1/privacy/policy"), ("GET", "/hux/v1/conversations/{id}/privacy"),
("GET", "/hux/v1/memory"), ("POST", "/hux/v1/memory"),
("POST", "/hux/v1/sources"), ("POST", "/hux/v1/passages"), ("POST", "/hux/v1/messages/{id}/citations"),
("POST", "/hux/v1/artifacts"), ("POST", "/hux/v1/artifacts/{id}/versions"),
("GET", "/hux/v1/modes"),
("GET", "/hux/v1/projects/{project_id}/conversations/{id}/mode"),
("POST", "/hux/v1/projects/{project_id}/conversations/{id}/multimodal/items"),
("GET", "/hux/v1/projects/{project_id}/conversations/{id}/releases"),
("GET", "/hux/v1/projects/{project_id}/conversations/{id}/releases/{release_id}"),
})
def worker_may_call(method: str, template: str) -> bool:
"""True when a ``trust: worker`` caller is allowed on this route (SO-08)."""
return (method, template) in WORKER_ROUTES
class Flags:
"""Snapshot of which cards are on for this process."""
def __init__(self, environ: Mapping[str, str] | None = None) -> None:
self._environ = dict(os.environ if environ is None else environ)
self._registry = flag_registry()
self._route_cards = frozenset(card for card, routes in CARD_ROUTES.items() if routes)
def bind_routes(self, routes: Mapping[str, set[str]]) -> None:
"""Bind capability flags to the route templates actually registered by this process."""
unknown = set(routes) - set(self._registry)
if unknown:
raise ValueError(f"routes registered for unknown cards: {sorted(unknown)}")
for card, actual in routes.items():
undeclared = actual - set(CARD_ROUTES.get(card, []))
if undeclared:
raise ValueError(f"undeclared routes for {card}: {sorted(undeclared)}")
self._route_cards = frozenset(
card for card, declared in CARD_ROUTES.items() if declared and set(declared) == routes.get(card, set())
)
def enabled(self, card: str) -> bool:
"""True only for a route-backed card whose configured flag chain is on."""
entry = self._registry.get(card)
enabled = bool(entry) and card in self._route_cards and flag_enabled(entry["flag"], self._environ)
if enabled and card == "HUX-12":
from hux.release_security import configured
return configured(self._environ)
return enabled
def require(self, card: str) -> None:
"""Raise FlagOff unless the card is enabled."""
if not self.enabled(card):
raise FlagOff(f"{card} is not enabled")
@property
def environ(self) -> Mapping[str, str]:
"""Read-only process configuration for route-family policy checks."""
return self._environ
def capabilities(self, identity: Identity, build: Mapping[str, str] | None = None) -> dict:
"""Serialise ``hux.capabilities.v1`` for one caller."""
cards = [
{"card": card, "flag": entry["flag"], "enabled": self.enabled(card), "routes": CARD_ROUTES.get(card, [])}
for card, entry in sorted(self._registry.items())
]
server = {k: v for k, v in (build or {}).items() if k in {"commit", "image_digest"} and v}
return {
"schema": "hux.capabilities.v1",
"contract_version": CONTRACT_VERSION,
"identity": identity.record(),
"cards": cards,
"server": server,
}
def build_from_environ(environ: Mapping[str, str] | None = None) -> dict[str, str]:
"""Commit and image digest the pod was started with, when the operator set them."""
environ = os.environ if environ is None else environ
tag = environ.get("HUX_IMAGE_TAG", "")
commit = environ.get("HUX_BUILD_COMMIT", "")
if tag:
match = RELEASE_TAG.fullmatch(tag)
if not match:
raise ValueError("HUX_IMAGE_TAG is not an immutable WebUI release tag")
if commit and commit != match.group(1):
raise ValueError("HUX build commit conflicts with the immutable image tag")
commit = match.group(1)
return {"commit": commit, "image_digest": environ.get("HUX_IMAGE_DIGEST", "")}

View File

@ -0,0 +1,185 @@
"""HUX-11 routes: capabilities, manifest, and trusted context bootstrap."""
from __future__ import annotations
import hashlib
import hmac
import os
import re
import stat
from pathlib import Path
from typing import Any
from hux import contracts, organization
from hux.errors import Conflict, Forbidden, Invalid
from hux.flags import CONTRACT_VERSION
from hux.http import Request, Response, Router
from hux.store import check_id, now_iso
CONTEXT_FAMILY = "context_bindings"
CONTEXT_SCHEMA = "hux.context_bootstrap.v1"
CONTEXT_KEY_BYTES = 32
CONTEXT_MESSAGE = b"hux.context.id.v1"
RAW_RE = re.compile(r"^[A-Za-z0-9._:/@+\-]{1,240}$")
CONTEXT_ID_RE = re.compile(r"^(ses|conv|prj)_[0-9a-f]{32}$")
SCHEMAS = contracts.load_all()
def _context_key(environ: dict[str, str]) -> bytes:
"""Read the exact 0600 regular key file; inline keys and links fail closed."""
raw_path = environ.get("HUX_CONTEXT_KEY_FILE", "")
if not raw_path:
raise Forbidden("context identity key is unavailable")
path = Path(raw_path)
try:
metadata = path.lstat()
except OSError as error:
raise Forbidden("context identity key is unavailable") from error
if not stat.S_ISREG(metadata.st_mode) or stat.S_IMODE(metadata.st_mode) != 0o600 or metadata.st_uid != os.geteuid():
raise Forbidden("context identity key is unsafe")
try:
value = path.read_bytes()
except OSError as error:
raise Forbidden("context identity key is unavailable") from error
if len(value) != CONTEXT_KEY_BYTES:
raise Forbidden("context identity key is invalid")
return value
def _source(body: dict[str, Any], name: str) -> str:
value = body.get(name)
if not isinstance(value, str) or not RAW_RE.fullmatch(value):
raise Invalid(f"{name} is malformed")
return value
def derive_context_id(key: bytes, purpose: str, identity: Any, raw: str) -> str:
"""Derive one frozen HMAC-bound context id."""
prefix = {"session": "ses", "conversation": "conv", "project": "prj"}[purpose]
message = b"\0".join(
(CONTEXT_MESSAGE, purpose.encode(), identity.tenant_slot.encode(), identity.subject.encode(), raw.encode())
)
return f"{prefix}_{hmac.new(key, message, hashlib.sha256).hexdigest()[:32]}"
def _exact_body(request: Request) -> tuple[dict[str, Any], str, str]:
if not isinstance(request.body, dict):
raise Invalid("body must be a JSON object")
expected = {"raw_session_id", "project_source", "session_id", "conversation_id", "project_id"}
if set(request.body) != expected:
raise Invalid("context bootstrap fields are not exact")
raw_session = _source(request.body, "raw_session_id")
project_source = _source(request.body, "project_source")
for name, prefix in (("session_id", "ses"), ("conversation_id", "conv"), ("project_id", "prj")):
value = request.body.get(name)
if not isinstance(value, str) or not CONTEXT_ID_RE.fullmatch(value) or not value.startswith(f"{prefix}_"):
raise Invalid(f"{name} is malformed")
check_id(value)
return request.body, raw_session, project_source
def _verify_record(record: dict[str, Any], schema: str, owner: str, project_id: str | None = None) -> None:
if record.get("schema") != schema or record.get("owner") != owner:
raise Conflict("context identity does not match existing record")
if project_id is not None and record.get("project_id") != project_id:
raise Conflict("context linkage does not match existing record")
if contracts.validate_record(record, SCHEMAS):
raise Conflict("existing context record is invalid")
def _response(request: Request, project: dict[str, Any], conversation: dict[str, Any], created: dict[str, bool]) -> Response:
body = {
"schema": CONTEXT_SCHEMA,
"contract_version": CONTRACT_VERSION,
"identity": request.identity.record(),
"session_id": request.body["session_id"],
"conversation_id": conversation["id"],
"project_id": project["id"],
"created": created,
"revisions": {"project": project["revision"], "conversation": conversation["revision"]},
}
return Response(201 if any(created.values()) else 200, body)
def bootstrap_context(request: Request) -> Response:
"""Create deterministic server-owned project and conversation records for trusted runtimes."""
if request.identity.trust not in {"relay", "worker"}:
raise Forbidden("context bootstrap requires relay or worker trust")
body, raw_session, project_source = _exact_body(request)
idem = request.idempotency_key()
if not idem:
raise Invalid("Idempotency-Key is required")
key = _context_key(request.flags._environ)
expected = {
"session_id": derive_context_id(key, "session", request.identity, raw_session),
"conversation_id": derive_context_id(key, "conversation", request.identity, raw_session),
"project_id": derive_context_id(key, "project", request.identity, project_source),
}
if any(not hmac.compare_digest(body[name], value) for name, value in expected.items()):
raise Invalid("context identifiers do not match authenticated inputs")
with request.store.lock(CONTEXT_FAMILY):
binding = request.store.get(CONTEXT_FAMILY, body["session_id"]) if request.store.exists(CONTEXT_FAMILY, body["session_id"]) else None
if binding:
valid_binding = (
binding.get("schema") == "hux.context_binding.v1"
and binding.get("id") == body["session_id"]
and binding.get("owner") == request.identity.subject
and all(binding.get(name) == body[name] for name in expected)
)
if not valid_binding:
raise Conflict("context binding does not match existing linkage")
for row in request.store.read(CONTEXT_FAMILY, "idempotency"):
if row.get("key") == idem and any(row.get(name) != body[name] for name in expected):
raise Conflict("Idempotency-Key was used for another context")
project = request.store.get(organization.PROJECTS, body["project_id"]) if request.store.exists(organization.PROJECTS, body["project_id"]) else None
conversation = request.store.get(organization.CONVERSATIONS, body["conversation_id"]) if request.store.exists(organization.CONVERSATIONS, body["conversation_id"]) else None
if project:
_verify_record(project, "hux.project.v1", request.identity.subject)
if conversation:
_verify_record(conversation, "hux.conversation.v1", request.identity.subject, body["project_id"])
if project is None and request.store.count(organization.PROJECTS) >= organization.MAX_PROJECTS:
raise Conflict("projects cap reached")
if conversation is None and request.store.count(organization.CONVERSATIONS) >= organization.MAX_CONVERSATIONS:
raise Conflict("conversations cap reached")
stamp = now_iso()
created = {"project": project is None, "conversation": conversation is None}
project = project or request.store.put(organization.PROJECTS, organization.checked({
"schema": "hux.project.v1", "id": body["project_id"], "owner": request.identity.subject,
"name": "Hermes", "tags": [], "pinned": False, "archived": False, "default_mode": "fast",
"created_at": stamp, "updated_at": stamp, "revision": 1,
}))
conversation = conversation or request.store.put(organization.CONVERSATIONS, organization.checked({
"schema": "hux.conversation.v1", "id": body["conversation_id"], "owner": request.identity.subject,
"project_id": body["project_id"], "title": "Hermes conversation", "tags": [], "pinned": False,
"archived": False, "mode": "fast", "artifact_ids": [], "created_at": stamp, "updated_at": stamp,
"revision": 1,
}))
if binding is None:
request.store.put(CONTEXT_FAMILY, {
"schema": "hux.context_binding.v1", "id": body["session_id"], "owner": request.identity.subject,
"session_id": body["session_id"], "conversation_id": body["conversation_id"],
"project_id": body["project_id"], "created_at": stamp,
})
if not any(row.get("key") == idem for row in request.store.read(CONTEXT_FAMILY, "idempotency")):
request.store.append(CONTEXT_FAMILY, "idempotency", {"key": idem, **expected})
request.audit("foundation.bootstrap", body["session_id"], reason="created" if any(created.values()) else "replayed")
return _response(request, project, conversation, created)
def capabilities(request: Request) -> Response:
"""``GET /hux/v1/capabilities``: what this tenant may call."""
request.audit("foundation.capabilities", "capabilities")
return Response(200, request.flags.capabilities(request.identity, request.flags_build))
def manifest(request: Request) -> Response:
"""``GET /hux/v1/manifest``: the on-disk layout version for rollback readers."""
request.audit("foundation.manifest", "manifest")
return Response(200, request.store.manifest(CONTRACT_VERSION))
def register(router: Router) -> None:
"""Attach HUX-11 routes."""
router.add("GET", "/hux/v1/capabilities", "HUX-11", "foundation.capabilities", capabilities)
router.add("GET", "/hux/v1/manifest", "HUX-11", "foundation.manifest", manifest)
router.add("POST", "/hux/v1/context/bootstrap", "HUX-11", "foundation.bootstrap", bootstrap_context, 4096)

View File

@ -0,0 +1,359 @@
"""Minimal HTTP layer for the per-tenant HUX service.
Stdlib only. A ``Router`` maps method + path template to a handler; family
modules register their routes with it. Every request resolves identity from
trusted headers, checks the owning card's flag, and writes an audit outcome.
Errors always leave as ``hux.error.v1``.
"""
from __future__ import annotations
import json
import math
import os
import re
import threading
import time
from collections.abc import Callable, Mapping
from dataclasses import dataclass, field
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from typing import Any
from urllib.parse import parse_qs, urlsplit
from hux import audit
from hux.errors import Forbidden, HuxError, Invalid, NotFound, RateLimited, TooLarge
from hux.flags import CONTRACT_VERSION, Flags, build_from_environ, worker_may_call
from hux.identity import Identity, resolve
from hux.store import TenantStore
MAX_BODY_BYTES = 1024 * 1024
MAX_QUERY_FIELDS = 64
DEFAULT_REQUEST_TIMEOUT_SECONDS = 10.0
DEFAULT_READS_PER_MINUTE = 300
DEFAULT_WRITES_PER_MINUTE = 30
MAX_RATE_BUCKETS = 4096
Handler = Callable[["Request"], "Response"]
def _bounded_number(environ: Mapping[str, str], name: str, default: float, low: float, high: float) -> float:
"""Read one numeric setting, using its safe default when malformed or outside bounds."""
try:
value = float(environ.get(name, default))
except (TypeError, ValueError):
return default
return value if math.isfinite(value) and low <= value <= high else default
class RateLimiter:
"""Small fixed-window limiter bounded by active subjects in the current minute."""
def __init__(self, reads: int, writes: int, clock: Callable[[], float] = time.monotonic) -> None:
self.limits = {"read": reads, "write": writes}
self.clock = clock
self._windows: dict[tuple[str, str], tuple[float, int]] = {}
self._lock = threading.Lock()
def check(self, subject: str, method: str) -> int | None:
"""Consume one request and return Retry-After seconds when its bucket is full."""
now = self.clock()
bucket = "read" if method in {"GET", "HEAD", "OPTIONS"} else "write"
key = (subject, bucket)
with self._lock:
# Each accepted identity can leave at most two current entries;
# expired identities disappear whenever any request arrives.
self._windows = {item: value for item, value in self._windows.items() if value[0] > now}
reset, used = self._windows.get(key, (now + 60.0, 0))
if key not in self._windows and len(self._windows) >= MAX_RATE_BUCKETS:
return 60
if used >= self.limits[bucket]:
return max(1, math.ceil(reset - now))
self._windows[key] = (reset, used + 1)
return None
@dataclass
class Request:
"""Everything a handler needs; no raw socket access."""
method: str
path: str
params: dict[str, str]
query: dict[str, str]
headers: Mapping[str, str]
body: Any
identity: Identity
store: TenantStore
flags: Flags
flags_build: dict[str, str] = field(default_factory=dict)
def if_match(self) -> int | None:
"""Parsed If-Match revision, or None when absent."""
raw = self.header("If-Match")
if raw == "":
return None
if not raw.isdigit():
raise Invalid("If-Match must be a revision integer")
return int(raw)
def idempotency_key(self) -> str:
"""Client idempotency key, validated against the contract pattern."""
raw = self.header("Idempotency-Key")
if raw and not re.match(r"^[A-Za-z0-9._:-]{8,120}$", raw):
raise Invalid("malformed Idempotency-Key")
return raw
def header(self, name: str) -> str:
"""Case-insensitive header lookup."""
for key, value in self.headers.items():
if key.lower() == name.lower():
return value.strip()
return ""
def audit(self, action: str, resource: str, outcome: str = "allow", reason: str = "") -> None:
"""Write an audit outcome for this request."""
audit.record(self.store, self.identity, action, resource, outcome, reason)
@dataclass
class Response:
"""JSON (or SSE) response."""
status: int = 200
body: Any = None
headers: dict[str, str] = field(default_factory=dict)
stream: Callable[[], Any] | None = None
@dataclass
class Route:
"""One registered handler."""
method: str
template: str
card: str
action: str
handler: Handler
max_body: int = MAX_BODY_BYTES
pattern: re.Pattern = field(init=False)
def __post_init__(self) -> None:
regex = re.sub(r"\{(\w+)\}", r"(?P<\1>[A-Za-z0-9._:-]+)", self.template)
self.pattern = re.compile(f"^{regex}$")
class Router:
"""Route table plus the request pipeline."""
def __init__(self, data_root: Path, environ: Mapping[str, str] | None = None) -> None:
self.data_root = Path(data_root)
self.environ = dict(os.environ if environ is None else environ)
self.flags = Flags(self.environ)
self.build = build_from_environ(self.environ)
self.routes: list[Route] = []
reads = int(_bounded_number(self.environ, "HUX_READS_PER_MINUTE", DEFAULT_READS_PER_MINUTE, 1, 10_000))
writes = int(_bounded_number(self.environ, "HUX_WRITES_PER_MINUTE", DEFAULT_WRITES_PER_MINUTE, 1, 10_000))
self.request_timeout = _bounded_number(
self.environ, "HUX_REQUEST_TIMEOUT_SECONDS", DEFAULT_REQUEST_TIMEOUT_SECONDS, 0.1, 60.0
)
self.rate_limiter = RateLimiter(reads, writes)
def add(self, method: str, template: str, card: str, action: str, handler: Handler, max_body: int = MAX_BODY_BYTES) -> None:
"""Register a handler; ``action`` is the audit action name (family.verb), ``max_body`` its byte cap."""
self.routes.append(Route(method, template, card, action, handler, max_body))
def bind_capability_routes(self) -> None:
"""Make feature negotiation depend on the route templates this process really registered."""
routes: dict[str, set[str]] = {}
for route in self.routes:
routes.setdefault(route.card, set()).add(route.template)
self.flags.bind_routes(routes)
def match(self, method: str, path: str) -> tuple[Route | None, dict[str, str], bool]:
"""Return (route, params, path_known)."""
known = False
for route in self.routes:
found = route.pattern.match(path)
if found:
known = True
if route.method == method:
return route, found.groupdict(), True
return None, {}, known
def body_limit(self, method: str, raw_path: str) -> int:
"""Maximum declared body for a matching route, before a socket read allocates it."""
route, _, _ = self.match(method, urlsplit(raw_path).path)
return route.max_body if route is not None else MAX_BODY_BYTES
def dispatch(self, method: str, raw_path: str, headers: Mapping[str, str], body: bytes) -> Response:
"""Run the full pipeline and never raise."""
parts = urlsplit(raw_path)
route: Route | None = None
try:
identity = resolve(headers, self.environ)
store = TenantStore(self.data_root, identity)
except HuxError as error:
return Response(error.status, error.record())
try:
retry_after = self.rate_limiter.check(identity.subject, method)
if retry_after is not None:
raise RateLimited(retry_after)
route, params, known = self.match(method, parts.path)
if route is None:
error = Invalid("method not allowed") if known else NotFound("no such route")
audit.record(store, identity, "http.route", parts.path, "not_found", error.message)
return Response(405 if known else 404, error.record())
# SO-08: the worker allowlist is checked before the flag so a
# worker cannot even learn which cards are on.
if identity.trust == "worker" and not worker_may_call(method, route.template):
raise Forbidden("route is not available to worker trust")
if identity.trust == "evidence":
allowed = {
("GET", "/hux/v1/capabilities"),
("GET", "/hux/v1/projects/{project_id}/conversations/{id}/releases"),
("GET", "/hux/v1/projects/{project_id}/conversations/{id}/releases/{release_id}"),
("POST", "/hux/v1/projects/{project_id}/conversations/{id}/releases/{release_id}/transitions"),
}
if (method, route.template) not in allowed:
raise Forbidden("route is not available to evidence trust")
self.flags.require(route.card)
try:
parsed_query = parse_qs(parts.query, max_num_fields=MAX_QUERY_FIELDS)
except ValueError as error:
raise Invalid("query has too many fields") from error
query = {key: values[-1] for key, values in parsed_query.items()}
payload = self._decode(body, route.max_body)
request = Request(method, parts.path, params, query, headers, payload, identity, store, self.flags, self.build)
response = route.handler(request)
except HuxError as error:
outcome = {"flag_off": "flag_off", "conflict": "conflict", "not_found": "not_found"}.get(error.code, "deny")
action = route.action if route is not None else "http.route"
audit.record(store, identity, action, parts.path, outcome, error.message)
headers = {"Retry-After": str(error.retry_after)} if isinstance(error, RateLimited) else {}
return Response(error.status, error.record(), headers)
except Exception: # noqa: BLE001 - the pipeline never raises; anything else is a 500 with no detail leaked
error = HuxError("internal error")
action = route.action if route is not None else "http.route"
audit.record(store, identity, action, parts.path, "deny", error.message)
return Response(error.status, error.record())
return response
@staticmethod
def _decode(body: bytes, max_body: int = MAX_BODY_BYTES) -> Any:
if not body:
return None
if len(body) > max_body:
raise TooLarge(f"body exceeds {max_body} bytes")
try:
return json.loads(body)
except json.JSONDecodeError as error:
raise Invalid(f"body is not JSON: {error.msg}") from error
def page(items: list[Any], next_cursor: Any = None) -> Response:
"""Standard list envelope."""
return Response(200, {"items": items, "next": next_cursor})
def make_handler(router: Router) -> type[BaseHTTPRequestHandler]:
"""Bind a Router to a BaseHTTPRequestHandler subclass."""
class HuxHandler(BaseHTTPRequestHandler):
server_version = "hux-foundation/1.0"
def log_message(self, fmt: str, *args: Any) -> None:
return
def _run(self) -> None:
try:
if self.headers.get("Transfer-Encoding"):
raise Invalid("Transfer-Encoding is not supported")
lengths = self.headers.get_all("Content-Length", [])
if len(lengths) > 1:
raise Invalid("duplicate Content-Length is not supported")
raw_length = lengths[0] if lengths else "0"
if not raw_length.isascii() or not raw_length.isdigit():
raise Invalid("Content-Length must be a non-negative integer")
if len(raw_length) > 10:
raise TooLarge("declared body length is too large")
length = int(raw_length)
limit = 0 if self.path == "/healthz" else router.body_limit(self.command, self.path)
if length > limit:
raise TooLarge(f"body exceeds {limit} bytes")
body = self.rfile.read(length) if length else b""
if len(body) != length:
raise Invalid("request body is incomplete or timed out")
if self.path == "/healthz":
body = {"status": "ok", "contract_version": CONTRACT_VERSION}
scheduler = getattr(router, "retention_scheduler", None)
if scheduler is not None:
body["retention"] = scheduler.health()
self._send(Response(200, body))
return
self._send(router.dispatch(self.command, self.path, dict(self.headers.items()), body))
except HuxError as error:
self._send(Response(error.status, error.record()))
except (OSError, TimeoutError):
error = Invalid("request body is incomplete or timed out")
self._send(Response(error.status, error.record()))
def _send(self, response: Response) -> None:
if response.stream is not None:
self.send_response(response.status)
self.send_header("Content-Type", "text/event-stream")
self.send_header("Cache-Control", "no-store")
self.send_header("X-Content-Type-Options", "nosniff")
for key, value in response.headers.items():
self.send_header(key, value)
self.end_headers()
for chunk in response.stream():
self.wfile.write(chunk)
self.wfile.flush()
return
data = json.dumps(response.body, sort_keys=True).encode()
self.send_response(response.status)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(data)))
self.send_header("Cache-Control", "no-store")
self.send_header("X-Content-Type-Options", "nosniff")
for key, value in response.headers.items():
self.send_header(key, value)
self.end_headers()
self.wfile.write(data)
do_GET = do_POST = do_PUT = do_PATCH = do_DELETE = _run
return HuxHandler
class BoundedHTTPServer(ThreadingHTTPServer):
"""Threading server that bounds header and body socket reads from accept onward."""
def __init__(self, address: tuple[str, int], handler: type[BaseHTTPRequestHandler], timeout: float, scheduler: Any = None) -> None:
self.request_timeout = timeout
self.scheduler = scheduler
super().__init__(address, handler)
def get_request(self) -> tuple[Any, Any]:
"""Accept one connection and apply its per-request socket deadline."""
request, address = super().get_request()
request.settimeout(self.request_timeout)
return request, address
def server_close(self) -> None:
"""Stop background work before closing the listening socket."""
if self.scheduler is not None:
self.scheduler.stop()
super().server_close()
def serve(router: Router, host: str = "127.0.0.1", port: int = 8790) -> ThreadingHTTPServer:
"""Create (but do not start) the server; callers call serve_forever()."""
from hux.retention_scheduler import RetentionScheduler
scheduler = RetentionScheduler(router)
router.retention_scheduler = scheduler
server = BoundedHTTPServer((host, port), make_handler(router), router.request_timeout, scheduler)
server.daemon_threads = True
scheduler.start()
return server

View File

@ -0,0 +1,196 @@
"""Caller identity resolved from the trusted hop's headers.
The chat router, the Telegram relay and the Worker are the only callers. Each
asserts identity in headers; the request body is never trusted for identity.
Router, relay, and worker callers must each present their distinct shared key,
compared in constant time against an inline or projected secret value.
"""
from __future__ import annotations
import hmac
import os
import re
import stat
import tempfile
from collections.abc import Mapping
from dataclasses import dataclass
from pathlib import Path
from hux.errors import Unauthorized
HEADER_SLOT = "X-Hermes-Tenant-Identity"
HEADER_SUBJECT = "X-Hux-Subject"
HEADER_SURFACE = "X-Hux-Surface"
HEADER_TRUST = "X-Hux-Trust"
HEADER_KEY = "X-Hux-Relay-Key"
SLOT_RE = re.compile(r"^slot-[0-9]{1,3}$")
SUBJECT_RE = re.compile(r"^usr_[0-9a-f]{16,64}$")
SURFACES = ("chat", "worker", "telegram", "voice", "api")
TRUSTS = ("router", "relay", "worker", "evidence")
KEY_ENV = {
"router": "HUX_ROUTER_KEY", "relay": "HUX_RELAY_KEY",
"worker": "HUX_WORKER_KEY", "evidence": "HUX_RELEASE_EVIDENCE_KEY",
}
MAX_KEY_BYTES = 4096
MAX_SUBJECT_BYTES = 128
SUBJECT_BINDING_ENV = "HUX_SUBJECT_BINDING_FILE"
@dataclass(frozen=True)
class Identity:
"""Who is calling, on which surface, asserted by which trusted hop."""
tenant_slot: str
subject: str
surface: str
trust: str
def record(self) -> dict[str, str]:
"""Serialise as ``common.identity``."""
return {"tenant_slot": self.tenant_slot, "subject": self.subject, "surface": self.surface, "trust": self.trust}
def _header(headers: Mapping[str, str], name: str) -> str:
for key, value in headers.items():
if key.lower() == name.lower():
return value.strip()
return ""
def _expected_key(environ: Mapping[str, str], trust: str) -> str:
"""Read a caller key, preferring a bounded Kubernetes secret file."""
if trust == "evidence":
from hux.release_security import evidence_key
try:
return evidence_key(environ)
except Exception: # noqa: BLE001 - authentication fails closed without revealing configuration detail
return ""
name = KEY_ENV[trust]
key_file = environ.get(f"{name}_FILE", "")
if key_file:
try:
path = Path(key_file)
mode = path.stat().st_mode
if not stat.S_ISREG(mode) or stat.S_IMODE(mode) != 0o400:
return ""
data = path.read_bytes()
except OSError:
return ""
if len(data) > MAX_KEY_BYTES:
return ""
try:
return data.decode("utf-8", errors="strict").strip()
except UnicodeDecodeError:
return ""
return environ.get(name, "")
def _read_subject_binding(path: Path) -> str | None:
"""Read a complete, permission-constrained binding without following links."""
flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0)
try:
descriptor = os.open(path, flags)
except FileNotFoundError:
return None
except OSError as error:
raise Unauthorized("subject binding is unavailable") from error
try:
mode = os.fstat(descriptor).st_mode
if not stat.S_ISREG(mode) or stat.S_IMODE(mode) not in {0o400, 0o440}:
raise Unauthorized("subject binding is invalid")
with os.fdopen(descriptor, "rb", closefd=False) as binding:
data = binding.read(MAX_SUBJECT_BYTES + 1)
except OSError as error:
raise Unauthorized("subject binding is unavailable") from error
finally:
os.close(descriptor)
if not data or len(data) > MAX_SUBJECT_BYTES:
raise Unauthorized("subject binding is invalid")
try:
value = data.decode("utf-8", errors="strict").strip()
except UnicodeDecodeError as error:
raise Unauthorized("subject binding is invalid") from error
if not SUBJECT_RE.fullmatch(value):
raise Unauthorized("subject binding is invalid")
return value
def _publish_subject_binding(path: Path, subject: str) -> None:
"""Publish an immutable first-writer binding; concurrent writers never see partial data."""
descriptor = -1
temporary = ""
try:
descriptor, temporary = tempfile.mkstemp(prefix=f".{path.name}.", dir=path.parent)
os.fchmod(descriptor, 0o440)
with os.fdopen(descriptor, "wb", closefd=False) as binding:
binding.write((subject + "\n").encode("utf-8"))
binding.flush()
os.fsync(descriptor)
os.close(descriptor)
descriptor = -1
try:
os.link(temporary, path, follow_symlinks=False)
except FileExistsError:
pass
except OSError as error:
raise Unauthorized("subject binding is unavailable") from error
finally:
if descriptor >= 0:
os.close(descriptor)
if temporary:
try:
os.unlink(temporary)
except FileNotFoundError:
pass
except OSError:
# The binding decision is still verified by a fresh read below.
pass
def _enforce_subject_binding(environ: Mapping[str, str], trust: str, subject: str) -> None:
"""Bind from an authenticated edge hop, then require every caller to match."""
raw_path = environ.get(SUBJECT_BINDING_ENV, "")
if not raw_path:
return
path = Path(raw_path)
bound = _read_subject_binding(path)
if bound is None:
if trust == "worker":
raise Unauthorized("worker subject is not bound")
_publish_subject_binding(path, subject)
bound = _read_subject_binding(path)
if bound is None or not hmac.compare_digest(bound, subject):
raise Unauthorized("subject does not match trusted binding")
def resolve(headers: Mapping[str, str], environ: Mapping[str, str] | None = None) -> Identity:
"""Build an Identity from request headers or raise Unauthorized.
``trust`` defaults to ``router``; every trusted hop requires its distinct
shared key to be configured and presented.
"""
environ = os.environ if environ is None else environ
slot = _header(headers, HEADER_SLOT)
subject = _header(headers, HEADER_SUBJECT)
surface = _header(headers, HEADER_SURFACE) or "chat"
trust = _header(headers, HEADER_TRUST) or "router"
if not SLOT_RE.match(slot):
raise Unauthorized("missing or malformed tenant slot")
own_slot = environ.get("HUX_TENANT_SLOT", "")
if own_slot and slot != own_slot:
raise Unauthorized("tenant slot does not belong to this pod")
if not SUBJECT_RE.match(subject):
raise Unauthorized("missing or malformed subject")
if surface not in SURFACES or trust not in TRUSTS:
raise Unauthorized("unknown surface or trust")
expected = _expected_key(environ, trust)
presented = _header(headers, HEADER_KEY)
if not expected or not presented or not hmac.compare_digest(expected, presented):
raise Unauthorized(f"{trust} key missing or wrong")
if trust == "router" and surface == "worker":
raise Unauthorized("worker surface needs worker trust")
if trust == "evidence" and surface != "api":
raise Unauthorized("evidence trust needs api surface")
_enforce_subject_binding(environ, trust, subject)
return Identity(slot, subject, surface, trust)

View File

@ -0,0 +1,459 @@
"""HUX-02 memory control: an append-only ledger with explicit user consent.
Every entry is a ``hux.memory.v1`` snapshot. The newest snapshot per id is
kept as a revisioned document (so If-Match works) and every snapshot is also
appended to ``ledger.jsonl``. Status moves only along
``rules.MEMORY_TRANSITIONS``; ``no_store`` and ``forgotten`` write a
content-free line plus a tombstone before anything is returned (SO-22), and
``retrieve`` consults the tombstones before the documents (SO-23).
"""
from __future__ import annotations
from datetime import datetime, timedelta, timezone
from typing import Any
from hux import contracts, redaction, rules
from hux.errors import Conflict, Forbidden, Invalid, NotFound
from hux.http import Request, Response, Router, page
from hux.identity import Identity
from hux.store import TenantStore, new_id, now_iso
FAMILY = "memory"
IDEM_FAMILY = "memory_idem"
LEDGER = "ledger"
TOMBSTONES = "tombstones"
DEFAULT_CONVERSATION = "conv_memory"
HUMAN_SURFACES = frozenset({"chat", "telegram", "voice"})
ACTIONS = ("approve", "reject", "forget", "edit", "remove_retrieval", "restore_retrieval")
KINDS = ("preference", "fact", "instruction", "context")
DEFAULT_DECAY_DAYS = 180
SCHEMAS = contracts.load_all()
def _parse(stamp: str) -> datetime:
return datetime.fromisoformat(stamp.replace("Z", "+00:00"))
def _emit(store: TenantStore, identity: Identity, record: dict[str, Any], kind: str, summary: str) -> None:
from hux import events
detail = {"memory_id": record["id"], "kind": record["kind"], "sensitivity": record["sensitivity"], "topic": record.get("topic", "general")}
conversation_id = record["provenance"].get("conversation_id", DEFAULT_CONVERSATION)
events.emit(store, identity, conversation_id, kind, summary, detail, [{"kind": "memory", "id": record["id"]}], sensitivity=record["sensitivity"])
def tombstoned(store: TenantStore) -> set[str]:
"""Ids that must never surface again, whatever the documents say."""
return {row["memory_id"] for row in store.read(FAMILY, TOMBSTONES)}
def _tombstone(store: TenantStore, memory_id: str, reason: str) -> None:
store.append(FAMILY, TOMBSTONES, {"memory_id": memory_id, "at": now_iso(), "reason": reason, "purged": False})
def effective_expiry(record: dict[str, Any]) -> datetime | None:
"""When the entry stops being active: ``expires_at``, or decay from the last approval (or creation)."""
ttl = record["ttl"]
if ttl["policy"] == "expires_at":
return _parse(ttl["expires_at"])
if ttl["policy"] == "decay":
approvals = [row["at"] for row in record["audit"] if row["action"] == "approved"]
return _parse(approvals[-1] if approvals else record["created_at"]) + timedelta(days=ttl["decay_days"])
return None
def _persist(store: TenantStore, record: dict[str, Any], expected: int | None) -> dict[str, Any]:
"""Validate against contract and rules, then write document + ledger line under the family lock."""
problems = rules.memory_policy_violations(record) + contracts.validate_record({**record, "revision": record.get("revision") or 1}, SCHEMAS)
if problems:
raise Invalid("memory entry violates policy", problems)
with store.lock(FAMILY):
stored = store.put(FAMILY, record, expected)
store.append(FAMILY, LEDGER, stored)
return stored
def _transition(store: TenantStore, identity: Identity, record: dict[str, Any], target: str, action: str, actor: dict[str, str],
note: str = "", expected: int | None = None, **changes: Any) -> dict[str, Any]:
"""Move an entry to ``target`` from its *current* stored state (F7).
The caller's ``record`` may be stale: the entry is re-read under the family
lock, ``expected`` (If-Match) is checked against that copy, and the write is
always conditional on the loaded revision so a stale unconditional write can
never resurrect a forgotten entry (SO-22, SO-44).
"""
with store.lock(FAMILY):
record = _current(store, record, expected)
if not rules.transition_allowed(rules.MEMORY_TRANSITIONS, record["status"], target):
raise Conflict(f"{record['status']} -> {target} is not a legal memory transition")
stamp = now_iso()
updated = {**record, **changes, "status": target, "updated_at": stamp, "audit": [*record["audit"], {"at": stamp, "action": action, "actor": actor, **({"note": note[:200]} if note else {})}]}
if target in {"forgotten", "rejected", "expired"}:
updated["content"], updated["retrievable"] = "", False
if target == "active":
updated["retrievable"] = changes.get("retrievable", True)
stored = _persist(store, updated, record["revision"])
if target == "forgotten":
_tombstone(store, stored["id"], action)
return stored
def _current(store: TenantStore, record: dict[str, Any], expected: int | None) -> dict[str, Any]:
"""Fresh copy of ``record`` from the store; Conflict when If-Match no longer matches it."""
current = store.get(FAMILY, record["id"])
if expected is not None and expected != current["revision"]:
raise Conflict(f"revision {expected} does not match current revision {current['revision']}", [str(current["revision"])])
return current
def load(store: TenantStore, memory_id: str, now: datetime | None = None) -> dict[str, Any]:
"""Newest snapshot with lazy expiry (SO-27): an entry past its expiry is written as expired before it is served."""
record = store.get(FAMILY, memory_id)
if record["status"] == "active":
expiry = effective_expiry(record)
if expiry is not None and expiry <= (now or datetime.now(timezone.utc)):
record = _transition(store, None, record, "expired", "expired", {"type": "system", "id": "retention"}, expected=record["revision"])
return record
def expire_due(store: TenantStore, now: datetime) -> int:
"""Retention pass: move every active entry past its expiry to expired. Returns the count."""
before = {row["id"]: row["status"] for row in store.scan(FAMILY)}
return sum(1 for memory_id, status in before.items() if status == "active" and load(store, memory_id, now)["status"] == "expired")
def purge_forgotten(store: TenantStore) -> int:
"""Retention pass (SO-47): blank ``content`` on every ledger snapshot of tombstoned ids, then mark them purged."""
with store.lock(FAMILY):
stones = store.read(FAMILY, TOMBSTONES)
pending = {row["memory_id"] for row in stones if not row.get("purged")}
if not pending:
return 0
rows = store.read(FAMILY, LEDGER)
store.rewrite(FAMILY, LEDGER, [{**row, "content": ""} if row.get("id") in pending else row for row in rows])
store.rewrite(FAMILY, TOMBSTONES, [{**row, "purged": True} for row in stones])
return len(pending)
def forget_from_conversation(store: TenantStore, identity: Identity, conversation_id: str) -> list[str]:
"""Forget every entry sourced from ``conversation_id`` (used by the privacy forget); returns the ids."""
from hux import events
forgotten: list[str] = []
for row in list(store.scan(FAMILY)):
if row["provenance"].get("conversation_id") != conversation_id:
continue
if row["status"] in {"proposed"}:
row = _transition(store, identity, row, "rejected", "rejected", {"type": "system", "id": "forget"}, "conversation forgotten")
if row["status"] in {"active", "expired"}:
row = _transition(store, identity, row, "forgotten", "forgotten", {"type": "system", "id": "forget"}, "conversation forgotten")
if row["status"] not in {"forgotten", "no_store"}:
_tombstone(store, row["id"], "conversation forgotten")
forgotten.append(row["id"])
events.redact_memory_references(store, row["id"])
return forgotten
def retrieve(store: TenantStore, query_terms: list[str], scope: dict[str, Any] | None = None) -> list[dict[str, Any]]:
"""Agent read hook (SO-23): tombstones first, then only active + retrievable entries in scope; never no_store, forgotten or expired."""
from hux import privacy
stones = tombstoned(store)
blocked = privacy.blocked_conversations(store)
if scope and scope.get("level") == "conversation" and scope.get("scope_id") in blocked:
return []
terms = [term.lower() for term in query_terms if term]
hits: list[dict[str, Any]] = []
for row in store.scan(FAMILY):
if row["id"] in stones or row["provenance"].get("conversation_id") in blocked:
continue
record = load(store, row["id"])
if record["status"] != "active" or not record["retrievable"]:
continue
if scope and record["scope"]["level"] != "global" and record["scope"] != scope:
continue
text = record["content"].lower()
if terms and not any(term in text for term in terms):
continue
hits.append(record)
return sorted(hits, key=lambda r: r["updated_at"], reverse=True)
# -- proposals ------------------------------------------------------------------
def _human(identity: Identity) -> bool:
return identity.surface in HUMAN_SURFACES and identity.trust != "worker"
def _actor(request: Request) -> dict[str, str]:
if _human(request.identity) and request.body.get("proposed_by", "assistant") == "user":
return {"type": "user", "id": request.identity.subject}
return {"type": "assistant", "id": "hermes"}
def _classify(store: TenantStore, body: dict[str, Any], content: str, conversation_id: str | None) -> dict[str, Any]:
"""Scrub ``content`` and decide topic, sensitivity floor and the privacy gate; shared by create and edit (F3)."""
from hux import privacy
hits: list[str] = []
content = redaction.scrub_value(content.strip(), hits)[:2000]
detected = privacy.detect_topics(content) + (["credentials"] if hits else [])
topic = body.get("topic") if body.get("topic") in rules.PRIVACY_TOPICS or body.get("topic") == "general" else None
topic = topic if topic and topic != "general" else (detected[0] if detected else "general")
sensitivity = body.get("sensitivity", "personal")
if sensitivity not in privacy.SENSITIVITY_RANK:
raise Invalid("unknown sensitivity")
floor = privacy.topic_sensitivity(detected + ([topic] if topic != "general" else []))
if privacy.SENSITIVITY_RANK[floor] > privacy.SENSITIVITY_RANK[sensitivity]:
sensitivity = floor
allowed, why = privacy.memory_write_allowed(store, conversation_id, {"sensitivity": sensitivity, "topic": topic})
if why == "private_mode":
raise Forbidden("memory writes are refused in private mode")
return {"content": content, "topic": topic, "sensitivity": sensitivity, "allowed": allowed, "why": why}
def _decline(entry: dict[str, Any], why: str, actor: dict[str, str], stamp: str) -> dict[str, Any]:
"""Content-free ``no_store`` decision record.
The topic moves into the note because rules.memory_policy_violations reads a
deny topic on any non-rejected status as a write, and a sensitive entry must
say approval_mode=ask.
"""
topic = entry.pop("topic")
entry.update(status="no_store", approval_mode="ask" if entry["sensitivity"] == "sensitive" else "no_store", content="", retrievable=False,
audit=[{"at": stamp, "action": "no_store", "actor": actor, "note": f"{why}; topic={topic}"[:200]}])
return entry
def _shape(request: Request, body: dict[str, Any]) -> tuple[dict[str, Any], str]:
"""Build the entry from a proposal body and decide its approval mode from policy; returns (entry, gate_reason)."""
if "id" in body or "revision" in body or "status" in body:
raise Invalid("id, revision and status are server-assigned")
content = body.get("content")
if not isinstance(content, str) or not content.strip() or len(content) > 2000:
raise Invalid("content must be a non-empty string of at most 2000 chars")
conversation_id = body.get("conversation_id") if isinstance(body.get("conversation_id"), str) else None
verdict = _classify(request.store, body, content, conversation_id)
topic, sensitivity = verdict["topic"], verdict["sensitivity"]
scope = body.get("scope") if isinstance(body.get("scope"), dict) else {"level": "global"}
source = body.get("source") if isinstance(body.get("source"), dict) else {"kind": "message", "id": "unspecified"}
ttl = body.get("ttl") if isinstance(body.get("ttl"), dict) else None
if ttl is None or (sensitivity == "sensitive" and ttl.get("policy") == "never"):
days = rules.PRIVACY_TOPICS[topic]["decay_days"] if topic in rules.PRIVACY_TOPICS else DEFAULT_DECAY_DAYS
ttl = {"policy": "decay", "decay_days": days}
actor = _actor(request)
stamp = now_iso()
provenance = {"surface": request.identity.surface, "actor": actor, "recorded_at": stamp}
if conversation_id:
provenance["conversation_id"] = conversation_id
if isinstance(body.get("run_id"), str):
provenance["run_id"] = body["run_id"][:120]
entry: dict[str, Any] = {
"schema": "hux.memory.v1", "id": new_id("mem"), "owner": request.identity.subject, "scope": scope,
"kind": body.get("kind") if body.get("kind") in KINDS else "fact", "content": verdict["content"], "status": "proposed",
"approval_mode": "ask", "sensitivity": sensitivity, "topic": topic, "ttl": ttl, "source": source,
"provenance": provenance, "created_at": stamp, "updated_at": stamp,
"audit": [{"at": stamp, "action": "proposed", "actor": actor}], "reason": str(body.get("reason") or "Proposed to be remembered.")[:280],
"retrievable": False, "identity": request.identity.record(),
}
if isinstance(body.get("supersedes"), str):
entry["supersedes"] = body["supersedes"]
references_forgotten = entry.get("supersedes") in tombstoned(request.store) or source.get("id") in tombstoned(request.store)
wants = body.get("approval_mode", "ask" if actor["type"] == "assistant" else "automatic")
why = verdict["why"]
if not verdict["allowed"] or wants == "no_store":
why = why if not verdict["allowed"] else "declined"
_decline(entry, why, actor, stamp)
elif sensitivity == "sensitive" or wants == "ask" or references_forgotten or actor["type"] != "user":
entry["approval_mode"] = "ask"
else:
entry.update(status="active", approval_mode="automatic", retrievable=True)
entry["audit"].append({"at": stamp, "action": "approved", "actor": actor, "note": "automatic"})
return entry, why
def _replay(request: Request, key: str) -> Response | None:
for row in request.store.read(IDEM_FAMILY, "keys"):
if row["idempotency_key"] == key:
request.audit("memory.propose", row["memory_id"], "allow", "replayed")
return Response(200, load(request.store, row["memory_id"]), {"HUX-Replayed": "true"})
return None
def post_memory(request: Request) -> Response:
"""``POST /hux/v1/memory``: propose an entry; policy decides active, proposed (ask) or no_store (202).
The Idempotency-Key lookup, the write and the key mapping all happen under
the family lock so concurrent retries with one key yield one record (F13b).
"""
body = request.body if isinstance(request.body, dict) else None
if body is None:
raise Invalid("body must be an object")
key = request.idempotency_key()
with request.store.lock(FAMILY):
replayed = _replay(request, key) if key else None
if replayed is not None:
return replayed
entry, why = _shape(request, body)
try:
stored = _persist(request.store, entry, None)
except Invalid:
request.store.append(FAMILY, TOMBSTONES, {"memory_id": entry["id"], "at": now_iso(), "reason": "policy_violation", "purged": True})
_emit(request.store, request.identity, entry, "memory.suppressed", "Memory proposal suppressed by policy")
raise
if key:
request.store.append(IDEM_FAMILY, "keys", {"idempotency_key": key, "memory_id": stored["id"], "at": stored["created_at"]})
if stored["status"] == "no_store":
_tombstone(request.store, stored["id"], why)
_emit(request.store, request.identity, stored, "memory.suppressed", f"Not remembered ({why})")
request.audit("memory.propose", stored["id"], "allow", "no_store")
return Response(202, stored)
kind = "memory.committed" if stored["status"] == "active" else "memory.proposed"
_emit(request.store, request.identity, stored, kind, "Remembered" if stored["status"] == "active" else "Proposed to remember")
request.audit("memory.propose", stored["id"])
return Response(201, stored)
def list_memory(request: Request) -> Response:
"""``GET /hux/v1/memory?status=&scope=``: owner-scoped list; ``scope`` is ``level`` or ``level:scope_id``."""
status, scope = request.query.get("status"), request.query.get("scope")
level, _, scope_id = (scope or "").partition(":")
items = []
for row in list(request.store.scan(FAMILY)):
record = load(request.store, row["id"])
if status and record["status"] != status:
continue
if level and (record["scope"]["level"] != level or (scope_id and record["scope"].get("scope_id") != scope_id)):
continue
items.append(record)
request.audit("memory.list", "memory")
return page(sorted(items, key=lambda r: r["updated_at"], reverse=True), None)
def get_memory(request: Request) -> Response:
"""``GET /hux/v1/memory/{id}``."""
memory_id = request.params["id"]
try:
record = load(request.store, memory_id)
except (NotFound, Invalid) as error:
raise NotFound("memory entry not found") from error
request.audit("memory.get", memory_id)
return Response(200, record, {"ETag": str(record["revision"])})
def _edit(request: Request, record: dict[str, Any], actor: dict[str, str], expected: int | None) -> dict[str, Any]:
"""Supersede ``record`` with edited content under the same privacy shaping as a proposal (F3).
A deny verdict (deny topic, restricted, memory disabled, conversation
forgotten) yields a content-free ``no_store`` record and leaves the old
entry untouched; a sensitive floor makes the replacement ``proposed`` with
``approval_mode: ask``. Private mode is refused outright.
"""
body = request.body if isinstance(request.body, dict) else {}
content = body.get("content")
if not isinstance(content, str) or not content.strip() or len(content) > 2000:
raise Invalid("edit needs new content of at most 2000 chars")
conversation_id = record["provenance"].get("conversation_id")
verdict = _classify(request.store, {**body, "sensitivity": record["sensitivity"]}, content, conversation_id)
stamp = now_iso()
fresh = {
**{k: v for k, v in record.items() if k not in {"revision", "supersedes"}}, "id": new_id("mem"), "content": verdict["content"],
"topic": verdict["topic"], "sensitivity": verdict["sensitivity"], "status": "active", "approval_mode": "automatic", "retrievable": True,
"created_at": stamp, "updated_at": stamp, "audit": [{"at": stamp, "action": "edited", "actor": actor, "note": f"edit of {record['id']}"}],
}
if not verdict["allowed"]:
stored = _persist(request.store, _decline(fresh, verdict["why"], actor, stamp), None)
_tombstone(request.store, stored["id"], verdict["why"])
return stored
old = _transition(request.store, request.identity, record, "forgotten", "superseded", actor, "edited", expected)
from hux import events
events.redact_memory_references(request.store, old["id"])
fresh["supersedes"] = old["id"]
fresh["audit"][0]["note"] = f"supersedes {old['id']}"
if fresh["sensitivity"] == "sensitive":
fresh.update(status="proposed", approval_mode="ask", retrievable=False)
else:
fresh["audit"].append({"at": stamp, "action": "approved", "actor": actor})
return _persist(request.store, fresh, None)
def _set_retrievable(store: TenantStore, record: dict[str, Any], retrievable: bool, actor: dict[str, str], expected: int | None) -> dict[str, Any]:
"""Flip ``retrievable`` on an active entry, re-reading it under the lock so a stale copy cannot overwrite a forget (F7)."""
with store.lock(FAMILY):
record = _current(store, record, expected)
if record["status"] != "active":
raise Conflict("retrieval can only change on active entries")
stamp = now_iso()
entry = {"at": stamp, "action": "approved" if retrievable else "retrieval_removed", "actor": actor, "note": "retrieval restored" if retrievable else ""}
return _persist(store, {**record, "retrievable": retrievable, "updated_at": stamp, "audit": [*record["audit"], entry]}, record["revision"])
def act_memory(request: Request) -> Response:
"""``POST /hux/v1/memory/{id}/{action}``: approve, reject, forget, edit, remove_retrieval, restore_retrieval; every move emits a memory.* event."""
memory_id, action = request.params["id"], request.params["action"]
if action not in ACTIONS:
raise NotFound("unknown memory action")
if not _human(request.identity):
raise Forbidden("memory decisions come from a human surface")
try:
record = load(request.store, memory_id)
except (NotFound, Invalid) as error:
raise NotFound("memory entry not found") from error
expected = request.if_match()
if expected is not None and expected != record["revision"]:
raise Conflict(f"revision {expected} does not match current revision {record['revision']}", [str(record["revision"])])
actor = {"type": "user", "id": request.identity.subject}
store, identity = request.store, request.identity
if action == "approve":
stored = _transition(store, identity, record, "active", "approved", actor, expected=expected)
_emit(store, identity, stored, "memory.committed", "Memory approved")
elif action == "reject":
stored = _transition(store, identity, record, "rejected", "rejected", actor, expected=expected)
_emit(store, identity, stored, "memory.suppressed", "Memory rejected")
elif action == "forget":
stored = _transition(store, identity, record, "forgotten", "forgotten", actor, expected=expected)
from hux import events
events.redact_memory_references(store, stored["id"])
_emit(store, identity, stored, "memory.forgotten", "Memory forgotten")
elif action == "edit":
stored = _edit(request, record, actor, expected)
if stored["status"] == "no_store":
_emit(store, identity, stored, "memory.suppressed", f"Edit not remembered ({stored['audit'][0]['note'].split(';')[0]})")
request.audit("memory.edit", memory_id, "allow", "no_store")
return Response(202, stored)
_emit(store, identity, stored, "memory.committed" if stored["status"] == "active" else "memory.proposed", f"Memory edited, supersedes {record['id']}")
else:
retrievable = action == "restore_retrieval"
stored = _set_retrievable(store, record, retrievable, actor, expected)
_emit(store, identity, stored, "memory.retrieval_removed" if not retrievable else "memory.committed", "Retrieval removed" if not retrievable else "Retrieval restored")
request.audit(f"memory.{action}", memory_id, "allow", "" if expected is not None else "unconditional_write")
return Response(200, stored, {"ETag": str(stored["revision"])})
def export_memory(request: Request) -> Response:
"""``GET /hux/v1/memory/export``: active, retrievable entries only, each stamped ``exported`` (SO-26)."""
stones = tombstoned(request.store)
items = []
for row in list(request.store.scan(FAMILY)):
record = load(request.store, row["id"])
if record["id"] in stones or record["status"] != "active" or not record["retrievable"]:
continue
stamp = now_iso()
items.append(_persist(request.store, {**record, "audit": [*record["audit"], {"at": stamp, "action": "exported", "actor": {"type": "user", "id": request.identity.subject}}]}, record["revision"]))
request.store.append(FAMILY, "exports", {"at": now_iso(), "ids": [item["id"] for item in items]})
request.audit("memory.export", "memory")
response = page(items, None)
response.headers["Content-Disposition"] = 'attachment; filename="hux-memory-export.json"'
return response
def register(router: Router) -> None:
"""Attach HUX-02 routes."""
router.add("GET", "/hux/v1/memory/export", "HUX-02", "memory.export", export_memory)
router.add("POST", "/hux/v1/memory", "HUX-02", "memory.propose", post_memory)
router.add("GET", "/hux/v1/memory", "HUX-02", "memory.list", list_memory)
router.add("GET", "/hux/v1/memory/{id}", "HUX-02", "memory.get", get_memory)
router.add("POST", "/hux/v1/memory/{id}/{action}", "HUX-02", "memory.act", act_memory)

View File

@ -0,0 +1,169 @@
"""HUX-06 provider-neutral friendly modes and scoped route selection.
Automatic modes express intent to Switchyard and never name a provider. An
exact route can only be selected through the explicit advanced path and must
exist in the operator-provided route catalog. Private mode is local-only.
"""
from __future__ import annotations
import hashlib
import json
import os
import re
from typing import Any
from hux import contracts, rules
from hux.errors import Conflict, Invalid, NotFound
from hux.http import Request, Response, Router, page
from hux.store import now_iso
CARD = "HUX-06"
FAMILY = "mode_selections"
MODE_NAMES = ("fast", "thoughtful", "research", "create", "private")
MANUAL_ROUTE = re.compile(r"^atlas/manual/(codex|claude|local)/[a-z0-9][a-z0-9/-]{0,100}$")
MAX_CATALOG_ROUTES = 256
SCHEMAS = contracts.load_all()
def _body(request: Request) -> dict[str, Any]:
if not isinstance(request.body, dict):
raise Invalid("body must be a JSON object")
extra = set(request.body) - {"project_id", "mode", "advanced", "override_route_id"}
if extra:
raise Invalid("unexpected mode fields", sorted(extra))
return request.body
def _scope(request: Request, project_id: Any) -> tuple[dict[str, Any], dict[str, Any]]:
"""Resolve a conversation and its mandatory owning project."""
from hux import organization
try:
conversation = request.store.get(organization.CONVERSATIONS, request.params["id"])
except (Invalid, NotFound) as error:
raise NotFound("conversation not found") from error
path_project = request.params.get("project_id")
actual = conversation.get("project_id")
if not isinstance(project_id, str) or project_id != path_project or not actual or project_id != actual:
raise NotFound("project or conversation not found")
try:
project = request.store.get(organization.PROJECTS, project_id)
except (Invalid, NotFound) as error:
raise NotFound("project or conversation not found") from error
return project, conversation
def _record_id(conversation_id: str) -> str:
digest = hashlib.sha256(conversation_id.encode()).hexdigest()[:24]
return f"mode_{digest}"
def _catalog() -> set[str]:
raw = os.environ.get("HUX_SWITCHYARD_ROUTE_CATALOG", "")
items = [item.strip() for item in raw.split(",") if item.strip()]
if len(items) > MAX_CATALOG_ROUTES:
return set()
return {item for item in items if MANUAL_ROUTE.fullmatch(item)}
def _contract(mode: str, advanced: bool, override: Any) -> dict[str, Any]:
if not isinstance(mode, str) or mode not in MODE_NAMES:
raise Invalid("unknown friendly mode")
if override is not None:
if not advanced or not isinstance(override, str) or not MANUAL_ROUTE.fullmatch(override):
raise Invalid("an exact route requires a valid advanced manual route")
if override not in _catalog():
raise Invalid("advanced route is not in the Switchyard catalog")
if mode == "private" and not override.startswith("atlas/manual/local/"):
raise Invalid("private mode is local-only")
elif advanced:
raise Invalid("advanced selection requires override_route_id")
try:
result = rules.mode_contract(mode, override)
except ValueError as error:
raise Invalid(str(error)) from error
problems = contracts.validate("mode.schema.json", result, SCHEMAS)
if problems:
raise Invalid("mode failed contract validation", problems)
if mode != "private" and result["switchyard"]["route_id"].startswith("atlas/manual/"):
raise Invalid("automatic modes may not pin a provider")
return result
def _fingerprint(body: dict[str, Any]) -> str:
allowed = {key: body.get(key) for key in ("project_id", "mode", "advanced", "override_route_id")}
return hashlib.sha256(json.dumps(allowed, sort_keys=True, separators=(",", ":")).encode()).hexdigest()
def _replay(request: Request, key: str, fingerprint: str) -> dict[str, Any] | None:
for row in request.store.read(FAMILY, "idempotency"):
if row.get("key") != key:
continue
if row.get("fingerprint") != fingerprint:
raise Conflict("Idempotency-Key was already used for a different selection")
return request.store.get(FAMILY, row["id"])
return None
def list_modes(request: Request) -> Response:
"""Return all provider-neutral mode contracts."""
items = [_contract(name, False, None) for name in MODE_NAMES]
request.audit("modes.list", "modes")
return page(items)
def get_selection(request: Request) -> Response:
"""Return the selected mode for one bound conversation."""
project_id = request.params["project_id"]
_, conversation = _scope(request, project_id)
record = request.store.get(FAMILY, _record_id(conversation["id"]))
request.audit("modes.read", record["id"])
return Response(200, record, {"ETag": str(record["revision"]), "Cache-Control": "no-store"})
def put_selection(request: Request) -> Response:
"""Select a mode using If-Match and Idempotency-Key."""
body = _body(request)
_, conversation = _scope(request, body.get("project_id"))
key = request.idempotency_key()
if not key:
raise Invalid("Idempotency-Key is required")
expected = request.if_match()
if expected is None:
raise Invalid("If-Match is required; use 0 for the first selection")
fingerprint = _fingerprint(body)
mode = _contract(body.get("mode"), body.get("advanced") is True, body.get("override_route_id"))
record_id = _record_id(conversation["id"])
with request.store.lock(FAMILY):
replayed = _replay(request, key, fingerprint)
if replayed is not None:
request.audit("modes.select", replayed["id"], reason="idempotent_replay")
return Response(200, replayed, {"ETag": str(replayed["revision"]), "HUX-Replayed": "true", "Cache-Control": "no-store"})
exists = request.store.exists(FAMILY, record_id)
if expected != (request.store.get(FAMILY, record_id)["revision"] if exists else 0):
raise Conflict("mode selection revision does not match If-Match")
stamp = now_iso()
record = {
"id": record_id,
"schema": "hux.mode_selection.v1",
"owner": request.identity.subject,
"project_id": body["project_id"],
"conversation_id": conversation["id"],
"mode": mode,
"updated_at": stamp,
}
stored = request.store.put(FAMILY, record, expected_revision=expected)
request.store.append(FAMILY, "idempotency", {"key": key, "fingerprint": fingerprint, "id": record_id, "at": stamp})
with request.store.lock("conversations"):
current = request.store.get("conversations", conversation["id"])
request.store.put("conversations", {**current, "mode": mode["mode"], "updated_at": stamp}, current["revision"])
request.audit("modes.select", record_id)
return Response(200, stored, {"ETag": str(stored["revision"]), "Cache-Control": "no-store"})
def register(router: Router) -> None:
"""Attach HUX-06 routes."""
router.add("GET", "/hux/v1/modes", CARD, "modes.list", list_modes)
router.add("GET", "/hux/v1/projects/{project_id}/conversations/{id}/mode", CARD, "modes.read", get_selection)
router.add("PUT", "/hux/v1/projects/{project_id}/conversations/{id}/mode", CARD, "modes.select", put_selection)

View File

@ -0,0 +1,276 @@
"""HUX-07 scoped multimodal metadata, lineage and transcript corrections.
This service never accepts media bytes and has no capture endpoint. It only
records metadata after an approved autonomy action, plus inert camera/screen
intents that a human-facing client may send through the HUX-05 approval lane.
"""
from __future__ import annotations
import hashlib
import json
from typing import Any
from hux import contracts, redaction
from hux.errors import Conflict, Forbidden, Invalid, NotFound, TooLarge
from hux.http import Request, Response, Router, page
from hux.store import check_id, new_id, now_iso
CARD = "HUX-07"
ITEMS = "multimodal_items"
CORRECTIONS = "transcript_corrections"
INTENTS = "capture_intents"
MAX_ITEMS = 2000
MAX_CORRECTIONS = 10000
MAX_INTENTS = 1000
EXECUTABLE_MIMES = frozenset({"text/html", "application/xhtml+xml", "image/svg+xml", "application/xml", "text/xml"})
SAFE_SUFFIXES = (".jpg", ".jpeg", ".png", ".webp", ".wav", ".webm", ".ogg", ".mp4", ".pdf", ".txt")
KIND_MIMES = {
"image": frozenset({"image/jpeg", "image/png", "image/webp"}),
"audio": frozenset({"audio/wav", "audio/webm", "audio/ogg"}),
"video": frozenset({"video/webm", "video/mp4"}),
"document": frozenset({"application/pdf", "text/plain"}),
}
ITEM_FIELDS = frozenset({"project_id", "kind", "source", "filename", "mime", "bytes", "hash", "approval_id", "lineage"})
SCHEMAS = contracts.load_all()
SCHEMAS["multimodal.schema.json"] = contracts.load_schema("multimodal.schema.json")
def _body(request: Request, allowed: frozenset[str]) -> dict[str, Any]:
if not isinstance(request.body, dict):
raise Invalid("body must be a JSON object")
extra = set(request.body) - allowed
if extra:
raise Invalid("unexpected multimodal fields", sorted(extra))
return request.body
def _scope(request: Request, project_id: Any) -> dict[str, Any]:
from hux import organization
try:
conversation = request.store.get(organization.CONVERSATIONS, request.params["id"])
except (Invalid, NotFound) as error:
raise NotFound("project or conversation not found") from error
path_project = request.params.get("project_id")
actual = conversation.get("project_id")
if not isinstance(project_id, str) or project_id != path_project or not actual or project_id != actual:
raise NotFound("project or conversation not found")
if not organization.project_exists(request.store, project_id):
raise NotFound("project or conversation not found")
return conversation
def _if_match(request: Request, current: int) -> None:
expected = request.if_match()
if expected is None:
raise Invalid("If-Match is required")
if expected != current:
raise Conflict("revision does not match If-Match")
def _key(request: Request) -> str:
key = request.idempotency_key()
if not key:
raise Invalid("Idempotency-Key is required")
return key
def _digest(body: dict[str, Any]) -> str:
return hashlib.sha256(json.dumps(body, sort_keys=True, separators=(",", ":")).encode()).hexdigest()
def _replay(request: Request, family: str, key: str, digest: str) -> dict[str, Any] | None:
for row in request.store.read(family, "idempotency"):
if row.get("key") != key:
continue
if row.get("digest") != digest:
raise Conflict("Idempotency-Key was already used with different metadata")
return request.store.get(family, row["id"])
return None
def _remember(request: Request, family: str, key: str, digest: str, record_id: str) -> None:
request.store.append(family, "idempotency", {"key": key, "digest": digest, "id": record_id, "at": now_iso()})
def _validate(record: dict[str, Any], pointer: str) -> None:
problems = contracts.validate("multimodal.schema.json", record, SCHEMAS, pointer)
if problems:
raise Invalid("multimodal record failed contract validation", problems)
def _approval(request: Request, approval_id: Any, source: Any) -> str:
from hux import policy
try:
approval = policy.load_approval(request.store, check_id(approval_id))
except (Invalid, NotFound) as error:
raise Forbidden("approved autonomy action is required") from error
capability = "external_side_effect" if source in {"camera", "screen"} else "artifact_write"
if approval.get("status") != "approved" or approval.get("conversation_id") != request.params["id"]:
raise Forbidden("approved autonomy action is required")
if approval.get("capability") != capability:
raise Forbidden("approval does not cover this multimodal action")
return approval["id"]
def _lineage(request: Request, body: Any, project_id: str) -> dict[str, Any] | None:
if body is None:
return None
if not isinstance(body, dict) or set(body) - {"parent_item_id", "artifact_id", "artifact_version"}:
raise Invalid("lineage must contain only a parent item or artifact version")
if "parent_item_id" in body:
if len(body) != 1:
raise Invalid("parent-item lineage cannot also name an artifact")
parent = request.store.get(ITEMS, check_id(body["parent_item_id"]))
if parent["conversation_id"] != request.params["id"] or parent["project_id"] != project_id:
raise NotFound("lineage item not found")
return {"parent_item_id": parent["id"]}
if set(body) != {"artifact_id", "artifact_version"} or not isinstance(body["artifact_version"], int):
raise Invalid("artifact lineage needs artifact_id and artifact_version")
from hux import artifacts
artifact = request.store.get(artifacts.FAMILY, check_id(body["artifact_id"]))
if artifact.get("project_id") != project_id or artifact.get("conversation_id") != request.params["id"]:
raise NotFound("lineage artifact not found")
version = next((row for row in artifact["versions"] if row["version"] == body["artifact_version"]), None)
if version is None:
raise NotFound("lineage artifact version not found")
if version["content_ref"]["mime"].lower() in EXECUTABLE_MIMES or artifact.get("type") in {"html", "svg"}:
raise Invalid("executable HTML and SVG lineage is not accepted")
return {"artifact_id": artifact["id"], "artifact_version": version["version"]}
def create_item(request: Request) -> Response:
"""Register metadata for approved media; bytes are never accepted here."""
body = _body(request, ITEM_FIELDS)
conversation = _scope(request, body.get("project_id"))
key, digest = _key(request), _digest(body)
with request.store.lock(ITEMS):
replayed = _replay(request, ITEMS, key, digest)
if replayed is not None:
return Response(200, replayed, {"ETag": str(replayed["revision"]), "HUX-Replayed": "true", "Cache-Control": "no-store"})
_if_match(request, 0)
if request.store.count(ITEMS) >= MAX_ITEMS:
raise TooLarge("multimodal item limit reached")
filename, mime = body.get("filename"), body.get("mime")
if not isinstance(filename, str) or not filename.lower().endswith(SAFE_SUFFIXES):
raise Invalid("filename is missing or executable")
if not isinstance(mime, str) or mime.lower() in EXECUTABLE_MIMES:
raise Invalid("executable HTML, XML and SVG are not accepted")
if mime.lower() not in KIND_MIMES.get(body.get("kind"), frozenset()):
raise Invalid("kind and MIME type do not agree")
approval_id = _approval(request, body.get("approval_id"), body.get("source"))
lineage = _lineage(request, body.get("lineage"), body["project_id"])
record = {
"schema": "hux.multimodal_item.v1", "id": new_id("mmi"), "owner": request.identity.subject,
"project_id": body["project_id"], "conversation_id": conversation["id"], "kind": body.get("kind"),
"source": body.get("source"), "filename": filename, "mime": mime.lower(), "bytes": body.get("bytes"),
"hash": body.get("hash"), "approval_id": approval_id, "status": "metadata_only", "created_at": now_iso(),
}
if lineage:
record["lineage"] = lineage
_validate({**record, "revision": 1}, "/$defs/item")
stored = request.store.put(ITEMS, record, expected_revision=0)
_remember(request, ITEMS, key, digest, stored["id"])
request.audit("multimodal.create", stored["id"])
return Response(201, stored, {"ETag": "1", "Cache-Control": "no-store"})
def list_items(request: Request) -> Response:
"""List metadata in one project/conversation scope."""
project_id = request.params["project_id"]
_scope(request, project_id)
items = [row for row in request.store.scan(ITEMS) if row["project_id"] == project_id and row["conversation_id"] == request.params["id"]]
request.audit("multimodal.list", request.params["id"])
response = page(items)
response.headers["Cache-Control"] = "no-store"
return response
def get_item(request: Request) -> Response:
"""Read one metadata record only within its path scope."""
project_id = request.params["project_id"]
_scope(request, project_id)
try:
item = request.store.get(ITEMS, check_id(request.params["item_id"]))
except (Invalid, NotFound) as error:
raise NotFound("multimodal item not found") from error
if item["project_id"] != project_id or item["conversation_id"] != request.params["id"]:
raise NotFound("multimodal item not found")
request.audit("multimodal.read", item["id"])
return Response(200, item, {"ETag": str(item["revision"]), "Cache-Control": "no-store"})
def correct_transcript(request: Request) -> Response:
"""Append an immutable transcript correction and point the media head to it."""
body = _body(request, frozenset({"project_id", "replacement_text"}))
_scope(request, body.get("project_id"))
key, digest = _key(request), _digest({**body, "item_id": request.params["item_id"]})
with request.store.lock(CORRECTIONS):
replayed = _replay(request, CORRECTIONS, key, digest)
if replayed is not None:
return Response(200, replayed, {"HUX-Replayed": "true", "Cache-Control": "no-store"})
item = request.store.get(ITEMS, check_id(request.params["item_id"]))
if item["project_id"] != body["project_id"] or item["conversation_id"] != request.params["id"]:
raise NotFound("multimodal item not found")
if item["kind"] not in {"audio", "video"}:
raise Invalid("only audio or video transcripts can be corrected")
_if_match(request, item["revision"])
text = body.get("replacement_text")
if not isinstance(text, str) or not text.strip() or len(text) > 10000:
raise Invalid("replacement_text must contain 1 to 10000 characters")
if request.store.count(CORRECTIONS) >= MAX_CORRECTIONS:
raise TooLarge("transcript correction limit reached")
record = {
"schema": "hux.transcript_correction.v1", "id": new_id("trc"), "owner": request.identity.subject,
"project_id": body["project_id"], "conversation_id": request.params["id"], "item_id": item["id"],
"replacement_text": text, "created_at": now_iso(),
}
_validate({**record, "revision": 1}, "/$defs/correction")
stored = request.store.put(CORRECTIONS, record, expected_revision=0)
with request.store.lock(ITEMS):
updated = request.store.put(ITEMS, {**item, "latest_correction_id": stored["id"]}, item["revision"])
_remember(request, CORRECTIONS, key, digest, stored["id"])
request.audit("multimodal.correct", stored["id"])
return Response(201, stored, {"ETag": str(updated["revision"]), "Cache-Control": "no-store"})
def create_capture_intent(request: Request) -> Response:
"""Record an inert camera/screen proposal; it never grants execution."""
body = _body(request, frozenset({"project_id", "source", "purpose"}))
_scope(request, body.get("project_id"))
key, digest = _key(request), _digest(body)
with request.store.lock(INTENTS):
replayed = _replay(request, INTENTS, key, digest)
if replayed is not None:
return Response(200, replayed, {"HUX-Replayed": "true", "Cache-Control": "no-store"})
_if_match(request, 0)
if request.store.count(INTENTS) >= MAX_INTENTS:
raise TooLarge("capture intent limit reached")
purpose = body.get("purpose")
if body.get("source") not in {"camera", "screen"} or not isinstance(purpose, str) or not purpose.strip():
raise Invalid("source and purpose are required")
purpose = redaction.scrub_text(purpose[:280])[0]
record = {
"schema": "hux.capture_intent.v1", "id": new_id("cap"), "owner": request.identity.subject,
"project_id": body["project_id"], "conversation_id": request.params["id"], "source": body["source"],
"purpose": purpose, "status": "proposed", "requires_approval": True, "execution_allowed": False,
"created_at": now_iso(),
}
_validate({**record, "revision": 1}, "/$defs/capture_intent")
stored = request.store.put(INTENTS, record, expected_revision=0)
_remember(request, INTENTS, key, digest, stored["id"])
request.audit("multimodal.intent", stored["id"])
return Response(201, stored, {"ETag": "1", "Cache-Control": "no-store"})
def register(router: Router) -> None:
"""Attach metadata-only HUX-07 routes."""
base = "/hux/v1/projects/{project_id}/conversations/{id}"
router.add("POST", base + "/multimodal/items", CARD, "multimodal.create", create_item)
router.add("GET", base + "/multimodal/items", CARD, "multimodal.list", list_items)
router.add("GET", base + "/multimodal/items/{item_id}", CARD, "multimodal.read", get_item)
router.add("POST", base + "/multimodal/items/{item_id}/transcript-corrections", CARD, "multimodal.correct", correct_transcript)
router.add("POST", base + "/capture-intents", CARD, "multimodal.intent", create_capture_intent)

View File

@ -0,0 +1,384 @@
"""HUX-03 organisation: projects, conversations, branch lineage and search.
Projects own conversations; conversations carry tags, pins, a mode and an
optional branch pointer to the conversation they forked from. Search covers
only the ``project.schema.json#/$defs/search_index`` fields this increment
indexes (title, tags, project_name, and artifact titles when the artifacts
lane is present); message text is not indexed here and the response says so.
"""
from __future__ import annotations
import re
from typing import Any
from hux import contracts, redaction
from hux.errors import Conflict, Invalid, NotFound
from hux.http import Request, Response, Router, page
from hux.store import TenantStore, check_id, new_id, now_iso
CARD = "HUX-03"
PROJECTS = "projects"
CONVERSATIONS = "conversations"
MAX_PROJECTS = 200
MAX_CONVERSATIONS = 2000
PROJECT_FIELDS = ("name", "description", "tags", "pinned", "archived", "default_mode")
CONVERSATION_FIELDS = ("title", "tags", "pinned", "archived", "mode", "project_id")
INDEXED = ("title", "tags", "project_name", "artifact_titles")
NOT_INDEXED = ("message_text",)
MESSAGE_KINDS = ("message.user", "message.assistant")
MESSAGE_SCAN_CONVERSATIONS = 100
MESSAGE_SCAN_EVENTS = 300
SEARCH_PAGE = 50
SCHEMAS = contracts.load_all()
TOKEN_RE = re.compile(r"[a-z0-9]+")
def project_exists(store: TenantStore, project_id: str) -> bool:
"""True when this tenant owns a project with that id (helper for other lanes)."""
return isinstance(project_id, str) and bool(re.match(r"^[a-z]{2,6}_[A-Za-z0-9._-]{4,80}$", project_id)) and store.exists(PROJECTS, project_id)
def conversation_exists(store: TenantStore, conversation_id: str) -> bool:
"""True when this tenant owns a conversation with that id (helper for other lanes)."""
return isinstance(conversation_id, str) and bool(re.match(r"^[a-z]{2,6}_[A-Za-z0-9._-]{4,80}$", conversation_id)) and store.exists(CONVERSATIONS, conversation_id)
def project_of(store: TenantStore, conversation_id: str | None) -> str | None:
"""The project a conversation belongs to, or None when unknown or unfiled."""
if not conversation_id or not conversation_exists(store, conversation_id):
return None
return store.get(CONVERSATIONS, conversation_id).get("project_id")
def checked(record: dict[str, Any]) -> dict[str, Any]:
"""Raise Invalid unless ``record`` satisfies its contract."""
problems = contracts.validate_record(record, SCHEMAS)
if problems:
raise Invalid("record fails contract", problems)
return record
def body_dict(request: Request) -> dict[str, Any]:
"""The JSON object body or Invalid."""
if not isinstance(request.body, dict):
raise Invalid("body must be a JSON object")
return request.body
def pick(body: dict[str, Any], fields: tuple[str, ...]) -> dict[str, Any]:
"""Only the client-settable fields, secret-scrubbed (F9); ids, owner, timestamps and revision are server-set."""
return redaction.scrub_value({k: body[k] for k in fields if k in body}, [])
def replay(store: TenantStore, family: str, key: str) -> dict[str, Any] | None:
"""The record an Idempotency-Key already created in this family, if any."""
for row in store.read(family, "idempotency"):
if row["key"] == key:
return store.get(family, row["id"])
return None
def create(request: Request, family: str, record: dict[str, Any], key: str, cap: int) -> tuple[dict[str, Any], int]:
"""Create under the family lock honouring Idempotency-Key and the family count cap."""
with request.store.lock(family):
if key:
existing = replay(request.store, family, key)
if existing is not None:
return existing, 200
if request.store.count(family) >= cap:
raise Conflict(f"{family} cap of {cap} reached")
stored = request.store.put(family, checked({**record, "revision": 1}))
if key:
request.store.append(family, "idempotency", {"key": key, "id": stored["id"]})
return stored, 201
def update(request: Request, family: str, fields: tuple[str, ...]) -> dict[str, Any]:
"""PATCH under If-Match; a missing If-Match is accepted but audited as unconditional (SO-44)."""
changes = pick(body_dict(request), fields)
expected = request.if_match()
with request.store.lock(family):
current = request.store.get(family, check_id(request.params["id"]))
if "project_id" in changes and changes["project_id"] is not None and not project_exists(request.store, changes["project_id"]):
raise NotFound("project not found")
record = {**current, **{k: v for k, v in changes.items() if v is not None}, "updated_at": now_iso()}
if changes.get("project_id", "") is None:
record.pop("project_id", None)
stored = request.store.put(family, checked(record), expected_revision=expected)
request.audit(f"{family}.update", stored["id"], reason="" if expected is not None else "unconditional_write")
return stored
def etag(record: dict[str, Any]) -> dict[str, str]:
"""Revision as ETag so clients can send it back in If-Match."""
return {"ETag": str(record["revision"])}
# -- projects ------------------------------------------------------------------
def create_project(request: Request) -> Response:
"""``POST /hux/v1/projects``."""
body = pick(body_dict(request), PROJECT_FIELDS)
stamp = now_iso()
record = {
"schema": "hux.project.v1", "id": new_id("prj"), "owner": request.identity.subject,
"tags": [], "pinned": False, "archived": False, **body, "created_at": stamp, "updated_at": stamp,
}
stored, status = create(request, PROJECTS, record, request.idempotency_key(), MAX_PROJECTS)
request.audit("projects.create", stored["id"], reason="replayed" if status == 200 else "")
return Response(status, stored, etag(stored))
def list_projects(request: Request) -> Response:
"""``GET /hux/v1/projects?archived=``: pinned first, then most recently updated."""
archived = request.query.get("archived")
items = [p for p in request.store.scan(PROJECTS) if archived is None or p["archived"] == (archived == "true")]
items.sort(key=lambda p: p["updated_at"], reverse=True)
items.sort(key=lambda p: not p["pinned"])
request.audit("projects.list", "projects")
return page(items)
def get_project(request: Request) -> Response:
"""``GET /hux/v1/projects/{id}``."""
record = request.store.get(PROJECTS, check_id(request.params["id"]))
request.audit("projects.read", record["id"])
return Response(200, record, etag(record))
def patch_project(request: Request) -> Response:
"""``PATCH /hux/v1/projects/{id}`` with If-Match."""
stored = update(request, PROJECTS, PROJECT_FIELDS)
return Response(200, stored, etag(stored))
# -- conversations -------------------------------------------------------------
def new_conversation(request: Request, body: dict[str, Any], extra: dict[str, Any] | None = None) -> Response:
"""Build, validate and store a conversation from client fields plus server-set extras."""
if body.get("project_id") is not None and not project_exists(request.store, body["project_id"]):
raise NotFound("project not found")
stamp = now_iso()
record = {
"schema": "hux.conversation.v1", "id": new_id("conv"), "owner": request.identity.subject,
"tags": [], "pinned": False, "archived": False, **{k: v for k, v in body.items() if v is not None},
**(extra or {}), "artifact_ids": [], "created_at": stamp, "updated_at": stamp,
}
stored, status = create(request, CONVERSATIONS, record, request.idempotency_key(), MAX_CONVERSATIONS)
request.audit("conversations.create", stored["id"], reason="replayed" if status == 200 else "")
return Response(status, stored, etag(stored))
def create_conversation(request: Request) -> Response:
"""``POST /hux/v1/conversations``."""
return new_conversation(request, pick(body_dict(request), CONVERSATION_FIELDS))
def list_conversations(request: Request) -> Response:
"""``GET /hux/v1/conversations?project_id=&tag=&pinned=&archived=``: newest activity first."""
q = request.query
flags = {k: q[k] == "true" for k in ("pinned", "archived") if k in q}
items = []
for index, record in enumerate(request.store.scan(CONVERSATIONS)):
if "project_id" in q and record.get("project_id") != q["project_id"]:
continue
if "tag" in q and q["tag"] not in record["tags"]:
continue
if any(record[k] != v for k, v in flags.items()):
continue
items.append((record.get("last_message_at", record["updated_at"]), index, record))
items.sort(key=lambda item: item[:2], reverse=True)
request.audit("conversations.list", "conversations")
return page([record for _, _, record in items])
def get_conversation(request: Request) -> Response:
"""``GET /hux/v1/conversations/{id}``."""
record = request.store.get(CONVERSATIONS, check_id(request.params["id"]))
request.audit("conversations.read", record["id"])
return Response(200, record, etag(record))
def patch_conversation(request: Request) -> Response:
"""``PATCH /hux/v1/conversations/{id}`` with If-Match."""
stored = update(request, CONVERSATIONS, CONVERSATION_FIELDS)
return Response(200, stored, etag(stored))
def branch_conversation(request: Request) -> Response:
"""``POST /hux/v1/conversations/{id}/branch``: fork at a message, keeping project, tags and mode."""
body = body_dict(request)
point = body.get("branch_point_message_id")
if not isinstance(point, str) or not 1 <= len(point) <= 120:
raise Invalid("branch_point_message_id required")
parent = request.store.get(CONVERSATIONS, check_id(request.params["id"]))
fields = {"title": body.get("title") or f"{parent['title']} (branch)"[:200], "tags": list(parent["tags"]),
"project_id": parent.get("project_id"), "mode": parent.get("mode")}
return new_conversation(request, fields, {"branch": {"parent_conversation_id": parent["id"], "branch_point_message_id": point}})
def lineage(request: Request) -> Response:
"""``GET /hux/v1/conversations/{id}/lineage``: ancestors root-first plus direct children."""
record = request.store.get(CONVERSATIONS, check_id(request.params["id"]))
ancestors: list[dict[str, Any]] = []
cursor, seen = record, {record["id"]}
while "branch" in cursor and len(ancestors) < 64:
parent_id = cursor["branch"]["parent_conversation_id"]
if parent_id in seen or not request.store.exists(CONVERSATIONS, parent_id):
break
cursor = request.store.get(CONVERSATIONS, parent_id)
seen.add(parent_id)
ancestors.insert(0, cursor)
children = [c for c in request.store.scan(CONVERSATIONS) if c.get("branch", {}).get("parent_conversation_id") == record["id"]]
request.audit("conversations.lineage", record["id"])
return Response(200, {"conversation": record, "ancestors": ancestors, "children": children})
# -- search --------------------------------------------------------------------
def artifact_index(store: TenantStore) -> dict[str, list[str]]:
"""Conversation id -> titles of the artifacts filed under it (F13c): both the conversation's ``artifact_ids`` and artifacts that name the conversation."""
try:
from hux import artifacts
except ModuleNotFoundError:
return {}
index: dict[str, list[str]] = {}
for artifact in store.scan(artifacts.FAMILY):
if isinstance(artifact.get("conversation_id"), str) and isinstance(artifact.get("title"), str):
index.setdefault(artifact["conversation_id"], []).append(artifact["title"])
for conversation in store.scan(CONVERSATIONS):
ids = [i for i in conversation.get("artifact_ids", []) if isinstance(i, str)]
titles = artifacts.artifact_titles(store, ids).values() if ids else []
for title in titles:
index.setdefault(conversation["id"], []).append(title)
return index
def artifact_titles(store: TenantStore, conversation: dict[str, Any], index: dict[str, list[str]] | None = None) -> list[str]:
"""Artifact titles for a conversation via the artifacts lane, or nothing when it is absent."""
index = artifact_index(store) if index is None else index
return list(dict.fromkeys(index.get(conversation["id"], [])))
def mark_forgotten(store: TenantStore, conversation_id: str) -> bool:
"""Blank a forgotten conversation's title and tags and archive it (F9); False when there is no document."""
with store.lock(CONVERSATIONS):
if not conversation_exists(store, conversation_id):
return False
current = store.get(CONVERSATIONS, conversation_id)
record = {**current, "title": "[forgotten]", "tags": [], "archived": True, "updated_at": now_iso()}
store.put(CONVERSATIONS, checked(record), expected_revision=current["revision"])
return True
def tokens(text: str) -> list[str]:
"""Lowercased alphanumeric terms."""
return TOKEN_RE.findall(text.lower())
def score(record: dict[str, Any], terms: list[str], project_name: str, titles: list[str]) -> int:
"""Title hit beats tag hit beats project/artifact hit; every term must match somewhere."""
fields = {"title": tokens(record["title"]), "tags": [t for tag in record["tags"] for t in tokens(tag)],
"project_name": tokens(project_name), "artifact_titles": [t for title in titles for t in tokens(title)]}
weight = {"title": 4, "tags": 3, "project_name": 2, "artifact_titles": 1}
total = 0
for term in terms:
hit = sum(weight[f] * words.count(term) for f, words in fields.items())
if not hit:
return 0
total += hit
return total
def message_text_score(store: TenantStore, conversation: dict[str, Any], terms: list[str]) -> int:
"""Bounded match over this conversation's stored message events.
Privacy wins: forgotten conversations were already excluded, private-mode
conversations are never scanned, and restricted or fully redacted events
stay invisible to search exactly as they are on the timeline.
"""
from hux import events
if events.is_private(store, conversation["id"]):
return 0
rows = store.read(events.FAMILY, conversation["id"])[-MESSAGE_SCAN_EVENTS:]
total = 0
for row in rows:
if row.get("kind") not in MESSAGE_KINDS or row.get("sensitivity") == "restricted":
continue
if (row.get("redaction") or {}).get("level") == "full":
continue
words = tokens(str(row.get("summary", "")))
total += sum(words.count(term) for term in terms)
return total
def search(request: Request) -> Response:
"""``GET /hux/v1/search?q=&project_id=&include=&cursor=``: best match first.
The default scope is the indexed fields only. ``include=message_text``
additionally scans the stored message events of the most recently active
``MESSAGE_SCAN_CONVERSATIONS`` candidates (bounded, privacy-enforced) and
pages deterministically: rank order is (score desc, updated_at desc, id),
the cursor is the offset into that total order.
"""
terms = tokens(request.query.get("q", "")[:200])
if not terms:
raise Invalid("q is required")
include = request.query.get("include")
if include not in (None, "message_text"):
raise Invalid("include supports only message_text")
cursor = request.query.get("cursor", "0")
if not cursor.isdigit():
raise Invalid("cursor must be a non-negative integer")
project_filter = request.query.get("project_id")
names = {p["id"]: p["name"] for p in request.store.scan(PROJECTS)}
index = artifact_index(request.store)
candidates = [
record for record in request.store.scan(CONVERSATIONS)
if not project_filter or record.get("project_id") == project_filter
]
scanned = set()
if include:
recent = sorted(candidates, key=lambda r: (r["updated_at"], r["id"]), reverse=True)
scanned = {record["id"] for record in recent[:MESSAGE_SCAN_CONVERSATIONS]}
ranked = []
for record in candidates:
points = score(record, terms, names.get(record.get("project_id", ""), ""), artifact_titles(request.store, record, index))
if record["id"] in scanned and not record.get("archived"):
points += message_text_score(request.store, record, terms)
if points:
ranked.append((points, record))
if include:
# Paginated mode needs one total order; ids break updated_at ties.
ranked.sort(key=lambda pair: (-pair[0], pair[1]["updated_at"], pair[1]["id"]))
else:
ranked.sort(key=lambda pair: (-pair[0], pair[1]["updated_at"]))
request.audit("search.query", "conversations")
if not include:
return Response(200, {"items": [r for _, r in ranked], "next": None, "scores": {r["id"]: s for s, r in ranked},
"indexed": list(INDEXED), "not_indexed": list(NOT_INDEXED)})
start = int(cursor)
window = ranked[start : start + SEARCH_PAGE]
return Response(200, {
"items": [r for _, r in window], "scores": {r["id"]: s for s, r in window},
"next": str(start + SEARCH_PAGE) if len(ranked) > start + SEARCH_PAGE else None,
"indexed": list(INDEXED) + ["message_text"], "not_indexed": [],
"message_scan_limit": MESSAGE_SCAN_CONVERSATIONS,
})
def register(router: Router) -> None:
"""Attach HUX-03 routes."""
router.add("POST", "/hux/v1/projects", CARD, "projects.create", create_project)
router.add("GET", "/hux/v1/projects", CARD, "projects.list", list_projects)
router.add("GET", "/hux/v1/projects/{id}", CARD, "projects.read", get_project)
router.add("PATCH", "/hux/v1/projects/{id}", CARD, "projects.update", patch_project)
router.add("POST", "/hux/v1/conversations", CARD, "conversations.create", create_conversation)
router.add("GET", "/hux/v1/conversations", CARD, "conversations.list", list_conversations)
router.add("GET", "/hux/v1/conversations/{id}", CARD, "conversations.read", get_conversation)
router.add("PATCH", "/hux/v1/conversations/{id}", CARD, "conversations.update", patch_conversation)
router.add("POST", "/hux/v1/conversations/{id}/branch", CARD, "conversations.branch", branch_conversation)
router.add("GET", "/hux/v1/conversations/{id}/lineage", CARD, "conversations.lineage", lineage)
router.add("GET", "/hux/v1/search", CARD, "search.query", search)

View File

@ -0,0 +1,428 @@
"""HUX-05 autonomy: policy documents per scope and the approval queue.
A policy fixes the autonomy level, explicit grants and budgets for a scope
(global, project or conversation). ``rules.effective_decision`` is the only
resolver: an approval request resolves to allow, ask or deny against the most
specific policy, and every external side effect asks regardless. Budgets,
the pre-side-effect gate and cancellation receipts live in ``hux.budgets``;
this module registers their routes so the family stays one card.
"""
from __future__ import annotations
import hashlib
import json
from datetime import datetime, timedelta, timezone
from typing import Any
from hux import contracts, rules
from hux.errors import BudgetExhausted, Conflict, Forbidden, Invalid
from hux.http import Request, Response, Router, page
from hux.identity import Identity
from hux.store import TenantStore, check_id, new_id
CARD = "HUX-05"
FAMILY = "policy"
APPROVALS = "approvals"
SCOPES = ("global", "project", "conversation")
APPROVAL_TTL = timedelta(hours=24)
ALWAYS_TTL = timedelta(days=30)
SESSION_TTL = timedelta(hours=24)
HUMAN_SURFACES = frozenset({"chat", "telegram", "voice"})
HUMAN_TRUSTS = frozenset({"router", "relay"})
DEFAULT_BUDGETS = {
"tokens_per_run": 200000, "tool_calls_per_run": 40, "wall_clock_seconds": 900,
"delegations_per_run": 4, "spend_units": 50, "subagents_per_run": 2,
}
SCHEMAS = contracts.load_all()
def clock() -> datetime:
"""Current UTC time; tests replace this to move approvals past expiry."""
return datetime.now(timezone.utc)
def now() -> datetime:
"""Indirection so monkeypatching ``clock`` reaches every module."""
return clock()
def iso(when: datetime) -> str:
"""RFC 3339 second-precision UTC string."""
return when.strftime("%Y-%m-%dT%H:%M:%SZ")
def parse(stamp: str) -> datetime:
"""Inverse of ``iso``."""
return datetime.fromisoformat(stamp.replace("Z", "+00:00"))
def run_key(run_id: str) -> str:
"""Stable hex handle for a free-form run id so it can name a document."""
return hashlib.sha256(run_id.encode()).hexdigest()[:32]
def is_human(identity: Identity) -> bool:
"""True when a person is behind the call: router/relay trust on a chat, telegram or voice surface."""
return identity.trust in HUMAN_TRUSTS and identity.surface in HUMAN_SURFACES
def require_human(identity: Identity, what: str) -> None:
"""Forbidden unless a human actor is behind the call (F1: policy writes and allow grants)."""
if not is_human(identity):
raise Forbidden(f"{what} only from a human surface")
def require_worker(identity: Identity, what: str) -> None:
"""Forbidden unless the separately keyed Worker gateway is making the call."""
if identity.trust != "worker" or identity.surface != "worker":
raise Forbidden(f"{what} only from worker trust")
def actor_for(identity: Identity) -> dict[str, str]:
"""The actor a record attributes to this caller: humans are users, hops are system."""
if is_human(identity):
return {"type": "user", "id": identity.subject}
return {"type": "system", "id": identity.surface}
def provenance(identity: Identity) -> dict[str, Any]:
"""Server-set provenance; nothing here comes from the body."""
return {"surface": identity.surface, "actor": actor_for(identity), "recorded_at": iso(now())}
def public(record: dict[str, Any], revisioned: bool = False) -> dict[str, Any]:
"""Strip store-internal fields before a record leaves the service."""
return {k: v for k, v in record.items() if not k.startswith("_") and (revisioned or k != "revision")}
def emit(store: TenantStore, identity: Identity, conversation_id: str | None, kind: str, summary: str, **extra: Any) -> None:
"""Record an activity event through the events lane when it is present."""
if not conversation_id:
return
try:
from hux import events
except ModuleNotFoundError:
return
events.emit(store, identity, conversation_id, kind, summary, **extra)
def checked(record: dict[str, Any]) -> dict[str, Any]:
"""Raise Invalid unless ``record`` satisfies its contract; internal fields are ignored."""
if record["schema"] == "hux.policy.v1":
candidate = public({"revision": 1, **record}, revisioned=True)
else:
candidate = public(record)
problems = contracts.validate_record(candidate, SCHEMAS)
if problems:
raise Invalid("record fails contract", problems)
return record
def body_dict(request: Request) -> dict[str, Any]:
"""The JSON object body or Invalid."""
if not isinstance(request.body, dict):
raise Invalid("body must be a JSON object")
return request.body
# -- policy documents --------------------------------------------------------
def policy_id(level: str, scope_id: str | None) -> str:
"""Document id for a scope; global has exactly one."""
if level not in SCOPES:
raise Invalid("scope must be global, project or conversation")
if level == "global":
return "pol_global"
if scope_id is None or len(scope_id) > 60:
raise Invalid("scope_id required and at most 60 characters for project and conversation scopes")
return f"pol_{level}.{check_id(scope_id)}"
def default_policy(identity: Identity) -> dict[str, Any]:
"""The ``safe`` policy every tenant starts with."""
return {
"schema": "hux.policy.v1", "id": "pol_global", "owner": identity.subject, "scope": {"level": "global"},
"autonomy": "safe", "grants": [], "budgets": dict(DEFAULT_BUDGETS),
"provenance": provenance(identity), "updated_at": iso(now()), "_budget_epoch": new_id("bep"),
}
def global_policy(store: TenantStore, identity: Identity) -> dict[str, Any]:
"""Read the global policy, creating the default on first touch."""
with store.lock(FAMILY):
if store.exists(FAMILY, "pol_global"):
return store.get(FAMILY, "pol_global")
return store.put(FAMILY, checked({**default_policy(identity), "revision": 1}))
def effective_policy(store: TenantStore, identity: Identity, level: str, scope_id: str | None) -> dict[str, Any]:
"""Most specific policy along conversation -> project -> global."""
chain: list[tuple[str, str | None]] = [(level, scope_id)]
if level == "conversation":
from hux import organization # lazy: organization never imports policy, but keep import order free
parent = organization.project_of(store, scope_id)
if parent:
chain.append(("project", parent))
for lvl, sid in chain:
if lvl != "global" and store.exists(FAMILY, policy_id(lvl, sid)):
return store.get(FAMILY, policy_id(lvl, sid))
return global_policy(store, identity)
def normalise_grants(grants: Any, identity: Identity) -> list[dict[str, Any]]:
"""Validate grants and pin every one to a server-set expiry of at most 30 days (SO-38)."""
if not isinstance(grants, list):
raise Invalid("grants must be a list")
ceiling = now() + ALWAYS_TTL
out = []
for grant in grants:
if not isinstance(grant, dict) or grant.get("capability") not in rules.CAPABILITIES:
raise Invalid("grant needs a known capability")
if grant.get("decision") not in ("allow", "ask", "deny"):
raise Invalid("grant decision must be allow, ask or deny")
if grant["decision"] == "allow":
require_human(identity, "allow grants")
expires = grant.get("expires_at")
if not isinstance(expires, str) or parse(expires) > ceiling:
expires = iso(ceiling)
out.append({"capability": grant["capability"], "decision": grant["decision"], "expires_at": expires, "granted_by": actor_for(identity)})
return out
def get_policy(request: Request) -> Response:
"""``GET /hux/v1/policy?scope=&scope_id=``: the effective policy for a scope."""
level = request.query.get("scope", "global")
scope_id = request.query.get("scope_id")
policy_id(level, scope_id)
record = effective_policy(request.store, request.identity, level, scope_id)
request.audit("policy.read", record["id"])
return Response(200, public(record, revisioned=True), {"ETag": str(record["revision"])})
def put_policy(request: Request) -> Response:
"""``PUT /hux/v1/policy``: replace the policy for the scope named in the body; humans only (F1)."""
require_human(request.identity, "policy writes")
body = body_dict(request)
scope = body.get("scope") or {}
if not isinstance(scope, dict):
raise Invalid("scope must be an object")
record_id = policy_id(scope.get("level", ""), scope.get("scope_id"))
if body.get("autonomy") not in ("ask_first", "safe", "autonomous"):
raise Invalid("autonomy must be ask_first, safe or autonomous")
if body["autonomy"] == "autonomous":
require_human(request.identity, "autonomy escalation")
budgets = body.get("budgets", dict(DEFAULT_BUDGETS))
record = {
"schema": "hux.policy.v1", "id": record_id, "owner": request.identity.subject, "scope": scope,
"autonomy": body["autonomy"], "grants": normalise_grants(body.get("grants", []), request.identity),
"budgets": budgets, "provenance": provenance(request.identity), "updated_at": iso(now()),
"_budget_epoch": new_id("bep"),
}
expected = request.if_match()
with request.store.lock(FAMILY):
exists = request.store.exists(FAMILY, record_id)
stored = request.store.put(FAMILY, checked(record), expected_revision=expected)
reason = "unconditional_write" if exists and expected is None else ""
request.audit("policy.write", record_id, reason=reason)
return Response(200, public(stored, revisioned=True), {"ETag": str(stored["revision"])})
def add_grant(store: TenantStore, identity: Identity, level: str, scope_id: str | None, capability: str, ttl: timedelta) -> None:
"""Append an allow grant to a scope's policy after a session/always decision; the decider must be human."""
require_human(identity, "allow grants")
record_id = policy_id(level, scope_id)
with store.lock(FAMILY):
if store.exists(FAMILY, record_id):
record = store.get(FAMILY, record_id)
else:
scope = {"level": level, **({"scope_id": scope_id} if scope_id else {})}
record = {**global_policy(store, identity), "id": record_id, "scope": scope}
record.pop("revision")
expected = record.pop("revision", None)
budget_epoch = record.get("_budget_epoch") or f"{record['id']}:{expected or 1}"
grants = [g for g in record["grants"] if g["capability"] != capability]
grants.append({"capability": capability, "decision": "allow", "expires_at": iso(now() + min(ttl, ALWAYS_TTL)), "granted_by": actor_for(identity)})
record = {
**record, "grants": grants[-64:], "provenance": provenance(identity),
"updated_at": iso(now()), "_budget_epoch": budget_epoch,
}
store.put(FAMILY, checked(record), expected_revision=expected)
# -- approvals -----------------------------------------------------------------
def refresh(store: TenantStore, record: dict[str, Any]) -> dict[str, Any]:
"""Expire a pending approval whose window has closed (SO-42); terminal records never change."""
if record["status"] == "pending" and parse(record["expires_at"]) <= now():
with store.lock(APPROVALS):
record = store.put(APPROVALS, {**record, "status": "expired"})
return record
def load_approval(store: TenantStore, approval_id: str) -> dict[str, Any]:
"""Read one approval for this tenant; unknown ids are 404 whoever owns them."""
return refresh(store, store.get(APPROVALS, check_id(approval_id)))
def request_fingerprint(body: dict[str, Any]) -> str:
"""Canonical digest that binds an Idempotency-Key to exactly one approval request."""
canonical = json.dumps(body, sort_keys=True, separators=(",", ":"), ensure_ascii=False).encode()
return "sha256:" + hashlib.sha256(canonical).hexdigest()
def replay(store: TenantStore, key: str, fingerprint: str) -> dict[str, Any] | None:
"""The approval an Idempotency-Key already created, rejecting a changed request body."""
for row in store.read(APPROVALS, "idempotency"):
if row["key"] == key:
record = load_approval(store, row["id"])
if record.get("_request_fingerprint") != fingerprint:
raise Conflict("Idempotency-Key was already used for a different approval request")
return record
return None
def resolve_request(policy: dict[str, Any], capability: str, external: bool) -> str:
"""Effective decision for a request; external side effects never auto-allow (SO-39)."""
decision = rules.effective_decision(policy, capability, now())
if external and decision == "allow":
return "ask"
return decision
PRIVATE_DENIED = frozenset({"network", "web_search", "send_message", "shell", "delegate"})
def private_mode_denies(store: TenantStore, conversation_id: str | None, capability: str) -> bool:
"""Private conversations never release web, shell, messaging or delegation.
The mode catalog says so (``rules.MODE_CATALOG['private']['tools']``);
memory writes are already refused by the privacy state. The conversation
record's ``mode`` field is written only by the HUX-06 selection route.
"""
if capability not in PRIVATE_DENIED or not conversation_id:
return False
try:
record = store.get("conversations", conversation_id)
except Exception:
return False
return record.get("mode") == "private"
def create_approval(request: Request) -> Response:
"""``POST /hux/v1/approvals``: the agent hook asks before a gated action."""
require_worker(request.identity, "approval requests")
body = body_dict(request)
key = request.idempotency_key()
conversation_id = check_id(body.get("conversation_id"))
capability = body.get("capability")
if capability not in rules.CAPABILITIES:
raise Invalid("unknown capability")
from hux import budgets # lazy: budgets imports this module
run_id = budgets.checked_run_id(body.get("run_id"))
bound_conversation = budgets.run_conversation(request.store, run_id)
if bound_conversation != conversation_id:
raise Invalid("run is not authoritatively bound to this conversation")
req = body.get("request") if isinstance(body.get("request"), dict) else {}
external = bool(req.get("external", False))
evidence = req.get("evidence") if isinstance(req.get("evidence"), list) else []
if sum(1 for e in evidence if isinstance(e, dict) and e.get("kind") == "tool_call") != 1:
raise Invalid("an approval names exactly one tool_call; ask once per side effect (SO-37)")
fingerprint = request_fingerprint(body)
if key:
with request.store.lock(APPROVALS):
existing = replay(request.store, key, fingerprint)
if existing is not None:
request.audit("approvals.create", existing["id"], reason="replayed")
return Response(200, public(existing))
state = budgets.budget_state(request.store, request.identity, run_id, conversation_id)
if state["exhausted"]:
emit(request.store, request.identity, conversation_id, "budget.exhausted", f"Budget exhausted: {', '.join(state['exhausted'])}", run_id=state["run_id"])
raise BudgetExhausted("run budget exhausted", state["exhausted"])
policy = effective_policy(request.store, request.identity, "conversation", conversation_id)
decision = resolve_request(policy, capability, external)
if private_mode_denies(request.store, conversation_id, capability):
decision = "deny"
stamp = now()
record: dict[str, Any] = {
"schema": "hux.approval.v1", "id": new_id("apr"), "run_id": run_id, "conversation_id": conversation_id,
"capability": capability, "request": {**req, "external": external},
"status": {"allow": "approved", "deny": "denied", "ask": "pending"}[decision],
"requested_at": iso(stamp), "expires_at": iso(stamp + APPROVAL_TTL),
}
if decision != "ask":
record["decision"] = {"choice": "once" if decision == "allow" else "deny", "by": {"type": "system", "id": "policy"}, "at": iso(stamp)}
if key:
record["idempotency_key"] = key
record["_request_fingerprint"] = fingerprint
with request.store.lock(APPROVALS):
if key:
existing = replay(request.store, key, fingerprint)
if existing is not None:
request.audit("approvals.create", existing["id"], reason="replayed")
return Response(200, public(existing))
stored = request.store.put(APPROVALS, checked(record))
if key:
request.store.append(APPROVALS, "idempotency", {"key": key, "id": stored["id"]})
request.audit("approvals.create", stored["id"], reason=f"policy_{decision}")
kind = "approval.requested" if decision == "ask" else "approval.resolved"
emit(request.store, request.identity, conversation_id, kind, f"{capability}: {record['request'].get('summary', '')}"[:280],
run_id=record["run_id"], evidence=[{"kind": "approval", "id": stored["id"]}])
return Response(201, public(stored))
def list_approvals(request: Request) -> Response:
"""``GET /hux/v1/approvals?status=``: the queue, oldest request first."""
wanted = request.query.get("status")
if wanted and wanted not in rules.APPROVAL_TRANSITIONS:
raise Invalid("unknown status")
items = [refresh(request.store, r) for r in request.store.scan(APPROVALS)]
items = [public(r) for r in items if not wanted or r["status"] == wanted]
items.sort(key=lambda r: (r["requested_at"], r["id"]))
request.audit("approvals.list", wanted or "all")
return page(items)
def decide_approval(request: Request) -> Response:
"""``POST /hux/v1/approvals/{id}``: a human answers once, session, always or deny (SO-35)."""
identity = request.identity
if identity.trust not in HUMAN_TRUSTS or identity.surface not in HUMAN_SURFACES:
raise Forbidden("approvals are decided only from a human surface")
choice = body_dict(request).get("choice")
if choice not in ("once", "session", "always", "deny"):
raise Invalid("choice must be once, session, always or deny")
target = "denied" if choice == "deny" else "approved"
with request.store.lock(APPROVALS):
record = load_approval(request.store, request.params["id"])
if not rules.transition_allowed(rules.APPROVAL_TRANSITIONS, record["status"], target):
raise Conflict(f"approval is {record['status']}; only pending approvals can be decided")
record = {**record, "status": target, "decision": {"choice": choice, "by": actor_for(identity), "at": iso(now())}}
stored = request.store.put(APPROVALS, checked(record))
if choice == "session":
add_grant(request.store, identity, "conversation", record["conversation_id"], record["capability"], SESSION_TTL)
elif choice == "always":
add_grant(request.store, identity, "global", None, record["capability"], ALWAYS_TTL)
request.audit("approvals.decide", stored["id"], reason=choice)
emit(request.store, identity, record["conversation_id"], "approval.resolved", f"{record['capability']} {target} ({choice})",
run_id=record["run_id"], evidence=[{"kind": "approval", "id": stored["id"]}])
return Response(200, public(stored))
def get_approval(request: Request) -> Response:
"""``GET /hux/v1/approvals/{id}``: one record."""
record = load_approval(request.store, request.params["id"])
request.audit("approvals.read", record["id"])
return Response(200, public(record))
def register(router: Router) -> None:
"""Attach HUX-05 routes, including the budget, gate and stop routes from ``hux.budgets``."""
from hux import budgets
router.add("GET", "/hux/v1/policy", CARD, "policy.read", get_policy)
router.add("PUT", "/hux/v1/policy", CARD, "policy.write", put_policy)
router.add("POST", "/hux/v1/approvals", CARD, "approvals.create", create_approval)
router.add("GET", "/hux/v1/approvals", CARD, "approvals.list", list_approvals)
router.add("GET", "/hux/v1/approvals/{id}", CARD, "approvals.read", get_approval)
router.add("POST", "/hux/v1/approvals/{id}", CARD, "approvals.decide", decide_approval)
budgets.register_routes(router)

Some files were not shown because too many files have changed in this diff Show More