hermes: resume failed PR publications with trusted bounded retries

This commit is contained in:
jenkins 2026-09-13 19:11:35 -05:00
parent fe3e9ba16b
commit 0a51ffe4a7
19 changed files with 888 additions and 36 deletions

View File

@ -67,7 +67,7 @@ EXPECTED_SHA256 = {
# sha256 of services/hermes/scm-common/scripts/gitea_api.py: the ConfigMap # sha256 of services/hermes/scm-common/scripts/gitea_api.py: the ConfigMap
# hermes-scm-boundary-v2 mounts that exact file at /opt/scm/gitea_api.py, # hermes-scm-boundary-v2 mounts that exact file at /opt/scm/gitea_api.py,
# so this pin is derivable from merged source and equal to the deployed one. # so this pin is derivable from merged source and equal to the deployed one.
GITEA_CLIENT: {"76efd16dedbeb74425b12fbbdbfaa391854771292077e0463bf22706855ae6dc"}, GITEA_CLIENT: {"5c457f63370ebed8a76648d755b98d0ee9c5f06ec6f657fb96cd6b8544eb51cd"},
} }
DEADLINE_ERROR = "deadline-exceeded" DEADLINE_ERROR = "deadline-exceeded"

View File

@ -182,6 +182,7 @@ spec:
- {name: CLAUDE_CODE_OAUTH_TOKEN_FILE, value: /claude-oauth-access/token} - {name: CLAUDE_CODE_OAUTH_TOKEN_FILE, value: /claude-oauth-access/token}
- {name: HERMES_CLAUDE_BIN, value: /opt/coordinator/claude_oauth_exec} - {name: HERMES_CLAUDE_BIN, value: /opt/coordinator/claude_oauth_exec}
- {name: HERMES_CLAUDE_NATIVE_BIN, value: /worker-data/tools/bin/claude} - {name: HERMES_CLAUDE_NATIVE_BIN, value: /worker-data/tools/bin/claude}
# Codex OAuth is deliberately not copied from Hermes' shared refresh lineage.
- {name: HERMES_EXECUTION_DISABLED_PROVIDER, value: codex} - {name: HERMES_EXECUTION_DISABLED_PROVIDER, value: codex}
- {name: HERMES_AUTO_ROUTER_PROFILE, value: agent} - {name: HERMES_AUTO_ROUTER_PROFILE, value: agent}
- {name: PYTHONPATH, value: /opt/hermes} - {name: PYTHONPATH, value: /opt/hermes}

View File

@ -106,6 +106,8 @@ configMapGenerator:
- supervisor_state.py=scripts/supervisor_state.py - supervisor_state.py=scripts/supervisor_state.py
- publication_retry.py=scripts/publication_retry.py - publication_retry.py=scripts/publication_retry.py
- publication_retry_recovery.py=scripts/publication_retry_recovery.py - publication_retry_recovery.py=scripts/publication_retry_recovery.py
- publication_retry_scheduler.py=scripts/publication_retry_scheduler.py
- publication_retry_fence.py=scripts/publication_retry_fence.py
- scm_resume_bootstrap.py=scripts/scm_resume_bootstrap.py - scm_resume_bootstrap.py=scripts/scm_resume_bootstrap.py
- bootstrap_soteria_publication_retry.py=scripts/bootstrap_soteria_publication_retry.py - bootstrap_soteria_publication_retry.py=scripts/bootstrap_soteria_publication_retry.py
- seed_legacy_scm_roots.py=scripts/seed_legacy_scm_roots.py - seed_legacy_scm_roots.py=scripts/seed_legacy_scm_roots.py

View File

@ -41,6 +41,10 @@ PORT = int(os.environ.get("HERMES_EXECUTION_CLIENT_PORT", "9009"))
RESULT_FIELDS = frozenset( RESULT_FIELDS = frozenset(
{"status", "summary", "changed_files", "tests_run", "artifacts", "findings", "blockers"} {"status", "summary", "changed_files", "tests_run", "artifacts", "findings", "blockers"}
) )
MEDIATOR_TERMINAL_FIELDS = frozenset({
"scm_submission", "scm_resume", "publication_retry_transient",
"publication_retry_lease_fence", "lease_fence",
})
LOG = logging.getLogger(__name__) LOG = logging.getLogger(__name__)
COORDINATOR_REJECTION_CATEGORIES = { COORDINATOR_REJECTION_CATEGORIES = {
"assignment": ( "assignment": (
@ -122,7 +126,12 @@ def _validate_result(payload: Any) -> dict[str, Any]:
value = structured.get(name) value = structured.get(name)
if not isinstance(value, list) or any(not isinstance(item, str) for item in value): if not isinstance(value, list) or any(not isinstance(item, str) for item in value):
raise ProtocolError(f"terminal result {name} must be a text list") raise ProtocolError(f"terminal result {name} must be a text list")
return payload return dict(payload)
def _worker_terminal(payload: dict[str, Any]) -> dict[str, Any]:
"""Copy ordinary worker output while removing mediator-only authority fields."""
return {name: value for name, value in payload.items() if name not in MEDIATOR_TERMINAL_FIELDS}
class ClientBoundary: class ClientBoundary:
@ -312,10 +321,9 @@ class ClientBoundary:
return {"scm_submission": self.resumed} return {"scm_submission": self.resumed}
def finish(self, request: dict[str, Any]) -> dict[str, Any]: def finish(self, request: dict[str, Any]) -> dict[str, Any]:
payload = _validate_result(request.get("payload")) payload = _worker_terminal(_validate_result(request.get("payload")))
# The model-facing caller cannot classify a retry as transient. Only a # SCM evidence and retry control are private mediator outputs. A worker
# preceding mediator resume may attach this coordinator control signal. # may report ordinary terminal fields but cannot carry authority forward.
payload.pop("publication_retry_transient", None)
with self.lock: with self.lock:
assignment, binding = self._current_for(request.get("binding")) assignment, binding = self._current_for(request.get("binding"))
structured = payload["structured"] structured = payload["structured"]

View File

@ -343,7 +343,14 @@ class Coordinator:
_task_value(task, "current_run_id", None) _task_value(task, "current_run_id", None)
) if task is not None else None ) if task is not None else None
if current_run_id is None or current_run_id != run_id: if current_run_id is None or current_run_id != run_id:
self.store.finalize(binding, "stale") from publication_retry_fence import recover, terminal_receipt
receipt = terminal_receipt(record)
if (current_run_id is None and _task_value(task, "status") == "blocked"
and receipt is not None and recover(
binding["board"], binding["task_id"], str(run_id), connection, receipt)):
self.store.finalize(binding, "finalized")
else:
self.store.finalize(binding, "stale")
return return
_append_terminal_activity(kanban_db, record) _append_terminal_activity(kanban_db, record)
metadata = { metadata = {
@ -428,10 +435,15 @@ class Coordinator:
reason = reason or str(structured.get("summary") or "worker failed") reason = reason or str(structured.get("summary") or "worker failed")
if retry_exhausted: if retry_exhausted:
reason = "Publication retry budget exhausted; inspect the retained SCM evidence." reason = "Publication retry budget exhausted; inspect the retained SCM evidence."
changed = kanban_db.block_task( from publication_retry_fence import block as block_with_fence
connection, binding["task_id"], reason=reason, retry_receipt = retry if retry is not None else (
kind="transient" if payload.get("capacity_failure") and not retry_exhausted else "capability", assignment.get("scm_resume") if isinstance(assignment, dict)
expected_run_id=run_id, and payload.get("publication_retry_transient") is True and not retry_exhausted else None
)
changed = block_with_fence(
kanban_db, connection, binding["board"], binding["task_id"], str(run_id), reason,
"transient" if payload.get("capacity_failure") and not retry_exhausted else "capability",
retry_receipt,
) )
self.store.finalize(binding, "finalized" if changed else "stale") self.store.finalize(binding, "finalized" if changed else "stale")
finally: finally:

View File

@ -128,7 +128,8 @@ def recover_results(pool: Any) -> None:
def _lease_failure_state( def _lease_failure_state(
kanban_db: Any, connection: Any, binding: dict[str, Any], run_id: int, retry_exhausted: bool = False kanban_db: Any, connection: Any, binding: dict[str, Any], run_id: int, retry_exhausted: bool = False,
retry_receipt: Any = None,
) -> str: ) -> str:
"""Classify one exhausted lease from authoritative Kanban evidence only.""" """Classify one exhausted lease from authoritative Kanban evidence only."""
task = kanban_db.get_task(connection, binding["task_id"]) task = kanban_db.get_task(connection, binding["task_id"])
@ -137,22 +138,24 @@ def _lease_failure_state(
if task is not None if task is not None
else None else None
) )
status = str(_task_value(task, "status", "") or "")
if current != run_id: if current != run_id:
# Native block_task clears current_run_id. Only a receipt-bearing run
# may reconstruct that exact coordinator fence; every other moved run
# remains stale rather than treating a human block as pool-owned.
if current is None and status == "blocked" and retry_receipt is not None and not retry_exhausted:
return "finalized"
return "stale" return "stale"
if str(_task_value(task, "status", "") or "") != "running": if status != "running":
return "finalized" return "finalized"
if kanban_db.block_task( from publication_retry_fence import block as block_with_fence
connection, if block_with_fence(
binding["task_id"], kanban_db, connection, binding["board"], binding["task_id"], str(run_id),
reason=("Publication retry budget exhausted; inspect the retained SCM evidence." if retry_exhausted else "Publication retry budget exhausted; inspect the retained SCM evidence." if retry_exhausted else
( f"Distributed worker lease expired after {binding['attempt']} fenced attempts",
"Distributed worker lease expired after " "capability" if retry_exhausted else "transient", retry_receipt if not retry_exhausted else None,
f"{binding['attempt']} fenced attempts"
)),
kind="capability" if retry_exhausted else "transient",
expected_run_id=run_id,
): ):
return "finalized" return "fenced" if retry_receipt is not None and not retry_exhausted else "finalized"
# Kanban refused the exact-run park yet still reports the run as running, so # Kanban refused the exact-run park yet still reports the run as running, so
# there is no terminal evidence. Keep the row for the next pass to retry. # there is no terminal evidence. Keep the row for the next pass to retry.
raise ProtocolError("Kanban refused the exact-run lease park") raise ProtocolError("Kanban refused the exact-run lease park")
@ -175,14 +178,24 @@ def record_lease_failure(pool: Any, record: dict[str, Any]) -> None:
pool.store.finalize(binding, "stale") pool.store.finalize(binding, "stale")
return return
retry_exhausted = False retry_exhausted = False
retry_receipt = None
if _resume_ordinal(record.get("payload")) is not None: if _resume_ordinal(record.get("payload")) is not None:
retry_exhausted = not supervisor_state.reissue_publication_retry( retry_exhausted = not supervisor_state.reissue_publication_retry(
binding["board"], binding["task_id"], binding["run_id"] binding["board"], binding["task_id"], binding["run_id"]
) )
retry_receipt = record.get("payload", {}).get("scm_resume")
with pool._kanban_lock, kanban_db.scoped_current_board(binding["board"]): with pool._kanban_lock, kanban_db.scoped_current_board(binding["board"]):
connection = kanban_db.connect(board=binding["board"]) connection = kanban_db.connect(board=binding["board"])
try: try:
state = _lease_failure_state(kanban_db, connection, binding, run_id, retry_exhausted) state = _lease_failure_state(kanban_db, connection, binding, run_id, retry_exhausted, retry_receipt)
if retry_receipt is not None and not retry_exhausted and state in {"fenced", "finalized"}:
# A crash after native block but before the sidecar write reaches
# here with ``finalized``. Reconstruct only the exact marker.
from publication_retry_fence import recover
fence = recover(binding["board"], binding["task_id"], str(run_id), connection, retry_receipt)
if fence is None or not pool.store.record_publication_lease_failure(binding, retry_receipt, fence[1]):
raise ProtocolError("publication lease fence cannot be recovered")
state = "finalized"
finally: finally:
connection.close() connection.close()
pool.store.finalize(binding, state) pool.store.finalize(binding, state)
@ -320,6 +333,12 @@ def reconcile(pool: Any) -> None:
_defer("board-registry", error) _defer("board-registry", error)
return return
_settled("board-registry") _settled("board-registry")
board_names = [name for raw in boards if (name := cli_lane_dispatch._board_slug(raw))]
from publication_retry_scheduler import reopen_due
try:
reopen_due(pool, kanban_db, board_names)
except Exception as error: # noqa: BLE001 - dispatch still serves unrelated ready work
_defer("publication-retry", error)
for raw_board in boards: for raw_board in boards:
board = cli_lane_dispatch._board_slug(raw_board) board = cli_lane_dispatch._board_slug(raw_board)
if not board or not ordinals: if not board or not ordinals:

View File

@ -132,6 +132,59 @@ class PoolStore:
"SELECT * FROM assignments WHERE state IN ('assigned','running','result')" "SELECT * FROM assignments WHERE state IN ('assigned','running','result')"
) )
def terminal_record(self, board: str, task_id: str, run_id: str) -> dict[str, Any] | None:
"""Return one final pool record without exposing nonterminal work."""
with self._connect() as connection:
row = connection.execute(
"SELECT * FROM assignments WHERE board=? AND task_id=? AND run_id=? "
"AND state IN ('finalized','stale')", (board, task_id, run_id)
).fetchone()
return self._record(row)
def record_publication_lease_failure(self, binding: dict[str, Any], receipt: dict[str, Any], marker: str) -> bool:
"""Persist coordinator-owned OOM evidence without replacing a worker result."""
source = receipt.get("source") if isinstance(receipt, dict) else None
digest = receipt.get("result_digest") if isinstance(receipt, dict) else None
expected_marker = f"[hermes-publication-retry-fence:{binding['run_id']}:{digest}]"
if (
not isinstance(source, dict) or source.get("board") != binding["board"]
or source.get("task_id") != binding["task_id"]
or not isinstance(source.get("worker_ordinal"), int)
or source["worker_ordinal"] != binding["worker_ordinal"]
or not isinstance(digest, str) or len(digest) != 64 or marker != expected_marker
):
raise ProtocolError("publication lease evidence is invalid")
result = {
"structured": {"status": "blocked", "summary": "Publication retry worker lease expired."},
"capacity_failure": True,
"scm_submission": None,
"scm_resume": receipt,
"publication_retry_lease_fence": marker,
}
encoded = canonical_json(result).decode()
with self._lock, self._connect() as connection:
row = connection.execute(
"SELECT payload_json,result_json,state FROM assignments WHERE board=? AND task_id=? AND run_id=? "
"AND worker_ordinal=? AND attempt=?",
tuple(binding[name] for name in ("board", "task_id", "run_id", "worker_ordinal", "attempt")),
).fetchone()
if row is None or row["state"] not in {LEASE_FAILED, "finalized"}:
return False
try:
payload = json.loads(row["payload_json"])
except (TypeError, json.JSONDecodeError) as error:
raise ProtocolError("publication lease assignment is malformed") from error
if payload.get("scm_resume") != receipt:
return False
if row["result_json"]:
return row["result_json"] == encoded
changed = connection.execute(
"UPDATE assignments SET result_digest=?,result_json=?,updated_at=? WHERE board=? AND task_id=? "
"AND run_id=? AND worker_ordinal=? AND attempt=? AND result_json IS NULL",
(payload_digest(result), encoded, time.time(), *(binding[name] for name in ("board", "task_id", "run_id", "worker_ordinal", "attempt"))),
).rowcount
return bool(changed)
def known_runs(self) -> set[tuple[str, str, str]]: def known_runs(self) -> set[tuple[str, str, str]]:
"""Every run identity this store already owns a row for, in any state. """Every run identity this store already owns a row for, in any state.

View File

@ -173,7 +173,8 @@ not understand. You have no Kubernetes identity and no SCM credential. Do not at
read Secrets, mutate workloads, use exec/attach/port-forward, reach node roots, or bypass read Secrets, mutate workloads, use exec/attach/port-forward, reach node roots, or bypass
the reviewed SCM boundary. Commit intended changes locally on the assigned feature branch; the reviewed SCM boundary. Commit intended changes locally on the assigned feature branch;
the worker boundary handles the bounded push and draft pull request after validation. the worker boundary handles the bounded push and draft pull request after validation.
Switchyard owns provider/model/effort selection and cross-provider fallback. This worker uses only its provisioned native provider. Hermes' credential-owning
Switchyard lane serves Sol and Astra outside this isolated workspace.
Return a final JSON object matching the supplied schema. Use status=incomplete when work, Return a final JSON object matching the supplied schema. Use status=incomplete when work,
tests, commits, or verification remain. Use blocked only for a concrete task obstacle. tests, commits, or verification remain. Use blocked only for a concrete task obstacle.
Completed must have no blockers. List changed files, tests, artifacts, findings, and blockers. Completed must have no blockers. List changed files, tests, artifacts, findings, and blockers.

View File

@ -0,0 +1,174 @@
#!/usr/bin/env python3
"""Bind coordinator publication blocks to one immutable native event."""
from __future__ import annotations
import json
from typing import Any
import supervisor_state
PREFIX = "[hermes-publication-retry-fence:"
def marker(receipt: Any, run_id: str) -> str:
"""Return a bounded marker derived only from a validated retry receipt."""
digest = receipt.get("result_digest") if isinstance(receipt, dict) else None
if not isinstance(digest, str) or len(digest) != 64 or not run_id.isdecimal():
raise ValueError("publication retry fence is invalid")
return f"{PREFIX}{run_id}:{digest}]"
def _event(connection: Any, task_id: str, run_id: str, value: str) -> int:
"""Read the just-written native block event before recording its fence."""
row = connection.execute(
"SELECT id,run_id,kind,payload FROM task_events WHERE task_id=? ORDER BY id DESC LIMIT 1", (task_id,)
).fetchone()
try:
payload = json.loads(row[3] or "{}")
except (TypeError, ValueError, json.JSONDecodeError) as error:
raise ValueError("publication retry block event is malformed") from error
if (row is None or int(row[0]) < 1 or int(row[1] or 0) != int(run_id)
or row[2] != "blocked" or not isinstance(payload, dict)):
raise ValueError("publication retry block event is unavailable")
if value not in str(payload.get("reason") or ""):
raise ValueError("publication retry block event lacks its fence")
return int(row[0])
def record(board: str, task_id: str, run_id: str, connection: Any, value: str) -> None:
"""Persist a coordinator-owned event identity only after native blocking."""
event_id = _event(connection, task_id, run_id, value)
with supervisor_state._connect(board) as state:
state.execute(
"CREATE TABLE IF NOT EXISTS publication_retry_fences("
"board TEXT NOT NULL,child_task_id TEXT NOT NULL,run_id TEXT NOT NULL,"
"event_id INTEGER NOT NULL,marker TEXT NOT NULL,PRIMARY KEY(board,child_task_id,run_id))"
)
existing = state.execute(
"SELECT event_id,marker FROM publication_retry_fences WHERE board=? AND child_task_id=? AND run_id=?",
(board, task_id, run_id),
).fetchone()
if existing is None:
state.execute("INSERT INTO publication_retry_fences VALUES(?,?,?,?,?)", (board, task_id, run_id, event_id, value))
elif tuple(existing) != (event_id, value):
raise ValueError("publication retry fence conflicts with native history")
def read(board: str, task_id: str, run_id: str) -> tuple[int, str] | None:
"""Return one exact coordinator block fence, never a task-body claim."""
try:
with supervisor_state._connect(board) as state:
row = state.execute(
"SELECT event_id,marker FROM publication_retry_fences WHERE board=? AND child_task_id=? AND run_id=?",
(board, task_id, run_id),
).fetchone()
except Exception:
return None
return (int(row[0]), str(row[1])) if row and int(row[0]) > 0 and str(row[1]).startswith(PREFIX) else None
def terminal_receipt(record: Any) -> dict[str, Any] | None:
"""Return only a signed-pool publication failure that may reconstruct a fence."""
if not isinstance(record, dict):
return None
result, assignment = record.get("result"), record.get("payload")
if not isinstance(result, dict) or not isinstance(assignment, dict):
return None
structured = result.get("structured")
if (not isinstance(structured, dict) or structured.get("status") != "blocked"
or result.get("capacity_failure") is not True or result.get("scm_submission") is not None):
return None
receipt = result.get("scm_resume")
if receipt is None and result.get("publication_retry_transient") is True:
receipt = assignment.get("scm_resume")
source = receipt.get("source") if isinstance(receipt, dict) else None
if (not isinstance(source, dict) or source.get("board") != record.get("board")
or source.get("task_id") != record.get("task_id")
or source.get("worker_ordinal") != record.get("worker_ordinal")):
return None
return receipt
def recover(board: str, task_id: str, run_id: str, connection: Any, receipt: Any) -> tuple[int, str] | None:
"""Reconstruct a missing sidecar fence only from the current native event."""
try:
with supervisor_state._connect(board) as state:
sealed = state.execute(
"SELECT receipt_json FROM publication_retries WHERE board=? AND child_task_id=?",
(board, task_id),
).fetchone()
if sealed is None or json.loads(sealed[0]) != receipt:
return None
value = marker(receipt, run_id)
record(board, task_id, run_id, connection, value)
except (OSError, TypeError, ValueError, json.JSONDecodeError):
return None
return read(board, task_id, run_id)
def guarded_unblock(kanban_db: Any, connection: Any, task_id: str, root_task_id: str, run_id: str, fence: tuple[int, str]) -> bool:
"""Atomically reopen only the current coordinator-owned native block event."""
if (not isinstance(fence, tuple) or len(fence) != 2 or not isinstance(fence[0], int)
or fence[0] < 1 or not isinstance(fence[1], str) or not fence[1].startswith(PREFIX)
or not str(run_id).isdecimal() or not root_task_id):
return False
transaction = getattr(kanban_db, "write_txn", None)
append_event = getattr(kanban_db, "_append_event", None)
if not callable(transaction) or not callable(append_event):
return False
try:
with transaction(connection):
event = connection.execute(
"SELECT id,run_id,kind,payload FROM task_events WHERE task_id=? ORDER BY id DESC LIMIT 1",
(task_id,),
).fetchone()
try:
payload = json.loads(event[3] or "{}") if event is not None else {}
except (TypeError, ValueError, json.JSONDecodeError):
return False
if (event is None or int(event[0]) != fence[0] or int(event[1] or 0) != int(run_id)
or event[2] != "blocked" or not isinstance(payload, dict)
or fence[1] not in str(payload.get("reason") or "")):
return False
task = connection.execute(
"SELECT status,current_run_id FROM tasks WHERE id=?", (task_id,)
).fetchone()
parent = connection.execute(
"SELECT 1 FROM task_links WHERE child_id=? AND parent_id=?", (task_id, root_task_id)
).fetchone()
if task is None or task[0] != "blocked" or task[1] is not None or parent is None:
return False
unfinished = connection.execute(
"SELECT 1 FROM task_links l JOIN tasks p ON p.id=l.parent_id "
"WHERE l.child_id=? AND p.status != 'done' LIMIT 1", (task_id,)
).fetchone()
status = "todo" if unfinished is not None else "ready"
changed = connection.execute(
"UPDATE tasks SET status=?,current_run_id=NULL,consecutive_failures=0,last_failure_error=NULL "
"WHERE id=? AND status='blocked' AND current_run_id IS NULL", (status, task_id),
).rowcount
if changed != 1:
return False
append_event(connection, task_id, "unblocked", {"status": status, "publication_retry_fence": fence[1]})
return True
except (AttributeError, TypeError, ValueError):
return False
def block(kanban_db: Any, connection: Any, board: str, task_id: str, run_id: str, reason: str, kind: str, receipt: Any) -> bool:
"""Block normally, then retain a native event fence for an automatic retry."""
try:
value = marker(receipt, run_id) if receipt is not None else ""
except ValueError:
# A malformed retained receipt must still park its exact live run, but
# can never acquire automatic-retry ownership.
value = ""
changed = kanban_db.block_task(
connection, task_id, reason=f"{reason}\n{value}" if value else reason, kind=kind,
expected_run_id=int(run_id),
)
if changed and value:
record(board, task_id, run_id, connection, value)
return bool(changed)

View File

@ -0,0 +1,136 @@
#!/usr/bin/env python3
"""Reopen only due, coordinator-owned SCM publication retries."""
from __future__ import annotations
import json
import time
from typing import Any
import supervisor_state
from publication_retry import PublicationRetryError
from publication_retry_fence import guarded_unblock, read as read_fence
INITIAL_BACKOFF_SECONDS = 300
def _value(task: Any, name: str) -> Any:
return task.get(name) if isinstance(task, dict) else getattr(task, name, None)
def _due(board: str, task_id: str, now: int) -> tuple[dict[str, Any], str] | None:
"""Return a validated unissued receipt only after one durable backoff."""
with supervisor_state._connect(board) as connection:
row = connection.execute(
"SELECT source_run_id,source_ordinal,issued_run_id,resolved_run_id,retry_after,last_reissued_run_id "
"FROM publication_retries WHERE board=? AND child_task_id=?", (board, task_id)
).fetchone()
if row is None or row[2] or row[3]:
return None
if int(row[4]) == 0:
connection.execute(
"UPDATE publication_retries SET retry_after=? WHERE board=? AND child_task_id=? AND retry_after=0 "
"AND issued_run_id='' AND resolved_run_id=''", (now + INITIAL_BACKOFF_SECONDS, board, task_id)
)
return None
if int(row[4]) > now:
return None
expected_run = str(row[5] or row[0])
try:
receipt = supervisor_state.publication_retry(board, task_id, "")
except (OSError, ValueError, PublicationRetryError):
return None
if receipt is None or str(receipt.get("source", {}).get("run_id", "")) != str(row[0]):
return None
return receipt, expected_run
def _same_evidence(left: Any, right: Any, *, source_run: bool) -> bool:
"""Compare immutable receipt evidence; a corrected title may differ on reissue."""
if not isinstance(left, dict) or not isinstance(right, dict):
return False
keys = ("source", "baseline_sha", "head", "body", "structured")
if any(left.get(key) != right.get(key) for key in keys):
return False
return not source_run or (left.get("title") == right.get("title") and left.get("result_digest") == right.get("result_digest"))
def _pool_owned(pool: Any, board: str, task_id: str, run_id: str, receipt: dict[str, Any]) -> bool:
"""Require a terminal exact run and no competing live pool assignment."""
if any(record.get("board") == board and record.get("task_id") == task_id for record in pool.store.active_assignments()):
return False
record = pool.store.terminal_record(board, task_id, run_id)
if record is None or record.get("state") != "finalized" or int(record.get("worker_ordinal", -1)) != int(receipt["source"]["worker_ordinal"]):
return False
result, payload = record.get("result"), record.get("payload")
structured = result.get("structured") if isinstance(result, dict) else None
if (not isinstance(structured, dict) or structured.get("status") != "blocked"
or result.get("capacity_failure") is not True or result.get("scm_submission") is not None):
return False
source_run = run_id == str(receipt["source"]["run_id"])
evidence = result.get("scm_resume") if source_run and isinstance(result, dict) else payload.get("scm_resume") if isinstance(payload, dict) else None
return _same_evidence(evidence, receipt, source_run=source_run)
def _reopen(kanban_db: Any, board: str, task_id: str, expected_run: str, receipt: dict[str, Any]) -> bool:
"""Unblock an unchanged transient task only when its private parent chain holds."""
child = supervisor_state.get_child(board, task_id)
if child is None:
return False
try:
if supervisor_state.publication_retry(board, task_id, "") != receipt:
return False
except (OSError, ValueError, PublicationRetryError):
return False
fence = read_fence(board, task_id, expected_run)
if fence is None:
return False
with kanban_db.scoped_current_board(board):
connection = kanban_db.connect(board=board)
try:
task = kanban_db.get_task(connection, task_id)
parents = kanban_db.parent_ids(connection, task_id)
if (
task is None or _value(task, "status") != "blocked" or _value(task, "block_kind") != "transient"
or _value(task, "current_run_id") is not None
or child["root_task_id"] not in {str(value) for value in parents}
):
return False
return guarded_unblock(
kanban_db, connection, task_id, child["root_task_id"], expected_run, fence
)
finally:
connection.close()
def reopen_due(pool: Any, kanban_db: Any, boards: list[str], *, now: int | None = None) -> int:
"""Schedule due receipts only on their free source ordinal, once per pass."""
current = int(time.time()) if now is None else int(now)
reopened = 0
available = set(pool.store.available_ordinals())
for board in boards:
with kanban_db.scoped_current_board(board):
connection = kanban_db.connect(board=board)
try:
tasks = list(kanban_db.list_tasks(connection))
finally:
connection.close()
for task in tasks:
task_id = str(_value(task, "id") or "")
if (
not task_id or _value(task, "status") != "blocked"
or _value(task, "block_kind") != "transient" or _value(task, "current_run_id") is not None
):
continue
candidate = _due(board, task_id, current)
if candidate is None:
continue
receipt, expected_run = candidate
ordinal = int(receipt["source"]["worker_ordinal"])
if ordinal not in available or not _pool_owned(pool, board, task_id, expected_run, receipt):
continue
if _reopen(kanban_db, board, task_id, expected_run, receipt):
available.remove(ordinal)
reopened += 1
return reopened

View File

@ -285,7 +285,7 @@ def test_manifest_rolls_out_the_bounded_codex_deadline_and_poller_module():
environment = {item["name"]: item["value"] for item in exporter["env"]} environment = {item["name"]: item["value"] for item in exporter["env"]}
assert annotations["ai.bstein.dev/config-rev"] == ( assert annotations["ai.bstein.dev/config-rev"] == (
"20260825-claude-quota-expiry" "20260913-soteria-kanban-recovery-v2"
) )
assert environment["ATLAS_AI_CODEX_QUERY_TIMEOUT_SECONDS"] == "45" assert environment["ATLAS_AI_CODEX_QUERY_TIMEOUT_SECONDS"] == "45"
assert environment["ATLAS_AI_AUTHENTICATION_GRACE_SECONDS"] == "1200" assert environment["ATLAS_AI_AUTHENTICATION_GRACE_SECONDS"] == "1200"
@ -303,9 +303,13 @@ def test_manifest_rolls_out_the_bounded_codex_deadline_and_poller_module():
) )
mounts = {item["name"]: item for item in exporter["volumeMounts"]} mounts = {item["name"]: item for item in exporter["volumeMounts"]}
assert mounts["claude-oauth-access"]["readOnly"] is True assert mounts["claude-oauth-access"]["readOnly"] is True
kustomization = (HERMES / "kustomization.yaml").read_text() kustomization = yaml.safe_load((HERMES / "kustomization.yaml").read_text())
assert "ai_usage_claude.py=scripts/ai_usage_claude.py" in kustomization generator = next(
assert "ai_usage_polling.py=scripts/ai_usage_polling.py" in kustomization item for item in kustomization["configMapGenerator"]
if item["name"] == "hermes-coordinator"
)
assert "ai_usage_claude.py=scripts/ai_usage_claude.py" in generator["files"]
assert "ai_usage_polling.py=scripts/ai_usage_polling.py" in generator["files"]
def test_health_endpoint_uses_poller_state_and_ignores_provider_failure(): def test_health_endpoint_uses_poller_state_and_ignores_provider_failure():

View File

@ -462,3 +462,25 @@ def test_recovery_defers_io_errors_and_lease_failure_releases_ordinal(
assert kanban.blocked[0][1]["expected_run_id"] == 23 assert kanban.blocked[0][1]["expected_run_id"] == 23
assert kanban.blocked[0][1]["kind"] == "transient" assert kanban.blocked[0][1]["kind"] == "transient"
assert store.available_ordinals() == [0, 1, 2] assert store.available_ordinals() == [0, 1, 2]
def test_result_recovery_reconstructs_only_an_exact_native_publication_fence(tmp_path, monkeypatch):
"""A crash after native block preserves the result only with its exact receipt fence."""
blocked_task = task(status="blocked", current_run_id=None)
install_kanban(monkeypatch, [blocked_task], tmp_path)
store = protocol.PoolStore(tmp_path / "fence-recovery.db")
receipt = {"source": {"board": "metis", "task_id": "t_deadbeef", "worker_ordinal": 0}, "result_digest": "a" * 64}
assigned = assignment_payload(scm_resume=receipt)
store.add(binding(), assigned)
failed = {"structured": {"status": "blocked"}, "capacity_failure": True, "scm_submission": None, "scm_resume": receipt}
import publication_retry_fence
seen = []
monkeypatch.setattr(publication_retry_fence, "recover", lambda *args: seen.append(args) or (17, "fence"))
coordinator.Coordinator(MASTER, store).finalize({**binding(), "payload": assigned, "result": failed})
assert seen and store._connect().execute("SELECT state FROM assignments").fetchone()[0] == "finalized"
store = protocol.PoolStore(tmp_path / "fence-recovery-denied.db")
store.add(binding(), assigned)
monkeypatch.setattr(publication_retry_fence, "recover", lambda *_args: None)
coordinator.Coordinator(MASTER, store).finalize({**binding(), "payload": assigned, "result": failed})
assert store._connect().execute("SELECT state FROM assignments").fetchone()[0] == "stale"

View File

@ -192,3 +192,83 @@ def test_a_durable_row_in_any_state_blocks_conflicting_re_adoption(tmp_path, mon
assert store.known_runs() == {("metis", "t_deadbeef", "23")} assert store.known_runs() == {("metis", "t_deadbeef", "23")}
pool.reconcile() pool.reconcile()
assert states(store) == [("23", 0, 1, state)] assert states(store) == [("23", 0, 1, state)]
def test_lease_fence_records_exact_terminal_evidence_idempotently(tmp_path):
"""An expired publication run becomes schedulable only after fenced evidence."""
store = pool_store.PoolStore(tmp_path / "lease-evidence.db")
exact = binding(attempt=3)
receipt = {
"source": {"board": "metis", "task_id": "t_deadbeef", "worker_ordinal": 0},
"result_digest": "a" * 64,
}
store.add(exact, assignment_payload(scm_resume=receipt))
with store._connect() as connection:
connection.execute("UPDATE assignments SET state='lease_failed'")
marker = "[hermes-publication-retry-fence:23:" + "a" * 64 + "]"
assert store.record_publication_lease_failure(exact, receipt, marker)
assert store.record_publication_lease_failure(exact, receipt, marker)
terminal = store.terminal_record("metis", "t_deadbeef", "23")
assert terminal is None # Evidence alone never releases the terminal state.
assert store.finalize(exact, "finalized")
result = store.terminal_record("metis", "t_deadbeef", "23")["result"]
assert result["capacity_failure"] is True and result["publication_retry_lease_fence"] == marker
def test_lease_expiry_auto_retry_evidence_requires_fenced_native_outcome(tmp_path, monkeypatch):
"""A fenced OOM park gains exact terminal evidence for the one retry scheduler."""
kanban = install_kanban(monkeypatch, [task()], tmp_path)
store = pool_store.PoolStore(tmp_path / "lease-auto-retry.db")
exact = binding(attempt=3)
receipt = {
"source": {"board": "metis", "task_id": "t_deadbeef", "worker_ordinal": 0},
"result_digest": "d" * 64,
}
store.add(exact, assignment_payload(scm_resume=receipt))
store.offer(0)
with store._connect() as connection:
connection.execute("UPDATE assignments SET lease_until=1")
monkeypatch.setattr(coordinator.supervisor_state, "reissue_publication_retry", lambda *_args: True)
import publication_retry_fence
marker = "[hermes-publication-retry-fence:23:" + "d" * 64 + "]"
monkeypatch.setattr(
publication_retry_fence, "block",
lambda db, connection, board, task_id, run_id, reason, kind, fenced: (
db.block_task(connection, task_id, reason=reason, kind=kind, expected_run_id=int(run_id)) or True
),
)
monkeypatch.setattr(publication_retry_fence, "recover", lambda *_args: (17, marker))
coordinator.Coordinator(MASTER, store).expire_leases()
result = store.terminal_record("metis", "t_deadbeef", "23")["result"]
assert result["scm_resume"] == receipt and result["publication_retry_lease_fence"] == marker
assert kanban.blocked[-1][1]["kind"] == "transient"
def test_crashed_fenced_lease_reconstructs_only_the_current_native_event(tmp_path, monkeypatch):
"""A native blocked/null-run shape resumes only through the preserved fence."""
blocked = task(status="blocked", current_run_id=None)
install_kanban(monkeypatch, [blocked], tmp_path)
store = pool_store.PoolStore(tmp_path / "lease-crash-fence.db")
exact = binding(attempt=3)
receipt = {
"source": {"board": "metis", "task_id": "t_deadbeef", "worker_ordinal": 0},
"result_digest": "e" * 64,
}
store.add(exact, assignment_payload(scm_resume=receipt))
with store._connect() as connection:
connection.execute("UPDATE assignments SET state='lease_failed'")
monkeypatch.setattr(coordinator.supervisor_state, "reissue_publication_retry", lambda *_args: True)
import publication_retry_fence
marker = "[hermes-publication-retry-fence:23:" + "e" * 64 + "]"
monkeypatch.setattr(publication_retry_fence, "recover", lambda *_args: (41, marker))
coordinator.Coordinator(MASTER, store).expire_leases()
assert store.terminal_record("metis", "t_deadbeef", "23")["result"]["publication_retry_lease_fence"] == marker
store = pool_store.PoolStore(tmp_path / "lease-crash-human.db")
store.add(exact, assignment_payload(scm_resume=receipt))
with store._connect() as connection:
connection.execute("UPDATE assignments SET state='lease_failed'")
monkeypatch.setattr(publication_retry_fence, "recover", lambda *_args: None)
coordinator.Coordinator(MASTER, store).expire_leases()
assert states(store) == [("23", 0, 3, "lease_failed")]

View File

@ -31,6 +31,9 @@ from testing.tests.test_hermes_execution_pool_mediator import ( # noqa: E402
) )
MEDIATOR_FIELDS = {"scm_submission", "scm_resume", "publication_retry_transient", "publication_retry_lease_fence", "lease_fence"}
def attempt_assignment(attempt, continuation_kind=""): def attempt_assignment(attempt, continuation_kind=""):
"""One signed assignment envelope bound to an exact retry attempt.""" """One signed assignment envelope bound to an exact retry attempt."""
return protocol.sign_envelope( return protocol.sign_envelope(
@ -603,6 +606,32 @@ def test_lost_publication_checkpoint_prevents_the_push(tmp_path, monkeypatch):
assert not any("push" in call for call in calls) assert not any("push" in call for call in calls)
def test_worker_cannot_sign_forged_scm_control_on_blocked_or_failed_finish():
"""Only private mediator SCM work may add terminal publication evidence."""
forged = {
"scm_submission": {"branch": "attacker/ref", "pull_request": "https://scm/pulls/99", "head": "f" * 40},
"scm_resume": {"result_digest": "e" * 64},
"publication_retry_transient": True,
"publication_retry_lease_fence": "forged-fence", "lease_fence": "forged-lease",
}
boundary = client.ClientBoundary(KEY, FailingSCM(protocol.ProtocolError("refused")))
boundary.current = assignment()
posted = []
boundary._post = lambda _path, envelope: posted.append(envelope) or protocol.sign_envelope(
KEY, "ack", binding(), {"accepted": True, "duplicate": False}
)
blocked = {**RESULT, "status": "blocked", "blockers": ["worker policy"]}
boundary.finish({"binding": binding(), "payload": {"structured": blocked, "returncode": 1, **forged}})
terminal = protocol.verify_envelope(KEY, posted.pop())["payload"]
assert not MEDIATOR_FIELDS.intersection(terminal)
boundary.current = assignment()
boundary.finish({"binding": binding(), "payload": {"structured": dict(RESULT), "returncode": 0, **forged}})
terminal = protocol.verify_envelope(KEY, posted.pop())["payload"]
assert terminal["capacity_failure"] is True
assert not MEDIATOR_FIELDS.intersection(terminal)
def test_worker_cannot_spoof_a_transient_publication_retry_result(): def test_worker_cannot_spoof_a_transient_publication_retry_result():
boundary = client.ClientBoundary(KEY, FailingSCM(protocol.ProtocolError("unused"))) boundary = client.ClientBoundary(KEY, FailingSCM(protocol.ProtocolError("unused")))
boundary.current = assignment(payload={"scm_resume": {}}) boundary.current = assignment(payload={"scm_resume": {}})
@ -624,12 +653,13 @@ def test_a_successful_submission_records_both_the_draft_and_the_exact_branch():
return { return {
"workspace": "/workspace/runs/metis/t_deadbeef/42", "workspace": "/workspace/runs/metis/t_deadbeef/42",
"branch": "wt/t_deadbeef-attempt-2", "branch": "wt/t_deadbeef-attempt-2",
"pull_request": "https://scm/pulls/9", "pull_request": "https://scm/pulls/9", "head": "a" * 40,
} }
boundary = client.ClientBoundary(KEY, Recording()) boundary = client.ClientBoundary(KEY, Recording())
boundary.current = assignment() boundary.current = assignment()
boundary._post = lambda _path, _envelope: protocol.sign_envelope( posted = []
boundary._post = lambda _path, envelope: posted.append(envelope) or protocol.sign_envelope(
KEY, "ack", binding(), {"accepted": True, "duplicate": False} KEY, "ack", binding(), {"accepted": True, "duplicate": False}
) )
finished = boundary.finish( finished = boundary.finish(
@ -643,3 +673,7 @@ def test_a_successful_submission_records_both_the_draft_and_the_exact_branch():
assert finished["structured"]["artifacts"] == [ assert finished["structured"]["artifacts"] == [
"https://scm/pulls/9", "branch:wt/t_deadbeef-attempt-2", "https://scm/pulls/9", "branch:wt/t_deadbeef-attempt-2",
] ]
terminal = protocol.verify_envelope(KEY, posted[0])["payload"]
assert terminal["scm_submission"] == {
"branch": "wt/t_deadbeef-attempt-2", "pull_request": "https://scm/pulls/9", "head": "a" * 40,
}

View File

@ -222,9 +222,11 @@ def test_flux_orders_observer_rbac_before_hermes_prunes_old_authority():
).read_text() ).read_text()
) )
assert "dependsOn" not in observer["spec"] assert "dependsOn" not in observer["spec"]
assert {item["name"] for item in hermes["spec"]["dependsOn"]} >= { assert "hermes-observer-rbac" in {
"hermes-observer-rbac", item["name"] for item in hermes["spec"]["dependsOn"]
"hermes-scm-broker", }
assert "hermes-scm-broker" not in {
item["name"] for item in hermes["spec"]["dependsOn"]
} }
assert {item["name"] for item in bindings["spec"]["dependsOn"]} == { assert {item["name"] for item in bindings["spec"]["dependsOn"]} == {
"hermes-observer-rbac", "hermes-observer-rbac",

View File

@ -0,0 +1,121 @@
"""Transactional publication-retry fence coverage."""
from __future__ import annotations
import json
import sqlite3
from contextlib import contextmanager
from testing.tests.test_hermes_cli_support import _load
fence = _load("publication_retry_fence")
class Native:
"""Minimal existing-native transaction surface used by the adapter."""
@staticmethod
@contextmanager
def write_txn(connection):
connection.execute("BEGIN IMMEDIATE")
try:
yield
except Exception:
connection.rollback()
raise
else:
connection.commit()
@staticmethod
def _append_event(connection, task_id, kind, payload):
connection.execute(
"INSERT INTO task_events(task_id,run_id,kind,payload) VALUES(?,?,?,?)",
(task_id, None, kind, json.dumps(payload, sort_keys=True)),
)
def _database():
connection = sqlite3.connect(":memory:")
connection.row_factory = sqlite3.Row
connection.executescript("""
CREATE TABLE tasks(id TEXT PRIMARY KEY,status TEXT,current_run_id INTEGER,
consecutive_failures INTEGER,last_failure_error TEXT);
CREATE TABLE task_links(child_id TEXT,parent_id TEXT);
CREATE TABLE task_events(id INTEGER PRIMARY KEY AUTOINCREMENT,task_id TEXT,
run_id INTEGER,kind TEXT,payload TEXT);
""")
connection.execute("INSERT INTO tasks VALUES('root','done',NULL,0,NULL)")
connection.execute("INSERT INTO tasks VALUES('child','blocked',NULL,1,'broker')")
connection.execute("INSERT INTO task_links VALUES('child','root')")
return connection
def _blocked(connection, marker):
event_id = connection.execute(
"INSERT INTO task_events(task_id,run_id,kind,payload) VALUES(?,?,?,?)",
("child", 8, "blocked", json.dumps({"reason": "transport\\n" + marker})),
).lastrowid
connection.commit()
return event_id
def test_adapter_reopens_only_the_exact_native_fence():
"""The existing native transaction protects the event check and state change."""
connection = _database()
marker = "[hermes-publication-retry-fence:8:" + "a" * 64 + "]"
event_id = _blocked(connection, marker)
assert fence.guarded_unblock(Native, connection, "child", "root", "8", (event_id, marker))
assert connection.execute("SELECT status FROM tasks WHERE id='child'").fetchone()[0] == "ready"
assert connection.execute("SELECT kind FROM task_events ORDER BY id DESC LIMIT 1").fetchone()[0] == "unblocked"
def test_adapter_keeps_a_later_human_transient_block_closed():
"""A later event, even with the same block kind, invalidates coordinator ownership."""
connection = _database()
marker = "[hermes-publication-retry-fence:8:" + "b" * 64 + "]"
event_id = _blocked(connection, marker)
connection.execute(
"INSERT INTO task_events(task_id,run_id,kind,payload) VALUES(?,?,?,?)",
("child", None, "blocked", json.dumps({"reason": "human decision"})),
)
connection.commit()
assert not fence.guarded_unblock(Native, connection, "child", "root", "8", (event_id, marker))
assert connection.execute("SELECT status FROM tasks WHERE id='child'").fetchone()[0] == "blocked"
def test_block_records_the_exact_native_event_as_coordinator_owned(tmp_path, monkeypatch):
"""Only the coordinator's successful fenced block creates a sidecar ownership row."""
monkeypatch.setattr(fence.supervisor_state, "KANBAN_ROOT", tmp_path / "boards")
connection = _database()
connection.execute("UPDATE tasks SET status='running',current_run_id=8 WHERE id='child'")
connection.commit()
class BlockingNative(Native):
@staticmethod
def block_task(conn, task_id, *, reason, kind, expected_run_id):
changed = conn.execute(
"UPDATE tasks SET status='blocked',current_run_id=NULL WHERE id=? AND status='running' AND current_run_id=?",
(task_id, expected_run_id),
).rowcount
if changed:
conn.execute(
"INSERT INTO task_events(task_id,run_id,kind,payload) VALUES(?,?,?,?)",
(task_id, expected_run_id, "blocked", json.dumps({"reason": reason, "kind": kind})),
)
return bool(changed)
receipt = {"result_digest": "c" * 64}
assert fence.block(BlockingNative, connection, "soteria", "child", "8", "transport", "transient", receipt)
event_id = connection.execute("SELECT id FROM task_events ORDER BY id DESC LIMIT 1").fetchone()[0]
marker = "[hermes-publication-retry-fence:8:" + "c" * 64 + "]"
assert fence.read("soteria", "child", "8") == (event_id, marker)
# Simulate a crash after native block but before its sidecar fence record.
with fence.supervisor_state._connect("soteria") as state:
state.execute("DELETE FROM publication_retry_fences")
state.execute(
"INSERT INTO publication_retries(board,child_task_id,source_run_id,source_ordinal,receipt_json) VALUES(?,?,?,?,?)",
("soteria", "child", "8", 0, json.dumps(receipt)),
)
assert fence.recover("soteria", "child", "8", connection, receipt) == (event_id, marker)

View File

@ -0,0 +1,182 @@
"""Scheduler coverage for coordinator-owned publication retry receipts."""
from __future__ import annotations
import json
from contextlib import nullcontext
from pathlib import Path
from types import SimpleNamespace
from testing.tests.test_hermes_cli_support import HERMES, _load
state = _load("supervisor_state")
retry = __import__("publication_retry")
scheduler = _load("publication_retry_scheduler")
class Native:
"""Small native surface that records only safe unblock transitions."""
def __init__(self, task, root, run):
self.task, self.root, self.run, self.unblocks, self.event_id = task, root, run, 0, 17
def scoped_current_board(self, _board):
return nullcontext()
def connect(self, *, board):
return SimpleNamespace(
execute=lambda _sql, _args: SimpleNamespace(fetchone=lambda: self.run),
close=lambda: None,
)
def list_tasks(self, _connection):
return [self.task]
def get_task(self, _connection, _task_id):
return self.task
def parent_ids(self, _connection, _task_id):
return [self.root]
def unblock_task(self, _connection, _task_id):
if self.task.status != "blocked":
return False
self.task.status = "ready"
self.unblocks += 1
return True
def unblock_task_if_event(self, connection, _task_id, *, expected_event_id, **_kwargs):
return self.unblock_task(connection, _task_id) if expected_event_id == self.event_id else False
class Store:
"""Pool records needed to prove the terminal retry belongs to Hermes."""
def __init__(self, records, active=()):
self.records, self.active = records, list(active)
def active_assignments(self):
return self.active
def available_ordinals(self):
return [0]
def terminal_record(self, board, task_id, run_id):
return self.records.get((board, task_id, run_id))
def _receipt(root, child, baseline, head):
structured = {
"status": "completed", "summary": "Replace cache literals.", "changed_files": ["a.go"],
"tests_run": ["go test ./..."], "artifacts": [], "findings": [], "blockers": [],
}
source = {
"board": "soteria", "task_id": child, "run_id": "8", "worker_ordinal": 0, "attempt": 1,
"root_task_id": root, "repo_url": "https://scm.bstein.dev/titan/soteria.git",
"branch": "hermes-repair/cache", "base_branch": "main",
}
title, body = structured["summary"], json.dumps(structured, sort_keys=True)
return {"source": source, "baseline_sha": baseline, "head": head, "title": title, "body": body,
"structured": structured, "result_digest": retry.receipt_digest(structured, title, body)}
def _setup(tmp_path, monkeypatch):
board, root, child, baseline, head = "soteria", "t_root", "t_child", "a" * 40, "b" * 40
monkeypatch.setattr(state, "KANBAN_ROOT", tmp_path / "boards")
lineage = state.Lineage(root, "hermes-repair/cache", "https://scm.bstein.dev/titan/soteria/pulls/3", "soteria", "main")
state.record_submission(board, root, lineage, baseline)
state.record_child(board, child, root, root, "repair", baseline, "replace cache literals")
receipt = _receipt(root, child, baseline, head)
binding = {name: receipt["source"][name] for name in ("board", "task_id", "run_id", "worker_ordinal", "attempt")}
state.record_publication_retry(board, child, binding, receipt)
task = SimpleNamespace(id=child, status="blocked", block_kind="transient", current_run_id=None)
native = Native(task, root, (8, "blocked", "blocked"))
source_record = {"state": "finalized", "worker_ordinal": 0, "payload": {}, "result": {
"scm_resume": receipt, "structured": {"status": "blocked"}, "capacity_failure": True,
"scm_submission": None,
}}
return board, root, child, baseline, receipt, task, native, source_record
def test_initial_retry_waits_then_reopens_once_with_exact_source_record(tmp_path, monkeypatch):
"""A clean initial refusal waits, then opens one source-ordinal retry."""
board, _root, child, _baseline, receipt, _task, native, source = _setup(tmp_path, monkeypatch)
pool = SimpleNamespace(store=Store({(board, child, "8"): source}))
monkeypatch.setattr(scheduler, "read_fence", lambda *_args: (17, "fence"))
monkeypatch.setattr(
scheduler, "guarded_unblock",
lambda _db, connection, _task, _root, _run, fence: native.unblock_task_if_event(
connection, _task, expected_event_id=fence[0]
),
)
assert scheduler.reopen_due(pool, native, [board], now=100) == 0
assert scheduler.reopen_due(pool, native, [board], now=401) == 1
assert native.unblocks == 1
assert scheduler.reopen_due(pool, native, [board], now=402) == 0
assert native.unblocks == 1
native.task.status, native.task.block_kind = "blocked", "capability"
assert scheduler.reopen_due(pool, native, [board], now=403) == 0
native.task.block_kind = "transient"
native.event_id = 18
assert scheduler.reopen_due(pool, native, [board], now=403) == 0
native.event_id = 17
lineage = state.get_root(board, "t_root")
state.record_submission(board, child, lineage, "c" * 40)
assert scheduler.reopen_due(pool, native, [board], now=404) == 0
def test_reissued_transient_receipt_reopens_only_its_last_terminal_run(tmp_path, monkeypatch):
"""A transient fresh-run failure uses the single release and same ordinal."""
board, _root, child, _baseline, receipt, task, native, source = _setup(tmp_path, monkeypatch)
monkeypatch.setattr(state.time, "time", lambda: 100)
state.issue_publication_retry(board, child, "9")
assert state.reissue_publication_retry(board, child, "9") is True
monkeypatch.setattr(state.time, "time", lambda: 401)
native.run = (9, "blocked", "blocked")
current = {"worker_ordinal": 0, "payload": {"scm_resume": receipt}, "result": {}}
pool = SimpleNamespace(store=Store({(board, child, "8"): source, (board, child, "9"): current}))
current["state"] = "finalized"
current["result"] = {"structured": {"status": "blocked"}, "capacity_failure": True, "scm_submission": None}
monkeypatch.setattr(scheduler, "read_fence", lambda *_args: (17, "fence"))
monkeypatch.setattr(
scheduler, "guarded_unblock",
lambda _db, connection, _task, _root, _run, fence: native.unblock_task_if_event(
connection, _task, expected_event_id=fence[0]
),
)
assert scheduler.reopen_due(pool, native, [board], now=401) == 1
assert task.status == "ready" and native.unblocks == 1
def test_reopen_rejects_unproven_or_human_changed_terminal_blocks(tmp_path, monkeypatch):
"""No stale pool row or later transient human action can reopen a card."""
board, _root, child, _baseline, _receipt, task, native, source = _setup(tmp_path, monkeypatch)
monkeypatch.setattr(scheduler, "read_fence", lambda *_args: (17, "fence"))
monkeypatch.setattr(scheduler, "guarded_unblock", lambda *_args: True)
# A non-terminal row has no final coordinator block evidence.
pending = dict(source, state="running")
assert scheduler.reopen_due(SimpleNamespace(store=Store({(board, child, "8"): pending})), native, [board], now=401) == 0
# A concurrent pool owner is an authoritative later-live assignment.
active = {"board": board, "task_id": child, "run_id": "10"}
assert scheduler.reopen_due(SimpleNamespace(store=Store({(board, child, "8"): source}, [active])), native, [board], now=401) == 0
# A terminal result without coordinator capacity evidence is not owned.
denied = dict(source, result={"structured": {"status": "blocked"}, "capacity_failure": False, "scm_submission": None})
assert scheduler.reopen_due(SimpleNamespace(store=Store({(board, child, "8"): denied})), native, [board], now=401) == 0
# A later human transient block changes the native event identity and remains blocked.
monkeypatch.setattr(scheduler, "guarded_unblock", lambda *_args: False)
assert scheduler.reopen_due(SimpleNamespace(store=Store({(board, child, "8"): source})), native, [board], now=401) == 0
assert task.status == "blocked" and native.unblocks == 0
def test_missing_receipt_never_turns_a_transient_block_into_retry_work(tmp_path, monkeypatch):
"""A generic coordinator or human transient block has no publication authority."""
board, _root, child, _baseline, _receipt, _task, native, source = _setup(tmp_path, monkeypatch)
with state._connect(board) as connection:
connection.execute("DELETE FROM publication_retries WHERE board=? AND child_task_id=?", (board, child))
monkeypatch.setattr(scheduler, "read_fence", lambda *_args: (17, "fence"))
monkeypatch.setattr(scheduler, "guarded_unblock", lambda *_args: True)
pool = SimpleNamespace(store=Store({(board, child, "8"): source}))
assert scheduler.reopen_due(pool, native, [board], now=401) == 0

View File

@ -189,6 +189,7 @@ def test_execution_worker_and_mediator_separate_claude_token_and_hmac(
master = "e" * 64 master = "e" * 64
(vault / "execution-pool-key").write_text(master) (vault / "execution-pool-key").write_text(master)
(vault / "scm-task-grant-key").write_text("scm-boundary-key")
stage.stage_execution_mediator() stage.stage_execution_mediator()
expected = hmac.new( expected = hmac.new(
master.encode(), b"hermes-execution-pool-v2:worker:1", hashlib.sha256 master.encode(), b"hermes-execution-pool-v2:worker:1", hashlib.sha256

View File

@ -132,7 +132,7 @@ def test_flux_wires_the_sibling_runtime_rollout_service_and_narrow_policy():
deployment = _documents(HERMES / "switchyard-deployment.yaml")[0] deployment = _documents(HERMES / "switchyard-deployment.yaml")[0]
template = deployment["spec"]["template"] template = deployment["spec"]["template"]
assert template["metadata"]["annotations"]["ai.bstein.dev/config-rev"].endswith( assert template["metadata"]["annotations"]["ai.bstein.dev/config-rev"].endswith(
"voice-route-preflight-v1" "capability-effort-v5"
) )
classifier = next( classifier = next(
item for item in template["spec"]["containers"] if item["name"] == "classifier-broker" item for item in template["spec"]["containers"] if item["name"] == "classifier-broker"