#!/usr/bin/env python3 """Run durable Codex and Claude workers from Hermes' authoritative Kanban.""" from __future__ import annotations import concurrent.futures import ctypes import errno import hashlib import json import os import re import selectors import signal import sqlite3 import stat import subprocess import sys import threading import time import uuid import urllib.error import urllib.request from dataclasses import asdict, dataclass from datetime import datetime, timezone from pathlib import Path from typing import Any, Callable import cli_lane_goal DATA_ROOT = Path(os.environ.get("HERMES_HOME", "/opt/data")) SWITCHYARD_URL = os.environ.get( "HERMES_SWITCHYARD_URL", "http://hermes-switchyard.hermes.svc.cluster.local:9005/v1/chat/completions", ) STATE_ROOT = DATA_ROOT / "cli-lanes" CODEX_BIN = DATA_ROOT / "tools/bin/codex" CLAUDE_BIN = DATA_ROOT / "tools/bin/claude" CLAUDE_SETTINGS = DATA_ROOT / "home/.claude/settings.json" RESULT_SCHEMA_PATH = STATE_ROOT / "worker-result.schema.json" EFFORTS = ("low", "medium", "high", "xhigh") EXTERNAL_PREFIX = "cli-" DEFAULT_CLAIM_TTL = 7 * 24 * 60 * 60 DEFAULT_MAX_RUNTIME = 12 * 60 * 60 HEARTBEAT_SECONDS = 20 PROVIDER_HEALTH_MAX_AGE_SECONDS = 5 * 60 PROVIDER_AUTH_FAILURE_MAX_AGE_SECONDS = 12 * 60 * 60 KANBAN_STORAGE_ATTEMPTS = 5 ARTIFACT_GC_INTERVAL_SECONDS = 5 * 60 ARTIFACT_RETENTION_AGE_SECONDS = 30 * 24 * 60 * 60 ARTIFACT_RETENTION_COUNT = 256 ARTIFACT_RETENTION_BYTES = 128 * 1024 * 1024 MAX_TERMINAL_RECORD_BYTES = 1024 * 1024 QUARANTINE_HASH_BYTES = 64 * 1024 PROVIDER_HEALTH_PATHS = { "codex": DATA_ROOT / "provider-health/codex.json", "claude": DATA_ROOT / "provider-health/claude.json", } WORKTREE_LOCK = threading.Lock() BOARD_CORRUPTION_ERRORS: dict[str, str] = {} LAST_ARTIFACT_GC = 0.0 CAPACITY_PATTERN = re.compile( r"(?:rate.?limit|capacity|overload|usage.?limit|quota|credit|exhaust|429|529|authentication|oauth|token.*expired)", re.I, ) NO_CLAUDE_SESSION = "No conversation found with session ID:" CLAUDE_SESSION_COLLISION = "Session ID already in use" NO_CODEX_THREAD = "no rollout found for thread id" RESULT_SCHEMA: dict[str, Any] = { "type": "object", "additionalProperties": False, "required": [ "status", "summary", "changed_files", "tests_run", "artifacts", "findings", "blockers", ], "properties": { "status": { "type": "string", "enum": sorted(cli_lane_goal.RESULT_STATUSES), }, "summary": {"type": "string"}, "changed_files": {"type": "array", "items": {"type": "string"}}, "tests_run": {"type": "array", "items": {"type": "string"}}, "artifacts": {"type": "array", "items": {"type": "string"}}, "findings": { "type": "array", "items": {"type": "string"}, "description": ( "Defects, risks, or observations discovered by a review or diagnosis. " "Findings do not prevent the assigned review or diagnosis from completing." ), }, "blockers": { "type": "array", "items": {"type": "string"}, "description": ( "Concrete obstacles that prevent completion of the assigned task itself. " "This must be empty when status is completed; review findings belong in findings." ), }, }, } @dataclass(frozen=True) class Route: """One provider/model/effort decision for an observable worker boundary.""" provider: str model: str effort: str profile: str classifier: str reason: str latency_ms: int fallback_chain: tuple[str, ...] @dataclass class ProcessResult: """Captured outcome from one provider CLI invocation.""" returncode: int output: str structured: dict[str, Any] | None capacity_failure: bool class TerminalFinalizationPending(RuntimeError): """An accepted worker result is durable but not committed to Kanban yet.""" @dataclass(frozen=True) class TerminalIdentity: """Replay authority derived only from a journal's directory and filename.""" board: str task_id: str run_id: int state: str @dataclass class TerminalSnapshot: """One inode-bound, bounded journal read held open across finalization.""" document: dict[str, Any] file_stat: os.stat_result descriptor: int directory_descriptor: int def close(self) -> None: """Release the pinned file and directory descriptors.""" os.close(self.descriptor) os.close(self.directory_descriptor) @dataclass class TerminalRecoverySnapshot: """One raw directory entry pinned before recovery classifies its payload.""" document: dict[str, Any] | None file_stat: os.stat_result descriptor: int | None directory_descriptor: int prefix: bytes invalid_reason: str | None def close(self) -> None: """Release the descriptors retained across classification/quarantine.""" if self.descriptor is not None: os.close(self.descriptor) self.descriptor = None if self.directory_descriptor >= 0: os.close(self.directory_descriptor) self.directory_descriptor = -1 def utc_now() -> str: return datetime.now(timezone.utc).isoformat() def atomic_json(path: Path, value: dict[str, Any], mode: int = 0o600) -> None: """Durably replace a small state document without following temp symlinks.""" path.parent.mkdir(parents=True, exist_ok=True, mode=0o700) os.chmod(path.parent, 0o700, follow_symlinks=False) temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp") flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL flags |= getattr(os, "O_NOFOLLOW", 0) descriptor = None try: descriptor = os.open(temporary, flags, mode) with os.fdopen(descriptor, "w", encoding="utf-8") as stream: descriptor = None json.dump(value, stream, indent=2, sort_keys=True) stream.write("\n") stream.flush() os.fsync(stream.fileno()) os.chmod(temporary, mode, follow_symlinks=False) os.replace(temporary, path) _fsync_directory(path.parent) finally: if descriptor is not None: os.close(descriptor) try: temporary.unlink() except FileNotFoundError: pass def _fsync_directory(directory: Path) -> None: """Persist directory-entry changes after an atomic rename or quarantine.""" flags = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) flags |= getattr(os, "O_NOFOLLOW", 0) descriptor = os.open(directory, flags) try: os.fsync(descriptor) finally: os.close(descriptor) def load_json(path: Path) -> dict[str, Any]: try: flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) descriptor = os.open(path, flags) with os.fdopen(descriptor, "r", encoding="utf-8") as stream: value = json.load(stream) except (OSError, UnicodeError, json.JSONDecodeError): return {} return value if isinstance(value, dict) else {} def parse_assignee(assignee: str) -> tuple[str | None, str | None]: """Parse external lane overrides while leaving cli-auto fully automatic.""" value = str(assignee or "").strip().lower() if value == "cli-auto": return None, None match = re.fullmatch(r"cli-(codex|claude)-(low|medium|high|xhigh)", value) if not match: raise ValueError(f"unsupported external lane assignee: {assignee}") return match.group(1), match.group(2) def fresh_unavailable_provider(now: float | None = None) -> str | None: """Return one recently proven-down provider for automatic route exclusion.""" current = time.time() if now is None else now unavailable: list[str] = [] for provider, path in PROVIDER_HEALTH_PATHS.items(): health = load_json(path) try: age = current - path.stat().st_mtime except OSError: continue max_age = ( PROVIDER_AUTH_FAILURE_MAX_AGE_SECONDS if health.get("authenticated") is False else PROVIDER_HEALTH_MAX_AGE_SECONDS ) if ( 0 <= age <= max_age and health.get("state") == "unavailable" ): unavailable.append(provider) # If both providers are down, keep the normal attempt/fallback path so the # card records authoritative current errors instead of trusting snapshots. return unavailable[0] if len(unavailable) == 1 else None def _decode_worker_target(value: str) -> tuple[str, str, str]: """Decode the selected model header emitted by a worker decision route.""" parts = value.split("/", 3) if len(parts) != 4 or parts[0] != "worker": raise RuntimeError(f"invalid Switchyard worker target: {value}") provider, model, effort = parts[1:] if provider not in {"codex", "claude"} or effort not in EFFORTS: raise RuntimeError(f"unsupported Switchyard worker target: {value}") return provider, model, effort def select_route( prompt: str, assignee: str, *, exclude_provider: str | None = None, exclude_reason: str | None = None, switchyard_url: str = SWITCHYARD_URL, open_request: Callable[..., Any] = urllib.request.urlopen, ) -> Route: """Ask Switchyard to select one native CLI worker at this boundary.""" started = time.monotonic() manual_provider, manual_effort = parse_assignee(assignee) if manual_provider and manual_effort: route_id = f"atlas/worker/manual/{manual_provider}/{manual_effort}" source = "switchyard-manual" else: route_id = "atlas/worker/auto/maximum" source = "switchyard-classifier" context = prompt if exclude_provider: reason = exclude_reason or "failed or exhausted capacity at this boundary" context += ( f"\n\nRouting constraint: the {exclude_provider} provider {reason}. " "Do not select it." ) payload = json.dumps( { "model": route_id, "messages": [{"role": "user", "content": context}], "stream": False, "max_tokens": 1, } ).encode("utf-8") request = urllib.request.Request( switchyard_url, data=payload, headers={"Content-Type": "application/json"}, method="POST", ) try: with open_request(request, timeout=60) as response: selected = str(response.headers.get("x-model-router-selected-model") or "") rationale = str(response.headers.get("x-model-router-rationale") or "") response_body = response.read() except (OSError, urllib.error.URLError) as exc: raise RuntimeError(f"Switchyard worker routing failed: {exc}") from exc provider, model, effort = _decode_worker_target(selected) # Switchyard preserves the stable tier target in the selection header and # top-level response model. The worker broker's assistant content contains # the steward-resolved provider model required by the native CLI. try: response_document = json.loads(response_body) resolved_target = str(response_document["choices"][0]["message"]["content"] or "") resolved_provider, resolved_model, resolved_effort = _decode_worker_target(resolved_target) required_prefix = "gpt-" if provider == "codex" else "claude-" if ( resolved_provider == provider and resolved_effort == effort and resolved_model.startswith(required_prefix) ): model = resolved_model except (AttributeError, IndexError, KeyError, RuntimeError, TypeError, ValueError, json.JSONDecodeError): pass if exclude_provider and provider == exclude_provider: alternate = "claude" if provider == "codex" else "codex" guarded = select_route( prompt, f"cli-{alternate}-{effort}", switchyard_url=switchyard_url, open_request=open_request, ) return Route( provider=guarded.provider, model=guarded.model, effort=guarded.effort, profile=guarded.profile, classifier=f"{source}-health-guard", reason=( f"Switchyard selected excluded {exclude_provider}; preserved " f"its {effort} effort on healthy-provider route. {guarded.reason}" ), latency_ms=int((time.monotonic() - started) * 1000), fallback_chain=(), ) return Route( provider=provider, model=model, effort=effort, profile=f"{provider}-{effort}", classifier=source, reason=rationale or f"Switchyard selected {selected}", latency_ms=int((time.monotonic() - started) * 1000), fallback_chain=(), ) def state_path(board: str, task_id: str) -> Path: safe_board = re.sub(r"[^a-zA-Z0-9_.-]+", "-", board) safe_task = re.sub(r"[^a-zA-Z0-9_.-]+", "-", task_id) return STATE_ROOT / safe_board / f"{safe_task}.json" def _result_path(state_file: Path, run_id: Any, sequence: int) -> Path: """Return a unique provider-result path; completed turns are never reused.""" safe_run = re.sub(r"[^a-zA-Z0-9_.-]+", "-", str(run_id or "unknown")) return state_file.with_name( f"{state_file.stem}.run-{safe_run}.provider-{sequence}.result.json" ) def _candidate_path(state_file: Path, run_id: Any, sequence: int) -> Path: """Return the durable path for one exact structured worker response.""" safe_run = re.sub(r"[^a-zA-Z0-9_.-]+", "-", str(run_id or "unknown")) return state_file.with_name( f"{state_file.stem}.run-{safe_run}.candidate-{sequence}.json" ) def _terminal_path( state_file: Path, run_id: Any, state: str = "pending", ) -> Path: """Return the identity-bound replay path for an accepted terminal response.""" if state not in {"pending", "committed"}: raise ValueError(f"invalid terminal journal state: {state}") safe_run = re.sub(r"[^a-zA-Z0-9_.-]+", "-", str(run_id or "unknown")) return state_file.with_name( f"{state_file.stem}.run-{safe_run}.terminal.{state}.json" ) _TERMINAL_NAME = re.compile( r"^(?P[a-zA-Z0-9_.-]+)\.run-(?P[0-9]+)\." r"terminal\.(?Ppending|committed)\.json$" ) _TERMINAL_EVIDENCE_NAME = re.compile( r"^(?P[a-zA-Z0-9_.-]+)\.run-(?P[0-9]+)\." r"terminal\.(?Pprepared|conflict)-(?P[a-f0-9]{32})\.json$" ) _RETIRE_NAME = re.compile( r"^\.retire\.(?P[a-f0-9]{16})\." r"(?P[a-f0-9]{16})\.(?P[0-9]+)$" ) _LEGACY_RETIRE_NAME = re.compile(r"^\.retire\.[a-f0-9]{32}\.[0-9]+$") _SAFE_BOARD = re.compile(r"^[a-zA-Z0-9_.-]+$") def _terminal_identity(path: Path) -> TerminalIdentity | None: """Parse replay authority lexically before opening a journal or board.""" try: relative = path.relative_to(STATE_ROOT) except ValueError: return None if len(relative.parts) != 2: return None board, filename = relative.parts match = _TERMINAL_NAME.fullmatch(filename) try: board_stat = path.parent.stat(follow_symlinks=False) except OSError: return None if ( not match or not stat.S_ISDIR(board_stat.st_mode) or not _SAFE_BOARD.fullmatch(board) or board in {".", ".."} or match.group("task") in {".", ".."} ): return None return TerminalIdentity( board=board, task_id=match.group("task"), run_id=int(match.group("run")), state=match.group("state"), ) def _terminal_evidence_identity(path: Path) -> TerminalIdentity | None: """Parse immutable prepared/conflict evidence authority from its path.""" try: relative = path.relative_to(STATE_ROOT) except ValueError: return None if len(relative.parts) != 2: return None board, filename = relative.parts match = _TERMINAL_EVIDENCE_NAME.fullmatch(filename) try: board_stat = path.parent.stat(follow_symlinks=False) except OSError: return None if ( not match or not stat.S_ISDIR(board_stat.st_mode) or not _SAFE_BOARD.fullmatch(board) or board in {".", ".."} or match.group("task") in {".", ".."} ): return None return TerminalIdentity( board=board, task_id=match.group("task"), run_id=int(match.group("run")), state=match.group("state"), ) def _read_bounded(descriptor: int, limit: int) -> bytes: """Read at most ``limit`` bytes from a regular file descriptor.""" chunks: list[bytes] = [] remaining = max(0, limit) while remaining: chunk = os.read(descriptor, min(64 * 1024, remaining)) if not chunk: break chunks.append(chunk) remaining -= len(chunk) return b"".join(chunks) def _open_small_json_snapshot(path: Path) -> TerminalSnapshot | None: """Open and pin one bounded, singly-linked JSON document.""" directory_flags = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) directory_flags |= getattr(os, "O_NOFOLLOW", 0) file_flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) directory = None descriptor = None try: directory = os.open(path.parent, directory_flags) descriptor = os.open(path.name, file_flags, dir_fd=directory) opened = os.fstat(descriptor) if ( not stat.S_ISREG(opened.st_mode) or opened.st_nlink != 1 or opened.st_size > MAX_TERMINAL_RECORD_BYTES ): return None payload = _read_bounded(descriptor, MAX_TERMINAL_RECORD_BYTES + 1) if len(payload) > MAX_TERMINAL_RECORD_BYTES: return None current = os.stat( path.name, dir_fd=directory, follow_symlinks=False, ) if ( current.st_dev != opened.st_dev or current.st_ino != opened.st_ino or current.st_size != opened.st_size or len(payload) != opened.st_size ): return None value = json.loads(payload.decode("utf-8")) if not isinstance(value, dict): return None snapshot = TerminalSnapshot(value, opened, descriptor, directory) descriptor = None directory = None return snapshot except (OSError, UnicodeError, json.JSONDecodeError): return None finally: if descriptor is not None: os.close(descriptor) if directory is not None: os.close(directory) def _open_terminal_recovery_snapshot(path: Path) -> TerminalRecoverySnapshot | None: """Pin the entry recovery inspected, even when its payload is malformed.""" directory_flags = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) directory_flags |= getattr(os, "O_NOFOLLOW", 0) file_flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) directory = None descriptor = None try: directory = os.open(path.parent, directory_flags) observed = os.stat( path.name, dir_fd=directory, follow_symlinks=False, ) if not stat.S_ISREG(observed.st_mode): snapshot = TerminalRecoverySnapshot( None, observed, None, directory, b"", "non-regular", ) directory = None return snapshot try: descriptor = os.open(path.name, file_flags, dir_fd=directory) except OSError: snapshot = TerminalRecoverySnapshot( None, observed, None, directory, b"", "open-failed", ) directory = None return snapshot opened = os.fstat(descriptor) if opened.st_dev != observed.st_dev or opened.st_ino != observed.st_ino: os.close(descriptor) descriptor = None snapshot = TerminalRecoverySnapshot( None, observed, None, directory, b"", "identity-changed-during-open", ) directory = None return snapshot observed = opened if observed.st_nlink != 1: snapshot = TerminalRecoverySnapshot( None, observed, descriptor, directory, b"", "hardlinked", ) descriptor = None directory = None return snapshot if observed.st_size > MAX_TERMINAL_RECORD_BYTES: prefix = _read_bounded(descriptor, QUARANTINE_HASH_BYTES) snapshot = TerminalRecoverySnapshot( None, observed, descriptor, directory, prefix, "oversized", ) descriptor = None directory = None return snapshot payload = _read_bounded(descriptor, MAX_TERMINAL_RECORD_BYTES + 1) after_read = os.fstat(descriptor) unchanged = ( after_read.st_dev == observed.st_dev and after_read.st_ino == observed.st_ino and after_read.st_size == observed.st_size and after_read.st_mtime_ns == observed.st_mtime_ns and after_read.st_ctime_ns == observed.st_ctime_ns and len(payload) == observed.st_size ) document = None invalid_reason = "unstable-payload" if unchanged: try: parsed = json.loads(payload.decode("utf-8")) if isinstance(parsed, dict): document = parsed invalid_reason = None else: invalid_reason = "non-object-payload" except (UnicodeError, json.JSONDecodeError): invalid_reason = "malformed-payload" snapshot = TerminalRecoverySnapshot( document, after_read, descriptor, directory, payload[:QUARANTINE_HASH_BYTES], invalid_reason, ) descriptor = None directory = None return snapshot except OSError: return None finally: if descriptor is not None: os.close(descriptor) if directory is not None: os.close(directory) def _open_terminal_snapshot( path: Path, identity: TerminalIdentity, ) -> TerminalSnapshot | None: """Open a terminal journal only after its lexical identity is verified.""" if _terminal_identity(path) != identity: return None return _open_small_json_snapshot(path) def _load_terminal_json(path: Path, identity: TerminalIdentity) -> dict[str, Any]: """Read one small, singly-linked journal through its validated directory.""" snapshot = _open_terminal_snapshot(path, identity) if snapshot is None: return {} try: return snapshot.document finally: snapshot.close() def _load_small_json(path: Path) -> dict[str, Any]: """Read one bounded immutable evidence document without following links.""" snapshot = _open_small_json_snapshot(path) if snapshot is None: return {} try: return snapshot.document finally: snapshot.close() def _persist_candidate( state: dict[str, Any], state_file: Path, structured: dict[str, Any], *, route: Route, returncode: int, goal_turn: int, ) -> Path: """Persist every structured response before any judge can supersede it.""" sequence = max(0, int(state.get("candidate_sequence", 0) or 0)) + 1 path = _candidate_path(state_file, state.get("run_id"), sequence) while path.exists(): sequence += 1 path = _candidate_path(state_file, state.get("run_id"), sequence) atomic_json( path, { "board": state.get("board"), "task_id": state.get("task_id"), "expected_run_id": state.get("run_id"), "goal_turn": goal_turn, "provider": route.provider, "model": route.model, "effort": route.effort, "returncode": returncode, "structured": structured, "recorded_at": utc_now(), }, ) state["candidate_sequence"] = sequence state["last_candidate_file"] = str(path) atomic_json(state_file, state) return path def _write_terminal_record( state_file: Path, *, board: str, task_id: str, run_id: Any, structured: dict[str, Any], summary: str, metadata: dict[str, Any], ) -> tuple[Path, dict[str, Any]]: """Journal an accepted result before attempting the Kanban transaction.""" path = _terminal_path(state_file, run_id) record = { "board": board, "task_id": task_id, "expected_run_id": run_id, "result": json.dumps(structured, sort_keys=True), "summary": summary, "metadata": metadata, "kanban_state": "pending", "recorded_at": utc_now(), } atomic_json(path, record) return path, record def _terminal_record_valid( record: dict[str, Any], identity: TerminalIdentity | None = None, ) -> bool: """Reject malformed or non-terminal replay journals without side effects.""" if not isinstance(record, dict): return False required_strings = ("board", "task_id", "result", "summary") if not all( isinstance(record.get(key), str) and record[key] for key in required_strings ): return False if not isinstance(record.get("expected_run_id"), int): return False if not isinstance(record.get("metadata"), dict): return False try: structured = json.loads(record["result"]) except (TypeError, json.JSONDecodeError): return False if not isinstance(structured, dict): return False required = set(RESULT_SCHEMA["required"]) if set(structured) != required: return False if ( type(structured["status"]) is not str or structured["status"] not in cli_lane_goal.RESULT_STATUSES or type(structured["summary"]) is not str or not structured["summary"].strip() ): return False for key in ("changed_files", "tests_run", "artifacts", "findings", "blockers"): value = structured[key] if type(value) is not list or any(type(item) is not str for item in value): return False valid = ( structured["status"] == "completed" and structured["blockers"] == [] and cli_lane_goal.unfinished_result_reason(structured) is None and record["summary"] == structured["summary"] and record.get("kanban_state") in {"pending", "prepared", "committed"} ) if not valid or identity is None: return valid return ( record.get("board") == identity.board and record.get("task_id") == identity.task_id and record.get("expected_run_id") == identity.run_id and record.get("kanban_state") == identity.state ) def build_prompt(context: str, workspace: Path, handoff: str = "") -> str: """Create a bounded worker contract with an explicit machine-readable result.""" return f"""You are a durable coding worker managed by Hermes Kanban. Work only on this objective and its acceptance criteria: {context} Workspace: {workspace} {handoff} Operate autonomously inside the workspace. Inspect before editing, preserve unrelated user changes, run proportionate tests, and do not claim completion without evidence. You have owner-level Kubernetes access in every namespace. Prefer Flux-tracked manifests for durable changes, but use kubectl, Flux, exec, port-forwarding, rollout operations, and existing Vault workflows when the objective or incident requires them. Persist any desired-state mutation back to Git. Do not force-push, hard-reset, clean untracked files, or expose credentials. Return a final JSON object matching the supplied schema. Use status=incomplete when required work, tests, commands, commits, pushes, or verification are still running or remain to be done. Use status=blocked only when an obstacle prevents completion of the assigned task itself. Never use status=completed for a progress report. For review or diagnostic tasks, put defects and risks in findings; those findings can make the reviewed change unfit to ship without blocking completion of the review. The blockers array must be empty whenever status is completed. List changed files, tests run, durable artifact paths, findings, and task blockers explicitly. """ def git_handoff(workspace: Path, prior_output: str) -> str: """Build an explicit cross-provider handoff without transferring hidden state.""" def read_git(*args: str) -> str: completed = subprocess.run( ["git", "-C", str(workspace), *args], text=True, capture_output=True, timeout=30, check=False, ) return (completed.stdout or completed.stderr).strip()[-12000:] return ( "\nCross-provider handoff from a failed or exhausted worker:\n" f"Git status:\n{read_git('status', '--short', '--branch')}\n\n" f"Current diff summary:\n{read_git('diff', '--stat')}\n\n" f"Prior worker tail:\n{prior_output[-10000:]}\n" "Reinspect the workspace and verify all inherited claims before continuing." ) def workspace_artifacts(workspace: Path, values: Any) -> list[str]: """Return existing regular artifacts contained by the task worktree.""" if not isinstance(values, list): return [] root = workspace.resolve() artifacts: list[str] = [] for value in values: if not isinstance(value, str) or not value.strip(): continue candidate = Path(value.strip()).expanduser() if not candidate.is_absolute(): candidate = root / candidate try: resolved = candidate.resolve(strict=True) resolved.relative_to(root) except (OSError, RuntimeError, ValueError): continue if resolved.is_file(): artifacts.append(str(resolved)) return list(dict.fromkeys(artifacts)) def _extract_json(value: Any) -> dict[str, Any] | None: if isinstance(value, dict) and value.get("status") in cli_lane_goal.RESULT_STATUSES: return value if not isinstance(value, str): return None candidates = [value] match = re.search(r"\{.*\}", value, re.DOTALL) if match: candidates.append(match.group(0)) for candidate in candidates: try: parsed = json.loads(candidate) except json.JSONDecodeError: continue if isinstance(parsed, dict) and parsed.get("status") in cli_lane_goal.RESULT_STATUSES: return parsed return None def _event_payload(provider: str, line: str, state: dict[str, Any], state_file: Path) -> dict[str, Any] | None: """Persist provider session identifiers before interpreting final output.""" try: event = json.loads(line) except json.JSONDecodeError: return None if not isinstance(event, dict): return None if provider == "codex" and event.get("type") == "thread.started": thread_id = event.get("thread_id") or event.get("thread", {}).get("id") if thread_id: state["codex_thread_id"] = str(thread_id) atomic_json(state_file, state) if provider == "claude" and event.get("session_id"): state["claude_session_id"] = str(event["session_id"]) atomic_json(state_file, state) for key in ("structured_output", "result", "output", "text"): parsed = _extract_json(event.get(key)) if parsed: return parsed item = event.get("item") if isinstance(item, dict): for key in ("text", "content"): parsed = _extract_json(item.get(key)) if parsed: return parsed return None def stream_process( command: list[str], *, provider: str, cwd: Path, env: dict[str, str], log_path: Path, state: dict[str, Any], state_file: Path, heartbeat: Callable[[str], bool], max_runtime: int, ) -> ProcessResult: """Stream JSONL to Kanban logs while maintaining the authoritative lease.""" process = subprocess.Popen( command, cwd=cwd, env=env, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, bufsize=1, start_new_session=True, ) assert process.stdout is not None selector = selectors.DefaultSelector() selector.register(process.stdout, selectors.EVENT_READ) started = time.monotonic() last_heartbeat = 0.0 known_descendants: dict[int, tuple[int, int]] = {} lines: list[str] = [] structured: dict[str, Any] | None = None log_path.parent.mkdir(parents=True, exist_ok=True) forced_failure = "" with log_path.open("a", encoding="utf-8") as log: log.write(f"\n[{utc_now()}] starting {provider} worker\n") log.flush() while process.poll() is None: known_descendants.update(_descendant_processes(process.pid)) now = time.monotonic() if now - started > max_runtime: _terminate_worker_process(process, known_descendants) forced_failure = "worker exceeded its maximum runtime" lines.append(forced_failure + "\n") break if now - last_heartbeat >= HEARTBEAT_SECONDS: if not heartbeat(f"{provider} worker active for {round(now - started)}s"): _terminate_worker_process(process, known_descendants) forced_failure = "Kanban lease was lost; provider process terminated" lines.append(forced_failure + "\n") break last_heartbeat = now for key, _ in selector.select(timeout=1.0): line = key.fileobj.readline() if not line: continue lines.append(line) if len(lines) > 4000: lines = lines[-4000:] log.write(line) log.flush() parsed = _event_payload(provider, line, state, state_file) structured = parsed or structured _terminate_worker_process(process, known_descendants) remainder = process.stdout.read() if remainder: lines.append(remainder) log.write(remainder) for line in remainder.splitlines(): parsed = _event_payload(provider, line, state, state_file) structured = parsed or structured if forced_failure: log.write(forced_failure + "\n") log.write(f"\n[{utc_now()}] {provider} exit={process.returncode}\n") selector.close() output = "".join(lines)[-100000:] returncode = int(process.returncode if process.returncode is not None else 1) return ProcessResult( returncode=returncode, output=output, structured=structured, capacity_failure=returncode != 0 and bool(CAPACITY_PATTERN.search(output)), ) def _process_record(pid: int) -> tuple[int, int, int] | None: """Return one Linux process's parent, group, and start-time identity.""" try: raw = Path(f"/proc/{pid}/stat").read_text(encoding="utf-8") fields = raw[raw.rfind(")") + 2 :].split() return int(fields[1]), int(fields[2]), int(fields[19]) except (IndexError, OSError, ValueError): return None def _descendant_processes(root_pid: int) -> dict[int, tuple[int, int]]: """Snapshot descendants as pid -> (process group, start-time identity).""" records: dict[int, tuple[int, int, int]] = {} for entry in Path("/proc").iterdir(): if not entry.name.isdigit(): continue pid = int(entry.name) record = _process_record(pid) if record is not None: records[pid] = record family = {root_pid} changed = True while changed: changed = False for pid, (parent, _, _) in records.items(): if pid not in family and parent in family: family.add(pid) changed = True return { pid: (records[pid][1], records[pid][2]) for pid in family if pid != root_pid and pid in records } def _process_identity_matches(pid: int, start_time: int) -> bool: record = _process_record(pid) return record is not None and record[2] == start_time def _signal_worker_tree( root_pid: int, descendants: dict[int, tuple[int, int]], sig: signal.Signals, ) -> None: """Signal captured descendants across terminal-created process groups.""" groups = {root_pid} for pid, (process_group, start_time) in descendants.items(): if _process_identity_matches(pid, start_time): groups.add(process_group) for process_group in groups: try: os.killpg(process_group, sig) except ProcessLookupError: pass for pid, (_, start_time) in descendants.items(): if not _process_identity_matches(pid, start_time): continue try: os.kill(pid, sig) except ProcessLookupError: pass def _terminate_worker_process( process: subprocess.Popen[str], known_descendants: dict[int, tuple[int, int]] | None = None, ) -> None: """Reap a worker and terminal descendants that created their own groups.""" descendants = dict(known_descendants or {}) descendants.update(_descendant_processes(process.pid)) if process.poll() is None: _signal_worker_tree(process.pid, descendants, signal.SIGTERM) try: process.wait(timeout=10) except subprocess.TimeoutExpired: pass # Provider terminal tools create their own sessions, so killing only the # native CLI's process group leaves test/build subprocesses orphaned. The # identity check prevents a reused PID from targeting an unrelated worker. _signal_worker_tree(process.pid, descendants, signal.SIGKILL) if process.poll() is None: process.wait(timeout=10) def _base_env() -> dict[str, str]: """Build a worker environment while preserving Vault-backed CLI homes.""" env = os.environ.copy() env.update( { "HOME": str(DATA_ROOT / "home"), "CODEX_HOME": os.environ.get("CODEX_HOME", "/runtime-access/codex"), "CLAUDE_CONFIG_DIR": os.environ.get( "CLAUDE_CONFIG_DIR", "/runtime-access/claude" ), "GIT_TERMINAL_PROMPT": "0", } ) env["PATH"] = f"{DATA_ROOT / 'tools/bin'}:/opt/coordinator:" + env.get("PATH", "") return env def _codex_command(route: Route, prompt: str, workspace: Path, state: dict[str, Any], result_file: Path) -> list[str]: base = [ str(CODEX_BIN), "--dangerously-bypass-approvals-and-sandbox", "--dangerously-bypass-hook-trust", "exec", ] thread_id = str(state.get("codex_thread_id") or "") common = [ "--json", "-m", route.model, "-c", f'model_reasoning_effort="{route.effort}"', "--output-schema", str(RESULT_SCHEMA_PATH), "-o", str(result_file), ] if thread_id: return [*base, "resume", *common, thread_id, prompt] return [ *base, *common, "-C", str(workspace), prompt, ] def _claude_command(route: Route, prompt: str, state: dict[str, Any], resume: bool) -> list[str]: session_id = str(state["claude_session_id"]) session = ["--resume", session_id] if resume else ["--session-id", session_id] denied = [ "Bash(git push --force *)", "Bash(git reset --hard *)", "Bash(git clean -f *)", ] return [ str(CLAUDE_BIN), "--dangerously-skip-permissions", "--autocompact", "auto", "--settings", str(CLAUDE_SETTINGS), "--disallowedTools", *denied, "--model", route.model, "--effort", route.effort, "--output-format", "stream-json", "--verbose", "--json-schema", json.dumps(RESULT_SCHEMA, separators=(",", ":")), *session, "-p", prompt, ] def run_provider( route: Route, prompt: str, workspace: Path, state: dict[str, Any], state_file: Path, log_path: Path, heartbeat: Callable[[str], bool], max_runtime: int, ) -> ProcessResult: """Run or resume one provider session using its pinned structured CLI.""" state.setdefault("attempts", []).append({"route": asdict(route), "started_at": utc_now()}) if route.provider == "claude" and not state.get("claude_session_id"): # The reservation is durable before Claude starts, closing the crash gap. state["claude_session_id"] = str(uuid.uuid4()) state["current_route"] = asdict(route) state["updated_at"] = utc_now() env = _base_env() if route.provider == "codex": result_sequence = max(0, int(state.get("result_sequence", 0) or 0)) + 1 result_file = _result_path(state_file, state.get("run_id"), result_sequence) while result_file.exists(): result_sequence += 1 result_file = _result_path( state_file, state.get("run_id"), result_sequence, ) result_file.parent.mkdir(parents=True, exist_ok=True) result_file.touch(mode=0o600, exist_ok=False) state["result_sequence"] = result_sequence state["current_result_file"] = str(result_file) atomic_json(state_file, state) command = _codex_command(route, prompt, workspace, state, result_file) result = stream_process( command, provider="codex", cwd=workspace, env=env, log_path=log_path, state=state, state_file=state_file, heartbeat=heartbeat, max_runtime=max_runtime, ) if ( state.get("codex_thread_id") and result.returncode != 0 and NO_CODEX_THREAD in result.output.lower() ): state.pop("codex_thread_id", None) result_sequence += 1 result_file = _result_path( state_file, state.get("run_id"), result_sequence, ) while result_file.exists(): result_sequence += 1 result_file = _result_path( state_file, state.get("run_id"), result_sequence, ) result_file.touch(mode=0o600, exist_ok=False) state["result_sequence"] = result_sequence state["current_result_file"] = str(result_file) atomic_json(state_file, state) result = stream_process( _codex_command(route, prompt, workspace, state, result_file), provider="codex", cwd=workspace, env=env, log_path=log_path, state=state, state_file=state_file, heartbeat=heartbeat, max_runtime=max_runtime, ) file_result = load_json(result_file) if file_result.get("status") in cli_lane_goal.RESULT_STATUSES: result.structured = file_result try: result_file.chmod(0o600) except OSError: pass return result atomic_json(state_file, state) resume = bool(state.get("claude_started")) result = stream_process( _claude_command(route, prompt, state, resume), provider="claude", cwd=workspace, env=env, log_path=log_path, state=state, state_file=state_file, heartbeat=heartbeat, max_runtime=max_runtime, ) if resume and result.returncode == 1 and NO_CLAUDE_SESSION in result.output: result = stream_process( _claude_command(route, prompt, state, False), provider="claude", cwd=workspace, env=env, log_path=log_path, state=state, state_file=state_file, heartbeat=heartbeat, max_runtime=max_runtime, ) elif not resume and result.returncode == 1 and CLAUDE_SESSION_COLLISION in result.output: result = stream_process( _claude_command(route, prompt, state, True), provider="claude", cwd=workspace, env=env, log_path=log_path, state=state, state_file=state_file, heartbeat=heartbeat, max_runtime=max_runtime, ) if result.returncode == 0 or NO_CLAUDE_SESSION not in result.output: state["claude_started"] = True atomic_json(state_file, state) return result def _task_value(task: Any, name: str, default: Any = None) -> Any: return getattr(task, name, default) def _task_context(kanban_db: Any, conn: Any, task_id: str) -> str: value = kanban_db.build_worker_context(conn, task_id) if isinstance(value, str): return value return json.dumps(value, indent=2, default=str) def _resolve_workspace(kanban_db: Any, conn: Any, task: Any, board: str) -> Path: # External coding lanes always receive their own linked worktree. A task # that does not resolve to a Git repository is blocked rather than sharing # a mutable checkout with another unattended worker. # Git worktree creation updates common-repository metadata. Serialize that # short materialization step while allowing the provider workers themselves # to run concurrently in independent worktrees. with WORKTREE_LOCK: value, branch_name = kanban_db._resolve_worktree_workspace(task, board=board) kanban_db.set_branch_name(conn, str(_task_value(task, "id")), branch_name) return Path(value).resolve() def _board_call( kanban_db: Any, board: str, operation: Callable[[Any], Any], ) -> Any: """Run one bounded Kanban operation on a fresh, promptly closed connection.""" last_storage_error: Exception | None = None for attempt in range(KANBAN_STORAGE_ATTEMPTS): conn = None try: with kanban_db.scoped_current_board(board): conn = kanban_db.connect(board=board) result = operation(conn) BOARD_CORRUPTION_ERRORS.pop(board, None) return result except (OSError, sqlite3.Error) as error: last_storage_error = error _record_board_access_error(board, error) if attempt + 1 < KANBAN_STORAGE_ATTEMPTS: time.sleep(0.2 * (attempt + 1)) finally: if conn is not None: conn.close() assert last_storage_error is not None raise last_storage_error def _rename_noreplace( source: str, destination: str, *, source_dir: int, destination_dir: int, ) -> None: """Atomically rename a directory entry without replacing a collision.""" renameat2 = getattr(ctypes.CDLL(None, use_errno=True), "renameat2", None) if renameat2 is None: raise OSError(errno.ENOSYS, "renameat2 is unavailable") renameat2.argtypes = ( ctypes.c_int, ctypes.c_char_p, ctypes.c_int, ctypes.c_char_p, ctypes.c_uint, ) renameat2.restype = ctypes.c_int result = renameat2( source_dir, ctypes.c_char_p(os.fsencode(source)), destination_dir, ctypes.c_char_p(os.fsencode(destination)), 1, # RENAME_NOREPLACE ) if result != 0: error = ctypes.get_errno() raise OSError(error, os.strerror(error), source, destination) def _retire_terminal_entry( path: Path, source_stat: os.stat_result, *, board_descriptor: int, quarantine_descriptor: int, authority_name: str | None = None, ) -> str: """Retire only the inode previously inspected; preserve any replacement.""" try: current = os.stat( path.name, dir_fd=board_descriptor, follow_symlinks=False, ) except FileNotFoundError: return "missing" if current.st_dev != source_stat.st_dev or current.st_ino != source_stat.st_ino: return "replacement" # A replacement can itself be staged repeatedly while recovery races a # writer. Keep every generation bound to the original canonical pending # name instead of hashing an intermediate .retire name and hiding it from # the next recovery pass. path_digest = hashlib.sha256( (authority_name or path.name).encode("utf-8") ).hexdigest()[:16] source_digest = hashlib.sha256( f"{source_stat.st_dev:x}\0{source_stat.st_ino:x}".encode("utf-8") ).hexdigest()[:16] for sequence in range(32): staging = f".retire.{path_digest}.{source_digest}.{sequence}" try: _rename_noreplace( path.name, staging, source_dir=board_descriptor, destination_dir=quarantine_descriptor, ) except FileExistsError: continue break else: return "collision" staged = os.stat( staging, dir_fd=quarantine_descriptor, follow_symlinks=False, ) if staged.st_dev == source_stat.st_dev and staged.st_ino == source_stat.st_ino: os.unlink(staging, dir_fd=quarantine_descriptor) os.fsync(quarantine_descriptor) os.fsync(board_descriptor) return "retired" # The path changed after the pre-check but before the rename. Put the # replacement back without overwriting an even newer entry. try: _rename_noreplace( staging, path.name, source_dir=quarantine_descriptor, destination_dir=board_descriptor, ) except FileExistsError: return "replacement-staged" os.fsync(quarantine_descriptor) os.fsync(board_descriptor) return "replacement" def _write_json_noreplace(path: Path, value: dict[str, Any]) -> bool: """Durably create one immutable JSON artifact without replacing a peer.""" payload = (json.dumps(value, indent=2, sort_keys=True) + "\n").encode("utf-8") if len(payload) > MAX_TERMINAL_RECORD_BYTES + 4096: raise ValueError("terminal evidence exceeds the bounded artifact limit") path.parent.mkdir(parents=True, exist_ok=True, mode=0o700) os.chmod(path.parent, 0o700, follow_symlinks=False) directory_flags = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) directory_flags |= getattr(os, "O_NOFOLLOW", 0) create_flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL create_flags |= getattr(os, "O_NOFOLLOW", 0) directory = os.open(path.parent, directory_flags) temporary = f".{path.name}.{uuid.uuid4().hex}.tmp" temporary_stat = None descriptor = None try: descriptor = os.open(temporary, create_flags, 0o600, dir_fd=directory) temporary_stat = os.fstat(descriptor) with os.fdopen(descriptor, "wb") as stream: descriptor = None stream.write(payload) stream.flush() os.fsync(stream.fileno()) written = os.fstat(stream.fileno()) if stat.S_IMODE(written.st_mode) != 0o600 or written.st_nlink != 1: raise OSError("terminal evidence is not private and singly linked") try: _rename_noreplace( temporary, path.name, source_dir=directory, destination_dir=directory, ) except FileExistsError: return False os.fsync(directory) return True finally: if descriptor is not None: os.close(descriptor) if temporary_stat is not None: try: _retire_terminal_entry( Path(temporary), temporary_stat, board_descriptor=directory, quarantine_descriptor=directory, ) except OSError: pass os.close(directory) def _terminal_document_digest(document: dict[str, Any]) -> str: """Return the stable identity for one exact accepted result document.""" immutable = { key: document.get(key) for key in ( "board", "task_id", "expected_run_id", "result", "summary", "metadata", ) } canonical = json.dumps(immutable, sort_keys=True, separators=(",", ":")) return hashlib.sha256(canonical.encode("utf-8")).hexdigest()[:32] def _terminal_evidence_path( identity: TerminalIdentity, state: str, document: dict[str, Any], ) -> Path: """Return a deterministic immutable prepared or conflict evidence path.""" if state not in {"prepared", "conflict"}: raise ValueError(f"invalid terminal evidence state: {state}") base = state_path(identity.board, identity.task_id) digest = _terminal_document_digest(document) return base.with_name( f"{base.stem}.run-{identity.run_id}.terminal.{state}-{digest}.json" ) def _same_terminal_document(left: dict[str, Any], right: dict[str, Any]) -> bool: """Compare the immutable result-bearing fields of terminal evidence.""" keys = ("board", "task_id", "expected_run_id", "result", "summary", "metadata") return all(left.get(key) == right.get(key) for key in keys) def _terminal_evidence_valid( document: dict[str, Any], identity: TerminalIdentity, state: str, ) -> bool: """Validate semantic terminal content plus its immutable evidence state.""" if document.get("kanban_state") != state: return False semantic = dict(document) semantic_identity = identity if state == "conflict": semantic["kanban_state"] = "prepared" semantic_identity = TerminalIdentity( identity.board, identity.task_id, identity.run_id, "prepared", ) return _terminal_record_valid(semantic, semantic_identity) def _persist_prepared_evidence( identity: TerminalIdentity, document: dict[str, Any], ) -> Path: """Fsync an immutable accepted result before entering the DB boundary.""" prepared = dict(document) prepared["kanban_state"] = "prepared" prepared["prepared_at"] = utc_now() path = _terminal_evidence_path(identity, "prepared", document) if _write_json_noreplace(path, prepared): return path if _terminal_evidence_identity(path) != TerminalIdentity( identity.board, identity.task_id, identity.run_id, "prepared" ): raise OSError("prepared terminal evidence identity is invalid") existing = _load_small_json(path) prepared_identity = TerminalIdentity( identity.board, identity.task_id, identity.run_id, "prepared", ) if not _terminal_evidence_valid( existing, prepared_identity, "prepared", ) or not _same_terminal_document(existing, prepared): raise OSError("prepared terminal evidence path contains a conflicting result") return path def _persist_conflict_evidence( identity: TerminalIdentity, document: dict[str, Any], reason: str, ) -> Path: """Preserve a losing valid result in full under deterministic retention.""" conflict = dict(document) conflict["kanban_state"] = "conflict" conflict["conflict_reason"] = reason conflict["conflicted_at"] = utc_now() path = _terminal_evidence_path(identity, "conflict", document) if _write_json_noreplace(path, conflict): return path if _terminal_evidence_identity(path) != TerminalIdentity( identity.board, identity.task_id, identity.run_id, "conflict" ): raise OSError("terminal conflict evidence identity is invalid") existing = _load_small_json(path) conflict_identity = TerminalIdentity( identity.board, identity.task_id, identity.run_id, "conflict", ) if not _terminal_evidence_valid( existing, conflict_identity, "conflict", ) or not _same_terminal_document(existing, conflict): raise OSError("terminal conflict evidence path contains a different result") return path def _quarantine_terminal( path: Path, identity: TerminalIdentity | None, reason: str, *, snapshot: TerminalRecoverySnapshot | None = None, ) -> Path: """Record bounded metadata, then retire only the inspected journal inode.""" board_dir = path.parent try: board_stat = board_dir.stat(follow_symlinks=False) except OSError: board_stat = None if board_stat is None or not stat.S_ISDIR(board_stat.st_mode): print( f"refused terminal journal path through unsafe board directory: {path}", file=sys.stderr, flush=True, ) return path board_flags = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) board_flags |= getattr(os, "O_NOFOLLOW", 0) file_flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0) create_flags = os.O_WRONLY | os.O_CREAT | os.O_EXCL create_flags |= getattr(os, "O_NOFOLLOW", 0) board_descriptor = None quarantine_descriptor = None source_descriptor = None source_stat = None destination = path retirement = "not-attempted" owns_source = snapshot is None try: if snapshot is None: board_descriptor = os.open(board_dir, board_flags) source_stat = os.stat( path.name, dir_fd=board_descriptor, follow_symlinks=False, ) else: board_descriptor = snapshot.directory_descriptor source_descriptor = snapshot.descriptor source_stat = snapshot.file_stat pinned_board = os.fstat(board_descriptor) if ( pinned_board.st_dev != board_stat.st_dev or pinned_board.st_ino != board_stat.st_ino ): raise OSError("terminal recovery board identity changed") safe_source = stat.S_ISREG(source_stat.st_mode) and source_stat.st_nlink == 1 digest = None hashed_bytes = 0 hash_complete = False if safe_source and snapshot is None: source_descriptor = os.open( path.name, file_flags, dir_fd=board_descriptor, ) opened_stat = os.fstat(source_descriptor) if ( opened_stat.st_dev != source_stat.st_dev or opened_stat.st_ino != source_stat.st_ino or opened_stat.st_nlink != 1 ): raise OSError("terminal journal identity changed while opening") # Keep this descriptor open through retirement. The staging entry # must match the exact fstat-verified inode, not merely a path stat # captured before a concurrent replacement. source_stat = opened_stat prefix = _read_bounded(source_descriptor, QUARANTINE_HASH_BYTES) hashed_bytes = len(prefix) hash_complete = opened_stat.st_size <= QUARANTINE_HASH_BYTES digest = hashlib.sha256(prefix).hexdigest() elif safe_source and source_descriptor is not None: prefix = snapshot.prefix hashed_bytes = len(prefix) hash_complete = ( source_stat.st_size <= QUARANTINE_HASH_BYTES and hashed_bytes == source_stat.st_size ) digest = hashlib.sha256(prefix).hexdigest() source_kind = ( "regular" if safe_source else "hardlink" if stat.S_ISREG(source_stat.st_mode) and source_stat.st_nlink != 1 else "non-regular" ) diagnostic = { "hash_complete": hash_complete, "hashed_bytes": hashed_bytes, "original_name": path.name, "reason": reason, "sha256": digest, "size": source_stat.st_size, "source_kind": source_kind, } payload = (json.dumps(diagnostic, sort_keys=True) + "\n").encode("utf-8") fingerprint = ( (digest or hashlib.sha256(path.name.encode("utf-8")).hexdigest())[:16] ) safe_reason = ( re.sub(r"[^a-zA-Z0-9_.-]+", "-", reason).strip("-") or "invalid" ) try: os.mkdir("quarantine", 0o700, dir_fd=board_descriptor) except FileExistsError: pass quarantine_descriptor = os.open( "quarantine", board_flags, dir_fd=board_descriptor, ) for sequence in range(32): name = f"{path.name}.{safe_reason}.{fingerprint}.{sequence}.quarantine" try: destination_descriptor = os.open( name, create_flags, 0o600, dir_fd=quarantine_descriptor, ) except FileExistsError: continue try: with os.fdopen(destination_descriptor, "wb") as target: target.write(payload) target.flush() os.fsync(target.fileno()) target_stat = os.fstat(target.fileno()) if ( stat.S_IMODE(target_stat.st_mode) != 0o600 or target_stat.st_nlink != 1 ): raise OSError("quarantine destination is not private and singly linked") except Exception: os.unlink(name, dir_fd=quarantine_descriptor) raise destination = board_dir / "quarantine" / name break else: raise OSError("could not reserve a unique quarantine destination") os.fsync(quarantine_descriptor) retirement = _retire_terminal_entry( path, source_stat, board_descriptor=board_descriptor, quarantine_descriptor=quarantine_descriptor, ) except OSError as error: # A bad quarantine target must not make an attacker-controlled pending # path eligible for replay forever. The same identity-safe staging # protocol works in the board directory when quarantine is unusable. if board_descriptor is not None and source_stat is not None: try: retirement = _retire_terminal_entry( path, source_stat, board_descriptor=board_descriptor, quarantine_descriptor=( quarantine_descriptor if quarantine_descriptor is not None else board_descriptor ), ) except OSError: retirement = "deferred" print( f"terminal journal quarantine degraded safely: {type(error).__name__}: {error}", file=sys.stderr, flush=True, ) finally: if owns_source and source_descriptor is not None: os.close(source_descriptor) if quarantine_descriptor is not None: os.close(quarantine_descriptor) if owns_source and board_descriptor is not None: os.close(board_descriptor) print( "quarantined terminal journal " f"{path.name}: {reason}; identity={identity or 'unparseable'}; " f"retirement={retirement}", file=sys.stderr, flush=True, ) return destination def _recover_exact_run( kanban_db: Any, identity: TerminalIdentity | None, reason: str, ) -> bool: """Make an exact external run retryable after journal recovery fails.""" if identity is None: return False def operation(conn: Any) -> bool: task = kanban_db.get_task(conn, identity.task_id) if task is None: return False if ( str(_task_value(task, "status", "")) != "running" or _task_value(task, "current_run_id", None) != identity.run_id or not _external(task) ): return False return bool( kanban_db.reclaim_task( conn, identity.task_id, reason=f"terminal journal recovery failed ({reason}); exact run may retry", expected_run_id=identity.run_id, ) ) try: recovered = bool(_board_call(kanban_db, identity.board, operation)) except Exception as error: _record_board_access_error(identity.board, error) return False if recovered: print( f"reclaimed {identity.board}/{identity.task_id} run {identity.run_id} " f"after terminal journal recovery: {reason}", file=sys.stderr, flush=True, ) return recovered def _retire_snapshot( path: Path, snapshot: TerminalSnapshot, *, authority_name: str | None = None, ) -> str: """Retire only the directory entry still naming an opened snapshot inode.""" return _retire_terminal_entry( path, snapshot.file_stat, board_descriptor=snapshot.directory_descriptor, quarantine_descriptor=snapshot.directory_descriptor, authority_name=authority_name, ) def _retire_snapshot_after_db(path: Path, snapshot: TerminalSnapshot) -> str: """Keep a committed DB result recoverable across retirement fsync errors.""" try: return _retire_snapshot(path, snapshot) except OSError as error: print( f"terminal journal retirement deferred after DB commit: " f"{type(error).__name__}: {error}", file=sys.stderr, flush=True, ) return "deferred" def _discard_evidence(path: Path) -> str: """Identity-safely remove one immutable prepared evidence artifact.""" snapshot = _open_small_json_snapshot(path) if snapshot is None: return "missing" try: return _retire_snapshot(path, snapshot) finally: snapshot.close() def _promote_prepared_evidence( identity: TerminalIdentity, document: dict[str, Any], prepared: Path, ) -> tuple[Path, str]: """Publish the DB-winning result without overwriting first-writer evidence.""" committed = _terminal_path( state_path(identity.board, identity.task_id), identity.run_id, "committed", ) committed_document = dict(document) committed_document["kanban_state"] = "committed" committed_document["committed_at"] = utc_now() state = "committed" if not _write_json_noreplace(committed, committed_document): committed_identity = TerminalIdentity( identity.board, identity.task_id, identity.run_id, "committed", ) existing = _load_terminal_json(committed, committed_identity) if ( not _terminal_record_valid(existing, committed_identity) or existing.get("kanban_state") != "committed" or not _same_terminal_document(existing, committed_document) ): _persist_conflict_evidence( identity, document, "canonical committed evidence already contains a different result", ) state = "conflict" _discard_evidence(prepared) return committed, state def _finalize_document_db( kanban_db: Any, identity: TerminalIdentity, document: dict[str, Any], ) -> str: """Apply one validated exact-run result to Kanban under DB run guards.""" def operation(conn: Any) -> str: task = kanban_db.get_task(conn, identity.task_id) if task is None: return "stale" status = str(_task_value(task, "status", "")) if status == "done": return ( "committed" if ( _task_value(task, "completed_run_id", None) == identity.run_id and str(_task_value(task, "result", "") or "") == document["result"] ) else "stale" ) current_run_id = _task_value(task, "current_run_id", None) completion_guard: dict[str, int] if status == "running" and current_run_id == identity.run_id: completion_guard = {"expected_run_id": identity.run_id} elif status in {"ready", "blocked"} and current_run_id is None: # The journal may have survived an older post-persistence error # path that ended its run. The patched DB verifies atomically that # this is still the latest ended run before allowing completion. completion_guard = {"replay_ended_run_id": identity.run_id} else: return "stale" completed = kanban_db.complete_task( conn, identity.task_id, result=document["result"], summary=document["summary"], metadata=document["metadata"], **completion_guard, ) if completed: return "committed" # A duplicate finalizer can lose the guarded UPDATE to an identical # first writer. Re-read rather than reporting a false pending state. latest = kanban_db.get_task(conn, identity.task_id) if latest is not None and str(_task_value(latest, "status", "")) == "done": return ( "committed" if ( _task_value(latest, "completed_run_id", None) == identity.run_id and str(_task_value(latest, "result", "") or "") == document["result"] ) else "stale" ) return "pending" return str(_board_call(kanban_db, identity.board, operation)) def _resolve_pending_after_winner( path: Path, identity: TerminalIdentity, winning_document: dict[str, Any], ) -> bool: """Retire duplicates and preserve differing replacements as conflicts.""" for _attempt in range(8): snapshot = _open_terminal_snapshot(path, identity) if snapshot is None: return False document = snapshot.document if not _terminal_record_valid(document, identity): snapshot.close() return False differing = not _same_terminal_document(document, winning_document) try: if differing: _persist_conflict_evidence( identity, document, "different valid result lost the exact-run first-writer race", ) retirement = _retire_snapshot_after_db(path, snapshot) finally: snapshot.close() if retirement in {"retired", "missing"}: if differing: print( f"preserved terminal result conflict for {identity.board}/" f"{identity.task_id} run {identity.run_id}", file=sys.stderr, flush=True, ) return differing if retirement not in {"replacement", "replacement-staged", "collision"}: return differing return False def _finalize_terminal_record( kanban_db: Any, path: Path, _record: dict[str, Any] | None = None, *, snapshot: TerminalRecoverySnapshot | None = None, ) -> str: """Commit one inode-pinned exact-run journal, or leave it safely pending.""" identity = _terminal_identity(path) if identity is None or identity.state != "pending": if snapshot is not None: snapshot.close() return "invalid" pinned: TerminalSnapshot | TerminalRecoverySnapshot | None = snapshot if pinned is None: pinned = _open_terminal_snapshot(path, identity) if pinned is None or pinned.document is None: if pinned is not None: pinned.close() return "invalid" document = pinned.document try: if not _terminal_record_valid(document): return "invalid" if not _terminal_record_valid(document, identity): return "foreign" prepared = _persist_prepared_evidence(identity, document) outcome = _finalize_document_db(kanban_db, identity, document) if outcome == "committed": _committed, evidence_state = _promote_prepared_evidence( identity, document, prepared, ) retirement = _retire_snapshot_after_db(path, pinned) elif outcome == "stale": _persist_conflict_evidence( identity, document, "valid terminal result no longer matches the authoritative task run", ) _discard_evidence(prepared) retirement = _retire_snapshot_after_db(path, pinned) outcome = "conflict" evidence_state = "conflict" else: return outcome finally: pinned.close() if outcome == "committed" and retirement not in {"retired", "missing"}: _resolve_pending_after_winner(path, identity, document) if evidence_state == "conflict": print( f"terminal first-writer conflict recorded for {identity.board}/" f"{identity.task_id} run {identity.run_id}", file=sys.stderr, flush=True, ) return outcome def _terminal_entry_absent(path: Path) -> bool: """Confirm absence through a nofollow directory descriptor.""" directory_flags = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) directory_flags |= getattr(os, "O_NOFOLLOW", 0) directory = None try: directory = os.open(path.parent, directory_flags) os.stat(path.name, dir_fd=directory, follow_symlinks=False) return False except FileNotFoundError: return True except OSError: return False finally: if directory is not None: os.close(directory) def _staged_terminal_authority( path: Path, snapshot: TerminalRecoverySnapshot, ) -> tuple[TerminalIdentity, Path] | None: """Bind a staged inode to the canonical pending name it was moved from.""" try: relative = path.relative_to(STATE_ROOT) except ValueError: return None if len(relative.parts) != 2: return None board, filename = relative.parts match = _RETIRE_NAME.fullmatch(filename) legacy = _LEGACY_RETIRE_NAME.fullmatch(filename) document = snapshot.document if ( (match is None and legacy is None) or document is None or not _SAFE_BOARD.fullmatch(board) or board in {".", ".."} or document.get("board") != board or not isinstance(document.get("task_id"), str) or not isinstance(document.get("expected_run_id"), int) ): return None identity = TerminalIdentity( board, document["task_id"], document["expected_run_id"], "pending", ) canonical = _terminal_path( state_path(identity.board, identity.task_id), identity.run_id, "pending", ) if ( canonical.parent != path.parent or ( match is not None and hashlib.sha256(canonical.name.encode("utf-8")).hexdigest()[:16] != match.group("path_digest") ) or not _terminal_record_valid(document, identity) ): return None return identity, canonical def _recover_retirement_staging() -> int: """Promote valid hidden journals to durable prepared evidence.""" recovered = 0 for path in sorted(STATE_ROOT.glob("*/.retire.*")): snapshot = _open_terminal_recovery_snapshot(path) if snapshot is None: continue authority = _staged_terminal_authority(path, snapshot) if authority is None: try: _quarantine_terminal( path, None, "untrusted-retirement-staging", snapshot=snapshot, ) finally: snapshot.close() continue identity, canonical = authority try: _persist_prepared_evidence(identity, snapshot.document or {}) retirement = _retire_snapshot( path, snapshot, authority_name=canonical.name, ) except Exception as error: _record_board_access_error(identity.board, error) continue finally: snapshot.close() if retirement in {"retired", "missing"}: recovered += 1 return recovered def _recover_prepared_finalizations(kanban_db: Any) -> int: """Recover an accepted result staged before an interrupted DB boundary.""" recovered = 0 prepared_paths = list(STATE_ROOT.glob("*/*.terminal.prepared-*.json")) def durable_order(path: Path) -> tuple[int, str]: try: return path.stat(follow_symlinks=False).st_mtime_ns, str(path) except OSError: return 2**63 - 1, str(path) # When no DB winner exists yet, the first durably prepared result owns the # run. The DB transaction remains the final arbiter across runner processes. for path in sorted(prepared_paths, key=durable_order): identity = _terminal_evidence_identity(path) snapshot = _open_small_json_snapshot(path) if identity is None or identity.state != "prepared" or snapshot is None: if snapshot is not None: snapshot.close() _quarantine_terminal(path, identity, "malformed-prepared-evidence") continue document = snapshot.document try: expected = _terminal_evidence_path(identity, "prepared", document) if ( expected.name != path.name or not _terminal_evidence_valid(document, identity, "prepared") ): _quarantine_terminal(path, identity, "foreign-prepared-evidence") continue try: outcome = _finalize_document_db(kanban_db, identity, document) except Exception as error: _record_board_access_error(identity.board, error) continue try: if outcome == "committed": _promote_prepared_evidence(identity, document, path) pending = _terminal_path( state_path(identity.board, identity.task_id), identity.run_id, "pending", ) pending_identity = TerminalIdentity( identity.board, identity.task_id, identity.run_id, "pending", ) _resolve_pending_after_winner( pending, pending_identity, document, ) recovered += 1 elif outcome == "stale": _persist_conflict_evidence( identity, document, "prepared result no longer matches the authoritative task run", ) _discard_evidence(path) except Exception as error: _record_board_access_error(identity.board, error) continue finally: snapshot.close() return recovered def recover_pending_finalizations() -> int: """Replay accepted exact-run results before scheduling more provider work.""" from hermes_cli import kanban_db _recover_retirement_staging() recovered = _recover_prepared_finalizations(kanban_db) for path in sorted(STATE_ROOT.glob("*/*.terminal.pending.json")): identity = _terminal_identity(path) last_invalid_reason = "malformed-name" if identity is None else None retired_invalid = False for _replacement_attempt in range(16): snapshot = _open_terminal_recovery_snapshot(path) if snapshot is None: if ( identity is not None and retired_invalid and _terminal_entry_absent(path) ): _recover_exact_run( kanban_db, identity, last_invalid_reason or "malformed-payload", ) break record = snapshot.document or {} reason = None if identity is None: reason = "malformed-name" elif not _terminal_record_valid(record): reason = snapshot.invalid_reason or "malformed-payload" elif not _terminal_record_valid(record, identity): reason = "foreign-identity" if reason is not None: try: _quarantine_terminal( path, identity, reason, snapshot=snapshot, ) finally: snapshot.close() retired_invalid = True last_invalid_reason = reason continue assert identity is not None try: outcome = _finalize_terminal_record( kanban_db, path, record, snapshot=snapshot, ) except Exception as error: _record_board_access_error(identity.board, error) break if outcome == "committed": recovered += 1 break if outcome in {"invalid", "foreign"}: # The pathname changed after the recovery snapshot closed. # Reclassify and pin the new inode instead of quarantining it # using authority derived from the older entry. last_invalid_reason = outcome continue break else: print( f"terminal journal replacement churn deferred for {path.name}", file=sys.stderr, flush=True, ) return recovered def _has_pending_finalization(board: str, task_id: str, run_id: Any) -> bool: """Keep an exact run claimed while its accepted result awaits replay.""" if not isinstance(run_id, int): return False path = _terminal_path(state_path(board, task_id), run_id, "pending") try: path.stat() except FileNotFoundError: pass except OSError: return False else: identity = _terminal_identity(path) if identity is not None: record = _load_terminal_json(path, identity) if _terminal_record_valid(record, identity): return True base = state_path(board, task_id) for prepared in base.parent.glob( f"{base.stem}.run-{run_id}.terminal.prepared-*.json" ): identity = _terminal_evidence_identity(prepared) if identity is None or identity.state != "prepared": continue record = _load_small_json(prepared) if _terminal_evidence_valid(record, identity, "prepared"): return True for staged in base.parent.glob(".retire.*"): snapshot = _open_terminal_recovery_snapshot(staged) if snapshot is None: continue try: authority = _staged_terminal_authority(staged, snapshot) if authority is not None and authority[0] == TerminalIdentity( board, task_id, run_id, "pending", ): return True finally: snapshot.close() return False def _artifact_gc_candidates(board_dir: Path) -> list[Path]: """List only bounded artifact classes; pending journals are excluded.""" candidates: list[Path] = [] patterns = ( "*.provider-*.result.json", "*.candidate-*.json", "*.terminal.committed.json", "*.terminal.conflict-*.json", ".retire.*", ) for pattern in patterns: candidates.extend(board_dir.glob(pattern)) quarantine = board_dir / "quarantine" if quarantine.is_dir(): candidates.extend(quarantine.glob("*.quarantine")) return candidates def _unlink_artifact_if_same(path: Path, observed: os.stat_result) -> bool: """Remove only the exact artifact inode selected by retention.""" directory_flags = os.O_RDONLY | getattr(os, "O_DIRECTORY", 0) directory_flags |= getattr(os, "O_NOFOLLOW", 0) descriptor = None try: descriptor = os.open(path.parent, directory_flags) outcome = _retire_terminal_entry( path, observed, board_descriptor=descriptor, quarantine_descriptor=descriptor, ) return outcome == "retired" except OSError: return False finally: if descriptor is not None: os.close(descriptor) def gc_lane_artifacts( *, now: float | None = None, max_age_seconds: int = ARTIFACT_RETENTION_AGE_SECONDS, max_count: int = ARTIFACT_RETENTION_COUNT, max_bytes: int = ARTIFACT_RETENTION_BYTES, ) -> int: """Bound non-pending lane artifacts by age, count, and bytes per board.""" current = time.time() if now is None else now removed = 0 for board_dir in STATE_ROOT.iterdir() if STATE_ROOT.is_dir() else (): if not board_dir.is_dir() or board_dir.is_symlink(): continue board_removed = 0 entries: list[tuple[Path, os.stat_result]] = [] for path in _artifact_gc_candidates(board_dir): if path.name.startswith(".retire."): snapshot = _open_terminal_recovery_snapshot(path) if snapshot is not None: try: if _staged_terminal_authority(path, snapshot) is not None: # Accepted authority is first promoted/replayed by # recovery; retention never deletes it directly. continue finally: snapshot.close() try: file_stat = path.stat(follow_symlinks=False) except OSError: continue if not stat.S_ISREG(file_stat.st_mode): if _unlink_artifact_if_same(path, file_stat): removed += 1 board_removed += 1 continue if max_age_seconds >= 0 and current - file_stat.st_mtime > max_age_seconds: if _unlink_artifact_if_same(path, file_stat): removed += 1 board_removed += 1 else: entries.append((path, file_stat)) retained_count = 0 retained_bytes = 0 for path, file_stat in sorted( entries, key=lambda item: item[1].st_mtime, reverse=True ): exceeds_count = max_count >= 0 and retained_count >= max_count exceeds_bytes = ( max_bytes >= 0 and retained_bytes + file_stat.st_size > max_bytes ) if exceeds_count or exceeds_bytes: if _unlink_artifact_if_same(path, file_stat): removed += 1 board_removed += 1 continue retained_count += 1 retained_bytes += file_stat.st_size if board_removed: _fsync_directory(board_dir) return removed def maybe_gc_lane_artifacts(*, now: float | None = None) -> int: """Run artifact retention on a bounded cadence, not every dispatcher tick.""" global LAST_ARTIFACT_GC current = time.time() if now is None else now if current - LAST_ARTIFACT_GC < ARTIFACT_GC_INTERVAL_SECONDS: return 0 LAST_ARTIFACT_GC = current try: return gc_lane_artifacts(now=current) except OSError as error: print( f"lane artifact retention deferred: {type(error).__name__}: {error}", file=sys.stderr, flush=True, ) return 0 def execute_claim(board: str, task_id: str) -> None: """Execute one already-claimed task and commit its outcome to Kanban.""" from hermes_cli import kanban_db run_id = None with kanban_db.scoped_current_board(board): conn = kanban_db.connect(board=board) try: task = kanban_db.get_task(conn, task_id) if task is None: return run_id = _task_value(task, "current_run_id", None) assignee = str(_task_value(task, "assignee", "cli-auto") or "cli-auto") state_file = state_path(board, task_id) state = load_json(state_file) state.update( { "board": board, "task_id": task_id, "run_id": run_id, "assignee": assignee, } ) atomic_json(state_file, state) log_path = Path(kanban_db.worker_log_path(task_id, board=board)) workspace = _resolve_workspace(kanban_db, conn, task, board) kanban_db.set_workspace_path(conn, task_id, str(workspace)) context = _task_context(kanban_db, conn, task_id) except Exception as error: failure_reason = ( f"Direct CLI lane preparation failed: {type(error).__name__}: {error}" ) conn.close() _board_call( kanban_db, board, lambda fresh: kanban_db.block_task( fresh, task_id, reason=failure_reason, kind="capability", expected_run_id=run_id, ), ) return finally: conn.close() def heartbeat(note: str) -> bool: try: return bool( _board_call( kanban_db, board, lambda fresh: kanban_db.heartbeat_worker( fresh, task_id, note=note, expected_run_id=run_id, ), ) ) except (OSError, sqlite3.Error): # The claim TTL is deliberately long. Keep the expensive # provider process alive during a transient volume stall and # retry on the next heartbeat instead of losing its work. return True def comment(body: str) -> None: try: _board_call( kanban_db, board, lambda fresh: kanban_db.add_comment( fresh, task_id, "cli-lane-runner", body, ), ) except (OSError, sqlite3.Error): # Route state is also written to the durable lane-state file; # a later heartbeat or terminal result remains authoritative. pass try: previous_route = state.get("current_route") excluded_provider = ( fresh_unavailable_provider() if assignee == "cli-auto" else None ) unavailable_provider = excluded_provider route = select_route( context, assignee, exclude_provider=excluded_provider, exclude_reason="is unavailable according to fresh native health" if excluded_provider else None, ) if excluded_provider: comment( f"Provider health guard excluded {excluded_provider} before automatic routing.", ) comment( f"CLI route: {route.provider}/{route.model} at {route.effort}; classifier={route.classifier}; {route.reason}", ) resume_handoff = "" if ( isinstance(previous_route, dict) and previous_route.get("provider") and previous_route.get("provider") != route.provider ): try: previous_output = log_path.read_text(encoding="utf-8")[-10000:] except OSError: previous_output = "Previous provider log was unavailable after restart." resume_handoff = git_handoff(workspace, previous_output) comment( f"Restart-time provider change: {previous_route.get('provider')} -> {route.provider}; explicit workspace handoff attached.", ) prompt = build_prompt(context, workspace, resume_handoff) max_runtime = int( _task_value(task, "max_runtime_seconds", 0) or DEFAULT_MAX_RUNTIME ) goal_mode = bool(_task_value(task, "goal_mode", False)) goal_max_turns = max(1, int(_task_value(task, "goal_max_turns", 1) or 1)) goal_turn = max(1, int(state.get("goal_turn", 0) or 0) + 1) deadline = time.monotonic() + max_runtime while True: state["goal_turn"] = goal_turn atomic_json(state_file, state) remaining = max(1, int(deadline - time.monotonic())) result = run_provider( route, prompt, workspace, state, state_file, log_path, heartbeat, remaining, ) candidate_file = None if result.structured: candidate_file = _persist_candidate( state, state_file, dict(result.structured), route=route, returncode=result.returncode, goal_turn=goal_turn, ) if result.capacity_failure: unavailable_provider = route.provider retry_context = ( context + "\n\nRouting boundary: the first provider failed from capacity/authentication. " + "Select the alternate hosted provider at an appropriate effort." ) alternate = "claude" if route.provider == "codex" else "codex" fallback = select_route( retry_context, f"cli-{alternate}-{route.effort}", ) comment( f"Provider fallback: {route.provider} -> {fallback.provider}; Jetson reclassified the retry boundary.", ) result = run_provider( fallback, build_prompt( context, workspace, git_handoff(workspace, result.output), ), workspace, state, state_file, log_path, heartbeat, max(1, int(deadline - time.monotonic())), ) route = fallback if result.structured: candidate_file = _persist_candidate( state, state_file, dict(result.structured), route=route, returncode=result.returncode, goal_turn=goal_turn, ) structured = result.structured if structured: structured = dict(structured) structured["artifacts"] = workspace_artifacts( workspace, structured.get("artifacts"), ) metadata = { "executor": "direct-cli-lane", "provider": route.provider, "model": route.model, "effort": route.effort, "classifier": route.classifier, "state_file": str(state_file), "codex_thread_id": state.get("codex_thread_id"), "claude_session_id": state.get("claude_session_id"), "goal_mode": goal_mode, "goal_turn": goal_turn, } if candidate_file is not None: metadata["candidate_file"] = str(candidate_file) if structured: for key in ( "changed_files", "tests_run", "artifacts", "findings", "blockers", ): value = structured.get(key) metadata[key] = value if isinstance(value, list) else [] completion_problem = None if structured and result.returncode == 0: completion_problem = cli_lane_goal.unfinished_result_reason(structured) if ( structured.get("status") == "completed" and completion_problem is None and goal_mode ): heartbeat("local goal-completion judge active") rejection_history = state.get("goal_rejections", []) if not isinstance(rejection_history, list): rejection_history = [] judge_context = context if goal_turn > 1 or rejection_history: judge_context += ( "\n\nAuthoritative Hermes goal-controller evidence: " f"current turn {goal_turn}/{goal_max_turns}; prior rejected " f"reports: {json.dumps(rejection_history[-5:])}." ) accepted, judge_reason = cli_lane_goal.judge_goal_completion( judge_context, structured, ) metadata["goal_judge_reason"] = judge_reason if not accepted: completion_problem = f"local goal judge requested continuation: {judge_reason}" if ( structured and structured.get("status") == "completed" and result.returncode == 0 and completion_problem is None ): terminal_file = _terminal_path(state_file, run_id, "pending") metadata["terminal_record"] = str( _terminal_path(state_file, run_id, "committed") ) try: terminal_file, terminal_record = _write_terminal_record( state_file, board=board, task_id=task_id, run_id=run_id, structured=structured, summary=str(structured.get("summary") or "Completed"), metadata=metadata, ) outcome = _finalize_terminal_record( kanban_db, terminal_file, terminal_record, ) except Exception as error: identity = _terminal_identity(terminal_file) replayable = bool( identity is not None and _has_pending_finalization( identity.board, identity.task_id, identity.run_id, ) ) if not replayable: _recover_exact_run( kanban_db, identity, f"persistence raised {type(error).__name__}", ) raise TerminalFinalizationPending( "accepted worker result remains outside the generic block path; " f"terminal replayable={replayable}; persistence/finalization " f"raised {type(error).__name__}" ) from error if outcome != "committed": raise TerminalFinalizationPending( "accepted worker result is durably journaled but " f"Kanban finalization is {outcome}" ) break can_continue = ( goal_mode and completion_problem is not None and goal_turn < goal_max_turns and deadline - time.monotonic() > 30 ) if can_continue: rejection_history = state.get("goal_rejections", []) if not isinstance(rejection_history, list): rejection_history = [] state["goal_rejections"] = [ *rejection_history[-4:], completion_problem, ] goal_turn += 1 comment( f"Goal completion rejected; continuing turn {goal_turn}/{goal_max_turns}: {completion_problem}", ) escalation_context = ( context + "\n\nThe previous worker turn missed its completion quality mark: " + completion_problem + "\nSelect a route that can finish and verify the remaining work." ) excluded = unavailable_provider if excluded is None and assignee == "cli-auto": excluded = fresh_unavailable_provider() next_route = select_route( escalation_context, assignee, exclude_provider=excluded, exclude_reason="is unavailable according to fresh native health" if excluded else None, ) comment( f"Goal route {goal_turn}/{goal_max_turns}: " f"{next_route.provider}/{next_route.model} at {next_route.effort}; " f"classifier={next_route.classifier}; {next_route.reason}", ) handoff = ( git_handoff(workspace, result.output) if next_route.provider != route.provider else "" ) route = next_route prompt = build_prompt( context, workspace, handoff + "\nGoal-loop continuation: the previous final report was rejected because " + completion_problem + ". Reinspect live state, finish the outstanding work, and return new final evidence.", ) continue reason = completion_problem if reason is None and structured: blockers = structured.get("blockers", []) reason = "; ".join(str(item) for item in blockers) reason = reason or str(structured.get("summary") or "") if reason is None: reason = result.output[-4000:] _board_call( kanban_db, board, lambda fresh: kanban_db.block_task( fresh, task_id, reason=reason or f"{route.provider} worker failed with exit {result.returncode}", kind="transient" if result.capacity_failure else "capability", expected_run_id=run_id, ), ) break except TerminalFinalizationPending as error: comment(str(error)) except Exception as error: failure_reason = f"Direct CLI lane failed: {type(error).__name__}: {error}" _board_call( kanban_db, board, lambda fresh: kanban_db.block_task( fresh, task_id, reason=failure_reason, kind="capability", expected_run_id=run_id, ), ) def _board_slug(board: Any) -> str: if isinstance(board, dict): return str(board.get("slug") or board.get("id") or "") return str(getattr(board, "slug", None) or getattr(board, "id", None) or board) def _external(task: Any) -> bool: return str(_task_value(task, "assignee", "") or "").startswith(EXTERNAL_PREFIX) def _connect_healthy_board(kanban_db: Any, board: str) -> Any | None: """Open one board without letting localized storage faults stop other lanes.""" try: return kanban_db.connect(board=board) except Exception as error: _record_board_access_error(board, error) return None def _record_board_access_error(board: str, error: Exception) -> None: """Report one board access failure once while allowing other lanes to run.""" failure_kind = "storage" if isinstance(error, (OSError, sqlite3.Error)) else "access" detail = f"{failure_kind} {type(error).__name__}: {error}" if BOARD_CORRUPTION_ERRORS.get(board) == detail: return print( f"temporarily skipping Kanban board {board!r}: {detail}", file=sys.stderr, flush=True, ) BOARD_CORRUPTION_ERRORS[board] = detail def recover_orphans() -> None: """Return external running tasks to ready after a runner/pod restart.""" from hermes_cli import kanban_db recover_pending_finalizations() try: boards = kanban_db.list_boards(include_archived=False) except Exception as error: _record_board_access_error("board-registry", error) return for raw_board in boards: board = _board_slug(raw_board) if not board: continue with kanban_db.scoped_current_board(board): conn = _connect_healthy_board(kanban_db, board) if conn is None: continue try: for task in kanban_db.list_tasks(conn): if _external(task) and str(_task_value(task, "status", "")) == "running": task_id = str(_task_value(task, "id")) run_id = _task_value(task, "current_run_id", None) if not isinstance(run_id, int): continue if _has_pending_finalization(board, task_id, run_id): continue kanban_db.reclaim_task( conn, task_id, reason="direct CLI lane restarted; provider session will resume", expected_run_id=run_id, ) BOARD_CORRUPTION_ERRORS.pop(board, None) except Exception as error: _record_board_access_error(board, error) finally: conn.close() def claim_ready(active: set[tuple[str, str]], limit: int) -> list[tuple[str, str]]: """Atomically claim external ready tasks across all non-archived boards.""" from hermes_cli import kanban_db claimed: list[tuple[str, str]] = [] if limit <= 0: return claimed for raw_board in kanban_db.list_boards(include_archived=False): board = _board_slug(raw_board) if not board: continue with kanban_db.scoped_current_board(board): conn = _connect_healthy_board(kanban_db, board) if conn is None: continue try: kanban_db.recompute_ready(conn) tasks = kanban_db.list_tasks(conn) BOARD_CORRUPTION_ERRORS.pop(board, None) for task in tasks: task_id = str(_task_value(task, "id", "")) assignee = str(_task_value(task, "assignee", "") or "") if task_id and not assignee and str(_task_value(task, "status", "")) == "ready": if kanban_db.assign_task(conn, task_id, "cli-auto"): task = kanban_db.get_task(conn, task_id) assignee = "cli-auto" if ( not task_id or (board, task_id) in active or not assignee.startswith(EXTERNAL_PREFIX) or str(_task_value(task, "status", "")) != "ready" ): continue try: result = kanban_db.claim_task( conn, task_id, ttl_seconds=DEFAULT_CLAIM_TTL, claimer="direct-cli-lane", ) except Exception: continue if result is not None: claimed.append((board, task_id)) if len(claimed) >= limit: return claimed except Exception as error: _record_board_access_error(board, error) finally: conn.close() return claimed def main() -> int: """Continuously bridge external Kanban lanes to provider CLIs.""" RESULT_SCHEMA_PATH.parent.mkdir(parents=True, exist_ok=True) atomic_json(RESULT_SCHEMA_PATH, RESULT_SCHEMA, 0o644) recover_orphans() workers = max(1, min(int(os.environ.get("HERMES_CLI_LANE_CONCURRENCY", "4")), 8)) futures: dict[concurrent.futures.Future[None], tuple[str, str]] = {} with concurrent.futures.ThreadPoolExecutor(max_workers=workers) as pool: while True: for future in list(futures): if future.done(): try: future.result() except Exception as error: print(f"worker future failed: {error}", file=sys.stderr, flush=True) del futures[future] recover_pending_finalizations() maybe_gc_lane_artifacts() active = set(futures.values()) try: newly_claimed = claim_ready(active, workers - len(futures)) BOARD_CORRUPTION_ERRORS.pop("board-registry", None) except Exception as error: _record_board_access_error("board-registry", error) newly_claimed = [] for board, task_id in newly_claimed: future = pool.submit(execute_claim, board, task_id) futures[future] = (board, task_id) time.sleep(5) return 0 if __name__ == "__main__": raise SystemExit(main())