335 lines
12 KiB
Python
335 lines
12 KiB
Python
|
|
#!/usr/bin/env python3
|
||
|
|
"""Orchestration for the Hermes full-handoff acceptance harness.
|
||
|
|
|
||
|
|
Three things happen before any check runs, and each of them can stop the run on
|
||
|
|
its own:
|
||
|
|
|
||
|
|
* the catalog is validated structurally — every rule exists, every deny check
|
||
|
|
carries a real attempt, no operator-side step can impersonate;
|
||
|
|
* each vantage is resolved and asked who it is; and
|
||
|
|
* the operator and in-pod identities are compared. If they are the same
|
||
|
|
principal there is only one vantage, the dual-vantage evidence would be a
|
||
|
|
restatement rather than a corroboration, and the run is a NO_GO.
|
||
|
|
|
||
|
|
Everything after that is uniform: run a check's steps against its vantage, hand
|
||
|
|
the outcomes to the named evaluator, and record the classification. A vantage
|
||
|
|
that could not be resolved yields ``NOT_RUN`` for every check that needs it,
|
||
|
|
which is exactly the fail-closed answer.
|
||
|
|
"""
|
||
|
|
|
||
|
|
from __future__ import annotations
|
||
|
|
|
||
|
|
import json
|
||
|
|
from collections.abc import Iterable, Mapping, Sequence
|
||
|
|
|
||
|
|
from hermes_handoff_catalog import (
|
||
|
|
CHAT,
|
||
|
|
NODE,
|
||
|
|
OPERATOR,
|
||
|
|
SELF,
|
||
|
|
SWITCHYARD,
|
||
|
|
VANTAGE_NAMES,
|
||
|
|
Targets,
|
||
|
|
)
|
||
|
|
from hermes_handoff_checks_access import access_checks
|
||
|
|
from hermes_handoff_checks_delivery import delivery_checks
|
||
|
|
from hermes_handoff_checks_platform import platform_checks
|
||
|
|
from hermes_handoff_checks_workers import worker_checks
|
||
|
|
from hermes_handoff_evaluators import EVALUATORS, evaluate
|
||
|
|
from hermes_handoff_exec import Outcome, Runner, Vantage, operator_vantage, pod_vantage
|
||
|
|
from hermes_handoff_model import (
|
||
|
|
ATTEMPT,
|
||
|
|
EPHEMERAL,
|
||
|
|
NOT_RUN,
|
||
|
|
STEP_KINDS,
|
||
|
|
CheckResult,
|
||
|
|
CheckSpec,
|
||
|
|
Report,
|
||
|
|
VantageRecord,
|
||
|
|
utc_now,
|
||
|
|
)
|
||
|
|
from hermes_handoff_policy import IMPERSONATION_ARGS
|
||
|
|
|
||
|
|
UNRECORDED = "[output not recorded]"
|
||
|
|
|
||
|
|
|
||
|
|
def build_catalog(targets: Targets) -> list[CheckSpec]:
|
||
|
|
"""Return the whole acceptance catalog for a set of targets."""
|
||
|
|
return [
|
||
|
|
*platform_checks(targets),
|
||
|
|
*access_checks(targets),
|
||
|
|
*delivery_checks(targets),
|
||
|
|
*worker_checks(targets),
|
||
|
|
]
|
||
|
|
|
||
|
|
|
||
|
|
def validate_catalog(specs: Sequence[CheckSpec]) -> list[str]:
|
||
|
|
"""Return the structural problems that make a catalog unsafe to run."""
|
||
|
|
problems: list[str] = []
|
||
|
|
seen: set[str] = set()
|
||
|
|
for spec in specs:
|
||
|
|
if spec.id in seen:
|
||
|
|
problems.append(f"{spec.id}: duplicate check id")
|
||
|
|
seen.add(spec.id)
|
||
|
|
if spec.rule not in EVALUATORS and spec.scope != EPHEMERAL:
|
||
|
|
problems.append(f"{spec.id}: unknown rule {spec.rule!r}")
|
||
|
|
for step in spec.steps:
|
||
|
|
if step.vantage not in VANTAGE_NAMES:
|
||
|
|
problems.append(f"{spec.id}: step {step.key} names unknown vantage {step.vantage!r}")
|
||
|
|
if step.kind not in STEP_KINDS:
|
||
|
|
problems.append(f"{spec.id}: step {step.key} has unknown kind {step.kind!r}")
|
||
|
|
if step.vantage == OPERATOR and set(step.argv) & set(IMPERSONATION_ARGS):
|
||
|
|
problems.append(f"{spec.id}: step {step.key} impersonates from the operator vantage")
|
||
|
|
if spec.rule == "denied" and not any(step.kind == ATTEMPT for step in spec.steps):
|
||
|
|
problems.append(f"{spec.id}: a deny check needs a real attempt, not only a review")
|
||
|
|
return problems
|
||
|
|
|
||
|
|
|
||
|
|
def resolve_identity(runner: Runner, vantage: Vantage) -> tuple[str, str]:
|
||
|
|
"""Return the authenticated username of a vantage and any failure detail."""
|
||
|
|
outcome = runner.run(("kubectl", "auth", "whoami", "-o", "json"), vantage)
|
||
|
|
if not outcome.ok:
|
||
|
|
return ("", outcome.error or outcome.combined[:200])
|
||
|
|
try:
|
||
|
|
payload = json.loads(outcome.stdout)
|
||
|
|
except ValueError as exc:
|
||
|
|
return ("", f"unparsable identity: {exc}")
|
||
|
|
username = ((payload.get("status") or {}).get("userInfo") or {}).get("username")
|
||
|
|
if not isinstance(username, str) or not username:
|
||
|
|
return ("", "identity response carried no username")
|
||
|
|
return (username, "")
|
||
|
|
|
||
|
|
|
||
|
|
def _first_pod(runner: Runner, operator: Vantage, namespace: str, selector: str) -> tuple[str, str]:
|
||
|
|
outcome = runner.run(
|
||
|
|
(
|
||
|
|
"kubectl",
|
||
|
|
"--namespace",
|
||
|
|
namespace,
|
||
|
|
"get",
|
||
|
|
"pods",
|
||
|
|
"--selector",
|
||
|
|
selector,
|
||
|
|
"--field-selector",
|
||
|
|
"status.phase=Running",
|
||
|
|
"-o",
|
||
|
|
"name",
|
||
|
|
),
|
||
|
|
operator,
|
||
|
|
)
|
||
|
|
if not outcome.ok:
|
||
|
|
return ("", outcome.error or outcome.combined[:200])
|
||
|
|
names = [line.strip().removeprefix("pod/") for line in outcome.stdout.splitlines() if line.strip()]
|
||
|
|
if not names:
|
||
|
|
return ("", f"no running pod matches {selector}")
|
||
|
|
return (sorted(names)[0], "")
|
||
|
|
|
||
|
|
|
||
|
|
def resolve_vantages(
|
||
|
|
runner: Runner, targets: Targets, kubeconfig: str | None = None, context: str | None = None
|
||
|
|
) -> tuple[dict[str, Vantage], list[VantageRecord]]:
|
||
|
|
"""Resolve every vantage the catalog names, recording what each one is."""
|
||
|
|
operator = operator_vantage(kubeconfig, context)
|
||
|
|
username, detail = resolve_identity(runner, operator)
|
||
|
|
records = [
|
||
|
|
VantageRecord(
|
||
|
|
name=OPERATOR,
|
||
|
|
description=operator.description or "external read-only operator",
|
||
|
|
identity=username,
|
||
|
|
available=bool(username),
|
||
|
|
detail=detail,
|
||
|
|
)
|
||
|
|
]
|
||
|
|
vantages: dict[str, Vantage] = {OPERATOR: operator}
|
||
|
|
if not username:
|
||
|
|
return (vantages, records)
|
||
|
|
|
||
|
|
for name, selector, container, description in (
|
||
|
|
(SELF, f"app={targets.agent_deployment}", targets.agent_container, "the Hermes agent itself"),
|
||
|
|
(
|
||
|
|
SWITCHYARD,
|
||
|
|
f"app={targets.switchyard_deployment}",
|
||
|
|
targets.switchyard_container,
|
||
|
|
"the Switchyard routing evidence log",
|
||
|
|
),
|
||
|
|
(NODE, f"app={targets.node_daemonset}", targets.node_container, "a node hardening probe"),
|
||
|
|
):
|
||
|
|
pod, failure = _first_pod(runner, operator, targets.namespace, selector)
|
||
|
|
if not pod:
|
||
|
|
records.append(
|
||
|
|
VantageRecord(name=name, description=description, available=False, detail=failure)
|
||
|
|
)
|
||
|
|
continue
|
||
|
|
vantage = pod_vantage(targets.namespace, pod, container, operator)
|
||
|
|
identity, identity_detail = resolve_identity(runner, vantage) if name == SELF else ("", "")
|
||
|
|
records.append(
|
||
|
|
VantageRecord(
|
||
|
|
name=name,
|
||
|
|
description=f"{description} ({targets.namespace}/{pod})",
|
||
|
|
identity=identity,
|
||
|
|
available=True,
|
||
|
|
detail=identity_detail,
|
||
|
|
)
|
||
|
|
)
|
||
|
|
vantages[name] = vantage
|
||
|
|
|
||
|
|
chat_pod = f"{targets.chat_statefulset}-{targets.chat_ordinal}"
|
||
|
|
present = runner.run(
|
||
|
|
("kubectl", "--namespace", targets.namespace, "get", f"pod/{chat_pod}", "-o", "name"),
|
||
|
|
operator,
|
||
|
|
)
|
||
|
|
if present.ok and present.stdout.strip():
|
||
|
|
vantages[CHAT] = pod_vantage(targets.namespace, chat_pod, targets.chat_container, operator)
|
||
|
|
records.append(
|
||
|
|
VantageRecord(
|
||
|
|
name=CHAT,
|
||
|
|
description=f"the chat surface ({targets.namespace}/{chat_pod})",
|
||
|
|
available=True,
|
||
|
|
)
|
||
|
|
)
|
||
|
|
else:
|
||
|
|
records.append(
|
||
|
|
VantageRecord(
|
||
|
|
name=CHAT,
|
||
|
|
description=f"the chat surface ({targets.namespace}/{chat_pod})",
|
||
|
|
available=False,
|
||
|
|
detail=present.error or present.combined[:200] or "chat tenant pod not found",
|
||
|
|
)
|
||
|
|
)
|
||
|
|
return (vantages, records)
|
||
|
|
|
||
|
|
|
||
|
|
def vantage_problems(records: Iterable[VantageRecord]) -> list[str]:
|
||
|
|
"""Return the identity problems that invalidate dual-vantage evidence."""
|
||
|
|
by_name = {record.name: record for record in records}
|
||
|
|
operator = by_name.get(OPERATOR)
|
||
|
|
hermes = by_name.get(SELF)
|
||
|
|
problems: list[str] = []
|
||
|
|
if operator is None or not operator.available:
|
||
|
|
problems.append("the operator vantage is unavailable; no evidence can be collected")
|
||
|
|
return problems
|
||
|
|
if hermes is None or not hermes.available:
|
||
|
|
problems.append("the in-pod Hermes vantage is unavailable; self-probe evidence is missing")
|
||
|
|
return problems
|
||
|
|
if not hermes.identity:
|
||
|
|
problems.append("the in-pod vantage did not report an identity")
|
||
|
|
elif operator.identity == hermes.identity:
|
||
|
|
problems.append(
|
||
|
|
"the operator and in-pod vantages authenticate as the same principal "
|
||
|
|
f"({operator.identity}); run the harness with a separate operator kubeconfig"
|
||
|
|
)
|
||
|
|
return problems
|
||
|
|
|
||
|
|
|
||
|
|
def run_check(
|
||
|
|
runner: Runner, spec: CheckSpec, vantages: Mapping[str, Vantage]
|
||
|
|
) -> CheckResult:
|
||
|
|
"""Run one check's steps and classify the result."""
|
||
|
|
outcomes: dict[str, Outcome] = {}
|
||
|
|
recorded: list[Outcome] = []
|
||
|
|
for step in spec.steps:
|
||
|
|
vantage = vantages.get(step.vantage)
|
||
|
|
if vantage is None:
|
||
|
|
outcome = Outcome(
|
||
|
|
argv=step.argv, vantage=step.vantage, error=f"vantage {step.vantage} is unavailable"
|
||
|
|
)
|
||
|
|
else:
|
||
|
|
outcome = runner.run(step.argv, vantage, step.max_bytes)
|
||
|
|
outcomes[step.key] = outcome
|
||
|
|
recorded.append(
|
||
|
|
outcome
|
||
|
|
if step.record
|
||
|
|
else Outcome(
|
||
|
|
argv=outcome.argv,
|
||
|
|
vantage=outcome.vantage,
|
||
|
|
returncode=outcome.returncode,
|
||
|
|
stdout=UNRECORDED if outcome.stdout else "",
|
||
|
|
stderr=UNRECORDED if outcome.stderr else "",
|
||
|
|
truncated=outcome.truncated,
|
||
|
|
duration_ms=outcome.duration_ms,
|
||
|
|
error=outcome.error,
|
||
|
|
)
|
||
|
|
)
|
||
|
|
evaluation = evaluate(spec, outcomes)
|
||
|
|
return CheckResult(
|
||
|
|
spec=spec,
|
||
|
|
status=evaluation.status,
|
||
|
|
reason=evaluation.reason,
|
||
|
|
evidence=evaluation.evidence,
|
||
|
|
outcomes=recorded,
|
||
|
|
)
|
||
|
|
|
||
|
|
|
||
|
|
def run_catalog(
|
||
|
|
runner: Runner, specs: Sequence[CheckSpec], vantages: Mapping[str, Vantage]
|
||
|
|
) -> list[CheckResult]:
|
||
|
|
"""Run every check in the catalog.
|
||
|
|
|
||
|
|
Nothing is skipped once the deadline passes: the runner returns a
|
||
|
|
deadline outcome for each remaining command, which classifies as
|
||
|
|
``NOT_RUN``, so a truncated sweep is visible in the report rather than
|
||
|
|
silently shorter.
|
||
|
|
"""
|
||
|
|
return [run_check(runner, spec, vantages) for spec in specs]
|
||
|
|
|
||
|
|
|
||
|
|
def build_report(
|
||
|
|
runner: Runner,
|
||
|
|
targets: Targets,
|
||
|
|
specs: Sequence[CheckSpec],
|
||
|
|
vantages: Mapping[str, Vantage],
|
||
|
|
records: Sequence[VantageRecord],
|
||
|
|
mode: str,
|
||
|
|
started_at: str,
|
||
|
|
extra_results: Sequence[CheckResult] = (),
|
||
|
|
) -> Report:
|
||
|
|
"""Run the catalog and assemble the report, including structural problems."""
|
||
|
|
problems = validate_catalog(specs) + vantage_problems(records)
|
||
|
|
report = Report(
|
||
|
|
mode=mode,
|
||
|
|
started_at=started_at,
|
||
|
|
baseline={
|
||
|
|
"namespace": targets.namespace,
|
||
|
|
"baseline_commit": targets.baseline_commit,
|
||
|
|
"dependency_pull_requests": list(targets.dependency_pull_requests),
|
||
|
|
"expected_suspensions": list(targets.expected_suspensions),
|
||
|
|
"node_count": targets.node_count,
|
||
|
|
"pool_replicas": targets.pool_replicas,
|
||
|
|
},
|
||
|
|
vantages=list(records),
|
||
|
|
harness_errors=problems,
|
||
|
|
)
|
||
|
|
armed_ids = {result.spec.id for result in extra_results}
|
||
|
|
report.results = [
|
||
|
|
*run_catalog(runner, [spec for spec in specs if spec.id not in armed_ids], vantages),
|
||
|
|
*extra_results,
|
||
|
|
]
|
||
|
|
report.results.sort(key=lambda result: result.spec.id)
|
||
|
|
if runner.remaining_seconds <= 0:
|
||
|
|
report.harness_errors.append("the run deadline expired before every check completed")
|
||
|
|
report.finished_at = utc_now()
|
||
|
|
return report
|
||
|
|
|
||
|
|
|
||
|
|
def unavailable_report(mode: str, started_at: str, reason: str, targets: Targets) -> Report:
|
||
|
|
"""Return a NO_GO report for a run that could not start at all."""
|
||
|
|
return Report(
|
||
|
|
mode=mode,
|
||
|
|
started_at=started_at,
|
||
|
|
finished_at=utc_now(),
|
||
|
|
baseline={"namespace": targets.namespace},
|
||
|
|
harness_errors=[reason],
|
||
|
|
results=[
|
||
|
|
CheckResult(
|
||
|
|
spec=CheckSpec(
|
||
|
|
id="harness.startup",
|
||
|
|
title="The harness could start and resolve its vantages",
|
||
|
|
group="harness",
|
||
|
|
rule="allowed",
|
||
|
|
),
|
||
|
|
status=NOT_RUN,
|
||
|
|
reason=reason,
|
||
|
|
)
|
||
|
|
],
|
||
|
|
)
|