336 lines
14 KiB
Python
336 lines
14 KiB
Python
|
|
#!/usr/bin/env python3
|
||
|
|
"""JSON-shaped evaluators for the handoff acceptance harness.
|
||
|
|
|
||
|
|
These rules cover the evidence that arrives as structured output — Kubernetes
|
||
|
|
objects, forge pull requests, provider health files, and the two append-only
|
||
|
|
evidence logs the platform keeps. They share the fail-closed contract of every
|
||
|
|
other rule: an absent path, an unparsable payload, or an empty result set is
|
||
|
|
``NOT_RUN``, never a quiet pass.
|
||
|
|
"""
|
||
|
|
|
||
|
|
from __future__ import annotations
|
||
|
|
|
||
|
|
import datetime as dt
|
||
|
|
import json
|
||
|
|
from typing import Any
|
||
|
|
|
||
|
|
from hermes_handoff_exec import Outcome
|
||
|
|
from hermes_handoff_model import FAIL, NOT_RUN, PASS, CheckSpec
|
||
|
|
from hermes_handoff_rules import (
|
||
|
|
Evaluation,
|
||
|
|
dotted,
|
||
|
|
evaluator,
|
||
|
|
is_list,
|
||
|
|
missing_steps,
|
||
|
|
parsed_step,
|
||
|
|
)
|
||
|
|
|
||
|
|
|
||
|
|
@evaluator("json_field")
|
||
|
|
def evaluate_json_field(spec: CheckSpec, outcomes: dict[str, Outcome]) -> Evaluation:
|
||
|
|
"""Compare dotted JSON paths of one step's output against expected values."""
|
||
|
|
blocked = missing_steps(spec, outcomes)
|
||
|
|
if blocked:
|
||
|
|
return blocked
|
||
|
|
parsed = parsed_step(spec, outcomes)
|
||
|
|
if isinstance(parsed, Evaluation):
|
||
|
|
return parsed
|
||
|
|
payload, key = parsed
|
||
|
|
evidence: dict[str, Any] = {}
|
||
|
|
mismatches: list[str] = []
|
||
|
|
for path, expected in spec.expect["fields"].items():
|
||
|
|
try:
|
||
|
|
actual = dotted(payload, path)
|
||
|
|
except (KeyError, ValueError, IndexError):
|
||
|
|
return Evaluation(NOT_RUN, f"path {path} is absent from step {key}", evidence)
|
||
|
|
evidence[path] = actual
|
||
|
|
allowed = expected if isinstance(expected, list) else [expected]
|
||
|
|
if actual not in allowed:
|
||
|
|
mismatches.append(f"{path}={actual!r} expected one of {allowed!r}")
|
||
|
|
if mismatches:
|
||
|
|
return Evaluation(FAIL, "; ".join(mismatches), evidence)
|
||
|
|
return Evaluation(PASS, "observed fields match the release contract", evidence)
|
||
|
|
|
||
|
|
|
||
|
|
@evaluator("json_numeric")
|
||
|
|
def evaluate_json_numeric(spec: CheckSpec, outcomes: dict[str, Outcome]) -> Evaluation:
|
||
|
|
"""Compare numeric JSON paths against inclusive minimum/maximum bounds."""
|
||
|
|
blocked = missing_steps(spec, outcomes)
|
||
|
|
if blocked:
|
||
|
|
return blocked
|
||
|
|
parsed = parsed_step(spec, outcomes)
|
||
|
|
if isinstance(parsed, Evaluation):
|
||
|
|
return parsed
|
||
|
|
payload, key = parsed
|
||
|
|
evidence: dict[str, Any] = {}
|
||
|
|
problems: list[str] = []
|
||
|
|
for path, bounds in spec.expect["fields"].items():
|
||
|
|
try:
|
||
|
|
actual = dotted(payload, path)
|
||
|
|
except (KeyError, ValueError, IndexError):
|
||
|
|
return Evaluation(NOT_RUN, f"path {path} is absent from step {key}", evidence)
|
||
|
|
if not isinstance(actual, (int, float)) or isinstance(actual, bool):
|
||
|
|
return Evaluation(NOT_RUN, f"path {path} is not numeric in step {key}", evidence)
|
||
|
|
evidence[path] = actual
|
||
|
|
if "min" in bounds and actual < bounds["min"]:
|
||
|
|
problems.append(f"{path}={actual} below minimum {bounds['min']}")
|
||
|
|
if "max" in bounds and actual > bounds["max"]:
|
||
|
|
problems.append(f"{path}={actual} above maximum {bounds['max']}")
|
||
|
|
if problems:
|
||
|
|
return Evaluation(FAIL, "; ".join(problems), evidence)
|
||
|
|
return Evaluation(PASS, "numeric fields are within the expected bounds", evidence)
|
||
|
|
|
||
|
|
|
||
|
|
@evaluator("json_recent")
|
||
|
|
def evaluate_json_recent(spec: CheckSpec, outcomes: dict[str, Outcome]) -> Evaluation:
|
||
|
|
"""Pass when a timestamp field is fresh enough to be evidence about now.
|
||
|
|
|
||
|
|
Stale provider or routing evidence is worse than none: it describes a state
|
||
|
|
the release no longer has. A timestamp older than the allowed age is a FAIL,
|
||
|
|
and one the harness cannot parse is NOT_RUN.
|
||
|
|
"""
|
||
|
|
blocked = missing_steps(spec, outcomes)
|
||
|
|
if blocked:
|
||
|
|
return blocked
|
||
|
|
parsed = parsed_step(spec, outcomes)
|
||
|
|
if isinstance(parsed, Evaluation):
|
||
|
|
return parsed
|
||
|
|
payload, key = parsed
|
||
|
|
now = spec.expect["now"]
|
||
|
|
evidence: dict[str, Any] = {}
|
||
|
|
stale: list[str] = []
|
||
|
|
for path, max_age in spec.expect["fields"].items():
|
||
|
|
try:
|
||
|
|
raw = dotted(payload, path)
|
||
|
|
except (KeyError, ValueError, IndexError):
|
||
|
|
return Evaluation(NOT_RUN, f"path {path} is absent from step {key}", evidence)
|
||
|
|
try:
|
||
|
|
observed = dt.datetime.fromisoformat(str(raw).replace("Z", "+00:00"))
|
||
|
|
except ValueError:
|
||
|
|
return Evaluation(NOT_RUN, f"path {path} is not an ISO timestamp", evidence)
|
||
|
|
if observed.tzinfo is None:
|
||
|
|
observed = observed.replace(tzinfo=dt.timezone.utc)
|
||
|
|
age = (now - observed).total_seconds()
|
||
|
|
evidence[path] = {"age_seconds": int(age)}
|
||
|
|
if age > max_age:
|
||
|
|
stale.append(f"{path} is {int(age)}s old, limit {max_age}s")
|
||
|
|
if stale:
|
||
|
|
return Evaluation(FAIL, "; ".join(stale), evidence)
|
||
|
|
return Evaluation(PASS, "observed timestamps are fresh", evidence)
|
||
|
|
|
||
|
|
|
||
|
|
@evaluator("all_of_field")
|
||
|
|
def evaluate_all_of_field(spec: CheckSpec, outcomes: dict[str, Outcome]) -> Evaluation:
|
||
|
|
"""Pass when every selected element of a JSON array matches expected fields."""
|
||
|
|
blocked = missing_steps(spec, outcomes)
|
||
|
|
if blocked:
|
||
|
|
return blocked
|
||
|
|
parsed = parsed_step(spec, outcomes)
|
||
|
|
if isinstance(parsed, Evaluation):
|
||
|
|
return parsed
|
||
|
|
payload, key = parsed
|
||
|
|
try:
|
||
|
|
elements = dotted(payload, spec.expect.get("array", ""))
|
||
|
|
except (KeyError, ValueError, IndexError):
|
||
|
|
return Evaluation(NOT_RUN, f"array path is absent from step {key}")
|
||
|
|
if not is_list(elements):
|
||
|
|
return Evaluation(NOT_RUN, f"step {key} did not return an array")
|
||
|
|
key_field = spec.expect.get("key_field", "")
|
||
|
|
wanted = spec.expect.get("keys")
|
||
|
|
exempt = set(spec.expect.get("exempt_keys", ()))
|
||
|
|
seen: list[Any] = []
|
||
|
|
problems: list[str] = []
|
||
|
|
evidence: dict[str, Any] = {}
|
||
|
|
for element in elements:
|
||
|
|
identity = dotted(element, key_field) if key_field else None
|
||
|
|
if wanted is not None and identity not in wanted:
|
||
|
|
continue
|
||
|
|
if identity in exempt:
|
||
|
|
continue
|
||
|
|
seen.append(identity)
|
||
|
|
for path in spec.expect.get("non_empty", ()):
|
||
|
|
try:
|
||
|
|
value = dotted(element, path)
|
||
|
|
except (KeyError, ValueError, IndexError):
|
||
|
|
value = None
|
||
|
|
if value in (None, "", [], {}):
|
||
|
|
problems.append(f"{identity}: {path} is empty")
|
||
|
|
for path, expected in spec.expect.get("fields", {}).items():
|
||
|
|
try:
|
||
|
|
actual = dotted(element, path)
|
||
|
|
except (KeyError, ValueError, IndexError):
|
||
|
|
actual = None
|
||
|
|
allowed = expected if isinstance(expected, list) else [expected]
|
||
|
|
if actual not in allowed:
|
||
|
|
problems.append(f"{identity}: {path}={actual!r} expected one of {allowed!r}")
|
||
|
|
evidence["examined"] = seen[:32]
|
||
|
|
if wanted is not None:
|
||
|
|
absent = [item for item in wanted if item not in seen]
|
||
|
|
if absent:
|
||
|
|
return Evaluation(NOT_RUN, f"expected keys absent from step {key}: {absent}", evidence)
|
||
|
|
elif not seen:
|
||
|
|
return Evaluation(NOT_RUN, f"step {key} returned nothing to examine", evidence)
|
||
|
|
if problems:
|
||
|
|
return Evaluation(FAIL, "; ".join(problems[:8]), evidence)
|
||
|
|
return Evaluation(PASS, f"{len(seen)} element(s) match the release contract", evidence)
|
||
|
|
|
||
|
|
|
||
|
|
@evaluator("json_record")
|
||
|
|
def evaluate_json_record(spec: CheckSpec, outcomes: dict[str, Outcome]) -> Evaluation:
|
||
|
|
"""Capture named JSON paths as evidence; pass when all of them resolve.
|
||
|
|
|
||
|
|
Recording is a release requirement in its own right — the runbook has to
|
||
|
|
name the exact heads it was verified against — so an unreachable source is a
|
||
|
|
NOT_RUN rather than an empty section.
|
||
|
|
"""
|
||
|
|
blocked = missing_steps(spec, outcomes)
|
||
|
|
if blocked:
|
||
|
|
return blocked
|
||
|
|
parsed = parsed_step(spec, outcomes)
|
||
|
|
if isinstance(parsed, Evaluation):
|
||
|
|
return parsed
|
||
|
|
payload, key = parsed
|
||
|
|
evidence: dict[str, Any] = {}
|
||
|
|
for label, path in spec.expect["record"].items():
|
||
|
|
try:
|
||
|
|
evidence[label] = dotted(payload, path)
|
||
|
|
except (KeyError, ValueError, IndexError):
|
||
|
|
return Evaluation(NOT_RUN, f"path {path} is absent from step {key}", evidence)
|
||
|
|
return Evaluation(PASS, "state recorded", evidence)
|
||
|
|
|
||
|
|
|
||
|
|
|
||
|
|
@evaluator("routing_evidence")
|
||
|
|
def evaluate_routing_evidence(spec: CheckSpec, outcomes: dict[str, Outcome]) -> Evaluation:
|
||
|
|
"""Summarise a Switchyard routing log tail into provider/effort coverage.
|
||
|
|
|
||
|
|
Route identifiers are ``<lane>/<provider>/<family>/<effort>``, so provider,
|
||
|
|
model family, effort tier, and lane all come from the same field. Fallback
|
||
|
|
and failure evidence is a separate ``fallback_reason`` on the record that
|
||
|
|
was retried. The parser tolerates a clipped first line: a tail of a growing
|
||
|
|
file starts mid-record, and discarding one row is better than a NOT_RUN that
|
||
|
|
reads as an outage.
|
||
|
|
"""
|
||
|
|
blocked = missing_steps(spec, outcomes)
|
||
|
|
if blocked:
|
||
|
|
return blocked
|
||
|
|
key = spec.expect["step"]
|
||
|
|
outcome = outcomes[key]
|
||
|
|
if not outcome.ok:
|
||
|
|
return Evaluation(NOT_RUN, f"step {key} failed (rc={outcome.returncode})")
|
||
|
|
records = []
|
||
|
|
for line in outcome.stdout.splitlines():
|
||
|
|
line = line.strip()
|
||
|
|
if not line:
|
||
|
|
continue
|
||
|
|
try:
|
||
|
|
records.append(json.loads(line))
|
||
|
|
except ValueError:
|
||
|
|
continue
|
||
|
|
routes = [
|
||
|
|
str(record.get("model") or "").split("/")
|
||
|
|
for record in records
|
||
|
|
if str(record.get("model") or "").count("/") == 3
|
||
|
|
]
|
||
|
|
fallbacks = [record for record in records if record.get("fallback_reason")]
|
||
|
|
stamps = sorted(str(record.get("ts")) for record in records if record.get("ts"))
|
||
|
|
evidence: dict[str, Any] = {
|
||
|
|
"records": len(records),
|
||
|
|
"routed": len(routes),
|
||
|
|
"lanes": sorted({route[0] for route in routes}),
|
||
|
|
"providers": sorted({route[1] for route in routes}),
|
||
|
|
"families": sorted({route[2] for route in routes}),
|
||
|
|
"efforts": sorted({route[3] for route in routes}),
|
||
|
|
"fallback_events": len(fallbacks),
|
||
|
|
"fallback_reasons": sorted({str(record["fallback_reason"]) for record in fallbacks}),
|
||
|
|
"latest_ts": stamps[-1] if stamps else "",
|
||
|
|
}
|
||
|
|
if not records:
|
||
|
|
return Evaluation(NOT_RUN, "routing log tail contained no parsable records", evidence)
|
||
|
|
missing = [
|
||
|
|
f"{label}:{item}"
|
||
|
|
for label, wanted in (
|
||
|
|
("provider", spec.expect.get("providers", ())),
|
||
|
|
("effort", spec.expect.get("efforts", ())),
|
||
|
|
("lane", spec.expect.get("lanes", ())),
|
||
|
|
)
|
||
|
|
for item in wanted
|
||
|
|
if item not in evidence[f"{label}s"]
|
||
|
|
]
|
||
|
|
if spec.expect.get("require_fallback_evidence") and not fallbacks:
|
||
|
|
missing.append("fallback:none-observed")
|
||
|
|
max_age = spec.expect.get("max_age_seconds")
|
||
|
|
now = spec.expect.get("now")
|
||
|
|
if max_age and now:
|
||
|
|
try:
|
||
|
|
newest = dt.datetime.fromisoformat(evidence["latest_ts"].replace("Z", "+00:00"))
|
||
|
|
except ValueError:
|
||
|
|
return Evaluation(NOT_RUN, "routing log has no parsable newest timestamp", evidence)
|
||
|
|
age = (now - newest).total_seconds()
|
||
|
|
evidence["age_seconds"] = int(age)
|
||
|
|
if age > max_age:
|
||
|
|
missing.append(f"freshness:{int(age)}s-exceeds-{max_age}s")
|
||
|
|
if missing:
|
||
|
|
return Evaluation(FAIL, f"routing evidence missing {', '.join(missing)}", evidence)
|
||
|
|
return Evaluation(PASS, "routing evidence covers the required lanes", evidence)
|
||
|
|
|
||
|
|
|
||
|
|
@evaluator("flux_health")
|
||
|
|
def evaluate_flux_health(spec: CheckSpec, outcomes: dict[str, Outcome]) -> Evaluation:
|
||
|
|
"""Classify a compact Flux projection into suspensions and unhealthy objects.
|
||
|
|
|
||
|
|
Each row is ``namespace/name<TAB>suspend<TAB>Type=Status,...``. An object is
|
||
|
|
unhealthy when it reports ``Healthy=False``, or ``Ready=False`` with no
|
||
|
|
Healthy condition and no in-flight ``Reconciling=True``. That last clause
|
||
|
|
matters: Flux flips Ready to False for the duration of a reconcile, so
|
||
|
|
without it the verdict depends on when the snapshot was taken. Suspended
|
||
|
|
objects are excluded from the health tally and judged against the
|
||
|
|
expected-suspension list instead, because a deliberate park is a different
|
||
|
|
fact from a broken reconcile.
|
||
|
|
"""
|
||
|
|
blocked = missing_steps(spec, outcomes)
|
||
|
|
if blocked:
|
||
|
|
return blocked
|
||
|
|
key = spec.expect["step"]
|
||
|
|
outcome = outcomes[key]
|
||
|
|
if not outcome.ok:
|
||
|
|
return Evaluation(NOT_RUN, f"step {key} failed (rc={outcome.returncode})")
|
||
|
|
expected = set(spec.expect.get("expected_suspensions", ()))
|
||
|
|
suspended: list[str] = []
|
||
|
|
unhealthy: list[str] = []
|
||
|
|
total = 0
|
||
|
|
for line in outcome.stdout.splitlines():
|
||
|
|
if not line.strip():
|
||
|
|
continue
|
||
|
|
total += 1
|
||
|
|
name, _, rest = line.partition("\t")
|
||
|
|
suspend, _, conditions = rest.partition("\t")
|
||
|
|
if suspend.strip().lower() == "true":
|
||
|
|
suspended.append(name.strip())
|
||
|
|
continue
|
||
|
|
states = dict(
|
||
|
|
pair.split("=", 1) for pair in conditions.strip().strip(",").split(",") if "=" in pair
|
||
|
|
)
|
||
|
|
healthy = states.get("Healthy")
|
||
|
|
reconciling = states.get("Reconciling") == "True"
|
||
|
|
ready_failed = states.get("Ready") == "False"
|
||
|
|
if healthy == "False" or (healthy is None and ready_failed and not reconciling):
|
||
|
|
unhealthy.append(name.strip())
|
||
|
|
unexpected = sorted(set(suspended) - expected)
|
||
|
|
evidence = {
|
||
|
|
"total": total,
|
||
|
|
"suspended": sorted(suspended),
|
||
|
|
"unexpected_suspensions": unexpected,
|
||
|
|
"unhealthy": sorted(unhealthy),
|
||
|
|
}
|
||
|
|
if not total:
|
||
|
|
return Evaluation(NOT_RUN, f"step {key} returned no Flux objects", evidence)
|
||
|
|
problems = []
|
||
|
|
if unexpected:
|
||
|
|
problems.append(f"unexpected suspensions: {', '.join(unexpected)}")
|
||
|
|
if unhealthy:
|
||
|
|
problems.append(f"unhealthy: {', '.join(sorted(unhealthy)[:8])}")
|
||
|
|
if problems:
|
||
|
|
return Evaluation(FAIL, "; ".join(problems), evidence)
|
||
|
|
return Evaluation(PASS, f"{total} Flux objects reconciled as expected", evidence)
|