Repairs the blockers from the independent review of the previous head. The role-aware verdict contract was correct but reachable only through one call site and only on cards written with real newlines, so most real review cards never used it. * The role-blind call is no longer a weaker classifier that can reject before the role-aware judge runs. Without a card there is no defensible role-dependent judgement, so `unfinished_result_reason()` applies only the card-independent checks. That closes the short-circuit at every call site, including the one PR15 moves to `cli_lane_execution`, and it makes a journalled terminal record accepted under one version of these semantics re-validate under any other instead of being quarantined into a re-dispatch of an already-accepted task. * The lane resolves the role from the card and passes it, and records it in the Kanban metadata. The verdict contract binds only where the lane can buy another turn: in single-shot mode a rejection discards the worker's real result, so review cards keep the relaxation without gaining any rejection single-shot mode did not already have. * Card scope expands the literal \n escapes the board stores in one-line bodies, so explicit `Hermes-Task-Role` / `Hermes-Expected-Output` directives are honoured on the 6 of 78 live cards that carry no real newline, and the read-only, verdict and mutation heuristics stop being cut apart by them. * Card scope now ends at the first non-card H2 and at the runner's controller evidence, which is emitted under its own heading. Goal-controller rejection history can no longer sit inside the card, and an upstream heading rename fails closed instead of admitting history into role resolution. * The inference recognises the SHIP/BLOCK-shaped deliverables real cards actually use: 21 of 78 live cards resolve to review, up from 10, with no implementation card misclassified. Cards asking for a findings list rather than a verdict deliberately stay on the model judge, since the verdict is the review contract's only gate. * A declared review role no longer outranks mutation evidence: a report that changed files falls back to the implementation regime. * Judge reasons go through the agent runtime's canonical redactor, extended for the two shapes it deliberately passes through and this lane handles - `scheme://user:secret@host` and a credential named in prose - while a 40-hex commit SHA survives as evidence. Regressions cover the recovered t_dbdcd739 incident, a verbatim snapshot of every live card on three boards with a hand-labelled expected role, the upgrade and single-shot properties against the previous gate, the upstream context-heading contract, and the end-to-end `execute_claim` shape that used to burn every goal turn. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
121 lines
3.5 KiB
Python
121 lines
3.5 KiB
Python
"""Completion-gate tests for Hermes external Kanban workers."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import importlib.util
|
|
import json
|
|
import sys
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
|
|
SCRIPT = (
|
|
Path(__file__).parents[2]
|
|
/ "services/hermes/scripts/cli_lane_goal.py"
|
|
)
|
|
SPEC = importlib.util.spec_from_file_location("cli_lane_goal_test", SCRIPT)
|
|
assert SPEC and SPEC.loader
|
|
goal = importlib.util.module_from_spec(SPEC)
|
|
sys.modules[SPEC.name] = goal
|
|
SPEC.loader.exec_module(goal)
|
|
|
|
|
|
def _result(**overrides):
|
|
value = {
|
|
"status": "completed",
|
|
"summary": "All acceptance criteria passed and the branch was pushed.",
|
|
"changed_files": ["src/example.py"],
|
|
"tests_run": ["pytest -q: 12 passed"],
|
|
"artifacts": [],
|
|
"blockers": [],
|
|
}
|
|
value.update(overrides)
|
|
return value
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"result",
|
|
[
|
|
_result(status="incomplete", summary="The full suite is still running."),
|
|
_result(summary="The broad rerun remains active before the final push."),
|
|
_result(tests_run=["Full suite — in progress (11m)"]),
|
|
_result(blockers=["remote head was not verified"]),
|
|
],
|
|
)
|
|
def test_unfinished_completion_evidence_is_rejected(result):
|
|
assert goal.unfinished_result_reason(result, role=goal.IMPLEMENTATION_ROLE)
|
|
|
|
|
|
def test_completed_pending_state_test_name_is_not_a_false_positive():
|
|
result = _result(tests_run=["pytest tests/test_pending_build_state.py: 8 passed"])
|
|
|
|
assert goal.unfinished_result_reason(result, role=goal.IMPLEMENTATION_ROLE) is None
|
|
|
|
|
|
def test_completed_review_findings_are_not_task_blockers():
|
|
result = _result(
|
|
summary="Review complete; the pull request is not merge-ready.",
|
|
findings=["A casting-table variant fails open."],
|
|
)
|
|
|
|
assert goal.unfinished_result_reason(result, role=goal.IMPLEMENTATION_ROLE) is None
|
|
|
|
|
|
class JudgeResponse:
|
|
def __init__(self, verdict: str, reason: str):
|
|
self.body = json.dumps(
|
|
{
|
|
"choices": [
|
|
{
|
|
"message": {
|
|
"content": json.dumps(
|
|
{"verdict": verdict, "reason": reason}
|
|
)
|
|
}
|
|
}
|
|
]
|
|
}
|
|
).encode()
|
|
|
|
def __enter__(self):
|
|
return self
|
|
|
|
def __exit__(self, *_args):
|
|
return False
|
|
|
|
def read(self):
|
|
return self.body
|
|
|
|
|
|
def test_goal_judge_uses_local_structured_verdict():
|
|
observed = {}
|
|
|
|
def request(req, timeout):
|
|
observed["payload"] = json.loads(req.data)
|
|
observed["timeout"] = timeout
|
|
return JudgeResponse("continue", "remote head evidence is missing")
|
|
|
|
accepted, reason = goal.judge_goal_completion(
|
|
"Run tests, push, and verify the remote head.",
|
|
_result(summary="Tests passed."),
|
|
open_request=request,
|
|
)
|
|
|
|
assert accepted is False
|
|
assert reason == "remote head evidence is missing"
|
|
assert observed["payload"]["model"] == "qwen2.5:14b-instruct-q4_0"
|
|
assert observed["payload"]["response_format"]["json_schema"]["strict"] is True
|
|
assert observed["timeout"] == 120
|
|
|
|
|
|
def test_goal_judge_fails_closed_on_invalid_response():
|
|
accepted, reason = goal.judge_goal_completion(
|
|
"Finish the task.",
|
|
_result(),
|
|
open_request=lambda *_args, **_kwargs: JudgeResponse("unknown", "bad"),
|
|
)
|
|
|
|
assert accepted is False
|
|
assert "local completion judge unavailable" in reason
|