hermes: bind terminal commit to journal inode
This commit is contained in:
parent
3ce59ee1b4
commit
2caad7e769
@ -5,6 +5,7 @@ from __future__ import annotations
|
|||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
import tempfile
|
import tempfile
|
||||||
|
import threading
|
||||||
import unittest
|
import unittest
|
||||||
from types import SimpleNamespace
|
from types import SimpleNamespace
|
||||||
from unittest import mock
|
from unittest import mock
|
||||||
@ -160,16 +161,36 @@ class AutomaticDecompositionSafetyTests(unittest.TestCase):
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
self.assertTrue(
|
with mock.patch.object(
|
||||||
kanban_db.complete_task(
|
kanban_db, "_fire_kanban_lifecycle_hook"
|
||||||
self.connection,
|
) as lifecycle:
|
||||||
task_id,
|
self.assertTrue(
|
||||||
result="durable terminal result",
|
kanban_db.complete_task(
|
||||||
summary="durable terminal result",
|
self.connection,
|
||||||
replay_ended_run_id=run_id,
|
task_id,
|
||||||
|
result="durable terminal result",
|
||||||
|
summary="durable terminal result",
|
||||||
|
replay_ended_run_id=run_id,
|
||||||
|
)
|
||||||
)
|
)
|
||||||
)
|
lifecycle.assert_called_once()
|
||||||
self.assertEqual(self._status(task_id), "done")
|
self.assertEqual(lifecycle.call_args.args[:2], ("kanban_task_completed", task_id))
|
||||||
|
self.assertEqual(lifecycle.call_args.kwargs["run_id"], run_id)
|
||||||
|
task = kanban_db.get_task(self.connection, task_id)
|
||||||
|
self.assertEqual(task.status, "done")
|
||||||
|
self.assertEqual(task.completed_run_id, run_id)
|
||||||
|
runs = kanban_db.list_runs(self.connection, task_id)
|
||||||
|
self.assertEqual(len(runs), 1)
|
||||||
|
self.assertEqual(runs[0].id, run_id)
|
||||||
|
self.assertEqual(runs[0].status, "done")
|
||||||
|
self.assertEqual(runs[0].outcome, "completed")
|
||||||
|
self.assertEqual(runs[0].summary, "durable terminal result")
|
||||||
|
completed_events = self.connection.execute(
|
||||||
|
"SELECT run_id FROM task_events "
|
||||||
|
"WHERE task_id = ? AND kind = 'completed' ORDER BY id",
|
||||||
|
(task_id,),
|
||||||
|
).fetchall()
|
||||||
|
self.assertEqual([row["run_id"] for row in completed_events], [run_id])
|
||||||
|
|
||||||
def test_older_ended_run_cannot_complete_over_a_replacement(self) -> None:
|
def test_older_ended_run_cannot_complete_over_a_replacement(self) -> None:
|
||||||
task_id = kanban_db.create_task(self.connection, title="replacement guard")
|
task_id = kanban_db.create_task(self.connection, title="replacement guard")
|
||||||
@ -208,6 +229,179 @@ class AutomaticDecompositionSafetyTests(unittest.TestCase):
|
|||||||
)
|
)
|
||||||
self.assertIn(self._status(task_id), {"blocked", "triage"})
|
self.assertIn(self._status(task_id), {"blocked", "triage"})
|
||||||
|
|
||||||
|
def test_equal_result_bytes_remain_bound_to_the_completing_run(self) -> None:
|
||||||
|
task_id = kanban_db.create_task(self.connection, title="same result retries")
|
||||||
|
self.assertIsNotNone(kanban_db.claim_task(self.connection, task_id))
|
||||||
|
first_run = kanban_db.get_task(self.connection, task_id).current_run_id
|
||||||
|
self.assertTrue(
|
||||||
|
kanban_db.block_task(
|
||||||
|
self.connection,
|
||||||
|
task_id,
|
||||||
|
reason="first attempt ended",
|
||||||
|
expected_run_id=first_run,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
self.assertTrue(kanban_db.unblock_task(self.connection, task_id))
|
||||||
|
self.assertIsNotNone(kanban_db.claim_task(self.connection, task_id))
|
||||||
|
second_run = kanban_db.get_task(self.connection, task_id).current_run_id
|
||||||
|
self.assertTrue(
|
||||||
|
kanban_db.complete_task(
|
||||||
|
self.connection,
|
||||||
|
task_id,
|
||||||
|
result="identical result bytes",
|
||||||
|
summary="second run wins",
|
||||||
|
expected_run_id=second_run,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
task = kanban_db.get_task(self.connection, task_id)
|
||||||
|
self.assertEqual(task.completed_run_id, second_run)
|
||||||
|
self.assertNotEqual(task.completed_run_id, first_run)
|
||||||
|
runs = {run.id: run for run in kanban_db.list_runs(self.connection, task_id)}
|
||||||
|
self.assertEqual(runs[first_run].outcome, "blocked")
|
||||||
|
self.assertEqual(runs[second_run].outcome, "completed")
|
||||||
|
|
||||||
|
def test_exact_run_reclaim_succeeds_for_the_authoritative_run(self) -> None:
|
||||||
|
task_id = kanban_db.create_task(self.connection, title="recover exact run")
|
||||||
|
self.assertIsNotNone(kanban_db.claim_task(self.connection, task_id))
|
||||||
|
run_id = kanban_db.get_task(self.connection, task_id).current_run_id
|
||||||
|
|
||||||
|
self.assertTrue(
|
||||||
|
kanban_db.reclaim_task(
|
||||||
|
self.connection,
|
||||||
|
task_id,
|
||||||
|
reason="invalid exact journal",
|
||||||
|
expected_run_id=run_id,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
task = kanban_db.get_task(self.connection, task_id)
|
||||||
|
self.assertEqual(task.status, "ready")
|
||||||
|
self.assertIsNone(task.current_run_id)
|
||||||
|
|
||||||
|
def test_stale_reclaim_before_transaction_preserves_replacement_run(self) -> None:
|
||||||
|
task_id = kanban_db.create_task(self.connection, title="replacement before txn")
|
||||||
|
self.assertIsNotNone(kanban_db.claim_task(self.connection, task_id))
|
||||||
|
old_run = kanban_db.get_task(self.connection, task_id).current_run_id
|
||||||
|
self.assertTrue(
|
||||||
|
kanban_db.block_task(
|
||||||
|
self.connection,
|
||||||
|
task_id,
|
||||||
|
reason="old recovery run",
|
||||||
|
expected_run_id=old_run,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
self.assertTrue(kanban_db.unblock_task(self.connection, task_id))
|
||||||
|
self.assertIsNotNone(kanban_db.claim_task(self.connection, task_id))
|
||||||
|
replacement = kanban_db.get_task(self.connection, task_id)
|
||||||
|
replacement_run = replacement.current_run_id
|
||||||
|
self.assertNotEqual(old_run, replacement_run)
|
||||||
|
signals = []
|
||||||
|
|
||||||
|
self.assertFalse(
|
||||||
|
kanban_db.reclaim_task(
|
||||||
|
self.connection,
|
||||||
|
task_id,
|
||||||
|
reason="stale journal",
|
||||||
|
expected_run_id=old_run,
|
||||||
|
signal_fn=lambda *args: signals.append(args),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
latest = kanban_db.get_task(self.connection, task_id)
|
||||||
|
self.assertEqual(latest.status, "running")
|
||||||
|
self.assertEqual(latest.current_run_id, replacement_run)
|
||||||
|
self.assertEqual(latest.claim_lock, replacement.claim_lock)
|
||||||
|
self.assertEqual(signals, [])
|
||||||
|
|
||||||
|
def test_reclaim_update_guard_preserves_run_changed_inside_transaction(self) -> None:
|
||||||
|
task_id = kanban_db.create_task(self.connection, title="replacement in txn")
|
||||||
|
self.assertIsNotNone(kanban_db.claim_task(self.connection, task_id))
|
||||||
|
old_run = kanban_db.get_task(self.connection, task_id).current_run_id
|
||||||
|
replacement = {}
|
||||||
|
|
||||||
|
def install_replacement(*_args, **_kwargs):
|
||||||
|
cursor = self.connection.execute(
|
||||||
|
"INSERT INTO task_runs (task_id, status, claim_lock, started_at) "
|
||||||
|
"VALUES (?, 'running', ?, strftime('%s','now'))",
|
||||||
|
(task_id, "replacement-lock"),
|
||||||
|
)
|
||||||
|
replacement["run_id"] = int(cursor.lastrowid)
|
||||||
|
self.connection.execute(
|
||||||
|
"UPDATE tasks SET current_run_id = ?, claim_lock = ? WHERE id = ?",
|
||||||
|
(replacement["run_id"], "replacement-lock", task_id),
|
||||||
|
)
|
||||||
|
return {}
|
||||||
|
|
||||||
|
with mock.patch.object(
|
||||||
|
kanban_db,
|
||||||
|
"_terminate_reclaimed_worker",
|
||||||
|
side_effect=install_replacement,
|
||||||
|
):
|
||||||
|
self.assertFalse(
|
||||||
|
kanban_db.reclaim_task(
|
||||||
|
self.connection,
|
||||||
|
task_id,
|
||||||
|
reason="journal for old run",
|
||||||
|
expected_run_id=old_run,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
latest = kanban_db.get_task(self.connection, task_id)
|
||||||
|
self.assertEqual(latest.status, "running")
|
||||||
|
self.assertEqual(latest.current_run_id, replacement["run_id"])
|
||||||
|
self.assertEqual(latest.claim_lock, "replacement-lock")
|
||||||
|
|
||||||
|
def test_concurrent_exact_reclaim_and_completion_have_one_winner(self) -> None:
|
||||||
|
task_id = kanban_db.create_task(self.connection, title="concurrent finalizer")
|
||||||
|
self.assertIsNotNone(kanban_db.claim_task(self.connection, task_id))
|
||||||
|
run_id = kanban_db.get_task(self.connection, task_id).current_run_id
|
||||||
|
barrier = threading.Barrier(2)
|
||||||
|
outcomes = {}
|
||||||
|
errors = []
|
||||||
|
|
||||||
|
def reclaim() -> None:
|
||||||
|
try:
|
||||||
|
with kanban_db.connect_closing() as connection:
|
||||||
|
barrier.wait()
|
||||||
|
outcomes["reclaim"] = kanban_db.reclaim_task(
|
||||||
|
connection,
|
||||||
|
task_id,
|
||||||
|
reason="concurrent invalid journal",
|
||||||
|
expected_run_id=run_id,
|
||||||
|
)
|
||||||
|
except BaseException as error: # pragma: no cover - assertion relay
|
||||||
|
errors.append(error)
|
||||||
|
|
||||||
|
def finalize() -> None:
|
||||||
|
try:
|
||||||
|
with kanban_db.connect_closing() as connection:
|
||||||
|
barrier.wait()
|
||||||
|
outcomes["complete"] = kanban_db.complete_task(
|
||||||
|
connection,
|
||||||
|
task_id,
|
||||||
|
result="durable winner",
|
||||||
|
summary="durable winner",
|
||||||
|
expected_run_id=run_id,
|
||||||
|
)
|
||||||
|
except BaseException as error: # pragma: no cover - assertion relay
|
||||||
|
errors.append(error)
|
||||||
|
|
||||||
|
threads = [threading.Thread(target=reclaim), threading.Thread(target=finalize)]
|
||||||
|
for thread in threads:
|
||||||
|
thread.start()
|
||||||
|
for thread in threads:
|
||||||
|
thread.join(timeout=10)
|
||||||
|
self.assertFalse(thread.is_alive())
|
||||||
|
|
||||||
|
self.assertEqual(errors, [])
|
||||||
|
self.assertEqual(set(outcomes), {"reclaim", "complete"})
|
||||||
|
self.assertEqual(sum(bool(value) for value in outcomes.values()), 1)
|
||||||
|
latest = kanban_db.get_task(self.connection, task_id)
|
||||||
|
self.assertIn(latest.status, {"done", "ready"})
|
||||||
|
self.assertIsNone(latest.current_run_id)
|
||||||
|
runs = kanban_db.list_runs(self.connection, task_id)
|
||||||
|
self.assertEqual(len(runs), 1)
|
||||||
|
self.assertIsNotNone(runs[0].ended_at)
|
||||||
|
|
||||||
def test_fresh_triage_task_still_auto_promotes(self) -> None:
|
def test_fresh_triage_task_still_auto_promotes(self) -> None:
|
||||||
task_id = kanban_db.create_task(
|
task_id = kanban_db.create_task(
|
||||||
self.connection,
|
self.connection,
|
||||||
|
|||||||
@ -72,9 +72,11 @@ complete_branch_after = ''' # Journal recovery may arrive after a legacy erro
|
|||||||
# accepted run. Verify that it is still the latest run in this same
|
# accepted run. Verify that it is still the latest run in this same
|
||||||
# transaction; a replacement attempt must make the replay fail closed.
|
# transaction; a replacement attempt must make the replay fail closed.
|
||||||
with write_txn(conn):
|
with write_txn(conn):
|
||||||
|
replayed_run_id = None
|
||||||
if expected_run_id is not None and replay_ended_run_id is not None:
|
if expected_run_id is not None and replay_ended_run_id is not None:
|
||||||
raise ValueError("expected_run_id and replay_ended_run_id are mutually exclusive")
|
raise ValueError("expected_run_id and replay_ended_run_id are mutually exclusive")
|
||||||
if replay_ended_run_id is not None:
|
if replay_ended_run_id is not None:
|
||||||
|
replayed_run_id = int(replay_ended_run_id)
|
||||||
cur = conn.execute(
|
cur = conn.execute(
|
||||||
"""
|
"""
|
||||||
UPDATE tasks
|
UPDATE tasks
|
||||||
@ -95,8 +97,22 @@ complete_branch_after = ''' # Journal recovery may arrive after a legacy erro
|
|||||||
ORDER BY id DESC
|
ORDER BY id DESC
|
||||||
LIMIT 1
|
LIMIT 1
|
||||||
)
|
)
|
||||||
|
AND EXISTS (
|
||||||
|
SELECT 1 FROM task_runs
|
||||||
|
WHERE id = ?
|
||||||
|
AND task_id = ?
|
||||||
|
AND ended_at IS NOT NULL
|
||||||
|
)
|
||||||
""",
|
""",
|
||||||
(result, now, task_id, int(replay_ended_run_id), task_id),
|
(
|
||||||
|
result,
|
||||||
|
now,
|
||||||
|
task_id,
|
||||||
|
replayed_run_id,
|
||||||
|
task_id,
|
||||||
|
replayed_run_id,
|
||||||
|
task_id,
|
||||||
|
),
|
||||||
)
|
)
|
||||||
elif expected_run_id is None:
|
elif expected_run_id is None:
|
||||||
cur = conn.execute(
|
cur = conn.execute(
|
||||||
@ -124,6 +140,315 @@ db = replace_once(
|
|||||||
"transactional exact ended-run completion",
|
"transactional exact ended-run completion",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
run_completion_before = ''' run_id = _end_run(
|
||||||
|
conn, task_id,
|
||||||
|
outcome="completed", status="done",
|
||||||
|
summary=summary if summary is not None else result,
|
||||||
|
metadata=metadata,
|
||||||
|
)
|
||||||
|
# If complete_task was called on a never-claimed task (ready or
|
||||||
|
# blocked → done with no run in flight), synthesize a
|
||||||
|
# zero-duration run so the handoff fields are persisted in
|
||||||
|
# attempt history instead of silently lost.
|
||||||
|
if run_id is None and (summary or metadata or result):
|
||||||
|
run_id = _synthesize_ended_run(
|
||||||
|
conn, task_id,
|
||||||
|
outcome="completed",
|
||||||
|
summary=summary if summary is not None else result,
|
||||||
|
metadata=metadata,
|
||||||
|
)
|
||||||
|
'''
|
||||||
|
run_completion_after = ''' if replayed_run_id is None:
|
||||||
|
run_id = _end_run(
|
||||||
|
conn, task_id,
|
||||||
|
outcome="completed", status="done",
|
||||||
|
summary=summary if summary is not None else result,
|
||||||
|
metadata=metadata,
|
||||||
|
)
|
||||||
|
# If complete_task was called on a never-claimed task (ready or
|
||||||
|
# blocked → done with no run in flight), synthesize a
|
||||||
|
# zero-duration run so the handoff fields are persisted in
|
||||||
|
# attempt history instead of silently lost.
|
||||||
|
if run_id is None and (summary or metadata or result):
|
||||||
|
run_id = _synthesize_ended_run(
|
||||||
|
conn, task_id,
|
||||||
|
outcome="completed",
|
||||||
|
summary=summary if summary is not None else result,
|
||||||
|
metadata=metadata,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
run_id = replayed_run_id
|
||||||
|
replayed = conn.execute(
|
||||||
|
"""
|
||||||
|
UPDATE task_runs
|
||||||
|
SET status = 'done',
|
||||||
|
outcome = 'completed',
|
||||||
|
summary = ?,
|
||||||
|
error = NULL,
|
||||||
|
metadata = ?,
|
||||||
|
claim_lock = NULL,
|
||||||
|
claim_expires = NULL,
|
||||||
|
worker_pid = NULL
|
||||||
|
WHERE id = ?
|
||||||
|
AND task_id = ?
|
||||||
|
AND ended_at IS NOT NULL
|
||||||
|
""",
|
||||||
|
(
|
||||||
|
summary if summary is not None else result,
|
||||||
|
json.dumps(metadata, ensure_ascii=False) if metadata else None,
|
||||||
|
run_id,
|
||||||
|
task_id,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
if replayed.rowcount != 1:
|
||||||
|
raise RuntimeError("exact ended-run completion lost its run identity")
|
||||||
|
conn.execute(
|
||||||
|
"UPDATE tasks SET completed_run_id = ? WHERE id = ?",
|
||||||
|
(run_id, task_id),
|
||||||
|
)
|
||||||
|
'''
|
||||||
|
db = replace_once(
|
||||||
|
db,
|
||||||
|
run_completion_before,
|
||||||
|
run_completion_after,
|
||||||
|
"exact completion run provenance",
|
||||||
|
)
|
||||||
|
|
||||||
|
reclaim_before = '''def reclaim_task(
|
||||||
|
conn: sqlite3.Connection,
|
||||||
|
task_id: str,
|
||||||
|
*,
|
||||||
|
reason: Optional[str] = None,
|
||||||
|
signal_fn=None,
|
||||||
|
) -> bool:
|
||||||
|
"""Operator-driven reclaim: release the claim and reset to ``ready``.
|
||||||
|
|
||||||
|
Unlike :func:`release_stale_claims` which only acts on tasks whose
|
||||||
|
``claim_expires`` has passed, this function reclaims immediately
|
||||||
|
regardless of TTL. Intended for the dashboard/CLI recovery flow
|
||||||
|
when an operator wants to abort a running worker without waiting
|
||||||
|
for the TTL to expire (e.g. after seeing a hallucination warning).
|
||||||
|
|
||||||
|
Returns True if a reclaim happened, False if the task isn't in a
|
||||||
|
reclaimable state (not running, or doesn't exist).
|
||||||
|
"""
|
||||||
|
row = conn.execute(
|
||||||
|
"SELECT status, claim_lock, worker_pid FROM tasks WHERE id = ?",
|
||||||
|
(task_id,),
|
||||||
|
).fetchone()
|
||||||
|
if not row:
|
||||||
|
return False
|
||||||
|
if row["status"] != "running" and row["claim_lock"] is None:
|
||||||
|
# Nothing to reclaim — already ready / blocked / done.
|
||||||
|
return False
|
||||||
|
prev_lock = row["claim_lock"]
|
||||||
|
termination = _terminate_reclaimed_worker(
|
||||||
|
row["worker_pid"], prev_lock, signal_fn=signal_fn,
|
||||||
|
)
|
||||||
|
with write_txn(conn):
|
||||||
|
cur = conn.execute(
|
||||||
|
"UPDATE tasks SET status = 'ready', claim_lock = NULL, "
|
||||||
|
"claim_expires = NULL, worker_pid = NULL "
|
||||||
|
"WHERE id = ? AND status IN ('running', 'ready', 'blocked') "
|
||||||
|
"AND claim_lock IS ?",
|
||||||
|
(task_id, prev_lock),
|
||||||
|
)
|
||||||
|
if cur.rowcount != 1:
|
||||||
|
return False
|
||||||
|
run_id = _end_run(
|
||||||
|
conn, task_id,
|
||||||
|
outcome="reclaimed", status="reclaimed",
|
||||||
|
error=(
|
||||||
|
f"manual_reclaim: {reason}" if reason
|
||||||
|
else f"manual_reclaim lock={prev_lock}"
|
||||||
|
),
|
||||||
|
metadata=termination,
|
||||||
|
)
|
||||||
|
payload = {
|
||||||
|
"manual": True,
|
||||||
|
"reason": reason,
|
||||||
|
"prev_lock": prev_lock,
|
||||||
|
}
|
||||||
|
payload.update(termination)
|
||||||
|
_append_event(
|
||||||
|
conn, task_id, "reclaimed",
|
||||||
|
payload,
|
||||||
|
run_id=run_id,
|
||||||
|
)
|
||||||
|
# Operator intervention — they've looked at the task, so the
|
||||||
|
# consecutive-failures counter is now stale. Give the next retry
|
||||||
|
# a fresh budget. (_clear_failure_counter opens its own write_txn,
|
||||||
|
# so it runs after the enclosing one commits.)
|
||||||
|
_clear_failure_counter(conn, task_id)
|
||||||
|
return True
|
||||||
|
'''
|
||||||
|
reclaim_after = '''def reclaim_task(
|
||||||
|
conn: sqlite3.Connection,
|
||||||
|
task_id: str,
|
||||||
|
*,
|
||||||
|
reason: Optional[str] = None,
|
||||||
|
signal_fn=None,
|
||||||
|
expected_run_id: Optional[int] = None,
|
||||||
|
) -> bool:
|
||||||
|
"""Operator-driven reclaim: release the claim and reset to ``ready``.
|
||||||
|
|
||||||
|
Unlike :func:`release_stale_claims` which only acts on tasks whose
|
||||||
|
``claim_expires`` has passed, this function reclaims immediately
|
||||||
|
regardless of TTL. Intended for the dashboard/CLI recovery flow
|
||||||
|
when an operator wants to abort a running worker without waiting
|
||||||
|
for the TTL to expire (e.g. after seeing a hallucination warning).
|
||||||
|
|
||||||
|
When ``expected_run_id`` is supplied, the run identity is checked after
|
||||||
|
``BEGIN IMMEDIATE`` and included in the guarded update. A replacement run
|
||||||
|
is therefore neither signalled nor reclaimed by stale recovery evidence.
|
||||||
|
|
||||||
|
Returns True if a reclaim happened, False if the task isn't in a
|
||||||
|
reclaimable state (not running, or doesn't exist), or its run changed.
|
||||||
|
"""
|
||||||
|
with write_txn(conn):
|
||||||
|
row = conn.execute(
|
||||||
|
"SELECT status, claim_lock, worker_pid, current_run_id "
|
||||||
|
"FROM tasks WHERE id = ?",
|
||||||
|
(task_id,),
|
||||||
|
).fetchone()
|
||||||
|
if not row:
|
||||||
|
return False
|
||||||
|
if expected_run_id is not None and row["current_run_id"] != expected_run_id:
|
||||||
|
return False
|
||||||
|
if row["status"] != "running" and row["claim_lock"] is None:
|
||||||
|
# Nothing to reclaim — already ready / blocked / done.
|
||||||
|
return False
|
||||||
|
prev_lock = row["claim_lock"]
|
||||||
|
termination = _terminate_reclaimed_worker(
|
||||||
|
row["worker_pid"], prev_lock, signal_fn=signal_fn,
|
||||||
|
)
|
||||||
|
if expected_run_id is None:
|
||||||
|
cur = conn.execute(
|
||||||
|
"UPDATE tasks SET status = 'ready', claim_lock = NULL, "
|
||||||
|
"claim_expires = NULL, worker_pid = NULL "
|
||||||
|
"WHERE id = ? AND status IN ('running', 'ready', 'blocked') "
|
||||||
|
"AND claim_lock IS ?",
|
||||||
|
(task_id, prev_lock),
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
cur = conn.execute(
|
||||||
|
"UPDATE tasks SET status = 'ready', claim_lock = NULL, "
|
||||||
|
"claim_expires = NULL, worker_pid = NULL "
|
||||||
|
"WHERE id = ? AND status IN ('running', 'ready', 'blocked') "
|
||||||
|
"AND claim_lock IS ? AND current_run_id = ?",
|
||||||
|
(task_id, prev_lock, int(expected_run_id)),
|
||||||
|
)
|
||||||
|
if cur.rowcount != 1:
|
||||||
|
return False
|
||||||
|
run_id = _end_run(
|
||||||
|
conn, task_id,
|
||||||
|
outcome="reclaimed", status="reclaimed",
|
||||||
|
error=(
|
||||||
|
f"manual_reclaim: {reason}" if reason
|
||||||
|
else f"manual_reclaim lock={prev_lock}"
|
||||||
|
),
|
||||||
|
metadata=termination,
|
||||||
|
)
|
||||||
|
payload = {
|
||||||
|
"manual": True,
|
||||||
|
"reason": reason,
|
||||||
|
"prev_lock": prev_lock,
|
||||||
|
}
|
||||||
|
payload.update(termination)
|
||||||
|
_append_event(
|
||||||
|
conn, task_id, "reclaimed",
|
||||||
|
payload,
|
||||||
|
run_id=run_id,
|
||||||
|
)
|
||||||
|
# Operator intervention — they've looked at the task, so the
|
||||||
|
# consecutive-failures counter is now stale. Give the next retry
|
||||||
|
# a fresh budget. (_clear_failure_counter opens its own write_txn,
|
||||||
|
# so it runs after the enclosing one commits.)
|
||||||
|
_clear_failure_counter(conn, task_id)
|
||||||
|
return True
|
||||||
|
'''
|
||||||
|
db = replace_once(
|
||||||
|
db,
|
||||||
|
reclaim_before,
|
||||||
|
reclaim_after,
|
||||||
|
"transactional exact-run reclaim",
|
||||||
|
)
|
||||||
|
|
||||||
|
task_field_before = ''' current_run_id: Optional[int] = None
|
||||||
|
workflow_template_id: Optional[str] = None
|
||||||
|
'''
|
||||||
|
task_field_after = ''' current_run_id: Optional[int] = None
|
||||||
|
# Durable provenance for the run that authoritatively completed this task.
|
||||||
|
completed_run_id: Optional[int] = None
|
||||||
|
workflow_template_id: Optional[str] = None
|
||||||
|
'''
|
||||||
|
db = replace_once(
|
||||||
|
db,
|
||||||
|
task_field_before,
|
||||||
|
task_field_after,
|
||||||
|
"completed-run task field",
|
||||||
|
)
|
||||||
|
|
||||||
|
task_row_before = ''' current_run_id=(
|
||||||
|
row["current_run_id"] if "current_run_id" in keys else None
|
||||||
|
),
|
||||||
|
workflow_template_id=(
|
||||||
|
'''
|
||||||
|
task_row_after = ''' current_run_id=(
|
||||||
|
row["current_run_id"] if "current_run_id" in keys else None
|
||||||
|
),
|
||||||
|
completed_run_id=(
|
||||||
|
row["completed_run_id"] if "completed_run_id" in keys else None
|
||||||
|
),
|
||||||
|
workflow_template_id=(
|
||||||
|
'''
|
||||||
|
db = replace_once(
|
||||||
|
db,
|
||||||
|
task_row_before,
|
||||||
|
task_row_after,
|
||||||
|
"completed-run row mapping",
|
||||||
|
)
|
||||||
|
|
||||||
|
task_schema_before = ''' -- run is in-flight). Denormalised for cheap reads.
|
||||||
|
current_run_id INTEGER,
|
||||||
|
-- Forward-compat for v2 workflow routing. In v1 the kernel writes
|
||||||
|
'''
|
||||||
|
task_schema_after = ''' -- run is in-flight). Denormalised for cheap reads.
|
||||||
|
current_run_id INTEGER,
|
||||||
|
-- Immutable provenance for the task result's authoritative run.
|
||||||
|
completed_run_id INTEGER,
|
||||||
|
-- Forward-compat for v2 workflow routing. In v1 the kernel writes
|
||||||
|
'''
|
||||||
|
db = replace_once(
|
||||||
|
db,
|
||||||
|
task_schema_before,
|
||||||
|
task_schema_after,
|
||||||
|
"completed-run schema",
|
||||||
|
)
|
||||||
|
|
||||||
|
task_migration_before = ''' if "current_run_id" not in cols:
|
||||||
|
_add_column_if_missing(
|
||||||
|
conn, "tasks", "current_run_id", "current_run_id INTEGER"
|
||||||
|
)
|
||||||
|
if "workflow_template_id" not in cols:
|
||||||
|
'''
|
||||||
|
task_migration_after = ''' if "current_run_id" not in cols:
|
||||||
|
_add_column_if_missing(
|
||||||
|
conn, "tasks", "current_run_id", "current_run_id INTEGER"
|
||||||
|
)
|
||||||
|
if "completed_run_id" not in cols:
|
||||||
|
_add_column_if_missing(
|
||||||
|
conn, "tasks", "completed_run_id", "completed_run_id INTEGER"
|
||||||
|
)
|
||||||
|
if "workflow_template_id" not in cols:
|
||||||
|
'''
|
||||||
|
db = replace_once(
|
||||||
|
db,
|
||||||
|
task_migration_before,
|
||||||
|
task_migration_after,
|
||||||
|
"completed-run migration",
|
||||||
|
)
|
||||||
|
|
||||||
specify_signature_before = '''def specify_triage_task(
|
specify_signature_before = '''def specify_triage_task(
|
||||||
conn: sqlite3.Connection,
|
conn: sqlite3.Connection,
|
||||||
task_id: str,
|
task_id: str,
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@ -24,10 +24,12 @@ def recover_running_tasks(kanban_db: Any) -> list[str]:
|
|||||||
if str(getattr(task, "status", "")) != "running":
|
if str(getattr(task, "status", "")) != "running":
|
||||||
continue
|
continue
|
||||||
task_id = str(getattr(task, "id", ""))
|
task_id = str(getattr(task, "id", ""))
|
||||||
if not task_id or not kanban_db.reclaim_task(
|
run_id = getattr(task, "current_run_id", None)
|
||||||
|
if not task_id or not isinstance(run_id, int) or not kanban_db.reclaim_task(
|
||||||
connection,
|
connection,
|
||||||
task_id,
|
task_id,
|
||||||
reason=REASON,
|
reason=REASON,
|
||||||
|
expected_run_id=run_id,
|
||||||
):
|
):
|
||||||
continue
|
continue
|
||||||
kanban_db.add_comment(
|
kanban_db.add_comment(
|
||||||
|
|||||||
File diff suppressed because it is too large
Load Diff
@ -42,7 +42,7 @@ class FakeKanban:
|
|||||||
self.tasks = tasks
|
self.tasks = tasks
|
||||||
self.has_board = board_exists
|
self.has_board = board_exists
|
||||||
self.connection = FakeConnection()
|
self.connection = FakeConnection()
|
||||||
self.reclaimed: list[tuple[str, str]] = []
|
self.reclaimed: list[tuple[str, str, int]] = []
|
||||||
self.comments: list[tuple[str, str, str]] = []
|
self.comments: list[tuple[str, str, str]] = []
|
||||||
|
|
||||||
def board_exists(self, board: str) -> bool:
|
def board_exists(self, board: str) -> bool:
|
||||||
@ -61,9 +61,16 @@ class FakeKanban:
|
|||||||
assert connection is self.connection
|
assert connection is self.connection
|
||||||
return self.tasks
|
return self.tasks
|
||||||
|
|
||||||
def reclaim_task(self, connection, task_id: str, *, reason: str) -> bool:
|
def reclaim_task(
|
||||||
|
self,
|
||||||
|
connection,
|
||||||
|
task_id: str,
|
||||||
|
*,
|
||||||
|
reason: str,
|
||||||
|
expected_run_id: int,
|
||||||
|
) -> bool:
|
||||||
assert connection is self.connection
|
assert connection is self.connection
|
||||||
self.reclaimed.append((task_id, reason))
|
self.reclaimed.append((task_id, reason, expected_run_id))
|
||||||
return task_id != "t_race"
|
return task_id != "t_race"
|
||||||
|
|
||||||
def add_comment(self, connection, task_id: str, author: str, body: str) -> None:
|
def add_comment(self, connection, task_id: str, author: str, body: str) -> None:
|
||||||
@ -75,16 +82,17 @@ def test_recover_running_tasks_requeues_only_claimed_workers() -> None:
|
|||||||
module = _load_module()
|
module = _load_module()
|
||||||
kanban = FakeKanban(
|
kanban = FakeKanban(
|
||||||
[
|
[
|
||||||
SimpleNamespace(id="t_running", status="running"),
|
SimpleNamespace(id="t_running", status="running", current_run_id=10),
|
||||||
SimpleNamespace(id="t_done", status="done"),
|
SimpleNamespace(id="t_done", status="done"),
|
||||||
SimpleNamespace(id="t_race", status="running"),
|
SimpleNamespace(id="t_race", status="running", current_run_id=11),
|
||||||
|
SimpleNamespace(id="t_missing_run", status="running", current_run_id=None),
|
||||||
]
|
]
|
||||||
)
|
)
|
||||||
|
|
||||||
assert module.recover_running_tasks(kanban) == ["t_running"]
|
assert module.recover_running_tasks(kanban) == ["t_running"]
|
||||||
assert kanban.reclaimed == [
|
assert kanban.reclaimed == [
|
||||||
("t_running", module.REASON),
|
("t_running", module.REASON, 10),
|
||||||
("t_race", module.REASON),
|
("t_race", module.REASON, 11),
|
||||||
]
|
]
|
||||||
assert kanban.comments == [
|
assert kanban.comments == [
|
||||||
(
|
(
|
||||||
|
|||||||
Loading…
x
Reference in New Issue
Block a user