201 lines
7.8 KiB
Python
201 lines
7.8 KiB
Python
#!/usr/bin/env python3
|
|
"""Atomically install the reviewed Soteria Kanban recovery candidate once."""
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import hashlib
|
|
import json
|
|
import os
|
|
import shutil
|
|
import sqlite3
|
|
import tempfile
|
|
import time
|
|
from pathlib import Path
|
|
|
|
|
|
SOURCE_SHA = "6ce2088512ae5bffe5a379aab00447086ab7cd74f4822b478702e55df1a9fb7d"
|
|
CANDIDATE_SHA = "a81c6dab96832578d9181db6b3935df5b7d1937569e5c59aa177bd19ead42e7e"
|
|
DEFAULT_DATABASE = Path("/opt/data/kanban/boards/soteria/kanban.db")
|
|
DEFAULT_CANDIDATE = Path("/opt/data/kanban/recovery/soteria/recovered.db")
|
|
COUNTS = {"tasks": 4, "task_events": 159, "task_comments": 8, "task_runs": 4,
|
|
"supervisor_roots": 3, "supervisor_children": 0}
|
|
ROOTS = {"t_f1593f8c", "t_c7c42600", "t_f4f726e1"}
|
|
REQUIRED_TABLES = {
|
|
"tasks", "task_events", "task_comments", "task_runs", "task_links",
|
|
"task_attachments", "kanban_notify_subs", "supervisor_roots", "supervisor_children",
|
|
}
|
|
|
|
|
|
class RecoveryError(RuntimeError):
|
|
"""The candidate or current board did not match the reviewed recovery plan."""
|
|
|
|
|
|
def _sha(path: Path) -> str:
|
|
"""Hash one local artifact without accepting symlink indirection."""
|
|
if path.is_symlink() or not path.is_file():
|
|
raise RecoveryError("recovery artifact is unavailable")
|
|
digest = hashlib.sha256()
|
|
with path.open("rb") as source:
|
|
for block in iter(lambda: source.read(1024 * 1024), b""):
|
|
digest.update(block)
|
|
return digest.hexdigest()
|
|
|
|
|
|
def _integrity(path: Path) -> bool:
|
|
"""Return whether SQLite can fully validate a database without writing it."""
|
|
try:
|
|
connection = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
|
|
try:
|
|
return [row[0] for row in connection.execute("PRAGMA integrity_check")] == ["ok"]
|
|
finally:
|
|
connection.close()
|
|
except sqlite3.Error:
|
|
return False
|
|
|
|
|
|
def _verify_candidate(path: Path) -> None:
|
|
"""Require the exact forensic artifact and its reviewed data contract."""
|
|
if _sha(path) != CANDIDATE_SHA or not _integrity(path):
|
|
raise RecoveryError("recovery candidate does not match reviewed evidence")
|
|
try:
|
|
connection = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
|
|
try:
|
|
tables = {str(row[0]) for row in connection.execute(
|
|
"SELECT name FROM sqlite_master WHERE type='table'"
|
|
)}
|
|
if not REQUIRED_TABLES <= tables:
|
|
raise RecoveryError("recovery candidate is missing required tables")
|
|
counts = {table: int(connection.execute(f"SELECT count(*) FROM {table}").fetchone()[0])
|
|
for table in COUNTS}
|
|
roots = {str(row[0]) for row in connection.execute(
|
|
"SELECT root_task_id FROM supervisor_roots"
|
|
)}
|
|
finally:
|
|
connection.close()
|
|
except sqlite3.Error as error:
|
|
raise RecoveryError("recovery candidate schema is unreadable") from error
|
|
if counts != COUNTS or roots != ROOTS:
|
|
raise RecoveryError("recovery candidate data contract differs")
|
|
|
|
|
|
def _copy(source: Path, destination: Path) -> None:
|
|
"""Durably copy one evidence artifact into a new destination."""
|
|
with source.open("rb") as reader, destination.open("xb") as writer:
|
|
shutil.copyfileobj(reader, writer)
|
|
writer.flush()
|
|
os.fsync(writer.fileno())
|
|
|
|
|
|
def _marker(database: Path) -> Path:
|
|
return database.with_name(f"{database.name}.recovery-{CANDIDATE_SHA[:16]}.json")
|
|
|
|
|
|
def _write_marker(database: Path) -> None:
|
|
"""Durably publish completion only after the replacement is in place."""
|
|
marker = _marker(database)
|
|
descriptor, temporary = tempfile.mkstemp(prefix=f".{marker.name}.", dir=database.parent)
|
|
try:
|
|
with os.fdopen(descriptor, "w", encoding="utf-8") as handle:
|
|
json.dump({"candidate_sha": CANDIDATE_SHA, "source_sha": SOURCE_SHA}, handle, sort_keys=True)
|
|
handle.write("\n")
|
|
handle.flush()
|
|
os.fsync(handle.fileno())
|
|
os.chmod(temporary, 0o600)
|
|
os.replace(temporary, marker)
|
|
directory = os.open(database.parent, os.O_RDONLY)
|
|
try:
|
|
os.fsync(directory)
|
|
finally:
|
|
os.close(directory)
|
|
except Exception:
|
|
Path(temporary).unlink(missing_ok=True)
|
|
raise
|
|
|
|
|
|
def _completed(database: Path) -> bool:
|
|
"""Allow later restarts only after the recorded replacement remains healthy."""
|
|
try:
|
|
value = json.loads(_marker(database).read_text())
|
|
except (OSError, ValueError, TypeError):
|
|
return False
|
|
return value == {"candidate_sha": CANDIDATE_SHA, "source_sha": SOURCE_SHA} and _integrity(database)
|
|
|
|
|
|
def _has_nonempty_journal(database: Path) -> bool:
|
|
"""Refuse a source with unapplied writes whose provenance was not reviewed."""
|
|
for suffix in ("-wal", "-journal"):
|
|
sidecar = database.with_name(database.name + suffix)
|
|
if sidecar.exists() and sidecar.stat().st_size:
|
|
return True
|
|
return False
|
|
|
|
|
|
def _retained_source(database: Path) -> bool:
|
|
"""Require the exact original evidence before finalizing an interrupted install."""
|
|
pattern = f"{database.name}.corrupt.recovery-*-{SOURCE_SHA[:16]}.bak"
|
|
for backup in database.parent.glob(pattern):
|
|
try:
|
|
if _sha(backup) == SOURCE_SHA:
|
|
return True
|
|
except RecoveryError:
|
|
continue
|
|
return False
|
|
|
|
|
|
def recover(database: Path = DEFAULT_DATABASE, candidate: Path = DEFAULT_CANDIDATE) -> dict[str, str]:
|
|
"""Replace only the exact quarantined source with the verified candidate."""
|
|
if _marker(database).exists():
|
|
if _completed(database):
|
|
return {"state": "already-recovered", "database": str(database)}
|
|
raise RecoveryError("completed recovery marker does not match healthy board")
|
|
if _has_nonempty_journal(database):
|
|
raise RecoveryError("current board has unreviewed journal writes")
|
|
current = _sha(database)
|
|
if current == CANDIDATE_SHA and _integrity(database):
|
|
if not _retained_source(database):
|
|
raise RecoveryError("interrupted recovery lacks retained source evidence")
|
|
_write_marker(database)
|
|
return {"state": "recovery-finalized", "database": str(database)}
|
|
_verify_candidate(candidate)
|
|
if current != SOURCE_SHA:
|
|
raise RecoveryError("current board does not match quarantined source")
|
|
stamp = f"{int(time.time())}-{SOURCE_SHA[:16]}"
|
|
backup = database.with_name(f"{database.name}.corrupt.recovery-{stamp}.bak")
|
|
_copy(database, backup)
|
|
for suffix in ("-wal", "-shm", "-journal"):
|
|
sidecar = database.with_name(database.name + suffix)
|
|
if sidecar.exists():
|
|
os.replace(sidecar, backup.with_name(backup.name + suffix))
|
|
descriptor, temporary = tempfile.mkstemp(prefix=f".{database.name}.recovery-", dir=database.parent)
|
|
os.close(descriptor)
|
|
replacement = Path(temporary)
|
|
replacement.unlink()
|
|
try:
|
|
_copy(candidate, replacement)
|
|
if not _integrity(replacement):
|
|
raise RecoveryError("copied candidate failed integrity verification")
|
|
os.chmod(replacement, database.stat().st_mode & 0o777)
|
|
os.replace(replacement, database)
|
|
except Exception:
|
|
replacement.unlink(missing_ok=True)
|
|
raise
|
|
_write_marker(database)
|
|
return {"state": "recovered", "database": str(database), "backup": str(backup)}
|
|
|
|
|
|
def main() -> int:
|
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
parser.add_argument("--database", type=Path, default=DEFAULT_DATABASE)
|
|
parser.add_argument("--candidate", type=Path, default=DEFAULT_CANDIDATE)
|
|
args = parser.parse_args()
|
|
try:
|
|
print(json.dumps(recover(args.database, args.candidate), sort_keys=True))
|
|
except (OSError, RecoveryError) as error:
|
|
print(f"Soteria Kanban recovery refused: {error}")
|
|
return 1
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|