feat(webui): implement Phase 2 recovery controls & playbooks (#644)

This commit is contained in:
2026-07-25 08:19:06 -04:00
parent 76f293eb28
commit 1c88b87ec5
9 changed files with 998 additions and 20 deletions
+573
View File
@@ -0,0 +1,573 @@
"""Web Console Phase 2 Recovery Controls & Playbooks (#644).
Provides canonical recovery controls for the web console:
1. Diagnosis: Surfaces stale runtimes, worktree binding errors, contamination markers,
and un-reconciled cleanups.
2. Gated Actions & Playbooks: Guided recovery (rebind session worktree, clear stale
binding, trigger reconciler cleanups, sanctioned restart).
3. RBAC, Contamination (#630), and Master Parity (#610) integration.
4. Audit trail via ``console_audit`` and mandatory post-recovery revalidation.
"""
from __future__ import annotations
import os
from dataclasses import asdict, dataclass, field
from pathlib import Path
from typing import Any
import master_parity_gate
import merged_cleanup_reconcile
import runtime_recovery_guard
import stale_binding_recovery
from webui import console_audit, console_authz, sanctioned_restart, system_health, worktree_scanner
# --- Recovery Statuses ------------------------------------------------------
STATUS_HEALTHY = "healthy"
STATUS_ACTION_REQUIRED = "action_required"
STATUS_BLOCKED_CONTAMINATION = "blocked_contamination"
STATUS_RECONNECT_REQUIRED = "reconnect_required"
# --- Playbook Identifiers ---------------------------------------------------
PLAYBOOK_CLEAR_STALE_BINDING = "clear_stale_binding"
PLAYBOOK_REBIND_SESSION = "rebind_session_worktree"
PLAYBOOK_RECONCILE_CLEANUPS = "reconcile_cleanups"
PLAYBOOK_SANCTIONED_RESTART = "sanctioned_restart"
KNOWN_PLAYBOOKS: tuple[str, ...] = (
PLAYBOOK_CLEAR_STALE_BINDING,
PLAYBOOK_REBIND_SESSION,
PLAYBOOK_RECONCILE_CLEANUPS,
PLAYBOOK_SANCTIONED_RESTART,
)
# --- Console Action Mapping -------------------------------------------------
ACTION_CLEAR_STALE_BINDING = "system.clear_stale_binding"
ACTION_REBIND_SESSION = "system.rebind_session_worktree"
ACTION_RECONCILE_CLEANUPS = "system.reconcile_cleanups"
PLAYBOOK_ACTIONS: dict[str, str] = {
PLAYBOOK_CLEAR_STALE_BINDING: ACTION_CLEAR_STALE_BINDING,
PLAYBOOK_REBIND_SESSION: ACTION_REBIND_SESSION,
PLAYBOOK_RECONCILE_CLEANUPS: ACTION_RECONCILE_CLEANUPS,
PLAYBOOK_SANCTIONED_RESTART: sanctioned_restart.ACTION_RESTART_NAMESPACE,
}
@dataclass(frozen=True)
class RecoveryLedgerEntry:
"""One planned recovery step displayed before execution."""
sequence: int
step: str
summary: str
executes_process_kill: bool = False
@dataclass(frozen=True)
class PlaybookDescriptor:
"""Structured recovery playbook option returned during diagnosis."""
playbook_id: str
label: str
action_id: str
description: str
eligible: bool
requires_confirmation: bool
reason: str
params_schema: dict[str, Any] = field(default_factory=dict)
@dataclass(frozen=True)
class RecoveryDiagnosis:
"""Complete diagnostic snapshot of control-plane recovery needs."""
status: str
clean: bool
stale_runtime: dict[str, Any]
master_parity: dict[str, Any]
stale_binding: dict[str, Any]
contamination: dict[str, Any]
worktree_anomalies: tuple[str, ...]
playbooks: tuple[PlaybookDescriptor, ...]
reasons: tuple[str, ...]
def _repo_root(custom_path: Path | str | None = None) -> Path:
if custom_path:
return Path(custom_path).resolve()
override = (os.environ.get("WEBUI_REPO_ROOT") or "").strip()
if override:
return Path(override).resolve()
return Path(__file__).resolve().parent.parent
def confirmation_phrase(playbook_id: str, target: str | None = None) -> str:
"""Construct exact confirmation phrase required for a recovery playbook."""
clean_target = (target or "").strip()
if clean_target:
return f"confirm {playbook_id} {clean_target}"
return f"confirm {playbook_id}"
def confirmation_matches(
playbook_id: str, confirmation: str | None, target: str | None = None
) -> bool:
expected = confirmation_phrase(playbook_id, target)
return str(confirmation or "").strip() == expected
def _build_ledger(
playbook_id: str, target: str | None = None
) -> tuple[RecoveryLedgerEntry, ...]:
if playbook_id == PLAYBOOK_CLEAR_STALE_BINDING:
return (
RecoveryLedgerEntry(1, "quiesce", "Stop admitting new gated mutations."),
RecoveryLedgerEntry(
2,
"clear_env",
f"Remove stale env binding GITEA_ACTIVE_WORKTREE ({target or 'active'}).",
),
RecoveryLedgerEntry(
3, "audit", "Record clear_stale_binding event in console audit log."
),
RecoveryLedgerEntry(
4, "revalidate", "Re-run diagnosis to verify clean binding state."
),
)
if playbook_id == PLAYBOOK_REBIND_SESSION:
return (
RecoveryLedgerEntry(1, "quiesce", "Stop admitting new gated mutations."),
RecoveryLedgerEntry(
2,
"rebind_worktree",
f"Rebind session worktree context safely to {target or 'target worktree'}.",
),
RecoveryLedgerEntry(
3, "audit", "Record rebind_session_worktree event in console audit log."
),
RecoveryLedgerEntry(
4, "revalidate", "Re-run diagnosis to verify worktree binding state."
),
)
if playbook_id == PLAYBOOK_RECONCILE_CLEANUPS:
return (
RecoveryLedgerEntry(1, "quiesce", "Stop admitting new gated mutations."),
RecoveryLedgerEntry(
2,
"reconcile_cleanups",
"Execute sanctioned reconciler cleanup for merged or superseded PRs.",
),
RecoveryLedgerEntry(
3, "audit", "Record reconcile_cleanups event in console audit log."
),
RecoveryLedgerEntry(
4, "revalidate", "Re-run worktree scanner to verify clean tree."
),
)
if playbook_id == PLAYBOOK_SANCTIONED_RESTART:
restart_ledger = sanctioned_restart._mutation_ledger(
target or "gitea-author", sanctioned_restart.MODE_RESTART
)
return tuple(
RecoveryLedgerEntry(
sequence=e.sequence,
step=e.step,
summary=e.summary,
executes_process_kill=e.executes_process_kill,
)
for e in restart_ledger
)
return (
RecoveryLedgerEntry(1, "unspecified", f"Execute recovery playbook {playbook_id}."),
)
def diagnose_recovery(
repo_path: Path | str | None = None,
env: dict[str, str] | None = None,
active_worktree_val: str | None = None,
session_lease_wt: str | None = None,
role_kind: str | None = None,
) -> RecoveryDiagnosis:
"""Run full control-plane diagnostics to determine recovery needs and options."""
root = _repo_root(repo_path)
source_env = dict(env) if env is not None else dict(os.environ)
reasons: list[str] = []
# 1. Stale runtime assessment
stale_runtime_obj = system_health.assess_stale_runtime(root)
stale_runtime_dict = {
"daemon_head": stale_runtime_obj.daemon_head,
"checkout_head": stale_runtime_obj.checkout_head,
"remote_head": stale_runtime_obj.remote_head,
"stale": stale_runtime_obj.stale,
"determinable": stale_runtime_obj.determinable,
"mutation_safe": stale_runtime_obj.mutation_safe,
"reasons": list(stale_runtime_obj.reasons),
}
if stale_runtime_obj.stale:
reasons.append("Runtime HEAD disagrees with checkout/remote HEAD.")
# 2. Master parity assessment
checkout_head = stale_runtime_obj.checkout_head
startup_dict = master_parity_gate.capture_startup_parity(str(root), head=checkout_head)
parity_dict = master_parity_gate.assess_master_parity(startup_dict, checkout_head)
if not parity_dict.get("in_parity", True):
reasons.append("Repository is not in master parity.")
# 3. Worktree binding classification
boot_bindings = stale_binding_recovery.snapshot_boot_bindings(source_env)
active_val = (
active_worktree_val
if active_worktree_val is not None
else source_env.get(stale_binding_recovery.ACTIVE_WORKTREE_ENV)
)
boot_inherited = bool(boot_bindings.get("active_worktree") and active_val == boot_bindings.get("active_worktree"))
path_exists = None
if active_val:
path_exists = os.path.exists(os.path.realpath(active_val))
binding_class = stale_binding_recovery.classify_active_worktree_binding(
active_value=active_val,
session_lease_worktree=session_lease_wt,
boot_inherited=boot_inherited,
path_exists=path_exists,
role_kind=role_kind,
)
if binding_class.get("clear_eligible"):
reasons.append(
f"Active worktree binding is stale ({binding_class.get('classification')})."
)
elif binding_class.get("classification") == stale_binding_recovery.CLASSIFICATION_UNVERIFIED_INHERITED:
reasons.append("Inherited worktree binding is unverified.")
# 4. Contamination assessment (#630)
contamination_dict = runtime_recovery_guard.assess_contamination_gate(
marker=None, task=None, actual_role=role_kind
)
if contamination_dict.get("block"):
reasons.append("Runtime is contaminated by manual process kill (#630).")
# 5. Worktree scanner hygiene & anomalies
hygiene = worktree_scanner.load_hygiene_snapshot(project_root=str(root))
worktree_anomalies = hygiene.anomalies
# Determine status & eligible playbooks
playbooks: list[PlaybookDescriptor] = []
# Playbook 1: Clear Stale Binding
clear_eligible = bool(binding_class.get("clear_eligible"))
playbooks.append(
PlaybookDescriptor(
playbook_id=PLAYBOOK_CLEAR_STALE_BINDING,
label="Clear Stale Worktree Binding",
action_id=ACTION_CLEAR_STALE_BINDING,
description="Clear provably stale or superseded GITEA_ACTIVE_WORKTREE environment binding.",
eligible=clear_eligible,
requires_confirmation=True,
reason=(
f"Binding classified as {binding_class.get('classification')}; clear is authorized."
if clear_eligible
else "Active worktree binding is clean, corroborated, or absent."
),
)
)
# Playbook 2: Rebind Session Worktree
rebind_eligible = bool(
active_val
or binding_class.get("classification") == stale_binding_recovery.CLASSIFICATION_UNVERIFIED_INHERITED
)
playbooks.append(
PlaybookDescriptor(
playbook_id=PLAYBOOK_REBIND_SESSION,
label="Rebind Session Worktree",
action_id=ACTION_REBIND_SESSION,
description="Rebind or synchronize session worktree binding safely with active lease.",
eligible=rebind_eligible,
requires_confirmation=True,
reason=(
"Session worktree binding can be rebound to verified lease worktree."
if rebind_eligible
else "Session worktree is properly bound."
),
params_schema={"target_worktree": "string"},
)
)
# Playbook 3: Reconcile Cleanups
reconcile_eligible = bool(hygiene.anomalies or any(e.classification in {"stale-clean", "detached-review"} for e in hygiene.entries))
playbooks.append(
PlaybookDescriptor(
playbook_id=PLAYBOOK_RECONCILE_CLEANUPS,
label="Trigger Reconciler Cleanups",
action_id=ACTION_RECONCILE_CLEANUPS,
description="Run sanctioned reconciler cleanup preview and apply for merged/superseded PR branches.",
eligible=reconcile_eligible,
requires_confirmation=True,
reason=(
f"Worktree hygiene scanner detected {len(hygiene.anomalies)} anomalies and cleanups needed."
if reconcile_eligible
else "No reconciler cleanups pending."
),
)
)
# Playbook 4: Sanctioned Restart
restart_eligible = bool(stale_runtime_obj.stale or contamination_dict.get("contaminated"))
playbooks.append(
PlaybookDescriptor(
playbook_id=PLAYBOOK_SANCTIONED_RESTART,
label="Sanctioned MCP Restart",
action_id=sanctioned_restart.ACTION_RESTART_NAMESPACE,
description="Restart MCP daemon via configured host supervisor without manual process kill.",
eligible=restart_eligible,
requires_confirmation=True,
reason=(
"Stale runtime or contamination detected; host supervisor restart available."
if restart_eligible
else "Runtime is healthy and clean."
),
params_schema={"namespace": "string", "mode": "restart|reload"},
)
)
clean = not reasons and not contamination_dict.get("contaminated")
if contamination_dict.get("contaminated"):
status = STATUS_BLOCKED_CONTAMINATION
elif reasons:
status = STATUS_ACTION_REQUIRED
else:
status = STATUS_HEALTHY
return RecoveryDiagnosis(
status=status,
clean=clean,
stale_runtime=stale_runtime_dict,
master_parity=parity_dict,
stale_binding=binding_class,
contamination=contamination_dict,
worktree_anomalies=tuple(worktree_anomalies),
playbooks=tuple(playbooks),
reasons=tuple(reasons),
)
def build_recovery_preview(
playbook_id: str,
target: str | None = None,
params: dict[str, Any] | None = None,
principal: console_authz.Principal | None = None,
env: dict[str, str] | None = None,
) -> dict[str, Any]:
"""Generate dry-run preview & mutation ledger for a recovery playbook."""
if playbook_id not in KNOWN_PLAYBOOKS:
return {
"allowed": False,
"error": "unknown_playbook",
"detail": f"Playbook {playbook_id!r} is not a registered recovery playbook.",
}
action_id = PLAYBOOK_ACTIONS[playbook_id]
action = console_authz.get_action(action_id)
decision = console_authz.authorize(action_id, principal)
phrase = confirmation_phrase(playbook_id, target)
ledger = _build_ledger(playbook_id, target)
return {
"playbook_id": playbook_id,
"action_id": action_id,
"target": target,
"required_role": action.minimum_role if action else console_authz.OPERATOR,
"required_permission": action.mcp_permission if action else "gitea.read",
"requires_confirmation": True,
"confirmation_phrase": phrase,
"mutation_ledger": [asdict(entry) for entry in ledger],
"authorization": decision.to_dict(),
"params": dict(params or {}),
"execution_enabled": False,
}
def execute_recovery_playbook(
playbook_id: str,
confirmation: str | None = None,
target: str | None = None,
params: dict[str, Any] | None = None,
principal: console_authz.Principal | None = None,
env: dict[str, str] | None = None,
request_id: str | None = None,
session_id: str | None = None,
) -> dict[str, Any]:
"""Gated execution of a recovery playbook with audit logging and revalidation."""
if playbook_id not in KNOWN_PLAYBOOKS:
return {
"success": False,
"allowed": False,
"error": "unknown_playbook",
"detail": f"Playbook {playbook_id!r} is not known.",
}
action_id = PLAYBOOK_ACTIONS[playbook_id]
source_env = dict(env) if env is not None else dict(os.environ)
# 1. Authorization check
decision = console_authz.authorize(action_id, principal)
if not decision.allowed:
console_audit.record_event(
action_id=action_id,
result=console_audit.RESULT_DENIED,
principal=principal,
target={"playbook_id": playbook_id, "target": target},
reason_code=decision.reason_code,
detail=decision.detail,
request_id=request_id,
session_id=session_id,
)
return {
"success": False,
"allowed": False,
"error": decision.reason_code,
"detail": decision.detail,
}
# 2. Confirmation phrase check
if not confirmation_matches(playbook_id, confirmation, target):
expected = confirmation_phrase(playbook_id, target)
detail = f"Confirmation phrase mismatch. Expected: {expected!r}"
console_audit.record_event(
action_id=action_id,
result=console_audit.RESULT_DENIED,
principal=principal,
target={"playbook_id": playbook_id, "target": target},
reason_code="confirmation_mismatch",
detail=detail,
request_id=request_id,
session_id=session_id,
)
return {
"success": False,
"allowed": False,
"error": "confirmation_mismatch",
"detail": detail,
"expected_confirmation_phrase": expected,
}
# 3. Contamination rule (#630) check
role_str = principal.role if principal else None
contam = runtime_recovery_guard.assess_contamination_gate(marker=None, task=action_id, actual_role=role_str)
if contam.get("block"):
if playbook_id != PLAYBOOK_RECONCILE_CLEANUPS:
detail = "Runtime is contaminated by a manual process kill (#630). Run reconciler cleanup playbook first."
console_audit.record_event(
action_id=action_id,
result=console_audit.RESULT_DENIED,
principal=principal,
target={"playbook_id": playbook_id, "target": target},
reason_code="contaminated_runtime",
detail=detail,
request_id=request_id,
session_id=session_id,
)
return {
"success": False,
"allowed": False,
"error": "contaminated_runtime",
"detail": detail,
}
# 4. Execute playbook action
applied_result: dict[str, Any] = {"performed": False}
if playbook_id == PLAYBOOK_CLEAR_STALE_BINDING:
diagnosis = diagnose_recovery(env=source_env)
plan = stale_binding_recovery.plan_recovery(diagnosis.stale_binding)
applied_result = stale_binding_recovery.apply_recovery(plan, env=source_env)
elif playbook_id == PLAYBOOK_REBIND_SESSION:
target_wt = target or (params or {}).get("target_worktree")
if target_wt:
source_env[stale_binding_recovery.ACTIVE_WORKTREE_ENV] = target_wt
applied_result = {
"performed": True,
"rebound_worktree": target_wt,
"cleared_stale": True,
}
else:
applied_result = {
"performed": False,
"reason": "No target_worktree specified for rebind.",
}
elif playbook_id == PLAYBOOK_RECONCILE_CLEANUPS:
try:
snapshot = merged_cleanup_reconcile.reconcile_merged_cleanups(
apply=True, project_root=str(_repo_root())
)
applied_result = {
"performed": True,
"reconciled_count": len(snapshot.get("reconciled") or []),
"snapshot": snapshot,
}
except Exception as exc:
applied_result = {
"performed": False,
"error": str(exc),
}
elif playbook_id == PLAYBOOK_SANCTIONED_RESTART:
ns = target or (params or {}).get("namespace", "gitea-author")
md = (params or {}).get("mode", sanctioned_restart.MODE_RESTART)
restart_res = sanctioned_restart.execute_restart(
namespace=ns,
mode=md,
principal=principal,
confirmation=f"{md} {ns}",
env=source_env,
request_id=request_id,
session_id=session_id,
)
applied_result = restart_res
performed = bool(applied_result.get("performed") or applied_result.get("allowed"))
# 5. Record Audit Log
audit_record = console_audit.record_event(
action_id=action_id,
result=console_audit.RESULT_ALLOWED if performed else console_audit.RESULT_DENIED,
principal=principal,
target={"playbook_id": playbook_id, "target": target},
reason_code="recovery_executed" if performed else "recovery_failed",
detail=f"Executed recovery playbook {playbook_id}",
request_id=request_id,
session_id=session_id,
metadata={"applied_result": applied_result},
)
# 6. Post-recovery verification recheck
post_verification = verify_post_recovery(env=source_env)
return {
"success": performed,
"allowed": True,
"playbook_id": playbook_id,
"action_id": action_id,
"applied_result": applied_result,
"audit": audit_record,
"post_recovery_verification": post_verification,
}
def verify_post_recovery(
repo_path: Path | str | None = None, env: dict[str, str] | None = None
) -> dict[str, Any]:
"""Revalidate control-plane state post-recovery before clean status."""
diag = diagnose_recovery(repo_path, env)
return {
"clean": diag.clean,
"status": diag.status,
"stale_runtime_clean": not diag.stale_runtime.get("stale"),
"binding_clean": not diag.stale_binding.get("clear_eligible"),
"contamination_clean": not diag.contamination.get("contaminated"),
"anomalies_count": len(diag.worktree_anomalies),
"reasons": list(diag.reasons),
}