feat(mcp): enforce scoped recovery playbook before full restart (Closes #669)

Add recovery_playbook.py with the narrow-to-broad recovery ladder, symptom
routing, attempt-log helpers, and escalation metrics. Wire the attempt-log
gate into restart_coordinator so rolling/full/host restarts require prior
insufficient narrower attempts (or break-glass). Document the ladder and
update gitea_request_mcp_restart for prior_recovery_attempts_json.

Co-Authored-By: Grok 4.5 <[email protected]>
This commit is contained in:
2026-07-25 17:35:11 -04:00
co-authored by Grok 4.5
parent 9c69bfcd80
commit 461e1dac78
7 changed files with 1004 additions and 16 deletions
@@ -35,6 +35,17 @@ BREAK_GLASS_ENV = "GITEA_BREAKGLASS_RESTART_AUTHORIZATION"
QUIET_SESSIONS: list[dict] = []
QUIET_LEASES: list[dict] = []
# #669: broad restarts need a prior narrow-attempt log (unless break-glass).
PRIOR_NARROW_ATTEMPTS_JSON = json.dumps(
[
{
"action": "client_reconnect",
"outcome": "insufficient",
"reason": "still flapping after reconnect",
}
]
)
class _FakeDB:
"""Minimal control-plane DB stand-in for the restart inventory."""
@@ -128,6 +139,7 @@ class TestConjunction(_RestartToolHarness):
preview = self._call(
role="operator",
restart_class="full_mcp_restart",
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
)
self.assertTrue(preview["allow_restart"],
@@ -136,6 +148,7 @@ class TestConjunction(_RestartToolHarness):
result = self._call(
role="operator",
restart_class="full_mcp_restart",
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
dry_run=False,
drain_proof_json=self._clean_proof_for(preview),
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
@@ -342,6 +355,7 @@ class TestExistingPathsStillWork(_RestartToolHarness):
result = self._call(
role="operator",
restart_class="full_mcp_restart",
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
dry_run=False,
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
)