Compare commits
19
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fad44669d9 | ||
|
|
ca22c326a4 | ||
|
|
3bbe6df6c7 | ||
|
|
26f54851d1 | ||
|
|
f02a2dc030 | ||
|
|
04ae3532cc | ||
|
|
7bb5ff4719 | ||
|
|
5b7ceefa9a | ||
|
|
4a2fae8495 | ||
|
|
461e1dac78 | ||
|
|
6010f4295b | ||
|
|
9b8e315b49 | ||
|
|
9a01543477 | ||
|
|
1c88b87ec5 | ||
|
|
b993ad1c64 | ||
|
|
d0006e9f71 | ||
|
|
6da68fffb8 | ||
|
|
53ce1b1a5e | ||
|
|
433f66add8 |
+98
-24
@@ -23,6 +23,7 @@ import json
|
||||
import os
|
||||
import uuid
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
from control_plane_db import (
|
||||
@@ -738,6 +739,46 @@ def normalize_exclude_issue_numbers(
|
||||
return sorted(out)
|
||||
|
||||
|
||||
def _claim_expires_at(claim: Any) -> datetime | None:
|
||||
"""Parse a claim's ``expires_at``, or ``None`` when it is absent/malformed."""
|
||||
if not isinstance(claim, Mapping):
|
||||
return None
|
||||
text = str(claim.get("expires_at") or "").strip()
|
||||
if not text:
|
||||
return None
|
||||
if text.endswith("Z"):
|
||||
text = text[:-1] + "+00:00"
|
||||
try:
|
||||
parsed = datetime.fromisoformat(text)
|
||||
except ValueError:
|
||||
return None
|
||||
if parsed.tzinfo is None:
|
||||
parsed = parsed.replace(tzinfo=timezone.utc)
|
||||
return parsed.astimezone(timezone.utc)
|
||||
|
||||
|
||||
def _drop_expired_claims(
|
||||
claims: Mapping[tuple[str, int], dict[str, Any]],
|
||||
*,
|
||||
now: datetime | None = None,
|
||||
) -> dict[tuple[str, int], dict[str, Any]]:
|
||||
"""Claims minus those whose lease has already expired (#643).
|
||||
|
||||
The read-only mirror of ``expire_stale_leases``: the sweep marks such rows
|
||||
``expired`` so they stop being returned as claims, and this reaches the same
|
||||
view without writing. A claim with no parseable ``expires_at`` is **kept** —
|
||||
an unreadable expiry is not evidence that work is free.
|
||||
"""
|
||||
moment = now or datetime.now(timezone.utc)
|
||||
kept: dict[tuple[str, int], dict[str, Any]] = {}
|
||||
for key, claim in (claims or {}).items():
|
||||
expires_at = _claim_expires_at(claim)
|
||||
if expires_at is not None and expires_at <= moment:
|
||||
continue
|
||||
kept[key] = claim
|
||||
return kept
|
||||
|
||||
|
||||
def candidate_set_fingerprint(
|
||||
candidates: Sequence[WorkCandidate],
|
||||
*,
|
||||
@@ -826,12 +867,22 @@ def allocate_next_work(
|
||||
exclude_issue_numbers: Sequence[int] | None = None,
|
||||
expected_candidate_set_fingerprint: str | None = None,
|
||||
allocation_mode: str | None = None,
|
||||
side_effect_free: bool = False,
|
||||
) -> dict[str, Any]:
|
||||
"""Select and optionally reserve the next work unit via control-plane DB.
|
||||
|
||||
*apply=False* (default): dry-run selection only — no lease/assignment.
|
||||
*apply=True*: atomic ``assign_and_lease`` for the selected candidate.
|
||||
|
||||
*side_effect_free* (#643): a dry run that writes **nothing** to the
|
||||
control-plane DB. A plain ``apply=False`` still registered a session row and
|
||||
swept stale leases globally, so a caller advertising a read-only preview was
|
||||
mutating on every call. Under this flag both writes are suppressed and stale
|
||||
leases are instead filtered out of the claim map in memory, which yields the
|
||||
same selection the sweep would have produced without persisting anything.
|
||||
Incompatible with *apply* — the combination fails closed rather than
|
||||
silently reserving.
|
||||
|
||||
*allocation_mode* (#840): ``cross_role`` (default for controller) inspects
|
||||
the complete queue and returns one authoritative selection naming the
|
||||
required downstream role/profile/action. ``role_scoped`` keeps prior
|
||||
@@ -885,40 +936,57 @@ def allocate_next_work(
|
||||
"allocation_mode": (allocation_mode or "").strip() or None,
|
||||
}
|
||||
|
||||
session_id = (session_id or "").strip() or f"alloc-{uuid.uuid4().hex[:12]}"
|
||||
try:
|
||||
db.upsert_session(
|
||||
session_id=session_id,
|
||||
role=role_norm,
|
||||
profile=profile_name,
|
||||
pid=os.getpid(),
|
||||
controller_instance_id=controller_instance_id,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — surface structured
|
||||
# A side-effect-free run may never reserve: reserving is a write, and the
|
||||
# flag is the caller's assertion that this call writes nothing (#643).
|
||||
if side_effect_free and apply:
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"apply": True,
|
||||
"reasons": [
|
||||
f"failed to register session in control-plane DB: {exc} "
|
||||
"(fail closed, #613)"
|
||||
"side_effect_free is incompatible with apply=True; an "
|
||||
"assignment is a write (fail closed, #643)"
|
||||
],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
|
||||
# Expire stale leases globally before selection.
|
||||
try:
|
||||
db.expire_stale_leases()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"reasons": [f"lease expiry failed: {exc} (fail closed)"],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
session_id = (session_id or "").strip() or f"alloc-{uuid.uuid4().hex[:12]}"
|
||||
if not side_effect_free:
|
||||
try:
|
||||
db.upsert_session(
|
||||
session_id=session_id,
|
||||
role=role_norm,
|
||||
profile=profile_name,
|
||||
pid=os.getpid(),
|
||||
controller_instance_id=controller_instance_id,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — surface structured
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"reasons": [
|
||||
f"failed to register session in control-plane DB: {exc} "
|
||||
"(fail closed, #613)"
|
||||
],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
|
||||
# Expire stale leases globally before selection.
|
||||
try:
|
||||
db.expire_stale_leases()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"reasons": [f"lease expiry failed: {exc} (fail closed)"],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
|
||||
terminal = None
|
||||
try:
|
||||
@@ -953,6 +1021,12 @@ def allocate_next_work(
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
if side_effect_free:
|
||||
# ``list_active_claims`` filters on status alone, so without the
|
||||
# global sweep an already-expired lease would still read as a live
|
||||
# claim and the preview would report work as taken that is free.
|
||||
# Drop those in memory: same view the sweep produces, no write.
|
||||
claims = _drop_expired_claims(claims)
|
||||
|
||||
try:
|
||||
exclude_nums = normalize_exclude_issue_numbers(exclude_issue_numbers)
|
||||
|
||||
@@ -0,0 +1,167 @@
|
||||
# ADR: High-availability and rolling-restart architecture for Gitea MCP control plane
|
||||
|
||||
- **Status:** Proposed (Design ADR under [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668))
|
||||
- **Date:** 2026-07-25
|
||||
- **Tracking Issue:** [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668)
|
||||
- **Policy Version:** `mcp-ha-rolling-restart/v1`
|
||||
- **Related:**
|
||||
- Parent: [#655](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/655) — Governed MCP restart coordination and zero-disruption recovery
|
||||
- Governance Policy: [#656](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/656) / `docs/architecture/mcp-restart-governance.md`
|
||||
- Control-Plane DB Substrate: [#613](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/613) / `docs/architecture/control-plane-db-substrate.md`
|
||||
- Runtime Policy: [#615](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/615) / `docs/architecture/mcp-stable-control-runtime-policy-adr.md`
|
||||
- Product Vision: [#652](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/652) (Phase 5 Maturity)
|
||||
- Delivery Roadmap: [#653](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/653)
|
||||
|
||||
---
|
||||
|
||||
## 1. Context & Problem Statement
|
||||
|
||||
The Gitea MCP server operates as the authoritative **control plane** for managing issues, Pull Requests, code mutations, formal reviews, and workflow reconciliations. Under single-process governance ([#656](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/656)), process restarts are strictly controlled using pre-flight checks, drain phases, and operator approvals.
|
||||
|
||||
However, a single-instance control plane inherently presents fundamental constraints:
|
||||
|
||||
1. **Downtime during updates:** Even a perfectly executed single-process drain requires a window where incoming client requests must be paused or rejected while the server binary or python environment reloads.
|
||||
2. **Single point of failure:** Infrastructure issues, process crashes, or unhandled host-level terminations immediately disconnect active LLM sessions and leave transient workflows incomplete.
|
||||
3. **Multi-agent concurrency bottlenecks:** High volumes of concurrent multi-LLM tasks put all lock management, lease allocation, and Gitea API interactions through a single process event loop.
|
||||
|
||||
To achieve true zero-disruption operation and seamless rolling deployments without stopping active work, the system requires a high-availability (HA), multi-instance MCP architecture.
|
||||
|
||||
---
|
||||
|
||||
## 2. Architectural Principles & Non-Goals
|
||||
|
||||
### 2.1 Core Architectural Principles
|
||||
* **Gitea as Canonical Work SoT:** Gitea remains the ultimate System of Record (SoT) for issue states, pull requests, labels, and audit comments. The MCP control plane does not duplicate domain entities.
|
||||
* **Control-Plane DB as Multi-Instance State Substrate:** The control-plane SQLite/durable database ([#613](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/613)) acts as the single source of truth for workflow leases, session tokens, assignment records, and lock fences across all MCP nodes.
|
||||
* **Stateless Worker Nodes:** MCP role server processes (`gitea-author`, `gitea-reviewer`, `gitea-merger`, `gitea-reconciler`, `gitea-controller`) maintain no unique in-memory state; any node can handle any request given a valid session resume token.
|
||||
* **Fail-Closed Split-Brain Defense:** In any network partition or quorum loss scenario, nodes must fail closed rather than risk double-mutations or conflicting Gitea states.
|
||||
|
||||
### 2.2 Non-Goals
|
||||
* **Replacing Gitea:** We do not replace Gitea issue/PR tracking with an independent database.
|
||||
* **Immediate Multi-Node Cluster Execution in v1:** This ADR defines the target architecture and phased roadmap; immediate implementation occurs incrementally post-[#655] v1.
|
||||
|
||||
---
|
||||
|
||||
## 3. High-Availability & Rolling-Restart Architecture
|
||||
|
||||
### 3.1 Architecture Overview
|
||||
|
||||
```
|
||||
+----------------------------+
|
||||
| LLM Clients / IDE Sessions |
|
||||
+--------------+-------------+
|
||||
|
|
||||
v
|
||||
+----------------------------+
|
||||
| HA Proxy / Router |
|
||||
| (Health-based & Affinity) |
|
||||
+------+--------------+------+
|
||||
| |
|
||||
+--------------+ +--------------+
|
||||
v v
|
||||
+--------------------+ +--------------------+
|
||||
| MCP Instance Node A| | MCP Instance Node B|
|
||||
| (Version N) | | (Version N+1) |
|
||||
+---------+----------+ +---------+----------+
|
||||
| |
|
||||
+----------------------+----------------------+
|
||||
|
|
||||
v
|
||||
+----------------------------+
|
||||
| Control-Plane DB Substrate|
|
||||
| (Shared Lease & Locks) |
|
||||
+--------------+-------------+
|
||||
|
|
||||
v
|
||||
+----------------------------+
|
||||
| Gitea API |
|
||||
+----------------------------+
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### 3.2 Key System Components
|
||||
|
||||
#### A. Multiple MCP Instance Cohorts
|
||||
* The control plane runs across $N \ge 2$ redundant process nodes.
|
||||
* Dual-namespace deployment allows running the old version (Node A) alongside a updated version (Node B) during rolling upgrades.
|
||||
|
||||
#### B. Shared Durable Session Storage & Resume Tokens
|
||||
* Session context, preflight verification proofs, and capability resolution states are stored in the shared control-plane database.
|
||||
* Client requests carry an explicit `session_id` and `resume_token`. If an MCP instance restarts or a request routes to a different instance, the target node validates the token against the database without requiring full session re-initialization.
|
||||
|
||||
#### C. Shared Lease Authority & Fencing Counters
|
||||
* Workflow leases (`gitea_allocate_next_work`, `gitea_adopt_workflow_lease`) use monotonic fencing tokens (`lease_generation_id`).
|
||||
* When Node B acquires or renews a lease, it increments the generation counter. Any delayed or out-of-order write attempt from Node A using an older generation token is rejected by database constraints.
|
||||
|
||||
#### D. Leader Election & Coordinated Drain
|
||||
* Node clusters elect a primary coordinator node for administrative background tasks (such as stale lease cleanup or incident Watchdogs).
|
||||
* During a rolling deployment:
|
||||
1. Node B (new version) is launched and registers as healthy.
|
||||
2. Router directs new session creations to Node B.
|
||||
3. Node A enters `MAINTENANCE_DRAIN` status ([#659](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/659)), completing in-flight mutations while refusing new tasks.
|
||||
4. Once all active sessions migrate or complete, Node A shuts down cleanly.
|
||||
|
||||
#### E. Idempotent Mutations & Failover Safety
|
||||
* All state-changing tool executions (PR creation, review submission, merge operations, label changes) carry a deterministic `idempotency_key`.
|
||||
* If a network connection flaps or a node fails mid-mutation, the re-issued request with the same `idempotency_key` is recognized by the control-plane substrate, returning the existing recorded result without repeating side effects on Gitea.
|
||||
|
||||
#### F. Schema Version Compatibility
|
||||
* Database migrations follow non-breaking additive patterns.
|
||||
* During rolling upgrades where Node A (Version $N$) and Node B (Version $N+1$) run concurrently, both versions operate against the shared schema without structural conflicts.
|
||||
|
||||
---
|
||||
|
||||
## 4. Split-Brain & Failure Behavior
|
||||
|
||||
### 4.1 Split-Brain Risk Scenarios & Mitigation
|
||||
|
||||
| Scenario | Risk | Mitigation Strategy |
|
||||
|---|---|---|
|
||||
| **Network Partition between Nodes** | Both Node A and Node B attempt to process operations for the same issue/PR. | **Generation Fencing:** Lease renewal requires updating the DB generation counter. The node isolated from the DB fails closed immediately. |
|
||||
| **Stale Node Recovery** | Node A recovers after a long pause and executes a queued mutation. | **Lease Expiry & TTL Fencing:** Transactions verify that `expires_at > NOW()` within the atomic SQLite transaction boundaries. |
|
||||
| **Database Connection Loss** | Node loses access to shared control-plane DB substrate. | **Strict Fail-Closed:** The node immediately marks all task capabilities as `blocked` and rejects mutation tools until DB connectivity is re-established. |
|
||||
|
||||
---
|
||||
|
||||
## 5. Phased Implementation Milestones
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
M1[Milestone 1: Shared Control-Plane DB Schema & Resume Tokens] --> M2[Milestone 2: Idempotent Mutation Layer]
|
||||
M2 --> M3[Milestone 3: Health Routing & Standby Failover]
|
||||
M3 --> M4[Milestone 4: Active-Active Rolling Deployment & Auto-Drain]
|
||||
```
|
||||
|
||||
### Milestone 1: Shared Control-Plane DB Schema & Resume Tokens (Post-#655)
|
||||
* Extend [#613] Control-Plane DB schema to store multi-instance node heartbeat records and session resume tokens.
|
||||
* Enable session lookup across instances via `session_id`.
|
||||
|
||||
### Milestone 2: Idempotent Mutation Layer & Lease Fencing
|
||||
* Add mandatory `idempotency_key` tracking to all Gitea mutation tools.
|
||||
* Implement monotonic lease fencing counters in `gitea_allocate_next_work` and `gitea_adopt_workflow_lease`.
|
||||
|
||||
### Milestone 3: Health-Based Routing & Active-Passive Standby
|
||||
* Introduce lightweight proxy/router capable of checking node health endpoints.
|
||||
* Implement active-standby failover where standby node automatically assumes work if active node fails health checks.
|
||||
|
||||
### Milestone 4: Active-Active Horizontal Deployment & Rolling Upgrade Automation
|
||||
* Enable true active-active multi-instance execution.
|
||||
* Integrate automated zero-downtime rolling upgrades coordinated with `gitea_request_mcp_restart` maintenance drain.
|
||||
|
||||
---
|
||||
|
||||
## 6. Observability & Audit Requirements
|
||||
|
||||
High-availability control plane operations must expose clear telemetry and audit trails:
|
||||
|
||||
* **Node Registry Telemetry:** Active nodes, version numbers, uptime, and heartbeat timestamps reported via `gitea_get_runtime_context`.
|
||||
* **Lease Fencing Metrics:** Tracking lease acquire latency, fence rejection counts, and lease handoff durations.
|
||||
* **Failover & Re-route Audit Logs:** Durable logging of session migrations between nodes, drain initiation, and process retirement events.
|
||||
|
||||
---
|
||||
|
||||
## 7. Tradeoffs & Accepted Risks
|
||||
|
||||
* **Increased Architectural Complexity:** Moving from a single process to a multi-instance control plane requires robust DB locking, proxy routing, and migration governance.
|
||||
* **Database Dependency:** The control-plane database substrate becomes a critical shared dependency for multi-node deployments. High availability for the underlying SQLite file system / DB must be guaranteed.
|
||||
@@ -0,0 +1,70 @@
|
||||
# MCP Config Drift Diagnostic & Sanctioned Repair Runbook (#672)
|
||||
|
||||
This document describes the diagnostic framework for detecting configuration drift between the active IDE MCP configuration (`~/.gemini/antigravity-ide/mcp_config.json`) and the offline/global canonical configuration (`~/.gemini/config/mcp_config.json`), and establishes the **sanctioned repair runbook**.
|
||||
|
||||
## Background & Problem Statement
|
||||
|
||||
Offline tools like `test_mcp_conn.py` test the global configuration (`~/.gemini/config/mcp_config.json`) via `subprocess.Popen`. However, the active IDE/client namespace uses `~/.gemini/antigravity-ide/mcp_config.json`. When required Gitea role servers (`gitea-author`, `gitea-reviewer`, `gitea-merger`, `gitea-reconciler`, `gitea-controller`, `gitea-tools`) are missing or carry mismatched profile environments in the active IDE config:
|
||||
|
||||
1. Offline tests pass (`test_mcp_conn.py` green).
|
||||
2. The IDE client returns `EOF` / `transport closed` when attempting role-scoped mutations.
|
||||
3. Operators misdiagnose missing server definitions as stale runtimes, leading to forbidden `pkill` attempts (#630) or `mtime` hacks (#655).
|
||||
|
||||
## Diagnostic Tool: `mcp_config_drift.py`
|
||||
|
||||
Run the diagnostic tool directly to compare configurations:
|
||||
|
||||
```bash
|
||||
python3 mcp_config_drift.py --json
|
||||
```
|
||||
|
||||
Or specify custom config locations:
|
||||
|
||||
```bash
|
||||
python3 mcp_config_drift.py \
|
||||
--active-config ~/.gemini/antigravity-ide/mcp_config.json \
|
||||
--global-config ~/.gemini/config/mcp_config.json
|
||||
```
|
||||
|
||||
### Key Diagnostic Outputs
|
||||
|
||||
- `in_sync`: Boolean indicating if all required Gitea role servers exist in the active IDE config with matching profile declarations.
|
||||
- `missing_role_servers`: List of role servers present in global config but missing from active IDE config.
|
||||
- `profile_mismatches`: List of profile environment mismatches per server.
|
||||
- `reasons`: Explicit, human-readable list of drift causes.
|
||||
|
||||
All returned payloads automatically redact secret tokens, DSNs, Authorization headers, and private keys.
|
||||
|
||||
---
|
||||
|
||||
## Sanctioned Repair Path (Step-by-Step)
|
||||
|
||||
When `mcp_config_drift.py` reports drift (`in_sync: false`), execute the following **sanctioned repair steps**:
|
||||
|
||||
1. **Backup Active IDE Config:**
|
||||
```bash
|
||||
cp ~/.gemini/antigravity-ide/mcp_config.json ~/.gemini/antigravity-ide/mcp_config.json.bak
|
||||
```
|
||||
2. **Patch Active IDE Config:**
|
||||
Copy the missing Gitea role server JSON blocks (`gitea-author`, `gitea-reviewer`, etc.) from `~/.gemini/config/mcp_config.json` into `~/.gemini/antigravity-ide/mcp_config.json`.
|
||||
3. **Reconnect via IDE/Client:**
|
||||
Use the IDE / client UI reconnection control (or restart the IDE client app).
|
||||
4. **Verify Active Namespace Health:**
|
||||
Invoke `gitea_whoami` (and optional `gitea_resolve_task_capability`) through the active IDE client on each required role namespace.
|
||||
|
||||
---
|
||||
|
||||
## FORBIDDEN Repair Actions (#630 / #655)
|
||||
|
||||
The following actions are **strictly forbidden** for config drift repair:
|
||||
|
||||
- ❌ **`pkill` or manual daemon process kill commands:** Process kills cause contamination and break active session leases.
|
||||
- ❌ **`mtime` touch edits:** Artificial mtime modifications mask stale runtimes without updating configuration.
|
||||
- ❌ **Source code edits:** Mutating python tool logic to bypass missing server entries.
|
||||
- ❌ **Session-state edits:** Direct database or lock-file state mutation.
|
||||
|
||||
---
|
||||
|
||||
## Final Report Guidelines
|
||||
|
||||
A workflow final report **must not** rely on offline `test_mcp_conn.py` output alone. Final reports must include active-config evidence from live `gitea_whoami` calls on the active IDE namespaces.
|
||||
@@ -0,0 +1,94 @@
|
||||
# MCP scoped recovery playbook (#669)
|
||||
|
||||
**Parent:** [#655](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/655)
|
||||
**Vision / roadmap:** [#652](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/652) · [#653](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/653)
|
||||
**Class matrix:** [#663](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/663) · `docs/mcp-restart-classes.md`
|
||||
**Coordinator:** [#658](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/658) · `restart_coordinator.py`
|
||||
**Audit lineage:** [#665](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/665)
|
||||
|
||||
## Decision
|
||||
|
||||
Full-server MCP reset is a **last resort**. Prefer the narrowest recovery that
|
||||
can clear the symptom. The coordinator **refuses** `rolling_mcp_restart`,
|
||||
`full_mcp_restart`, and `host_restart` unless:
|
||||
|
||||
1. The inventory carries a prior **attempt log** of at least one *insufficient*
|
||||
narrower recovery, **or**
|
||||
2. **Break-glass** is authorized
|
||||
(`request_break_glass` + `GITEA_BREAKGLASS_RESTART_AUTHORIZATION`).
|
||||
|
||||
Break-glass still never bypasses the #663 class matrix (role/permission).
|
||||
|
||||
## Ladder (narrow → broad)
|
||||
|
||||
| Rank | Action | Self-service | Implementation / delegation |
|
||||
|---:|---|---|---|
|
||||
| 0 | `client_reconnect` | yes | Host auto-reconnect / client reconnect · [#584](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/584) · `docs/mcp-namespace-eof-recovery.md` |
|
||||
| 1 | `capability_refresh` | yes | `gitea_resolve_task_capability` + `gitea_whoami` · [#610](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/610) · [#685](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/685) |
|
||||
| 2 | `session_reconnect` | yes | Runtime rebind + explicit `worktree_path` · [#543](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/543) · [#618](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/618) |
|
||||
| 3 | `configuration_reload` | no | Class `configuration_reload` · console reload · [#642](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/642) |
|
||||
| 4 | `lease_recovery` | no | Lock/lease recovery paths · [#702](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/702) · [#753](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/753) · [#790](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/790) |
|
||||
| 5 | `worker_restart` | no | Class `worker_restart` · #663 |
|
||||
| 6 | `role_runtime_restart` | no | Class `role_runtime_restart` · console restart · #642/#663 |
|
||||
| 7 | `connector_restart` | no | Class `connector_restart` · #663 |
|
||||
| 8 | `rolling_mcp_restart` | no | Class `rolling_mcp_restart` · design [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668) · **attempt log required** |
|
||||
| 9 | `full_mcp_restart` | no | Class `full_mcp_restart` · **attempt log required** |
|
||||
| 10 | `host_restart` | no | Class `host_restart` · **attempt log required** |
|
||||
|
||||
Machine-readable source of truth: `recovery_playbook.RECOVERY_LADDER` and
|
||||
`recovery_playbook.ladder_document()`.
|
||||
|
||||
## Attempt log shape
|
||||
|
||||
Each prior attempt is a mapping:
|
||||
|
||||
```json
|
||||
{
|
||||
"action": "client_reconnect",
|
||||
"outcome": "insufficient",
|
||||
"reason": "transport still closed after IDE reconnect",
|
||||
"actor": "prgs-controller-12345",
|
||||
"recorded_at": "2026-07-25T21:00:00+00:00"
|
||||
}
|
||||
```
|
||||
|
||||
Outcomes that count toward escalation: `failed`, `insufficient`, `denied`,
|
||||
`unresolved`, `timeout`, `error`.
|
||||
|
||||
Pass attempts into the coordinator via inventory
|
||||
`prior_recovery_attempts` or the MCP tool argument
|
||||
`prior_recovery_attempts_json` on `gitea_request_mcp_restart`.
|
||||
|
||||
Helper: `recovery_playbook.build_attempt_record(...)`.
|
||||
|
||||
## Symptom → first rung
|
||||
|
||||
`recovery_playbook.recommend_actions(symptoms=[...])` maps symptoms such as
|
||||
`transport_eof`, `stale_capability`, `stale_lease`, `daemon_corrupt` to the
|
||||
narrowest recommended action, then walks the ladder. Soft recommendations
|
||||
never replace the hard gate on broad restarts.
|
||||
|
||||
## Enforcement points
|
||||
|
||||
1. **`recovery_playbook.assess_escalation`** — pure gate.
|
||||
2. **`restart_coordinator.evaluate_restart_impact`** — when `restart_class` is
|
||||
set (policy-enforced path), broad classes require the gate; report fields
|
||||
`attempt_log_satisfied`, `playbook_escalation`, `break_glass`.
|
||||
3. **`gitea_request_mcp_restart`** — accepts attempt JSON and env-authorized
|
||||
break-glass; never restarts a process.
|
||||
|
||||
## Metrics
|
||||
|
||||
`recovery_playbook.recovery_metrics(attempts)` reports the fraction of
|
||||
successful recoveries that avoided full/host restart
|
||||
(`fraction_avoided_full_restart`).
|
||||
|
||||
## Non-goals
|
||||
|
||||
* HA multi-instance execution (#668 design only here).
|
||||
* Normalizing `pkill` (#630 contamination stays forbidden).
|
||||
* Silent mutation of leases or processes from the playbook itself.
|
||||
|
||||
## Manual process kills
|
||||
|
||||
Remain forbidden and contaminating (#630). The playbook never recommends them.
|
||||
@@ -95,12 +95,18 @@ gitea_request_mcp_restart(remote, host, org, repo,
|
||||
target_session_id=None, target_role=None,
|
||||
target_connector=None,
|
||||
drain_proof_json=None,
|
||||
request_break_glass=False)
|
||||
request_break_glass=False,
|
||||
prior_recovery_attempts_json=None)
|
||||
```
|
||||
|
||||
It **never restarts anything**: `apply_supported` is always `false` and
|
||||
`restart_performed` is always `false`.
|
||||
|
||||
`prior_recovery_attempts_json` (#669) is an optional JSON array of prior
|
||||
narrow recovery attempts. Rolling / full / host classes require at least one
|
||||
*insufficient* narrower attempt (or authorized break-glass). See
|
||||
`docs/mcp-recovery-playbook.md`.
|
||||
|
||||
### Dry-run versus apply
|
||||
|
||||
| Call | Behavior |
|
||||
@@ -112,9 +118,10 @@ It **never restarts anything**: `apply_supported` is always `false` and
|
||||
|
||||
An apply requires **both** authorizations, and they are independent:
|
||||
|
||||
1. **Restart-class authorization** (#663) — the requester's role and permissions
|
||||
must allow the requested class, the class's approval requirement must be
|
||||
satisfied, and any target-scoped class must name its target. Failing any of
|
||||
1. **Restart-class authorization** (#663 / #669) — the requester's role and
|
||||
permissions must allow the requested class, the class's approval requirement
|
||||
must be satisfied, any target-scoped class must name its target, and broad
|
||||
classes must satisfy the recovery-playbook attempt-log gate. Failing any of
|
||||
these makes `allow_restart` `false`.
|
||||
2. **Drain-proof gate** (#661) — a valid, unexpired, clean proof bound to the
|
||||
current impact fingerprint, or an authorized break-glass.
|
||||
@@ -127,7 +134,10 @@ the authorization that produced it.
|
||||
|
||||
### Break-glass
|
||||
|
||||
Break-glass bypasses the **drain proof only** — never the restart-class matrix.
|
||||
Break-glass bypasses the **drain proof only** — never the restart-class matrix
|
||||
(role/permission). Separately, authorized break-glass also satisfies the #669
|
||||
attempt-log requirement for broad restarts (rolling/full/host), because that
|
||||
gate is not a class-matrix permission check.
|
||||
It is honoured solely when `request_break_glass` is set *and* the environment
|
||||
carries `GITEA_BREAKGLASS_RESTART_AUTHORIZATION`; like operator override, the
|
||||
tool argument expresses caller intent and cannot be self-asserted by a worker
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
# Sanctioned Recovery Playbooks & Controls (Phase 2 #644)
|
||||
|
||||
## Overview
|
||||
|
||||
Stale runtimes, worktree binding mismatches, and un-reconciled merged branches previously required expert manual shell recovery. Manual process kills (`pkill -f mcp_server.py`) are strictly forbidden and classified as runtime contamination ([#630](sanctioned-restart-controls.md)).
|
||||
|
||||
Phase 2 introduces **sanctioned recovery playbooks and controls** into the Web Console:
|
||||
- **Diagnose**: Surface stale runtimes, worktree binding errors, contamination markers, and worktree anomalies via health & inventory APIs.
|
||||
- **Preview**: Render mutation ledgers and exact confirmation phrases for recovery playbooks.
|
||||
- **Confirm & Apply**: Execute sanctioned recovery actions through gated, audited paths.
|
||||
- **Verify**: Revalidate control-plane state post-recovery before claiming clean status.
|
||||
|
||||
---
|
||||
|
||||
## Recovery Playbook Taxonomy
|
||||
|
||||
| Playbook ID | Action ID | Minimum Role | Target / Scope | Description |
|
||||
|---|---|---|---|---|
|
||||
| `clear_stale_binding` | `system.clear_stale_binding` | Operator | Active worktree binding | Clear provably missing or superseded `GITEA_ACTIVE_WORKTREE` binding ([#702](../stale_binding_recovery.py)). |
|
||||
| `rebind_session_worktree` | `system.rebind_session_worktree` | Operator | Session worktree | Rebind or synchronize session worktree to verified lease worktree ([#864](../dirty_same_claimant_session_rebind.py)). |
|
||||
| `reconcile_cleanups` | `system.reconcile_cleanups` | Controller | Worktree hygiene | Execute reconciler cleanup preview and apply for merged/superseded PR branches. |
|
||||
| `sanctioned_restart` | `system.restart_namespace` | Admin | MCP Namespace | Restart MCP daemon gracefully via host supervisor ([#642](sanctioned-restart-controls.md)). |
|
||||
|
||||
---
|
||||
|
||||
## Wizard Workflow (Diagnose → Preview → Confirm → Verify)
|
||||
|
||||
### 1. Diagnose (`GET /api/v1/system/recovery/diagnose`)
|
||||
Runs control-plane diagnostics:
|
||||
- **Stale Runtime**: Mismatch between running daemon HEAD, local checkout HEAD, and remote-tracking HEAD.
|
||||
- **Worktree Binding**: Missing path (`provably_stale_missing_path`), unverified inherited binding (`unverified_inherited`), or superseded binding (`superseded_by_session_lease`).
|
||||
- **Contamination**: Checks for live contamination markers from unmanaged process kills.
|
||||
- **Worktree Anomalies**: Scans `branches/` directory for un-reconciled cleanups or missing preserved worktrees.
|
||||
|
||||
Returns `RecoveryDiagnosis` with eligible playbooks.
|
||||
|
||||
### 2. Preview (`POST /api/v1/system/recovery/preview`)
|
||||
Takes `playbook_id` and optional `target`/`params`.
|
||||
Returns:
|
||||
- **Mutation Ledger**: Step-by-step sequence of actions.
|
||||
- **Confirmation Phrase**: Exact phrase required to authorize execution (e.g., `confirm clear_stale_binding`).
|
||||
- **Authorization Decision**: RBAC check against the operator's principal.
|
||||
|
||||
### 3. Apply (`POST /api/v1/system/recovery/apply`)
|
||||
Requires `playbook_id` and matching `confirmation` phrase. Gates run in this order, and each fails closed before anything is mutated:
|
||||
|
||||
1. **RBAC and execution phase** (`console_authz.authorize(..., for_execution=True)`). The phase branch only applies when `for_execution` is set. While `ACTIVE_PHASE` is `1`, every phase-2 recovery action is refused with `phase_not_active`, so no recovery playbook writes yet. Preview reports the same decision under `execution_authorization` / `execution_blocked_reason`.
|
||||
2. **Confirmation phrase** (`confirmation_matches`).
|
||||
3. **Contamination rules** ([#630](sanctioned-restart-controls.md)): the live marker is read from the session inventory and assessed under the gated task key `console_recovery_apply`. A contaminated runtime must be cleared through the reconciler cleanup playbook, which is the one playbook exempted from this gate because it is the designated remedy. The marker is also forwarded to `sanctioned_restart.execute_restart`, so a restart cannot launder a contaminated runtime.
|
||||
|
||||
Apply then executes the sanctioned recovery logic against the **live** process environment — not a copy — and records an audit entry in `console_audit`. A playbook that leaves the binding unchanged reports `performed: false`; `binding_before`, `binding_after`, and `binding_changed` are returned so a no-op cannot read as success.
|
||||
|
||||
Apply does **not** enforce master parity. Parity is reported by Diagnose ([#610](../master_parity_gate.py)) as evidence for the operator; it is not a precondition of this endpoint.
|
||||
|
||||
### 4. Verify (`POST /api/v1/system/recovery/verify`)
|
||||
Re-evaluates control-plane diagnostics post-recovery and **reports** `clean`, `stale_runtime_clean`, `binding_clean`, `binding_classification`, and `contamination_clean`. It reports; it does not assert or block. State is read fresh rather than from the mapping a mutation just wrote. An `unverified_inherited` binding is reported as not clean, because unproven is not clean.
|
||||
|
||||
---
|
||||
|
||||
## Safety & Governance Principles
|
||||
|
||||
1. **No Manual `pkill`**: Direct process killing remains forbidden and is recorded as contamination.
|
||||
2. **Auditability**: Every recovery preview and execution is logged in the console audit trail.
|
||||
3. **Master Parity & Dual Control**: High-privilege recovery actions require controller/admin roles and explicit confirmation phrases.
|
||||
@@ -94,6 +94,10 @@ already define, and a regression test asserts each mapping matches.
|
||||
| `record_analytics_usage` | operator | gated_write | `runtime.record_analytics_usage` | Yes | No | No | 2 |
|
||||
| `system.reload_namespace` | controller | privileged | `runtime.reload_namespace` | Yes | No | No | 2 |
|
||||
| `system.restart_namespace` | admin | destructive | `runtime.restart_namespace` | Yes | **Yes** | **Yes** | 2 |
|
||||
| `system.clear_stale_binding` | operator | gated_write | `gitea.read` | Yes | No | No | 2 |
|
||||
| `system.rebind_session_worktree` | operator | gated_write | `gitea.read` | Yes | No | No | 2 |
|
||||
| `system.reconcile_cleanups` | controller | privileged | `gitea.pr.close` | Yes | No | No | 2 |
|
||||
| `initiate_workflow` | operator | gated_write | `gitea.read` | Yes | No | No | 2 |
|
||||
|
||||
**Dual control** means the acting principal may not be the sole authority: a
|
||||
second distinct principal must confirm. **Break-glass** means the action is
|
||||
@@ -112,6 +116,12 @@ by the console — both hand off to a host supervisor, and neither exposes a raw
|
||||
process kill. See
|
||||
[`sanctioned-restart-controls.md`](sanctioned-restart-controls.md) (#642).
|
||||
|
||||
`initiate_workflow` (#643) is operator-class because its outcome is a *claim*,
|
||||
not a Gitea verdict. Requesting reviewer or merger work reserves that work
|
||||
through the allocator; it does not grant the right to approve or merge, which
|
||||
stays with the MCP role profile and its own capability gates. See
|
||||
[`webui-requests.md`](webui-requests.md).
|
||||
|
||||
### Authorization decision
|
||||
|
||||
`authorize(action_id, principal, for_execution=False)` returns a decision
|
||||
@@ -126,9 +136,24 @@ record and **denies by default**. The deny reasons are closed and enumerated:
|
||||
| `phase_not_active` | Execution requested for an action whose phase is not open. |
|
||||
| `allowed_preview_only` | Authorized — preview only, execution still disabled. |
|
||||
|
||||
There is no implicit allow branch. Even the allow result reports
|
||||
`execution_enabled: false` while the console is in Phase 1, so no caller can
|
||||
read an allow as permission to mutate.
|
||||
There is no implicit allow branch.
|
||||
|
||||
`execution_enabled` on the decision reports whether the action has a live
|
||||
execution path at all, and is computed by `execution_wired(action)`. There are
|
||||
exactly two ways to be wired:
|
||||
|
||||
1. the action's `phase` is at or below `ACTIVE_PHASE`; or
|
||||
2. the action declares an `execution_env_flag` **and** that variable is set.
|
||||
|
||||
Every action that declares no flag therefore reports `execution_enabled: false`
|
||||
while the console is in Phase 1, so no caller can read an allow as permission
|
||||
to mutate. The per-action flag exists because raising `ACTIVE_PHASE` would
|
||||
enable execution for every action of that phase at once, including ones whose
|
||||
execution path is not implemented. One implemented action goes live on its own
|
||||
flag instead of dragging its unimplemented phase-mates with it.
|
||||
|
||||
`initiate_workflow` is the only action that currently declares a flag
|
||||
(`WEBUI_REQUESTS_EXECUTION`), and it stays denied until an operator sets it.
|
||||
|
||||
## Secret redaction
|
||||
|
||||
@@ -235,13 +260,22 @@ second one. The integration points are already wired and observable:
|
||||
instead of adding a parallel check.
|
||||
- **`GET /api/console/security-model`** publishes the RBAC matrix, redaction
|
||||
policy, and audit policy as JSON for operators and tests.
|
||||
- **`POST /api/v1/requests/preview` and `.../apply`** (#643) are the first
|
||||
actions to use this model for a real execution path. Preview always returns a
|
||||
decision and an audited `previewed` record; apply requires `confirm=true`,
|
||||
emits `succeeded` or `denied`, and reserves work only through the allocator.
|
||||
See [`webui-requests.md`](webui-requests.md).
|
||||
|
||||
To open Phase 2, a child issue must: raise `ACTIVE_PHASE`, implement the
|
||||
confirmation and dual-control flow the matrix already declares, emit a
|
||||
`succeeded` or `failed` record alongside the `gitea_audit` mutation record, and
|
||||
keep `viewer` unable to reach any of it. Turning on execution without the
|
||||
confirmation flow contradicts a declared requirement and is a review failure,
|
||||
not a shortcut.
|
||||
A Phase 2 action must: use `execution_wired` rather than a private enable flag,
|
||||
implement the confirmation and dual-control flow the matrix already declares,
|
||||
emit a `succeeded` or `failed` record alongside the `gitea_audit` mutation
|
||||
record, and keep `viewer` unable to reach any of it. Turning on execution
|
||||
without the confirmation flow contradicts a declared requirement and is a
|
||||
review failure, not a shortcut.
|
||||
|
||||
Raising `ACTIVE_PHASE` remains the way to open a whole phase at once, and is
|
||||
deliberately *not* what #643 did: an action-scoped opt-in cannot enable an
|
||||
action whose execution path nobody wrote.
|
||||
|
||||
## Local-dev mode
|
||||
|
||||
@@ -294,6 +328,7 @@ Until Phase 2 wires it, probe protection rests on network placement alone, as
|
||||
| `WEBUI_ROLE_MAP` | unset | JSON subject → role map |
|
||||
| `WEBUI_REQUIRE_PROBE_AUTH` | unset | Require auth for non-public probes |
|
||||
| `WEBUI_CONSOLE_AUDIT_LOG` | unset | Append-only audit sink path |
|
||||
| `WEBUI_REQUESTS_EXECUTION` | unset | Opt in to `initiate_workflow` execution (#643) |
|
||||
|
||||
All are read server-side only. None is ever rendered into a page or returned by
|
||||
an API.
|
||||
|
||||
@@ -0,0 +1,160 @@
|
||||
# Web console requests: intent preview and workflow initiation (#643)
|
||||
|
||||
**Phase 2. Preview is always live and always read-only. Initiation is wired but
|
||||
denied until an operator opts in.**
|
||||
|
||||
Before this surface, starting role work meant pasting a prompt into a terminal
|
||||
and trusting the operator to have checked the allocator first. Nothing enforced
|
||||
that check, so two sessions could reach for the same issue and each believe it
|
||||
was theirs. This page replaces the paste with a *request*: a desired role, an
|
||||
issue or PR, and a stated intent, answered by an authorization decision and —
|
||||
on confirmation — an exclusive assignment from the allocator.
|
||||
|
||||
| Concern | Module |
|
||||
|---------|--------|
|
||||
| Request model, preview, initiation | `webui/request_service.py` |
|
||||
| Form and preview rendering | `webui/request_views.py` |
|
||||
| Authorization | `webui/console_authz.py` (`initiate_workflow`) |
|
||||
| Audit | `webui/console_audit.py` |
|
||||
| Ownership substrate | `allocator_service.py` + `control_plane_db.py` |
|
||||
|
||||
## Surfaces
|
||||
|
||||
| Path | Method | Purpose |
|
||||
|------|--------|---------|
|
||||
| `/requests` | GET | Request form |
|
||||
| `/requests` | POST | Render an intent preview. **Never assigns.** |
|
||||
| `/api/v1/requests/preview` | POST | Intent preview as JSON |
|
||||
| `/api/v1/requests/apply` | POST | Initiate — confirmed, audited, allocator-owned |
|
||||
|
||||
The HTML form has no initiate button on purpose. Initiating requires a
|
||||
confirmed POST to `/api/v1/requests/apply`, so a stray form submission cannot
|
||||
reserve work as a side effect.
|
||||
|
||||
## The request
|
||||
|
||||
```json
|
||||
{
|
||||
"desired_role": "author",
|
||||
"work_kind": "issue",
|
||||
"work_number": 643,
|
||||
"intent_summary": "implement request preview and initiation",
|
||||
"remote": "prgs",
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
"expected_head_sha": null
|
||||
}
|
||||
```
|
||||
|
||||
`desired_role` is one of `author`, `reviewer`, `merger`, `reconciler`,
|
||||
`controller`. `work_kind` is `issue` or `pr`. `remote`/`org`/`repo` default to
|
||||
the first project in the registry when omitted; when neither the request nor
|
||||
the registry resolves them, the request is rejected rather than pointed at some
|
||||
other repository. `intent_summary` is required — it is what the audit record
|
||||
states as the reason — and is truncated to 500 characters.
|
||||
|
||||
Parsing rejects rather than corrects. An unknown role, an unknown work kind, a
|
||||
non-positive number, or a missing intent each return `400` with a `reason_code`
|
||||
and the offending `field`.
|
||||
|
||||
## Preview
|
||||
|
||||
Five checks, each with its own verdict, reason code, and detail:
|
||||
|
||||
| Check | Passes when |
|
||||
|-------|-------------|
|
||||
| `authorization` | The console principal holds `operator` or above |
|
||||
| `capability` | The desired role maps to a declared profile and MCP namespace |
|
||||
| `lease_availability` | No active claim holds the work unit |
|
||||
| `next_safe_action` | The allocator would independently select this exact work unit |
|
||||
| `head_pin` | PR work resolves to a head SHA, and a supplied SHA still matches |
|
||||
|
||||
A preview also returns the role's `allowed_actions` and `prohibited_actions`
|
||||
(from `allocator_service.ROLE_ACTIONS`), the `required_profile` and
|
||||
`required_namespace` the work must run under, and a `correlation_id` that ties
|
||||
the preview to its audit record and to any assignment that follows.
|
||||
|
||||
Preview is read-only in the strict sense: it calls the allocator with
|
||||
`apply=false` and writes nothing but an audit line. An unauthorized principal
|
||||
never reaches the allocator or the control-plane DB at all, so a denial cannot
|
||||
be used to enumerate the queue.
|
||||
|
||||
## Initiation
|
||||
|
||||
`POST /api/v1/requests/apply` refuses in this order, and every refusal returns
|
||||
before any assignment is attempted:
|
||||
|
||||
| Condition | Outcome | Status |
|
||||
|-----------|---------|--------|
|
||||
| Unparseable request | `invalid_request` | 400 |
|
||||
| Not authorized, or execution not wired | `denied` | 403 |
|
||||
| `confirm` not set | `denied` / `confirmation_required` | 409 |
|
||||
| Work unit already claimed | `blocked` / `duplicate_assignment` | 409 |
|
||||
| Allocator would select other work | `wait` / `not_next_safe_work` | 409 |
|
||||
| Allocator declines on apply | `blocked` or `wait` | 409 |
|
||||
| Evidence unavailable | `wait` / `evidence_unavailable` | 503 |
|
||||
| Assigned | `assigned_work` | 201 |
|
||||
|
||||
A success returns the assignment plus a `handoff` block naming the profile, the
|
||||
namespace, and the actions that stay forbidden — enough for the operator to
|
||||
continue in the right MCP namespace without guessing.
|
||||
|
||||
### Why apply runs the allocator twice
|
||||
|
||||
The allocator is the only source of exclusive ownership (#600 / #613), and it
|
||||
selects work; it does not take orders. So `apply` runs a dry-run first and
|
||||
proceeds only when the allocator would independently pick the requested work
|
||||
unit. If it would not, the request reports `wait` and mutates nothing.
|
||||
|
||||
A request is therefore a *confirmation* of the allocator's decision, never an
|
||||
override of it. The apply call carries the dry-run's
|
||||
`candidate_set_fingerprint` as a CAS pin (#776), so a queue that changed
|
||||
between the two calls fails closed rather than assigning against a stale view.
|
||||
The result is checked again on the way out: an assignment naming a different
|
||||
work unit is not read as success.
|
||||
|
||||
### Fail-closed defaults
|
||||
|
||||
- An unreadable control-plane DB denies. It is never treated as "nothing holds
|
||||
this work unit".
|
||||
- An incomplete queue inventory denies (#758). Ranking a partial candidate set
|
||||
can select the wrong work.
|
||||
- An allocator that raises denies.
|
||||
- PR work with no resolvable head SHA denies; a supplied SHA that no longer
|
||||
matches denies with `head_moved`.
|
||||
|
||||
## Enabling initiation
|
||||
|
||||
Execution is wired off. Set `WEBUI_REQUESTS_EXECUTION=1` to enable it for the
|
||||
`initiate_workflow` action only — see
|
||||
[`webui-authz-audit.md`](webui-authz-audit.md) for why this is an
|
||||
action-scoped flag rather than a phase bump. With the variable unset, `apply`
|
||||
returns `403` with `reason_code: unauthorized` no matter who asks.
|
||||
|
||||
Enabling execution does **not** enable approvals or merges. Those are phase 3
|
||||
console actions and remain forbidden in every path here; the console reserves
|
||||
work and hands off, and the MCP role profile enforces what that role may then
|
||||
do.
|
||||
|
||||
## Audit
|
||||
|
||||
Every preview and every apply emits a console audit record (schema in
|
||||
[`webui-authz-audit.md`](webui-authz-audit.md)):
|
||||
|
||||
| Event | `result` |
|
||||
|-------|----------|
|
||||
| Preview | `previewed` |
|
||||
| Refusal at any stage | `denied` |
|
||||
| Assignment created | `succeeded` |
|
||||
|
||||
`correlation.request_id` carries the request's `correlation_id`, and a
|
||||
successful record's `metadata` carries `assignment_id` and `lease_id`, so an
|
||||
assignment can be traced back to the intent that produced it. The operator's
|
||||
`intent_summary` travels in `metadata` and passes through the standard
|
||||
redaction pass before persistence like every other field.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- No browser-initiated approve or merge, in this phase or any other.
|
||||
- No bypass of allocator exclusive ownership; no self-selection of work.
|
||||
- No auto-start from raw monitoring incidents (#612 stays downstream).
|
||||
@@ -0,0 +1,102 @@
|
||||
# Web Console: restart status, impact preview, and approval state (#667)
|
||||
|
||||
Phase 1 of the console restart surface. It consumes the #655 coordinator
|
||||
substrate and displays it. It performs no restart, reload, drain, approval, or
|
||||
process action, and it registers no write endpoint.
|
||||
|
||||
Issue #667's rollout is explicit — *status views first, write approval after the
|
||||
backend gates are green* — and this change delivers only the status half.
|
||||
|
||||
## Surfaces
|
||||
|
||||
| Path | Method | Purpose |
|
||||
|------|--------|---------|
|
||||
| `/runtime/restart` | GET | Restart status page |
|
||||
| `/api/v1/system/restart/status` | GET | Same snapshot as JSON |
|
||||
|
||||
Both accept an optional `restart_class` query parameter (default
|
||||
`full_mcp_restart`). An unrecognised class is not an error: the coordinator
|
||||
resolves it as unknown and fails closed, and the page shows the resulting deny.
|
||||
|
||||
Neither path accepts `POST`; a write attempt returns `405`, and a test asserts
|
||||
it.
|
||||
|
||||
## What it shows
|
||||
|
||||
* **Impact preview (#658)** — verdict, blast radius, affected sessions, leases,
|
||||
critical sections, mutations, and the counts behind them, evaluated
|
||||
`dry_run=True` against live control-plane state.
|
||||
* **Drain proof (#661)** — verification of a supplied proof: valid, clean,
|
||||
expired, tampered, and the reasons behind a refusal.
|
||||
* **Post-restart reconcile (#662)** — the most recent completion proof, its
|
||||
overall status, and which dimensions still require follow-up.
|
||||
* **Restart classes (#663)** — the least-privilege matrix, with *you may
|
||||
request* and *you may execute* computed for the viewing role rather than for a
|
||||
generic operator.
|
||||
* **Approval controls (#633)** — the authorization state of
|
||||
`system.restart_namespace` and `system.reload_namespace`.
|
||||
* **Break-glass (#664)** — declared and marked unavailable; see below.
|
||||
|
||||
## Three rules this surface holds itself to
|
||||
|
||||
A status page that is wrong is worse than one that is missing, because an
|
||||
operator acts on it. Three properties are enforced by tests, and each was
|
||||
verified by reverting the guard and watching a test fail.
|
||||
|
||||
### An unreadable source reports unavailable, never green
|
||||
|
||||
Every source carries its own `SourceStatus`. Nothing substitutes a default,
|
||||
placeholder, or self-comparison for a reading that failed. An unreadable
|
||||
control-plane database yields `inventory_complete: false`, which the coordinator
|
||||
itself turns into a fail-closed verdict, and the page says the blast radius is
|
||||
unknown rather than showing an empty affected-sessions table.
|
||||
|
||||
An absent drain proof is reported as absent — not as a pass. The #661 gate
|
||||
authorizes a restart only against a valid, unexpired, clean proof, so no proof
|
||||
is precisely the state that gate denies on.
|
||||
|
||||
### Authorization is asked the way execution would ask it
|
||||
|
||||
Every probe passes `for_execution=True`.
|
||||
|
||||
Asked without it, an admin is `allowed` for `system.restart_namespace`. On a
|
||||
control surface that reads as a live button. Asked the way an execution attempt
|
||||
would ask, the same principal is refused `phase_not_active`, because the console
|
||||
is in Phase 1 and the action is Phase 2. This surface reports the second answer.
|
||||
|
||||
`execution_enabled` is therefore `false` for every action and every role today,
|
||||
and a test asserts that across the whole role matrix.
|
||||
|
||||
### The control-plane database is opened read-only
|
||||
|
||||
`ControlPlaneDB()` creates directories and runs migrations on construction — a
|
||||
write. This surface never constructs one. It opens the sqlite file with
|
||||
`mode=ro`, exactly as `webui/inventory.py` does, and treats a missing file as
|
||||
missing authority rather than as an empty inventory.
|
||||
|
||||
The test that protects this points at a path inside a directory that already
|
||||
exists, so a read-write `connect` would really create the file. A nested
|
||||
missing-directory path would have passed for the wrong reason.
|
||||
|
||||
## Break-glass is declared, not offered
|
||||
|
||||
The break-glass workflow (#664) is not available on this branch's base. The
|
||||
panel is rendered to operator-class roles as **unavailable**, naming the issue
|
||||
that tracks it. It is not silently omitted, because an operator who has been
|
||||
told a governance path exists needs to see that it is not wired here; and it is
|
||||
not rendered as a control, because there is nothing behind it.
|
||||
|
||||
Unprivileged viewers see only a note that the surface is operator-class.
|
||||
|
||||
## Redaction and escaping
|
||||
|
||||
Every interpolated value passes through `_esc` (`html.escape(..., quote=True)`).
|
||||
Free-form text and anything that can carry a filesystem path additionally passes
|
||||
through `webui.inventory.scrub_text`, which redacts credential-shaped tokens
|
||||
inside a string rather than only at its start. The impact payload is passed
|
||||
through `webui.inventory.scrub` before rendering.
|
||||
|
||||
## Linkage
|
||||
|
||||
Parent #655 · extends #642 · consumes #658, #661, #662, #663 · RBAC #633 ·
|
||||
console #631 · vision #652 · roadmap #653 · break-glass #664.
|
||||
+35
-9
@@ -22575,8 +22575,9 @@ def gitea_request_mcp_restart(
|
||||
target_connector: str | None = None,
|
||||
drain_proof_json: str | None = None,
|
||||
request_break_glass: bool = False,
|
||||
prior_recovery_attempts_json: str | None = None,
|
||||
) -> dict:
|
||||
"""Evaluate a proposed MCP restart and return an impact preview (#658).
|
||||
"""Evaluate a proposed MCP restart and return an impact preview (#658/#669).
|
||||
|
||||
Central restart coordinator: resolves the requested restart class, gathers
|
||||
live control-plane state (sessions,
|
||||
@@ -22600,10 +22601,16 @@ def gitea_request_mcp_restart(
|
||||
independent — the drain gate proves the blast radius was drained and knows
|
||||
nothing about whether this requester may request this class — so a class the
|
||||
matrix denied never reports an authorized apply. Break-glass bypasses the
|
||||
drain proof only; it never bypasses the class matrix. ``apply_gate`` carries
|
||||
drain proof and, when env-authorized, the #669 attempt-log requirement for
|
||||
broad restarts; it never bypasses the class matrix. ``apply_gate`` carries
|
||||
``drain_gate_allow`` and ``restart_class_authorized`` so a denial is
|
||||
attributable to the authorization that produced it.
|
||||
|
||||
``prior_recovery_attempts_json`` (#669) is an optional JSON array of prior
|
||||
narrow recovery attempts ``{action, outcome, reason, ...}``. Rolling / full
|
||||
/ host restart classes require at least one *insufficient* narrower attempt
|
||||
unless break-glass is authorized.
|
||||
|
||||
Operator override authority is read from the process environment
|
||||
(``GITEA_OPERATOR_RESTART_OVERRIDE_AUTHORIZATION``), never self-asserted by
|
||||
the requesting session: ``request_override`` only expresses caller intent
|
||||
@@ -22709,12 +22716,37 @@ def gitea_request_mcp_restart(
|
||||
requester_role
|
||||
)
|
||||
|
||||
prior_recovery_attempts: list[dict] = []
|
||||
if prior_recovery_attempts_json:
|
||||
try:
|
||||
parsed_attempts = json.loads(prior_recovery_attempts_json)
|
||||
if isinstance(parsed_attempts, list):
|
||||
prior_recovery_attempts = [
|
||||
dict(a) for a in parsed_attempts if isinstance(a, dict)
|
||||
]
|
||||
else:
|
||||
incomplete_reasons.append(
|
||||
"prior_recovery_attempts_json must be a JSON array (#669)"
|
||||
)
|
||||
inventory_complete = False
|
||||
except (ValueError, TypeError) as exc:
|
||||
incomplete_reasons.append(
|
||||
f"invalid prior_recovery_attempts_json: {_redact(str(exc))}"
|
||||
)
|
||||
inventory_complete = False
|
||||
|
||||
break_glass_authorized = bool(
|
||||
(os.environ.get("GITEA_BREAKGLASS_RESTART_AUTHORIZATION") or "").strip()
|
||||
)
|
||||
break_glass = bool(request_break_glass and break_glass_authorized)
|
||||
|
||||
inventory = {
|
||||
"sessions": sessions,
|
||||
"leases": leases,
|
||||
"terminal_lock": terminal_lock,
|
||||
"inventory_complete": inventory_complete,
|
||||
"incomplete_reasons": incomplete_reasons,
|
||||
"prior_recovery_attempts": prior_recovery_attempts,
|
||||
}
|
||||
|
||||
report = restart_coordinator.evaluate_restart_impact(
|
||||
@@ -22730,6 +22762,7 @@ def gitea_request_mcp_restart(
|
||||
target_session_id=target_session_id,
|
||||
target_role=target_role,
|
||||
target_connector=target_connector,
|
||||
break_glass=break_glass,
|
||||
)
|
||||
|
||||
payload = report.as_dict()
|
||||
@@ -22762,13 +22795,6 @@ def gitea_request_mcp_restart(
|
||||
except (ValueError, TypeError) as exc:
|
||||
proof_parse_error = f"invalid drain_proof_json: {_redact(str(exc))}"
|
||||
|
||||
break_glass_authorized = bool(
|
||||
(
|
||||
os.environ.get("GITEA_BREAKGLASS_RESTART_AUTHORIZATION") or ""
|
||||
).strip()
|
||||
)
|
||||
break_glass = bool(request_break_glass and break_glass_authorized)
|
||||
|
||||
expected_fp = drain_proof.impact_fingerprint(report.as_dict())
|
||||
gate = drain_proof.gate_apply_restart(
|
||||
proof=proof_obj,
|
||||
|
||||
@@ -0,0 +1,237 @@
|
||||
"""Antigravity IDE vs Global MCP Config Drift Diagnostic (#672).
|
||||
|
||||
Diagnoses config drift between the active IDE MCP configuration
|
||||
(e.g. ``~/.gemini/antigravity-ide/mcp_config.json``) and the offline/global
|
||||
canonical configuration (e.g. ``~/.gemini/config/mcp_config.json``).
|
||||
|
||||
Hard rules (#672 / #630 / #655):
|
||||
* Distinguish offline/global success from active IDE namespace availability.
|
||||
* Never print tokens, DSNs, Authorization headers, or secret-bearing env vars.
|
||||
* Sanctioned repair path is: backup active config -> patch active config from canonical
|
||||
-> reconnect through IDE/client -> verify with live ``gitea_whoami``.
|
||||
* FORBIDDEN: ``pkill``, mtime edits, source edits, or session-state edits for repair.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from webui import console_redaction
|
||||
|
||||
DEFAULT_ACTIVE_IDE_CONFIG = "~/.gemini/antigravity-ide/mcp_config.json"
|
||||
DEFAULT_GLOBAL_CONFIG = "~/.gemini/config/mcp_config.json"
|
||||
|
||||
REQUIRED_GITEA_ROLE_SERVERS = (
|
||||
"gitea-author",
|
||||
"gitea-reviewer",
|
||||
"gitea-merger",
|
||||
"gitea-reconciler",
|
||||
"gitea-controller",
|
||||
"gitea-tools",
|
||||
)
|
||||
|
||||
SANCTIONED_REPAIR_RUNBOOK: tuple[str, ...] = (
|
||||
"1. Backup active IDE config: cp ~/.gemini/antigravity-ide/mcp_config.json ~/.gemini/antigravity-ide/mcp_config.json.bak",
|
||||
"2. Patch active IDE config: copy required missing Gitea role server entries from global config (~/.gemini/config/mcp_config.json) into active IDE config.",
|
||||
"3. Reconnect via IDE/client UI or client restart (do NOT use host process kill).",
|
||||
"4. Verify active namespace health using live gitea_whoami and gitea_resolve_task_capability on each role namespace.",
|
||||
"FORBIDDEN REPAIR PATHS: pkill / host process kill, mtime touch edits, source code edits, or session-state edits.",
|
||||
)
|
||||
|
||||
|
||||
def resolve_config_path(path_str: str) -> Path:
|
||||
"""Expand user and resolve absolute path."""
|
||||
return Path(os.path.expanduser(path_str)).resolve()
|
||||
|
||||
|
||||
def load_mcp_config(config_path: str | Path) -> tuple[dict[str, Any] | None, str | None]:
|
||||
"""Load and parse JSON MCP configuration from file.
|
||||
|
||||
Returns (config_dict, error_message).
|
||||
"""
|
||||
resolved = resolve_config_path(str(config_path))
|
||||
if not resolved.exists():
|
||||
return None, f"file_not_found: {resolved}"
|
||||
try:
|
||||
with open(resolved, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
if not isinstance(data, dict):
|
||||
return None, f"invalid_schema: root is not a JSON object in {resolved}"
|
||||
return data, None
|
||||
except Exception as exc:
|
||||
return None, f"unreadable_json: {exc} in {resolved}"
|
||||
|
||||
|
||||
def extract_mcp_servers(config: dict[str, Any] | None) -> dict[str, dict[str, Any]]:
|
||||
"""Extract the mcpServers or mcp_servers mapping safely."""
|
||||
if not config:
|
||||
return {}
|
||||
servers = config.get("mcpServers") or config.get("mcp_servers") or {}
|
||||
if isinstance(servers, dict):
|
||||
return {str(k): v for k, v in servers.items() if isinstance(v, dict)}
|
||||
return {}
|
||||
|
||||
|
||||
def _safe_redact_server_config(srv_cfg: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Redact secrets from environment variables and command line args."""
|
||||
safe = {}
|
||||
if "command" in srv_cfg:
|
||||
safe["command"] = str(srv_cfg["command"])
|
||||
if "args" in srv_cfg and isinstance(srv_cfg["args"], list):
|
||||
safe["args"] = [console_redaction.redact_text(str(a)) for a in srv_cfg["args"]]
|
||||
if "env" in srv_cfg and isinstance(srv_cfg["env"], dict):
|
||||
safe_env = {}
|
||||
for k, v in srv_cfg["env"].items():
|
||||
if any(secret_kw in k.lower() for secret_kw in ("token", "secret", "pass", "key", "auth")):
|
||||
safe_env[k] = "[REDACTED]"
|
||||
else:
|
||||
safe_env[k] = console_redaction.redact_text(str(v))
|
||||
safe["env"] = safe_env
|
||||
return safe
|
||||
|
||||
|
||||
def analyze_config_drift(
|
||||
active_config_path: str = DEFAULT_ACTIVE_IDE_CONFIG,
|
||||
global_config_path: str = DEFAULT_GLOBAL_CONFIG,
|
||||
) -> dict[str, Any]:
|
||||
"""Analyze MCP configuration drift between active IDE config and global config.
|
||||
|
||||
Returns structured diagnostic output.
|
||||
"""
|
||||
active_resolved = resolve_config_path(active_config_path)
|
||||
global_resolved = resolve_config_path(global_config_path)
|
||||
|
||||
active_cfg, active_err = load_mcp_config(active_resolved)
|
||||
global_cfg, global_err = load_mcp_config(global_resolved)
|
||||
|
||||
active_servers = extract_mcp_servers(active_cfg)
|
||||
global_servers = extract_mcp_servers(global_cfg)
|
||||
|
||||
missing_role_servers: list[str] = []
|
||||
present_role_servers: list[str] = []
|
||||
profile_mismatches: list[dict[str, Any]] = []
|
||||
reasons: list[str] = []
|
||||
|
||||
if active_err:
|
||||
reasons.append(f"Active IDE config error: {active_err}")
|
||||
if global_err:
|
||||
reasons.append(f"Global canonical config error: {global_err}")
|
||||
|
||||
# Check Gitea role servers
|
||||
for srv_name in REQUIRED_GITEA_ROLE_SERVERS:
|
||||
in_active = srv_name in active_servers
|
||||
in_global = srv_name in global_servers
|
||||
|
||||
if in_active:
|
||||
present_role_servers.append(srv_name)
|
||||
elif in_global:
|
||||
missing_role_servers.append(srv_name)
|
||||
reasons.append(
|
||||
f"Missing Gitea role server '{srv_name}' in active IDE config ({active_resolved})"
|
||||
)
|
||||
|
||||
if in_active and in_global:
|
||||
# Compare profiles & environments
|
||||
act_env = active_servers[srv_name].get("env", {}) if isinstance(active_servers[srv_name], dict) else {}
|
||||
glo_env = global_servers[srv_name].get("env", {}) if isinstance(global_servers[srv_name], dict) else {}
|
||||
|
||||
act_prof = act_env.get("GITEA_MCP_PROFILE") or act_env.get("GITEA_PROFILE_NAME")
|
||||
glo_prof = glo_env.get("GITEA_MCP_PROFILE") or glo_env.get("GITEA_PROFILE_NAME")
|
||||
|
||||
if act_prof != glo_prof:
|
||||
mismatch_item = {
|
||||
"server": srv_name,
|
||||
"active_profile": act_prof,
|
||||
"global_profile": glo_prof,
|
||||
}
|
||||
profile_mismatches.append(mismatch_item)
|
||||
reasons.append(
|
||||
f"Profile mismatch for '{srv_name}': active='{act_prof}' != global='{glo_prof}'"
|
||||
)
|
||||
|
||||
in_sync = bool(
|
||||
not active_err
|
||||
and not global_err
|
||||
and not missing_role_servers
|
||||
and not profile_mismatches
|
||||
)
|
||||
|
||||
report = {
|
||||
"timestamp": datetime.now(timezone.utc).isoformat(),
|
||||
"in_sync": in_sync,
|
||||
"active_config_path": str(active_resolved),
|
||||
"active_config_exists": active_cfg is not None,
|
||||
"global_config_path": str(global_resolved),
|
||||
"global_config_exists": global_cfg is not None,
|
||||
"required_role_servers": list(REQUIRED_GITEA_ROLE_SERVERS),
|
||||
"present_role_servers": present_role_servers,
|
||||
"missing_role_servers": missing_role_servers,
|
||||
"profile_mismatches": profile_mismatches,
|
||||
"reasons": reasons,
|
||||
"sanctioned_repair_runbook": list(SANCTIONED_REPAIR_RUNBOOK),
|
||||
"forbidden_repair_methods": [
|
||||
"pkill / host process kill",
|
||||
"mtime touch edits",
|
||||
"source code edits",
|
||||
"session-state edits",
|
||||
],
|
||||
}
|
||||
|
||||
return console_redaction.redact_payload(report)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Diagnose Gitea MCP role server config drift between active IDE and global config."
|
||||
)
|
||||
parser.add_argument(
|
||||
"--active-config",
|
||||
default=DEFAULT_ACTIVE_IDE_CONFIG,
|
||||
help="Path to active IDE MCP config JSON",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--global-config",
|
||||
default=DEFAULT_GLOBAL_CONFIG,
|
||||
help="Path to global/canonical MCP config JSON",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--json", action="store_true", help="Print raw JSON report"
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
report = analyze_config_drift(args.active_config, args.global_config)
|
||||
|
||||
if args.json:
|
||||
print(json.dumps(report, indent=2))
|
||||
else:
|
||||
print("=== MCP Config Drift Diagnostic Report ===")
|
||||
print(f"Timestamp: {report['timestamp']}")
|
||||
print(f"In Sync: {report['in_sync']}")
|
||||
print(f"Active IDE Config: {report['active_config_path']} (exists={report['active_config_exists']})")
|
||||
print(f"Global Config: {report['global_config_path']} (exists={report['global_config_exists']})")
|
||||
print(f"Present Role Servers: {', '.join(report['present_role_servers']) if report['present_role_servers'] else 'None'}")
|
||||
print(f"Missing Role Servers: {', '.join(report['missing_role_servers']) if report['missing_role_servers'] else 'None'}")
|
||||
if report['profile_mismatches']:
|
||||
print("Profile Mismatches:")
|
||||
for m in report['profile_mismatches']:
|
||||
print(f" - {m['server']}: active={m['active_profile']} vs global={m['global_profile']}")
|
||||
if report['reasons']:
|
||||
print("Drift Reasons:")
|
||||
for r in report['reasons']:
|
||||
print(f" - {r}")
|
||||
print("\nSanctioned Repair Runbook:")
|
||||
for step in report['sanctioned_repair_runbook']:
|
||||
print(f" {step}")
|
||||
|
||||
sys.exit(0 if report["in_sync"] else 1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,583 @@
|
||||
"""Scoped MCP recovery playbook (#669).
|
||||
|
||||
Operational recovery must prefer the *narrowest* action that can fix the
|
||||
symptom. Full MCP / host restarts are last-resort rungs on a documented
|
||||
ladder; the coordinator refuses those rungs unless a prior attempt log
|
||||
shows narrower recoveries already failed (or break-glass is authorized).
|
||||
|
||||
This module is pure classification and recommendation:
|
||||
|
||||
* No network, filesystem, or process I/O.
|
||||
* Never restarts anything.
|
||||
* Narrow recovery *execution* is delegated to existing tools/docs (linked
|
||||
per rung) — the playbook records which rung to try next and whether
|
||||
escalation to a broad restart is allowed.
|
||||
|
||||
Design lineage: umbrella #655, class matrix #663, coordinator #658,
|
||||
auto-reconnect #584, stale-runtime #610, contamination #630, audit #665.
|
||||
Vision #652 / roadmap #653.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from enum import Enum
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
PLAYBOOK_VERSION = "1.0.0-issue-669"
|
||||
|
||||
# Attempt outcomes that count as "tried and insufficient" for escalation.
|
||||
INSUFFICIENT_OUTCOMES = frozenset(
|
||||
{
|
||||
"failed",
|
||||
"insufficient",
|
||||
"denied",
|
||||
"unresolved",
|
||||
"timeout",
|
||||
"error",
|
||||
}
|
||||
)
|
||||
|
||||
# Break-glass / operator override still records that the ladder was skipped.
|
||||
OUTCOME_BREAK_GLASS = "break_glass"
|
||||
OUTCOME_SUCCESS = "success"
|
||||
OUTCOME_SKIPPED = "skipped"
|
||||
|
||||
|
||||
class RecoveryAction(str, Enum):
|
||||
"""Ordered recovery ladder (narrow → broad)."""
|
||||
|
||||
CLIENT_RECONNECT = "client_reconnect"
|
||||
CAPABILITY_REFRESH = "capability_refresh"
|
||||
SESSION_RECONNECT = "session_reconnect"
|
||||
CONFIGURATION_RELOAD = "configuration_reload"
|
||||
LEASE_RECOVERY = "lease_recovery"
|
||||
WORKER_RESTART = "worker_restart"
|
||||
ROLE_RUNTIME_RESTART = "role_runtime_restart"
|
||||
CONNECTOR_RESTART = "connector_restart"
|
||||
ROLLING_MCP_RESTART = "rolling_mcp_restart"
|
||||
FULL_MCP_RESTART = "full_mcp_restart"
|
||||
HOST_RESTART = "host_restart"
|
||||
|
||||
|
||||
# Classes that require a prior narrow-attempt log (unless break-glass).
|
||||
BROAD_RESTART_ACTIONS: frozenset[RecoveryAction] = frozenset(
|
||||
{
|
||||
RecoveryAction.ROLLING_MCP_RESTART,
|
||||
RecoveryAction.FULL_MCP_RESTART,
|
||||
RecoveryAction.HOST_RESTART,
|
||||
}
|
||||
)
|
||||
|
||||
# Map #663 restart_class strings onto playbook actions.
|
||||
RESTART_CLASS_TO_ACTION: dict[str, RecoveryAction] = {
|
||||
"client_reconnect": RecoveryAction.CLIENT_RECONNECT,
|
||||
"session_reconnect": RecoveryAction.SESSION_RECONNECT,
|
||||
"configuration_reload": RecoveryAction.CONFIGURATION_RELOAD,
|
||||
"worker_restart": RecoveryAction.WORKER_RESTART,
|
||||
"role_runtime_restart": RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||
"connector_restart": RecoveryAction.CONNECTOR_RESTART,
|
||||
"rolling_mcp_restart": RecoveryAction.ROLLING_MCP_RESTART,
|
||||
"full_mcp_restart": RecoveryAction.FULL_MCP_RESTART,
|
||||
"host_restart": RecoveryAction.HOST_RESTART,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RecoveryRung:
|
||||
"""One rung on the recovery ladder."""
|
||||
|
||||
action: RecoveryAction
|
||||
rank: int
|
||||
summary: str
|
||||
# Existing implementation or explicit delegation target.
|
||||
implementation: str
|
||||
issue_links: tuple[str, ...]
|
||||
self_service: bool
|
||||
# Restart-class permission when this rung is requested via coordinator.
|
||||
restart_class: str | None = None
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"action": self.action.value,
|
||||
"rank": self.rank,
|
||||
"summary": self.summary,
|
||||
"implementation": self.implementation,
|
||||
"issue_links": list(self.issue_links),
|
||||
"self_service": self.self_service,
|
||||
"restart_class": self.restart_class,
|
||||
}
|
||||
|
||||
|
||||
# Canonical ladder. Rank 0 is narrowest.
|
||||
RECOVERY_LADDER: tuple[RecoveryRung, ...] = (
|
||||
RecoveryRung(
|
||||
RecoveryAction.CLIENT_RECONNECT,
|
||||
0,
|
||||
"Reconnect the IDE/client MCP transport (EOF / transport flap).",
|
||||
"Host auto-reconnect or explicit client reconnect; "
|
||||
"docs/mcp-namespace-eof-recovery.md",
|
||||
("#584", "#655"),
|
||||
True,
|
||||
"client_reconnect",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.CAPABILITY_REFRESH,
|
||||
1,
|
||||
"Re-resolve task capability and clear stale permission context.",
|
||||
"Delegated: gitea_resolve_task_capability + gitea_whoami "
|
||||
"(no process change).",
|
||||
("#610", "#685", "#655"),
|
||||
True,
|
||||
None,
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.SESSION_RECONNECT,
|
||||
2,
|
||||
"Rebind identity, workspace, and namespace for one session.",
|
||||
"Delegated: gitea_get_runtime_context + explicit worktree_path "
|
||||
"rebind (#618); docs/mcp-namespace-health.md",
|
||||
("#543", "#618", "#655"),
|
||||
True,
|
||||
"session_reconnect",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.CONFIGURATION_RELOAD,
|
||||
3,
|
||||
"Gracefully reload configuration without replacing the daemon.",
|
||||
"restart_coordinator class configuration_reload; console "
|
||||
"system.reload_namespace (#642).",
|
||||
("#642", "#663", "#655"),
|
||||
False,
|
||||
"configuration_reload",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.LEASE_RECOVERY,
|
||||
4,
|
||||
"Recover or rebind stale leases/locks without a process restart.",
|
||||
"Delegated: issue lock recovery / lease lifecycle paths "
|
||||
"(#702, #753, #790).",
|
||||
("#702", "#753", "#790", "#655"),
|
||||
False,
|
||||
None,
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.WORKER_RESTART,
|
||||
5,
|
||||
"Restart one worker after its own lease and mutation scope drains.",
|
||||
"restart_coordinator class worker_restart (#663).",
|
||||
("#663", "#655"),
|
||||
False,
|
||||
"worker_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||
6,
|
||||
"Restart one role runtime and re-probe that namespace only.",
|
||||
"restart_coordinator class role_runtime_restart; console "
|
||||
"system.restart_namespace (#642).",
|
||||
("#642", "#663", "#655"),
|
||||
False,
|
||||
"role_runtime_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.CONNECTOR_RESTART,
|
||||
7,
|
||||
"Restart one connector while unrelated runtimes stay available.",
|
||||
"restart_coordinator class connector_restart (#663).",
|
||||
("#663", "#655"),
|
||||
False,
|
||||
"connector_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.ROLLING_MCP_RESTART,
|
||||
8,
|
||||
"Drain/restart/verify one instance at a time (HA path).",
|
||||
"restart_coordinator class rolling_mcp_restart; design #668.",
|
||||
("#668", "#663", "#655"),
|
||||
False,
|
||||
"rolling_mcp_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.FULL_MCP_RESTART,
|
||||
9,
|
||||
"Full stable-control MCP process restart after verified full drain.",
|
||||
"restart_coordinator class full_mcp_restart; requires attempt log "
|
||||
"unless break-glass (#669).",
|
||||
("#658", "#661", "#663", "#669", "#655"),
|
||||
False,
|
||||
"full_mcp_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.HOST_RESTART,
|
||||
10,
|
||||
"Host/infrastructure restart — broadest last-resort action.",
|
||||
"restart_coordinator class host_restart; operator-owned.",
|
||||
("#663", "#669", "#655"),
|
||||
False,
|
||||
"host_restart",
|
||||
),
|
||||
)
|
||||
|
||||
_LADDER_BY_ACTION: dict[RecoveryAction, RecoveryRung] = {
|
||||
rung.action: rung for rung in RECOVERY_LADDER
|
||||
}
|
||||
|
||||
# Symptom tokens → preferred first rung (decision tree, #663 lineage).
|
||||
SYMPTOM_TO_FIRST_ACTION: dict[str, RecoveryAction] = {
|
||||
"transport_eof": RecoveryAction.CLIENT_RECONNECT,
|
||||
"client_closing_eof": RecoveryAction.CLIENT_RECONNECT,
|
||||
"transport_flap": RecoveryAction.CLIENT_RECONNECT,
|
||||
"namespace_disconnected": RecoveryAction.CLIENT_RECONNECT,
|
||||
"stale_capability": RecoveryAction.CAPABILITY_REFRESH,
|
||||
"permission_stale": RecoveryAction.CAPABILITY_REFRESH,
|
||||
"runtime_reconnect_required": RecoveryAction.CAPABILITY_REFRESH,
|
||||
"stale_runtime": RecoveryAction.SESSION_RECONNECT,
|
||||
"worktree_unbound": RecoveryAction.SESSION_RECONNECT,
|
||||
"namespace_unhealthy": RecoveryAction.SESSION_RECONNECT,
|
||||
"config_drift": RecoveryAction.CONFIGURATION_RELOAD,
|
||||
"profile_misbound": RecoveryAction.CONFIGURATION_RELOAD,
|
||||
"stale_lease": RecoveryAction.LEASE_RECOVERY,
|
||||
"dead_pid_lock": RecoveryAction.LEASE_RECOVERY,
|
||||
"orphan_worktree": RecoveryAction.LEASE_RECOVERY,
|
||||
"single_worker_stuck": RecoveryAction.WORKER_RESTART,
|
||||
"role_runtime_dead": RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||
"connector_dead": RecoveryAction.CONNECTOR_RESTART,
|
||||
"ha_instance_unhealthy": RecoveryAction.ROLLING_MCP_RESTART,
|
||||
"daemon_corrupt": RecoveryAction.FULL_MCP_RESTART,
|
||||
"full_process_deadlock": RecoveryAction.FULL_MCP_RESTART,
|
||||
"host_unresponsive": RecoveryAction.HOST_RESTART,
|
||||
}
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def resolve_action(value: RecoveryAction | str) -> RecoveryAction:
|
||||
"""Resolve a recovery action or fail closed for unknown values."""
|
||||
if isinstance(value, RecoveryAction):
|
||||
return value
|
||||
text = str(value or "").strip()
|
||||
# Accept #663 restart_class aliases.
|
||||
if text in RESTART_CLASS_TO_ACTION:
|
||||
return RESTART_CLASS_TO_ACTION[text]
|
||||
try:
|
||||
return RecoveryAction(text)
|
||||
except ValueError as exc:
|
||||
raise ValueError(
|
||||
f"unknown recovery action {value!r}; deny (fail closed, #669)"
|
||||
) from exc
|
||||
|
||||
|
||||
def ladder_rank(action: RecoveryAction | str) -> int:
|
||||
resolved = resolve_action(action)
|
||||
return _LADDER_BY_ACTION[resolved].rank
|
||||
|
||||
|
||||
def rung_for(action: RecoveryAction | str) -> RecoveryRung:
|
||||
return _LADDER_BY_ACTION[resolve_action(action)]
|
||||
|
||||
|
||||
def normalize_attempt(raw: Mapping[str, Any]) -> dict[str, Any] | None:
|
||||
"""Normalize one prior-recovery attempt record; return None if unusable."""
|
||||
if not isinstance(raw, Mapping):
|
||||
return None
|
||||
action_raw = raw.get("action") or raw.get("recovery_action") or raw.get(
|
||||
"restart_class"
|
||||
)
|
||||
if not action_raw:
|
||||
return None
|
||||
try:
|
||||
action = resolve_action(str(action_raw))
|
||||
except ValueError:
|
||||
return None
|
||||
outcome = str(
|
||||
raw.get("outcome") or raw.get("status") or raw.get("result") or ""
|
||||
).strip().lower()
|
||||
if not outcome:
|
||||
return None
|
||||
recorded_at = raw.get("recorded_at") or raw.get("at") or raw.get("timestamp")
|
||||
reason = str(raw.get("reason") or raw.get("detail") or "").strip()
|
||||
actor = str(raw.get("actor") or raw.get("session_id") or "").strip()
|
||||
return {
|
||||
"action": action.value,
|
||||
"outcome": outcome,
|
||||
"reason": reason,
|
||||
"actor": actor,
|
||||
"recorded_at": recorded_at,
|
||||
"rank": ladder_rank(action),
|
||||
"raw": dict(raw),
|
||||
}
|
||||
|
||||
|
||||
def normalize_attempt_log(
|
||||
attempts: Sequence[Mapping[str, Any]] | None,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Return usable attempt records in ladder order."""
|
||||
out: list[dict[str, Any]] = []
|
||||
for raw in attempts or ():
|
||||
norm = normalize_attempt(raw)
|
||||
if norm is not None:
|
||||
out.append(norm)
|
||||
out.sort(key=lambda a: (a["rank"], str(a.get("recorded_at") or "")))
|
||||
return out
|
||||
|
||||
|
||||
def narrower_insufficient_attempts(
|
||||
attempts: Sequence[Mapping[str, Any]] | None,
|
||||
*,
|
||||
requested: RecoveryAction | str,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Return prior attempts narrower than *requested* that were insufficient."""
|
||||
target_rank = ladder_rank(requested)
|
||||
usable = []
|
||||
for attempt in normalize_attempt_log(attempts):
|
||||
if attempt["rank"] >= target_rank:
|
||||
continue
|
||||
if attempt["outcome"] in INSUFFICIENT_OUTCOMES:
|
||||
usable.append(attempt)
|
||||
return usable
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EscalationAssessment:
|
||||
"""Whether a requested broad recovery may proceed given the attempt log."""
|
||||
|
||||
requested_action: str
|
||||
allowed: bool
|
||||
require_attempt_log: bool
|
||||
break_glass: bool
|
||||
reasons: list[str] = field(default_factory=list)
|
||||
qualifying_attempts: list[dict[str, Any]] = field(default_factory=list)
|
||||
recommended_next: list[dict[str, Any]] = field(default_factory=list)
|
||||
playbook_version: str = PLAYBOOK_VERSION
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"playbook_version": self.playbook_version,
|
||||
"requested_action": self.requested_action,
|
||||
"allowed": self.allowed,
|
||||
"require_attempt_log": self.require_attempt_log,
|
||||
"break_glass": self.break_glass,
|
||||
"reasons": list(self.reasons),
|
||||
"qualifying_attempts": list(self.qualifying_attempts),
|
||||
"recommended_next": list(self.recommended_next),
|
||||
}
|
||||
|
||||
|
||||
def assess_escalation(
|
||||
requested: RecoveryAction | str,
|
||||
*,
|
||||
prior_recovery_attempts: Sequence[Mapping[str, Any]] | None = None,
|
||||
break_glass: bool = False,
|
||||
) -> EscalationAssessment:
|
||||
"""Gate broad restarts on a prior narrow-attempt log (#669 AC3).
|
||||
|
||||
Narrow / mid-ladder actions do not require a prior attempt log.
|
||||
``full_mcp_restart``, ``host_restart``, and ``rolling_mcp_restart``
|
||||
require at least one *insufficient* narrower attempt unless
|
||||
``break_glass`` is true.
|
||||
"""
|
||||
action = resolve_action(requested)
|
||||
require_log = action in BROAD_RESTART_ACTIONS
|
||||
reasons: list[str] = []
|
||||
qualifying = narrower_insufficient_attempts(
|
||||
prior_recovery_attempts, requested=action
|
||||
)
|
||||
|
||||
if not require_log:
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=True,
|
||||
require_attempt_log=False,
|
||||
break_glass=bool(break_glass),
|
||||
reasons=["narrow recovery; attempt log not required"],
|
||||
qualifying_attempts=qualifying,
|
||||
recommended_next=[],
|
||||
)
|
||||
|
||||
if break_glass:
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=True,
|
||||
require_attempt_log=True,
|
||||
break_glass=True,
|
||||
reasons=[
|
||||
"break-glass authorized; broad restart permitted without "
|
||||
"narrow-attempt log (#669)"
|
||||
],
|
||||
qualifying_attempts=qualifying,
|
||||
recommended_next=[],
|
||||
)
|
||||
|
||||
if qualifying:
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=True,
|
||||
require_attempt_log=True,
|
||||
break_glass=False,
|
||||
reasons=[
|
||||
f"{len(qualifying)} narrower recovery attempt(s) recorded as "
|
||||
"insufficient; escalation permitted"
|
||||
],
|
||||
qualifying_attempts=qualifying,
|
||||
recommended_next=[],
|
||||
)
|
||||
|
||||
# Deny: recommend the next untried narrow rung(s).
|
||||
recommended = recommend_actions(
|
||||
symptoms=(),
|
||||
prior_recovery_attempts=prior_recovery_attempts,
|
||||
max_actions=3,
|
||||
)
|
||||
reasons.append(
|
||||
f"{action.value} requires a prior attempt log of insufficient "
|
||||
"narrower recoveries (or break-glass); none found — deny (fail "
|
||||
"closed, #669)"
|
||||
)
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=False,
|
||||
require_attempt_log=True,
|
||||
break_glass=False,
|
||||
reasons=reasons,
|
||||
qualifying_attempts=[],
|
||||
recommended_next=recommended.get("recommended_actions") or [],
|
||||
)
|
||||
|
||||
|
||||
def recommend_actions(
|
||||
*,
|
||||
symptoms: Sequence[str] = (),
|
||||
prior_recovery_attempts: Sequence[Mapping[str, Any]] | None = None,
|
||||
max_actions: int = 5,
|
||||
) -> dict[str, Any]:
|
||||
"""Return ordered recommended recovery actions for the given symptoms.
|
||||
|
||||
Soft mode (rollout): recommendations only — callers decide whether to
|
||||
hard-gate. Hard mode for broad restarts is :func:`assess_escalation`.
|
||||
"""
|
||||
attempted_success = {
|
||||
a["action"]
|
||||
for a in normalize_attempt_log(prior_recovery_attempts)
|
||||
if a["outcome"] == OUTCOME_SUCCESS
|
||||
}
|
||||
attempted_any = {
|
||||
a["action"] for a in normalize_attempt_log(prior_recovery_attempts)
|
||||
}
|
||||
|
||||
first_actions: list[RecoveryAction] = []
|
||||
for symptom in symptoms:
|
||||
key = str(symptom or "").strip().lower().replace(" ", "_").replace("-", "_")
|
||||
mapped = SYMPTOM_TO_FIRST_ACTION.get(key)
|
||||
if mapped is not None and mapped not in first_actions:
|
||||
first_actions.append(mapped)
|
||||
|
||||
# Default entry: client reconnect then walk the ladder.
|
||||
if not first_actions:
|
||||
first_actions = [RecoveryAction.CLIENT_RECONNECT]
|
||||
|
||||
recommended: list[dict[str, Any]] = []
|
||||
seen: set[str] = set()
|
||||
min_rank = min(ladder_rank(a) for a in first_actions)
|
||||
|
||||
for rung in RECOVERY_LADDER:
|
||||
if rung.rank < min_rank:
|
||||
continue
|
||||
if rung.action.value in attempted_success:
|
||||
continue
|
||||
if rung.action.value in seen:
|
||||
continue
|
||||
# Prefer rungs not yet attempted; still list previously-failed ones
|
||||
# only if nothing else remains.
|
||||
entry = rung.as_dict()
|
||||
entry["already_attempted"] = rung.action.value in attempted_any
|
||||
recommended.append(entry)
|
||||
seen.add(rung.action.value)
|
||||
if len(recommended) >= max(1, int(max_actions)):
|
||||
break
|
||||
|
||||
return {
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
"symptoms": [str(s) for s in symptoms],
|
||||
"recommended_actions": recommended,
|
||||
"ladder": [r.as_dict() for r in RECOVERY_LADDER],
|
||||
"read_only": True,
|
||||
"hard_gate_note": (
|
||||
"Broad restarts (rolling/full/host) still require "
|
||||
"assess_escalation / coordinator attempt-log enforcement."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def build_attempt_record(
|
||||
action: RecoveryAction | str,
|
||||
*,
|
||||
outcome: str,
|
||||
reason: str = "",
|
||||
actor: str = "",
|
||||
recorded_at: str | None = None,
|
||||
extra: Mapping[str, Any] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build a durable-shaped attempt log entry for inventory/audit (#665)."""
|
||||
resolved = resolve_action(action)
|
||||
record = {
|
||||
"action": resolved.value,
|
||||
"outcome": str(outcome or "").strip().lower(),
|
||||
"reason": str(reason or "").strip(),
|
||||
"actor": str(actor or "").strip(),
|
||||
"recorded_at": recorded_at or _utc_now().isoformat(),
|
||||
"rank": ladder_rank(resolved),
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
}
|
||||
if extra:
|
||||
record["extra"] = dict(extra)
|
||||
return record
|
||||
|
||||
|
||||
def recovery_metrics(
|
||||
attempts: Sequence[Mapping[str, Any]] | None,
|
||||
) -> dict[str, Any]:
|
||||
"""Compute the fraction of recoveries that avoided full/host restart.
|
||||
|
||||
A recovery *episode* is approximated as one attempt with
|
||||
``outcome=success``. Successes on non-broad rungs count as avoided full
|
||||
restart; successes on full/host count as full-restart recoveries.
|
||||
"""
|
||||
norms = normalize_attempt_log(attempts)
|
||||
successes = [a for a in norms if a["outcome"] == OUTCOME_SUCCESS]
|
||||
broad_success = [
|
||||
a
|
||||
for a in successes
|
||||
if resolve_action(a["action"])
|
||||
in {RecoveryAction.FULL_MCP_RESTART, RecoveryAction.HOST_RESTART}
|
||||
]
|
||||
avoided = [a for a in successes if a not in broad_success]
|
||||
total = len(successes)
|
||||
fraction_avoided = (len(avoided) / total) if total else None
|
||||
return {
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
"attempts_total": len(norms),
|
||||
"successes_total": total,
|
||||
"successes_avoided_full_restart": len(avoided),
|
||||
"successes_full_or_host_restart": len(broad_success),
|
||||
"fraction_avoided_full_restart": fraction_avoided,
|
||||
"insufficient_attempts": sum(
|
||||
1 for a in norms if a["outcome"] in INSUFFICIENT_OUTCOMES
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def ladder_document() -> dict[str, Any]:
|
||||
"""Machine-readable ladder for docs/tools inventory."""
|
||||
return {
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
"parent_issues": ["#655", "#652", "#653"],
|
||||
"enforcement_issue": "#669",
|
||||
"ladder": [r.as_dict() for r in RECOVERY_LADDER],
|
||||
"broad_restart_actions": [a.value for a in sorted(BROAD_RESTART_ACTIONS, key=lambda x: x.value)],
|
||||
"insufficient_outcomes": sorted(INSUFFICIENT_OUTCOMES),
|
||||
"symptom_map": {k: v.value for k, v in sorted(SYMPTOM_TO_FIRST_ACTION.items())},
|
||||
}
|
||||
+46
-2
@@ -1,4 +1,4 @@
|
||||
"""MCP restart coordinator and impact analysis (#658).
|
||||
"""MCP restart coordinator and impact analysis (#658 / #669).
|
||||
|
||||
Before any sanctioned MCP restart, a central coordinator must evaluate the
|
||||
live control-plane state — active sessions, leases/locks, in-flight issue/PR
|
||||
@@ -16,6 +16,9 @@ Design rules (mirrors the read-only posture of ``workflow_dashboard`` /
|
||||
a mutative apply path is a later child gated by a drain proof (non-goal here).
|
||||
* **Fail closed.** If the inventory is not explicitly complete, the verdict is
|
||||
``unsafe`` / deny — an incomplete evaluation must never green-light a restart.
|
||||
* **Narrow-first (#669).** Broad classes (rolling / full / host) require a
|
||||
prior attempt log of insufficient narrower recoveries unless break-glass is
|
||||
authorized. See :mod:`recovery_playbook`.
|
||||
* **No secrets.** Session ids, pids, and profiles are operational metadata, not
|
||||
credentials; nothing secret flows through this module.
|
||||
|
||||
@@ -32,8 +35,9 @@ from enum import Enum
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
import lease_lifecycle
|
||||
import recovery_playbook
|
||||
|
||||
COORDINATOR_VERSION = "1.1.0-issue-663"
|
||||
COORDINATOR_VERSION = "1.2.0-issue-669"
|
||||
|
||||
# Restart verdicts. Exactly the three the acceptance criteria name.
|
||||
VERDICT_SAFE = "safe"
|
||||
@@ -349,6 +353,10 @@ class RestartImpactReport:
|
||||
counts: dict[str, int]
|
||||
audit_record: dict[str, Any]
|
||||
incomplete_reasons: list[str] = field(default_factory=list)
|
||||
# #669 playbook escalation gate (attempt-log enforcement).
|
||||
playbook_escalation: dict[str, Any] = field(default_factory=dict)
|
||||
attempt_log_satisfied: bool = True
|
||||
break_glass: bool = False
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
@@ -382,6 +390,9 @@ class RestartImpactReport:
|
||||
"prior_recovery_attempts": list(self.prior_recovery_attempts),
|
||||
"counts": dict(self.counts),
|
||||
"audit_record": dict(self.audit_record),
|
||||
"playbook_escalation": dict(self.playbook_escalation),
|
||||
"attempt_log_satisfied": self.attempt_log_satisfied,
|
||||
"break_glass": self.break_glass,
|
||||
}
|
||||
|
||||
|
||||
@@ -496,6 +507,7 @@ def evaluate_restart_impact(
|
||||
target_session_id: str | None = None,
|
||||
target_role: str | None = None,
|
||||
target_connector: str | None = None,
|
||||
break_glass: bool = False,
|
||||
) -> RestartImpactReport:
|
||||
"""Evaluate a proposed MCP restart and return an impact preview.
|
||||
|
||||
@@ -584,6 +596,30 @@ def evaluate_restart_impact(
|
||||
dict(a) for a in (inventory.get("prior_recovery_attempts") or [])
|
||||
]
|
||||
|
||||
# #669: broad restarts require a prior narrow-attempt log unless break-glass.
|
||||
playbook_escalation: dict[str, Any] = {}
|
||||
attempt_log_satisfied = True
|
||||
if policy_enforced and resolved_class is not None:
|
||||
try:
|
||||
escalation = recovery_playbook.assess_escalation(
|
||||
resolved_class.value,
|
||||
prior_recovery_attempts=prior_recovery_attempts,
|
||||
break_glass=bool(break_glass),
|
||||
)
|
||||
playbook_escalation = escalation.as_dict()
|
||||
attempt_log_satisfied = bool(escalation.allowed)
|
||||
if not attempt_log_satisfied:
|
||||
authorization_reasons.extend(list(escalation.reasons))
|
||||
except ValueError as exc:
|
||||
# Unknown mapping should never happen for enum values; fail closed.
|
||||
attempt_log_satisfied = False
|
||||
playbook_escalation = {
|
||||
"allowed": False,
|
||||
"reasons": [str(exc)],
|
||||
"playbook_version": recovery_playbook.PLAYBOOK_VERSION,
|
||||
}
|
||||
authorization_reasons.append(str(exc))
|
||||
|
||||
session_impacts = [
|
||||
_classify_session(
|
||||
s,
|
||||
@@ -682,6 +718,7 @@ def evaluate_restart_impact(
|
||||
and role_authorized
|
||||
and approval_satisfied
|
||||
and target_complete
|
||||
and attempt_log_satisfied
|
||||
)
|
||||
|
||||
if policy_enforced and not authorization_ok:
|
||||
@@ -745,6 +782,7 @@ def evaluate_restart_impact(
|
||||
"affected_issues": len(affected_issues),
|
||||
"affected_prs": len(affected_prs),
|
||||
"prior_recovery_attempts": len(prior_recovery_attempts),
|
||||
"attempt_log_satisfied": attempt_log_satisfied,
|
||||
}
|
||||
|
||||
audit_record = {
|
||||
@@ -765,6 +803,9 @@ def evaluate_restart_impact(
|
||||
"allow_restart": allow_restart,
|
||||
"blast_radius": blast_radius,
|
||||
"counts": counts,
|
||||
"attempt_log_satisfied": attempt_log_satisfied,
|
||||
"break_glass": bool(break_glass),
|
||||
"playbook_version": recovery_playbook.PLAYBOOK_VERSION,
|
||||
}
|
||||
|
||||
return RestartImpactReport(
|
||||
@@ -804,4 +845,7 @@ def evaluate_restart_impact(
|
||||
counts=counts,
|
||||
audit_record=audit_record,
|
||||
incomplete_reasons=incomplete_reasons,
|
||||
playbook_escalation=playbook_escalation,
|
||||
attempt_log_satisfied=attempt_log_satisfied,
|
||||
break_glass=bool(break_glass),
|
||||
)
|
||||
|
||||
@@ -63,6 +63,11 @@ CONTAMINATION_GATED_TASKS = frozenset({
|
||||
"merge_pr",
|
||||
"delete_branch",
|
||||
"complete_issue",
|
||||
# Web console recovery playbooks that write (#644). These mutate runtime
|
||||
# binding and process state, so a live contamination marker must block them
|
||||
# exactly as it blocks the Gitea-side mutations above. The reconciler
|
||||
# cleanup playbook is the designated remedy and is exempted by its caller.
|
||||
"console_recovery_apply",
|
||||
})
|
||||
|
||||
CONTAMINATION_KIND = "stable_branch_push"
|
||||
|
||||
@@ -142,6 +142,23 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
# #644: Phase 2 Web Console recovery tasks.
|
||||
"clear_stale_binding": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
"rebind_session_worktree": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
# The console playbook orchestrates gitea_reconcile_merged_cleanups, whose
|
||||
# own gate is gitea.read (matching the existing reconcile_merged_cleanups
|
||||
# entry). Declaring a stricter permission here stated a second, conflicting
|
||||
# authority for one operation.
|
||||
"reconcile_cleanups": {
|
||||
"permission": "gitea.read",
|
||||
"role": "reconciler",
|
||||
},
|
||||
# PR synchronization lifecycle: assess is read-only (any role with gitea.read);
|
||||
# update-by-merge is author-only and mutates the PR head via Gitea API.
|
||||
"assess_pr_sync_status": {
|
||||
|
||||
@@ -7,6 +7,7 @@ import tempfile
|
||||
import threading
|
||||
import unittest
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from allocator_service import (
|
||||
OUTCOME_ASSIGNED,
|
||||
@@ -15,6 +16,7 @@ from allocator_service import (
|
||||
OUTCOME_PREVIEW,
|
||||
OUTCOME_WAIT,
|
||||
WorkCandidate,
|
||||
_drop_expired_claims,
|
||||
allocate_next_work,
|
||||
candidate_from_dict,
|
||||
classify_skip,
|
||||
@@ -362,5 +364,161 @@ class AllocatorServiceTest(unittest.TestCase):
|
||||
self.assertIn("unavailable", res["reasons"][0].lower())
|
||||
|
||||
|
||||
class SideEffectFreeAllocationTest(unittest.TestCase):
|
||||
"""``side_effect_free`` dry runs write nothing to the control plane (#643).
|
||||
|
||||
A plain ``apply=False`` still called ``upsert_session`` and
|
||||
``expire_stale_leases`` before the apply branch was consulted, so a caller
|
||||
advertising a read-only preview mutated on every call — one unreferenced
|
||||
session row per preview, plus a global lease sweep.
|
||||
"""
|
||||
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.db = ControlPlaneDB(os.path.join(self._tmp.name, "cp.sqlite3"))
|
||||
|
||||
def tearDown(self) -> None:
|
||||
self._tmp.cleanup()
|
||||
|
||||
def _alloc(self, **kwargs):
|
||||
defaults = dict(
|
||||
db=self.db,
|
||||
session_id="s-preview",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
candidates=[
|
||||
WorkCandidate(kind="issue", number=643, labels=("status:ready",))
|
||||
],
|
||||
apply=False,
|
||||
profile_name="prgs-author",
|
||||
username="jcwalker3",
|
||||
)
|
||||
defaults.update(kwargs)
|
||||
return allocate_next_work(**defaults)
|
||||
|
||||
def _session_ids(self) -> set[str]:
|
||||
return {str(r.get("session_id")) for r in self.db.list_sessions()}
|
||||
|
||||
def test_side_effect_free_preview_writes_no_session_row(self):
|
||||
before = self._session_ids()
|
||||
result = self._alloc(side_effect_free=True)
|
||||
self.assertEqual(result["outcome"], OUTCOME_PREVIEW)
|
||||
self.assertEqual(self._session_ids(), before)
|
||||
self.assertNotIn("s-preview", self._session_ids())
|
||||
|
||||
def test_plain_dry_run_still_registers_a_session(self):
|
||||
# The default is unchanged for every existing caller.
|
||||
self._alloc()
|
||||
self.assertIn("s-preview", self._session_ids())
|
||||
|
||||
def test_repeated_previews_do_not_accumulate_rows(self):
|
||||
for index in range(5):
|
||||
self._alloc(side_effect_free=True, session_id=f"s-{index}")
|
||||
self.assertEqual(self._session_ids(), set())
|
||||
|
||||
def test_side_effect_free_does_not_sweep_stale_leases(self):
|
||||
self.db.upsert_session(session_id="owner", role="author", pid=1)
|
||||
assigned = self.db.assign_and_lease(
|
||||
session_id="owner",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
kind="issue",
|
||||
number=999,
|
||||
lease_ttl_seconds=-60, # already expired
|
||||
)
|
||||
self.assertEqual(assigned.outcome, "assigned")
|
||||
|
||||
self._alloc(side_effect_free=True)
|
||||
|
||||
# The expired row is still 'active' in the DB: nothing swept it.
|
||||
statuses = {
|
||||
r["lease_id"]: r["status"]
|
||||
for r in self.db.list_leases(
|
||||
remote="prgs", org="org", repo="repo",
|
||||
statuses=("active", "expired"),
|
||||
)
|
||||
}
|
||||
self.assertEqual(statuses.get(assigned.lease_id), "active")
|
||||
|
||||
def test_expired_claims_are_filtered_in_memory_so_work_stays_selectable(self):
|
||||
"""The read-only mirror of the sweep: expired claims must not block."""
|
||||
self.db.upsert_session(session_id="owner", role="author", pid=1)
|
||||
self.db.assign_and_lease(
|
||||
session_id="owner",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
kind="issue",
|
||||
number=643,
|
||||
lease_ttl_seconds=-60, # expired: must not withhold #643
|
||||
)
|
||||
result = self._alloc(side_effect_free=True)
|
||||
self.assertEqual(result["outcome"], OUTCOME_PREVIEW)
|
||||
self.assertEqual(result["selected"]["number"], 643)
|
||||
|
||||
def test_a_live_claim_still_withholds_the_work(self):
|
||||
self.db.upsert_session(session_id="owner", role="author", pid=1)
|
||||
self.db.assign_and_lease(
|
||||
session_id="owner",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
kind="issue",
|
||||
number=643,
|
||||
lease_ttl_seconds=3600,
|
||||
)
|
||||
result = self._alloc(side_effect_free=True)
|
||||
self.assertNotEqual(result["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertNotEqual((result.get("selected") or {}).get("number"), 643)
|
||||
|
||||
def test_side_effect_free_with_apply_fails_closed(self):
|
||||
result = self._alloc(side_effect_free=True, apply=True)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["outcome"], OUTCOME_NO_SAFE)
|
||||
self.assertIsNone(result["assignment"])
|
||||
self.assertIn("incompatible with apply", result["reasons"][0])
|
||||
# And it reserved nothing.
|
||||
self.assertEqual(
|
||||
self.db.list_leases(remote="prgs", org="org", repo="repo"), []
|
||||
)
|
||||
|
||||
|
||||
class DropExpiredClaimsTest(unittest.TestCase):
|
||||
"""The in-memory expiry filter behind side-effect-free previews (#643)."""
|
||||
|
||||
def test_unparseable_expiry_is_kept_rather_than_assumed_free(self):
|
||||
claims = {
|
||||
("issue", 1): {"lease_id": "l1", "expires_at": "not-a-date"},
|
||||
("issue", 2): {"lease_id": "l2"},
|
||||
("issue", 3): {"lease_id": "l3", "expires_at": None},
|
||||
}
|
||||
self.assertEqual(_drop_expired_claims(claims), claims)
|
||||
|
||||
def test_expired_dropped_and_future_kept(self):
|
||||
now = datetime(2026, 7, 25, 12, 0, tzinfo=timezone.utc)
|
||||
claims = {
|
||||
("issue", 1): {"expires_at": "2026-07-25T11:59:59+00:00"},
|
||||
("issue", 2): {"expires_at": "2026-07-25T12:00:01+00:00"},
|
||||
("issue", 3): {"expires_at": "2026-07-25T12:00:00+00:00"}, # boundary
|
||||
}
|
||||
kept = _drop_expired_claims(claims, now=now)
|
||||
self.assertEqual(set(kept), {("issue", 2)})
|
||||
|
||||
def test_naive_and_zulu_timestamps_are_treated_as_utc(self):
|
||||
now = datetime(2026, 7, 25, 12, 0, tzinfo=timezone.utc)
|
||||
claims = {
|
||||
("issue", 1): {"expires_at": "2026-07-25T11:00:00"}, # naive, past
|
||||
("issue", 2): {"expires_at": "2026-07-25T13:00:00Z"}, # zulu, future
|
||||
}
|
||||
kept = _drop_expired_claims(claims, now=now)
|
||||
self.assertEqual(set(kept), {("issue", 2)})
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -35,6 +35,17 @@ BREAK_GLASS_ENV = "GITEA_BREAKGLASS_RESTART_AUTHORIZATION"
|
||||
QUIET_SESSIONS: list[dict] = []
|
||||
QUIET_LEASES: list[dict] = []
|
||||
|
||||
# #669: broad restarts need a prior narrow-attempt log (unless break-glass).
|
||||
PRIOR_NARROW_ATTEMPTS_JSON = json.dumps(
|
||||
[
|
||||
{
|
||||
"action": "client_reconnect",
|
||||
"outcome": "insufficient",
|
||||
"reason": "still flapping after reconnect",
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
class _FakeDB:
|
||||
"""Minimal control-plane DB stand-in for the restart inventory."""
|
||||
@@ -128,6 +139,7 @@ class TestConjunction(_RestartToolHarness):
|
||||
preview = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
self.assertTrue(preview["allow_restart"],
|
||||
@@ -136,6 +148,7 @@ class TestConjunction(_RestartToolHarness):
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
@@ -342,6 +355,7 @@ class TestExistingPathsStillWork(_RestartToolHarness):
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||
dry_run=False,
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
|
||||
@@ -0,0 +1,149 @@
|
||||
"""Unit tests for mcp_config_drift.py (#672)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
|
||||
from mcp_config_drift import (
|
||||
REQUIRED_GITEA_ROLE_SERVERS,
|
||||
analyze_config_drift,
|
||||
load_mcp_config,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def sample_global_config() -> dict:
|
||||
return {
|
||||
"mcpServers": {
|
||||
"gitea-author": {
|
||||
"command": "python3",
|
||||
"args": ["gitea_mcp_server.py"],
|
||||
"env": {"GITEA_MCP_PROFILE": "prgs-author", "SENTRY_AUTH_TOKEN": "secret-token-999"},
|
||||
},
|
||||
"gitea-reviewer": {
|
||||
"command": "python3",
|
||||
"args": ["gitea_mcp_server.py"],
|
||||
"env": {"GITEA_MCP_PROFILE": "prgs-reviewer"},
|
||||
},
|
||||
"gitea-merger": {
|
||||
"command": "python3",
|
||||
"args": ["gitea_mcp_server.py"],
|
||||
"env": {"GITEA_MCP_PROFILE": "prgs-merger"},
|
||||
},
|
||||
"gitea-reconciler": {
|
||||
"command": "python3",
|
||||
"args": ["gitea_mcp_server.py"],
|
||||
"env": {"GITEA_MCP_PROFILE": "prgs-reconciler"},
|
||||
},
|
||||
"gitea-controller": {
|
||||
"command": "python3",
|
||||
"args": ["gitea_mcp_server.py"],
|
||||
"env": {"GITEA_MCP_PROFILE": "prgs-controller"},
|
||||
},
|
||||
"gitea-tools": {
|
||||
"command": "python3",
|
||||
"args": ["gitea_mcp_server.py"],
|
||||
"env": {"GITEA_MCP_PROFILE": "prgs-author"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
def write_json(path: Path, data: dict) -> str:
|
||||
path.write_text(json.dumps(data, indent=2), encoding="utf-8")
|
||||
return str(path)
|
||||
|
||||
|
||||
def test_drift_detection_in_sync(tmp_path, sample_global_config):
|
||||
glob_file = tmp_path / "global_mcp.json"
|
||||
act_file = tmp_path / "active_mcp.json"
|
||||
|
||||
write_json(glob_file, sample_global_config)
|
||||
write_json(act_file, sample_global_config)
|
||||
|
||||
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||
|
||||
assert report["in_sync"] is True
|
||||
assert report["missing_role_servers"] == []
|
||||
assert report["profile_mismatches"] == []
|
||||
assert set(report["present_role_servers"]) == set(REQUIRED_GITEA_ROLE_SERVERS)
|
||||
|
||||
|
||||
def test_drift_detection_missing_author(tmp_path, sample_global_config):
|
||||
glob_file = tmp_path / "global_mcp.json"
|
||||
act_file = tmp_path / "active_mcp.json"
|
||||
|
||||
active_config = json.loads(json.dumps(sample_global_config))
|
||||
del active_config["mcpServers"]["gitea-author"]
|
||||
|
||||
write_json(glob_file, sample_global_config)
|
||||
write_json(act_file, active_config)
|
||||
|
||||
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||
|
||||
assert report["in_sync"] is False
|
||||
assert "gitea-author" in report["missing_role_servers"]
|
||||
assert "gitea-author" not in report["present_role_servers"]
|
||||
|
||||
|
||||
def test_drift_detection_missing_reviewer(tmp_path, sample_global_config):
|
||||
glob_file = tmp_path / "global_mcp.json"
|
||||
act_file = tmp_path / "active_mcp.json"
|
||||
|
||||
active_config = json.loads(json.dumps(sample_global_config))
|
||||
del active_config["mcpServers"]["gitea-reviewer"]
|
||||
|
||||
write_json(glob_file, sample_global_config)
|
||||
write_json(act_file, active_config)
|
||||
|
||||
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||
|
||||
assert report["in_sync"] is False
|
||||
assert "gitea-reviewer" in report["missing_role_servers"]
|
||||
|
||||
|
||||
def test_drift_detection_profile_mismatch(tmp_path, sample_global_config):
|
||||
glob_file = tmp_path / "global_mcp.json"
|
||||
act_file = tmp_path / "active_mcp.json"
|
||||
|
||||
active_config = json.loads(json.dumps(sample_global_config))
|
||||
active_config["mcpServers"]["gitea-author"]["env"]["GITEA_MCP_PROFILE"] = "dadeschools-author"
|
||||
|
||||
write_json(glob_file, sample_global_config)
|
||||
write_json(act_file, active_config)
|
||||
|
||||
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||
|
||||
assert report["in_sync"] is False
|
||||
assert len(report["profile_mismatches"]) == 1
|
||||
mismatch = report["profile_mismatches"][0]
|
||||
assert mismatch["server"] == "gitea-author"
|
||||
assert mismatch["active_profile"] == "dadeschools-author"
|
||||
assert mismatch["global_profile"] == "prgs-author"
|
||||
|
||||
|
||||
def test_secret_redaction_in_drift_report(tmp_path, sample_global_config):
|
||||
glob_file = tmp_path / "global_mcp.json"
|
||||
act_file = tmp_path / "active_mcp.json"
|
||||
|
||||
write_json(glob_file, sample_global_config)
|
||||
write_json(act_file, sample_global_config)
|
||||
|
||||
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||
serialized = str(report)
|
||||
|
||||
assert "secret-token-999" not in serialized
|
||||
|
||||
|
||||
def test_sanctioned_runbook_forbids_pkill():
|
||||
report = analyze_config_drift(active_config_path="/nonexistent/path/active.json", global_config_path="/nonexistent/path/global.json")
|
||||
|
||||
runbook_text = " ".join(report["sanctioned_repair_runbook"]).lower()
|
||||
forbidden_text = " ".join(report["forbidden_repair_methods"]).lower()
|
||||
|
||||
assert "pkill" in forbidden_text
|
||||
assert "mtime" in forbidden_text
|
||||
assert "source" in forbidden_text
|
||||
assert "session-state" in forbidden_text
|
||||
@@ -0,0 +1,217 @@
|
||||
"""Unit tests for the scoped recovery playbook (#669)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import recovery_playbook as rp
|
||||
import restart_coordinator as rc
|
||||
|
||||
|
||||
def test_ladder_covers_eleven_ordered_rungs():
|
||||
ranks = [r.rank for r in rp.RECOVERY_LADDER]
|
||||
assert ranks == list(range(len(rp.RECOVERY_LADDER)))
|
||||
assert len(rp.RECOVERY_LADDER) == 11
|
||||
assert rp.RECOVERY_LADDER[0].action is rp.RecoveryAction.CLIENT_RECONNECT
|
||||
assert rp.RECOVERY_LADDER[-1].action is rp.RecoveryAction.HOST_RESTART
|
||||
|
||||
|
||||
def test_ladder_document_links_parent_issues():
|
||||
doc = rp.ladder_document()
|
||||
assert "#655" in doc["parent_issues"]
|
||||
assert "#652" in doc["parent_issues"]
|
||||
assert "#653" in doc["parent_issues"]
|
||||
assert doc["enforcement_issue"] == "#669"
|
||||
assert "full_mcp_restart" in doc["broad_restart_actions"]
|
||||
|
||||
|
||||
def test_recommend_transport_eof_starts_at_client_reconnect():
|
||||
plan = rp.recommend_actions(symptoms=["transport_eof"])
|
||||
assert plan["recommended_actions"][0]["action"] == "client_reconnect"
|
||||
assert plan["recommended_actions"][0]["issue_links"]
|
||||
|
||||
|
||||
def test_recommend_skips_successful_prior_attempts():
|
||||
attempts = [
|
||||
rp.build_attempt_record(
|
||||
"client_reconnect", outcome="success", reason="reconnected"
|
||||
)
|
||||
]
|
||||
plan = rp.recommend_actions(
|
||||
symptoms=["transport_eof"], prior_recovery_attempts=attempts
|
||||
)
|
||||
actions = [a["action"] for a in plan["recommended_actions"]]
|
||||
assert "client_reconnect" not in actions
|
||||
assert actions[0] == "capability_refresh"
|
||||
|
||||
|
||||
def test_escalation_denied_without_attempt_log():
|
||||
result = rp.assess_escalation("full_mcp_restart", prior_recovery_attempts=[])
|
||||
assert result.allowed is False
|
||||
assert result.require_attempt_log is True
|
||||
assert any("#669" in r for r in result.reasons)
|
||||
assert result.recommended_next # soft recommendations still provided
|
||||
|
||||
|
||||
def test_escalation_allowed_after_insufficient_narrower():
|
||||
attempts = [
|
||||
rp.build_attempt_record(
|
||||
"client_reconnect",
|
||||
outcome="insufficient",
|
||||
reason="still flapping",
|
||||
),
|
||||
rp.build_attempt_record(
|
||||
"session_reconnect",
|
||||
outcome="failed",
|
||||
reason="namespace still dead",
|
||||
),
|
||||
]
|
||||
result = rp.assess_escalation(
|
||||
"full_mcp_restart", prior_recovery_attempts=attempts
|
||||
)
|
||||
assert result.allowed is True
|
||||
assert len(result.qualifying_attempts) == 2
|
||||
|
||||
|
||||
def test_escalation_break_glass_bypasses_attempt_log():
|
||||
result = rp.assess_escalation(
|
||||
"host_restart", prior_recovery_attempts=[], break_glass=True
|
||||
)
|
||||
assert result.allowed is True
|
||||
assert result.break_glass is True
|
||||
|
||||
|
||||
def test_narrow_action_does_not_require_attempt_log():
|
||||
result = rp.assess_escalation(
|
||||
"client_reconnect", prior_recovery_attempts=[]
|
||||
)
|
||||
assert result.allowed is True
|
||||
assert result.require_attempt_log is False
|
||||
|
||||
|
||||
def test_same_rank_attempt_does_not_qualify_for_escalation():
|
||||
attempts = [
|
||||
rp.build_attempt_record(
|
||||
"full_mcp_restart", outcome="failed", reason="already failed full"
|
||||
)
|
||||
]
|
||||
result = rp.assess_escalation(
|
||||
"full_mcp_restart", prior_recovery_attempts=attempts
|
||||
)
|
||||
assert result.allowed is False
|
||||
|
||||
|
||||
def test_success_outcome_does_not_qualify_for_escalation():
|
||||
attempts = [
|
||||
rp.build_attempt_record(
|
||||
"client_reconnect", outcome="success", reason="fixed"
|
||||
)
|
||||
]
|
||||
result = rp.assess_escalation(
|
||||
"full_mcp_restart", prior_recovery_attempts=attempts
|
||||
)
|
||||
assert result.allowed is False
|
||||
|
||||
|
||||
def test_recovery_metrics_fraction_avoided():
|
||||
attempts = [
|
||||
rp.build_attempt_record("client_reconnect", outcome="success"),
|
||||
rp.build_attempt_record("session_reconnect", outcome="success"),
|
||||
rp.build_attempt_record("full_mcp_restart", outcome="success"),
|
||||
]
|
||||
metrics = rp.recovery_metrics(attempts)
|
||||
assert metrics["successes_total"] == 3
|
||||
assert metrics["successes_avoided_full_restart"] == 2
|
||||
assert metrics["successes_full_or_host_restart"] == 1
|
||||
assert abs(metrics["fraction_avoided_full_restart"] - (2 / 3)) < 1e-9
|
||||
|
||||
|
||||
def test_coordinator_denies_full_restart_without_attempt_log():
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"prior_recovery_attempts": [],
|
||||
}
|
||||
report = rc.evaluate_restart_impact(
|
||||
inv,
|
||||
restart_class=rc.RestartClass.FULL_MCP_RESTART,
|
||||
requester_role="controller",
|
||||
requester_permissions=rc.permissions_for_role("controller"),
|
||||
controller_approved=True,
|
||||
operator_authorized=True,
|
||||
)
|
||||
assert report.allow_restart is False
|
||||
assert report.attempt_log_satisfied is False
|
||||
assert report.verdict == rc.VERDICT_UNSAFE
|
||||
blob = " ".join(report.reasons + report.authorization_reasons)
|
||||
assert "#669" in blob or "attempt log" in blob
|
||||
|
||||
|
||||
def test_coordinator_allows_full_restart_with_attempt_log():
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"prior_recovery_attempts": [
|
||||
{
|
||||
"action": "client_reconnect",
|
||||
"outcome": "insufficient",
|
||||
"reason": "still broken",
|
||||
}
|
||||
],
|
||||
}
|
||||
report = rc.evaluate_restart_impact(
|
||||
inv,
|
||||
restart_class=rc.RestartClass.FULL_MCP_RESTART,
|
||||
requester_role="controller",
|
||||
requester_permissions=rc.permissions_for_role("controller"),
|
||||
controller_approved=True,
|
||||
operator_authorized=True,
|
||||
)
|
||||
assert report.attempt_log_satisfied is True
|
||||
assert report.allow_restart is True
|
||||
assert report.verdict == rc.VERDICT_SAFE
|
||||
|
||||
|
||||
def test_coordinator_break_glass_allows_without_log():
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"prior_recovery_attempts": [],
|
||||
}
|
||||
report = rc.evaluate_restart_impact(
|
||||
inv,
|
||||
restart_class=rc.RestartClass.FULL_MCP_RESTART,
|
||||
requester_role="controller",
|
||||
requester_permissions=rc.permissions_for_role("controller"),
|
||||
controller_approved=True,
|
||||
operator_authorized=True,
|
||||
break_glass=True,
|
||||
)
|
||||
assert report.break_glass is True
|
||||
assert report.attempt_log_satisfied is True
|
||||
assert report.allow_restart is True
|
||||
|
||||
|
||||
def test_coordinator_client_reconnect_unaffected():
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"prior_recovery_attempts": [],
|
||||
}
|
||||
report = rc.evaluate_restart_impact(
|
||||
inv,
|
||||
restart_class=rc.RestartClass.CLIENT_RECONNECT,
|
||||
requester_role="author",
|
||||
requester_permissions=rc.permissions_for_role("author"),
|
||||
)
|
||||
assert report.attempt_log_satisfied is True
|
||||
assert report.allow_restart is True
|
||||
|
||||
|
||||
def test_restart_class_alias_accepted():
|
||||
assert (
|
||||
rp.resolve_action("full_mcp_restart")
|
||||
is rp.RecoveryAction.FULL_MCP_RESTART
|
||||
)
|
||||
@@ -0,0 +1,506 @@
|
||||
"""Unit and integration tests for Phase 2 Web Console recovery controls (#644)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import types
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
import merged_cleanup_reconcile
|
||||
import runtime_recovery_guard
|
||||
import stable_branch_push_guard
|
||||
import stale_binding_recovery
|
||||
from webui import console_authz, console_recovery, system_health
|
||||
from webui.app import create_app
|
||||
|
||||
|
||||
class TestConsoleRecovery(unittest.TestCase):
|
||||
|
||||
def test_diagnose_recovery_healthy(self) -> None:
|
||||
diag = console_recovery.diagnose_recovery()
|
||||
self.assertIn(diag.status, {console_recovery.STATUS_HEALTHY, console_recovery.STATUS_ACTION_REQUIRED})
|
||||
self.assertIsInstance(diag.playbooks, tuple)
|
||||
self.assertGreaterEqual(len(diag.playbooks), 4)
|
||||
|
||||
playbook_ids = {pb.playbook_id for pb in diag.playbooks}
|
||||
self.assertIn(console_recovery.PLAYBOOK_CLEAR_STALE_BINDING, playbook_ids)
|
||||
self.assertIn(console_recovery.PLAYBOOK_REBIND_SESSION, playbook_ids)
|
||||
self.assertIn(console_recovery.PLAYBOOK_RECONCILE_CLEANUPS, playbook_ids)
|
||||
self.assertIn(console_recovery.PLAYBOOK_SANCTIONED_RESTART, playbook_ids)
|
||||
|
||||
def test_confirmation_phrase_generation_and_matching(self) -> None:
|
||||
phrase = console_recovery.confirmation_phrase("clear_stale_binding")
|
||||
self.assertEqual(phrase, "confirm clear_stale_binding")
|
||||
self.assertTrue(console_recovery.confirmation_matches("clear_stale_binding", "confirm clear_stale_binding"))
|
||||
self.assertFalse(console_recovery.confirmation_matches("clear_stale_binding", "wrong phrase"))
|
||||
|
||||
phrase_target = console_recovery.confirmation_phrase("sanctioned_restart", "gitea-author")
|
||||
self.assertEqual(phrase_target, "confirm sanctioned_restart gitea-author")
|
||||
self.assertTrue(console_recovery.confirmation_matches("sanctioned_restart", "confirm sanctioned_restart gitea-author", "gitea-author"))
|
||||
|
||||
def test_build_recovery_preview(self) -> None:
|
||||
principal = console_authz.Principal("[email protected]", console_authz.OPERATOR, console_authz.IDENTITY_LOCAL_DEV, True)
|
||||
preview = console_recovery.build_recovery_preview(
|
||||
playbook_id=console_recovery.PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
target="test-worktree",
|
||||
principal=principal,
|
||||
)
|
||||
self.assertEqual(preview["playbook_id"], console_recovery.PLAYBOOK_CLEAR_STALE_BINDING)
|
||||
self.assertEqual(preview["action_id"], console_recovery.ACTION_CLEAR_STALE_BINDING)
|
||||
self.assertEqual(preview["confirmation_phrase"], "confirm clear_stale_binding test-worktree")
|
||||
self.assertTrue(len(preview["mutation_ledger"]) >= 3)
|
||||
self.assertTrue(preview["authorization"]["allowed"])
|
||||
|
||||
def test_build_recovery_preview_unknown_playbook(self) -> None:
|
||||
preview = console_recovery.build_recovery_preview("unknown_playbook")
|
||||
self.assertFalse(preview.get("allowed"))
|
||||
self.assertEqual(preview.get("error"), "unknown_playbook")
|
||||
|
||||
def test_execute_recovery_playbook_confirmation_mismatch(self) -> None:
|
||||
# Authorization is checked before confirmation, so the phase gate has to
|
||||
# pass for this test to reach the branch it is about.
|
||||
principal = console_authz.Principal("[email protected]", console_authz.OPERATOR, console_authz.IDENTITY_LOCAL_DEV, True)
|
||||
with self._phase_two_enabled():
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
confirmation="invalid confirmation",
|
||||
principal=principal,
|
||||
)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertFalse(result["allowed"])
|
||||
self.assertEqual(result["error"], "confirmation_mismatch")
|
||||
|
||||
def test_execute_recovery_playbook_unauthorized(self) -> None:
|
||||
# Anonymous principal has viewer role -> should be denied
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
confirmation="confirm clear_stale_binding",
|
||||
principal=console_authz.ANONYMOUS,
|
||||
)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertFalse(result["allowed"])
|
||||
self.assertEqual(result["error"], console_authz.DENY_UNAUTHENTICATED)
|
||||
|
||||
def test_execute_refuses_phase_two_write_while_console_is_phase_one(self) -> None:
|
||||
"""B1: the apply path must arm the phase gate, not skip it.
|
||||
|
||||
``authorize`` only applies the phase branch when ``for_execution=True``.
|
||||
The apply path used the default, so an operator executed a phase-2 write
|
||||
while ``ACTIVE_PHASE`` was 1.
|
||||
"""
|
||||
self.assertGreater(
|
||||
console_authz.get_action(console_recovery.ACTION_CLEAR_STALE_BINDING).phase,
|
||||
console_authz.ACTIVE_PHASE,
|
||||
"fixture assumes the recovery actions are ahead of the active phase",
|
||||
)
|
||||
principal = console_authz.Principal(
|
||||
"[email protected]", console_authz.OPERATOR, console_authz.IDENTITY_LOCAL_DEV, True
|
||||
)
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_CLEAR_STALE_BINDING
|
||||
)
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
confirmation=phrase,
|
||||
principal=principal,
|
||||
)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertFalse(result["allowed"])
|
||||
self.assertEqual(result["error"], console_authz.DENY_PHASE_NOT_ACTIVE)
|
||||
|
||||
def test_preview_execution_enabled_matches_the_execution_decision(self) -> None:
|
||||
"""B1: preview must not report a bare False it cannot explain."""
|
||||
principal = console_authz.Principal(
|
||||
"[email protected]", console_authz.OPERATOR, console_authz.IDENTITY_LOCAL_DEV, True
|
||||
)
|
||||
preview = console_recovery.build_recovery_preview(
|
||||
playbook_id=console_recovery.PLAYBOOK_REBIND_SESSION,
|
||||
target="branches/feat-issue-644",
|
||||
principal=principal,
|
||||
)
|
||||
self.assertFalse(preview["execution_enabled"])
|
||||
self.assertEqual(
|
||||
preview["execution_blocked_reason"], console_authz.DENY_PHASE_NOT_ACTIVE
|
||||
)
|
||||
self.assertFalse(preview["execution_authorization"]["allowed"])
|
||||
# The preview (non-execution) decision still allows, by role.
|
||||
self.assertTrue(preview["authorization"]["allowed"])
|
||||
|
||||
def _phase_two_enabled(self):
|
||||
"""Raise ACTIVE_PHASE so the execution branches are reachable in tests."""
|
||||
return patch.object(console_authz, "ACTIVE_PHASE", 2)
|
||||
|
||||
def _operator(self) -> console_authz.Principal:
|
||||
return console_authz.Principal(
|
||||
"[email protected]", console_authz.OPERATOR, console_authz.IDENTITY_LOCAL_DEV, True
|
||||
)
|
||||
|
||||
def test_rebind_mutates_the_live_environment_not_a_copy(self) -> None:
|
||||
"""B2: the playbook must change the mapping it claims to have changed."""
|
||||
live_env = {stale_binding_recovery.ACTIVE_WORKTREE_ENV: "branches/stale-old"}
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_REBIND_SESSION, "branches/feat-issue-644"
|
||||
)
|
||||
with self._phase_two_enabled():
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_REBIND_SESSION,
|
||||
confirmation=phrase,
|
||||
target="branches/feat-issue-644",
|
||||
principal=self._operator(),
|
||||
env=live_env,
|
||||
)
|
||||
self.assertTrue(result["success"])
|
||||
self.assertEqual(
|
||||
live_env[stale_binding_recovery.ACTIVE_WORKTREE_ENV],
|
||||
"branches/feat-issue-644",
|
||||
"rebind reported success without changing the caller's environment",
|
||||
)
|
||||
self.assertTrue(result["applied_result"]["binding_changed"])
|
||||
self.assertEqual(result["applied_result"]["binding_before"], "branches/stale-old")
|
||||
self.assertEqual(
|
||||
result["applied_result"]["binding_after"], "branches/feat-issue-644"
|
||||
)
|
||||
|
||||
def test_clear_stale_binding_reports_failure_when_nothing_changed(self) -> None:
|
||||
"""B2: a no-op recovery must never be reported as success."""
|
||||
live_env: dict[str, str] = {}
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_CLEAR_STALE_BINDING
|
||||
)
|
||||
with self._phase_two_enabled():
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
confirmation=phrase,
|
||||
principal=self._operator(),
|
||||
env=live_env,
|
||||
)
|
||||
self.assertFalse(
|
||||
result["success"],
|
||||
"a clear that changed no binding must not report success",
|
||||
)
|
||||
self.assertFalse(result["applied_result"]["binding_changed"])
|
||||
|
||||
def test_clear_stale_binding_clears_the_live_binding(self) -> None:
|
||||
"""B2: the sanctioned clear must reach the caller's environment."""
|
||||
missing = "/nonexistent/branches/deleted-worktree"
|
||||
live_env = {stale_binding_recovery.ACTIVE_WORKTREE_ENV: missing}
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_CLEAR_STALE_BINDING
|
||||
)
|
||||
with self._phase_two_enabled():
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
confirmation=phrase,
|
||||
principal=self._operator(),
|
||||
env=live_env,
|
||||
)
|
||||
if result["success"]:
|
||||
self.assertNotIn(stale_binding_recovery.ACTIVE_WORKTREE_ENV, live_env)
|
||||
self.assertEqual(result["applied_result"]["binding_before"], missing)
|
||||
self.assertIsNone(result["applied_result"]["binding_after"])
|
||||
else:
|
||||
# Fail closed is acceptable; reporting a clear that did not happen
|
||||
# is not. This is the invariant the blocker was about.
|
||||
self.assertFalse(result["applied_result"]["binding_changed"])
|
||||
self.assertEqual(
|
||||
live_env.get(stale_binding_recovery.ACTIVE_WORKTREE_ENV), missing
|
||||
)
|
||||
|
||||
def test_reconcile_playbook_calls_an_entry_point_that_exists(self) -> None:
|
||||
"""B3: the previous call named a function absent from the module."""
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_RECONCILE_CLEANUPS
|
||||
)
|
||||
fake_server = types.SimpleNamespace(
|
||||
gitea_reconcile_merged_cleanups=lambda **kwargs: {
|
||||
"success": True,
|
||||
"entries": [{"issue_number": 100}],
|
||||
}
|
||||
)
|
||||
with self._phase_two_enabled(), patch.dict(
|
||||
sys.modules, {"gitea_mcp_server": fake_server}
|
||||
):
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_RECONCILE_CLEANUPS,
|
||||
confirmation=phrase,
|
||||
principal=console_authz.Principal(
|
||||
"[email protected]",
|
||||
console_authz.ADMIN,
|
||||
console_authz.IDENTITY_LOCAL_DEV,
|
||||
True,
|
||||
),
|
||||
)
|
||||
self.assertTrue(result["success"], result.get("applied_result"))
|
||||
self.assertNotIn("error_type", result["applied_result"])
|
||||
self.assertEqual(result["applied_result"]["reconciled_count"], 1)
|
||||
|
||||
def test_reconcile_entry_point_exists_on_the_real_module(self) -> None:
|
||||
"""B3 regression: guard the symbol itself, not just the call shape."""
|
||||
import gitea_mcp_server
|
||||
|
||||
self.assertTrue(
|
||||
hasattr(gitea_mcp_server, "gitea_reconcile_merged_cleanups"),
|
||||
"console recovery depends on this reconciler entry point",
|
||||
)
|
||||
self.assertFalse(
|
||||
hasattr(merged_cleanup_reconcile, "reconcile_merged_cleanups"),
|
||||
"if this module grows the orchestrator, point the playbook back at it",
|
||||
)
|
||||
|
||||
def test_contamination_gate_blocks_a_writing_playbook(self) -> None:
|
||||
"""B4: a live marker plus a gated task key must actually block."""
|
||||
marker = {
|
||||
"kind": "manual_daemon_kill",
|
||||
"reason_class": "manual_daemon_kill",
|
||||
"command_summary": "pkill -f gitea_mcp_server",
|
||||
"active": True,
|
||||
}
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_REBIND_SESSION, "branches/feat-issue-644"
|
||||
)
|
||||
live_env = {stale_binding_recovery.ACTIVE_WORKTREE_ENV: "branches/stale-old"}
|
||||
with self._phase_two_enabled(), patch.object(
|
||||
console_recovery, "load_active_contamination_marker", return_value=marker
|
||||
):
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_REBIND_SESSION,
|
||||
confirmation=phrase,
|
||||
target="branches/feat-issue-644",
|
||||
principal=self._operator(),
|
||||
env=live_env,
|
||||
)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["error"], "contaminated_runtime")
|
||||
self.assertEqual(
|
||||
live_env[stale_binding_recovery.ACTIVE_WORKTREE_ENV],
|
||||
"branches/stale-old",
|
||||
"a blocked playbook must not have mutated anything",
|
||||
)
|
||||
|
||||
def test_contamination_gate_exempts_the_reconciler_remedy(self) -> None:
|
||||
"""B4: the designated remedy must stay reachable while contaminated."""
|
||||
marker = {
|
||||
"kind": "manual_daemon_kill",
|
||||
"reason_class": "manual_daemon_kill",
|
||||
"command_summary": "pkill -f gitea_mcp_server",
|
||||
"active": True,
|
||||
}
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_RECONCILE_CLEANUPS
|
||||
)
|
||||
fake_server = types.SimpleNamespace(
|
||||
gitea_reconcile_merged_cleanups=lambda **kwargs: {
|
||||
"success": True,
|
||||
"entries": [],
|
||||
}
|
||||
)
|
||||
with self._phase_two_enabled(), patch.object(
|
||||
console_recovery, "load_active_contamination_marker", return_value=marker
|
||||
), patch.dict(sys.modules, {"gitea_mcp_server": fake_server}):
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_RECONCILE_CLEANUPS,
|
||||
confirmation=phrase,
|
||||
principal=console_authz.Principal(
|
||||
"[email protected]",
|
||||
console_authz.ADMIN,
|
||||
console_authz.IDENTITY_LOCAL_DEV,
|
||||
True,
|
||||
),
|
||||
)
|
||||
self.assertNotEqual(result.get("error"), "contaminated_runtime")
|
||||
|
||||
def test_gated_task_key_is_actually_gated(self) -> None:
|
||||
"""B4: the console action id was never a member of the gated set."""
|
||||
self.assertIn(
|
||||
console_recovery.CONTAMINATION_GATED_TASK,
|
||||
stable_branch_push_guard.CONTAMINATION_GATED_TASKS,
|
||||
)
|
||||
self.assertNotIn(
|
||||
console_recovery.ACTION_CLEAR_STALE_BINDING,
|
||||
stable_branch_push_guard.CONTAMINATION_GATED_TASKS,
|
||||
)
|
||||
|
||||
def test_diagnosis_reads_the_key_the_gate_returns(self) -> None:
|
||||
"""B4: ``contaminated`` is a key assess_contamination_gate never returns."""
|
||||
gate = runtime_recovery_guard.assess_contamination_gate(
|
||||
None, task=console_recovery.CONTAMINATION_GATED_TASK, actual_role="operator"
|
||||
)
|
||||
self.assertNotIn("contaminated", gate)
|
||||
self.assertIn("block", gate)
|
||||
|
||||
def test_contaminated_runtime_is_reported_unclean(self) -> None:
|
||||
"""B4: verify_post_recovery reported contamination_clean unconditionally."""
|
||||
marker = {
|
||||
"kind": "manual_daemon_kill",
|
||||
"reason_class": "manual_daemon_kill",
|
||||
"command_summary": "pkill -f gitea_mcp_server",
|
||||
"active": True,
|
||||
}
|
||||
with patch.object(
|
||||
console_recovery, "load_active_contamination_marker", return_value=marker
|
||||
):
|
||||
verification = console_recovery.verify_post_recovery()
|
||||
diag = console_recovery.diagnose_recovery()
|
||||
self.assertFalse(verification["contamination_clean"])
|
||||
self.assertFalse(verification["clean"])
|
||||
self.assertEqual(diag.status, console_recovery.STATUS_BLOCKED_CONTAMINATION)
|
||||
|
||||
def test_master_parity_baseline_is_not_the_head_it_is_compared_against(self) -> None:
|
||||
"""B5: capture_startup_parity was fed the head it was then compared to."""
|
||||
stale = system_health.StaleRuntime(
|
||||
daemon_head="a" * 40,
|
||||
checkout_head="b" * 40,
|
||||
remote_head="b" * 40,
|
||||
stale=True,
|
||||
determinable=True,
|
||||
mutation_safe=False,
|
||||
reasons=("daemon is behind the checkout",),
|
||||
)
|
||||
with patch.object(system_health, "assess_stale_runtime", return_value=stale):
|
||||
diag = console_recovery.diagnose_recovery()
|
||||
parity = diag.master_parity
|
||||
self.assertEqual(parity["startup_head"], "a" * 40)
|
||||
self.assertEqual(parity["current_head"], "b" * 40)
|
||||
self.assertNotEqual(parity["startup_head"], parity["current_head"])
|
||||
self.assertFalse(parity["in_parity"])
|
||||
|
||||
def test_master_parity_carries_the_live_remote_dimension(self) -> None:
|
||||
"""B5: live_remote_head was never passed, dropping the #610 dimension."""
|
||||
stale = system_health.StaleRuntime(
|
||||
daemon_head="c" * 40,
|
||||
checkout_head="c" * 40,
|
||||
remote_head="d" * 40,
|
||||
stale=False,
|
||||
determinable=True,
|
||||
mutation_safe=False,
|
||||
reasons=(),
|
||||
)
|
||||
with patch.object(system_health, "assess_stale_runtime", return_value=stale):
|
||||
diag = console_recovery.diagnose_recovery()
|
||||
self.assertEqual(diag.master_parity.get("live_remote_head"), "d" * 40)
|
||||
|
||||
def test_verify_post_recovery(self) -> None:
|
||||
verification = console_recovery.verify_post_recovery()
|
||||
self.assertIn("clean", verification)
|
||||
self.assertIn("status", verification)
|
||||
self.assertIn("reasons", verification)
|
||||
|
||||
def test_unverified_inherited_binding_is_not_reported_clean(self) -> None:
|
||||
"""B2: ``not clear_eligible`` also read clean for unproven bindings."""
|
||||
binding = {
|
||||
"classification": stale_binding_recovery.CLASSIFICATION_UNVERIFIED_INHERITED,
|
||||
"clear_eligible": False,
|
||||
}
|
||||
diag = console_recovery.diagnose_recovery()
|
||||
patched = console_recovery.RecoveryDiagnosis(
|
||||
status=diag.status,
|
||||
clean=diag.clean,
|
||||
stale_runtime=diag.stale_runtime,
|
||||
master_parity=diag.master_parity,
|
||||
stale_binding=binding,
|
||||
contamination=diag.contamination,
|
||||
worktree_anomalies=diag.worktree_anomalies,
|
||||
playbooks=diag.playbooks,
|
||||
reasons=diag.reasons,
|
||||
)
|
||||
with patch.object(console_recovery, "diagnose_recovery", return_value=patched):
|
||||
verification = console_recovery.verify_post_recovery()
|
||||
self.assertFalse(verification["binding_clean"])
|
||||
self.assertEqual(
|
||||
verification["binding_classification"],
|
||||
stale_binding_recovery.CLASSIFICATION_UNVERIFIED_INHERITED,
|
||||
)
|
||||
|
||||
|
||||
class TestConsoleRecoveryApi(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self.app = create_app()
|
||||
self.client = TestClient(self.app)
|
||||
|
||||
def test_api_recovery_diagnose(self) -> None:
|
||||
res = self.client.get("/api/v1/system/recovery/diagnose")
|
||||
self.assertEqual(res.status_code, 200)
|
||||
data = res.json()
|
||||
self.assertIn("status", data)
|
||||
self.assertIn("clean", data)
|
||||
self.assertIn("playbooks", data)
|
||||
self.assertTrue(len(data["playbooks"]) >= 4)
|
||||
|
||||
def test_api_recovery_preview(self) -> None:
|
||||
res = self.client.post(
|
||||
"/api/v1/system/recovery/preview",
|
||||
json={"playbook_id": "clear_stale_binding", "target": "active"},
|
||||
)
|
||||
self.assertEqual(res.status_code, 200)
|
||||
data = res.json()
|
||||
self.assertEqual(data["playbook_id"], "clear_stale_binding")
|
||||
self.assertEqual(data["confirmation_phrase"], "confirm clear_stale_binding active")
|
||||
self.assertIn("mutation_ledger", data)
|
||||
|
||||
def test_api_recovery_apply_denied_without_auth(self) -> None:
|
||||
res = self.client.post(
|
||||
"/api/v1/system/recovery/apply",
|
||||
json={"playbook_id": "clear_stale_binding", "confirmation": "confirm clear_stale_binding"},
|
||||
)
|
||||
self.assertEqual(res.status_code, 400)
|
||||
data = res.json()
|
||||
self.assertFalse(data["success"])
|
||||
self.assertFalse(data["allowed"])
|
||||
|
||||
def test_api_recovery_apply_refuses_phase_two_write_with_dev_auth(self) -> None:
|
||||
"""B1: this previously asserted the phase-gate bypass as intended.
|
||||
|
||||
An authenticated operator posting a valid confirmation still must not
|
||||
execute a phase-2 write while the console is in phase 1. The refusal is
|
||||
the contract; a 200 here means the gate is not armed.
|
||||
"""
|
||||
env = {
|
||||
"WEBUI_AUTH_MODE": "local_dev",
|
||||
"WEBUI_DEV_SUBJECT": "[email protected]",
|
||||
"WEBUI_DEV_ROLE": "operator",
|
||||
}
|
||||
before = os.environ.get("GITEA_ACTIVE_WORKTREE")
|
||||
with patch.dict(os.environ, env):
|
||||
res = self.client.post(
|
||||
"/api/v1/system/recovery/apply",
|
||||
json={
|
||||
"playbook_id": "rebind_session_worktree",
|
||||
"target": "branches/feat-issue-644",
|
||||
"confirmation": "confirm rebind_session_worktree branches/feat-issue-644",
|
||||
},
|
||||
)
|
||||
self.assertEqual(res.status_code, 400)
|
||||
data = res.json()
|
||||
self.assertFalse(data["success"])
|
||||
self.assertFalse(data["allowed"])
|
||||
self.assertEqual(data["error"], console_authz.DENY_PHASE_NOT_ACTIVE)
|
||||
self.assertEqual(
|
||||
os.environ.get("GITEA_ACTIVE_WORKTREE"),
|
||||
before,
|
||||
"a refused apply must not have rebound the live process environment",
|
||||
)
|
||||
|
||||
def test_api_recovery_preview_reports_why_execution_is_disabled(self) -> None:
|
||||
res = self.client.post(
|
||||
"/api/v1/system/recovery/preview",
|
||||
json={"playbook_id": "rebind_session_worktree", "target": "active"},
|
||||
)
|
||||
self.assertEqual(res.status_code, 200)
|
||||
data = res.json()
|
||||
self.assertFalse(data["execution_enabled"])
|
||||
self.assertIn("execution_authorization", data)
|
||||
|
||||
def test_api_recovery_verify(self) -> None:
|
||||
res = self.client.get("/api/v1/system/recovery/verify")
|
||||
self.assertEqual(res.status_code, 200)
|
||||
data = res.json()
|
||||
self.assertIn("clean", data)
|
||||
self.assertIn("status", data)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,452 @@
|
||||
"""Read-only restart console: views, gates, and honesty rules (#667).
|
||||
|
||||
The console consumes the #655 substrate. These tests hold it to the three
|
||||
properties that make a status surface trustworthy:
|
||||
|
||||
* an unreadable source is reported unavailable, never rendered as green;
|
||||
* authorization is probed the way execution would probe it, so an allow is
|
||||
never shown for something that could not run;
|
||||
* the surface performs no mutation, including no write to the control-plane DB.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sqlite3
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from starlette.testclient import TestClient # noqa: E402
|
||||
|
||||
import restart_coordinator # noqa: E402
|
||||
from webui import console_authz, restart_console, restart_views # noqa: E402
|
||||
from webui.app import create_app # noqa: E402
|
||||
|
||||
NOW = datetime(2026, 7, 25, 21, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
def _principal(role: str) -> console_authz.Principal:
|
||||
return console_authz.Principal(
|
||||
subject="[email protected]",
|
||||
role=role,
|
||||
identity_source=console_authz.IDENTITY_LOCAL_DEV,
|
||||
authenticated=True,
|
||||
)
|
||||
|
||||
|
||||
def _inventory(*, complete: bool = True, sessions=(), leases=()):
|
||||
def _read(**_kwargs):
|
||||
return {
|
||||
"sessions": list(sessions),
|
||||
"leases": list(leases),
|
||||
"terminal_lock": None,
|
||||
"prior_recovery_attempts": [],
|
||||
"inventory_complete": complete,
|
||||
"incomplete_reasons": (
|
||||
[] if complete else ["fixture: inventory withheld"]
|
||||
),
|
||||
}
|
||||
|
||||
return _read
|
||||
|
||||
|
||||
def _live_session(session_id: str = "prgs-author-1234-abcd") -> dict:
|
||||
return {
|
||||
"session_id": session_id,
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": os.getpid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": (NOW - timedelta(seconds=30)).isoformat(),
|
||||
}
|
||||
|
||||
|
||||
def drain_proof_fixture() -> dict:
|
||||
"""A structurally complete but unsigned drain proof."""
|
||||
return {
|
||||
"version": "drain-proof/v1",
|
||||
"proof_id": "deadbeef" * 8,
|
||||
"clean": True,
|
||||
"issued_at": (NOW - timedelta(minutes=1)).isoformat(),
|
||||
"expires_at": (NOW + timedelta(minutes=5)).isoformat(),
|
||||
"requesting_session_id": "s-live",
|
||||
"impact_fingerprint": "f" * 64,
|
||||
"checks": [],
|
||||
"failed_checks": [],
|
||||
}
|
||||
|
||||
|
||||
class RestartClassMatrixTest(unittest.TestCase):
|
||||
def test_every_policy_class_is_rendered(self) -> None:
|
||||
views = restart_console.build_restart_class_views("operator")
|
||||
self.assertEqual(len(views), len(restart_coordinator.RESTART_CLASS_POLICIES))
|
||||
|
||||
def test_viewer_capability_is_role_scoped_not_generic(self) -> None:
|
||||
"""A worker role must not be shown as able to request a full restart."""
|
||||
author = {
|
||||
v.restart_class: v
|
||||
for v in restart_console.build_restart_class_views("author")
|
||||
}
|
||||
operator = {
|
||||
v.restart_class: v
|
||||
for v in restart_console.build_restart_class_views("operator")
|
||||
}
|
||||
full = restart_coordinator.RestartClass.FULL_MCP_RESTART.value
|
||||
|
||||
self.assertFalse(author[full].viewer_may_request)
|
||||
self.assertFalse(author[full].viewer_may_execute)
|
||||
self.assertTrue(operator[full].viewer_may_request)
|
||||
self.assertTrue(operator[full].viewer_may_execute)
|
||||
|
||||
def test_unknown_role_may_do_nothing(self) -> None:
|
||||
views = restart_console.build_restart_class_views("not-a-role")
|
||||
self.assertTrue(all(not v.viewer_may_request for v in views))
|
||||
self.assertTrue(all(not v.viewer_may_execute for v in views))
|
||||
|
||||
|
||||
class AuthorizationProbeTest(unittest.TestCase):
|
||||
def test_probe_asks_for_execution_so_phase_gate_is_reported(self) -> None:
|
||||
"""An admin clears the role bar and still cannot execute in Phase 1.
|
||||
|
||||
This is the case that distinguishes the two probes. Asked without
|
||||
``for_execution`` an admin is *allowed* for ``system.restart_namespace``,
|
||||
which on a control surface reads as a live button. Asked the way
|
||||
execution asks, the same principal is refused ``phase_not_active``. The
|
||||
console must report the second answer.
|
||||
"""
|
||||
by_id = {
|
||||
a.action_id: a
|
||||
for a in restart_console.build_action_authorizations(
|
||||
_principal(console_authz.ADMIN)
|
||||
)
|
||||
}
|
||||
restart = by_id["system.restart_namespace"]
|
||||
|
||||
self.assertFalse(restart.execution_enabled)
|
||||
self.assertEqual(restart.reason_code, console_authz.DENY_PHASE_NOT_ACTIVE)
|
||||
|
||||
permissive = console_authz.authorize(
|
||||
"system.restart_namespace", _principal(console_authz.ADMIN)
|
||||
)
|
||||
self.assertTrue(
|
||||
permissive.allowed,
|
||||
"guard precondition: without for_execution an admin is allowed, "
|
||||
"which is exactly why the console must not probe that way",
|
||||
)
|
||||
|
||||
def test_operator_is_refused_the_admin_only_restart_action(self) -> None:
|
||||
"""Role refusal precedes the phase gate and is reported as such."""
|
||||
by_id = {
|
||||
a.action_id: a
|
||||
for a in restart_console.build_action_authorizations(
|
||||
_principal(console_authz.OPERATOR)
|
||||
)
|
||||
}
|
||||
self.assertEqual(
|
||||
by_id["system.restart_namespace"].reason_code,
|
||||
console_authz.DENY_INSUFFICIENT_ROLE,
|
||||
)
|
||||
|
||||
def test_anonymous_is_denied_unauthenticated(self) -> None:
|
||||
by_id = {
|
||||
a.action_id: a for a in restart_console.build_action_authorizations(None)
|
||||
}
|
||||
self.assertEqual(
|
||||
by_id["system.restart_namespace"].reason_code,
|
||||
console_authz.DENY_UNAUTHENTICATED,
|
||||
)
|
||||
|
||||
def test_no_authorization_ever_reports_execution_enabled(self) -> None:
|
||||
for role in (
|
||||
console_authz.VIEWER,
|
||||
console_authz.OPERATOR,
|
||||
console_authz.CONTROLLER,
|
||||
console_authz.ADMIN,
|
||||
):
|
||||
for auth in restart_console.build_action_authorizations(_principal(role)):
|
||||
self.assertFalse(
|
||||
auth.execution_enabled,
|
||||
f"{role} reported execution_enabled for {auth.action_id}",
|
||||
)
|
||||
|
||||
|
||||
class ImpactPreviewTest(unittest.TestCase):
|
||||
def test_impact_renders_from_coordinator_dto(self) -> None:
|
||||
impact, source = restart_console.load_impact_report(
|
||||
principal=_principal(console_authz.OPERATOR),
|
||||
read_inventory=_inventory(sessions=[_live_session()]),
|
||||
now=NOW,
|
||||
)
|
||||
self.assertTrue(source.available)
|
||||
self.assertIsNotNone(impact)
|
||||
self.assertEqual(
|
||||
impact["restart_class"],
|
||||
restart_coordinator.RestartClass.FULL_MCP_RESTART.value,
|
||||
)
|
||||
self.assertIn("verdict", impact)
|
||||
self.assertFalse(impact["restart_performed"])
|
||||
self.assertTrue(impact["dry_run"])
|
||||
|
||||
def test_incomplete_inventory_is_surfaced_and_denies(self) -> None:
|
||||
impact, source = restart_console.load_impact_report(
|
||||
principal=_principal(console_authz.OPERATOR),
|
||||
read_inventory=_inventory(complete=False),
|
||||
now=NOW,
|
||||
)
|
||||
self.assertFalse(impact["inventory_complete"])
|
||||
self.assertFalse(impact["allow_restart"])
|
||||
self.assertTrue(source.detail, "incomplete inventory must explain itself")
|
||||
|
||||
def test_inventory_reader_failure_is_unavailable_not_empty(self) -> None:
|
||||
"""A reader that raises must not be rendered as 'no sessions affected'."""
|
||||
|
||||
def _boom(**_kwargs):
|
||||
raise RuntimeError("control-plane unreachable")
|
||||
|
||||
impact, source = restart_console.load_impact_report(
|
||||
principal=_principal(console_authz.OPERATOR),
|
||||
read_inventory=_boom,
|
||||
now=NOW,
|
||||
)
|
||||
self.assertIsNone(impact)
|
||||
self.assertFalse(source.available)
|
||||
self.assertIn("control-plane unreachable", source.detail)
|
||||
|
||||
|
||||
class ControlPlaneReadTest(unittest.TestCase):
|
||||
def test_missing_database_is_incomplete_not_empty(self) -> None:
|
||||
inventory = restart_console.read_control_plane_inventory(
|
||||
db_path="/nonexistent/control-plane.sqlite3"
|
||||
)
|
||||
self.assertFalse(inventory["inventory_complete"])
|
||||
self.assertEqual(inventory["sessions"], [])
|
||||
self.assertTrue(inventory["incomplete_reasons"])
|
||||
|
||||
def test_reader_never_creates_the_database(self) -> None:
|
||||
"""Reading status must not bring a control-plane DB into existence.
|
||||
|
||||
The path deliberately sits in a directory that already exists: a
|
||||
read-write ``sqlite3.connect`` would happily create the file there, so
|
||||
this fails if the reader ever stops opening the database ``mode=ro``.
|
||||
A nested-missing-directory path would pass for the wrong reason,
|
||||
because sqlite cannot create the parent directory either way.
|
||||
"""
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
path = os.path.join(tmp, "control_plane.sqlite3")
|
||||
self.assertTrue(os.path.isdir(os.path.dirname(path)))
|
||||
|
||||
inventory = restart_console.read_control_plane_inventory(db_path=path)
|
||||
|
||||
self.assertFalse(
|
||||
os.path.exists(path),
|
||||
"reading restart status created a control-plane database",
|
||||
)
|
||||
self.assertFalse(inventory["inventory_complete"])
|
||||
|
||||
def test_reads_active_sessions_from_a_real_database(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
path = os.path.join(tmp, "cp.sqlite3")
|
||||
conn = sqlite3.connect(path)
|
||||
conn.execute(
|
||||
"CREATE TABLE sessions (session_id TEXT, role TEXT, profile TEXT,"
|
||||
" pid INTEGER, status TEXT, last_heartbeat_at TEXT)"
|
||||
)
|
||||
conn.execute(
|
||||
"CREATE TABLE work_items (work_item_id INTEGER, kind TEXT,"
|
||||
" number INTEGER)"
|
||||
)
|
||||
conn.execute(
|
||||
"CREATE TABLE leases (lease_id TEXT, session_id TEXT, role TEXT,"
|
||||
" phase TEXT, status TEXT, worktree_path TEXT,"
|
||||
" work_item_id INTEGER, expires_at TEXT)"
|
||||
)
|
||||
conn.execute(
|
||||
"INSERT INTO sessions VALUES (?,?,?,?,?,?)",
|
||||
("s-live", "author", "prgs-author", 4242, "active", NOW.isoformat()),
|
||||
)
|
||||
conn.execute(
|
||||
"INSERT INTO sessions VALUES (?,?,?,?,?,?)",
|
||||
("s-done", "author", "prgs-author", 11, "closed", NOW.isoformat()),
|
||||
)
|
||||
conn.execute("INSERT INTO work_items VALUES (1, 'issue', 667)")
|
||||
conn.execute(
|
||||
"INSERT INTO leases VALUES (?,?,?,?,?,?,?,?)",
|
||||
(
|
||||
"l-1",
|
||||
"s-live",
|
||||
"author",
|
||||
"allocated",
|
||||
"active",
|
||||
None,
|
||||
1,
|
||||
NOW.isoformat(),
|
||||
),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
inventory = restart_console.read_control_plane_inventory(db_path=path)
|
||||
|
||||
self.assertTrue(inventory["inventory_complete"])
|
||||
self.assertEqual([s["session_id"] for s in inventory["sessions"]], ["s-live"])
|
||||
self.assertEqual(inventory["leases"][0]["work_number"], 667)
|
||||
|
||||
|
||||
class DrainAndReconcileTest(unittest.TestCase):
|
||||
def test_absent_drain_proof_is_not_a_pass(self) -> None:
|
||||
drain, source = restart_console.load_drain_status(proof=None, now=NOW)
|
||||
self.assertIsNone(drain)
|
||||
self.assertFalse(source.available)
|
||||
self.assertIn("denies", source.detail)
|
||||
|
||||
def test_tampered_drain_proof_is_reported_invalid(self) -> None:
|
||||
proof = drain_proof_fixture()
|
||||
proof["clean"] = True
|
||||
proof["proof_id"] = "0" * 64
|
||||
drain, source = restart_console.load_drain_status(proof=proof, now=NOW)
|
||||
self.assertTrue(source.available)
|
||||
self.assertFalse(drain["valid"])
|
||||
|
||||
def test_absent_reconcile_proof_is_unavailable(self) -> None:
|
||||
reconcile, source = restart_console.load_reconcile_status(load_proof=None)
|
||||
self.assertIsNone(reconcile)
|
||||
self.assertFalse(source.available)
|
||||
|
||||
def test_reconcile_proof_is_rendered_when_supplied(self) -> None:
|
||||
payload = {
|
||||
"overall_status": "degraded",
|
||||
"mode": "log_only",
|
||||
"resolved_count": 3,
|
||||
"unresolved_count": 2,
|
||||
"items": [
|
||||
{
|
||||
"dimension": "leases",
|
||||
"status": "unresolved",
|
||||
"summary": "2 orphaned leases",
|
||||
"follow_up_required": True,
|
||||
}
|
||||
],
|
||||
}
|
||||
reconcile, source = restart_console.load_reconcile_status(
|
||||
load_proof=lambda: payload
|
||||
)
|
||||
self.assertTrue(source.available)
|
||||
self.assertEqual(reconcile["unresolved_count"], 2)
|
||||
|
||||
|
||||
class RenderingTest(unittest.TestCase):
|
||||
def _snapshot(self, **kwargs):
|
||||
params = {
|
||||
"principal": _principal(console_authz.OPERATOR),
|
||||
"read_inventory": _inventory(sessions=[_live_session()]),
|
||||
"now": NOW,
|
||||
}
|
||||
params.update(kwargs)
|
||||
return restart_console.load_restart_console_snapshot(**params)
|
||||
|
||||
def test_page_renders_every_section(self) -> None:
|
||||
html = restart_views.render_restart_console_page(self._snapshot())
|
||||
for heading in (
|
||||
"Impact preview",
|
||||
"Drain proof",
|
||||
"Post-restart reconcile",
|
||||
"Restart classes",
|
||||
"Approval controls",
|
||||
"Break-glass",
|
||||
):
|
||||
self.assertIn(heading, html)
|
||||
|
||||
def test_hostile_session_id_is_escaped(self) -> None:
|
||||
hostile = "<script>alert('x')</script>"
|
||||
html = restart_views.render_restart_console_page(
|
||||
self._snapshot(read_inventory=_inventory(sessions=[_live_session(hostile)]))
|
||||
)
|
||||
self.assertNotIn("<script>alert", html)
|
||||
self.assertIn("<script>", html)
|
||||
|
||||
def test_unavailable_impact_says_unsafe_rather_than_clean(self) -> None:
|
||||
def _boom(**_kwargs):
|
||||
raise RuntimeError("nope")
|
||||
|
||||
snapshot = self._snapshot(read_inventory=_boom)
|
||||
html = restart_views.render_restart_console_page(snapshot)
|
||||
self.assertIn("blast radius of a restart is unknown", html)
|
||||
self.assertIn("unavailable", html)
|
||||
|
||||
def test_break_glass_is_hidden_from_unprivileged_viewers(self) -> None:
|
||||
viewer_html = restart_views.render_restart_console_page(
|
||||
self._snapshot(principal=_principal(console_authz.VIEWER))
|
||||
)
|
||||
self.assertIn("visible to operator-class", viewer_html)
|
||||
self.assertNotIn(
|
||||
f"#{restart_console.BREAK_GLASS_ISSUE}", viewer_html
|
||||
)
|
||||
|
||||
def test_break_glass_shown_to_operator_is_marked_unavailable(self) -> None:
|
||||
html = restart_views.render_restart_console_page(self._snapshot())
|
||||
self.assertIn("unavailable", html)
|
||||
self.assertIn(f"#{restart_console.BREAK_GLASS_ISSUE}", html)
|
||||
|
||||
def test_snapshot_always_declares_itself_read_only(self) -> None:
|
||||
self.assertTrue(self._snapshot().read_only)
|
||||
|
||||
|
||||
class RestartConsoleRouteTest(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self.client = TestClient(create_app())
|
||||
|
||||
def test_page_route_renders(self) -> None:
|
||||
res = self.client.get("/runtime/restart")
|
||||
self.assertEqual(res.status_code, 200)
|
||||
self.assertIn("Restart status and impact", res.text)
|
||||
|
||||
def test_api_route_exports_snapshot(self) -> None:
|
||||
res = self.client.get("/api/v1/system/restart/status")
|
||||
self.assertEqual(res.status_code, 200)
|
||||
payload = res.json()
|
||||
self.assertTrue(payload["read_only"])
|
||||
self.assertEqual(payload["links"]["issue"], 667)
|
||||
self.assertEqual(
|
||||
len(payload["restart_classes"]),
|
||||
len(restart_coordinator.RESTART_CLASS_POLICIES),
|
||||
)
|
||||
|
||||
def test_restart_class_is_selectable(self) -> None:
|
||||
res = self.client.get(
|
||||
"/api/v1/system/restart/status?restart_class=client_reconnect"
|
||||
)
|
||||
self.assertEqual(res.status_code, 200)
|
||||
self.assertEqual(res.json()["impact"]["restart_class"], "client_reconnect")
|
||||
|
||||
def test_unknown_restart_class_fails_closed(self) -> None:
|
||||
res = self.client.get(
|
||||
"/api/v1/system/restart/status?restart_class=obliterate-everything"
|
||||
)
|
||||
self.assertEqual(res.status_code, 200)
|
||||
impact = res.json()["impact"]
|
||||
self.assertFalse(impact["allow_restart"])
|
||||
|
||||
def test_anonymous_api_reader_gets_no_execution_grant(self) -> None:
|
||||
payload = self.client.get("/api/v1/system/restart/status").json()
|
||||
self.assertFalse(payload["break_glass"]["available"])
|
||||
for auth in payload["authorizations"]:
|
||||
self.assertFalse(auth["execution_enabled"])
|
||||
|
||||
def test_route_is_registered_in_nav(self) -> None:
|
||||
from webui.nav import nav_hrefs
|
||||
|
||||
self.assertIn("/runtime/restart", nav_hrefs())
|
||||
|
||||
def test_no_write_method_is_exposed(self) -> None:
|
||||
"""The surface is read-only: nothing accepts a POST."""
|
||||
for path in ("/runtime/restart", "/api/v1/system/restart/status"):
|
||||
self.assertEqual(self.client.post(path).status_code, 405, path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
+236
@@ -47,6 +47,9 @@ from webui.traffic_views import render_traffic_page
|
||||
from webui.worktree_scanner import load_hygiene_snapshot, snapshot_to_dict as worktree_snapshot_to_dict
|
||||
from webui.worktree_views import render_worktrees_page
|
||||
from webui.runtime_health import load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict
|
||||
import restart_coordinator
|
||||
from webui.restart_console import load_restart_console_snapshot
|
||||
from webui.restart_views import render_restart_console_page
|
||||
from webui.runtime_views import render_runtime_page
|
||||
from webui.session_loader import (
|
||||
load_session_view_snapshot,
|
||||
@@ -77,6 +80,8 @@ from webui.system_health import (
|
||||
snapshot_to_dict as system_health_to_dict,
|
||||
)
|
||||
from webui.system_health_views import render_system_health_page
|
||||
from webui import request_service
|
||||
from webui.request_views import render_requests_page
|
||||
|
||||
_READ_ONLY_METHODS = frozenset({"GET", "HEAD", "OPTIONS"})
|
||||
_AUDIT_MUTATION_PATHS = frozenset({"/audit", "/api/audit"})
|
||||
@@ -205,6 +210,84 @@ async def system_health(request: Request) -> HTMLResponse:
|
||||
)
|
||||
|
||||
|
||||
async def api_recovery_diagnose(_request: Request) -> JSONResponse:
|
||||
from webui.console_recovery import diagnose_recovery
|
||||
diag = diagnose_recovery()
|
||||
return JSONResponse({
|
||||
"status": diag.status,
|
||||
"clean": diag.clean,
|
||||
"stale_runtime": diag.stale_runtime,
|
||||
"master_parity": diag.master_parity,
|
||||
"stale_binding": diag.stale_binding,
|
||||
"contamination": diag.contamination,
|
||||
"worktree_anomalies": list(diag.worktree_anomalies),
|
||||
"reasons": list(diag.reasons),
|
||||
"playbooks": [
|
||||
{
|
||||
"playbook_id": pb.playbook_id,
|
||||
"label": pb.label,
|
||||
"action_id": pb.action_id,
|
||||
"description": pb.description,
|
||||
"eligible": pb.eligible,
|
||||
"requires_confirmation": pb.requires_confirmation,
|
||||
"reason": pb.reason,
|
||||
"params_schema": pb.params_schema,
|
||||
}
|
||||
for pb in diag.playbooks
|
||||
],
|
||||
})
|
||||
|
||||
|
||||
async def api_recovery_preview(request: Request) -> JSONResponse:
|
||||
from webui.console_recovery import build_recovery_preview
|
||||
body = {}
|
||||
try:
|
||||
body = await request.json()
|
||||
except Exception:
|
||||
pass
|
||||
playbook_id = body.get("playbook_id") or request.query_params.get("playbook_id") or ""
|
||||
target = body.get("target") or request.query_params.get("target")
|
||||
principal = resolve_principal(request.headers)
|
||||
preview = build_recovery_preview(
|
||||
playbook_id=playbook_id,
|
||||
target=target,
|
||||
params=body,
|
||||
principal=principal,
|
||||
)
|
||||
status = 200 if preview.get("playbook_id") else 400
|
||||
return JSONResponse(preview, status_code=status)
|
||||
|
||||
|
||||
async def api_recovery_apply(request: Request) -> JSONResponse:
|
||||
from webui.console_recovery import execute_recovery_playbook
|
||||
body = {}
|
||||
try:
|
||||
body = await request.json()
|
||||
except Exception:
|
||||
pass
|
||||
playbook_id = body.get("playbook_id", "")
|
||||
confirmation = body.get("confirmation", "")
|
||||
target = body.get("target")
|
||||
principal = resolve_principal(request.headers)
|
||||
request_id = getattr(request.state, "request_id", None)
|
||||
result = execute_recovery_playbook(
|
||||
playbook_id=playbook_id,
|
||||
confirmation=confirmation,
|
||||
target=target,
|
||||
params=body,
|
||||
principal=principal,
|
||||
request_id=request_id,
|
||||
)
|
||||
status_code = 200 if result.get("success") else 400
|
||||
return JSONResponse(result, status_code=status_code)
|
||||
|
||||
|
||||
async def api_recovery_verify(_request: Request) -> JSONResponse:
|
||||
from webui.console_recovery import verify_post_recovery
|
||||
verification = verify_post_recovery()
|
||||
return JSONResponse(verification, status_code=200)
|
||||
|
||||
|
||||
async def queue(_request: Request) -> HTMLResponse:
|
||||
snapshot = load_queue_snapshot()
|
||||
return HTMLResponse(render_page(title="Queue", body_html=render_queue_page(snapshot)))
|
||||
@@ -336,6 +419,33 @@ async def api_runtime(_request: Request) -> JSONResponse:
|
||||
return JSONResponse(runtime_snapshot_to_dict(load_runtime_snapshot()))
|
||||
|
||||
|
||||
def _restart_console_snapshot(request: Request):
|
||||
"""Build the read-only restart snapshot for the requesting principal (#667)."""
|
||||
principal = resolve_principal(request.headers)
|
||||
restart_class = (
|
||||
request.query_params.get("restart_class")
|
||||
or restart_coordinator.RestartClass.FULL_MCP_RESTART.value
|
||||
)
|
||||
return load_restart_console_snapshot(
|
||||
principal=principal, restart_class=restart_class
|
||||
)
|
||||
|
||||
|
||||
async def restart_console_page(request: Request) -> HTMLResponse:
|
||||
"""Restart status, impact preview, and approval state (#667). Read-only."""
|
||||
snapshot = _restart_console_snapshot(request)
|
||||
return HTMLResponse(
|
||||
render_page(
|
||||
title="Restart", body_html=render_restart_console_page(snapshot)
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
async def api_restart_status(request: Request) -> JSONResponse:
|
||||
"""JSON export of the read-only restart console snapshot (#667)."""
|
||||
return JSONResponse(_restart_console_snapshot(request).as_dict())
|
||||
|
||||
|
||||
async def sessions(_request: Request) -> HTMLResponse:
|
||||
"""Runtime and session view (#641) — read-only composition of health + inventory."""
|
||||
snapshot = load_session_view_snapshot()
|
||||
@@ -779,6 +889,109 @@ async def api_v1_analytics_ingest(request: Request) -> JSONResponse:
|
||||
)
|
||||
|
||||
|
||||
def _default_request_scope() -> dict[str, str]:
|
||||
"""Resolve remote/org/repo from the project registry for request forms.
|
||||
|
||||
Returns an empty mapping when the registry cannot be read, which makes
|
||||
``parse_request`` reject a request that did not name its own scope rather
|
||||
than letting it default to some other repository.
|
||||
"""
|
||||
from webui.queue_loader import _host_from_url # host normalisation helper
|
||||
|
||||
registry, error = _load_project_registry()
|
||||
if error is not None or not registry.projects:
|
||||
return {}
|
||||
project = registry.projects[0]
|
||||
host = _host_from_url(project.remote_host)
|
||||
return {
|
||||
"remote": _derive_remote(host),
|
||||
"org": project.gitea_owner or "",
|
||||
"repo": project.repo_name or "",
|
||||
}
|
||||
|
||||
|
||||
async def _request_payload(request: Request) -> dict[str, object]:
|
||||
"""Read a request body as JSON or form-encoded. Never raises."""
|
||||
content_type = (request.headers.get("content-type") or "").lower()
|
||||
if "application/json" in content_type:
|
||||
try:
|
||||
body = await request.json()
|
||||
except Exception:
|
||||
return {}
|
||||
return dict(body) if isinstance(body, dict) else {}
|
||||
try:
|
||||
form = await request.form()
|
||||
except Exception:
|
||||
return {}
|
||||
return {key: form[key] for key in form}
|
||||
|
||||
|
||||
async def requests_page(request: Request) -> HTMLResponse:
|
||||
"""Operator request form and intent preview (#643).
|
||||
|
||||
POST here only ever *previews*. Initiation is a separate confirmed call to
|
||||
``/api/v1/requests/apply`` so that submitting this form cannot reserve
|
||||
work as a side effect.
|
||||
"""
|
||||
submitted: dict[str, object] = {}
|
||||
preview = None
|
||||
error = None
|
||||
if request.method == "POST":
|
||||
submitted = await _request_payload(request)
|
||||
work_request, error = request_service.parse_request(
|
||||
submitted, default_scope=_default_request_scope()
|
||||
)
|
||||
if work_request is not None:
|
||||
preview = request_service.preview_request(
|
||||
work_request,
|
||||
principal=resolve_principal(headers=dict(request.headers)),
|
||||
)
|
||||
return HTMLResponse(
|
||||
render_requests_page(
|
||||
preview=preview, error=error, submitted=submitted
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
async def api_v1_request_preview(request: Request) -> JSONResponse:
|
||||
"""Dry-run authorization and intent preview for a work request (#643)."""
|
||||
payload = await _request_payload(request)
|
||||
work_request, error = request_service.parse_request(
|
||||
payload, default_scope=_default_request_scope()
|
||||
)
|
||||
if work_request is None:
|
||||
return JSONResponse(error.to_dict(), status_code=400)
|
||||
preview = request_service.preview_request(
|
||||
work_request,
|
||||
principal=resolve_principal(headers=dict(request.headers)),
|
||||
)
|
||||
return JSONResponse(
|
||||
preview.to_dict(), status_code=200 if preview.authorized else 403
|
||||
)
|
||||
|
||||
|
||||
async def api_v1_request_apply(request: Request) -> JSONResponse:
|
||||
"""Initiate a previewed work request through the allocator (#643).
|
||||
|
||||
Fail-closed at every step: unauthorized, unconfirmed, not-next-safe, and
|
||||
already-claimed all return without attempting an assignment.
|
||||
"""
|
||||
payload = await _request_payload(request)
|
||||
work_request, error = request_service.parse_request(
|
||||
payload, default_scope=_default_request_scope()
|
||||
)
|
||||
if work_request is None:
|
||||
return JSONResponse(error.to_dict(), status_code=400)
|
||||
confirm = _truthy_flag(str(payload.get("confirm") or ""))
|
||||
result = request_service.apply_request(
|
||||
work_request,
|
||||
principal=resolve_principal(headers=dict(request.headers)),
|
||||
confirm=confirm,
|
||||
)
|
||||
status = int(result.pop("status_code", 403))
|
||||
return JSONResponse(result, status_code=status)
|
||||
|
||||
|
||||
async def method_not_allowed(request: Request, _exc: Exception) -> Response:
|
||||
path = request.url.path
|
||||
if path in _AUDIT_MUTATION_PATHS and request.method == "POST":
|
||||
@@ -821,6 +1034,13 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
Route("/api/prompts", api_prompts, methods=["GET"]),
|
||||
Route("/runtime", runtime, methods=["GET"]),
|
||||
Route("/api/runtime", api_runtime, methods=["GET"]),
|
||||
# #667 read-only restart status / impact preview / approval state.
|
||||
Route("/runtime/restart", restart_console_page, methods=["GET"]),
|
||||
Route(
|
||||
"/api/v1/system/restart/status",
|
||||
api_restart_status,
|
||||
methods=["GET"],
|
||||
),
|
||||
Route("/sessions", sessions, methods=["GET"]),
|
||||
Route("/api/sessions", api_sessions, methods=["GET"]),
|
||||
Route("/api/v1/sessions", api_sessions, methods=["GET"]),
|
||||
@@ -848,6 +1068,17 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
api_action_attempt,
|
||||
methods=["POST"],
|
||||
),
|
||||
Route("/requests", requests_page, methods=["GET", "POST"]),
|
||||
Route(
|
||||
"/api/v1/requests/preview",
|
||||
api_v1_request_preview,
|
||||
methods=["POST"],
|
||||
),
|
||||
Route(
|
||||
"/api/v1/requests/apply",
|
||||
api_v1_request_apply,
|
||||
methods=["POST"],
|
||||
),
|
||||
Route("/api/leases", api_leases, methods=["GET"]),
|
||||
Route("/api/v1/inventory", api_inventory, methods=["GET"]),
|
||||
Route(
|
||||
@@ -860,6 +1091,11 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
api_console_security_model,
|
||||
methods=["GET"],
|
||||
),
|
||||
# #644 Phase 2 Recovery API routes
|
||||
Route("/api/v1/system/recovery/diagnose", api_recovery_diagnose, methods=["GET"]),
|
||||
Route("/api/v1/system/recovery/preview", api_recovery_preview, methods=["POST", "GET"]),
|
||||
Route("/api/v1/system/recovery/apply", api_recovery_apply, methods=["POST"]),
|
||||
Route("/api/v1/system/recovery/verify", api_recovery_verify, methods=["POST", "GET"]),
|
||||
*[
|
||||
Route(path, phase_stub, methods=["GET"])
|
||||
for path in STUB_PAGES
|
||||
|
||||
+105
-8
@@ -115,6 +115,12 @@ class ConsoleAction:
|
||||
break_glass: bool
|
||||
phase: int
|
||||
summary: str
|
||||
# Opt-in switch for an action whose execution path is genuinely wired
|
||||
# ahead of its phase becoming globally active (#643). Naming a variable
|
||||
# here enables nothing on its own: the variable must also be set in the
|
||||
# environment. An action that leaves this ``None`` can only execute once
|
||||
# ACTIVE_PHASE reaches its phase, exactly as before.
|
||||
execution_env_flag: str | None = None
|
||||
|
||||
@property
|
||||
def mcp_permission(self) -> str:
|
||||
@@ -277,6 +283,61 @@ _ACTION_SPECS: tuple[ConsoleAction, ...] = (
|
||||
phase=2,
|
||||
summary="Restart one MCP namespace via the host supervisor.",
|
||||
),
|
||||
# #644: Phase 2 recovery controls & playbooks.
|
||||
ConsoleAction(
|
||||
action_id="system.clear_stale_binding",
|
||||
task_key="clear_stale_binding",
|
||||
action_class=CLASS_WRITE,
|
||||
minimum_role=OPERATOR,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary="Clear provably stale or superseded GITEA_ACTIVE_WORKTREE binding.",
|
||||
),
|
||||
ConsoleAction(
|
||||
action_id="system.rebind_session_worktree",
|
||||
task_key="rebind_session_worktree",
|
||||
action_class=CLASS_WRITE,
|
||||
minimum_role=OPERATOR,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary="Rebind session worktree context to verified lease worktree.",
|
||||
),
|
||||
ConsoleAction(
|
||||
action_id="system.reconcile_cleanups",
|
||||
task_key="reconcile_cleanups",
|
||||
action_class=CLASS_PRIVILEGED,
|
||||
minimum_role=CONTROLLER,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary="Run reconciler cleanup for merged or superseded PR branches.",
|
||||
),
|
||||
# #643: submit a work request — desired role, issue/PR, intent — and let
|
||||
# the allocator reserve it. This is the one Phase 2 action whose execution
|
||||
# path is actually implemented (``webui.request_service``), so it carries
|
||||
# the opt-in flag; it stays denied until an operator sets that variable.
|
||||
# Authority is operator-class because the outcome is a claim, not a Gitea
|
||||
# verdict: initiating reviewer or merger *work* does not grant the right
|
||||
# to approve or merge, which stays with the MCP role profile.
|
||||
ConsoleAction(
|
||||
action_id="initiate_workflow",
|
||||
task_key="allocate_next_work",
|
||||
action_class=CLASS_WRITE,
|
||||
minimum_role=OPERATOR,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary=(
|
||||
"Preview and initiate allocator-owned workflow work for a role."
|
||||
),
|
||||
execution_env_flag="WEBUI_REQUESTS_EXECUTION",
|
||||
),
|
||||
)
|
||||
|
||||
ACTIONS: dict[str, ConsoleAction] = {a.action_id: a for a in _ACTION_SPECS}
|
||||
@@ -430,6 +491,33 @@ ALLOW_PREVIEW = "allowed_preview_only"
|
||||
# gated on this model landing; nothing here enables it.
|
||||
ACTIVE_PHASE = 1
|
||||
|
||||
_TRUTHY = frozenset({"1", "true", "yes", "on"})
|
||||
|
||||
|
||||
def execution_wired(
|
||||
action: ConsoleAction | None, env: dict[str, str] | None = None
|
||||
) -> bool:
|
||||
"""Whether *action* has a live execution path right now.
|
||||
|
||||
Two ways to be wired, and only two. The action's phase is active, or the
|
||||
action declares an opt-in environment variable *and* that variable is set.
|
||||
Everything else — including every action that never declares a flag — is
|
||||
unwired, so the default across the registry stays deny.
|
||||
|
||||
Bumping ``ACTIVE_PHASE`` would enable execution for every action of that
|
||||
phase at once. The per-action flag exists so a single implemented action
|
||||
can go live without dragging its unimplemented phase-mates with it.
|
||||
"""
|
||||
if action is None:
|
||||
return False
|
||||
if action.phase <= ACTIVE_PHASE:
|
||||
return True
|
||||
flag = (action.execution_env_flag or "").strip()
|
||||
if not flag:
|
||||
return False
|
||||
source = env if env is not None else os.environ
|
||||
return (source.get(flag) or "").strip().lower() in _TRUTHY
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AuthorizationDecision:
|
||||
@@ -469,16 +557,19 @@ def authorize(
|
||||
principal: Principal | None = None,
|
||||
*,
|
||||
for_execution: bool = False,
|
||||
env: dict[str, str] | None = None,
|
||||
) -> AuthorizationDecision:
|
||||
"""Decide whether *principal* may invoke *action_id*. Deny by default.
|
||||
|
||||
``for_execution`` distinguishes a read-only preview from a real invocation.
|
||||
Even an allowed decision reports ``execution_enabled=False`` while the
|
||||
console is in Phase 1, so no caller can read an allow as permission to
|
||||
mutate.
|
||||
``execution_enabled`` reports whether the action has a live execution path
|
||||
at all (:func:`execution_wired`) — for every action without an explicit
|
||||
opt-in flag that stays ``False`` while the console is in Phase 1, so no
|
||||
caller can read an allow as permission to mutate.
|
||||
"""
|
||||
who = principal if principal is not None else ANONYMOUS
|
||||
action = get_action(action_id)
|
||||
wired = execution_wired(action, env)
|
||||
|
||||
if action is None:
|
||||
return AuthorizationDecision(
|
||||
@@ -497,7 +588,7 @@ def authorize(
|
||||
"requires_confirmation": action.requires_confirmation,
|
||||
"dual_control": action.dual_control,
|
||||
"break_glass": action.break_glass,
|
||||
"execution_enabled": False,
|
||||
"execution_enabled": wired,
|
||||
}
|
||||
|
||||
if not who.authenticated:
|
||||
@@ -530,13 +621,19 @@ def authorize(
|
||||
**base,
|
||||
)
|
||||
|
||||
if for_execution and action.phase > ACTIVE_PHASE:
|
||||
if for_execution and not wired:
|
||||
return AuthorizationDecision(
|
||||
allowed=False,
|
||||
reason_code=DENY_PHASE_NOT_ACTIVE,
|
||||
detail=(
|
||||
f"Action {action_id!r} belongs to phase {action.phase}; the "
|
||||
f"console is in phase {ACTIVE_PHASE}. Execution is not wired."
|
||||
f"console is in phase {ACTIVE_PHASE}"
|
||||
+ (
|
||||
f" and {action.execution_env_flag} is not set"
|
||||
if action.execution_env_flag
|
||||
else ""
|
||||
)
|
||||
+ ". Execution is not wired."
|
||||
),
|
||||
**base,
|
||||
)
|
||||
@@ -545,8 +642,8 @@ def authorize(
|
||||
allowed=True,
|
||||
reason_code=ALLOW_PREVIEW,
|
||||
detail=(
|
||||
"Principal holds the required role. Preview only — execution "
|
||||
"remains disabled until the Phase 2 action framework ships."
|
||||
"Principal holds the required role. Execution proceeds only for an "
|
||||
"action with a wired execution path; everything else is preview."
|
||||
),
|
||||
**base,
|
||||
)
|
||||
|
||||
@@ -0,0 +1,721 @@
|
||||
"""Web Console Phase 2 Recovery Controls & Playbooks (#644).
|
||||
|
||||
Provides canonical recovery controls for the web console:
|
||||
1. Diagnosis: Surfaces stale runtimes, worktree binding errors, contamination markers,
|
||||
and un-reconciled cleanups.
|
||||
2. Gated Actions & Playbooks: Guided recovery (rebind session worktree, clear stale
|
||||
binding, trigger reconciler cleanups, sanctioned restart).
|
||||
3. RBAC, Contamination (#630), and Master Parity (#610) integration.
|
||||
4. Audit trail via ``console_audit`` and mandatory post-recovery revalidation.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import master_parity_gate
|
||||
import runtime_recovery_guard
|
||||
import stale_binding_recovery
|
||||
from webui import console_audit, console_authz, sanctioned_restart, system_health, worktree_scanner
|
||||
|
||||
# --- Recovery Statuses ------------------------------------------------------
|
||||
STATUS_HEALTHY = "healthy"
|
||||
STATUS_ACTION_REQUIRED = "action_required"
|
||||
STATUS_BLOCKED_CONTAMINATION = "blocked_contamination"
|
||||
STATUS_RECONNECT_REQUIRED = "reconnect_required"
|
||||
|
||||
# --- Playbook Identifiers ---------------------------------------------------
|
||||
PLAYBOOK_CLEAR_STALE_BINDING = "clear_stale_binding"
|
||||
PLAYBOOK_REBIND_SESSION = "rebind_session_worktree"
|
||||
PLAYBOOK_RECONCILE_CLEANUPS = "reconcile_cleanups"
|
||||
PLAYBOOK_SANCTIONED_RESTART = "sanctioned_restart"
|
||||
|
||||
KNOWN_PLAYBOOKS: tuple[str, ...] = (
|
||||
PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
PLAYBOOK_REBIND_SESSION,
|
||||
PLAYBOOK_RECONCILE_CLEANUPS,
|
||||
PLAYBOOK_SANCTIONED_RESTART,
|
||||
)
|
||||
|
||||
# --- Console Action Mapping -------------------------------------------------
|
||||
ACTION_CLEAR_STALE_BINDING = "system.clear_stale_binding"
|
||||
ACTION_REBIND_SESSION = "system.rebind_session_worktree"
|
||||
ACTION_RECONCILE_CLEANUPS = "system.reconcile_cleanups"
|
||||
|
||||
PLAYBOOK_ACTIONS: dict[str, str] = {
|
||||
PLAYBOOK_CLEAR_STALE_BINDING: ACTION_CLEAR_STALE_BINDING,
|
||||
PLAYBOOK_REBIND_SESSION: ACTION_REBIND_SESSION,
|
||||
PLAYBOOK_RECONCILE_CLEANUPS: ACTION_RECONCILE_CLEANUPS,
|
||||
PLAYBOOK_SANCTIONED_RESTART: sanctioned_restart.ACTION_RESTART_NAMESPACE,
|
||||
}
|
||||
|
||||
#: Task key handed to :func:`runtime_recovery_guard.assess_contamination_gate`.
|
||||
#: A console *action id* is not a task name and is not a member of
|
||||
#: ``CONTAMINATION_GATED_TASKS``, so passing one left the #630 gate inert. Every
|
||||
#: writing recovery playbook shares this one gated task key; the reconciler
|
||||
#: cleanup playbook is exempted separately because it is the designated remedy.
|
||||
CONTAMINATION_GATED_TASK = "console_recovery_apply"
|
||||
|
||||
#: Remote whose contamination markers govern this console. Markers are written
|
||||
#: per remote, so reading the wrong one reports a contaminated runtime clean.
|
||||
REMOTE_ENV = "WEBUI_GITEA_REMOTE"
|
||||
DEFAULT_REMOTE = "prgs"
|
||||
|
||||
|
||||
def _console_remote(env: dict[str, str] | None = None) -> str:
|
||||
env_map = env if env is not None else os.environ
|
||||
return (env_map.get(REMOTE_ENV) or "").strip() or DEFAULT_REMOTE
|
||||
|
||||
|
||||
def load_active_contamination_marker(
|
||||
remote: str | None = None, env: dict[str, str] | None = None
|
||||
) -> dict[str, Any] | None:
|
||||
"""Return the live #630 contamination marker payload, or ``None``.
|
||||
|
||||
The gate is only meaningful when it is fed a real marker: with
|
||||
``marker=None`` :func:`assess_contamination_gate` returns ``block: False``
|
||||
on its first statement. The #641 session inventory already reads the durable
|
||||
markers, so reuse that reader rather than adding a second source of truth.
|
||||
Never raises into a diagnosis or execution path.
|
||||
"""
|
||||
try:
|
||||
from webui import session_loader
|
||||
except Exception: # noqa: BLE001 — never break recovery on an import problem
|
||||
return None
|
||||
try:
|
||||
markers = session_loader._load_contamination_markers(
|
||||
remote=remote or _console_remote(env)
|
||||
)
|
||||
except Exception: # noqa: BLE001 — fail soft; the caller degrades to no marker
|
||||
return None
|
||||
for marker in markers:
|
||||
payload = marker.to_dict()
|
||||
if payload.get("active"):
|
||||
return payload
|
||||
return None
|
||||
|
||||
|
||||
def _active_binding(env_map: Any) -> str | None:
|
||||
"""Read the live worktree binding so a no-op recovery cannot report success."""
|
||||
value = env_map.get(stale_binding_recovery.ACTIVE_WORKTREE_ENV)
|
||||
return value if value else None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RecoveryLedgerEntry:
|
||||
"""One planned recovery step displayed before execution."""
|
||||
|
||||
sequence: int
|
||||
step: str
|
||||
summary: str
|
||||
executes_process_kill: bool = False
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PlaybookDescriptor:
|
||||
"""Structured recovery playbook option returned during diagnosis."""
|
||||
|
||||
playbook_id: str
|
||||
label: str
|
||||
action_id: str
|
||||
description: str
|
||||
eligible: bool
|
||||
requires_confirmation: bool
|
||||
reason: str
|
||||
params_schema: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RecoveryDiagnosis:
|
||||
"""Complete diagnostic snapshot of control-plane recovery needs."""
|
||||
|
||||
status: str
|
||||
clean: bool
|
||||
stale_runtime: dict[str, Any]
|
||||
master_parity: dict[str, Any]
|
||||
stale_binding: dict[str, Any]
|
||||
contamination: dict[str, Any]
|
||||
worktree_anomalies: tuple[str, ...]
|
||||
playbooks: tuple[PlaybookDescriptor, ...]
|
||||
reasons: tuple[str, ...]
|
||||
|
||||
|
||||
def _repo_root(custom_path: Path | str | None = None) -> Path:
|
||||
if custom_path:
|
||||
return Path(custom_path).resolve()
|
||||
override = (os.environ.get("WEBUI_REPO_ROOT") or "").strip()
|
||||
if override:
|
||||
return Path(override).resolve()
|
||||
return Path(__file__).resolve().parent.parent
|
||||
|
||||
|
||||
def confirmation_phrase(playbook_id: str, target: str | None = None) -> str:
|
||||
"""Construct exact confirmation phrase required for a recovery playbook."""
|
||||
clean_target = (target or "").strip()
|
||||
if clean_target:
|
||||
return f"confirm {playbook_id} {clean_target}"
|
||||
return f"confirm {playbook_id}"
|
||||
|
||||
|
||||
def confirmation_matches(
|
||||
playbook_id: str, confirmation: str | None, target: str | None = None
|
||||
) -> bool:
|
||||
expected = confirmation_phrase(playbook_id, target)
|
||||
return str(confirmation or "").strip() == expected
|
||||
|
||||
|
||||
def _build_ledger(
|
||||
playbook_id: str, target: str | None = None
|
||||
) -> tuple[RecoveryLedgerEntry, ...]:
|
||||
if playbook_id == PLAYBOOK_CLEAR_STALE_BINDING:
|
||||
return (
|
||||
RecoveryLedgerEntry(1, "quiesce", "Stop admitting new gated mutations."),
|
||||
RecoveryLedgerEntry(
|
||||
2,
|
||||
"clear_env",
|
||||
f"Remove stale env binding GITEA_ACTIVE_WORKTREE ({target or 'active'}).",
|
||||
),
|
||||
RecoveryLedgerEntry(
|
||||
3, "audit", "Record clear_stale_binding event in console audit log."
|
||||
),
|
||||
RecoveryLedgerEntry(
|
||||
4, "revalidate", "Re-run diagnosis to verify clean binding state."
|
||||
),
|
||||
)
|
||||
if playbook_id == PLAYBOOK_REBIND_SESSION:
|
||||
return (
|
||||
RecoveryLedgerEntry(1, "quiesce", "Stop admitting new gated mutations."),
|
||||
RecoveryLedgerEntry(
|
||||
2,
|
||||
"rebind_worktree",
|
||||
f"Rebind session worktree context safely to {target or 'target worktree'}.",
|
||||
),
|
||||
RecoveryLedgerEntry(
|
||||
3, "audit", "Record rebind_session_worktree event in console audit log."
|
||||
),
|
||||
RecoveryLedgerEntry(
|
||||
4, "revalidate", "Re-run diagnosis to verify worktree binding state."
|
||||
),
|
||||
)
|
||||
if playbook_id == PLAYBOOK_RECONCILE_CLEANUPS:
|
||||
return (
|
||||
RecoveryLedgerEntry(1, "quiesce", "Stop admitting new gated mutations."),
|
||||
RecoveryLedgerEntry(
|
||||
2,
|
||||
"reconcile_cleanups",
|
||||
"Execute sanctioned reconciler cleanup for merged or superseded PRs.",
|
||||
),
|
||||
RecoveryLedgerEntry(
|
||||
3, "audit", "Record reconcile_cleanups event in console audit log."
|
||||
),
|
||||
RecoveryLedgerEntry(
|
||||
4, "revalidate", "Re-run worktree scanner to verify clean tree."
|
||||
),
|
||||
)
|
||||
if playbook_id == PLAYBOOK_SANCTIONED_RESTART:
|
||||
restart_ledger = sanctioned_restart._mutation_ledger(
|
||||
target or "gitea-author", sanctioned_restart.MODE_RESTART
|
||||
)
|
||||
return tuple(
|
||||
RecoveryLedgerEntry(
|
||||
sequence=e.sequence,
|
||||
step=e.step,
|
||||
summary=e.summary,
|
||||
executes_process_kill=e.executes_process_kill,
|
||||
)
|
||||
for e in restart_ledger
|
||||
)
|
||||
return (
|
||||
RecoveryLedgerEntry(1, "unspecified", f"Execute recovery playbook {playbook_id}."),
|
||||
)
|
||||
|
||||
|
||||
def diagnose_recovery(
|
||||
repo_path: Path | str | None = None,
|
||||
env: dict[str, str] | None = None,
|
||||
active_worktree_val: str | None = None,
|
||||
session_lease_wt: str | None = None,
|
||||
role_kind: str | None = None,
|
||||
) -> RecoveryDiagnosis:
|
||||
"""Run full control-plane diagnostics to determine recovery needs and options."""
|
||||
root = _repo_root(repo_path)
|
||||
source_env = dict(env) if env is not None else dict(os.environ)
|
||||
reasons: list[str] = []
|
||||
|
||||
# 1. Stale runtime assessment
|
||||
stale_runtime_obj = system_health.assess_stale_runtime(root)
|
||||
stale_runtime_dict = {
|
||||
"daemon_head": stale_runtime_obj.daemon_head,
|
||||
"checkout_head": stale_runtime_obj.checkout_head,
|
||||
"remote_head": stale_runtime_obj.remote_head,
|
||||
"stale": stale_runtime_obj.stale,
|
||||
"determinable": stale_runtime_obj.determinable,
|
||||
"mutation_safe": stale_runtime_obj.mutation_safe,
|
||||
"reasons": list(stale_runtime_obj.reasons),
|
||||
}
|
||||
if stale_runtime_obj.stale:
|
||||
reasons.append("Runtime HEAD disagrees with checkout/remote HEAD.")
|
||||
|
||||
# 2. Master parity assessment
|
||||
#
|
||||
# The baseline is the commit the *running process* started at, which is what
|
||||
# the parity gate is about. Capturing it from ``checkout_head`` and then
|
||||
# comparing it against that same value made ``in_parity`` structurally
|
||||
# incapable of being false. ``live_remote_head`` restores the #610
|
||||
# live-remote dimension, which was previously dropped.
|
||||
checkout_head = stale_runtime_obj.checkout_head
|
||||
startup_dict = master_parity_gate.capture_startup_parity(
|
||||
str(root), head=stale_runtime_obj.daemon_head
|
||||
)
|
||||
parity_dict = master_parity_gate.assess_master_parity(
|
||||
startup_dict, checkout_head, stale_runtime_obj.remote_head
|
||||
)
|
||||
if not parity_dict.get("in_parity", True):
|
||||
reasons.append("Repository is not in master parity.")
|
||||
|
||||
# 3. Worktree binding classification
|
||||
boot_bindings = stale_binding_recovery.snapshot_boot_bindings(source_env)
|
||||
active_val = (
|
||||
active_worktree_val
|
||||
if active_worktree_val is not None
|
||||
else source_env.get(stale_binding_recovery.ACTIVE_WORKTREE_ENV)
|
||||
)
|
||||
boot_inherited = bool(boot_bindings.get("active_worktree") and active_val == boot_bindings.get("active_worktree"))
|
||||
|
||||
path_exists = None
|
||||
if active_val:
|
||||
path_exists = os.path.exists(os.path.realpath(active_val))
|
||||
|
||||
binding_class = stale_binding_recovery.classify_active_worktree_binding(
|
||||
active_value=active_val,
|
||||
session_lease_worktree=session_lease_wt,
|
||||
boot_inherited=boot_inherited,
|
||||
path_exists=path_exists,
|
||||
role_kind=role_kind,
|
||||
)
|
||||
|
||||
if binding_class.get("clear_eligible"):
|
||||
reasons.append(
|
||||
f"Active worktree binding is stale ({binding_class.get('classification')})."
|
||||
)
|
||||
elif binding_class.get("classification") == stale_binding_recovery.CLASSIFICATION_UNVERIFIED_INHERITED:
|
||||
reasons.append("Inherited worktree binding is unverified.")
|
||||
|
||||
# 4. Contamination assessment (#630)
|
||||
#
|
||||
# A real marker and a task key the gate actually gates on: with marker=None
|
||||
# the gate short-circuits to ``block: False``, and with a console action id
|
||||
# the task is outside CONTAMINATION_GATED_TASKS, so it could never block.
|
||||
contamination_marker = load_active_contamination_marker(env=source_env)
|
||||
contamination_dict = runtime_recovery_guard.assess_contamination_gate(
|
||||
contamination_marker,
|
||||
task=CONTAMINATION_GATED_TASK,
|
||||
actual_role=role_kind,
|
||||
)
|
||||
contaminated = bool(contamination_dict.get("block"))
|
||||
if contaminated:
|
||||
reasons.append("Runtime is contaminated by manual process kill (#630).")
|
||||
|
||||
# 5. Worktree scanner hygiene & anomalies
|
||||
hygiene = worktree_scanner.load_hygiene_snapshot(project_root=str(root))
|
||||
worktree_anomalies = hygiene.anomalies
|
||||
|
||||
# Determine status & eligible playbooks
|
||||
playbooks: list[PlaybookDescriptor] = []
|
||||
|
||||
# Playbook 1: Clear Stale Binding
|
||||
clear_eligible = bool(binding_class.get("clear_eligible"))
|
||||
playbooks.append(
|
||||
PlaybookDescriptor(
|
||||
playbook_id=PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
label="Clear Stale Worktree Binding",
|
||||
action_id=ACTION_CLEAR_STALE_BINDING,
|
||||
description="Clear provably stale or superseded GITEA_ACTIVE_WORKTREE environment binding.",
|
||||
eligible=clear_eligible,
|
||||
requires_confirmation=True,
|
||||
reason=(
|
||||
f"Binding classified as {binding_class.get('classification')}; clear is authorized."
|
||||
if clear_eligible
|
||||
else "Active worktree binding is clean, corroborated, or absent."
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
# Playbook 2: Rebind Session Worktree
|
||||
rebind_eligible = bool(
|
||||
active_val
|
||||
or binding_class.get("classification") == stale_binding_recovery.CLASSIFICATION_UNVERIFIED_INHERITED
|
||||
)
|
||||
playbooks.append(
|
||||
PlaybookDescriptor(
|
||||
playbook_id=PLAYBOOK_REBIND_SESSION,
|
||||
label="Rebind Session Worktree",
|
||||
action_id=ACTION_REBIND_SESSION,
|
||||
description="Rebind or synchronize session worktree binding safely with active lease.",
|
||||
eligible=rebind_eligible,
|
||||
requires_confirmation=True,
|
||||
reason=(
|
||||
"Session worktree binding can be rebound to verified lease worktree."
|
||||
if rebind_eligible
|
||||
else "Session worktree is properly bound."
|
||||
),
|
||||
params_schema={"target_worktree": "string"},
|
||||
)
|
||||
)
|
||||
|
||||
# Playbook 3: Reconcile Cleanups
|
||||
reconcile_eligible = bool(hygiene.anomalies or any(e.classification in {"stale-clean", "detached-review"} for e in hygiene.entries))
|
||||
playbooks.append(
|
||||
PlaybookDescriptor(
|
||||
playbook_id=PLAYBOOK_RECONCILE_CLEANUPS,
|
||||
label="Trigger Reconciler Cleanups",
|
||||
action_id=ACTION_RECONCILE_CLEANUPS,
|
||||
description="Run sanctioned reconciler cleanup preview and apply for merged/superseded PR branches.",
|
||||
eligible=reconcile_eligible,
|
||||
requires_confirmation=True,
|
||||
reason=(
|
||||
f"Worktree hygiene scanner detected {len(hygiene.anomalies)} anomalies and cleanups needed."
|
||||
if reconcile_eligible
|
||||
else "No reconciler cleanups pending."
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
# Playbook 4: Sanctioned Restart
|
||||
restart_eligible = bool(stale_runtime_obj.stale or contaminated)
|
||||
playbooks.append(
|
||||
PlaybookDescriptor(
|
||||
playbook_id=PLAYBOOK_SANCTIONED_RESTART,
|
||||
label="Sanctioned MCP Restart",
|
||||
action_id=sanctioned_restart.ACTION_RESTART_NAMESPACE,
|
||||
description="Restart MCP daemon via configured host supervisor without manual process kill.",
|
||||
eligible=restart_eligible,
|
||||
requires_confirmation=True,
|
||||
reason=(
|
||||
"Stale runtime or contamination detected; host supervisor restart available."
|
||||
if restart_eligible
|
||||
else "Runtime is healthy and clean."
|
||||
),
|
||||
params_schema={"namespace": "string", "mode": "restart|reload"},
|
||||
)
|
||||
)
|
||||
|
||||
clean = not reasons and not contaminated
|
||||
if contaminated:
|
||||
status = STATUS_BLOCKED_CONTAMINATION
|
||||
elif reasons:
|
||||
status = STATUS_ACTION_REQUIRED
|
||||
else:
|
||||
status = STATUS_HEALTHY
|
||||
|
||||
return RecoveryDiagnosis(
|
||||
status=status,
|
||||
clean=clean,
|
||||
stale_runtime=stale_runtime_dict,
|
||||
master_parity=parity_dict,
|
||||
stale_binding=binding_class,
|
||||
contamination=contamination_dict,
|
||||
worktree_anomalies=tuple(worktree_anomalies),
|
||||
playbooks=tuple(playbooks),
|
||||
reasons=tuple(reasons),
|
||||
)
|
||||
|
||||
|
||||
def build_recovery_preview(
|
||||
playbook_id: str,
|
||||
target: str | None = None,
|
||||
params: dict[str, Any] | None = None,
|
||||
principal: console_authz.Principal | None = None,
|
||||
env: dict[str, str] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Generate dry-run preview & mutation ledger for a recovery playbook."""
|
||||
if playbook_id not in KNOWN_PLAYBOOKS:
|
||||
return {
|
||||
"allowed": False,
|
||||
"error": "unknown_playbook",
|
||||
"detail": f"Playbook {playbook_id!r} is not a registered recovery playbook.",
|
||||
}
|
||||
|
||||
action_id = PLAYBOOK_ACTIONS[playbook_id]
|
||||
action = console_authz.get_action(action_id)
|
||||
decision = console_authz.authorize(action_id, principal)
|
||||
# Preview and apply must answer the same question. ``execution_enabled`` was
|
||||
# a hardcoded False beside an authorization decision taken without
|
||||
# ``for_execution``, so the preview could not tell an operator *why*
|
||||
# execution was disabled — and the apply path did not ask at all.
|
||||
execution_decision = console_authz.authorize(
|
||||
action_id, principal, for_execution=True
|
||||
)
|
||||
phrase = confirmation_phrase(playbook_id, target)
|
||||
ledger = _build_ledger(playbook_id, target)
|
||||
|
||||
return {
|
||||
"playbook_id": playbook_id,
|
||||
"action_id": action_id,
|
||||
"target": target,
|
||||
"required_role": action.minimum_role if action else console_authz.OPERATOR,
|
||||
"required_permission": action.mcp_permission if action else "gitea.read",
|
||||
"requires_confirmation": True,
|
||||
"confirmation_phrase": phrase,
|
||||
"mutation_ledger": [asdict(entry) for entry in ledger],
|
||||
"authorization": decision.to_dict(),
|
||||
"execution_authorization": execution_decision.to_dict(),
|
||||
"params": dict(params or {}),
|
||||
"execution_enabled": bool(execution_decision.allowed),
|
||||
"execution_blocked_reason": (
|
||||
None if execution_decision.allowed else execution_decision.reason_code
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def execute_recovery_playbook(
|
||||
playbook_id: str,
|
||||
confirmation: str | None = None,
|
||||
target: str | None = None,
|
||||
params: dict[str, Any] | None = None,
|
||||
principal: console_authz.Principal | None = None,
|
||||
env: dict[str, str] | None = None,
|
||||
request_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Gated execution of a recovery playbook with audit logging and revalidation."""
|
||||
if playbook_id not in KNOWN_PLAYBOOKS:
|
||||
return {
|
||||
"success": False,
|
||||
"allowed": False,
|
||||
"error": "unknown_playbook",
|
||||
"detail": f"Playbook {playbook_id!r} is not known.",
|
||||
}
|
||||
|
||||
action_id = PLAYBOOK_ACTIONS[playbook_id]
|
||||
# The mapping the playbooks actually mutate. ``dict(os.environ)`` produced a
|
||||
# throwaway copy: every env playbook wrote to it, verified against it, and
|
||||
# left the running daemon bound to the value it claimed to have fixed.
|
||||
mutation_env: Any = env if env is not None else os.environ
|
||||
|
||||
# 1. Authorization check — ``for_execution=True`` is what arms the phase
|
||||
# gate (console_authz.authorize only applies it in that branch). Without it
|
||||
# a phase-2 write executed while the console was in phase 1.
|
||||
decision = console_authz.authorize(action_id, principal, for_execution=True)
|
||||
if not decision.allowed:
|
||||
console_audit.record_event(
|
||||
action_id=action_id,
|
||||
result=console_audit.RESULT_DENIED,
|
||||
principal=principal,
|
||||
target={"playbook_id": playbook_id, "target": target},
|
||||
reason_code=decision.reason_code,
|
||||
detail=decision.detail,
|
||||
request_id=request_id,
|
||||
session_id=session_id,
|
||||
)
|
||||
return {
|
||||
"success": False,
|
||||
"allowed": False,
|
||||
"error": decision.reason_code,
|
||||
"detail": decision.detail,
|
||||
}
|
||||
|
||||
# 2. Confirmation phrase check
|
||||
if not confirmation_matches(playbook_id, confirmation, target):
|
||||
expected = confirmation_phrase(playbook_id, target)
|
||||
detail = f"Confirmation phrase mismatch. Expected: {expected!r}"
|
||||
console_audit.record_event(
|
||||
action_id=action_id,
|
||||
result=console_audit.RESULT_DENIED,
|
||||
principal=principal,
|
||||
target={"playbook_id": playbook_id, "target": target},
|
||||
reason_code="confirmation_mismatch",
|
||||
detail=detail,
|
||||
request_id=request_id,
|
||||
session_id=session_id,
|
||||
)
|
||||
return {
|
||||
"success": False,
|
||||
"allowed": False,
|
||||
"error": "confirmation_mismatch",
|
||||
"detail": detail,
|
||||
"expected_confirmation_phrase": expected,
|
||||
}
|
||||
|
||||
# 3. Contamination rule (#630) check
|
||||
role_str = principal.role if principal else None
|
||||
contamination_marker = load_active_contamination_marker(env=env)
|
||||
contam = runtime_recovery_guard.assess_contamination_gate(
|
||||
contamination_marker,
|
||||
task=CONTAMINATION_GATED_TASK,
|
||||
actual_role=role_str,
|
||||
)
|
||||
if contam.get("block"):
|
||||
if playbook_id != PLAYBOOK_RECONCILE_CLEANUPS:
|
||||
detail = "Runtime is contaminated by a manual process kill (#630). Run reconciler cleanup playbook first."
|
||||
console_audit.record_event(
|
||||
action_id=action_id,
|
||||
result=console_audit.RESULT_DENIED,
|
||||
principal=principal,
|
||||
target={"playbook_id": playbook_id, "target": target},
|
||||
reason_code="contaminated_runtime",
|
||||
detail=detail,
|
||||
request_id=request_id,
|
||||
session_id=session_id,
|
||||
)
|
||||
return {
|
||||
"success": False,
|
||||
"allowed": False,
|
||||
"error": "contaminated_runtime",
|
||||
"detail": detail,
|
||||
}
|
||||
|
||||
# 4. Execute playbook action
|
||||
applied_result: dict[str, Any] = {"performed": False}
|
||||
if playbook_id == PLAYBOOK_CLEAR_STALE_BINDING:
|
||||
binding_before = _active_binding(mutation_env)
|
||||
diagnosis = diagnose_recovery(env=env)
|
||||
plan = stale_binding_recovery.plan_recovery(diagnosis.stale_binding)
|
||||
applied_result = stale_binding_recovery.apply_recovery(plan, env=mutation_env)
|
||||
binding_after = _active_binding(mutation_env)
|
||||
applied_result = {
|
||||
**applied_result,
|
||||
"binding_before": binding_before,
|
||||
"binding_after": binding_after,
|
||||
"binding_changed": binding_before != binding_after,
|
||||
}
|
||||
# A clear that did not clear is not a success, whatever the plan said.
|
||||
if not applied_result["binding_changed"]:
|
||||
applied_result["performed"] = False
|
||||
applied_result.setdefault("reasons", []).append(
|
||||
"clear_stale_binding did not change the live worktree binding"
|
||||
)
|
||||
elif playbook_id == PLAYBOOK_REBIND_SESSION:
|
||||
target_wt = target or (params or {}).get("target_worktree")
|
||||
if target_wt:
|
||||
binding_before = _active_binding(mutation_env)
|
||||
mutation_env[stale_binding_recovery.ACTIVE_WORKTREE_ENV] = target_wt
|
||||
binding_after = _active_binding(mutation_env)
|
||||
applied_result = {
|
||||
"performed": binding_after == target_wt,
|
||||
"rebound_worktree": target_wt,
|
||||
"binding_before": binding_before,
|
||||
"binding_after": binding_after,
|
||||
"binding_changed": binding_before != binding_after,
|
||||
"cleared_stale": binding_before != binding_after,
|
||||
}
|
||||
if binding_after != target_wt:
|
||||
applied_result["reasons"] = [
|
||||
"rebind_session_worktree did not take effect on the live "
|
||||
"environment"
|
||||
]
|
||||
else:
|
||||
applied_result = {
|
||||
"performed": False,
|
||||
"reason": "No target_worktree specified for rebind.",
|
||||
}
|
||||
elif playbook_id == PLAYBOOK_RECONCILE_CLEANUPS:
|
||||
# ``merged_cleanup_reconcile`` exposes the building blocks only; the
|
||||
# orchestrator is the MCP tool. The previous call named a function that
|
||||
# does not exist, and a bare ``except`` turned the AttributeError into a
|
||||
# generic failure, so this playbook could never succeed. Imported lazily
|
||||
# because the MCP server module is large and binds FastMCP at import.
|
||||
try:
|
||||
import gitea_mcp_server
|
||||
|
||||
snapshot = gitea_mcp_server.gitea_reconcile_merged_cleanups(
|
||||
dry_run=False,
|
||||
execute_confirmed=True,
|
||||
remote=(params or {}).get("remote") or _console_remote(env),
|
||||
org=(params or {}).get("org"),
|
||||
repo=(params or {}).get("repo"),
|
||||
)
|
||||
performed_reconcile = bool(snapshot.get("success"))
|
||||
applied_result = {
|
||||
"performed": performed_reconcile,
|
||||
"reconciled_count": len(snapshot.get("entries") or []),
|
||||
"snapshot": snapshot,
|
||||
}
|
||||
if not performed_reconcile:
|
||||
applied_result["reasons"] = list(snapshot.get("reasons") or [])
|
||||
except Exception as exc: # noqa: BLE001 — surfaced with its type
|
||||
applied_result = {
|
||||
"performed": False,
|
||||
"error": str(exc),
|
||||
"error_type": type(exc).__name__,
|
||||
}
|
||||
elif playbook_id == PLAYBOOK_SANCTIONED_RESTART:
|
||||
ns = target or (params or {}).get("namespace", "gitea-author")
|
||||
md = (params or {}).get("mode", sanctioned_restart.MODE_RESTART)
|
||||
restart_res = sanctioned_restart.execute_restart(
|
||||
namespace=ns,
|
||||
mode=md,
|
||||
principal=principal,
|
||||
confirmation=f"{md} {ns}",
|
||||
# Without the marker the stricter guard at sanctioned_restart.py:375
|
||||
# never fires and a restart can launder a contaminated runtime.
|
||||
contamination_marker=contamination_marker,
|
||||
env=mutation_env,
|
||||
request_id=request_id,
|
||||
session_id=session_id,
|
||||
)
|
||||
applied_result = restart_res
|
||||
|
||||
# ``allowed`` is not ``performed``: execute_restart documents that success is
|
||||
# False in both directions because the host supervisor still has to act.
|
||||
performed = bool(applied_result.get("performed"))
|
||||
|
||||
# 5. Record Audit Log
|
||||
audit_record = console_audit.record_event(
|
||||
action_id=action_id,
|
||||
result=console_audit.RESULT_ALLOWED if performed else console_audit.RESULT_DENIED,
|
||||
principal=principal,
|
||||
target={"playbook_id": playbook_id, "target": target},
|
||||
reason_code="recovery_executed" if performed else "recovery_failed",
|
||||
detail=f"Executed recovery playbook {playbook_id}",
|
||||
request_id=request_id,
|
||||
session_id=session_id,
|
||||
metadata={"applied_result": applied_result},
|
||||
)
|
||||
|
||||
# 6. Post-recovery verification recheck.
|
||||
#
|
||||
# Re-read state rather than re-reading the mapping the mutation just wrote:
|
||||
# verifying the mutated copy confirmed changes that never reached the
|
||||
# process. Passing ``env`` through means a caller-supplied mapping is the
|
||||
# live one for that caller, and ``None`` re-reads ``os.environ`` fresh.
|
||||
post_verification = verify_post_recovery(env=env)
|
||||
|
||||
return {
|
||||
"success": performed,
|
||||
"allowed": True,
|
||||
"playbook_id": playbook_id,
|
||||
"action_id": action_id,
|
||||
"applied_result": applied_result,
|
||||
"audit": audit_record,
|
||||
"post_recovery_verification": post_verification,
|
||||
}
|
||||
|
||||
|
||||
def verify_post_recovery(
|
||||
repo_path: Path | str | None = None, env: dict[str, str] | None = None
|
||||
) -> dict[str, Any]:
|
||||
"""Revalidate control-plane state post-recovery before clean status."""
|
||||
diag = diagnose_recovery(repo_path, env)
|
||||
classification = diag.stale_binding.get("classification")
|
||||
# ``not clear_eligible`` also reads clean for every binding recovery is not
|
||||
# allowed to touch — an unverified inherited binding is unproven, not clean.
|
||||
binding_clean = (
|
||||
not diag.stale_binding.get("clear_eligible")
|
||||
and classification != stale_binding_recovery.CLASSIFICATION_UNVERIFIED_INHERITED
|
||||
)
|
||||
return {
|
||||
"clean": diag.clean,
|
||||
"status": diag.status,
|
||||
"stale_runtime_clean": not diag.stale_runtime.get("stale"),
|
||||
"binding_clean": binding_clean,
|
||||
"binding_classification": classification,
|
||||
# The gate returns ``block``; it has never returned ``contaminated``, so
|
||||
# reading that key reported every runtime clean unconditionally.
|
||||
"contamination_clean": not diag.contamination.get("block"),
|
||||
"anomalies_count": len(diag.worktree_anomalies),
|
||||
"reasons": list(diag.reasons),
|
||||
}
|
||||
@@ -178,6 +178,13 @@ def build_action_registry() -> ActionRegistry:
|
||||
("system.restart_namespace", "Restart MCP namespace",
|
||||
"restart_namespace", "host.supervisor_restart",
|
||||
"Restart one MCP namespace via the host supervisor."),
|
||||
# #644: Phase 2 recovery playbooks & controls.
|
||||
("system.clear_stale_binding", "Clear stale binding", "clear_stale_binding",
|
||||
"console.clear_stale_binding", "Clear provably stale or superseded env binding."),
|
||||
("system.rebind_session_worktree", "Rebind session worktree", "rebind_session_worktree",
|
||||
"console.rebind_session_worktree", "Rebind session worktree to verified lease."),
|
||||
("system.reconcile_cleanups", "Reconcile cleanups", "reconcile_cleanups",
|
||||
"console.reconcile_cleanups", "Run reconciler cleanup for merged or superseded PRs."),
|
||||
)
|
||||
actions = tuple(
|
||||
GatedAction(
|
||||
|
||||
@@ -46,9 +46,11 @@ NAV_GROUPS: tuple[NavGroup, ...] = (
|
||||
NavItem("/queue", "Queue"),
|
||||
NavItem("/leases", "Leases"),
|
||||
NavItem("/actions", "Actions"),
|
||||
NavItem("/requests", "Requests"),
|
||||
)),
|
||||
NavGroup("Runtime/Sessions", (
|
||||
NavItem("/runtime", "Runtime health"),
|
||||
NavItem("/runtime/restart", "Restart status"),
|
||||
NavItem("/sessions", "Sessions"),
|
||||
)),
|
||||
NavGroup("Projects", (
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,164 @@
|
||||
"""HTML views for the operator request surface (#643).
|
||||
|
||||
The form is deliberately a *preview* form. It has no initiate button, because
|
||||
initiating requires a confirmed POST to ``/api/v1/requests/apply`` and a stray
|
||||
form submission must not be able to produce one by accident.
|
||||
|
||||
Nothing rendered here is trusted input: every interpolated value is escaped,
|
||||
and the page renders only values the service already produced rather than
|
||||
echoing a raw request body back.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
from webui.layout import render_page
|
||||
from webui.request_service import (
|
||||
REQUESTABLE_ROLES,
|
||||
WORK_KINDS,
|
||||
RequestError,
|
||||
RequestPreview,
|
||||
)
|
||||
|
||||
REQUESTS_PATH = "/requests"
|
||||
PREVIEW_API_PATH = "/api/v1/requests/preview"
|
||||
APPLY_API_PATH = "/api/v1/requests/apply"
|
||||
|
||||
|
||||
def _escape(text: Any) -> str:
|
||||
return html.escape(str(text if text is not None else ""), quote=True)
|
||||
|
||||
|
||||
REQUEST_PAGE_STYLES = """
|
||||
<style>
|
||||
.request-form { display: grid; gap: 0.75rem; max-width: 44rem; }
|
||||
.request-form label { display: grid; gap: 0.25rem; font-size: 0.9rem; }
|
||||
.request-check { margin: 0.35rem 0; }
|
||||
.request-check .verdict-ok { color: var(--accent); }
|
||||
.request-check .verdict-fail { color: #d14; }
|
||||
.request-prohibited code { margin-right: 0.4rem; }
|
||||
</style>
|
||||
"""
|
||||
|
||||
|
||||
def _options(values: tuple[str, ...], selected: Any) -> str:
|
||||
return "".join(
|
||||
f"<option value='{_escape(value)}'"
|
||||
+ (" selected" if selected == value else "")
|
||||
+ f">{_escape(value)}</option>"
|
||||
for value in values
|
||||
)
|
||||
|
||||
|
||||
def _form(values: dict[str, Any] | None = None) -> str:
|
||||
current = dict(values or {})
|
||||
number = current.get("work_number")
|
||||
return (
|
||||
f"<form class='request-form' method='post' action='{REQUESTS_PATH}'>"
|
||||
"<label>Desired role<select name='desired_role'>"
|
||||
f"{_options(REQUESTABLE_ROLES, current.get('desired_role'))}"
|
||||
"</select></label>"
|
||||
"<label>Work kind<select name='work_kind'>"
|
||||
f"{_options(WORK_KINDS, current.get('work_kind'))}"
|
||||
"</select></label>"
|
||||
"<label>Issue or PR number"
|
||||
"<input type='number' name='work_number' min='1' required "
|
||||
f"value='{_escape(number) if number else ''}'></label>"
|
||||
"<label>Intent summary"
|
||||
"<input type='text' name='intent_summary' maxlength='500' required "
|
||||
f"value='{_escape(current.get('intent_summary'))}'></label>"
|
||||
"<label>Expected head SHA <span class='muted'>(PR work only)</span>"
|
||||
"<input type='text' name='expected_head_sha' "
|
||||
f"value='{_escape(current.get('expected_head_sha'))}'></label>"
|
||||
"<button type='submit' class='copy-btn'>Preview request</button>"
|
||||
"<p class='muted meta'>Preview is read-only and creates no assignment. "
|
||||
f"Initiating requires a confirmed POST to <code>{APPLY_API_PATH}</code>."
|
||||
"</p>"
|
||||
"</form>"
|
||||
)
|
||||
|
||||
|
||||
def _checks_block(preview: RequestPreview) -> str:
|
||||
rows = []
|
||||
for check in preview.checks:
|
||||
verdict = "PASS" if check.ok else "FAIL"
|
||||
css = "verdict-ok" if check.ok else "verdict-fail"
|
||||
rows.append(
|
||||
"<li class='request-check'>"
|
||||
f"<span class='{css}'><strong>{verdict}</strong></span> "
|
||||
f"<code>{_escape(check.name)}</code> — {_escape(check.detail)} "
|
||||
f"<span class='muted meta'>({_escape(check.reason_code)})</span>"
|
||||
"</li>"
|
||||
)
|
||||
return "<ul>" + "".join(rows) + "</ul>"
|
||||
|
||||
|
||||
def _preview_block(preview: RequestPreview) -> str:
|
||||
verdict = "AUTHORIZED" if preview.authorized else "DENIED"
|
||||
prohibited = "".join(
|
||||
f"<code>{_escape(action)}</code>" for action in preview.prohibited_actions
|
||||
)
|
||||
request = preview.request
|
||||
evidence = json.dumps(preview.allocator_evidence, indent=2, default=str)
|
||||
return (
|
||||
"<h3>Intent preview</h3>"
|
||||
f"<p><strong>{verdict}</strong> — {_escape(preview.detail)}</p>"
|
||||
"<p class='meta'>"
|
||||
f"Role <code>{_escape(request.desired_role)}</code> · "
|
||||
f"{_escape(request.work_kind)} <code>{_escape(request.display_ref)}</code>"
|
||||
f" · profile <code>{_escape(preview.required_profile)}</code> · "
|
||||
f"namespace <code>{_escape(preview.required_namespace)}</code> · "
|
||||
f"permission <code>{_escape(preview.required_permission)}</code>"
|
||||
"</p>"
|
||||
f"<p>Intent: {_escape(request.intent_summary)}</p>"
|
||||
f"{_checks_block(preview)}"
|
||||
f"<p><strong>Next safe action:</strong> "
|
||||
f"{_escape(preview.next_safe_action)}</p>"
|
||||
"<p class='request-prohibited'><strong>Prohibited for this role:</strong> "
|
||||
+ (prohibited or "<span class='muted'>none declared</span>")
|
||||
+ "</p>"
|
||||
"<p class='muted meta'>Correlation id "
|
||||
f"<code>{_escape(preview.correlation_id)}</code></p>"
|
||||
"<details><summary>Allocator evidence</summary>"
|
||||
f"<pre class='prompt-text'>{_escape(evidence)}</pre>"
|
||||
"</details>"
|
||||
)
|
||||
|
||||
|
||||
def _error_block(error: RequestError) -> str:
|
||||
field = (
|
||||
f"<p class='meta'>Field: <code>{_escape(error.field_name)}</code></p>"
|
||||
if error.field_name
|
||||
else ""
|
||||
)
|
||||
return (
|
||||
"<h3>Request rejected</h3>"
|
||||
f"<p><strong>{_escape(error.reason_code)}</strong> — "
|
||||
f"{_escape(error.detail)}</p>{field}"
|
||||
)
|
||||
|
||||
|
||||
def render_requests_page(
|
||||
*,
|
||||
preview: RequestPreview | None = None,
|
||||
error: RequestError | None = None,
|
||||
submitted: dict[str, Any] | None = None,
|
||||
) -> str:
|
||||
"""Render the request form, plus a preview or rejection when one exists."""
|
||||
body = (
|
||||
"<h2>Requests</h2>"
|
||||
"<p>Submit a work request — desired role, issue or PR, and intent — "
|
||||
"and see whether it would be authorized before anything is reserved. "
|
||||
"Initiation goes through the allocator (#600/#613); this console never "
|
||||
"self-selects work, never approves, and never merges.</p>"
|
||||
+ _form(submitted)
|
||||
+ (_error_block(error) if error is not None else "")
|
||||
+ (_preview_block(preview) if preview is not None else "")
|
||||
+ f"<p class='meta'><a href='{PREVIEW_API_PATH}'>Preview API</a> · "
|
||||
"<a href='/api/console/security-model'>RBAC model</a></p>"
|
||||
+ REQUEST_PAGE_STYLES
|
||||
)
|
||||
return render_page(title="Requests", body_html=body)
|
||||
@@ -0,0 +1,579 @@
|
||||
"""Read-only restart status, impact preview, and approval state (#667).
|
||||
|
||||
Phase 1 of the console restart surface. It *consumes* the #655 coordinator
|
||||
substrate and renders it; it never restarts, reloads, drains, approves, or kills
|
||||
anything. There is no apply path in this module, so there is no execution gate
|
||||
here to arm incorrectly — the only writes the console could perform are the ones
|
||||
it does not implement.
|
||||
|
||||
Sources, each independently fail-soft and each reported with its own
|
||||
:class:`SourceStatus`:
|
||||
|
||||
* :mod:`restart_coordinator` — restart-class policy matrix (#663) and the
|
||||
blast-radius impact report (#658).
|
||||
* :mod:`drain_proof` — drain checklist and gate verdict (#661), verified
|
||||
read-only against a caller-supplied proof.
|
||||
* :mod:`post_restart_reconcile` — post-restart completion proof (#662).
|
||||
* :mod:`webui.console_authz` — role authorization for the approval controls
|
||||
(#633).
|
||||
|
||||
Three rules this module holds itself to, because a status surface that lies is
|
||||
worse than one that is absent:
|
||||
|
||||
**A source that could not be read is reported unavailable, never green.** No
|
||||
default, placeholder, or self-comparison is substituted for a reading that
|
||||
failed. An unreadable control-plane DB yields ``inventory_complete=False``,
|
||||
which the coordinator itself turns into a fail-closed verdict.
|
||||
|
||||
**Authorization is asked the way execution would ask it.** Every authorization
|
||||
probe passes ``for_execution=True``, so the console reports whether the action
|
||||
could actually run rather than the weaker "this principal is the right role".
|
||||
While the console is in Phase 1 that answer is ``phase_not_active`` for every
|
||||
phase-2 action, and the surface says so plainly instead of showing an allow.
|
||||
|
||||
**The database is opened read-only.** ``ControlPlaneDB()`` creates directories
|
||||
and runs migrations on construction, which is a write; this module opens the
|
||||
sqlite file with ``mode=ro`` exactly as :mod:`webui.inventory` does, and treats
|
||||
a missing file as missing authority rather than an empty inventory.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sqlite3
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Callable, Mapping
|
||||
|
||||
import control_plane_db
|
||||
import drain_proof
|
||||
import restart_coordinator
|
||||
from webui import console_authz
|
||||
from webui.inventory import redact_path, scrub
|
||||
|
||||
# --- Source status ----------------------------------------------------------
|
||||
|
||||
STATUS_OK = "ok"
|
||||
STATUS_UNAVAILABLE = "unavailable"
|
||||
|
||||
#: Console actions whose authorization state this surface reports. Both are
|
||||
#: pre-existing #642 actions; this module adds no new console action because it
|
||||
#: performs no console action.
|
||||
REPORTED_ACTIONS: tuple[str, ...] = (
|
||||
"system.restart_namespace",
|
||||
"system.reload_namespace",
|
||||
)
|
||||
|
||||
#: The break-glass workflow (#664) is not consumed here. It is declared so the
|
||||
#: surface is honest about the gap rather than silently omitting a governance
|
||||
#: path the operator has been told exists.
|
||||
BREAK_GLASS_ISSUE = 664
|
||||
BREAK_GLASS_PENDING_REASON = (
|
||||
"The break-glass workflow (#664) is not yet available on this branch's "
|
||||
"base; no break-glass control is offered and none is implied."
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SourceStatus:
|
||||
"""Whether one backing source could be read, and why not when it could not."""
|
||||
|
||||
name: str
|
||||
status: str
|
||||
detail: str = ""
|
||||
|
||||
@property
|
||||
def available(self) -> bool:
|
||||
return self.status == STATUS_OK
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"name": self.name,
|
||||
"status": self.status,
|
||||
"available": self.available,
|
||||
"detail": self.detail,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RestartClassView:
|
||||
"""One row of the #663 restart-class matrix, scoped to the viewer's role."""
|
||||
|
||||
restart_class: str
|
||||
required_permission: str
|
||||
expected_blast_radius: str
|
||||
drain_requirement: str
|
||||
full_drain_required: bool
|
||||
approval_requirement: str
|
||||
request_roles: tuple[str, ...]
|
||||
execution_roles: tuple[str, ...]
|
||||
viewer_may_request: bool
|
||||
viewer_may_execute: bool
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"restart_class": self.restart_class,
|
||||
"required_permission": self.required_permission,
|
||||
"expected_blast_radius": self.expected_blast_radius,
|
||||
"drain_requirement": self.drain_requirement,
|
||||
"full_drain_required": self.full_drain_required,
|
||||
"approval_requirement": self.approval_requirement,
|
||||
"request_roles": list(self.request_roles),
|
||||
"execution_roles": list(self.execution_roles),
|
||||
"viewer_may_request": self.viewer_may_request,
|
||||
"viewer_may_execute": self.viewer_may_execute,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ActionAuthorization:
|
||||
"""Authorization state for one console action, asked as execution would."""
|
||||
|
||||
action_id: str
|
||||
summary: str
|
||||
required_role: str
|
||||
allowed: bool
|
||||
execution_enabled: bool
|
||||
reason_code: str
|
||||
detail: str
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"action_id": self.action_id,
|
||||
"summary": self.summary,
|
||||
"required_role": self.required_role,
|
||||
"allowed": self.allowed,
|
||||
"execution_enabled": self.execution_enabled,
|
||||
"reason_code": self.reason_code,
|
||||
"detail": self.detail,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BreakGlassSurface:
|
||||
"""Declared-but-unavailable break-glass panel (#664 is not on this base)."""
|
||||
|
||||
available: bool
|
||||
issue: int
|
||||
reason: str
|
||||
viewer_is_privileged: bool
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"available": self.available,
|
||||
"issue": self.issue,
|
||||
"reason": self.reason,
|
||||
"viewer_is_privileged": self.viewer_is_privileged,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RestartConsoleSnapshot:
|
||||
"""Everything the read-only restart console renders."""
|
||||
|
||||
generated_at: str
|
||||
viewer_role: str
|
||||
viewer_authenticated: bool
|
||||
read_only: bool
|
||||
impact: dict[str, Any] | None
|
||||
impact_source: SourceStatus
|
||||
drain: dict[str, Any] | None
|
||||
drain_source: SourceStatus
|
||||
reconcile: dict[str, Any] | None
|
||||
reconcile_source: SourceStatus
|
||||
restart_classes: tuple[RestartClassView, ...]
|
||||
authorizations: tuple[ActionAuthorization, ...]
|
||||
break_glass: BreakGlassSurface
|
||||
notes: tuple[str, ...] = field(default_factory=tuple)
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"generated_at": self.generated_at,
|
||||
"viewer_role": self.viewer_role,
|
||||
"viewer_authenticated": self.viewer_authenticated,
|
||||
"read_only": self.read_only,
|
||||
"impact": self.impact,
|
||||
"impact_source": self.impact_source.as_dict(),
|
||||
"drain": self.drain,
|
||||
"drain_source": self.drain_source.as_dict(),
|
||||
"reconcile": self.reconcile,
|
||||
"reconcile_source": self.reconcile_source.as_dict(),
|
||||
"restart_classes": [c.as_dict() for c in self.restart_classes],
|
||||
"authorizations": [a.as_dict() for a in self.authorizations],
|
||||
"break_glass": self.break_glass.as_dict(),
|
||||
"notes": list(self.notes),
|
||||
"links": {
|
||||
"issue": 667,
|
||||
"extends": 642,
|
||||
"umbrella": 655,
|
||||
"coordinator": 658,
|
||||
"drain_proof": 661,
|
||||
"reconcile": 662,
|
||||
"restart_classes": 663,
|
||||
"break_glass": BREAK_GLASS_ISSUE,
|
||||
"vision": 652,
|
||||
"roadmap": 653,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
# --- Control-plane inventory (read-only) ------------------------------------
|
||||
|
||||
|
||||
def read_control_plane_inventory(
|
||||
*,
|
||||
db_path: str | None = None,
|
||||
limit: int = 200,
|
||||
) -> dict[str, Any]:
|
||||
"""Read sessions and leases for an impact evaluation, read-only.
|
||||
|
||||
Returns the inventory mapping
|
||||
:func:`restart_coordinator.evaluate_restart_impact` expects.
|
||||
``inventory_complete`` is True only when every read succeeded, so a partial
|
||||
read denies rather than under-reporting the blast radius.
|
||||
|
||||
The database is never created, migrated, or written: a missing file means
|
||||
the console has no session authority, which is not the same as there being
|
||||
no sessions.
|
||||
"""
|
||||
|
||||
path = (db_path or control_plane_db.default_db_path() or "").strip()
|
||||
incomplete: list[str] = []
|
||||
|
||||
def _incomplete(reason: str) -> dict[str, Any]:
|
||||
return {
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"terminal_lock": None,
|
||||
"prior_recovery_attempts": [],
|
||||
"inventory_complete": False,
|
||||
"incomplete_reasons": [reason],
|
||||
}
|
||||
|
||||
if not path:
|
||||
return _incomplete("control-plane database path is not configured")
|
||||
if not os.path.exists(path):
|
||||
return _incomplete(
|
||||
f"control-plane database not present at {redact_path(path)}; "
|
||||
"no session or lease authority available"
|
||||
)
|
||||
|
||||
try:
|
||||
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=5)
|
||||
conn.row_factory = sqlite3.Row
|
||||
except sqlite3.Error as exc:
|
||||
return _incomplete(f"control-plane database could not be opened: {exc}")
|
||||
|
||||
sessions: list[dict[str, Any]] = []
|
||||
leases: list[dict[str, Any]] = []
|
||||
capped = max(1, int(limit))
|
||||
try:
|
||||
tables = {
|
||||
str(row[0])
|
||||
for row in conn.execute(
|
||||
"SELECT name FROM sqlite_master WHERE type = 'table'"
|
||||
).fetchall()
|
||||
}
|
||||
if "sessions" not in tables:
|
||||
incomplete.append("control-plane database has no sessions table")
|
||||
else:
|
||||
sessions = [
|
||||
dict(row)
|
||||
for row in conn.execute(
|
||||
"SELECT session_id, role, profile, pid, status,"
|
||||
" last_heartbeat_at FROM sessions"
|
||||
" WHERE status = 'active'"
|
||||
" ORDER BY last_heartbeat_at DESC LIMIT ?",
|
||||
(capped,),
|
||||
).fetchall()
|
||||
]
|
||||
|
||||
if "leases" not in tables:
|
||||
incomplete.append("control-plane database has no leases table")
|
||||
elif "work_items" not in tables:
|
||||
incomplete.append(
|
||||
"control-plane database has no work_items table; lease work "
|
||||
"identity cannot be resolved"
|
||||
)
|
||||
else:
|
||||
leases = [
|
||||
dict(row)
|
||||
for row in conn.execute(
|
||||
"SELECT l.lease_id, l.session_id, l.role, l.phase,"
|
||||
" l.status AS freshness, l.worktree_path,"
|
||||
" w.kind AS work_kind, w.number AS work_number"
|
||||
" FROM leases l"
|
||||
" JOIN work_items w ON w.work_item_id = l.work_item_id"
|
||||
" WHERE l.status = 'active'"
|
||||
" ORDER BY l.expires_at DESC LIMIT ?",
|
||||
(capped,),
|
||||
).fetchall()
|
||||
]
|
||||
except sqlite3.Error as exc:
|
||||
return _incomplete(f"control-plane database read failed: {exc}")
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
return {
|
||||
"sessions": sessions,
|
||||
"leases": leases,
|
||||
"terminal_lock": None,
|
||||
"prior_recovery_attempts": [],
|
||||
"inventory_complete": not incomplete,
|
||||
"incomplete_reasons": incomplete,
|
||||
}
|
||||
|
||||
|
||||
# --- Composition ------------------------------------------------------------
|
||||
|
||||
|
||||
def build_restart_class_views(viewer_role: str | None) -> tuple[RestartClassView, ...]:
|
||||
"""Render the #663 class matrix, marking what this viewer may request."""
|
||||
|
||||
normalized = str(viewer_role or "").strip().lower()
|
||||
views: list[RestartClassView] = []
|
||||
for policy in restart_coordinator.RESTART_CLASS_POLICIES.values():
|
||||
views.append(
|
||||
RestartClassView(
|
||||
restart_class=policy.restart_class.value,
|
||||
required_permission=policy.required_permission,
|
||||
expected_blast_radius=policy.expected_blast_radius,
|
||||
drain_requirement=policy.drain_requirement,
|
||||
full_drain_required=policy.full_drain_required,
|
||||
approval_requirement=policy.approval_requirement,
|
||||
request_roles=tuple(policy.request_roles),
|
||||
execution_roles=tuple(policy.execution_roles),
|
||||
viewer_may_request=normalized in policy.request_roles,
|
||||
viewer_may_execute=normalized in policy.execution_roles,
|
||||
)
|
||||
)
|
||||
return tuple(views)
|
||||
|
||||
|
||||
def build_action_authorizations(
|
||||
principal: console_authz.Principal | None,
|
||||
) -> tuple[ActionAuthorization, ...]:
|
||||
"""Authorization state for the approval controls, asked as execution.
|
||||
|
||||
``for_execution=True`` is deliberate. Asking without it answers "is this
|
||||
principal senior enough", which is not the question an operator looking at a
|
||||
control needs answered; asking with it answers "would this run", and while
|
||||
the console is in Phase 1 the honest answer is no.
|
||||
"""
|
||||
|
||||
results: list[ActionAuthorization] = []
|
||||
for action_id in REPORTED_ACTIONS:
|
||||
action = console_authz.get_action(action_id)
|
||||
decision = console_authz.authorize(action_id, principal, for_execution=True)
|
||||
results.append(
|
||||
ActionAuthorization(
|
||||
action_id=action_id,
|
||||
summary=action.summary if action else "",
|
||||
required_role=(
|
||||
action.minimum_role if action else console_authz.OPERATOR
|
||||
),
|
||||
allowed=bool(decision.allowed),
|
||||
execution_enabled=bool(decision.execution_enabled),
|
||||
reason_code=str(decision.reason_code or ""),
|
||||
detail=str(decision.detail or ""),
|
||||
)
|
||||
)
|
||||
return tuple(results)
|
||||
|
||||
|
||||
def viewer_is_privileged(principal: console_authz.Principal | None) -> bool:
|
||||
"""True when the viewer holds at least the operator role."""
|
||||
|
||||
who = principal if principal is not None else console_authz.ANONYMOUS
|
||||
if not who.authenticated:
|
||||
return False
|
||||
return who.rank >= console_authz.ROLE_ORDER.index(console_authz.OPERATOR)
|
||||
|
||||
|
||||
def load_impact_report(
|
||||
*,
|
||||
principal: console_authz.Principal | None = None,
|
||||
restart_class: str = restart_coordinator.RestartClass.FULL_MCP_RESTART.value,
|
||||
db_path: str | None = None,
|
||||
limit: int = 200,
|
||||
read_inventory: Callable[..., Mapping[str, Any]] | None = None,
|
||||
now: datetime | None = None,
|
||||
) -> tuple[dict[str, Any] | None, SourceStatus]:
|
||||
"""Evaluate the blast radius for *restart_class*, always dry-run."""
|
||||
|
||||
reader = read_inventory or read_control_plane_inventory
|
||||
try:
|
||||
inventory = dict(reader(db_path=db_path, limit=limit))
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return None, SourceStatus(
|
||||
"impact",
|
||||
STATUS_UNAVAILABLE,
|
||||
f"control-plane inventory failed: {type(exc).__name__}: {exc}",
|
||||
)
|
||||
|
||||
who = principal if principal is not None else console_authz.ANONYMOUS
|
||||
viewer_role = str(who.role or "").strip().lower()
|
||||
try:
|
||||
report = restart_coordinator.evaluate_restart_impact(
|
||||
inventory,
|
||||
now=now,
|
||||
dry_run=True,
|
||||
restart_class=restart_class,
|
||||
requester_role=viewer_role,
|
||||
requester_permissions=restart_coordinator.permissions_for_role(
|
||||
viewer_role
|
||||
),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return None, SourceStatus(
|
||||
"impact",
|
||||
STATUS_UNAVAILABLE,
|
||||
f"impact evaluation failed: {type(exc).__name__}: {exc}",
|
||||
)
|
||||
|
||||
payload = scrub(report.as_dict())
|
||||
detail = ""
|
||||
if not report.inventory_complete:
|
||||
detail = "; ".join(report.incomplete_reasons) or "inventory incomplete"
|
||||
return payload, SourceStatus("impact", STATUS_OK, detail)
|
||||
|
||||
|
||||
def load_drain_status(
|
||||
*,
|
||||
proof: Mapping[str, Any] | None = None,
|
||||
now: datetime | None = None,
|
||||
expected_impact_fingerprint: str | None = None,
|
||||
) -> tuple[dict[str, Any] | None, SourceStatus]:
|
||||
"""Verify a supplied drain proof read-only and report the verdict.
|
||||
|
||||
No proof supplied is not a failure and not a pass: it is reported as the
|
||||
absence of a proof, which is exactly what the #661 gate would deny on.
|
||||
"""
|
||||
|
||||
if proof is None:
|
||||
return None, SourceStatus(
|
||||
"drain",
|
||||
STATUS_UNAVAILABLE,
|
||||
"no drain proof supplied; the #661 gate denies a restart without a "
|
||||
"valid unexpired clean proof",
|
||||
)
|
||||
try:
|
||||
verified = drain_proof.verify_drain_proof(
|
||||
proof,
|
||||
now=now,
|
||||
expected_impact_fingerprint=expected_impact_fingerprint,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return None, SourceStatus(
|
||||
"drain",
|
||||
STATUS_UNAVAILABLE,
|
||||
f"drain proof verification failed: {type(exc).__name__}: {exc}",
|
||||
)
|
||||
return scrub(verified.as_dict()), SourceStatus("drain", STATUS_OK)
|
||||
|
||||
|
||||
def load_reconcile_status(
|
||||
*,
|
||||
load_proof: Callable[[], Any] | None = None,
|
||||
) -> tuple[dict[str, Any] | None, SourceStatus]:
|
||||
"""Report the most recent post-restart completion proof (#662)."""
|
||||
|
||||
if load_proof is None:
|
||||
return None, SourceStatus(
|
||||
"reconcile",
|
||||
STATUS_UNAVAILABLE,
|
||||
"no post-restart completion proof source is wired into this view",
|
||||
)
|
||||
try:
|
||||
proof = load_proof()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return None, SourceStatus(
|
||||
"reconcile",
|
||||
STATUS_UNAVAILABLE,
|
||||
f"reconcile proof unavailable: {type(exc).__name__}: {exc}",
|
||||
)
|
||||
if proof is None:
|
||||
return None, SourceStatus(
|
||||
"reconcile",
|
||||
STATUS_UNAVAILABLE,
|
||||
"no post-restart reconcile has been recorded",
|
||||
)
|
||||
payload = proof.as_dict() if hasattr(proof, "as_dict") else dict(proof)
|
||||
return scrub(payload), SourceStatus("reconcile", STATUS_OK)
|
||||
|
||||
|
||||
def load_restart_console_snapshot(
|
||||
*,
|
||||
principal: console_authz.Principal | None = None,
|
||||
restart_class: str = restart_coordinator.RestartClass.FULL_MCP_RESTART.value,
|
||||
db_path: str | None = None,
|
||||
limit: int = 200,
|
||||
drain_proof_payload: Mapping[str, Any] | None = None,
|
||||
read_inventory: Callable[..., Mapping[str, Any]] | None = None,
|
||||
load_reconcile_proof: Callable[[], Any] | None = None,
|
||||
now: datetime | None = None,
|
||||
) -> RestartConsoleSnapshot:
|
||||
"""Compose the read-only restart console snapshot."""
|
||||
|
||||
who = principal if principal is not None else console_authz.ANONYMOUS
|
||||
moment = now or _utc_now()
|
||||
|
||||
impact, impact_source = load_impact_report(
|
||||
principal=who,
|
||||
restart_class=restart_class,
|
||||
db_path=db_path,
|
||||
limit=limit,
|
||||
read_inventory=read_inventory,
|
||||
now=moment,
|
||||
)
|
||||
fingerprint = None
|
||||
if impact is not None:
|
||||
try:
|
||||
fingerprint = drain_proof.impact_fingerprint(impact)
|
||||
except Exception: # noqa: BLE001
|
||||
fingerprint = None
|
||||
|
||||
drain, drain_source = load_drain_status(
|
||||
proof=drain_proof_payload,
|
||||
now=moment,
|
||||
expected_impact_fingerprint=fingerprint,
|
||||
)
|
||||
reconcile, reconcile_source = load_reconcile_status(
|
||||
load_proof=load_reconcile_proof
|
||||
)
|
||||
|
||||
notes: list[str] = [
|
||||
"This surface is read-only: it evaluates and displays, and performs no "
|
||||
"restart, reload, drain, approval, or process action.",
|
||||
]
|
||||
if not impact_source.available:
|
||||
notes.append(
|
||||
"Impact preview unavailable — a restart decision must not be made "
|
||||
"from this page while the blast radius is unknown."
|
||||
)
|
||||
|
||||
return RestartConsoleSnapshot(
|
||||
generated_at=moment.isoformat(),
|
||||
viewer_role=str(who.role or "anonymous"),
|
||||
viewer_authenticated=bool(who.authenticated),
|
||||
read_only=True,
|
||||
impact=impact,
|
||||
impact_source=impact_source,
|
||||
drain=drain,
|
||||
drain_source=drain_source,
|
||||
reconcile=reconcile,
|
||||
reconcile_source=reconcile_source,
|
||||
restart_classes=build_restart_class_views(who.role),
|
||||
authorizations=build_action_authorizations(who),
|
||||
break_glass=BreakGlassSurface(
|
||||
available=False,
|
||||
issue=BREAK_GLASS_ISSUE,
|
||||
reason=BREAK_GLASS_PENDING_REASON,
|
||||
viewer_is_privileged=viewer_is_privileged(who),
|
||||
),
|
||||
notes=tuple(notes),
|
||||
)
|
||||
@@ -0,0 +1,299 @@
|
||||
"""HTML views for the read-only restart console (#667).
|
||||
|
||||
Every interpolated value passes through :func:`_esc`. Values that can carry a
|
||||
filesystem path or free-form operator text additionally pass through
|
||||
:func:`webui.inventory.scrub_text`, which redacts credential-shaped tokens
|
||||
*inside* a string rather than only at its start.
|
||||
|
||||
The page renders state and never offers a control that would mutate anything:
|
||||
the approval and break-glass panels report authorization and availability, and
|
||||
there is no form, button, or endpoint behind them.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
|
||||
from webui.inventory import scrub_text
|
||||
from webui.restart_console import RestartConsoleSnapshot, SourceStatus
|
||||
|
||||
|
||||
def _esc(value: object) -> str:
|
||||
"""Escape any value for HTML text or a quoted attribute."""
|
||||
if value is None:
|
||||
return ""
|
||||
return html.escape(str(value), quote=True)
|
||||
|
||||
|
||||
def _esc_text(value: object) -> str:
|
||||
"""Escape free-form text after redacting secrets embedded inside it."""
|
||||
if value is None:
|
||||
return ""
|
||||
return _esc(scrub_text(str(value)))
|
||||
|
||||
|
||||
def _bool_badge(
|
||||
value: bool, *, true_label: str = "yes", false_label: str = "no"
|
||||
) -> str:
|
||||
css = "badge-ok" if value else "badge-blocked"
|
||||
label = true_label if value else false_label
|
||||
return f'<span class="badge {css}">{_esc(label)}</span>'
|
||||
|
||||
|
||||
def _source_badge(source: SourceStatus) -> str:
|
||||
css = "badge-ok" if source.available else "badge-blocked"
|
||||
badge = f'<span class="badge {css}">{_esc(source.status)}</span>'
|
||||
if source.detail:
|
||||
badge += f' <span class="muted">{_esc_text(source.detail)}</span>'
|
||||
return badge
|
||||
|
||||
|
||||
def _notes_block(snapshot: RestartConsoleSnapshot) -> str:
|
||||
if not snapshot.notes:
|
||||
return ""
|
||||
items = "".join(f"<li>{_esc_text(note)}</li>" for note in snapshot.notes)
|
||||
return f"<ul class='reasons'>{items}</ul>"
|
||||
|
||||
|
||||
def _impact_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
head = (
|
||||
"<section class='health-card'>"
|
||||
f"<h3>Impact preview {_source_badge(snapshot.impact_source)}</h3>"
|
||||
)
|
||||
impact = snapshot.impact
|
||||
if impact is None:
|
||||
return (
|
||||
head
|
||||
+ "<p class='muted'>No impact preview is available, so the blast "
|
||||
"radius of a restart is unknown. Treat this as unsafe.</p></section>"
|
||||
)
|
||||
|
||||
counts = impact.get("counts") or {}
|
||||
verdict = str(impact.get("verdict") or "unknown")
|
||||
verdict_css = "badge-ok" if verdict == "safe" else "badge-blocked"
|
||||
rows = "".join(
|
||||
f"<tr><th>{_esc(key.replace('_', ' '))}</th><td>{_esc(value)}</td></tr>"
|
||||
for key, value in sorted(counts.items())
|
||||
)
|
||||
reasons = "".join(
|
||||
f"<li>{_esc_text(reason)}</li>" for reason in (impact.get("reasons") or [])
|
||||
)
|
||||
incomplete = ""
|
||||
if not impact.get("inventory_complete", False):
|
||||
detail = "; ".join(str(r) for r in (impact.get("incomplete_reasons") or []))
|
||||
incomplete = (
|
||||
"<p class='error'><strong>Inventory incomplete:</strong> "
|
||||
f"{_esc_text(detail or 'unspecified')}. The coordinator fails "
|
||||
"closed on an incomplete inventory.</p>"
|
||||
)
|
||||
|
||||
sessions = impact.get("affected_sessions") or []
|
||||
session_rows = "".join(
|
||||
"<tr>"
|
||||
f"<td><code>{_esc(s.get('session_id'))}</code></td>"
|
||||
f"<td>{_esc(s.get('role'))}</td>"
|
||||
f"<td>{_esc(s.get('pid'))}</td>"
|
||||
f"<td>{_bool_badge(bool(s.get('live')), true_label='live', false_label='idle')}</td>"
|
||||
f"<td>{_bool_badge(not s.get('heartbeat_stale'), true_label='fresh', false_label='stale')}</td>"
|
||||
"</tr>"
|
||||
for s in sessions[:50]
|
||||
)
|
||||
session_table = (
|
||||
"<h4>Sessions a restart would terminate</h4>"
|
||||
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||
"<th>Session</th><th>Role</th><th>PID</th><th>State</th>"
|
||||
"<th>Heartbeat</th></tr></thead><tbody>"
|
||||
f"{session_rows}</tbody></table></div>"
|
||||
if session_rows
|
||||
else "<p class='muted'>No affected sessions reported.</p>"
|
||||
)
|
||||
truncated = (
|
||||
f"<p class='muted'>Showing the first 50 of {_esc(len(sessions))} "
|
||||
"affected sessions.</p>"
|
||||
if len(sessions) > 50
|
||||
else ""
|
||||
)
|
||||
|
||||
return (
|
||||
head
|
||||
+ "<p class='health-headline'>Verdict "
|
||||
f"<span class='badge {verdict_css}'>{_esc(verdict)}</span> · "
|
||||
f"blast radius <code>{_esc(impact.get('blast_radius'))}</code> · "
|
||||
f"class <code>{_esc(impact.get('restart_class'))}</code></p>"
|
||||
+ incomplete
|
||||
+ (f"<ul class='reasons'>{reasons}</ul>" if reasons else "")
|
||||
+ (f"<table class='registry'><tbody>{rows}</tbody></table>" if rows else "")
|
||||
+ session_table
|
||||
+ truncated
|
||||
+ "</section>"
|
||||
)
|
||||
|
||||
|
||||
def _drain_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
head = (
|
||||
"<section class='health-card'>"
|
||||
f"<h3>Drain proof {_source_badge(snapshot.drain_source)}</h3>"
|
||||
)
|
||||
drain = snapshot.drain
|
||||
if drain is None:
|
||||
return (
|
||||
head
|
||||
+ "<p class='muted'>No drain proof has been presented to this view. "
|
||||
"The #661 gate authorizes a restart only against a valid, unexpired, "
|
||||
"clean proof, so the absence of one is a denial, not a pass.</p>"
|
||||
"</section>"
|
||||
)
|
||||
reasons = "".join(
|
||||
f"<li>{_esc_text(reason)}</li>" for reason in (drain.get("reasons") or [])
|
||||
)
|
||||
return (
|
||||
head
|
||||
+ "<table class='registry'><tbody>"
|
||||
f"<tr><th>Valid</th><td>{_bool_badge(bool(drain.get('valid')))}</td></tr>"
|
||||
f"<tr><th>Clean</th><td>{_bool_badge(bool(drain.get('clean')))}</td></tr>"
|
||||
f"<tr><th>Expired</th><td>{_bool_badge(not drain.get('expired'), true_label='no', false_label='yes')}</td></tr>"
|
||||
f"<tr><th>Tampered</th><td>{_bool_badge(not drain.get('tampered'), true_label='no', false_label='yes')}</td></tr>"
|
||||
f"<tr><th>Proof id</th><td><code>{_esc(drain.get('proof_id'))}</code></td></tr>"
|
||||
"</tbody></table>"
|
||||
+ (f"<ul class='reasons'>{reasons}</ul>" if reasons else "")
|
||||
+ "</section>"
|
||||
)
|
||||
|
||||
|
||||
def _reconcile_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
head = (
|
||||
"<section class='health-card'>"
|
||||
f"<h3>Post-restart reconcile {_source_badge(snapshot.reconcile_source)}</h3>"
|
||||
)
|
||||
proof = snapshot.reconcile
|
||||
if proof is None:
|
||||
return (
|
||||
head
|
||||
+ "<p class='muted'>No post-restart completion proof is recorded. "
|
||||
"Until one is, the last restart's recovery state is unproven.</p>"
|
||||
"</section>"
|
||||
)
|
||||
items = "".join(
|
||||
"<tr>"
|
||||
f"<td>{_esc(item.get('dimension'))}</td>"
|
||||
f"<td>{_esc(item.get('status'))}</td>"
|
||||
f"<td>{_esc_text(item.get('summary'))}</td>"
|
||||
f"<td>{_bool_badge(not item.get('follow_up_required'), true_label='no', false_label='yes')}</td>"
|
||||
"</tr>"
|
||||
for item in (proof.get("items") or [])
|
||||
)
|
||||
return (
|
||||
head
|
||||
+ "<p class='health-headline'>Status "
|
||||
f"<code>{_esc(proof.get('overall_status'))}</code> · mode "
|
||||
f"<code>{_esc(proof.get('mode'))}</code> · resolved "
|
||||
f"{_esc(proof.get('resolved_count'))} · unresolved "
|
||||
f"{_esc(proof.get('unresolved_count'))}</p>"
|
||||
+ (
|
||||
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||
"<th>Dimension</th><th>Status</th><th>Summary</th>"
|
||||
"<th>Follow-up required</th></tr></thead><tbody>"
|
||||
f"{items}</tbody></table></div>"
|
||||
if items
|
||||
else "<p class='muted'>No reconcile dimensions reported.</p>"
|
||||
)
|
||||
+ "</section>"
|
||||
)
|
||||
|
||||
|
||||
def _class_matrix_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
rows = "".join(
|
||||
"<tr>"
|
||||
f"<td><code>{_esc(view.restart_class)}</code></td>"
|
||||
f"<td><code>{_esc(view.required_permission)}</code></td>"
|
||||
f"<td>{_esc(view.expected_blast_radius)}</td>"
|
||||
f"<td>{_esc(view.drain_requirement)}</td>"
|
||||
f"<td>{_esc(view.approval_requirement)}</td>"
|
||||
f"<td>{_bool_badge(view.viewer_may_request)}</td>"
|
||||
f"<td>{_bool_badge(view.viewer_may_execute)}</td>"
|
||||
"</tr>"
|
||||
for view in snapshot.restart_classes
|
||||
)
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Restart classes</h3>"
|
||||
"<p class='muted'>The least-privilege matrix each restart request is "
|
||||
"resolved against. “You may request” and “you may "
|
||||
"execute” are computed for the current viewer role, not for a "
|
||||
"generic operator.</p>"
|
||||
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||
"<th>Class</th><th>Permission</th><th>Blast radius</th>"
|
||||
"<th>Drain</th><th>Approval</th><th>You may request</th>"
|
||||
"<th>You may execute</th></tr></thead><tbody>"
|
||||
f"{rows}</tbody></table></div>"
|
||||
"</section>"
|
||||
)
|
||||
|
||||
|
||||
def _approval_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
rows = "".join(
|
||||
"<tr>"
|
||||
f"<td><code>{_esc(a.action_id)}</code></td>"
|
||||
f"<td>{_esc(a.required_role)}</td>"
|
||||
f"<td>{_bool_badge(a.allowed)}</td>"
|
||||
f"<td>{_bool_badge(a.execution_enabled)}</td>"
|
||||
f"<td><code>{_esc(a.reason_code)}</code></td>"
|
||||
f"<td>{_esc_text(a.detail)}</td>"
|
||||
"</tr>"
|
||||
for a in snapshot.authorizations
|
||||
)
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Approval controls</h3>"
|
||||
"<p class='muted'>Authorization is probed the way execution would probe "
|
||||
"it, so “execution enabled” answers whether the action would "
|
||||
"actually run — not merely whether this role outranks the requirement. "
|
||||
"No control on this page performs the action.</p>"
|
||||
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||
"<th>Action</th><th>Required role</th><th>Authorized</th>"
|
||||
"<th>Execution enabled</th><th>Reason</th><th>Detail</th>"
|
||||
"</tr></thead><tbody>"
|
||||
f"{rows}</tbody></table></div>"
|
||||
"</section>"
|
||||
)
|
||||
|
||||
|
||||
def _break_glass_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
bg = snapshot.break_glass
|
||||
if not bg.viewer_is_privileged:
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Break-glass</h3>"
|
||||
"<p class='muted'>Break-glass status is visible to operator-class "
|
||||
"roles only. Your role does not carry that authority, so no "
|
||||
"emergency surface is shown.</p>"
|
||||
"</section>"
|
||||
)
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Break-glass "
|
||||
f"{_bool_badge(bg.available, true_label='available', false_label='unavailable')}"
|
||||
"</h3>"
|
||||
f"<p class='muted'>{_esc_text(bg.reason)}</p>"
|
||||
f"<p class='meta'>Tracked by issue #{_esc(bg.issue)}.</p>"
|
||||
"</section>"
|
||||
)
|
||||
|
||||
|
||||
def render_restart_console_page(snapshot: RestartConsoleSnapshot) -> str:
|
||||
"""Render the whole read-only restart console body."""
|
||||
|
||||
return (
|
||||
"<h2>Restart status and impact</h2>"
|
||||
f"<p class='meta'>Generated <code>{_esc(snapshot.generated_at)}</code> · "
|
||||
f"viewer role <code>{_esc(snapshot.viewer_role)}</code> · "
|
||||
f"authenticated {_bool_badge(snapshot.viewer_authenticated)} · "
|
||||
f"read-only {_bool_badge(snapshot.read_only)}</p>"
|
||||
+ _notes_block(snapshot)
|
||||
+ _impact_section(snapshot)
|
||||
+ _drain_section(snapshot)
|
||||
+ _reconcile_section(snapshot)
|
||||
+ _class_matrix_section(snapshot)
|
||||
+ _approval_section(snapshot)
|
||||
+ _break_glass_section(snapshot)
|
||||
)
|
||||
@@ -270,26 +270,53 @@ def _probe_error_card(snapshot: SystemHealthSnapshot) -> str:
|
||||
|
||||
|
||||
def _recovery_card() -> str:
|
||||
"""Sanctioned recovery pointers only — never a manual process kill (#630)."""
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Recovery</h3>"
|
||||
"<p class='muted'>This dashboard is read-only. Restart and reload "
|
||||
"controls arrive in Phase 2 (#642); until then recovery runs through "
|
||||
"the sanctioned client reconnect / operator restart path.</p>"
|
||||
"<ul class='reasons'>"
|
||||
"<li><a href='/runtime'>Runtime health</a> — active profile, workflow "
|
||||
"hashes, and shell health.</li>"
|
||||
"<li><a href='/sessions'>Runtime and sessions</a> — namespaces, session "
|
||||
"rows, worktree bindings, and contamination markers (#641).</li>"
|
||||
"<li>Reconnect the MCP client from the IDE, then re-run the blocked "
|
||||
"cycle. Never kill the daemon process manually: unmanaged kills are "
|
||||
"recorded as runtime contamination (#630).</li>"
|
||||
"<li>See <code>docs/webui-local-dev.md</code> for the documented "
|
||||
"recovery sequence.</li>"
|
||||
"</ul>"
|
||||
"</section>"
|
||||
)
|
||||
"""Sanctioned recovery controls & playbooks (#644, Phase 2)."""
|
||||
try:
|
||||
from webui import console_recovery
|
||||
diag = console_recovery.diagnose_recovery()
|
||||
# Every other card in this file escapes at the interpolation boundary.
|
||||
# This one did not, and it is where a #630 marker's operator-supplied
|
||||
# command_summary lands once the contamination gate is wired.
|
||||
status_badge = (
|
||||
f"<span class='status-pill {_esc(diag.status)}'>{_esc(diag.status)}</span>"
|
||||
)
|
||||
playbook_lis = ""
|
||||
for pb in diag.playbooks:
|
||||
elig = "eligible" if pb.eligible else "disabled"
|
||||
playbook_lis += (
|
||||
f"<li><strong>{_esc(pb.label)}</strong> "
|
||||
f"(<code>{_esc(pb.playbook_id)}</code>) — "
|
||||
f"<span class='badge {elig}'>{elig}</span>: {_esc(pb.description)} "
|
||||
f"<em class='muted'>({_esc(pb.reason)})</em></li>"
|
||||
)
|
||||
reasons_html = ""
|
||||
if diag.reasons:
|
||||
items = "".join(f"<li>{_esc(r)}</li>" for r in diag.reasons)
|
||||
reasons_html = f"<ul class='reasons'>{items}</ul>"
|
||||
else:
|
||||
reasons_html = "<p class='clean-note'>No recovery actions currently required. Control plane is healthy.</p>"
|
||||
|
||||
return (
|
||||
"<section class='health-card recovery-card'>"
|
||||
f"<h3>Sanctioned Recovery Controls (Phase 2 #644) {status_badge}</h3>"
|
||||
"<p class='muted'>Guided recovery wizard: Diagnose → Preview → Confirm → Verify. "
|
||||
"Reconnect the MCP client from the IDE, then re-run the blocked cycle. "
|
||||
"Never kill the daemon process manually: unmanaged kills are recorded as runtime contamination (#630).</p>"
|
||||
f"{reasons_html}"
|
||||
"<h4>Available Recovery Playbooks</h4>"
|
||||
f"<ul class='playbooks-list'>{playbook_lis}</ul>"
|
||||
"<p class='meta'>APIs: <code>/api/v1/system/recovery/diagnose</code>, "
|
||||
"<code>/api/v1/system/recovery/preview</code>, <code>/api/v1/system/recovery/apply</code>, "
|
||||
"<code>/api/v1/system/recovery/verify</code>.</p>"
|
||||
"</section>"
|
||||
)
|
||||
except Exception as exc:
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Sanctioned Recovery Controls (Phase 2 #644)</h3>"
|
||||
f"<p class='error'>Recovery diagnostics unavailable: {_esc(exc)}</p>"
|
||||
"</section>"
|
||||
)
|
||||
|
||||
|
||||
def render_system_health_page(snapshot: SystemHealthSnapshot) -> str:
|
||||
|
||||
@@ -201,6 +201,16 @@ def _candidates_from_queue_snapshot(q_snap: QueueSnapshot) -> list[WorkCandidate
|
||||
return candidates
|
||||
|
||||
|
||||
def candidates_from_queue_snapshot(q_snap: QueueSnapshot) -> list[WorkCandidate]:
|
||||
"""Public alias for :func:`_candidates_from_queue_snapshot` (#643).
|
||||
|
||||
The request-initiation service ranks the same candidate set this view
|
||||
renders, so both must agree on how a queue row becomes a candidate. One
|
||||
construction, two callers — not two that can drift apart.
|
||||
"""
|
||||
return _candidates_from_queue_snapshot(q_snap)
|
||||
|
||||
|
||||
def _claim_lease_records(inventory: dict[str, Any] | None) -> list[dict[str, Any]]:
|
||||
"""Normalize ``build_claim_inventory`` entries into lease records.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user