Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f0c6255d7d | ||
|
|
d7ad2838ec | ||
|
|
c6d68dbc7b | ||
|
|
c83a10d7c2 | ||
|
|
ca22c326a4 | ||
|
|
3bbe6df6c7 | ||
|
|
71031c812e | ||
|
|
e43ddd3cbe | ||
|
|
a64ba08e27 | ||
|
|
26f54851d1 | ||
|
|
f02a2dc030 | ||
|
|
77d808e7d4 | ||
|
|
bb8c3a537b | ||
|
|
461e1dac78 | ||
|
|
6010f4295b | ||
|
|
9b8e315b49 | ||
|
|
9a01543477 | ||
|
|
e91b94db56 | ||
|
|
4f06d30e07 |
@@ -26,6 +26,7 @@ from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
import maintenance_drain
|
||||
from control_plane_db import (
|
||||
ControlPlaneDB,
|
||||
ControlPlaneError,
|
||||
@@ -936,6 +937,46 @@ def allocate_next_work(
|
||||
"allocation_mode": (allocation_mode or "").strip() or None,
|
||||
}
|
||||
|
||||
# #659 AC2: while maintenance drain is active, no new work is assigned —
|
||||
# for dry-run and apply alike, so a preview can never be read as evidence
|
||||
# that work was assignable during the drain. Checked before session
|
||||
# registration so a drained allocator leaves no new state behind.
|
||||
try:
|
||||
drain_record = db.read_maintenance_drain(remote=remote, org=org, repo=repo)
|
||||
except Exception as exc: # noqa: BLE001 — unreadable drain state fails closed
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"reasons": [
|
||||
f"maintenance-drain state lookup failed: {exc} (fail closed, #659)"
|
||||
],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
|
||||
drain_decision = maintenance_drain.classify_assignment(drain_record)
|
||||
if not drain_decision["assignment_allowed"]:
|
||||
return {
|
||||
"success": True,
|
||||
"outcome": OUTCOME_WAIT,
|
||||
"apply": apply,
|
||||
"role": role_norm,
|
||||
"allocation_mode": mode,
|
||||
"remote": remote,
|
||||
"org": org,
|
||||
"repo": repo,
|
||||
"selected": None,
|
||||
"reasons": list(drain_decision["reasons"]),
|
||||
"reason_code": drain_decision["reason_code"],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
"maintenance_drain": maintenance_drain.status_payload(
|
||||
drain_record, remote=remote, org=org, repo=repo
|
||||
),
|
||||
}
|
||||
|
||||
# A side-effect-free run may never reserve: reserving is a write, and the
|
||||
# flag is the caller's assertion that this call writes nothing (#643).
|
||||
if side_effect_free and apply:
|
||||
|
||||
+194
-1
@@ -31,8 +31,9 @@ from typing import Any, Iterator, Sequence
|
||||
|
||||
import dependency_graph
|
||||
import gitea_audit
|
||||
import maintenance_drain
|
||||
|
||||
SCHEMA_VERSION = 5
|
||||
SCHEMA_VERSION = 6
|
||||
|
||||
# Assignable work kinds only — raw monitoring incidents are never work items.
|
||||
WORK_KINDS = frozenset({"issue", "pr"})
|
||||
@@ -239,6 +240,31 @@ CREATE INDEX IF NOT EXISTS idx_session_checkpoints_session
|
||||
CREATE INDEX IF NOT EXISTS idx_session_checkpoints_work
|
||||
ON session_checkpoints(remote, org, repo, work_kind, work_number);
|
||||
|
||||
-- Graceful maintenance-drain state (#659). One current row per repository
|
||||
-- scope — drain is a *state*, not a history, so entering and exiting update
|
||||
-- the same row and every transition is audited to ``events``. Creating the
|
||||
-- table is the v5->v6 migration: additive, idempotent, and it never touches
|
||||
-- prior tables. ``state`` is CHECK-constrained so an unknown value can never
|
||||
-- be written and later read as "not draining".
|
||||
CREATE TABLE IF NOT EXISTS maintenance_drain (
|
||||
drain_id TEXT PRIMARY KEY,
|
||||
remote TEXT NOT NULL,
|
||||
org TEXT NOT NULL,
|
||||
repo TEXT NOT NULL,
|
||||
state TEXT NOT NULL DEFAULT 'inactive'
|
||||
CHECK (state IN ('inactive', 'draining')),
|
||||
reason TEXT NOT NULL DEFAULT '',
|
||||
requested_by TEXT NOT NULL DEFAULT '',
|
||||
requested_by_profile TEXT NOT NULL DEFAULT '',
|
||||
session_id TEXT NOT NULL DEFAULT '',
|
||||
entered_at TEXT NOT NULL DEFAULT '',
|
||||
exited_at TEXT NOT NULL DEFAULT '',
|
||||
drain_schema_version INTEGER NOT NULL DEFAULT 6,
|
||||
created_at TEXT NOT NULL,
|
||||
updated_at TEXT NOT NULL,
|
||||
UNIQUE (remote, org, repo)
|
||||
);
|
||||
|
||||
-- Model usage, token cost, latency, and performance events (#651)
|
||||
CREATE TABLE IF NOT EXISTS usage_events (
|
||||
usage_id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
@@ -3025,3 +3051,170 @@ class ControlPlaneDB:
|
||||
"live_lease_id": None if live_lease_id is None else str(live_lease_id),
|
||||
"reconcile_action": "reconcile_required" if stale else "safe_to_resume",
|
||||
}
|
||||
|
||||
# ── Maintenance drain (#659) ─────────────────────────────────────────────
|
||||
|
||||
@staticmethod
|
||||
def _maintenance_drain_row(row: sqlite3.Row | None) -> dict[str, Any] | None:
|
||||
"""Convert a ``maintenance_drain`` row to a plain record."""
|
||||
if row is None:
|
||||
return None
|
||||
return {key: row[key] for key in row.keys()}
|
||||
|
||||
def read_maintenance_drain(
|
||||
self, *, remote: str, org: str, repo: str
|
||||
) -> dict[str, Any] | None:
|
||||
"""Return the current drain record for a scope, or None if never set.
|
||||
|
||||
None and a stored ``inactive`` row mean the same thing to callers —
|
||||
``maintenance_drain.is_draining`` treats both as not draining — so the
|
||||
read never has to invent a record to answer the gate.
|
||||
"""
|
||||
with self._tx(immediate=False) as conn:
|
||||
row = conn.execute(
|
||||
"""
|
||||
SELECT * FROM maintenance_drain
|
||||
WHERE remote = ? AND org = ? AND repo = ?
|
||||
""",
|
||||
(str(remote or ""), str(org or ""), str(repo or "")),
|
||||
).fetchone()
|
||||
return self._maintenance_drain_row(row)
|
||||
|
||||
def set_maintenance_drain(
|
||||
self,
|
||||
*,
|
||||
remote: str,
|
||||
org: str,
|
||||
repo: str,
|
||||
state: str,
|
||||
reason: str = "",
|
||||
requested_by: str = "",
|
||||
requested_by_profile: str = "",
|
||||
session_id: str = "",
|
||||
) -> dict[str, Any]:
|
||||
"""Enter or exit maintenance drain for one repository scope (AC1).
|
||||
|
||||
The state transition is audited to ``events`` — entering and exiting
|
||||
are exactly the moments an operator has to be able to reconstruct
|
||||
later. Re-entering an already-draining scope is idempotent: it refreshes
|
||||
the reason/owner metadata, keeps the original ``entered_at``, and
|
||||
records no duplicate transition event.
|
||||
|
||||
Capability authorization happens above this layer (the drain tasks
|
||||
carry a non-``gitea.*`` permission in the task capability map); the DB
|
||||
records who asked and why, and never grants the right itself.
|
||||
"""
|
||||
state_norm = maintenance_drain.normalize_state(state)
|
||||
raw = {
|
||||
"reason": str(reason or ""),
|
||||
"requested_by": str(requested_by or ""),
|
||||
"requested_by_profile": str(requested_by_profile or ""),
|
||||
"session_id": str(session_id or ""),
|
||||
}
|
||||
clean = gitea_audit.redact(raw)
|
||||
remote_s, org_s, repo_s = str(remote or ""), str(org or ""), str(repo or "")
|
||||
now_s = _ts()
|
||||
|
||||
with self._tx() as conn:
|
||||
existing = conn.execute(
|
||||
"""
|
||||
SELECT * FROM maintenance_drain
|
||||
WHERE remote = ? AND org = ? AND repo = ?
|
||||
""",
|
||||
(remote_s, org_s, repo_s),
|
||||
).fetchone()
|
||||
|
||||
prior_state = (
|
||||
maintenance_drain.normalize_state(existing["state"])
|
||||
if existing is not None
|
||||
else maintenance_drain.STATE_INACTIVE
|
||||
)
|
||||
transitioned = prior_state != state_norm
|
||||
|
||||
prior_entered = (
|
||||
str(existing["entered_at"] or "") if existing is not None else ""
|
||||
)
|
||||
prior_exited = (
|
||||
str(existing["exited_at"] or "") if existing is not None else ""
|
||||
)
|
||||
if state_norm == maintenance_drain.STATE_DRAINING:
|
||||
# A re-entry keeps the original entry time (the drain never
|
||||
# stopped); a fresh entry stamps now and clears the old exit.
|
||||
entered_at = prior_entered if (not transitioned and prior_entered) else now_s
|
||||
exited_at = ""
|
||||
else:
|
||||
entered_at = prior_entered
|
||||
exited_at = now_s if (transitioned or not prior_exited) else prior_exited
|
||||
|
||||
if existing is None:
|
||||
drain_id = uuid.uuid4().hex
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO maintenance_drain(
|
||||
drain_id, remote, org, repo, state, reason,
|
||||
requested_by, requested_by_profile, session_id,
|
||||
entered_at, exited_at, drain_schema_version,
|
||||
created_at, updated_at
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
drain_id, remote_s, org_s, repo_s, state_norm,
|
||||
clean["reason"], clean["requested_by"],
|
||||
clean["requested_by_profile"], clean["session_id"],
|
||||
entered_at, exited_at,
|
||||
maintenance_drain.DRAIN_SCHEMA_VERSION, now_s, now_s,
|
||||
),
|
||||
)
|
||||
else:
|
||||
drain_id = str(existing["drain_id"])
|
||||
conn.execute(
|
||||
"""
|
||||
UPDATE maintenance_drain
|
||||
SET state = ?, reason = ?, requested_by = ?,
|
||||
requested_by_profile = ?, session_id = ?,
|
||||
entered_at = ?, exited_at = ?,
|
||||
drain_schema_version = ?, updated_at = ?
|
||||
WHERE drain_id = ?
|
||||
""",
|
||||
(
|
||||
state_norm, clean["reason"], clean["requested_by"],
|
||||
clean["requested_by_profile"], clean["session_id"],
|
||||
entered_at, exited_at,
|
||||
maintenance_drain.DRAIN_SCHEMA_VERSION, now_s, drain_id,
|
||||
),
|
||||
)
|
||||
|
||||
if transitioned:
|
||||
event_type = (
|
||||
"maintenance_drain_enter"
|
||||
if state_norm == maintenance_drain.STATE_DRAINING
|
||||
else "maintenance_drain_exit"
|
||||
)
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO events(work_item_id, event_type, message, created_at)
|
||||
VALUES (NULL, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
event_type,
|
||||
f"drain {drain_id} scope {remote_s}/{org_s}/{repo_s} "
|
||||
f"{prior_state} -> {state_norm} by "
|
||||
f"{clean['requested_by'] or '(unknown)'} "
|
||||
f"({clean['requested_by_profile'] or 'no profile'}); "
|
||||
f"reason: {clean['reason'] or '(none)'}",
|
||||
now_s,
|
||||
),
|
||||
)
|
||||
|
||||
row = conn.execute(
|
||||
"SELECT * FROM maintenance_drain WHERE drain_id = ?", (drain_id,)
|
||||
).fetchone()
|
||||
|
||||
record = self._maintenance_drain_row(row) or {}
|
||||
return {
|
||||
"record": record,
|
||||
"drain_id": drain_id,
|
||||
"state": state_norm,
|
||||
"prior_state": prior_state,
|
||||
"transitioned": transitioned,
|
||||
}
|
||||
|
||||
@@ -0,0 +1,167 @@
|
||||
# ADR: High-availability and rolling-restart architecture for Gitea MCP control plane
|
||||
|
||||
- **Status:** Proposed (Design ADR under [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668))
|
||||
- **Date:** 2026-07-25
|
||||
- **Tracking Issue:** [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668)
|
||||
- **Policy Version:** `mcp-ha-rolling-restart/v1`
|
||||
- **Related:**
|
||||
- Parent: [#655](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/655) — Governed MCP restart coordination and zero-disruption recovery
|
||||
- Governance Policy: [#656](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/656) / `docs/architecture/mcp-restart-governance.md`
|
||||
- Control-Plane DB Substrate: [#613](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/613) / `docs/architecture/control-plane-db-substrate.md`
|
||||
- Runtime Policy: [#615](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/615) / `docs/architecture/mcp-stable-control-runtime-policy-adr.md`
|
||||
- Product Vision: [#652](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/652) (Phase 5 Maturity)
|
||||
- Delivery Roadmap: [#653](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/653)
|
||||
|
||||
---
|
||||
|
||||
## 1. Context & Problem Statement
|
||||
|
||||
The Gitea MCP server operates as the authoritative **control plane** for managing issues, Pull Requests, code mutations, formal reviews, and workflow reconciliations. Under single-process governance ([#656](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/656)), process restarts are strictly controlled using pre-flight checks, drain phases, and operator approvals.
|
||||
|
||||
However, a single-instance control plane inherently presents fundamental constraints:
|
||||
|
||||
1. **Downtime during updates:** Even a perfectly executed single-process drain requires a window where incoming client requests must be paused or rejected while the server binary or python environment reloads.
|
||||
2. **Single point of failure:** Infrastructure issues, process crashes, or unhandled host-level terminations immediately disconnect active LLM sessions and leave transient workflows incomplete.
|
||||
3. **Multi-agent concurrency bottlenecks:** High volumes of concurrent multi-LLM tasks put all lock management, lease allocation, and Gitea API interactions through a single process event loop.
|
||||
|
||||
To achieve true zero-disruption operation and seamless rolling deployments without stopping active work, the system requires a high-availability (HA), multi-instance MCP architecture.
|
||||
|
||||
---
|
||||
|
||||
## 2. Architectural Principles & Non-Goals
|
||||
|
||||
### 2.1 Core Architectural Principles
|
||||
* **Gitea as Canonical Work SoT:** Gitea remains the ultimate System of Record (SoT) for issue states, pull requests, labels, and audit comments. The MCP control plane does not duplicate domain entities.
|
||||
* **Control-Plane DB as Multi-Instance State Substrate:** The control-plane SQLite/durable database ([#613](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/613)) acts as the single source of truth for workflow leases, session tokens, assignment records, and lock fences across all MCP nodes.
|
||||
* **Stateless Worker Nodes:** MCP role server processes (`gitea-author`, `gitea-reviewer`, `gitea-merger`, `gitea-reconciler`, `gitea-controller`) maintain no unique in-memory state; any node can handle any request given a valid session resume token.
|
||||
* **Fail-Closed Split-Brain Defense:** In any network partition or quorum loss scenario, nodes must fail closed rather than risk double-mutations or conflicting Gitea states.
|
||||
|
||||
### 2.2 Non-Goals
|
||||
* **Replacing Gitea:** We do not replace Gitea issue/PR tracking with an independent database.
|
||||
* **Immediate Multi-Node Cluster Execution in v1:** This ADR defines the target architecture and phased roadmap; immediate implementation occurs incrementally post-[#655] v1.
|
||||
|
||||
---
|
||||
|
||||
## 3. High-Availability & Rolling-Restart Architecture
|
||||
|
||||
### 3.1 Architecture Overview
|
||||
|
||||
```
|
||||
+----------------------------+
|
||||
| LLM Clients / IDE Sessions |
|
||||
+--------------+-------------+
|
||||
|
|
||||
v
|
||||
+----------------------------+
|
||||
| HA Proxy / Router |
|
||||
| (Health-based & Affinity) |
|
||||
+------+--------------+------+
|
||||
| |
|
||||
+--------------+ +--------------+
|
||||
v v
|
||||
+--------------------+ +--------------------+
|
||||
| MCP Instance Node A| | MCP Instance Node B|
|
||||
| (Version N) | | (Version N+1) |
|
||||
+---------+----------+ +---------+----------+
|
||||
| |
|
||||
+----------------------+----------------------+
|
||||
|
|
||||
v
|
||||
+----------------------------+
|
||||
| Control-Plane DB Substrate|
|
||||
| (Shared Lease & Locks) |
|
||||
+--------------+-------------+
|
||||
|
|
||||
v
|
||||
+----------------------------+
|
||||
| Gitea API |
|
||||
+----------------------------+
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### 3.2 Key System Components
|
||||
|
||||
#### A. Multiple MCP Instance Cohorts
|
||||
* The control plane runs across $N \ge 2$ redundant process nodes.
|
||||
* Dual-namespace deployment allows running the old version (Node A) alongside a updated version (Node B) during rolling upgrades.
|
||||
|
||||
#### B. Shared Durable Session Storage & Resume Tokens
|
||||
* Session context, preflight verification proofs, and capability resolution states are stored in the shared control-plane database.
|
||||
* Client requests carry an explicit `session_id` and `resume_token`. If an MCP instance restarts or a request routes to a different instance, the target node validates the token against the database without requiring full session re-initialization.
|
||||
|
||||
#### C. Shared Lease Authority & Fencing Counters
|
||||
* Workflow leases (`gitea_allocate_next_work`, `gitea_adopt_workflow_lease`) use monotonic fencing tokens (`lease_generation_id`).
|
||||
* When Node B acquires or renews a lease, it increments the generation counter. Any delayed or out-of-order write attempt from Node A using an older generation token is rejected by database constraints.
|
||||
|
||||
#### D. Leader Election & Coordinated Drain
|
||||
* Node clusters elect a primary coordinator node for administrative background tasks (such as stale lease cleanup or incident Watchdogs).
|
||||
* During a rolling deployment:
|
||||
1. Node B (new version) is launched and registers as healthy.
|
||||
2. Router directs new session creations to Node B.
|
||||
3. Node A enters `MAINTENANCE_DRAIN` status ([#659](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/659)), completing in-flight mutations while refusing new tasks.
|
||||
4. Once all active sessions migrate or complete, Node A shuts down cleanly.
|
||||
|
||||
#### E. Idempotent Mutations & Failover Safety
|
||||
* All state-changing tool executions (PR creation, review submission, merge operations, label changes) carry a deterministic `idempotency_key`.
|
||||
* If a network connection flaps or a node fails mid-mutation, the re-issued request with the same `idempotency_key` is recognized by the control-plane substrate, returning the existing recorded result without repeating side effects on Gitea.
|
||||
|
||||
#### F. Schema Version Compatibility
|
||||
* Database migrations follow non-breaking additive patterns.
|
||||
* During rolling upgrades where Node A (Version $N$) and Node B (Version $N+1$) run concurrently, both versions operate against the shared schema without structural conflicts.
|
||||
|
||||
---
|
||||
|
||||
## 4. Split-Brain & Failure Behavior
|
||||
|
||||
### 4.1 Split-Brain Risk Scenarios & Mitigation
|
||||
|
||||
| Scenario | Risk | Mitigation Strategy |
|
||||
|---|---|---|
|
||||
| **Network Partition between Nodes** | Both Node A and Node B attempt to process operations for the same issue/PR. | **Generation Fencing:** Lease renewal requires updating the DB generation counter. The node isolated from the DB fails closed immediately. |
|
||||
| **Stale Node Recovery** | Node A recovers after a long pause and executes a queued mutation. | **Lease Expiry & TTL Fencing:** Transactions verify that `expires_at > NOW()` within the atomic SQLite transaction boundaries. |
|
||||
| **Database Connection Loss** | Node loses access to shared control-plane DB substrate. | **Strict Fail-Closed:** The node immediately marks all task capabilities as `blocked` and rejects mutation tools until DB connectivity is re-established. |
|
||||
|
||||
---
|
||||
|
||||
## 5. Phased Implementation Milestones
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
M1[Milestone 1: Shared Control-Plane DB Schema & Resume Tokens] --> M2[Milestone 2: Idempotent Mutation Layer]
|
||||
M2 --> M3[Milestone 3: Health Routing & Standby Failover]
|
||||
M3 --> M4[Milestone 4: Active-Active Rolling Deployment & Auto-Drain]
|
||||
```
|
||||
|
||||
### Milestone 1: Shared Control-Plane DB Schema & Resume Tokens (Post-#655)
|
||||
* Extend [#613] Control-Plane DB schema to store multi-instance node heartbeat records and session resume tokens.
|
||||
* Enable session lookup across instances via `session_id`.
|
||||
|
||||
### Milestone 2: Idempotent Mutation Layer & Lease Fencing
|
||||
* Add mandatory `idempotency_key` tracking to all Gitea mutation tools.
|
||||
* Implement monotonic lease fencing counters in `gitea_allocate_next_work` and `gitea_adopt_workflow_lease`.
|
||||
|
||||
### Milestone 3: Health-Based Routing & Active-Passive Standby
|
||||
* Introduce lightweight proxy/router capable of checking node health endpoints.
|
||||
* Implement active-standby failover where standby node automatically assumes work if active node fails health checks.
|
||||
|
||||
### Milestone 4: Active-Active Horizontal Deployment & Rolling Upgrade Automation
|
||||
* Enable true active-active multi-instance execution.
|
||||
* Integrate automated zero-downtime rolling upgrades coordinated with `gitea_request_mcp_restart` maintenance drain.
|
||||
|
||||
---
|
||||
|
||||
## 6. Observability & Audit Requirements
|
||||
|
||||
High-availability control plane operations must expose clear telemetry and audit trails:
|
||||
|
||||
* **Node Registry Telemetry:** Active nodes, version numbers, uptime, and heartbeat timestamps reported via `gitea_get_runtime_context`.
|
||||
* **Lease Fencing Metrics:** Tracking lease acquire latency, fence rejection counts, and lease handoff durations.
|
||||
* **Failover & Re-route Audit Logs:** Durable logging of session migrations between nodes, drain initiation, and process retirement events.
|
||||
|
||||
---
|
||||
|
||||
## 7. Tradeoffs & Accepted Risks
|
||||
|
||||
* **Increased Architectural Complexity:** Moving from a single process to a multi-instance control plane requires robust DB locking, proxy routing, and migration governance.
|
||||
* **Database Dependency:** The control-plane database substrate becomes a critical shared dependency for multi-node deployments. High availability for the underlying SQLite file system / DB must be guaranteed.
|
||||
@@ -0,0 +1,83 @@
|
||||
# Incident #670: bare direct-to-master commit `2fa97c26` (retroactive audit)
|
||||
|
||||
Status: verified; disposition recommendation: **accept as-is, no revert** (final
|
||||
disposition owned by controller per issue #670).
|
||||
|
||||
## Summary
|
||||
|
||||
Commit `2fa97c26fbda555a1a83930ca5fdcea9d8e47b50`
|
||||
(`fix(mcp): load dotenv relative to project root`) landed on `prgs/master`
|
||||
as a single-parent commit with no PR wrapper and no review record, bypassing
|
||||
the sanctioned issue → branch → PR → review → merge workflow. It was
|
||||
discovered during the PR #654 post-merge audit. PR #654 itself merged
|
||||
cleanly via the Gitea API and did **not** introduce this commit.
|
||||
|
||||
## Verification evidence (acceptance criteria 1–3)
|
||||
|
||||
- **AC1 — present on `prgs/master`: yes.**
|
||||
`git merge-base --is-ancestor 2fa97c26fbda555a1a83930ca5fdcea9d8e47b50 prgs/master` → true.
|
||||
- **AC2 — no PR or review record: confirmed.**
|
||||
The commit is a single-parent, non-merge commit sitting directly on
|
||||
first-parent master between the #629 merge (`5ab5fe85`) and the #654
|
||||
merge (`ec903b0d`). A PR landing on master produces a merge commit (or a
|
||||
PR-linked head); neither exists here. The controller audit at issue-create
|
||||
time also found no PR wrapper and no review record for this SHA.
|
||||
- **AC3 — changed files and diff summary: confirmed.**
|
||||
`gitea_auth.py | 5 +++--` (+3/−2). Single parent
|
||||
`5ab5fe8583c07134d55dadf09381aecb67df246e`. The change moves
|
||||
`PROJECT_ROOT` derivation above `load_dotenv()` and loads
|
||||
`.env` relative to the project root instead of the process CWD.
|
||||
|
||||
## AC4 — why no immediate revert
|
||||
|
||||
- The dotenv fix is intentional and required for correct runtime behavior:
|
||||
without it, `load_dotenv()` resolves `.env` against the process working
|
||||
directory, which breaks MCP server launches whose CWD is not the project
|
||||
root.
|
||||
- The change is small (+3/−2), self-contained in `gitea_auth.py`, and has
|
||||
been running on master without incident since 2026-07-10.
|
||||
- Reverting would re-introduce a real bug to remove a provenance defect —
|
||||
the wrong trade. Provenance is repaired retroactively by this document,
|
||||
issue #670, and the hardening landed under #671.
|
||||
- If the controller later judges the change unsafe, a separate
|
||||
revert/repair issue is the sanctioned path (issue #670, recommended
|
||||
disposition option 4).
|
||||
|
||||
## AC5 — workflow-hardening linkage
|
||||
|
||||
Prevention already landed: **issue #671** (closed)
|
||||
*“Block direct pushes to stable branches from MCP workflow sessions”*,
|
||||
implemented by commit `5933d87647656643a67a50331c4c7b06ea751dad`
|
||||
(`feat(guard): block direct stable-branch pushes from MCP workflow sessions`).
|
||||
|
||||
Shipped guardrails include:
|
||||
|
||||
- `gitea_record_stable_branch_push_attempt` — classifies proposed commands
|
||||
for direct stable-branch push intent (`git push <remote> master`,
|
||||
refspecs, `HEAD:master`, `--force`, dry-run intent, `:master` delete),
|
||||
plus root/control-checkout local commits not carried by an issue branch,
|
||||
and writes a durable `stable_branch_contamination` marker.
|
||||
- `gitea_audit_stable_branch_contamination` — reconciler-only audit/clear
|
||||
path; a contaminated worker session cannot self-clear.
|
||||
- Review/merge/close/completion mutations fail closed while a
|
||||
contamination marker is active.
|
||||
|
||||
## AC6 — PR #654 was not the source
|
||||
|
||||
- `2fa97c26` is the **first parent** of the #654 merge commit
|
||||
`ec903b0d619e7a27d24aed272a890f4e5d381411`; it predates the #654 merge.
|
||||
- First-parent history `5ab5fe8..ec903b0`:
|
||||
`2fa97c2 fix(mcp): load dotenv relative to project root` followed by
|
||||
`ec903b0 Merge pull request 'feat: lifecycle role/hazard labels ... (#603)' (#654)`.
|
||||
- The #654 merger audit confirmed `ec903b0d` was a valid Gitea-API merge,
|
||||
the `git push prgs master` attempt during that run was a no-op, and the
|
||||
net change `2fa97c2..ec903b0` contained only the reviewed #603
|
||||
lifecycle-label files.
|
||||
- Conclusion: #654 merged reviewed content only; the unauthorized-path
|
||||
defect is solely the earlier bare commit `2fa97c26`.
|
||||
|
||||
## Explicit non-actions (unchanged by this audit)
|
||||
|
||||
- No revert of `2fa97c26`.
|
||||
- No force-push or history rewrite.
|
||||
- No master mutation from the audit session.
|
||||
@@ -0,0 +1,45 @@
|
||||
# MCP maintenance-drain mode (#659)
|
||||
|
||||
Graceful **maintenance drain** stops new work assignment and defers non-allowlisted
|
||||
mutations so sessions can finish critical handoffs and checkpoint before a
|
||||
restart. It is **not** a restart authorization: the drain *proof* and apply gate
|
||||
remain #661.
|
||||
|
||||
## State
|
||||
|
||||
Per repository scope (`remote`/`org`/`repo`) in the control-plane DB table
|
||||
`maintenance_drain` (schema v6):
|
||||
|
||||
| State | Meaning |
|
||||
|-------|---------|
|
||||
| `inactive` | Normal operation (also: no row) |
|
||||
| `draining` | Assignment stopped; non-allowlisted mutations deferred |
|
||||
|
||||
Enter/exit transitions are audited as `maintenance_drain_enter` /
|
||||
`maintenance_drain_exit` events.
|
||||
|
||||
## Tools
|
||||
|
||||
| Tool | Permission | Effect |
|
||||
|------|------------|--------|
|
||||
| `gitea_maintenance_drain_status` | `gitea.read` | Observe drain (every session) |
|
||||
| `gitea_enter_maintenance_drain` | `runtime.maintenance_drain` | Enter drain (capability-gated) |
|
||||
| `gitea_exit_maintenance_drain` | `runtime.maintenance_drain` | Exit drain |
|
||||
|
||||
`runtime.maintenance_drain` is intentionally **not** a `gitea.*` op, so ordinary
|
||||
author profiles cannot enter drain by accident.
|
||||
|
||||
## Enforcement
|
||||
|
||||
1. **Allocator** (`allocate_next_work`): while draining, returns `outcome=wait`
|
||||
with `reason_code=maintenance_drain_assignment_stopped` for dry-run and apply.
|
||||
2. **Mutation preflight** (`verify_preflight_purity`): non-allowlisted mutation
|
||||
tasks raise `MaintenanceDrainError` with a typed next action.
|
||||
3. **Allowlist** (safety only): heartbeats, lease release/abandon, session
|
||||
checkpoints, enter/exit drain. Reads always work.
|
||||
|
||||
## Restart relationship
|
||||
|
||||
Drain mode prepares the blast radius. Restart apply still requires a clean
|
||||
`DrainProof` (#661) or authorized break-glass. Status payloads never claim
|
||||
restart permission.
|
||||
@@ -0,0 +1,94 @@
|
||||
# MCP scoped recovery playbook (#669)
|
||||
|
||||
**Parent:** [#655](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/655)
|
||||
**Vision / roadmap:** [#652](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/652) · [#653](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/653)
|
||||
**Class matrix:** [#663](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/663) · `docs/mcp-restart-classes.md`
|
||||
**Coordinator:** [#658](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/658) · `restart_coordinator.py`
|
||||
**Audit lineage:** [#665](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/665)
|
||||
|
||||
## Decision
|
||||
|
||||
Full-server MCP reset is a **last resort**. Prefer the narrowest recovery that
|
||||
can clear the symptom. The coordinator **refuses** `rolling_mcp_restart`,
|
||||
`full_mcp_restart`, and `host_restart` unless:
|
||||
|
||||
1. The inventory carries a prior **attempt log** of at least one *insufficient*
|
||||
narrower recovery, **or**
|
||||
2. **Break-glass** is authorized
|
||||
(`request_break_glass` + `GITEA_BREAKGLASS_RESTART_AUTHORIZATION`).
|
||||
|
||||
Break-glass still never bypasses the #663 class matrix (role/permission).
|
||||
|
||||
## Ladder (narrow → broad)
|
||||
|
||||
| Rank | Action | Self-service | Implementation / delegation |
|
||||
|---:|---|---|---|
|
||||
| 0 | `client_reconnect` | yes | Host auto-reconnect / client reconnect · [#584](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/584) · `docs/mcp-namespace-eof-recovery.md` |
|
||||
| 1 | `capability_refresh` | yes | `gitea_resolve_task_capability` + `gitea_whoami` · [#610](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/610) · [#685](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/685) |
|
||||
| 2 | `session_reconnect` | yes | Runtime rebind + explicit `worktree_path` · [#543](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/543) · [#618](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/618) |
|
||||
| 3 | `configuration_reload` | no | Class `configuration_reload` · console reload · [#642](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/642) |
|
||||
| 4 | `lease_recovery` | no | Lock/lease recovery paths · [#702](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/702) · [#753](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/753) · [#790](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/790) |
|
||||
| 5 | `worker_restart` | no | Class `worker_restart` · #663 |
|
||||
| 6 | `role_runtime_restart` | no | Class `role_runtime_restart` · console restart · #642/#663 |
|
||||
| 7 | `connector_restart` | no | Class `connector_restart` · #663 |
|
||||
| 8 | `rolling_mcp_restart` | no | Class `rolling_mcp_restart` · design [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668) · **attempt log required** |
|
||||
| 9 | `full_mcp_restart` | no | Class `full_mcp_restart` · **attempt log required** |
|
||||
| 10 | `host_restart` | no | Class `host_restart` · **attempt log required** |
|
||||
|
||||
Machine-readable source of truth: `recovery_playbook.RECOVERY_LADDER` and
|
||||
`recovery_playbook.ladder_document()`.
|
||||
|
||||
## Attempt log shape
|
||||
|
||||
Each prior attempt is a mapping:
|
||||
|
||||
```json
|
||||
{
|
||||
"action": "client_reconnect",
|
||||
"outcome": "insufficient",
|
||||
"reason": "transport still closed after IDE reconnect",
|
||||
"actor": "prgs-controller-12345",
|
||||
"recorded_at": "2026-07-25T21:00:00+00:00"
|
||||
}
|
||||
```
|
||||
|
||||
Outcomes that count toward escalation: `failed`, `insufficient`, `denied`,
|
||||
`unresolved`, `timeout`, `error`.
|
||||
|
||||
Pass attempts into the coordinator via inventory
|
||||
`prior_recovery_attempts` or the MCP tool argument
|
||||
`prior_recovery_attempts_json` on `gitea_request_mcp_restart`.
|
||||
|
||||
Helper: `recovery_playbook.build_attempt_record(...)`.
|
||||
|
||||
## Symptom → first rung
|
||||
|
||||
`recovery_playbook.recommend_actions(symptoms=[...])` maps symptoms such as
|
||||
`transport_eof`, `stale_capability`, `stale_lease`, `daemon_corrupt` to the
|
||||
narrowest recommended action, then walks the ladder. Soft recommendations
|
||||
never replace the hard gate on broad restarts.
|
||||
|
||||
## Enforcement points
|
||||
|
||||
1. **`recovery_playbook.assess_escalation`** — pure gate.
|
||||
2. **`restart_coordinator.evaluate_restart_impact`** — when `restart_class` is
|
||||
set (policy-enforced path), broad classes require the gate; report fields
|
||||
`attempt_log_satisfied`, `playbook_escalation`, `break_glass`.
|
||||
3. **`gitea_request_mcp_restart`** — accepts attempt JSON and env-authorized
|
||||
break-glass; never restarts a process.
|
||||
|
||||
## Metrics
|
||||
|
||||
`recovery_playbook.recovery_metrics(attempts)` reports the fraction of
|
||||
successful recoveries that avoided full/host restart
|
||||
(`fraction_avoided_full_restart`).
|
||||
|
||||
## Non-goals
|
||||
|
||||
* HA multi-instance execution (#668 design only here).
|
||||
* Normalizing `pkill` (#630 contamination stays forbidden).
|
||||
* Silent mutation of leases or processes from the playbook itself.
|
||||
|
||||
## Manual process kills
|
||||
|
||||
Remain forbidden and contaminating (#630). The playbook never recommends them.
|
||||
@@ -95,12 +95,18 @@ gitea_request_mcp_restart(remote, host, org, repo,
|
||||
target_session_id=None, target_role=None,
|
||||
target_connector=None,
|
||||
drain_proof_json=None,
|
||||
request_break_glass=False)
|
||||
request_break_glass=False,
|
||||
prior_recovery_attempts_json=None)
|
||||
```
|
||||
|
||||
It **never restarts anything**: `apply_supported` is always `false` and
|
||||
`restart_performed` is always `false`.
|
||||
|
||||
`prior_recovery_attempts_json` (#669) is an optional JSON array of prior
|
||||
narrow recovery attempts. Rolling / full / host classes require at least one
|
||||
*insufficient* narrower attempt (or authorized break-glass). See
|
||||
`docs/mcp-recovery-playbook.md`.
|
||||
|
||||
### Dry-run versus apply
|
||||
|
||||
| Call | Behavior |
|
||||
@@ -112,9 +118,10 @@ It **never restarts anything**: `apply_supported` is always `false` and
|
||||
|
||||
An apply requires **both** authorizations, and they are independent:
|
||||
|
||||
1. **Restart-class authorization** (#663) — the requester's role and permissions
|
||||
must allow the requested class, the class's approval requirement must be
|
||||
satisfied, and any target-scoped class must name its target. Failing any of
|
||||
1. **Restart-class authorization** (#663 / #669) — the requester's role and
|
||||
permissions must allow the requested class, the class's approval requirement
|
||||
must be satisfied, any target-scoped class must name its target, and broad
|
||||
classes must satisfy the recovery-playbook attempt-log gate. Failing any of
|
||||
these makes `allow_restart` `false`.
|
||||
2. **Drain-proof gate** (#661) — a valid, unexpired, clean proof bound to the
|
||||
current impact fingerprint, or an authorized break-glass.
|
||||
@@ -127,7 +134,10 @@ the authorization that produced it.
|
||||
|
||||
### Break-glass
|
||||
|
||||
Break-glass bypasses the **drain proof only** — never the restart-class matrix.
|
||||
Break-glass bypasses the **drain proof only** — never the restart-class matrix
|
||||
(role/permission). Separately, authorized break-glass also satisfies the #669
|
||||
attempt-log requirement for broad restarts (rolling/full/host), because that
|
||||
gate is not a class-matrix permission check.
|
||||
It is honoured solely when `request_break_glass` is set *and* the environment
|
||||
carries `GITEA_BREAKGLASS_RESTART_AUTHORIZATION`; like operator override, the
|
||||
tool argument expresses caller intent and cannot be self-asserted by a worker
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
# Web Console: Notifications & Human-Attention Routing (#648)
|
||||
|
||||
- **Status:** Phase 3 Live
|
||||
- **Tracking Issue:** [#648](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/648)
|
||||
- **Parent Epic:** [#631](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/631)
|
||||
- **Attention Boundary Reference:** [#628](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/628)
|
||||
|
||||
---
|
||||
|
||||
## 1. Overview
|
||||
|
||||
The **Notifications & Human-Attention Console** (`/notifications`, `/api/v1/notifications`) provides intelligent event classification and human-attention routing for autonomous workflow operations.
|
||||
|
||||
To prevent alert fatigue while ensuring critical escalation boundaries are never missed, events are classified into three distinct **Attention Classes**:
|
||||
|
||||
1. **`human-required`** (Urgent Escalation Boundary):
|
||||
- Items requiring immediate human intervention or business decisions.
|
||||
- Triggers: Auth failures, hard stops, irrecoverable state, decision locks, failed report validations, critical probe errors.
|
||||
- Display: Highlighted in red (`badge-blocked`) with a `HUMAN REQUIRED` badge.
|
||||
|
||||
2. **`operator`** (Operational Inbox):
|
||||
- Items requiring controller or operator review/triage during routine execution.
|
||||
- Triggers: Blocked PRs (merge conflicts), stale leases, duplicate PRs on issues, unassigned ready work.
|
||||
- Display: Displayed in orange/yellow (`badge-claimed`).
|
||||
|
||||
3. **`routine`** (Background Workflow Transitions):
|
||||
- Normal, healthy workflow transitions and state progressions.
|
||||
- Triggers: Active PRs/issues in standard state, clean branch creation, routine heartbeats.
|
||||
- Display: Filtered out of default inbox views to eliminate notification spam; viewable on demand via the "Routine" or "All" tab.
|
||||
|
||||
---
|
||||
|
||||
## 2. API Endpoints
|
||||
|
||||
### `GET /api/v1/notifications`
|
||||
*Compatibility Alias:* `GET /api/notifications`
|
||||
|
||||
#### Query Parameters:
|
||||
- `project_id` (optional): Filter notifications by project ID.
|
||||
- `attention_class` (optional): `inbox` (default: human-required + operator), `human-required`, `operator`, `routine`, `all`.
|
||||
|
||||
#### Example JSON Response:
|
||||
```json
|
||||
{
|
||||
"project_id": "gitea-tools",
|
||||
"repo_label": "Scaled-Tech-Consulting/Gitea-Tools",
|
||||
"human_required_count": 0,
|
||||
"operator_count": 2,
|
||||
"routine_count": 5,
|
||||
"total_count": 7,
|
||||
"fetch_error": null,
|
||||
"inbox_items": [
|
||||
{
|
||||
"id": "notif-pr-block-742",
|
||||
"attention_class": "operator",
|
||||
"category": "blocker",
|
||||
"title": "Blocked PR #742",
|
||||
"summary": "PR #742 requires merge conflict resolution.",
|
||||
"work_kind": "pr",
|
||||
"work_number": 742,
|
||||
"project_id": "gitea-tools",
|
||||
"repo_label": "Scaled-Tech-Consulting/Gitea-Tools",
|
||||
"created_at": "2026-07-25T16:39:47Z",
|
||||
"deep_link": "/traffic",
|
||||
"requires_human": false,
|
||||
"extra": {}
|
||||
}
|
||||
],
|
||||
"all_items": [...]
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. UI Navigation
|
||||
|
||||
- Access via the **Traffic** navigation menu: **Traffic → Notifications**.
|
||||
- The main view displays:
|
||||
- **Metrics Summary Bar**: Highlighting counts for Human Required, Operator Inbox, and Routine items.
|
||||
- **Attention Filter Tabs**: Toggle between Inbox (Human + Operator), Human Required, Operator, Routine, and All.
|
||||
- **Structured Event Table**: Displays category, title, summary, work item links, and timestamps.
|
||||
@@ -0,0 +1,102 @@
|
||||
# Web Console: restart status, impact preview, and approval state (#667)
|
||||
|
||||
Phase 1 of the console restart surface. It consumes the #655 coordinator
|
||||
substrate and displays it. It performs no restart, reload, drain, approval, or
|
||||
process action, and it registers no write endpoint.
|
||||
|
||||
Issue #667's rollout is explicit — *status views first, write approval after the
|
||||
backend gates are green* — and this change delivers only the status half.
|
||||
|
||||
## Surfaces
|
||||
|
||||
| Path | Method | Purpose |
|
||||
|------|--------|---------|
|
||||
| `/runtime/restart` | GET | Restart status page |
|
||||
| `/api/v1/system/restart/status` | GET | Same snapshot as JSON |
|
||||
|
||||
Both accept an optional `restart_class` query parameter (default
|
||||
`full_mcp_restart`). An unrecognised class is not an error: the coordinator
|
||||
resolves it as unknown and fails closed, and the page shows the resulting deny.
|
||||
|
||||
Neither path accepts `POST`; a write attempt returns `405`, and a test asserts
|
||||
it.
|
||||
|
||||
## What it shows
|
||||
|
||||
* **Impact preview (#658)** — verdict, blast radius, affected sessions, leases,
|
||||
critical sections, mutations, and the counts behind them, evaluated
|
||||
`dry_run=True` against live control-plane state.
|
||||
* **Drain proof (#661)** — verification of a supplied proof: valid, clean,
|
||||
expired, tampered, and the reasons behind a refusal.
|
||||
* **Post-restart reconcile (#662)** — the most recent completion proof, its
|
||||
overall status, and which dimensions still require follow-up.
|
||||
* **Restart classes (#663)** — the least-privilege matrix, with *you may
|
||||
request* and *you may execute* computed for the viewing role rather than for a
|
||||
generic operator.
|
||||
* **Approval controls (#633)** — the authorization state of
|
||||
`system.restart_namespace` and `system.reload_namespace`.
|
||||
* **Break-glass (#664)** — declared and marked unavailable; see below.
|
||||
|
||||
## Three rules this surface holds itself to
|
||||
|
||||
A status page that is wrong is worse than one that is missing, because an
|
||||
operator acts on it. Three properties are enforced by tests, and each was
|
||||
verified by reverting the guard and watching a test fail.
|
||||
|
||||
### An unreadable source reports unavailable, never green
|
||||
|
||||
Every source carries its own `SourceStatus`. Nothing substitutes a default,
|
||||
placeholder, or self-comparison for a reading that failed. An unreadable
|
||||
control-plane database yields `inventory_complete: false`, which the coordinator
|
||||
itself turns into a fail-closed verdict, and the page says the blast radius is
|
||||
unknown rather than showing an empty affected-sessions table.
|
||||
|
||||
An absent drain proof is reported as absent — not as a pass. The #661 gate
|
||||
authorizes a restart only against a valid, unexpired, clean proof, so no proof
|
||||
is precisely the state that gate denies on.
|
||||
|
||||
### Authorization is asked the way execution would ask it
|
||||
|
||||
Every probe passes `for_execution=True`.
|
||||
|
||||
Asked without it, an admin is `allowed` for `system.restart_namespace`. On a
|
||||
control surface that reads as a live button. Asked the way an execution attempt
|
||||
would ask, the same principal is refused `phase_not_active`, because the console
|
||||
is in Phase 1 and the action is Phase 2. This surface reports the second answer.
|
||||
|
||||
`execution_enabled` is therefore `false` for every action and every role today,
|
||||
and a test asserts that across the whole role matrix.
|
||||
|
||||
### The control-plane database is opened read-only
|
||||
|
||||
`ControlPlaneDB()` creates directories and runs migrations on construction — a
|
||||
write. This surface never constructs one. It opens the sqlite file with
|
||||
`mode=ro`, exactly as `webui/inventory.py` does, and treats a missing file as
|
||||
missing authority rather than as an empty inventory.
|
||||
|
||||
The test that protects this points at a path inside a directory that already
|
||||
exists, so a read-write `connect` would really create the file. A nested
|
||||
missing-directory path would have passed for the wrong reason.
|
||||
|
||||
## Break-glass is declared, not offered
|
||||
|
||||
The break-glass workflow (#664) is not available on this branch's base. The
|
||||
panel is rendered to operator-class roles as **unavailable**, naming the issue
|
||||
that tracks it. It is not silently omitted, because an operator who has been
|
||||
told a governance path exists needs to see that it is not wired here; and it is
|
||||
not rendered as a control, because there is nothing behind it.
|
||||
|
||||
Unprivileged viewers see only a note that the surface is operator-class.
|
||||
|
||||
## Redaction and escaping
|
||||
|
||||
Every interpolated value passes through `_esc` (`html.escape(..., quote=True)`).
|
||||
Free-form text and anything that can carry a filesystem path additionally passes
|
||||
through `webui.inventory.scrub_text`, which redacts credential-shaped tokens
|
||||
inside a string rather than only at its start. The impact payload is passed
|
||||
through `webui.inventory.scrub` before rendering.
|
||||
|
||||
## Linkage
|
||||
|
||||
Parent #655 · extends #642 · consumes #658, #661, #662, #663 · RBAC #633 ·
|
||||
console #631 · vision #652 · roadmap #653 · break-glass #664.
|
||||
+275
-9
@@ -1472,6 +1472,9 @@ def verify_preflight_purity(
|
||||
# contaminated by manual MCP daemon process killing (reconciler-exempt).
|
||||
_enforce_runtime_recovery_contamination_gate(task, remote)
|
||||
|
||||
# #659 AC3: defer non-allowlisted mutations while maintenance drain is active.
|
||||
_enforce_maintenance_drain_gate(task, remote=remote, org=org, repo=repo)
|
||||
|
||||
ctx = _resolve_namespace_mutation_context(worktree_path)
|
||||
workspace = ctx["workspace_path"]
|
||||
canonical_root = ctx["canonical_repo_root"]
|
||||
@@ -2004,6 +2007,55 @@ def _enforce_runtime_recovery_contamination_gate(
|
||||
)
|
||||
|
||||
|
||||
def _enforce_maintenance_drain_gate(
|
||||
task: str | None,
|
||||
remote: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
) -> None:
|
||||
"""#659 AC3: defer non-allowlisted mutations while drain is active.
|
||||
|
||||
The single mutation chokepoint already used by every gated task, so drain
|
||||
coverage cannot drift per-tool. Allowlisted safety operations (heartbeat,
|
||||
release/abandon, checkpoint, drain exit) pass through so an in-flight
|
||||
session can still finish and hand off; everything else is deferred with a
|
||||
typed blocker. Unreadable drain state fails closed — a drain that cannot be
|
||||
read is not evidence that no drain is running.
|
||||
"""
|
||||
if _preflight_in_test_mode() and not os.environ.get(
|
||||
"GITEA_TEST_FORCE_MAINTENANCE_DRAIN"
|
||||
):
|
||||
return
|
||||
if maintenance_drain.is_allowlisted_task(task):
|
||||
return
|
||||
|
||||
try:
|
||||
_h, o, r = _resolve(remote, None, org, repo)
|
||||
except Exception: # noqa: BLE001 — scope resolution is best-effort here
|
||||
o, r = (org or ""), (repo or "")
|
||||
|
||||
db, errs = _control_plane_db_or_error()
|
||||
if db is None:
|
||||
raise RuntimeError(
|
||||
"maintenance-drain state could not be read: "
|
||||
f"{'; '.join(errs) or 'control-plane DB unavailable'} (fail closed, #659)"
|
||||
)
|
||||
try:
|
||||
record = db.read_maintenance_drain(remote=remote or "", org=o, repo=r)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
raise RuntimeError(
|
||||
f"maintenance-drain state could not be read: {_redact(str(exc))} "
|
||||
"(fail closed, #659)"
|
||||
) from exc
|
||||
|
||||
decision = maintenance_drain.classify_mutation(task, record)
|
||||
if not decision["allowed"]:
|
||||
raise maintenance_drain.MaintenanceDrainError(
|
||||
maintenance_drain.format_drain_block_error(decision),
|
||||
decision=decision,
|
||||
)
|
||||
|
||||
|
||||
def _enforce_stable_branch_contamination_gate(
|
||||
task: str | None,
|
||||
remote: str | None = None,
|
||||
@@ -2064,6 +2116,7 @@ import allocator_service # noqa: E402
|
||||
import allocator_dependencies # noqa: E402
|
||||
import dependency_graph # noqa: E402 # #784 durable dependency edges
|
||||
import control_plane_db # noqa: E402
|
||||
import maintenance_drain # noqa: E402 # #659 graceful maintenance-drain mode
|
||||
import lease_lifecycle # noqa: E402
|
||||
import lease_policy # noqa: E402
|
||||
import workflow_dashboard # noqa: E402 # #605 live queue/lease dashboard
|
||||
@@ -22559,6 +22612,193 @@ def gitea_workflow_dashboard(
|
||||
return payload
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_maintenance_drain_status(
|
||||
remote: str = "dadeschools",
|
||||
host: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
) -> dict:
|
||||
"""Read-only: current maintenance-drain state for a repository scope (#659 AC4).
|
||||
|
||||
Every session must be able to observe drain so it can stop creating new work
|
||||
and finish only allowlisted safety operations. Never mutates; never restarts.
|
||||
"""
|
||||
read_block = _profile_operation_gate("gitea.read")
|
||||
if read_block:
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": read_block,
|
||||
"permission_report": _permission_block_report("gitea.read"),
|
||||
}
|
||||
try:
|
||||
_h, o, r = _resolve(remote, host, org, repo)
|
||||
except ValueError as exc:
|
||||
return {"success": False, "read_only": True, "reasons": [str(exc)]}
|
||||
db, errs = _control_plane_db_or_error()
|
||||
if db is None:
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": errs or ["control-plane DB unavailable"],
|
||||
"maintenance_drain": maintenance_drain.status_payload(
|
||||
None, remote=remote, org=o, repo=r
|
||||
),
|
||||
}
|
||||
try:
|
||||
record = db.read_maintenance_drain(remote=remote, org=o, repo=r)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": [f"drain state unreadable: {_redact(str(exc))}"],
|
||||
}
|
||||
payload = maintenance_drain.status_payload(
|
||||
record, remote=remote, org=o, repo=r
|
||||
)
|
||||
return {"success": True, "read_only": True, "maintenance_drain": payload}
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_enter_maintenance_drain(
|
||||
reason: str = "",
|
||||
remote: str = "dadeschools",
|
||||
host: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
session_id: str | None = None,
|
||||
) -> dict:
|
||||
"""Enter graceful maintenance-drain mode for a repository scope (#659 AC1).
|
||||
|
||||
Stops new assignment and defers non-allowlisted mutations until exit. Requires
|
||||
``runtime.maintenance_drain`` (controller/lifecycle capability — not granted
|
||||
by ordinary Gitea author profiles). Audited in the control-plane event log.
|
||||
"""
|
||||
cap_block = _profile_operation_gate("runtime.maintenance_drain")
|
||||
if cap_block:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": cap_block,
|
||||
"permission_report": _permission_block_report(
|
||||
"runtime.maintenance_drain"
|
||||
),
|
||||
}
|
||||
try:
|
||||
_h, o, r = _resolve(remote, host, org, repo)
|
||||
except ValueError as exc:
|
||||
return {"success": False, "performed": False, "reasons": [str(exc)]}
|
||||
db, errs = _control_plane_db_or_error()
|
||||
if db is None:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": errs or ["control-plane DB unavailable"],
|
||||
}
|
||||
profile = get_profile() or {}
|
||||
try:
|
||||
result = db.set_maintenance_drain(
|
||||
remote=remote,
|
||||
org=o,
|
||||
repo=r,
|
||||
state=maintenance_drain.STATE_DRAINING,
|
||||
reason=reason or "operator-entered maintenance drain",
|
||||
requested_by=str(
|
||||
(profile.get("identity") or {}).get("username")
|
||||
or profile.get("expected_username")
|
||||
or ""
|
||||
),
|
||||
requested_by_profile=str(profile.get("profile_name") or ""),
|
||||
session_id=str(session_id or ""),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": [f"enter drain failed: {_redact(str(exc))}"],
|
||||
}
|
||||
record = result.get("record") or {}
|
||||
return {
|
||||
"success": True,
|
||||
"performed": True,
|
||||
"transitioned": bool(result.get("transitioned")),
|
||||
"state": result.get("state"),
|
||||
"prior_state": result.get("prior_state"),
|
||||
"drain_id": result.get("drain_id"),
|
||||
"maintenance_drain": maintenance_drain.status_payload(
|
||||
record, remote=remote, org=o, repo=r
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_exit_maintenance_drain(
|
||||
reason: str = "",
|
||||
remote: str = "dadeschools",
|
||||
host: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
session_id: str | None = None,
|
||||
) -> dict:
|
||||
"""Exit graceful maintenance-drain mode (#659 AC1). Restores assignment and mutations."""
|
||||
cap_block = _profile_operation_gate("runtime.maintenance_drain")
|
||||
if cap_block:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": cap_block,
|
||||
"permission_report": _permission_block_report(
|
||||
"runtime.maintenance_drain"
|
||||
),
|
||||
}
|
||||
try:
|
||||
_h, o, r = _resolve(remote, host, org, repo)
|
||||
except ValueError as exc:
|
||||
return {"success": False, "performed": False, "reasons": [str(exc)]}
|
||||
db, errs = _control_plane_db_or_error()
|
||||
if db is None:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": errs or ["control-plane DB unavailable"],
|
||||
}
|
||||
profile = get_profile() or {}
|
||||
try:
|
||||
result = db.set_maintenance_drain(
|
||||
remote=remote,
|
||||
org=o,
|
||||
repo=r,
|
||||
state=maintenance_drain.STATE_INACTIVE,
|
||||
reason=reason or "operator-exited maintenance drain",
|
||||
requested_by=str(
|
||||
(profile.get("identity") or {}).get("username")
|
||||
or profile.get("expected_username")
|
||||
or ""
|
||||
),
|
||||
requested_by_profile=str(profile.get("profile_name") or ""),
|
||||
session_id=str(session_id or ""),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": [f"exit drain failed: {_redact(str(exc))}"],
|
||||
}
|
||||
record = result.get("record") or {}
|
||||
return {
|
||||
"success": True,
|
||||
"performed": True,
|
||||
"transitioned": bool(result.get("transitioned")),
|
||||
"state": result.get("state"),
|
||||
"prior_state": result.get("prior_state"),
|
||||
"drain_id": result.get("drain_id"),
|
||||
"maintenance_drain": maintenance_drain.status_payload(
|
||||
record, remote=remote, org=o, repo=r
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_request_mcp_restart(
|
||||
remote: str = "dadeschools",
|
||||
@@ -22575,8 +22815,9 @@ def gitea_request_mcp_restart(
|
||||
target_connector: str | None = None,
|
||||
drain_proof_json: str | None = None,
|
||||
request_break_glass: bool = False,
|
||||
prior_recovery_attempts_json: str | None = None,
|
||||
) -> dict:
|
||||
"""Evaluate a proposed MCP restart and return an impact preview (#658).
|
||||
"""Evaluate a proposed MCP restart and return an impact preview (#658/#669).
|
||||
|
||||
Central restart coordinator: resolves the requested restart class, gathers
|
||||
live control-plane state (sessions,
|
||||
@@ -22600,10 +22841,16 @@ def gitea_request_mcp_restart(
|
||||
independent — the drain gate proves the blast radius was drained and knows
|
||||
nothing about whether this requester may request this class — so a class the
|
||||
matrix denied never reports an authorized apply. Break-glass bypasses the
|
||||
drain proof only; it never bypasses the class matrix. ``apply_gate`` carries
|
||||
drain proof and, when env-authorized, the #669 attempt-log requirement for
|
||||
broad restarts; it never bypasses the class matrix. ``apply_gate`` carries
|
||||
``drain_gate_allow`` and ``restart_class_authorized`` so a denial is
|
||||
attributable to the authorization that produced it.
|
||||
|
||||
``prior_recovery_attempts_json`` (#669) is an optional JSON array of prior
|
||||
narrow recovery attempts ``{action, outcome, reason, ...}``. Rolling / full
|
||||
/ host restart classes require at least one *insufficient* narrower attempt
|
||||
unless break-glass is authorized.
|
||||
|
||||
Operator override authority is read from the process environment
|
||||
(``GITEA_OPERATOR_RESTART_OVERRIDE_AUTHORIZATION``), never self-asserted by
|
||||
the requesting session: ``request_override`` only expresses caller intent
|
||||
@@ -22709,12 +22956,37 @@ def gitea_request_mcp_restart(
|
||||
requester_role
|
||||
)
|
||||
|
||||
prior_recovery_attempts: list[dict] = []
|
||||
if prior_recovery_attempts_json:
|
||||
try:
|
||||
parsed_attempts = json.loads(prior_recovery_attempts_json)
|
||||
if isinstance(parsed_attempts, list):
|
||||
prior_recovery_attempts = [
|
||||
dict(a) for a in parsed_attempts if isinstance(a, dict)
|
||||
]
|
||||
else:
|
||||
incomplete_reasons.append(
|
||||
"prior_recovery_attempts_json must be a JSON array (#669)"
|
||||
)
|
||||
inventory_complete = False
|
||||
except (ValueError, TypeError) as exc:
|
||||
incomplete_reasons.append(
|
||||
f"invalid prior_recovery_attempts_json: {_redact(str(exc))}"
|
||||
)
|
||||
inventory_complete = False
|
||||
|
||||
break_glass_authorized = bool(
|
||||
(os.environ.get("GITEA_BREAKGLASS_RESTART_AUTHORIZATION") or "").strip()
|
||||
)
|
||||
break_glass = bool(request_break_glass and break_glass_authorized)
|
||||
|
||||
inventory = {
|
||||
"sessions": sessions,
|
||||
"leases": leases,
|
||||
"terminal_lock": terminal_lock,
|
||||
"inventory_complete": inventory_complete,
|
||||
"incomplete_reasons": incomplete_reasons,
|
||||
"prior_recovery_attempts": prior_recovery_attempts,
|
||||
}
|
||||
|
||||
report = restart_coordinator.evaluate_restart_impact(
|
||||
@@ -22730,6 +23002,7 @@ def gitea_request_mcp_restart(
|
||||
target_session_id=target_session_id,
|
||||
target_role=target_role,
|
||||
target_connector=target_connector,
|
||||
break_glass=break_glass,
|
||||
)
|
||||
|
||||
payload = report.as_dict()
|
||||
@@ -22762,13 +23035,6 @@ def gitea_request_mcp_restart(
|
||||
except (ValueError, TypeError) as exc:
|
||||
proof_parse_error = f"invalid drain_proof_json: {_redact(str(exc))}"
|
||||
|
||||
break_glass_authorized = bool(
|
||||
(
|
||||
os.environ.get("GITEA_BREAKGLASS_RESTART_AUTHORIZATION") or ""
|
||||
).strip()
|
||||
)
|
||||
break_glass = bool(request_break_glass and break_glass_authorized)
|
||||
|
||||
expected_fp = drain_proof.impact_fingerprint(report.as_dict())
|
||||
gate = drain_proof.gate_apply_restart(
|
||||
proof=proof_obj,
|
||||
|
||||
@@ -0,0 +1,281 @@
|
||||
"""Graceful MCP maintenance-drain mode (#659).
|
||||
|
||||
Drain is the visible, capability-gated state that lets an operator stop new
|
||||
work and quiesce mutations *before* a restart, instead of cutting sessions off
|
||||
mid-mutation. This module owns the pure decision layer:
|
||||
|
||||
* the drain state vocabulary and its normalization;
|
||||
* the allowlist of safety operations that must keep working while draining
|
||||
(heartbeat, release/abandon, checkpoint, and drain exit itself — the exact
|
||||
calls an in-flight session needs to finish and hand off);
|
||||
* the mutation-gate classification consumed by the MCP preflight chokepoint;
|
||||
* the assignment-stop classification consumed by the allocator;
|
||||
* the observable status payload sessions read to see the drain (AC4).
|
||||
|
||||
Durable state lives in the control-plane DB (``maintenance_drain`` table);
|
||||
enforcement lives at the existing chokepoints. Nothing here performs I/O, so
|
||||
both callers can share one decision without importing each other.
|
||||
|
||||
Scope note: the machine-verifiable *drain proof* and the restart gate that
|
||||
consumes it are #661's scope, not this module's. Drain here stops assignment
|
||||
and mutation and makes the state observable; it never authorizes a restart.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Mapping
|
||||
|
||||
# ── State vocabulary ──────────────────────────────────────────────────────────
|
||||
|
||||
STATE_INACTIVE = "inactive"
|
||||
STATE_DRAINING = "draining"
|
||||
DRAIN_STATES = frozenset({STATE_INACTIVE, STATE_DRAINING})
|
||||
|
||||
# Typed blocker code surfaced to clients (never a bare string at call sites).
|
||||
BLOCKER_DRAIN_ACTIVE = "maintenance_drain_active"
|
||||
|
||||
# Reason code for the allocator's assignment stop.
|
||||
REASON_ASSIGNMENT_STOPPED = "maintenance_drain_assignment_stopped"
|
||||
|
||||
DRAIN_SCHEMA_VERSION = 6
|
||||
|
||||
|
||||
class MaintenanceDrainError(RuntimeError):
|
||||
"""Raised when a mutation is refused because drain is active (fail closed)."""
|
||||
|
||||
def __init__(self, message: str, *, decision: Mapping[str, Any] | None = None):
|
||||
super().__init__(message)
|
||||
self.decision = dict(decision or {})
|
||||
self.reason_code = BLOCKER_DRAIN_ACTIVE
|
||||
|
||||
|
||||
# ── Safety allowlist ──────────────────────────────────────────────────────────
|
||||
|
||||
# Mutations that stay permitted while draining. Every entry is a *quiesce*
|
||||
# operation: it either proves an in-flight task is still alive, hands its claim
|
||||
# back, records the durable state a restart needs, or ends the drain. Nothing
|
||||
# that creates new work, new branches, new PRs, or new review/merge verdicts is
|
||||
# on this list — that is the whole point of the drain.
|
||||
ALLOWLISTED_DRAIN_TASKS: frozenset[str] = frozenset(
|
||||
{
|
||||
# Liveness of work already in flight.
|
||||
"heartbeat_issue_lock",
|
||||
"heartbeat_reviewer_pr_lease",
|
||||
"post_heartbeat",
|
||||
# Handing claims back so nothing is stranded across the restart.
|
||||
"release_workflow_lease",
|
||||
"release_reviewer_pr_lease",
|
||||
"release_merger_pr_lease",
|
||||
"abandon_workflow_lease",
|
||||
# Durable recovery state (#660) must be writable *during* drain.
|
||||
"write_session_checkpoint",
|
||||
"checkpoint_session",
|
||||
# The drain controls themselves — exit must never be self-blocked.
|
||||
"enter_maintenance_drain",
|
||||
"exit_maintenance_drain",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def normalize_task(task: str | None) -> str:
|
||||
"""Normalize a task name, tolerating the ``gitea_`` tool-name prefix."""
|
||||
name = str(task or "").strip()
|
||||
if name.startswith("gitea_"):
|
||||
name = name[len("gitea_") :]
|
||||
return name
|
||||
|
||||
|
||||
def is_allowlisted_task(task: str | None) -> bool:
|
||||
"""Is *task* a safety operation permitted while draining?"""
|
||||
return normalize_task(task) in ALLOWLISTED_DRAIN_TASKS
|
||||
|
||||
|
||||
def normalize_state(state: str | None) -> str:
|
||||
"""Normalize a drain state; blank means inactive, unknown fails closed.
|
||||
|
||||
Blank normalizes to ``inactive`` (no drain record = not draining), but an
|
||||
unrecognized non-blank value raises: silently treating ``"drainig"`` as
|
||||
inactive would disable the gate.
|
||||
"""
|
||||
value = str(state or "").strip().lower()
|
||||
if not value:
|
||||
return STATE_INACTIVE
|
||||
if value not in DRAIN_STATES:
|
||||
raise MaintenanceDrainError(
|
||||
f"unknown maintenance-drain state {value!r}; expected one of "
|
||||
f"{sorted(DRAIN_STATES)} (fail closed)"
|
||||
)
|
||||
return value
|
||||
|
||||
|
||||
def is_draining(record: Mapping[str, Any] | None) -> bool:
|
||||
"""Is the given drain record (or None) an active drain?"""
|
||||
if not record:
|
||||
return False
|
||||
return normalize_state(record.get("state")) == STATE_DRAINING
|
||||
|
||||
|
||||
# ── Decisions ─────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def classify_mutation(
|
||||
task: str | None,
|
||||
record: Mapping[str, Any] | None,
|
||||
) -> dict[str, Any]:
|
||||
"""Decide whether *task* may mutate under the given drain record.
|
||||
|
||||
Returns a decision dict with ``allowed``/``deferred`` and, when refused, a
|
||||
typed ``reason_code`` plus the one exact next action the caller may take.
|
||||
Deferred (not failed): the operation is legal again after drain exits, so
|
||||
the caller is told to wait rather than to retry a different way.
|
||||
"""
|
||||
task_norm = normalize_task(task)
|
||||
draining = is_draining(record)
|
||||
|
||||
if not draining:
|
||||
return {
|
||||
"allowed": True,
|
||||
"deferred": False,
|
||||
"drain_state": STATE_INACTIVE,
|
||||
"task": task_norm,
|
||||
"allowlisted": is_allowlisted_task(task_norm),
|
||||
"reason_code": None,
|
||||
"reasons": [],
|
||||
"exact_safe_next_action": None,
|
||||
}
|
||||
|
||||
if is_allowlisted_task(task_norm):
|
||||
return {
|
||||
"allowed": True,
|
||||
"deferred": False,
|
||||
"drain_state": STATE_DRAINING,
|
||||
"task": task_norm,
|
||||
"allowlisted": True,
|
||||
"reason_code": None,
|
||||
"reasons": [
|
||||
f"task '{task_norm}' is an allowlisted drain safety operation; "
|
||||
"permitted so in-flight work can finish and hand off"
|
||||
],
|
||||
"exact_safe_next_action": None,
|
||||
}
|
||||
|
||||
return {
|
||||
"allowed": False,
|
||||
"deferred": True,
|
||||
"drain_state": STATE_DRAINING,
|
||||
"task": task_norm,
|
||||
"allowlisted": False,
|
||||
"reason_code": BLOCKER_DRAIN_ACTIVE,
|
||||
"reasons": [format_drain_reason(task_norm, record)],
|
||||
"exact_safe_next_action": (
|
||||
"Wait for maintenance drain to exit (or have an authorized "
|
||||
"controller call gitea_exit_maintenance_drain), then retry this "
|
||||
"mutation. Reads and gitea_maintenance_drain_status stay available."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def classify_assignment(record: Mapping[str, Any] | None) -> dict[str, Any]:
|
||||
"""Decide whether the allocator may assign new work (AC2)."""
|
||||
if not is_draining(record):
|
||||
return {
|
||||
"assignment_allowed": True,
|
||||
"drain_state": STATE_INACTIVE,
|
||||
"reason_code": None,
|
||||
"reasons": [],
|
||||
}
|
||||
return {
|
||||
"assignment_allowed": False,
|
||||
"drain_state": STATE_DRAINING,
|
||||
"reason_code": REASON_ASSIGNMENT_STOPPED,
|
||||
"reasons": [
|
||||
"maintenance drain is active: new work assignment is stopped and "
|
||||
"no lease was created (fail closed, #659)" + _scope_suffix(record)
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def format_drain_reason(task: str | None, record: Mapping[str, Any] | None) -> str:
|
||||
"""Human-readable refusal line for a drained mutation."""
|
||||
task_norm = normalize_task(task) or "(unnamed task)"
|
||||
return (
|
||||
f"maintenance drain is active: mutation '{task_norm}' is deferred; only "
|
||||
"allowlisted drain safety operations "
|
||||
f"({', '.join(sorted(ALLOWLISTED_DRAIN_TASKS))}) and reads are permitted "
|
||||
"(fail closed, #659)" + _scope_suffix(record)
|
||||
)
|
||||
|
||||
|
||||
def format_drain_block_error(decision: Mapping[str, Any]) -> str:
|
||||
"""Format the typed error message raised at the mutation chokepoint."""
|
||||
reasons = list(decision.get("reasons") or [])
|
||||
head = reasons[0] if reasons else "maintenance drain is active (fail closed)"
|
||||
action = decision.get("exact_safe_next_action")
|
||||
return f"{head}. Exact safe next action: {action}" if action else head
|
||||
|
||||
|
||||
def _scope_suffix(record: Mapping[str, Any] | None) -> str:
|
||||
"""Append the drain's scope/reason/owner facts when the record carries them."""
|
||||
if not record:
|
||||
return ""
|
||||
bits: list[str] = []
|
||||
scope = "/".join(
|
||||
str(record.get(key) or "") for key in ("remote", "org", "repo")
|
||||
).strip("/")
|
||||
if scope:
|
||||
bits.append(f"scope {scope}")
|
||||
if record.get("reason"):
|
||||
bits.append(f"reason: {record['reason']}")
|
||||
if record.get("requested_by"):
|
||||
bits.append(f"entered by {record['requested_by']}")
|
||||
if record.get("entered_at"):
|
||||
bits.append(f"at {record['entered_at']}")
|
||||
return f" ({'; '.join(bits)})" if bits else ""
|
||||
|
||||
|
||||
# ── Observability (AC4) ───────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def status_payload(
|
||||
record: Mapping[str, Any] | None,
|
||||
*,
|
||||
remote: str = "",
|
||||
org: str = "",
|
||||
repo: str = "",
|
||||
) -> dict[str, Any]:
|
||||
"""Build the session-observable drain status payload.
|
||||
|
||||
Always answers, including when no drain record exists: an absent record is
|
||||
a definitive "not draining", not an unknown.
|
||||
"""
|
||||
draining = is_draining(record)
|
||||
rec: Mapping[str, Any] = record or {}
|
||||
return {
|
||||
"drain_state": STATE_DRAINING if draining else STATE_INACTIVE,
|
||||
"draining": draining,
|
||||
"remote": str(rec.get("remote") or "") or remote,
|
||||
"org": str(rec.get("org") or "") or org,
|
||||
"repo": str(rec.get("repo") or "") or repo,
|
||||
"reason": str(rec.get("reason") or ""),
|
||||
"requested_by": str(rec.get("requested_by") or ""),
|
||||
"requested_by_profile": str(rec.get("requested_by_profile") or ""),
|
||||
"session_id": str(rec.get("session_id") or ""),
|
||||
"entered_at": str(rec.get("entered_at") or ""),
|
||||
"exited_at": str(rec.get("exited_at") or ""),
|
||||
"assignment_stopped": draining,
|
||||
"mutations_deferred": draining,
|
||||
"allowlisted_tasks": sorted(ALLOWLISTED_DRAIN_TASKS),
|
||||
"reads_permitted": True,
|
||||
"record_present": bool(record),
|
||||
"schema_version": DRAIN_SCHEMA_VERSION,
|
||||
"drain_proof_scope": (
|
||||
"drain proof and the restart gate that consumes it are #661 scope; "
|
||||
"this status never authorizes a restart"
|
||||
),
|
||||
"safe_next_action": (
|
||||
"Wait for drain to exit before retrying deferred mutations; "
|
||||
"allowlisted safety operations and reads remain available."
|
||||
if draining
|
||||
else "None; maintenance drain is not active."
|
||||
),
|
||||
}
|
||||
@@ -0,0 +1,583 @@
|
||||
"""Scoped MCP recovery playbook (#669).
|
||||
|
||||
Operational recovery must prefer the *narrowest* action that can fix the
|
||||
symptom. Full MCP / host restarts are last-resort rungs on a documented
|
||||
ladder; the coordinator refuses those rungs unless a prior attempt log
|
||||
shows narrower recoveries already failed (or break-glass is authorized).
|
||||
|
||||
This module is pure classification and recommendation:
|
||||
|
||||
* No network, filesystem, or process I/O.
|
||||
* Never restarts anything.
|
||||
* Narrow recovery *execution* is delegated to existing tools/docs (linked
|
||||
per rung) — the playbook records which rung to try next and whether
|
||||
escalation to a broad restart is allowed.
|
||||
|
||||
Design lineage: umbrella #655, class matrix #663, coordinator #658,
|
||||
auto-reconnect #584, stale-runtime #610, contamination #630, audit #665.
|
||||
Vision #652 / roadmap #653.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from enum import Enum
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
PLAYBOOK_VERSION = "1.0.0-issue-669"
|
||||
|
||||
# Attempt outcomes that count as "tried and insufficient" for escalation.
|
||||
INSUFFICIENT_OUTCOMES = frozenset(
|
||||
{
|
||||
"failed",
|
||||
"insufficient",
|
||||
"denied",
|
||||
"unresolved",
|
||||
"timeout",
|
||||
"error",
|
||||
}
|
||||
)
|
||||
|
||||
# Break-glass / operator override still records that the ladder was skipped.
|
||||
OUTCOME_BREAK_GLASS = "break_glass"
|
||||
OUTCOME_SUCCESS = "success"
|
||||
OUTCOME_SKIPPED = "skipped"
|
||||
|
||||
|
||||
class RecoveryAction(str, Enum):
|
||||
"""Ordered recovery ladder (narrow → broad)."""
|
||||
|
||||
CLIENT_RECONNECT = "client_reconnect"
|
||||
CAPABILITY_REFRESH = "capability_refresh"
|
||||
SESSION_RECONNECT = "session_reconnect"
|
||||
CONFIGURATION_RELOAD = "configuration_reload"
|
||||
LEASE_RECOVERY = "lease_recovery"
|
||||
WORKER_RESTART = "worker_restart"
|
||||
ROLE_RUNTIME_RESTART = "role_runtime_restart"
|
||||
CONNECTOR_RESTART = "connector_restart"
|
||||
ROLLING_MCP_RESTART = "rolling_mcp_restart"
|
||||
FULL_MCP_RESTART = "full_mcp_restart"
|
||||
HOST_RESTART = "host_restart"
|
||||
|
||||
|
||||
# Classes that require a prior narrow-attempt log (unless break-glass).
|
||||
BROAD_RESTART_ACTIONS: frozenset[RecoveryAction] = frozenset(
|
||||
{
|
||||
RecoveryAction.ROLLING_MCP_RESTART,
|
||||
RecoveryAction.FULL_MCP_RESTART,
|
||||
RecoveryAction.HOST_RESTART,
|
||||
}
|
||||
)
|
||||
|
||||
# Map #663 restart_class strings onto playbook actions.
|
||||
RESTART_CLASS_TO_ACTION: dict[str, RecoveryAction] = {
|
||||
"client_reconnect": RecoveryAction.CLIENT_RECONNECT,
|
||||
"session_reconnect": RecoveryAction.SESSION_RECONNECT,
|
||||
"configuration_reload": RecoveryAction.CONFIGURATION_RELOAD,
|
||||
"worker_restart": RecoveryAction.WORKER_RESTART,
|
||||
"role_runtime_restart": RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||
"connector_restart": RecoveryAction.CONNECTOR_RESTART,
|
||||
"rolling_mcp_restart": RecoveryAction.ROLLING_MCP_RESTART,
|
||||
"full_mcp_restart": RecoveryAction.FULL_MCP_RESTART,
|
||||
"host_restart": RecoveryAction.HOST_RESTART,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RecoveryRung:
|
||||
"""One rung on the recovery ladder."""
|
||||
|
||||
action: RecoveryAction
|
||||
rank: int
|
||||
summary: str
|
||||
# Existing implementation or explicit delegation target.
|
||||
implementation: str
|
||||
issue_links: tuple[str, ...]
|
||||
self_service: bool
|
||||
# Restart-class permission when this rung is requested via coordinator.
|
||||
restart_class: str | None = None
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"action": self.action.value,
|
||||
"rank": self.rank,
|
||||
"summary": self.summary,
|
||||
"implementation": self.implementation,
|
||||
"issue_links": list(self.issue_links),
|
||||
"self_service": self.self_service,
|
||||
"restart_class": self.restart_class,
|
||||
}
|
||||
|
||||
|
||||
# Canonical ladder. Rank 0 is narrowest.
|
||||
RECOVERY_LADDER: tuple[RecoveryRung, ...] = (
|
||||
RecoveryRung(
|
||||
RecoveryAction.CLIENT_RECONNECT,
|
||||
0,
|
||||
"Reconnect the IDE/client MCP transport (EOF / transport flap).",
|
||||
"Host auto-reconnect or explicit client reconnect; "
|
||||
"docs/mcp-namespace-eof-recovery.md",
|
||||
("#584", "#655"),
|
||||
True,
|
||||
"client_reconnect",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.CAPABILITY_REFRESH,
|
||||
1,
|
||||
"Re-resolve task capability and clear stale permission context.",
|
||||
"Delegated: gitea_resolve_task_capability + gitea_whoami "
|
||||
"(no process change).",
|
||||
("#610", "#685", "#655"),
|
||||
True,
|
||||
None,
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.SESSION_RECONNECT,
|
||||
2,
|
||||
"Rebind identity, workspace, and namespace for one session.",
|
||||
"Delegated: gitea_get_runtime_context + explicit worktree_path "
|
||||
"rebind (#618); docs/mcp-namespace-health.md",
|
||||
("#543", "#618", "#655"),
|
||||
True,
|
||||
"session_reconnect",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.CONFIGURATION_RELOAD,
|
||||
3,
|
||||
"Gracefully reload configuration without replacing the daemon.",
|
||||
"restart_coordinator class configuration_reload; console "
|
||||
"system.reload_namespace (#642).",
|
||||
("#642", "#663", "#655"),
|
||||
False,
|
||||
"configuration_reload",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.LEASE_RECOVERY,
|
||||
4,
|
||||
"Recover or rebind stale leases/locks without a process restart.",
|
||||
"Delegated: issue lock recovery / lease lifecycle paths "
|
||||
"(#702, #753, #790).",
|
||||
("#702", "#753", "#790", "#655"),
|
||||
False,
|
||||
None,
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.WORKER_RESTART,
|
||||
5,
|
||||
"Restart one worker after its own lease and mutation scope drains.",
|
||||
"restart_coordinator class worker_restart (#663).",
|
||||
("#663", "#655"),
|
||||
False,
|
||||
"worker_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||
6,
|
||||
"Restart one role runtime and re-probe that namespace only.",
|
||||
"restart_coordinator class role_runtime_restart; console "
|
||||
"system.restart_namespace (#642).",
|
||||
("#642", "#663", "#655"),
|
||||
False,
|
||||
"role_runtime_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.CONNECTOR_RESTART,
|
||||
7,
|
||||
"Restart one connector while unrelated runtimes stay available.",
|
||||
"restart_coordinator class connector_restart (#663).",
|
||||
("#663", "#655"),
|
||||
False,
|
||||
"connector_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.ROLLING_MCP_RESTART,
|
||||
8,
|
||||
"Drain/restart/verify one instance at a time (HA path).",
|
||||
"restart_coordinator class rolling_mcp_restart; design #668.",
|
||||
("#668", "#663", "#655"),
|
||||
False,
|
||||
"rolling_mcp_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.FULL_MCP_RESTART,
|
||||
9,
|
||||
"Full stable-control MCP process restart after verified full drain.",
|
||||
"restart_coordinator class full_mcp_restart; requires attempt log "
|
||||
"unless break-glass (#669).",
|
||||
("#658", "#661", "#663", "#669", "#655"),
|
||||
False,
|
||||
"full_mcp_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.HOST_RESTART,
|
||||
10,
|
||||
"Host/infrastructure restart — broadest last-resort action.",
|
||||
"restart_coordinator class host_restart; operator-owned.",
|
||||
("#663", "#669", "#655"),
|
||||
False,
|
||||
"host_restart",
|
||||
),
|
||||
)
|
||||
|
||||
_LADDER_BY_ACTION: dict[RecoveryAction, RecoveryRung] = {
|
||||
rung.action: rung for rung in RECOVERY_LADDER
|
||||
}
|
||||
|
||||
# Symptom tokens → preferred first rung (decision tree, #663 lineage).
|
||||
SYMPTOM_TO_FIRST_ACTION: dict[str, RecoveryAction] = {
|
||||
"transport_eof": RecoveryAction.CLIENT_RECONNECT,
|
||||
"client_closing_eof": RecoveryAction.CLIENT_RECONNECT,
|
||||
"transport_flap": RecoveryAction.CLIENT_RECONNECT,
|
||||
"namespace_disconnected": RecoveryAction.CLIENT_RECONNECT,
|
||||
"stale_capability": RecoveryAction.CAPABILITY_REFRESH,
|
||||
"permission_stale": RecoveryAction.CAPABILITY_REFRESH,
|
||||
"runtime_reconnect_required": RecoveryAction.CAPABILITY_REFRESH,
|
||||
"stale_runtime": RecoveryAction.SESSION_RECONNECT,
|
||||
"worktree_unbound": RecoveryAction.SESSION_RECONNECT,
|
||||
"namespace_unhealthy": RecoveryAction.SESSION_RECONNECT,
|
||||
"config_drift": RecoveryAction.CONFIGURATION_RELOAD,
|
||||
"profile_misbound": RecoveryAction.CONFIGURATION_RELOAD,
|
||||
"stale_lease": RecoveryAction.LEASE_RECOVERY,
|
||||
"dead_pid_lock": RecoveryAction.LEASE_RECOVERY,
|
||||
"orphan_worktree": RecoveryAction.LEASE_RECOVERY,
|
||||
"single_worker_stuck": RecoveryAction.WORKER_RESTART,
|
||||
"role_runtime_dead": RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||
"connector_dead": RecoveryAction.CONNECTOR_RESTART,
|
||||
"ha_instance_unhealthy": RecoveryAction.ROLLING_MCP_RESTART,
|
||||
"daemon_corrupt": RecoveryAction.FULL_MCP_RESTART,
|
||||
"full_process_deadlock": RecoveryAction.FULL_MCP_RESTART,
|
||||
"host_unresponsive": RecoveryAction.HOST_RESTART,
|
||||
}
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def resolve_action(value: RecoveryAction | str) -> RecoveryAction:
|
||||
"""Resolve a recovery action or fail closed for unknown values."""
|
||||
if isinstance(value, RecoveryAction):
|
||||
return value
|
||||
text = str(value or "").strip()
|
||||
# Accept #663 restart_class aliases.
|
||||
if text in RESTART_CLASS_TO_ACTION:
|
||||
return RESTART_CLASS_TO_ACTION[text]
|
||||
try:
|
||||
return RecoveryAction(text)
|
||||
except ValueError as exc:
|
||||
raise ValueError(
|
||||
f"unknown recovery action {value!r}; deny (fail closed, #669)"
|
||||
) from exc
|
||||
|
||||
|
||||
def ladder_rank(action: RecoveryAction | str) -> int:
|
||||
resolved = resolve_action(action)
|
||||
return _LADDER_BY_ACTION[resolved].rank
|
||||
|
||||
|
||||
def rung_for(action: RecoveryAction | str) -> RecoveryRung:
|
||||
return _LADDER_BY_ACTION[resolve_action(action)]
|
||||
|
||||
|
||||
def normalize_attempt(raw: Mapping[str, Any]) -> dict[str, Any] | None:
|
||||
"""Normalize one prior-recovery attempt record; return None if unusable."""
|
||||
if not isinstance(raw, Mapping):
|
||||
return None
|
||||
action_raw = raw.get("action") or raw.get("recovery_action") or raw.get(
|
||||
"restart_class"
|
||||
)
|
||||
if not action_raw:
|
||||
return None
|
||||
try:
|
||||
action = resolve_action(str(action_raw))
|
||||
except ValueError:
|
||||
return None
|
||||
outcome = str(
|
||||
raw.get("outcome") or raw.get("status") or raw.get("result") or ""
|
||||
).strip().lower()
|
||||
if not outcome:
|
||||
return None
|
||||
recorded_at = raw.get("recorded_at") or raw.get("at") or raw.get("timestamp")
|
||||
reason = str(raw.get("reason") or raw.get("detail") or "").strip()
|
||||
actor = str(raw.get("actor") or raw.get("session_id") or "").strip()
|
||||
return {
|
||||
"action": action.value,
|
||||
"outcome": outcome,
|
||||
"reason": reason,
|
||||
"actor": actor,
|
||||
"recorded_at": recorded_at,
|
||||
"rank": ladder_rank(action),
|
||||
"raw": dict(raw),
|
||||
}
|
||||
|
||||
|
||||
def normalize_attempt_log(
|
||||
attempts: Sequence[Mapping[str, Any]] | None,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Return usable attempt records in ladder order."""
|
||||
out: list[dict[str, Any]] = []
|
||||
for raw in attempts or ():
|
||||
norm = normalize_attempt(raw)
|
||||
if norm is not None:
|
||||
out.append(norm)
|
||||
out.sort(key=lambda a: (a["rank"], str(a.get("recorded_at") or "")))
|
||||
return out
|
||||
|
||||
|
||||
def narrower_insufficient_attempts(
|
||||
attempts: Sequence[Mapping[str, Any]] | None,
|
||||
*,
|
||||
requested: RecoveryAction | str,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Return prior attempts narrower than *requested* that were insufficient."""
|
||||
target_rank = ladder_rank(requested)
|
||||
usable = []
|
||||
for attempt in normalize_attempt_log(attempts):
|
||||
if attempt["rank"] >= target_rank:
|
||||
continue
|
||||
if attempt["outcome"] in INSUFFICIENT_OUTCOMES:
|
||||
usable.append(attempt)
|
||||
return usable
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EscalationAssessment:
|
||||
"""Whether a requested broad recovery may proceed given the attempt log."""
|
||||
|
||||
requested_action: str
|
||||
allowed: bool
|
||||
require_attempt_log: bool
|
||||
break_glass: bool
|
||||
reasons: list[str] = field(default_factory=list)
|
||||
qualifying_attempts: list[dict[str, Any]] = field(default_factory=list)
|
||||
recommended_next: list[dict[str, Any]] = field(default_factory=list)
|
||||
playbook_version: str = PLAYBOOK_VERSION
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"playbook_version": self.playbook_version,
|
||||
"requested_action": self.requested_action,
|
||||
"allowed": self.allowed,
|
||||
"require_attempt_log": self.require_attempt_log,
|
||||
"break_glass": self.break_glass,
|
||||
"reasons": list(self.reasons),
|
||||
"qualifying_attempts": list(self.qualifying_attempts),
|
||||
"recommended_next": list(self.recommended_next),
|
||||
}
|
||||
|
||||
|
||||
def assess_escalation(
|
||||
requested: RecoveryAction | str,
|
||||
*,
|
||||
prior_recovery_attempts: Sequence[Mapping[str, Any]] | None = None,
|
||||
break_glass: bool = False,
|
||||
) -> EscalationAssessment:
|
||||
"""Gate broad restarts on a prior narrow-attempt log (#669 AC3).
|
||||
|
||||
Narrow / mid-ladder actions do not require a prior attempt log.
|
||||
``full_mcp_restart``, ``host_restart``, and ``rolling_mcp_restart``
|
||||
require at least one *insufficient* narrower attempt unless
|
||||
``break_glass`` is true.
|
||||
"""
|
||||
action = resolve_action(requested)
|
||||
require_log = action in BROAD_RESTART_ACTIONS
|
||||
reasons: list[str] = []
|
||||
qualifying = narrower_insufficient_attempts(
|
||||
prior_recovery_attempts, requested=action
|
||||
)
|
||||
|
||||
if not require_log:
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=True,
|
||||
require_attempt_log=False,
|
||||
break_glass=bool(break_glass),
|
||||
reasons=["narrow recovery; attempt log not required"],
|
||||
qualifying_attempts=qualifying,
|
||||
recommended_next=[],
|
||||
)
|
||||
|
||||
if break_glass:
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=True,
|
||||
require_attempt_log=True,
|
||||
break_glass=True,
|
||||
reasons=[
|
||||
"break-glass authorized; broad restart permitted without "
|
||||
"narrow-attempt log (#669)"
|
||||
],
|
||||
qualifying_attempts=qualifying,
|
||||
recommended_next=[],
|
||||
)
|
||||
|
||||
if qualifying:
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=True,
|
||||
require_attempt_log=True,
|
||||
break_glass=False,
|
||||
reasons=[
|
||||
f"{len(qualifying)} narrower recovery attempt(s) recorded as "
|
||||
"insufficient; escalation permitted"
|
||||
],
|
||||
qualifying_attempts=qualifying,
|
||||
recommended_next=[],
|
||||
)
|
||||
|
||||
# Deny: recommend the next untried narrow rung(s).
|
||||
recommended = recommend_actions(
|
||||
symptoms=(),
|
||||
prior_recovery_attempts=prior_recovery_attempts,
|
||||
max_actions=3,
|
||||
)
|
||||
reasons.append(
|
||||
f"{action.value} requires a prior attempt log of insufficient "
|
||||
"narrower recoveries (or break-glass); none found — deny (fail "
|
||||
"closed, #669)"
|
||||
)
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=False,
|
||||
require_attempt_log=True,
|
||||
break_glass=False,
|
||||
reasons=reasons,
|
||||
qualifying_attempts=[],
|
||||
recommended_next=recommended.get("recommended_actions") or [],
|
||||
)
|
||||
|
||||
|
||||
def recommend_actions(
|
||||
*,
|
||||
symptoms: Sequence[str] = (),
|
||||
prior_recovery_attempts: Sequence[Mapping[str, Any]] | None = None,
|
||||
max_actions: int = 5,
|
||||
) -> dict[str, Any]:
|
||||
"""Return ordered recommended recovery actions for the given symptoms.
|
||||
|
||||
Soft mode (rollout): recommendations only — callers decide whether to
|
||||
hard-gate. Hard mode for broad restarts is :func:`assess_escalation`.
|
||||
"""
|
||||
attempted_success = {
|
||||
a["action"]
|
||||
for a in normalize_attempt_log(prior_recovery_attempts)
|
||||
if a["outcome"] == OUTCOME_SUCCESS
|
||||
}
|
||||
attempted_any = {
|
||||
a["action"] for a in normalize_attempt_log(prior_recovery_attempts)
|
||||
}
|
||||
|
||||
first_actions: list[RecoveryAction] = []
|
||||
for symptom in symptoms:
|
||||
key = str(symptom or "").strip().lower().replace(" ", "_").replace("-", "_")
|
||||
mapped = SYMPTOM_TO_FIRST_ACTION.get(key)
|
||||
if mapped is not None and mapped not in first_actions:
|
||||
first_actions.append(mapped)
|
||||
|
||||
# Default entry: client reconnect then walk the ladder.
|
||||
if not first_actions:
|
||||
first_actions = [RecoveryAction.CLIENT_RECONNECT]
|
||||
|
||||
recommended: list[dict[str, Any]] = []
|
||||
seen: set[str] = set()
|
||||
min_rank = min(ladder_rank(a) for a in first_actions)
|
||||
|
||||
for rung in RECOVERY_LADDER:
|
||||
if rung.rank < min_rank:
|
||||
continue
|
||||
if rung.action.value in attempted_success:
|
||||
continue
|
||||
if rung.action.value in seen:
|
||||
continue
|
||||
# Prefer rungs not yet attempted; still list previously-failed ones
|
||||
# only if nothing else remains.
|
||||
entry = rung.as_dict()
|
||||
entry["already_attempted"] = rung.action.value in attempted_any
|
||||
recommended.append(entry)
|
||||
seen.add(rung.action.value)
|
||||
if len(recommended) >= max(1, int(max_actions)):
|
||||
break
|
||||
|
||||
return {
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
"symptoms": [str(s) for s in symptoms],
|
||||
"recommended_actions": recommended,
|
||||
"ladder": [r.as_dict() for r in RECOVERY_LADDER],
|
||||
"read_only": True,
|
||||
"hard_gate_note": (
|
||||
"Broad restarts (rolling/full/host) still require "
|
||||
"assess_escalation / coordinator attempt-log enforcement."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def build_attempt_record(
|
||||
action: RecoveryAction | str,
|
||||
*,
|
||||
outcome: str,
|
||||
reason: str = "",
|
||||
actor: str = "",
|
||||
recorded_at: str | None = None,
|
||||
extra: Mapping[str, Any] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build a durable-shaped attempt log entry for inventory/audit (#665)."""
|
||||
resolved = resolve_action(action)
|
||||
record = {
|
||||
"action": resolved.value,
|
||||
"outcome": str(outcome or "").strip().lower(),
|
||||
"reason": str(reason or "").strip(),
|
||||
"actor": str(actor or "").strip(),
|
||||
"recorded_at": recorded_at or _utc_now().isoformat(),
|
||||
"rank": ladder_rank(resolved),
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
}
|
||||
if extra:
|
||||
record["extra"] = dict(extra)
|
||||
return record
|
||||
|
||||
|
||||
def recovery_metrics(
|
||||
attempts: Sequence[Mapping[str, Any]] | None,
|
||||
) -> dict[str, Any]:
|
||||
"""Compute the fraction of recoveries that avoided full/host restart.
|
||||
|
||||
A recovery *episode* is approximated as one attempt with
|
||||
``outcome=success``. Successes on non-broad rungs count as avoided full
|
||||
restart; successes on full/host count as full-restart recoveries.
|
||||
"""
|
||||
norms = normalize_attempt_log(attempts)
|
||||
successes = [a for a in norms if a["outcome"] == OUTCOME_SUCCESS]
|
||||
broad_success = [
|
||||
a
|
||||
for a in successes
|
||||
if resolve_action(a["action"])
|
||||
in {RecoveryAction.FULL_MCP_RESTART, RecoveryAction.HOST_RESTART}
|
||||
]
|
||||
avoided = [a for a in successes if a not in broad_success]
|
||||
total = len(successes)
|
||||
fraction_avoided = (len(avoided) / total) if total else None
|
||||
return {
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
"attempts_total": len(norms),
|
||||
"successes_total": total,
|
||||
"successes_avoided_full_restart": len(avoided),
|
||||
"successes_full_or_host_restart": len(broad_success),
|
||||
"fraction_avoided_full_restart": fraction_avoided,
|
||||
"insufficient_attempts": sum(
|
||||
1 for a in norms if a["outcome"] in INSUFFICIENT_OUTCOMES
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def ladder_document() -> dict[str, Any]:
|
||||
"""Machine-readable ladder for docs/tools inventory."""
|
||||
return {
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
"parent_issues": ["#655", "#652", "#653"],
|
||||
"enforcement_issue": "#669",
|
||||
"ladder": [r.as_dict() for r in RECOVERY_LADDER],
|
||||
"broad_restart_actions": [a.value for a in sorted(BROAD_RESTART_ACTIONS, key=lambda x: x.value)],
|
||||
"insufficient_outcomes": sorted(INSUFFICIENT_OUTCOMES),
|
||||
"symptom_map": {k: v.value for k, v in sorted(SYMPTOM_TO_FIRST_ACTION.items())},
|
||||
}
|
||||
+46
-2
@@ -1,4 +1,4 @@
|
||||
"""MCP restart coordinator and impact analysis (#658).
|
||||
"""MCP restart coordinator and impact analysis (#658 / #669).
|
||||
|
||||
Before any sanctioned MCP restart, a central coordinator must evaluate the
|
||||
live control-plane state — active sessions, leases/locks, in-flight issue/PR
|
||||
@@ -16,6 +16,9 @@ Design rules (mirrors the read-only posture of ``workflow_dashboard`` /
|
||||
a mutative apply path is a later child gated by a drain proof (non-goal here).
|
||||
* **Fail closed.** If the inventory is not explicitly complete, the verdict is
|
||||
``unsafe`` / deny — an incomplete evaluation must never green-light a restart.
|
||||
* **Narrow-first (#669).** Broad classes (rolling / full / host) require a
|
||||
prior attempt log of insufficient narrower recoveries unless break-glass is
|
||||
authorized. See :mod:`recovery_playbook`.
|
||||
* **No secrets.** Session ids, pids, and profiles are operational metadata, not
|
||||
credentials; nothing secret flows through this module.
|
||||
|
||||
@@ -32,8 +35,9 @@ from enum import Enum
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
import lease_lifecycle
|
||||
import recovery_playbook
|
||||
|
||||
COORDINATOR_VERSION = "1.1.0-issue-663"
|
||||
COORDINATOR_VERSION = "1.2.0-issue-669"
|
||||
|
||||
# Restart verdicts. Exactly the three the acceptance criteria name.
|
||||
VERDICT_SAFE = "safe"
|
||||
@@ -349,6 +353,10 @@ class RestartImpactReport:
|
||||
counts: dict[str, int]
|
||||
audit_record: dict[str, Any]
|
||||
incomplete_reasons: list[str] = field(default_factory=list)
|
||||
# #669 playbook escalation gate (attempt-log enforcement).
|
||||
playbook_escalation: dict[str, Any] = field(default_factory=dict)
|
||||
attempt_log_satisfied: bool = True
|
||||
break_glass: bool = False
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
@@ -382,6 +390,9 @@ class RestartImpactReport:
|
||||
"prior_recovery_attempts": list(self.prior_recovery_attempts),
|
||||
"counts": dict(self.counts),
|
||||
"audit_record": dict(self.audit_record),
|
||||
"playbook_escalation": dict(self.playbook_escalation),
|
||||
"attempt_log_satisfied": self.attempt_log_satisfied,
|
||||
"break_glass": self.break_glass,
|
||||
}
|
||||
|
||||
|
||||
@@ -496,6 +507,7 @@ def evaluate_restart_impact(
|
||||
target_session_id: str | None = None,
|
||||
target_role: str | None = None,
|
||||
target_connector: str | None = None,
|
||||
break_glass: bool = False,
|
||||
) -> RestartImpactReport:
|
||||
"""Evaluate a proposed MCP restart and return an impact preview.
|
||||
|
||||
@@ -584,6 +596,30 @@ def evaluate_restart_impact(
|
||||
dict(a) for a in (inventory.get("prior_recovery_attempts") or [])
|
||||
]
|
||||
|
||||
# #669: broad restarts require a prior narrow-attempt log unless break-glass.
|
||||
playbook_escalation: dict[str, Any] = {}
|
||||
attempt_log_satisfied = True
|
||||
if policy_enforced and resolved_class is not None:
|
||||
try:
|
||||
escalation = recovery_playbook.assess_escalation(
|
||||
resolved_class.value,
|
||||
prior_recovery_attempts=prior_recovery_attempts,
|
||||
break_glass=bool(break_glass),
|
||||
)
|
||||
playbook_escalation = escalation.as_dict()
|
||||
attempt_log_satisfied = bool(escalation.allowed)
|
||||
if not attempt_log_satisfied:
|
||||
authorization_reasons.extend(list(escalation.reasons))
|
||||
except ValueError as exc:
|
||||
# Unknown mapping should never happen for enum values; fail closed.
|
||||
attempt_log_satisfied = False
|
||||
playbook_escalation = {
|
||||
"allowed": False,
|
||||
"reasons": [str(exc)],
|
||||
"playbook_version": recovery_playbook.PLAYBOOK_VERSION,
|
||||
}
|
||||
authorization_reasons.append(str(exc))
|
||||
|
||||
session_impacts = [
|
||||
_classify_session(
|
||||
s,
|
||||
@@ -682,6 +718,7 @@ def evaluate_restart_impact(
|
||||
and role_authorized
|
||||
and approval_satisfied
|
||||
and target_complete
|
||||
and attempt_log_satisfied
|
||||
)
|
||||
|
||||
if policy_enforced and not authorization_ok:
|
||||
@@ -745,6 +782,7 @@ def evaluate_restart_impact(
|
||||
"affected_issues": len(affected_issues),
|
||||
"affected_prs": len(affected_prs),
|
||||
"prior_recovery_attempts": len(prior_recovery_attempts),
|
||||
"attempt_log_satisfied": attempt_log_satisfied,
|
||||
}
|
||||
|
||||
audit_record = {
|
||||
@@ -765,6 +803,9 @@ def evaluate_restart_impact(
|
||||
"allow_restart": allow_restart,
|
||||
"blast_radius": blast_radius,
|
||||
"counts": counts,
|
||||
"attempt_log_satisfied": attempt_log_satisfied,
|
||||
"break_glass": bool(break_glass),
|
||||
"playbook_version": recovery_playbook.PLAYBOOK_VERSION,
|
||||
}
|
||||
|
||||
return RestartImpactReport(
|
||||
@@ -804,4 +845,7 @@ def evaluate_restart_impact(
|
||||
counts=counts,
|
||||
audit_record=audit_record,
|
||||
incomplete_reasons=incomplete_reasons,
|
||||
playbook_escalation=playbook_escalation,
|
||||
attempt_log_satisfied=attempt_log_satisfied,
|
||||
break_glass=bool(break_glass),
|
||||
)
|
||||
|
||||
@@ -414,6 +414,24 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"role": "controller",
|
||||
},
|
||||
|
||||
# #659 maintenance drain. Same reasoning as the lifecycle controls above:
|
||||
# entering/exiting drain quiesces a whole namespace, so it carries a
|
||||
# non-``gitea.*`` permission that no configured Gitea profile satisfies by
|
||||
# accident (AC1 — capability-gated and audited). Reading drain state is
|
||||
# ordinary read authority: every session must be able to see the drain (AC4).
|
||||
"enter_maintenance_drain": {
|
||||
"permission": "runtime.maintenance_drain",
|
||||
"role": "controller",
|
||||
},
|
||||
"exit_maintenance_drain": {
|
||||
"permission": "runtime.maintenance_drain",
|
||||
"role": "controller",
|
||||
},
|
||||
"maintenance_drain_status": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
|
||||
# #601 first-class lease lifecycle — inspect/list need read; mutations gate on
|
||||
# ownership in the control-plane DB (not a separate Gitea write permission).
|
||||
"list_workflow_leases": {
|
||||
|
||||
@@ -35,6 +35,17 @@ BREAK_GLASS_ENV = "GITEA_BREAKGLASS_RESTART_AUTHORIZATION"
|
||||
QUIET_SESSIONS: list[dict] = []
|
||||
QUIET_LEASES: list[dict] = []
|
||||
|
||||
# #669: broad restarts need a prior narrow-attempt log (unless break-glass).
|
||||
PRIOR_NARROW_ATTEMPTS_JSON = json.dumps(
|
||||
[
|
||||
{
|
||||
"action": "client_reconnect",
|
||||
"outcome": "insufficient",
|
||||
"reason": "still flapping after reconnect",
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
class _FakeDB:
|
||||
"""Minimal control-plane DB stand-in for the restart inventory."""
|
||||
@@ -128,6 +139,7 @@ class TestConjunction(_RestartToolHarness):
|
||||
preview = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
self.assertTrue(preview["allow_restart"],
|
||||
@@ -136,6 +148,7 @@ class TestConjunction(_RestartToolHarness):
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
@@ -342,6 +355,7 @@ class TestExistingPathsStillWork(_RestartToolHarness):
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||
dry_run=False,
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
|
||||
@@ -0,0 +1,237 @@
|
||||
"""Tests for graceful MCP maintenance-drain mode (#659).
|
||||
|
||||
Acceptance coverage:
|
||||
|
||||
1. Enter/exit is durable and audited (DB substrate).
|
||||
2. New work assignment stops during drain (allocator WAIT).
|
||||
3. Mutations deferred except allowlisted safety ops.
|
||||
4. Sessions can observe drain state.
|
||||
5. Fail-closed on unreadable drain state.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
import maintenance_drain
|
||||
from control_plane_db import ControlPlaneDB
|
||||
from allocator_service import WorkCandidate, allocate_next_work, OUTCOME_WAIT
|
||||
|
||||
|
||||
class TestDrainDecisions(unittest.TestCase):
|
||||
def test_inactive_allows_mutations_and_assignment(self):
|
||||
decision = maintenance_drain.classify_mutation("create_pr", None)
|
||||
self.assertTrue(decision["allowed"])
|
||||
self.assertFalse(decision["deferred"])
|
||||
assign = maintenance_drain.classify_assignment(None)
|
||||
self.assertTrue(assign["assignment_allowed"])
|
||||
|
||||
def test_draining_defers_non_allowlisted_mutation(self):
|
||||
record = {"state": "draining", "remote": "prgs", "org": "o", "repo": "r"}
|
||||
decision = maintenance_drain.classify_mutation("create_pr", record)
|
||||
self.assertFalse(decision["allowed"])
|
||||
self.assertTrue(decision["deferred"])
|
||||
self.assertEqual(decision["reason_code"], maintenance_drain.BLOCKER_DRAIN_ACTIVE)
|
||||
self.assertIn("create_pr", decision["reasons"][0])
|
||||
|
||||
def test_allowlisted_safety_ops_pass_during_drain(self):
|
||||
record = {"state": "draining"}
|
||||
for task in (
|
||||
"heartbeat_issue_lock",
|
||||
"gitea_release_reviewer_pr_lease",
|
||||
"write_session_checkpoint",
|
||||
"exit_maintenance_drain",
|
||||
):
|
||||
with self.subTest(task=task):
|
||||
decision = maintenance_drain.classify_mutation(task, record)
|
||||
self.assertTrue(decision["allowed"], decision)
|
||||
|
||||
def test_assignment_stopped_during_drain(self):
|
||||
record = {"state": "draining", "reason": "upgrade"}
|
||||
decision = maintenance_drain.classify_assignment(record)
|
||||
self.assertFalse(decision["assignment_allowed"])
|
||||
self.assertEqual(
|
||||
decision["reason_code"], maintenance_drain.REASON_ASSIGNMENT_STOPPED
|
||||
)
|
||||
|
||||
def test_unknown_state_fails_closed(self):
|
||||
with self.assertRaises(maintenance_drain.MaintenanceDrainError):
|
||||
maintenance_drain.normalize_state("drainig")
|
||||
|
||||
def test_status_payload_always_answers(self):
|
||||
inactive = maintenance_drain.status_payload(None, remote="prgs", org="o", repo="r")
|
||||
self.assertFalse(inactive["draining"])
|
||||
self.assertTrue(inactive["reads_permitted"])
|
||||
active = maintenance_drain.status_payload(
|
||||
{"state": "draining", "reason": "reboot", "requested_by": "ops"},
|
||||
remote="prgs",
|
||||
org="o",
|
||||
repo="r",
|
||||
)
|
||||
self.assertTrue(active["draining"])
|
||||
self.assertTrue(active["assignment_stopped"])
|
||||
self.assertTrue(active["mutations_deferred"])
|
||||
self.assertIn("heartbeat_issue_lock", active["allowlisted_tasks"])
|
||||
|
||||
|
||||
class TestDrainDB(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.db = ControlPlaneDB(db_path=str(Path(self._tmp.name) / "cp.sqlite3"))
|
||||
|
||||
def test_enter_exit_idempotent_and_audited(self):
|
||||
first = self.db.set_maintenance_drain(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
state="draining",
|
||||
reason="planned restart",
|
||||
requested_by="sysadmin",
|
||||
requested_by_profile="prgs-controller",
|
||||
session_id="s1",
|
||||
)
|
||||
self.assertTrue(first["transitioned"])
|
||||
self.assertEqual(first["state"], "draining")
|
||||
self.assertTrue(maintenance_drain.is_draining(first["record"]))
|
||||
|
||||
again = self.db.set_maintenance_drain(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
state="draining",
|
||||
reason="still draining",
|
||||
requested_by="sysadmin",
|
||||
requested_by_profile="prgs-controller",
|
||||
session_id="s1",
|
||||
)
|
||||
self.assertFalse(again["transitioned"])
|
||||
self.assertEqual(again["record"]["entered_at"], first["record"]["entered_at"])
|
||||
|
||||
exited = self.db.set_maintenance_drain(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
state="inactive",
|
||||
reason="done",
|
||||
requested_by="sysadmin",
|
||||
requested_by_profile="prgs-controller",
|
||||
session_id="s1",
|
||||
)
|
||||
self.assertTrue(exited["transitioned"])
|
||||
self.assertFalse(maintenance_drain.is_draining(exited["record"]))
|
||||
self.assertTrue(exited["record"]["exited_at"])
|
||||
|
||||
# Events recorded for transitions only (enter + exit).
|
||||
with self.db._tx(immediate=False) as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT event_type FROM events WHERE event_type LIKE 'maintenance_drain_%' "
|
||||
"ORDER BY event_id"
|
||||
).fetchall()
|
||||
types = [r[0] for r in rows]
|
||||
self.assertEqual(types, ["maintenance_drain_enter", "maintenance_drain_exit"])
|
||||
|
||||
def test_read_missing_is_none_not_error(self):
|
||||
self.assertIsNone(
|
||||
self.db.read_maintenance_drain(remote="prgs", org="o", repo="r")
|
||||
)
|
||||
|
||||
|
||||
class TestAllocatorStopsDuringDrain(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.db = ControlPlaneDB(db_path=str(Path(self._tmp.name) / "cp.sqlite3"))
|
||||
|
||||
def test_allocate_returns_wait_while_draining(self):
|
||||
self.db.set_maintenance_drain(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
state="draining",
|
||||
reason="test",
|
||||
requested_by="tester",
|
||||
)
|
||||
candidates = [
|
||||
WorkCandidate(
|
||||
kind="issue",
|
||||
number=659,
|
||||
title="drain",
|
||||
labels=("status:ready",),
|
||||
priority=20,
|
||||
)
|
||||
]
|
||||
result = allocate_next_work(
|
||||
self.db,
|
||||
role="author",
|
||||
session_id="test-session",
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
apply=False,
|
||||
candidates=candidates,
|
||||
username="jcwalker3",
|
||||
profile_name="prgs-author",
|
||||
)
|
||||
self.assertEqual(result["outcome"], OUTCOME_WAIT)
|
||||
self.assertIsNone(result.get("selected"))
|
||||
self.assertEqual(
|
||||
result.get("reason_code"),
|
||||
maintenance_drain.REASON_ASSIGNMENT_STOPPED,
|
||||
)
|
||||
self.assertTrue(result["maintenance_drain"]["draining"])
|
||||
|
||||
def test_allocate_works_when_inactive(self):
|
||||
candidates = [
|
||||
WorkCandidate(
|
||||
kind="issue",
|
||||
number=659,
|
||||
title="drain",
|
||||
labels=("status:ready",),
|
||||
priority=20,
|
||||
)
|
||||
]
|
||||
result = allocate_next_work(
|
||||
self.db,
|
||||
role="author",
|
||||
session_id="test-session-2",
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
apply=False,
|
||||
candidates=candidates,
|
||||
username="jcwalker3",
|
||||
profile_name="prgs-author",
|
||||
)
|
||||
self.assertNotEqual(
|
||||
result.get("reason_code"),
|
||||
maintenance_drain.REASON_ASSIGNMENT_STOPPED,
|
||||
)
|
||||
|
||||
|
||||
class TestCapabilityMap(unittest.TestCase):
|
||||
def test_drain_tasks_mapped(self):
|
||||
import task_capability_map as tcm
|
||||
|
||||
self.assertEqual(
|
||||
tcm.required_permission("enter_maintenance_drain"),
|
||||
"runtime.maintenance_drain",
|
||||
)
|
||||
self.assertEqual(
|
||||
tcm.required_permission("exit_maintenance_drain"),
|
||||
"runtime.maintenance_drain",
|
||||
)
|
||||
self.assertEqual(
|
||||
tcm.required_permission("maintenance_drain_status"),
|
||||
"gitea.read",
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,217 @@
|
||||
"""Unit tests for the scoped recovery playbook (#669)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import recovery_playbook as rp
|
||||
import restart_coordinator as rc
|
||||
|
||||
|
||||
def test_ladder_covers_eleven_ordered_rungs():
|
||||
ranks = [r.rank for r in rp.RECOVERY_LADDER]
|
||||
assert ranks == list(range(len(rp.RECOVERY_LADDER)))
|
||||
assert len(rp.RECOVERY_LADDER) == 11
|
||||
assert rp.RECOVERY_LADDER[0].action is rp.RecoveryAction.CLIENT_RECONNECT
|
||||
assert rp.RECOVERY_LADDER[-1].action is rp.RecoveryAction.HOST_RESTART
|
||||
|
||||
|
||||
def test_ladder_document_links_parent_issues():
|
||||
doc = rp.ladder_document()
|
||||
assert "#655" in doc["parent_issues"]
|
||||
assert "#652" in doc["parent_issues"]
|
||||
assert "#653" in doc["parent_issues"]
|
||||
assert doc["enforcement_issue"] == "#669"
|
||||
assert "full_mcp_restart" in doc["broad_restart_actions"]
|
||||
|
||||
|
||||
def test_recommend_transport_eof_starts_at_client_reconnect():
|
||||
plan = rp.recommend_actions(symptoms=["transport_eof"])
|
||||
assert plan["recommended_actions"][0]["action"] == "client_reconnect"
|
||||
assert plan["recommended_actions"][0]["issue_links"]
|
||||
|
||||
|
||||
def test_recommend_skips_successful_prior_attempts():
|
||||
attempts = [
|
||||
rp.build_attempt_record(
|
||||
"client_reconnect", outcome="success", reason="reconnected"
|
||||
)
|
||||
]
|
||||
plan = rp.recommend_actions(
|
||||
symptoms=["transport_eof"], prior_recovery_attempts=attempts
|
||||
)
|
||||
actions = [a["action"] for a in plan["recommended_actions"]]
|
||||
assert "client_reconnect" not in actions
|
||||
assert actions[0] == "capability_refresh"
|
||||
|
||||
|
||||
def test_escalation_denied_without_attempt_log():
|
||||
result = rp.assess_escalation("full_mcp_restart", prior_recovery_attempts=[])
|
||||
assert result.allowed is False
|
||||
assert result.require_attempt_log is True
|
||||
assert any("#669" in r for r in result.reasons)
|
||||
assert result.recommended_next # soft recommendations still provided
|
||||
|
||||
|
||||
def test_escalation_allowed_after_insufficient_narrower():
|
||||
attempts = [
|
||||
rp.build_attempt_record(
|
||||
"client_reconnect",
|
||||
outcome="insufficient",
|
||||
reason="still flapping",
|
||||
),
|
||||
rp.build_attempt_record(
|
||||
"session_reconnect",
|
||||
outcome="failed",
|
||||
reason="namespace still dead",
|
||||
),
|
||||
]
|
||||
result = rp.assess_escalation(
|
||||
"full_mcp_restart", prior_recovery_attempts=attempts
|
||||
)
|
||||
assert result.allowed is True
|
||||
assert len(result.qualifying_attempts) == 2
|
||||
|
||||
|
||||
def test_escalation_break_glass_bypasses_attempt_log():
|
||||
result = rp.assess_escalation(
|
||||
"host_restart", prior_recovery_attempts=[], break_glass=True
|
||||
)
|
||||
assert result.allowed is True
|
||||
assert result.break_glass is True
|
||||
|
||||
|
||||
def test_narrow_action_does_not_require_attempt_log():
|
||||
result = rp.assess_escalation(
|
||||
"client_reconnect", prior_recovery_attempts=[]
|
||||
)
|
||||
assert result.allowed is True
|
||||
assert result.require_attempt_log is False
|
||||
|
||||
|
||||
def test_same_rank_attempt_does_not_qualify_for_escalation():
|
||||
attempts = [
|
||||
rp.build_attempt_record(
|
||||
"full_mcp_restart", outcome="failed", reason="already failed full"
|
||||
)
|
||||
]
|
||||
result = rp.assess_escalation(
|
||||
"full_mcp_restart", prior_recovery_attempts=attempts
|
||||
)
|
||||
assert result.allowed is False
|
||||
|
||||
|
||||
def test_success_outcome_does_not_qualify_for_escalation():
|
||||
attempts = [
|
||||
rp.build_attempt_record(
|
||||
"client_reconnect", outcome="success", reason="fixed"
|
||||
)
|
||||
]
|
||||
result = rp.assess_escalation(
|
||||
"full_mcp_restart", prior_recovery_attempts=attempts
|
||||
)
|
||||
assert result.allowed is False
|
||||
|
||||
|
||||
def test_recovery_metrics_fraction_avoided():
|
||||
attempts = [
|
||||
rp.build_attempt_record("client_reconnect", outcome="success"),
|
||||
rp.build_attempt_record("session_reconnect", outcome="success"),
|
||||
rp.build_attempt_record("full_mcp_restart", outcome="success"),
|
||||
]
|
||||
metrics = rp.recovery_metrics(attempts)
|
||||
assert metrics["successes_total"] == 3
|
||||
assert metrics["successes_avoided_full_restart"] == 2
|
||||
assert metrics["successes_full_or_host_restart"] == 1
|
||||
assert abs(metrics["fraction_avoided_full_restart"] - (2 / 3)) < 1e-9
|
||||
|
||||
|
||||
def test_coordinator_denies_full_restart_without_attempt_log():
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"prior_recovery_attempts": [],
|
||||
}
|
||||
report = rc.evaluate_restart_impact(
|
||||
inv,
|
||||
restart_class=rc.RestartClass.FULL_MCP_RESTART,
|
||||
requester_role="controller",
|
||||
requester_permissions=rc.permissions_for_role("controller"),
|
||||
controller_approved=True,
|
||||
operator_authorized=True,
|
||||
)
|
||||
assert report.allow_restart is False
|
||||
assert report.attempt_log_satisfied is False
|
||||
assert report.verdict == rc.VERDICT_UNSAFE
|
||||
blob = " ".join(report.reasons + report.authorization_reasons)
|
||||
assert "#669" in blob or "attempt log" in blob
|
||||
|
||||
|
||||
def test_coordinator_allows_full_restart_with_attempt_log():
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"prior_recovery_attempts": [
|
||||
{
|
||||
"action": "client_reconnect",
|
||||
"outcome": "insufficient",
|
||||
"reason": "still broken",
|
||||
}
|
||||
],
|
||||
}
|
||||
report = rc.evaluate_restart_impact(
|
||||
inv,
|
||||
restart_class=rc.RestartClass.FULL_MCP_RESTART,
|
||||
requester_role="controller",
|
||||
requester_permissions=rc.permissions_for_role("controller"),
|
||||
controller_approved=True,
|
||||
operator_authorized=True,
|
||||
)
|
||||
assert report.attempt_log_satisfied is True
|
||||
assert report.allow_restart is True
|
||||
assert report.verdict == rc.VERDICT_SAFE
|
||||
|
||||
|
||||
def test_coordinator_break_glass_allows_without_log():
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"prior_recovery_attempts": [],
|
||||
}
|
||||
report = rc.evaluate_restart_impact(
|
||||
inv,
|
||||
restart_class=rc.RestartClass.FULL_MCP_RESTART,
|
||||
requester_role="controller",
|
||||
requester_permissions=rc.permissions_for_role("controller"),
|
||||
controller_approved=True,
|
||||
operator_authorized=True,
|
||||
break_glass=True,
|
||||
)
|
||||
assert report.break_glass is True
|
||||
assert report.attempt_log_satisfied is True
|
||||
assert report.allow_restart is True
|
||||
|
||||
|
||||
def test_coordinator_client_reconnect_unaffected():
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"prior_recovery_attempts": [],
|
||||
}
|
||||
report = rc.evaluate_restart_impact(
|
||||
inv,
|
||||
restart_class=rc.RestartClass.CLIENT_RECONNECT,
|
||||
requester_role="author",
|
||||
requester_permissions=rc.permissions_for_role("author"),
|
||||
)
|
||||
assert report.attempt_log_satisfied is True
|
||||
assert report.allow_restart is True
|
||||
|
||||
|
||||
def test_restart_class_alias_accepted():
|
||||
assert (
|
||||
rp.resolve_action("full_mcp_restart")
|
||||
is rp.RecoveryAction.FULL_MCP_RESTART
|
||||
)
|
||||
@@ -0,0 +1,465 @@
|
||||
"""Unit tests for Phase 3 Notifications and Human-Attention Console (#648)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from webui.app import create_app
|
||||
from webui.notifications import (
|
||||
ATTENTION_HUMAN_REQUIRED,
|
||||
ATTENTION_OPERATOR,
|
||||
ATTENTION_ROUTINE,
|
||||
CATEGORY_AUTH,
|
||||
CATEGORY_BLOCKER,
|
||||
CATEGORY_LEASE,
|
||||
CATEGORY_SYSTEM,
|
||||
CATEGORY_VALIDATION,
|
||||
CATEGORY_WORKFLOW,
|
||||
NotificationItem,
|
||||
NotificationSnapshot,
|
||||
classify_attention_event,
|
||||
load_notifications_snapshot,
|
||||
snapshot_to_dict,
|
||||
)
|
||||
from webui.notification_views import render_notifications_page
|
||||
from webui.project_registry import load_registry
|
||||
from webui.queue_loader import QueueItem, QueueSnapshot
|
||||
from webui.lease_loader import CollisionWarning, LeaseSnapshot
|
||||
from webui.system_health import DependencyProbe, SystemHealthSnapshot, VersionInfo, StaleRuntime
|
||||
|
||||
|
||||
def test_classify_attention_event_rules():
|
||||
# 1. Critical escalation boundaries -> human-required
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_AUTH, "Auth error", "Unauthorized access attempt", is_auth_failure=True
|
||||
)
|
||||
assert att_cls == ATTENTION_HUMAN_REQUIRED
|
||||
assert req_human is True
|
||||
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_SYSTEM, "Hard stop", "Hard stop triggered", is_hard_stop=True
|
||||
)
|
||||
assert att_cls == ATTENTION_HUMAN_REQUIRED
|
||||
assert req_human is True
|
||||
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_VALIDATION, "Validation Error", "Report validation failed", is_validation_failure=True
|
||||
)
|
||||
assert att_cls == ATTENTION_HUMAN_REQUIRED
|
||||
assert req_human is True
|
||||
|
||||
# 2. Operational issues -> operator
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_BLOCKER, "PR Blocked", "Merge conflict detected", is_blocker=True
|
||||
)
|
||||
assert att_cls == ATTENTION_OPERATOR
|
||||
assert req_human is False
|
||||
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_LEASE, "Lease Expired", "Session lease expired", is_stale=True
|
||||
)
|
||||
assert att_cls == ATTENTION_OPERATOR
|
||||
assert req_human is False
|
||||
|
||||
# 3. Routine workflow transitions -> routine
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_WORKFLOW, "PR Active", "PR in review"
|
||||
)
|
||||
assert att_cls == ATTENTION_ROUTINE
|
||||
assert req_human is False
|
||||
|
||||
|
||||
def test_notification_snapshot_aggregation():
|
||||
reg = load_registry()
|
||||
proj_id = reg.projects[0].id if reg.projects else "gitea-tools"
|
||||
|
||||
mock_queue = QueueSnapshot(
|
||||
project_id=proj_id,
|
||||
repo_label="org/repo",
|
||||
prs=(
|
||||
QueueItem(
|
||||
number=101,
|
||||
title="Blocked PR",
|
||||
badges=("blocked",),
|
||||
extra={},
|
||||
),
|
||||
QueueItem(
|
||||
number=102,
|
||||
title="Normal PR",
|
||||
badges=("in-review",),
|
||||
extra={},
|
||||
),
|
||||
),
|
||||
issues=(),
|
||||
pr_pagination=None,
|
||||
issue_pagination=None,
|
||||
)
|
||||
|
||||
mock_leases = LeaseSnapshot(
|
||||
project_id=proj_id,
|
||||
repo_label="org/repo",
|
||||
issue_lock=None,
|
||||
claim_inventory={},
|
||||
reviewer_leases=(
|
||||
{
|
||||
"pr_number": 101,
|
||||
"status": "expired",
|
||||
"is_expired": True,
|
||||
},
|
||||
),
|
||||
duplicate_prs=(
|
||||
CollisionWarning(
|
||||
kind="duplicate_pr",
|
||||
message="Multiple open PRs for issue #101",
|
||||
issue_number=101,
|
||||
pr_numbers=(101, 103),
|
||||
),
|
||||
),
|
||||
duplicate_branches=(),
|
||||
collision_history=(),
|
||||
fetch_error=None,
|
||||
)
|
||||
|
||||
mock_version = VersionInfo(
|
||||
git_sha="abc1234",
|
||||
git_describe="v1.0.0",
|
||||
control_plane_schema_version=1,
|
||||
python_version="3.11",
|
||||
known=True,
|
||||
)
|
||||
|
||||
mock_stale = StaleRuntime(
|
||||
daemon_head="abc1234",
|
||||
checkout_head="abc1234",
|
||||
remote_head="abc1234",
|
||||
stale=False,
|
||||
determinable=True,
|
||||
mutation_safe=True,
|
||||
reasons=(),
|
||||
)
|
||||
|
||||
mock_health = SystemHealthSnapshot(
|
||||
status="degraded",
|
||||
ready=False,
|
||||
readiness_complete=True,
|
||||
readiness_reasons=("Auth failure",),
|
||||
service="webui",
|
||||
mode="test",
|
||||
version=mock_version,
|
||||
started_at="2026-07-25T00:00:00Z",
|
||||
uptime_seconds=100.0,
|
||||
timestamp="2026-07-25T00:00:00Z",
|
||||
deep_probes_requested=True,
|
||||
dependencies=(
|
||||
DependencyProbe(
|
||||
name="auth_service",
|
||||
kind="auth",
|
||||
status="unauthorized",
|
||||
detail="Token expired",
|
||||
required=True,
|
||||
),
|
||||
),
|
||||
mcp_namespaces=(),
|
||||
stale_runtime=mock_stale,
|
||||
probe_errors=(),
|
||||
)
|
||||
|
||||
snapshot = load_notifications_snapshot(
|
||||
proj_id,
|
||||
load_queue=lambda _id: mock_queue,
|
||||
load_leases=lambda **_kwargs: mock_leases,
|
||||
load_health=lambda **_kwargs: mock_health,
|
||||
)
|
||||
|
||||
assert snapshot.project_id == proj_id
|
||||
assert snapshot.total_count == 5
|
||||
assert snapshot.human_required_count >= 1 # auth probe failure
|
||||
assert snapshot.operator_count >= 3 # blocked PR + expired lease + duplicate PR collision
|
||||
assert snapshot.routine_count >= 1 # normal PR
|
||||
|
||||
# Inbox items should include operator and human-required items only
|
||||
inbox_classes = {item.attention_class for item in snapshot.inbox_items}
|
||||
assert ATTENTION_ROUTINE not in inbox_classes
|
||||
assert ATTENTION_OPERATOR in inbox_classes
|
||||
assert ATTENTION_HUMAN_REQUIRED in inbox_classes
|
||||
|
||||
|
||||
def test_snapshot_to_dict_and_redaction():
|
||||
item = NotificationItem(
|
||||
id="notif-1",
|
||||
attention_class=ATTENTION_HUMAN_REQUIRED,
|
||||
category=CATEGORY_AUTH,
|
||||
title="Auth Error",
|
||||
summary="Failed auth header: Bearer secret_token_12345",
|
||||
work_kind="system",
|
||||
work_number=None,
|
||||
project_id="test-proj",
|
||||
repo_label="org/repo",
|
||||
created_at="2026-07-25T16:00:00Z",
|
||||
requires_human=True,
|
||||
)
|
||||
snap = NotificationSnapshot(
|
||||
project_id="test-proj",
|
||||
repo_label="org/repo",
|
||||
items=(item,),
|
||||
human_required_count=1,
|
||||
operator_count=0,
|
||||
routine_count=0,
|
||||
total_count=1,
|
||||
)
|
||||
|
||||
data = snapshot_to_dict(snap)
|
||||
assert data["project_id"] == "test-proj"
|
||||
assert data["human_required_count"] == 1
|
||||
assert len(data["inbox_items"]) == 1
|
||||
|
||||
# Redaction test
|
||||
summary = data["inbox_items"][0]["summary"]
|
||||
assert "secret_token_12345" not in summary
|
||||
assert "<redacted>" in summary or "Bearer" in summary
|
||||
|
||||
|
||||
def test_notifications_html_views():
|
||||
item = NotificationItem(
|
||||
id="notif-1",
|
||||
attention_class=ATTENTION_HUMAN_REQUIRED,
|
||||
category=CATEGORY_AUTH,
|
||||
title="Critical Auth Failure",
|
||||
summary="Auth failure details",
|
||||
work_kind="issue",
|
||||
work_number=42,
|
||||
project_id="test-proj",
|
||||
repo_label="org/repo",
|
||||
created_at="2026-07-25T16:00:00Z",
|
||||
requires_human=True,
|
||||
)
|
||||
snap = NotificationSnapshot(
|
||||
project_id="test-proj",
|
||||
repo_label="org/repo",
|
||||
items=(item,),
|
||||
human_required_count=1,
|
||||
operator_count=0,
|
||||
routine_count=0,
|
||||
total_count=1,
|
||||
)
|
||||
|
||||
html = render_notifications_page(snap, filter_class="inbox")
|
||||
assert "Notifications & Attention Inbox" in html or "Notifications & Attention Inbox" in html
|
||||
assert "Critical Auth Failure" in html
|
||||
assert "HUMAN REQUIRED" in html
|
||||
assert "Human Required" in html
|
||||
|
||||
|
||||
def test_notifications_app_routes():
|
||||
app = create_app()
|
||||
client = TestClient(app)
|
||||
|
||||
# 1. HTML Route
|
||||
res = client.get("/notifications")
|
||||
assert res.status_code == 200
|
||||
assert "Notifications" in res.text
|
||||
assert "Attention Inbox" in res.text
|
||||
|
||||
# 2. API Route /api/v1/notifications
|
||||
res_api = client.get("/api/v1/notifications")
|
||||
assert res_api.status_code == 200
|
||||
json_data = res_api.json()
|
||||
assert "human_required_count" in json_data
|
||||
assert "operator_count" in json_data
|
||||
assert "routine_count" in json_data
|
||||
assert "inbox_items" in json_data
|
||||
|
||||
# 3. Compatibility Alias /api/notifications
|
||||
res_alias = client.get("/api/notifications")
|
||||
assert res_alias.status_code == 200
|
||||
assert res_alias.json()["project_id"] == json_data["project_id"]
|
||||
|
||||
|
||||
def test_classify_ignores_human_authored_title_and_summary_keywords():
|
||||
"""B1: keywords in human-authored titles must not escalate routine work (#905)."""
|
||||
# Routine transition whose title/summary mention critical-boundary words
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_WORKFLOW,
|
||||
"record irrecoverable decision lock provenance",
|
||||
"PR #999 'record irrecoverable decision lock provenance' is in routine state in-review.",
|
||||
)
|
||||
assert att_cls == ATTENTION_ROUTINE
|
||||
assert req_human is False
|
||||
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_WORKFLOW,
|
||||
"fix unauthorized token path",
|
||||
"Issue #1 'fix unauthorized token path' state: claimed. hard stop docs only.",
|
||||
)
|
||||
assert att_cls == ATTENTION_ROUTINE
|
||||
assert req_human is False
|
||||
|
||||
# Structured flags still escalate (machine-driven)
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_SYSTEM,
|
||||
"anything",
|
||||
"anything with hard stop in text",
|
||||
is_hard_stop=True,
|
||||
)
|
||||
assert att_cls == ATTENTION_HUMAN_REQUIRED
|
||||
assert req_human is True
|
||||
|
||||
|
||||
def test_notification_ids_are_unique_across_probe_errors_and_collisions():
|
||||
"""B2: published notification ids must be unique within a snapshot (#905)."""
|
||||
reg = load_registry()
|
||||
proj_id = reg.projects[0].id if reg.projects else "gitea-tools"
|
||||
|
||||
mock_queue = QueueSnapshot(
|
||||
project_id=proj_id,
|
||||
repo_label="org/repo",
|
||||
prs=(),
|
||||
issues=(),
|
||||
pr_pagination=None,
|
||||
issue_pagination=None,
|
||||
)
|
||||
mock_leases = LeaseSnapshot(
|
||||
project_id=proj_id,
|
||||
repo_label="org/repo",
|
||||
issue_lock=None,
|
||||
claim_inventory={},
|
||||
reviewer_leases=(),
|
||||
duplicate_prs=(
|
||||
CollisionWarning(
|
||||
kind="duplicate_pr",
|
||||
message="Multiple open PRs for issue #10",
|
||||
issue_number=10,
|
||||
pr_numbers=(10, 11),
|
||||
),
|
||||
CollisionWarning(
|
||||
kind="duplicate_branch",
|
||||
message="Another collision without issue",
|
||||
issue_number=None,
|
||||
pr_numbers=(12, 13),
|
||||
),
|
||||
CollisionWarning(
|
||||
kind="duplicate_pr",
|
||||
message="Second issue collision",
|
||||
issue_number=10,
|
||||
pr_numbers=(14, 15),
|
||||
),
|
||||
),
|
||||
duplicate_branches=(),
|
||||
collision_history=(),
|
||||
fetch_error=None,
|
||||
)
|
||||
mock_version = VersionInfo(
|
||||
git_sha="abc1234",
|
||||
git_describe="v1.0.0",
|
||||
control_plane_schema_version=1,
|
||||
python_version="3.11",
|
||||
known=True,
|
||||
)
|
||||
mock_stale = StaleRuntime(
|
||||
daemon_head="abc1234",
|
||||
checkout_head="abc1234",
|
||||
remote_head="abc1234",
|
||||
stale=False,
|
||||
determinable=True,
|
||||
mutation_safe=True,
|
||||
reasons=(),
|
||||
)
|
||||
mock_health = SystemHealthSnapshot(
|
||||
status="degraded",
|
||||
ready=False,
|
||||
readiness_complete=True,
|
||||
readiness_reasons=(),
|
||||
service="webui",
|
||||
mode="test",
|
||||
version=mock_version,
|
||||
started_at="2026-07-25T00:00:00Z",
|
||||
uptime_seconds=100.0,
|
||||
timestamp="2026-07-25T00:00:00Z",
|
||||
deep_probes_requested=True,
|
||||
dependencies=(),
|
||||
mcp_namespaces=(),
|
||||
stale_runtime=mock_stale,
|
||||
probe_errors=("error alpha", "error beta"),
|
||||
)
|
||||
|
||||
snapshot = load_notifications_snapshot(
|
||||
proj_id,
|
||||
load_queue=lambda _id: mock_queue,
|
||||
load_leases=lambda **_kwargs: mock_leases,
|
||||
load_health=lambda **_kwargs: mock_health,
|
||||
)
|
||||
ids = [item.id for item in snapshot.items]
|
||||
assert len(ids) == len(set(ids)), f"duplicate notification ids: {ids}"
|
||||
assert any(i.startswith(f"notif-sys-err-{proj_id}-") for i in ids)
|
||||
assert any(i.startswith("notif-collision-") for i in ids)
|
||||
|
||||
|
||||
def test_probe_errors_do_not_set_fetch_error():
|
||||
"""B3: probe_errors must not be reported as fetch_error (#905)."""
|
||||
reg = load_registry()
|
||||
proj_id = reg.projects[0].id if reg.projects else "gitea-tools"
|
||||
|
||||
mock_queue = QueueSnapshot(
|
||||
project_id=proj_id,
|
||||
repo_label="org/repo",
|
||||
prs=(),
|
||||
issues=(),
|
||||
pr_pagination=None,
|
||||
issue_pagination=None,
|
||||
fetch_error=None,
|
||||
)
|
||||
mock_leases = LeaseSnapshot(
|
||||
project_id=proj_id,
|
||||
repo_label="org/repo",
|
||||
issue_lock=None,
|
||||
claim_inventory={},
|
||||
reviewer_leases=(),
|
||||
duplicate_prs=(),
|
||||
duplicate_branches=(),
|
||||
collision_history=(),
|
||||
fetch_error=None,
|
||||
)
|
||||
mock_version = VersionInfo(
|
||||
git_sha="abc1234",
|
||||
git_describe="v1.0.0",
|
||||
control_plane_schema_version=1,
|
||||
python_version="3.11",
|
||||
known=True,
|
||||
)
|
||||
mock_stale = StaleRuntime(
|
||||
daemon_head="abc1234",
|
||||
checkout_head="abc1234",
|
||||
remote_head="abc1234",
|
||||
stale=False,
|
||||
determinable=True,
|
||||
mutation_safe=True,
|
||||
reasons=(),
|
||||
)
|
||||
mock_health = SystemHealthSnapshot(
|
||||
status="degraded",
|
||||
ready=False,
|
||||
readiness_complete=True,
|
||||
readiness_reasons=(),
|
||||
service="webui",
|
||||
mode="test",
|
||||
version=mock_version,
|
||||
started_at="2026-07-25T00:00:00Z",
|
||||
uptime_seconds=100.0,
|
||||
timestamp="2026-07-25T00:00:00Z",
|
||||
deep_probes_requested=True,
|
||||
dependencies=(),
|
||||
mcp_namespaces=(),
|
||||
stale_runtime=mock_stale,
|
||||
probe_errors=("probe blew up",),
|
||||
)
|
||||
|
||||
snapshot = load_notifications_snapshot(
|
||||
proj_id,
|
||||
load_queue=lambda _id: mock_queue,
|
||||
load_leases=lambda **_kwargs: mock_leases,
|
||||
load_health=lambda **_kwargs: mock_health,
|
||||
)
|
||||
assert snapshot.fetch_error is None
|
||||
# probe errors still appear as items
|
||||
assert any("probe blew up" in item.summary for item in snapshot.items)
|
||||
@@ -0,0 +1,452 @@
|
||||
"""Read-only restart console: views, gates, and honesty rules (#667).
|
||||
|
||||
The console consumes the #655 substrate. These tests hold it to the three
|
||||
properties that make a status surface trustworthy:
|
||||
|
||||
* an unreadable source is reported unavailable, never rendered as green;
|
||||
* authorization is probed the way execution would probe it, so an allow is
|
||||
never shown for something that could not run;
|
||||
* the surface performs no mutation, including no write to the control-plane DB.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sqlite3
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from starlette.testclient import TestClient # noqa: E402
|
||||
|
||||
import restart_coordinator # noqa: E402
|
||||
from webui import console_authz, restart_console, restart_views # noqa: E402
|
||||
from webui.app import create_app # noqa: E402
|
||||
|
||||
NOW = datetime(2026, 7, 25, 21, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
def _principal(role: str) -> console_authz.Principal:
|
||||
return console_authz.Principal(
|
||||
subject="[email protected]",
|
||||
role=role,
|
||||
identity_source=console_authz.IDENTITY_LOCAL_DEV,
|
||||
authenticated=True,
|
||||
)
|
||||
|
||||
|
||||
def _inventory(*, complete: bool = True, sessions=(), leases=()):
|
||||
def _read(**_kwargs):
|
||||
return {
|
||||
"sessions": list(sessions),
|
||||
"leases": list(leases),
|
||||
"terminal_lock": None,
|
||||
"prior_recovery_attempts": [],
|
||||
"inventory_complete": complete,
|
||||
"incomplete_reasons": (
|
||||
[] if complete else ["fixture: inventory withheld"]
|
||||
),
|
||||
}
|
||||
|
||||
return _read
|
||||
|
||||
|
||||
def _live_session(session_id: str = "prgs-author-1234-abcd") -> dict:
|
||||
return {
|
||||
"session_id": session_id,
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": os.getpid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": (NOW - timedelta(seconds=30)).isoformat(),
|
||||
}
|
||||
|
||||
|
||||
def drain_proof_fixture() -> dict:
|
||||
"""A structurally complete but unsigned drain proof."""
|
||||
return {
|
||||
"version": "drain-proof/v1",
|
||||
"proof_id": "deadbeef" * 8,
|
||||
"clean": True,
|
||||
"issued_at": (NOW - timedelta(minutes=1)).isoformat(),
|
||||
"expires_at": (NOW + timedelta(minutes=5)).isoformat(),
|
||||
"requesting_session_id": "s-live",
|
||||
"impact_fingerprint": "f" * 64,
|
||||
"checks": [],
|
||||
"failed_checks": [],
|
||||
}
|
||||
|
||||
|
||||
class RestartClassMatrixTest(unittest.TestCase):
|
||||
def test_every_policy_class_is_rendered(self) -> None:
|
||||
views = restart_console.build_restart_class_views("operator")
|
||||
self.assertEqual(len(views), len(restart_coordinator.RESTART_CLASS_POLICIES))
|
||||
|
||||
def test_viewer_capability_is_role_scoped_not_generic(self) -> None:
|
||||
"""A worker role must not be shown as able to request a full restart."""
|
||||
author = {
|
||||
v.restart_class: v
|
||||
for v in restart_console.build_restart_class_views("author")
|
||||
}
|
||||
operator = {
|
||||
v.restart_class: v
|
||||
for v in restart_console.build_restart_class_views("operator")
|
||||
}
|
||||
full = restart_coordinator.RestartClass.FULL_MCP_RESTART.value
|
||||
|
||||
self.assertFalse(author[full].viewer_may_request)
|
||||
self.assertFalse(author[full].viewer_may_execute)
|
||||
self.assertTrue(operator[full].viewer_may_request)
|
||||
self.assertTrue(operator[full].viewer_may_execute)
|
||||
|
||||
def test_unknown_role_may_do_nothing(self) -> None:
|
||||
views = restart_console.build_restart_class_views("not-a-role")
|
||||
self.assertTrue(all(not v.viewer_may_request for v in views))
|
||||
self.assertTrue(all(not v.viewer_may_execute for v in views))
|
||||
|
||||
|
||||
class AuthorizationProbeTest(unittest.TestCase):
|
||||
def test_probe_asks_for_execution_so_phase_gate_is_reported(self) -> None:
|
||||
"""An admin clears the role bar and still cannot execute in Phase 1.
|
||||
|
||||
This is the case that distinguishes the two probes. Asked without
|
||||
``for_execution`` an admin is *allowed* for ``system.restart_namespace``,
|
||||
which on a control surface reads as a live button. Asked the way
|
||||
execution asks, the same principal is refused ``phase_not_active``. The
|
||||
console must report the second answer.
|
||||
"""
|
||||
by_id = {
|
||||
a.action_id: a
|
||||
for a in restart_console.build_action_authorizations(
|
||||
_principal(console_authz.ADMIN)
|
||||
)
|
||||
}
|
||||
restart = by_id["system.restart_namespace"]
|
||||
|
||||
self.assertFalse(restart.execution_enabled)
|
||||
self.assertEqual(restart.reason_code, console_authz.DENY_PHASE_NOT_ACTIVE)
|
||||
|
||||
permissive = console_authz.authorize(
|
||||
"system.restart_namespace", _principal(console_authz.ADMIN)
|
||||
)
|
||||
self.assertTrue(
|
||||
permissive.allowed,
|
||||
"guard precondition: without for_execution an admin is allowed, "
|
||||
"which is exactly why the console must not probe that way",
|
||||
)
|
||||
|
||||
def test_operator_is_refused_the_admin_only_restart_action(self) -> None:
|
||||
"""Role refusal precedes the phase gate and is reported as such."""
|
||||
by_id = {
|
||||
a.action_id: a
|
||||
for a in restart_console.build_action_authorizations(
|
||||
_principal(console_authz.OPERATOR)
|
||||
)
|
||||
}
|
||||
self.assertEqual(
|
||||
by_id["system.restart_namespace"].reason_code,
|
||||
console_authz.DENY_INSUFFICIENT_ROLE,
|
||||
)
|
||||
|
||||
def test_anonymous_is_denied_unauthenticated(self) -> None:
|
||||
by_id = {
|
||||
a.action_id: a for a in restart_console.build_action_authorizations(None)
|
||||
}
|
||||
self.assertEqual(
|
||||
by_id["system.restart_namespace"].reason_code,
|
||||
console_authz.DENY_UNAUTHENTICATED,
|
||||
)
|
||||
|
||||
def test_no_authorization_ever_reports_execution_enabled(self) -> None:
|
||||
for role in (
|
||||
console_authz.VIEWER,
|
||||
console_authz.OPERATOR,
|
||||
console_authz.CONTROLLER,
|
||||
console_authz.ADMIN,
|
||||
):
|
||||
for auth in restart_console.build_action_authorizations(_principal(role)):
|
||||
self.assertFalse(
|
||||
auth.execution_enabled,
|
||||
f"{role} reported execution_enabled for {auth.action_id}",
|
||||
)
|
||||
|
||||
|
||||
class ImpactPreviewTest(unittest.TestCase):
|
||||
def test_impact_renders_from_coordinator_dto(self) -> None:
|
||||
impact, source = restart_console.load_impact_report(
|
||||
principal=_principal(console_authz.OPERATOR),
|
||||
read_inventory=_inventory(sessions=[_live_session()]),
|
||||
now=NOW,
|
||||
)
|
||||
self.assertTrue(source.available)
|
||||
self.assertIsNotNone(impact)
|
||||
self.assertEqual(
|
||||
impact["restart_class"],
|
||||
restart_coordinator.RestartClass.FULL_MCP_RESTART.value,
|
||||
)
|
||||
self.assertIn("verdict", impact)
|
||||
self.assertFalse(impact["restart_performed"])
|
||||
self.assertTrue(impact["dry_run"])
|
||||
|
||||
def test_incomplete_inventory_is_surfaced_and_denies(self) -> None:
|
||||
impact, source = restart_console.load_impact_report(
|
||||
principal=_principal(console_authz.OPERATOR),
|
||||
read_inventory=_inventory(complete=False),
|
||||
now=NOW,
|
||||
)
|
||||
self.assertFalse(impact["inventory_complete"])
|
||||
self.assertFalse(impact["allow_restart"])
|
||||
self.assertTrue(source.detail, "incomplete inventory must explain itself")
|
||||
|
||||
def test_inventory_reader_failure_is_unavailable_not_empty(self) -> None:
|
||||
"""A reader that raises must not be rendered as 'no sessions affected'."""
|
||||
|
||||
def _boom(**_kwargs):
|
||||
raise RuntimeError("control-plane unreachable")
|
||||
|
||||
impact, source = restart_console.load_impact_report(
|
||||
principal=_principal(console_authz.OPERATOR),
|
||||
read_inventory=_boom,
|
||||
now=NOW,
|
||||
)
|
||||
self.assertIsNone(impact)
|
||||
self.assertFalse(source.available)
|
||||
self.assertIn("control-plane unreachable", source.detail)
|
||||
|
||||
|
||||
class ControlPlaneReadTest(unittest.TestCase):
|
||||
def test_missing_database_is_incomplete_not_empty(self) -> None:
|
||||
inventory = restart_console.read_control_plane_inventory(
|
||||
db_path="/nonexistent/control-plane.sqlite3"
|
||||
)
|
||||
self.assertFalse(inventory["inventory_complete"])
|
||||
self.assertEqual(inventory["sessions"], [])
|
||||
self.assertTrue(inventory["incomplete_reasons"])
|
||||
|
||||
def test_reader_never_creates_the_database(self) -> None:
|
||||
"""Reading status must not bring a control-plane DB into existence.
|
||||
|
||||
The path deliberately sits in a directory that already exists: a
|
||||
read-write ``sqlite3.connect`` would happily create the file there, so
|
||||
this fails if the reader ever stops opening the database ``mode=ro``.
|
||||
A nested-missing-directory path would pass for the wrong reason,
|
||||
because sqlite cannot create the parent directory either way.
|
||||
"""
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
path = os.path.join(tmp, "control_plane.sqlite3")
|
||||
self.assertTrue(os.path.isdir(os.path.dirname(path)))
|
||||
|
||||
inventory = restart_console.read_control_plane_inventory(db_path=path)
|
||||
|
||||
self.assertFalse(
|
||||
os.path.exists(path),
|
||||
"reading restart status created a control-plane database",
|
||||
)
|
||||
self.assertFalse(inventory["inventory_complete"])
|
||||
|
||||
def test_reads_active_sessions_from_a_real_database(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
path = os.path.join(tmp, "cp.sqlite3")
|
||||
conn = sqlite3.connect(path)
|
||||
conn.execute(
|
||||
"CREATE TABLE sessions (session_id TEXT, role TEXT, profile TEXT,"
|
||||
" pid INTEGER, status TEXT, last_heartbeat_at TEXT)"
|
||||
)
|
||||
conn.execute(
|
||||
"CREATE TABLE work_items (work_item_id INTEGER, kind TEXT,"
|
||||
" number INTEGER)"
|
||||
)
|
||||
conn.execute(
|
||||
"CREATE TABLE leases (lease_id TEXT, session_id TEXT, role TEXT,"
|
||||
" phase TEXT, status TEXT, worktree_path TEXT,"
|
||||
" work_item_id INTEGER, expires_at TEXT)"
|
||||
)
|
||||
conn.execute(
|
||||
"INSERT INTO sessions VALUES (?,?,?,?,?,?)",
|
||||
("s-live", "author", "prgs-author", 4242, "active", NOW.isoformat()),
|
||||
)
|
||||
conn.execute(
|
||||
"INSERT INTO sessions VALUES (?,?,?,?,?,?)",
|
||||
("s-done", "author", "prgs-author", 11, "closed", NOW.isoformat()),
|
||||
)
|
||||
conn.execute("INSERT INTO work_items VALUES (1, 'issue', 667)")
|
||||
conn.execute(
|
||||
"INSERT INTO leases VALUES (?,?,?,?,?,?,?,?)",
|
||||
(
|
||||
"l-1",
|
||||
"s-live",
|
||||
"author",
|
||||
"allocated",
|
||||
"active",
|
||||
None,
|
||||
1,
|
||||
NOW.isoformat(),
|
||||
),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
inventory = restart_console.read_control_plane_inventory(db_path=path)
|
||||
|
||||
self.assertTrue(inventory["inventory_complete"])
|
||||
self.assertEqual([s["session_id"] for s in inventory["sessions"]], ["s-live"])
|
||||
self.assertEqual(inventory["leases"][0]["work_number"], 667)
|
||||
|
||||
|
||||
class DrainAndReconcileTest(unittest.TestCase):
|
||||
def test_absent_drain_proof_is_not_a_pass(self) -> None:
|
||||
drain, source = restart_console.load_drain_status(proof=None, now=NOW)
|
||||
self.assertIsNone(drain)
|
||||
self.assertFalse(source.available)
|
||||
self.assertIn("denies", source.detail)
|
||||
|
||||
def test_tampered_drain_proof_is_reported_invalid(self) -> None:
|
||||
proof = drain_proof_fixture()
|
||||
proof["clean"] = True
|
||||
proof["proof_id"] = "0" * 64
|
||||
drain, source = restart_console.load_drain_status(proof=proof, now=NOW)
|
||||
self.assertTrue(source.available)
|
||||
self.assertFalse(drain["valid"])
|
||||
|
||||
def test_absent_reconcile_proof_is_unavailable(self) -> None:
|
||||
reconcile, source = restart_console.load_reconcile_status(load_proof=None)
|
||||
self.assertIsNone(reconcile)
|
||||
self.assertFalse(source.available)
|
||||
|
||||
def test_reconcile_proof_is_rendered_when_supplied(self) -> None:
|
||||
payload = {
|
||||
"overall_status": "degraded",
|
||||
"mode": "log_only",
|
||||
"resolved_count": 3,
|
||||
"unresolved_count": 2,
|
||||
"items": [
|
||||
{
|
||||
"dimension": "leases",
|
||||
"status": "unresolved",
|
||||
"summary": "2 orphaned leases",
|
||||
"follow_up_required": True,
|
||||
}
|
||||
],
|
||||
}
|
||||
reconcile, source = restart_console.load_reconcile_status(
|
||||
load_proof=lambda: payload
|
||||
)
|
||||
self.assertTrue(source.available)
|
||||
self.assertEqual(reconcile["unresolved_count"], 2)
|
||||
|
||||
|
||||
class RenderingTest(unittest.TestCase):
|
||||
def _snapshot(self, **kwargs):
|
||||
params = {
|
||||
"principal": _principal(console_authz.OPERATOR),
|
||||
"read_inventory": _inventory(sessions=[_live_session()]),
|
||||
"now": NOW,
|
||||
}
|
||||
params.update(kwargs)
|
||||
return restart_console.load_restart_console_snapshot(**params)
|
||||
|
||||
def test_page_renders_every_section(self) -> None:
|
||||
html = restart_views.render_restart_console_page(self._snapshot())
|
||||
for heading in (
|
||||
"Impact preview",
|
||||
"Drain proof",
|
||||
"Post-restart reconcile",
|
||||
"Restart classes",
|
||||
"Approval controls",
|
||||
"Break-glass",
|
||||
):
|
||||
self.assertIn(heading, html)
|
||||
|
||||
def test_hostile_session_id_is_escaped(self) -> None:
|
||||
hostile = "<script>alert('x')</script>"
|
||||
html = restart_views.render_restart_console_page(
|
||||
self._snapshot(read_inventory=_inventory(sessions=[_live_session(hostile)]))
|
||||
)
|
||||
self.assertNotIn("<script>alert", html)
|
||||
self.assertIn("<script>", html)
|
||||
|
||||
def test_unavailable_impact_says_unsafe_rather_than_clean(self) -> None:
|
||||
def _boom(**_kwargs):
|
||||
raise RuntimeError("nope")
|
||||
|
||||
snapshot = self._snapshot(read_inventory=_boom)
|
||||
html = restart_views.render_restart_console_page(snapshot)
|
||||
self.assertIn("blast radius of a restart is unknown", html)
|
||||
self.assertIn("unavailable", html)
|
||||
|
||||
def test_break_glass_is_hidden_from_unprivileged_viewers(self) -> None:
|
||||
viewer_html = restart_views.render_restart_console_page(
|
||||
self._snapshot(principal=_principal(console_authz.VIEWER))
|
||||
)
|
||||
self.assertIn("visible to operator-class", viewer_html)
|
||||
self.assertNotIn(
|
||||
f"#{restart_console.BREAK_GLASS_ISSUE}", viewer_html
|
||||
)
|
||||
|
||||
def test_break_glass_shown_to_operator_is_marked_unavailable(self) -> None:
|
||||
html = restart_views.render_restart_console_page(self._snapshot())
|
||||
self.assertIn("unavailable", html)
|
||||
self.assertIn(f"#{restart_console.BREAK_GLASS_ISSUE}", html)
|
||||
|
||||
def test_snapshot_always_declares_itself_read_only(self) -> None:
|
||||
self.assertTrue(self._snapshot().read_only)
|
||||
|
||||
|
||||
class RestartConsoleRouteTest(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self.client = TestClient(create_app())
|
||||
|
||||
def test_page_route_renders(self) -> None:
|
||||
res = self.client.get("/runtime/restart")
|
||||
self.assertEqual(res.status_code, 200)
|
||||
self.assertIn("Restart status and impact", res.text)
|
||||
|
||||
def test_api_route_exports_snapshot(self) -> None:
|
||||
res = self.client.get("/api/v1/system/restart/status")
|
||||
self.assertEqual(res.status_code, 200)
|
||||
payload = res.json()
|
||||
self.assertTrue(payload["read_only"])
|
||||
self.assertEqual(payload["links"]["issue"], 667)
|
||||
self.assertEqual(
|
||||
len(payload["restart_classes"]),
|
||||
len(restart_coordinator.RESTART_CLASS_POLICIES),
|
||||
)
|
||||
|
||||
def test_restart_class_is_selectable(self) -> None:
|
||||
res = self.client.get(
|
||||
"/api/v1/system/restart/status?restart_class=client_reconnect"
|
||||
)
|
||||
self.assertEqual(res.status_code, 200)
|
||||
self.assertEqual(res.json()["impact"]["restart_class"], "client_reconnect")
|
||||
|
||||
def test_unknown_restart_class_fails_closed(self) -> None:
|
||||
res = self.client.get(
|
||||
"/api/v1/system/restart/status?restart_class=obliterate-everything"
|
||||
)
|
||||
self.assertEqual(res.status_code, 200)
|
||||
impact = res.json()["impact"]
|
||||
self.assertFalse(impact["allow_restart"])
|
||||
|
||||
def test_anonymous_api_reader_gets_no_execution_grant(self) -> None:
|
||||
payload = self.client.get("/api/v1/system/restart/status").json()
|
||||
self.assertFalse(payload["break_glass"]["available"])
|
||||
for auth in payload["authorizations"]:
|
||||
self.assertFalse(auth["execution_enabled"])
|
||||
|
||||
def test_route_is_registered_in_nav(self) -> None:
|
||||
from webui.nav import nav_hrefs
|
||||
|
||||
self.assertIn("/runtime/restart", nav_hrefs())
|
||||
|
||||
def test_no_write_method_is_exposed(self) -> None:
|
||||
"""The surface is read-only: nothing accepts a POST."""
|
||||
for path in ("/runtime/restart", "/api/v1/system/restart/status"):
|
||||
self.assertEqual(self.client.post(path).status_code, 405, path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -47,6 +47,9 @@ from webui.traffic_views import render_traffic_page
|
||||
from webui.worktree_scanner import load_hygiene_snapshot, snapshot_to_dict as worktree_snapshot_to_dict
|
||||
from webui.worktree_views import render_worktrees_page
|
||||
from webui.runtime_health import load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict
|
||||
import restart_coordinator
|
||||
from webui.restart_console import load_restart_console_snapshot
|
||||
from webui.restart_views import render_restart_console_page
|
||||
from webui.runtime_views import render_runtime_page
|
||||
from webui.session_loader import (
|
||||
load_session_view_snapshot,
|
||||
@@ -77,6 +80,11 @@ from webui.system_health import (
|
||||
snapshot_to_dict as system_health_to_dict,
|
||||
)
|
||||
from webui.system_health_views import render_system_health_page
|
||||
from webui.notifications import (
|
||||
load_notifications_snapshot,
|
||||
snapshot_to_dict as notifications_snapshot_to_dict,
|
||||
)
|
||||
from webui.notification_views import render_notifications_page
|
||||
from webui import request_service
|
||||
from webui.request_views import render_requests_page
|
||||
|
||||
@@ -416,6 +424,33 @@ async def api_runtime(_request: Request) -> JSONResponse:
|
||||
return JSONResponse(runtime_snapshot_to_dict(load_runtime_snapshot()))
|
||||
|
||||
|
||||
def _restart_console_snapshot(request: Request):
|
||||
"""Build the read-only restart snapshot for the requesting principal (#667)."""
|
||||
principal = resolve_principal(request.headers)
|
||||
restart_class = (
|
||||
request.query_params.get("restart_class")
|
||||
or restart_coordinator.RestartClass.FULL_MCP_RESTART.value
|
||||
)
|
||||
return load_restart_console_snapshot(
|
||||
principal=principal, restart_class=restart_class
|
||||
)
|
||||
|
||||
|
||||
async def restart_console_page(request: Request) -> HTMLResponse:
|
||||
"""Restart status, impact preview, and approval state (#667). Read-only."""
|
||||
snapshot = _restart_console_snapshot(request)
|
||||
return HTMLResponse(
|
||||
render_page(
|
||||
title="Restart", body_html=render_restart_console_page(snapshot)
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
async def api_restart_status(request: Request) -> JSONResponse:
|
||||
"""JSON export of the read-only restart console snapshot (#667)."""
|
||||
return JSONResponse(_restart_console_snapshot(request).as_dict())
|
||||
|
||||
|
||||
async def sessions(_request: Request) -> HTMLResponse:
|
||||
"""Runtime and session view (#641) — read-only composition of health + inventory."""
|
||||
snapshot = load_session_view_snapshot()
|
||||
@@ -859,6 +894,22 @@ async def api_v1_analytics_ingest(request: Request) -> JSONResponse:
|
||||
)
|
||||
|
||||
|
||||
async def notifications_route(request: Request) -> HTMLResponse:
|
||||
project_id = request.query_params.get("project_id")
|
||||
attention_class = request.query_params.get("attention_class") or "inbox"
|
||||
snap = load_notifications_snapshot(project_id)
|
||||
html = render_notifications_page(
|
||||
snap, filter_class=attention_class, filter_project=project_id
|
||||
)
|
||||
return HTMLResponse(html)
|
||||
|
||||
|
||||
async def api_notifications(request: Request) -> JSONResponse:
|
||||
project_id = request.query_params.get("project_id")
|
||||
snap = load_notifications_snapshot(project_id)
|
||||
data = notifications_snapshot_to_dict(snap)
|
||||
return JSONResponse(data)
|
||||
|
||||
def _default_request_scope() -> dict[str, str]:
|
||||
"""Resolve remote/org/repo from the project registry for request forms.
|
||||
|
||||
@@ -990,6 +1041,9 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
Route("/api/queue", api_queue, methods=["GET"]),
|
||||
Route("/traffic", traffic, methods=["GET"]),
|
||||
Route("/api/traffic", api_traffic, methods=["GET"]),
|
||||
Route("/notifications", notifications_route, methods=["GET"]),
|
||||
Route("/api/notifications", api_notifications, methods=["GET"]),
|
||||
Route("/api/v1/notifications", api_notifications, methods=["GET"]),
|
||||
Route("/projects", projects, methods=["GET"]),
|
||||
Route("/projects/{project_id}", project_detail, methods=["GET"]),
|
||||
Route("/api/projects", api_projects, methods=["GET"]),
|
||||
@@ -1004,6 +1058,13 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
Route("/api/prompts", api_prompts, methods=["GET"]),
|
||||
Route("/runtime", runtime, methods=["GET"]),
|
||||
Route("/api/runtime", api_runtime, methods=["GET"]),
|
||||
# #667 read-only restart status / impact preview / approval state.
|
||||
Route("/runtime/restart", restart_console_page, methods=["GET"]),
|
||||
Route(
|
||||
"/api/v1/system/restart/status",
|
||||
api_restart_status,
|
||||
methods=["GET"],
|
||||
),
|
||||
Route("/sessions", sessions, methods=["GET"]),
|
||||
Route("/api/sessions", api_sessions, methods=["GET"]),
|
||||
Route("/api/v1/sessions", api_sessions, methods=["GET"]),
|
||||
|
||||
@@ -46,10 +46,12 @@ NAV_GROUPS: tuple[NavGroup, ...] = (
|
||||
NavItem("/queue", "Queue"),
|
||||
NavItem("/leases", "Leases"),
|
||||
NavItem("/actions", "Actions"),
|
||||
NavItem("/notifications", "Notifications"),
|
||||
NavItem("/requests", "Requests"),
|
||||
)),
|
||||
NavGroup("Runtime/Sessions", (
|
||||
NavItem("/runtime", "Runtime health"),
|
||||
NavItem("/runtime/restart", "Restart status"),
|
||||
NavItem("/sessions", "Sessions"),
|
||||
)),
|
||||
NavGroup("Projects", (
|
||||
|
||||
@@ -0,0 +1,158 @@
|
||||
"""HTML rendering for Phase 3 Notifications and Human-Attention Console (#648)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from html import escape
|
||||
from typing import Sequence
|
||||
|
||||
from webui.layout import render_page
|
||||
from webui.notifications import (
|
||||
ATTENTION_HUMAN_REQUIRED,
|
||||
ATTENTION_OPERATOR,
|
||||
ATTENTION_ROUTINE,
|
||||
NotificationItem,
|
||||
NotificationSnapshot,
|
||||
)
|
||||
|
||||
|
||||
def _render_attention_badge(attention_class: str) -> str:
|
||||
cls = "badge"
|
||||
if attention_class == ATTENTION_HUMAN_REQUIRED:
|
||||
cls += " badge-blocked"
|
||||
elif attention_class == ATTENTION_OPERATOR:
|
||||
cls += " badge-claimed"
|
||||
else:
|
||||
cls += " muted"
|
||||
return f'<span class="{cls}">{escape(attention_class)}</span>'
|
||||
|
||||
|
||||
def _render_notification_row(item: NotificationItem) -> str:
|
||||
category_label = escape(item.category.upper())
|
||||
id_str = escape(item.id)
|
||||
title_str = escape(item.title)
|
||||
summary_str = escape(item.summary)
|
||||
att_badge = _render_attention_badge(item.attention_class)
|
||||
|
||||
work_item_html = "—"
|
||||
if item.work_number and item.work_kind:
|
||||
kind_label = escape(item.work_kind.upper())
|
||||
num_str = f"#{item.work_number}"
|
||||
link = item.deep_link or "#"
|
||||
work_item_html = f'<a href="{escape(link)}"><code>{kind_label} {num_str}</code></a>'
|
||||
|
||||
requires_human_label = (
|
||||
'<span class="badge badge-blocked" style="font-size:0.75rem;">HUMAN REQUIRED</span>'
|
||||
if item.requires_human
|
||||
else ""
|
||||
)
|
||||
|
||||
return f"""<tr>
|
||||
<td><code>{category_label}</code><br><span class="muted" style="font-size:0.75rem;">{id_str}</span></td>
|
||||
<td>
|
||||
<div><strong>{title_str}</strong> {att_badge} {requires_human_label}</div>
|
||||
<div class="muted" style="font-size:0.85rem; margin-top:0.25rem;">{summary_str}</div>
|
||||
</td>
|
||||
<td>{work_item_html}</td>
|
||||
<td><span class="muted" style="font-size:0.8rem;">{escape(item.created_at[:19])}</span></td>
|
||||
</tr>"""
|
||||
|
||||
|
||||
def _render_notifications_table(items: Sequence[NotificationItem], empty_message: str) -> str:
|
||||
if not items:
|
||||
return f'<p class="muted" style="padding:1rem 0;">{escape(empty_message)}</p>'
|
||||
|
||||
rows = "".join(_render_notification_row(item) for item in items)
|
||||
return f"""<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th style="width: 18%;">Category & ID</th>
|
||||
<th style="width: 52%;">Title & Attention Summary</th>
|
||||
<th style="width: 15%;">Work Item</th>
|
||||
<th style="width: 15%;">Time</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{rows}
|
||||
</tbody>
|
||||
</table>"""
|
||||
|
||||
|
||||
def render_notifications_page(
|
||||
snapshot: NotificationSnapshot,
|
||||
*,
|
||||
filter_class: str = "inbox",
|
||||
filter_project: str | None = None,
|
||||
) -> str:
|
||||
"""Render the notifications and attention inbox page."""
|
||||
title = "Notifications & Attention Inbox"
|
||||
|
||||
err_html = ""
|
||||
if snapshot.fetch_error:
|
||||
err_html = f'<div class="stub" style="border-color:#e53e3e; background:#fff5f5; color:#c53030; margin-bottom:1rem;"><p><strong>Fetch Warning:</strong> {escape(snapshot.fetch_error)}</p></div>'
|
||||
|
||||
# Determine items to render based on filter_class
|
||||
if filter_class == ATTENTION_HUMAN_REQUIRED:
|
||||
display_items = snapshot.human_required_items
|
||||
active_tab_title = "Human-Required Escalations"
|
||||
elif filter_class == ATTENTION_OPERATOR:
|
||||
display_items = snapshot.operator_items
|
||||
active_tab_title = "Operator Inbox Items"
|
||||
elif filter_class == ATTENTION_ROUTINE:
|
||||
display_items = snapshot.routine_items
|
||||
active_tab_title = "Routine Workflow Transitions"
|
||||
elif filter_class == "all":
|
||||
display_items = snapshot.items
|
||||
active_tab_title = "All Events (including Routine)"
|
||||
else: # "inbox" default
|
||||
display_items = snapshot.inbox_items
|
||||
active_tab_title = "Attention Inbox (Human + Operator)"
|
||||
|
||||
hr_cls = "badge-blocked" if snapshot.human_required_count > 0 else "muted"
|
||||
op_cls = "badge-claimed" if snapshot.operator_count > 0 else "muted"
|
||||
|
||||
metrics_html = f"""<div style="display:flex; gap:1rem; margin-bottom:1.5rem;">
|
||||
<div class="health-card" style="flex:1;">
|
||||
<span class="muted" style="font-size:0.85rem;">Human Required</span>
|
||||
<h2 style="margin:0.2rem 0;"><span class="badge {hr_cls}" style="font-size:1.4rem;">{snapshot.human_required_count}</span></h2>
|
||||
<p class="muted" style="font-size:0.8rem; margin:0;">Critical escalation boundary</p>
|
||||
</div>
|
||||
<div class="health-card" style="flex:1;">
|
||||
<span class="muted" style="font-size:0.85rem;">Operator Inbox</span>
|
||||
<h2 style="margin:0.2rem 0;"><span class="badge {op_cls}" style="font-size:1.4rem;">{snapshot.operator_count}</span></h2>
|
||||
<p class="muted" style="font-size:0.8rem; margin:0;">Operational items needing review</p>
|
||||
</div>
|
||||
<div class="health-card" style="flex:1;">
|
||||
<span class="muted" style="font-size:0.85rem;">Routine Transitions</span>
|
||||
<h2 style="margin:0.2rem 0;"><span class="badge muted" style="font-size:1.4rem;">{snapshot.routine_count}</span></h2>
|
||||
<p class="muted" style="font-size:0.8rem; margin:0;">Background transitions (filtered)</p>
|
||||
</div>
|
||||
</div>"""
|
||||
|
||||
# Filter navigation links
|
||||
def _tab_link(target_class: str, label: str) -> str:
|
||||
is_active = (filter_class == target_class)
|
||||
style = "font-weight:bold; border-bottom:2px solid currentColor;" if is_active else "color:#4a5568;"
|
||||
return f'<a href="/notifications?attention_class={target_class}" style="margin-right:1.25rem; text-decoration:none; padding-bottom:0.25rem; {style}">{label}</a>'
|
||||
|
||||
tabs_html = f"""<div style="margin-bottom:1.25rem; border-bottom:1px solid #e2e8f0; padding-bottom:0.5rem;">
|
||||
{_tab_link("inbox", f"Attention Inbox ({snapshot.human_required_count + snapshot.operator_count})")}
|
||||
{_tab_link("human-required", f"Human Required ({snapshot.human_required_count})")}
|
||||
{_tab_link("operator", f"Operator ({snapshot.operator_count})")}
|
||||
{_tab_link("routine", f"Routine ({snapshot.routine_count})")}
|
||||
{_tab_link("all", f"All Events ({snapshot.total_count})")}
|
||||
</div>"""
|
||||
|
||||
table_html = _render_notifications_table(
|
||||
display_items,
|
||||
f"No items match attention filter '{filter_class}'.",
|
||||
)
|
||||
|
||||
body = f"""<h2>{escape(title)}</h2>
|
||||
<p class="muted">Phase 3 console surface for human-attention routing (#648). Routine workflow transitions are filtered by default to eliminate notification fatigue.</p>
|
||||
{err_html}
|
||||
{metrics_html}
|
||||
{tabs_html}
|
||||
<h3>{escape(active_tab_title)}</h3>
|
||||
{table_html}"""
|
||||
|
||||
return render_page(title=title, body_html=body)
|
||||
@@ -0,0 +1,486 @@
|
||||
"""Notifications and human-attention routing module for Phase 3 web console (#648).
|
||||
|
||||
Defines attention classes, event classification rules, and inbox aggregation so
|
||||
operators receive direct alerts only for human-required escalation boundaries
|
||||
(#628) while routine workflow transitions remain available for pull-based review.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Callable
|
||||
|
||||
from webui import console_redaction
|
||||
from webui.project_registry import load_registry
|
||||
from webui.queue_loader import QueueSnapshot, load_queue_snapshot
|
||||
from webui.lease_loader import LeaseSnapshot, load_lease_snapshot
|
||||
from webui.system_health import SystemHealthSnapshot, load_system_health
|
||||
|
||||
# Attention class definitions (#628, #648)
|
||||
ATTENTION_ROUTINE = "routine"
|
||||
ATTENTION_OPERATOR = "operator"
|
||||
ATTENTION_HUMAN_REQUIRED = "human-required"
|
||||
|
||||
ATTENTION_CLASSES = (
|
||||
ATTENTION_ROUTINE,
|
||||
ATTENTION_OPERATOR,
|
||||
ATTENTION_HUMAN_REQUIRED,
|
||||
)
|
||||
|
||||
# Notification categories
|
||||
CATEGORY_AUTH = "auth"
|
||||
CATEGORY_BLOCKER = "blocker"
|
||||
CATEGORY_LEASE = "lease"
|
||||
CATEGORY_VALIDATION = "validation"
|
||||
CATEGORY_WORKFLOW = "workflow"
|
||||
CATEGORY_SYSTEM = "system"
|
||||
|
||||
CATEGORIES = (
|
||||
CATEGORY_AUTH,
|
||||
CATEGORY_BLOCKER,
|
||||
CATEGORY_LEASE,
|
||||
CATEGORY_VALIDATION,
|
||||
CATEGORY_WORKFLOW,
|
||||
CATEGORY_SYSTEM,
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NotificationItem:
|
||||
"""A single notification or inbox event."""
|
||||
|
||||
id: str
|
||||
attention_class: str # "routine", "operator", "human-required"
|
||||
category: str # "auth", "blocker", "lease", "validation", etc.
|
||||
title: str
|
||||
summary: str
|
||||
work_kind: str | None # "issue", "pr", "session", "system"
|
||||
work_number: int | None
|
||||
project_id: str
|
||||
repo_label: str
|
||||
created_at: str
|
||||
deep_link: str | None = None
|
||||
requires_human: bool = False
|
||||
extra: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"id": self.id,
|
||||
"attention_class": self.attention_class,
|
||||
"category": self.category,
|
||||
"title": self.title,
|
||||
"summary": console_redaction.redact_text(self.summary),
|
||||
"work_kind": self.work_kind,
|
||||
"work_number": self.work_number,
|
||||
"project_id": self.project_id,
|
||||
"repo_label": self.repo_label,
|
||||
"created_at": self.created_at,
|
||||
"deep_link": self.deep_link,
|
||||
"requires_human": self.requires_human,
|
||||
"extra": self.extra,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NotificationSnapshot:
|
||||
"""Snapshot of notifications and attention inbox state."""
|
||||
|
||||
project_id: str
|
||||
repo_label: str
|
||||
items: tuple[NotificationItem, ...]
|
||||
human_required_count: int
|
||||
operator_count: int
|
||||
routine_count: int
|
||||
total_count: int
|
||||
fetch_error: str | None = None
|
||||
|
||||
@property
|
||||
def inbox_items(self) -> tuple[NotificationItem, ...]:
|
||||
"""Items requiring operator or human attention (excluding routine)."""
|
||||
return tuple(
|
||||
item
|
||||
for item in self.items
|
||||
if item.attention_class in {ATTENTION_OPERATOR, ATTENTION_HUMAN_REQUIRED}
|
||||
)
|
||||
|
||||
@property
|
||||
def human_required_items(self) -> tuple[NotificationItem, ...]:
|
||||
return tuple(
|
||||
item for item in self.items if item.attention_class == ATTENTION_HUMAN_REQUIRED
|
||||
)
|
||||
|
||||
@property
|
||||
def operator_items(self) -> tuple[NotificationItem, ...]:
|
||||
return tuple(
|
||||
item for item in self.items if item.attention_class == ATTENTION_OPERATOR
|
||||
)
|
||||
|
||||
@property
|
||||
def routine_items(self) -> tuple[NotificationItem, ...]:
|
||||
return tuple(
|
||||
item for item in self.items if item.attention_class == ATTENTION_ROUTINE
|
||||
)
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"project_id": self.project_id,
|
||||
"repo_label": self.repo_label,
|
||||
"human_required_count": self.human_required_count,
|
||||
"operator_count": self.operator_count,
|
||||
"routine_count": self.routine_count,
|
||||
"total_count": self.total_count,
|
||||
"fetch_error": self.fetch_error,
|
||||
"inbox_items": [item.as_dict() for item in self.inbox_items],
|
||||
"all_items": [item.as_dict() for item in self.items],
|
||||
}
|
||||
|
||||
|
||||
def classify_attention_event(
|
||||
category: str,
|
||||
title: str,
|
||||
summary: str,
|
||||
*,
|
||||
is_hard_stop: bool = False,
|
||||
is_auth_failure: bool = False,
|
||||
is_irrecoverable: bool = False,
|
||||
is_decision_lock: bool = False,
|
||||
is_validation_failure: bool = False,
|
||||
is_stale: bool = False,
|
||||
is_blocker: bool = False,
|
||||
) -> tuple[str, bool]:
|
||||
"""Classify an event into an attention class and human requirement flag.
|
||||
|
||||
Rules (#628, #648):
|
||||
1. Critical boundaries (hard stop, auth failure, irrecoverable state,
|
||||
decision lock, validation failure) -> ATTENTION_HUMAN_REQUIRED (requires_human=True).
|
||||
2. Operational queues (blocker, stale lease, unassigned ready work, queue collision)
|
||||
-> ATTENTION_OPERATOR (requires_human=False).
|
||||
3. Routine state transitions (clean progression, healthy heartbeats) -> ATTENTION_ROUTINE (requires_human=False).
|
||||
|
||||
Classification uses structured flags and category only. Human-authored
|
||||
``title`` / ``summary`` text is never substring-matched for escalation
|
||||
(PR #905 review B1) — callers that need text signals must set flags from
|
||||
machine-generated status/detail fields before calling this function.
|
||||
"""
|
||||
del title, summary # kept for API stability; never used for classification
|
||||
if (
|
||||
is_hard_stop
|
||||
or is_auth_failure
|
||||
or is_irrecoverable
|
||||
or is_decision_lock
|
||||
or is_validation_failure
|
||||
or category in {CATEGORY_AUTH, CATEGORY_VALIDATION}
|
||||
):
|
||||
return ATTENTION_HUMAN_REQUIRED, True
|
||||
|
||||
if is_stale or is_blocker or category in {CATEGORY_BLOCKER, CATEGORY_LEASE}:
|
||||
return ATTENTION_OPERATOR, False
|
||||
|
||||
return ATTENTION_ROUTINE, False
|
||||
|
||||
|
||||
def load_notifications_snapshot(
|
||||
project_id: str | None = None,
|
||||
*,
|
||||
load_queue: Callable[..., QueueSnapshot] | None = None,
|
||||
load_leases: Callable[..., LeaseSnapshot] | None = None,
|
||||
load_health: Callable[..., SystemHealthSnapshot] | None = None,
|
||||
) -> NotificationSnapshot:
|
||||
"""Load and classify attention notifications across queue, leases, and system health."""
|
||||
registry = load_registry()
|
||||
project = None
|
||||
if project_id:
|
||||
for entry in registry.projects:
|
||||
if entry.id == project_id:
|
||||
project = entry
|
||||
break
|
||||
else:
|
||||
project = registry.projects[0] if registry.projects else None
|
||||
|
||||
if project is None:
|
||||
return NotificationSnapshot(
|
||||
project_id=project_id or "",
|
||||
repo_label="",
|
||||
items=(),
|
||||
human_required_count=0,
|
||||
operator_count=0,
|
||||
routine_count=0,
|
||||
total_count=0,
|
||||
fetch_error="project not found in registry",
|
||||
)
|
||||
|
||||
queue_loader_fn = load_queue or load_queue_snapshot
|
||||
lease_loader_fn = load_leases or load_lease_snapshot
|
||||
health_loader_fn = load_health or load_system_health
|
||||
|
||||
try:
|
||||
queue_snap = queue_loader_fn(project.id)
|
||||
except TypeError:
|
||||
queue_snap = queue_loader_fn(project_id=project.id)
|
||||
|
||||
try:
|
||||
lease_snap = lease_loader_fn(project_id=project.id)
|
||||
except TypeError:
|
||||
lease_snap = lease_loader_fn(project.id)
|
||||
|
||||
try:
|
||||
health_snap = health_loader_fn(project_id=project.id)
|
||||
except TypeError:
|
||||
try:
|
||||
health_snap = health_loader_fn(project.id)
|
||||
except TypeError:
|
||||
health_snap = health_loader_fn()
|
||||
|
||||
items: list[NotificationItem] = []
|
||||
now_iso = datetime.now(timezone.utc).isoformat()
|
||||
|
||||
# 1. System health alerts (highest priority)
|
||||
for err_idx, probe_err in enumerate(getattr(health_snap, "probe_errors", ())):
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_SYSTEM,
|
||||
"System Health Probe Error",
|
||||
probe_err,
|
||||
is_blocker=True,
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-sys-err-{project.id}-{err_idx}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_SYSTEM,
|
||||
title="System Health Error",
|
||||
summary=f"System health error: {probe_err}",
|
||||
work_kind="system",
|
||||
work_number=None,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link="/system",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
|
||||
for probe in getattr(health_snap, "dependencies", ()):
|
||||
if probe.status not in ("ok", "healthy"):
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_SYSTEM,
|
||||
f"Probe Failure: {probe.name}",
|
||||
probe.detail or probe.status,
|
||||
is_hard_stop=("stop" in probe.status or "fatal" in probe.status),
|
||||
is_auth_failure=("auth" in probe.name.lower() or "unauthorized" in probe.status.lower()),
|
||||
is_blocker=True,
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-probe-{probe.name}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_AUTH if "auth" in probe.name.lower() else CATEGORY_SYSTEM,
|
||||
title=f"Health Probe Alert: {probe.name}",
|
||||
summary=f"Probe '{probe.name}' reported status '{probe.status}': {probe.detail}",
|
||||
work_kind="system",
|
||||
work_number=None,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link="/system",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
|
||||
# 2. Queue items (PRs and Issues)
|
||||
for pr in queue_snap.prs:
|
||||
if "blocked" in pr.badges:
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_BLOCKER,
|
||||
f"PR #{pr.number} Blocked",
|
||||
f"PR #{pr.number} '{pr.title}' is blocked or has merge conflicts.",
|
||||
is_blocker=True,
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-pr-block-{pr.number}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_BLOCKER,
|
||||
title=f"Blocked PR #{pr.number}",
|
||||
summary=f"PR #{pr.number} ({pr.title}) requires merge conflict resolution.",
|
||||
work_kind="pr",
|
||||
work_number=pr.number,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link=f"/traffic",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
elif "stale" in pr.badges:
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_WORKFLOW,
|
||||
f"PR #{pr.number} Stale",
|
||||
f"PR #{pr.number} '{pr.title}' has had no activity for over 14 days.",
|
||||
is_stale=True,
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-pr-stale-{pr.number}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_WORKFLOW,
|
||||
title=f"Stale PR #{pr.number}",
|
||||
summary=f"PR #{pr.number} ({pr.title}) is stale.",
|
||||
work_kind="pr",
|
||||
work_number=pr.number,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link=f"/queue",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
else:
|
||||
# Routine PR transition
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_WORKFLOW,
|
||||
f"PR #{pr.number} Active",
|
||||
f"PR #{pr.number} '{pr.title}' is in routine state {', '.join(pr.badges)}.",
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-pr-routine-{pr.number}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_WORKFLOW,
|
||||
title=f"Routine PR #{pr.number}",
|
||||
summary=f"PR #{pr.number} ({pr.title}) state: {', '.join(pr.badges)}.",
|
||||
work_kind="pr",
|
||||
work_number=pr.number,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link=f"/queue",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
|
||||
for issue in queue_snap.issues:
|
||||
if "duplicate" in issue.badges:
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_BLOCKER,
|
||||
f"Issue #{issue.number} Duplicate PRs",
|
||||
f"Issue #{issue.number} has multiple linked PRs.",
|
||||
is_blocker=True,
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-issue-dup-{issue.number}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_BLOCKER,
|
||||
title=f"Duplicate PRs on Issue #{issue.number}",
|
||||
summary=f"Issue #{issue.number} ({issue.title}) linked to multiple PRs.",
|
||||
work_kind="issue",
|
||||
work_number=issue.number,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link=f"/traffic",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
elif "claimed" in issue.badges or "in-review" in issue.badges:
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_WORKFLOW,
|
||||
f"Issue #{issue.number} Active",
|
||||
f"Issue #{issue.number} '{issue.title}' in state {', '.join(issue.badges)}.",
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-issue-routine-{issue.number}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_WORKFLOW,
|
||||
title=f"Routine Issue #{issue.number}",
|
||||
summary=f"Issue #{issue.number} ({issue.title}) state: {', '.join(issue.badges)}.",
|
||||
work_kind="issue",
|
||||
work_number=issue.number,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link=f"/queue",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
|
||||
# 3. Leases / Collisions
|
||||
for lease in lease_snap.reviewer_leases:
|
||||
if lease.get("is_expired") or lease.get("status") == "expired":
|
||||
pr_num = lease.get("pr_number") or lease.get("work_item_number")
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_LEASE,
|
||||
f"Reviewer Lease Expired for PR #{pr_num}",
|
||||
f"Reviewer lease for PR #{pr_num} has expired.",
|
||||
is_stale=True,
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-lease-exp-pr-{pr_num}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_LEASE,
|
||||
title=f"Expired Reviewer Lease (PR #{pr_num})",
|
||||
summary=f"Reviewer lease for PR #{pr_num} expired.",
|
||||
work_kind="pr",
|
||||
work_number=pr_num,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link="/leases",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
|
||||
for col_idx, collision in enumerate(lease_snap.duplicate_prs):
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_BLOCKER,
|
||||
f"Duplicate PR Collision ({collision.kind})",
|
||||
collision.message,
|
||||
is_blocker=True,
|
||||
)
|
||||
issue_part = collision.issue_number if collision.issue_number is not None else "none"
|
||||
kind_part = (collision.kind or "unknown").replace(" ", "-")
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-collision-{kind_part}-{issue_part}-{col_idx}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_BLOCKER,
|
||||
title=f"Collision Alert ({collision.kind})",
|
||||
summary=collision.message,
|
||||
work_kind="issue" if collision.issue_number else "pr",
|
||||
work_number=collision.issue_number,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link="/leases",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
|
||||
human_req_count = sum(1 for i in items if i.attention_class == ATTENTION_HUMAN_REQUIRED)
|
||||
operator_count = sum(1 for i in items if i.attention_class == ATTENTION_OPERATOR)
|
||||
routine_count = sum(1 for i in items if i.attention_class == ATTENTION_ROUTINE)
|
||||
|
||||
# Fetch errors are transport/load failures only — not probe results that
|
||||
# already surface as first-class notification items (PR #905 review B3).
|
||||
fetch_err = queue_snap.fetch_error or lease_snap.fetch_error
|
||||
if isinstance(fetch_err, (tuple, list)):
|
||||
fetch_err = "; ".join(fetch_err) if fetch_err else None
|
||||
|
||||
return NotificationSnapshot(
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
items=tuple(items),
|
||||
human_required_count=human_req_count,
|
||||
operator_count=operator_count,
|
||||
routine_count=routine_count,
|
||||
total_count=len(items),
|
||||
fetch_error=fetch_err,
|
||||
)
|
||||
|
||||
|
||||
def snapshot_to_dict(snapshot: NotificationSnapshot) -> dict[str, Any]:
|
||||
"""JSON-serializable export for /api/v1/notifications."""
|
||||
return snapshot.as_dict()
|
||||
@@ -0,0 +1,579 @@
|
||||
"""Read-only restart status, impact preview, and approval state (#667).
|
||||
|
||||
Phase 1 of the console restart surface. It *consumes* the #655 coordinator
|
||||
substrate and renders it; it never restarts, reloads, drains, approves, or kills
|
||||
anything. There is no apply path in this module, so there is no execution gate
|
||||
here to arm incorrectly — the only writes the console could perform are the ones
|
||||
it does not implement.
|
||||
|
||||
Sources, each independently fail-soft and each reported with its own
|
||||
:class:`SourceStatus`:
|
||||
|
||||
* :mod:`restart_coordinator` — restart-class policy matrix (#663) and the
|
||||
blast-radius impact report (#658).
|
||||
* :mod:`drain_proof` — drain checklist and gate verdict (#661), verified
|
||||
read-only against a caller-supplied proof.
|
||||
* :mod:`post_restart_reconcile` — post-restart completion proof (#662).
|
||||
* :mod:`webui.console_authz` — role authorization for the approval controls
|
||||
(#633).
|
||||
|
||||
Three rules this module holds itself to, because a status surface that lies is
|
||||
worse than one that is absent:
|
||||
|
||||
**A source that could not be read is reported unavailable, never green.** No
|
||||
default, placeholder, or self-comparison is substituted for a reading that
|
||||
failed. An unreadable control-plane DB yields ``inventory_complete=False``,
|
||||
which the coordinator itself turns into a fail-closed verdict.
|
||||
|
||||
**Authorization is asked the way execution would ask it.** Every authorization
|
||||
probe passes ``for_execution=True``, so the console reports whether the action
|
||||
could actually run rather than the weaker "this principal is the right role".
|
||||
While the console is in Phase 1 that answer is ``phase_not_active`` for every
|
||||
phase-2 action, and the surface says so plainly instead of showing an allow.
|
||||
|
||||
**The database is opened read-only.** ``ControlPlaneDB()`` creates directories
|
||||
and runs migrations on construction, which is a write; this module opens the
|
||||
sqlite file with ``mode=ro`` exactly as :mod:`webui.inventory` does, and treats
|
||||
a missing file as missing authority rather than an empty inventory.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sqlite3
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Callable, Mapping
|
||||
|
||||
import control_plane_db
|
||||
import drain_proof
|
||||
import restart_coordinator
|
||||
from webui import console_authz
|
||||
from webui.inventory import redact_path, scrub
|
||||
|
||||
# --- Source status ----------------------------------------------------------
|
||||
|
||||
STATUS_OK = "ok"
|
||||
STATUS_UNAVAILABLE = "unavailable"
|
||||
|
||||
#: Console actions whose authorization state this surface reports. Both are
|
||||
#: pre-existing #642 actions; this module adds no new console action because it
|
||||
#: performs no console action.
|
||||
REPORTED_ACTIONS: tuple[str, ...] = (
|
||||
"system.restart_namespace",
|
||||
"system.reload_namespace",
|
||||
)
|
||||
|
||||
#: The break-glass workflow (#664) is not consumed here. It is declared so the
|
||||
#: surface is honest about the gap rather than silently omitting a governance
|
||||
#: path the operator has been told exists.
|
||||
BREAK_GLASS_ISSUE = 664
|
||||
BREAK_GLASS_PENDING_REASON = (
|
||||
"The break-glass workflow (#664) is not yet available on this branch's "
|
||||
"base; no break-glass control is offered and none is implied."
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SourceStatus:
|
||||
"""Whether one backing source could be read, and why not when it could not."""
|
||||
|
||||
name: str
|
||||
status: str
|
||||
detail: str = ""
|
||||
|
||||
@property
|
||||
def available(self) -> bool:
|
||||
return self.status == STATUS_OK
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"name": self.name,
|
||||
"status": self.status,
|
||||
"available": self.available,
|
||||
"detail": self.detail,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RestartClassView:
|
||||
"""One row of the #663 restart-class matrix, scoped to the viewer's role."""
|
||||
|
||||
restart_class: str
|
||||
required_permission: str
|
||||
expected_blast_radius: str
|
||||
drain_requirement: str
|
||||
full_drain_required: bool
|
||||
approval_requirement: str
|
||||
request_roles: tuple[str, ...]
|
||||
execution_roles: tuple[str, ...]
|
||||
viewer_may_request: bool
|
||||
viewer_may_execute: bool
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"restart_class": self.restart_class,
|
||||
"required_permission": self.required_permission,
|
||||
"expected_blast_radius": self.expected_blast_radius,
|
||||
"drain_requirement": self.drain_requirement,
|
||||
"full_drain_required": self.full_drain_required,
|
||||
"approval_requirement": self.approval_requirement,
|
||||
"request_roles": list(self.request_roles),
|
||||
"execution_roles": list(self.execution_roles),
|
||||
"viewer_may_request": self.viewer_may_request,
|
||||
"viewer_may_execute": self.viewer_may_execute,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ActionAuthorization:
|
||||
"""Authorization state for one console action, asked as execution would."""
|
||||
|
||||
action_id: str
|
||||
summary: str
|
||||
required_role: str
|
||||
allowed: bool
|
||||
execution_enabled: bool
|
||||
reason_code: str
|
||||
detail: str
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"action_id": self.action_id,
|
||||
"summary": self.summary,
|
||||
"required_role": self.required_role,
|
||||
"allowed": self.allowed,
|
||||
"execution_enabled": self.execution_enabled,
|
||||
"reason_code": self.reason_code,
|
||||
"detail": self.detail,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BreakGlassSurface:
|
||||
"""Declared-but-unavailable break-glass panel (#664 is not on this base)."""
|
||||
|
||||
available: bool
|
||||
issue: int
|
||||
reason: str
|
||||
viewer_is_privileged: bool
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"available": self.available,
|
||||
"issue": self.issue,
|
||||
"reason": self.reason,
|
||||
"viewer_is_privileged": self.viewer_is_privileged,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RestartConsoleSnapshot:
|
||||
"""Everything the read-only restart console renders."""
|
||||
|
||||
generated_at: str
|
||||
viewer_role: str
|
||||
viewer_authenticated: bool
|
||||
read_only: bool
|
||||
impact: dict[str, Any] | None
|
||||
impact_source: SourceStatus
|
||||
drain: dict[str, Any] | None
|
||||
drain_source: SourceStatus
|
||||
reconcile: dict[str, Any] | None
|
||||
reconcile_source: SourceStatus
|
||||
restart_classes: tuple[RestartClassView, ...]
|
||||
authorizations: tuple[ActionAuthorization, ...]
|
||||
break_glass: BreakGlassSurface
|
||||
notes: tuple[str, ...] = field(default_factory=tuple)
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"generated_at": self.generated_at,
|
||||
"viewer_role": self.viewer_role,
|
||||
"viewer_authenticated": self.viewer_authenticated,
|
||||
"read_only": self.read_only,
|
||||
"impact": self.impact,
|
||||
"impact_source": self.impact_source.as_dict(),
|
||||
"drain": self.drain,
|
||||
"drain_source": self.drain_source.as_dict(),
|
||||
"reconcile": self.reconcile,
|
||||
"reconcile_source": self.reconcile_source.as_dict(),
|
||||
"restart_classes": [c.as_dict() for c in self.restart_classes],
|
||||
"authorizations": [a.as_dict() for a in self.authorizations],
|
||||
"break_glass": self.break_glass.as_dict(),
|
||||
"notes": list(self.notes),
|
||||
"links": {
|
||||
"issue": 667,
|
||||
"extends": 642,
|
||||
"umbrella": 655,
|
||||
"coordinator": 658,
|
||||
"drain_proof": 661,
|
||||
"reconcile": 662,
|
||||
"restart_classes": 663,
|
||||
"break_glass": BREAK_GLASS_ISSUE,
|
||||
"vision": 652,
|
||||
"roadmap": 653,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
# --- Control-plane inventory (read-only) ------------------------------------
|
||||
|
||||
|
||||
def read_control_plane_inventory(
|
||||
*,
|
||||
db_path: str | None = None,
|
||||
limit: int = 200,
|
||||
) -> dict[str, Any]:
|
||||
"""Read sessions and leases for an impact evaluation, read-only.
|
||||
|
||||
Returns the inventory mapping
|
||||
:func:`restart_coordinator.evaluate_restart_impact` expects.
|
||||
``inventory_complete`` is True only when every read succeeded, so a partial
|
||||
read denies rather than under-reporting the blast radius.
|
||||
|
||||
The database is never created, migrated, or written: a missing file means
|
||||
the console has no session authority, which is not the same as there being
|
||||
no sessions.
|
||||
"""
|
||||
|
||||
path = (db_path or control_plane_db.default_db_path() or "").strip()
|
||||
incomplete: list[str] = []
|
||||
|
||||
def _incomplete(reason: str) -> dict[str, Any]:
|
||||
return {
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"terminal_lock": None,
|
||||
"prior_recovery_attempts": [],
|
||||
"inventory_complete": False,
|
||||
"incomplete_reasons": [reason],
|
||||
}
|
||||
|
||||
if not path:
|
||||
return _incomplete("control-plane database path is not configured")
|
||||
if not os.path.exists(path):
|
||||
return _incomplete(
|
||||
f"control-plane database not present at {redact_path(path)}; "
|
||||
"no session or lease authority available"
|
||||
)
|
||||
|
||||
try:
|
||||
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=5)
|
||||
conn.row_factory = sqlite3.Row
|
||||
except sqlite3.Error as exc:
|
||||
return _incomplete(f"control-plane database could not be opened: {exc}")
|
||||
|
||||
sessions: list[dict[str, Any]] = []
|
||||
leases: list[dict[str, Any]] = []
|
||||
capped = max(1, int(limit))
|
||||
try:
|
||||
tables = {
|
||||
str(row[0])
|
||||
for row in conn.execute(
|
||||
"SELECT name FROM sqlite_master WHERE type = 'table'"
|
||||
).fetchall()
|
||||
}
|
||||
if "sessions" not in tables:
|
||||
incomplete.append("control-plane database has no sessions table")
|
||||
else:
|
||||
sessions = [
|
||||
dict(row)
|
||||
for row in conn.execute(
|
||||
"SELECT session_id, role, profile, pid, status,"
|
||||
" last_heartbeat_at FROM sessions"
|
||||
" WHERE status = 'active'"
|
||||
" ORDER BY last_heartbeat_at DESC LIMIT ?",
|
||||
(capped,),
|
||||
).fetchall()
|
||||
]
|
||||
|
||||
if "leases" not in tables:
|
||||
incomplete.append("control-plane database has no leases table")
|
||||
elif "work_items" not in tables:
|
||||
incomplete.append(
|
||||
"control-plane database has no work_items table; lease work "
|
||||
"identity cannot be resolved"
|
||||
)
|
||||
else:
|
||||
leases = [
|
||||
dict(row)
|
||||
for row in conn.execute(
|
||||
"SELECT l.lease_id, l.session_id, l.role, l.phase,"
|
||||
" l.status AS freshness, l.worktree_path,"
|
||||
" w.kind AS work_kind, w.number AS work_number"
|
||||
" FROM leases l"
|
||||
" JOIN work_items w ON w.work_item_id = l.work_item_id"
|
||||
" WHERE l.status = 'active'"
|
||||
" ORDER BY l.expires_at DESC LIMIT ?",
|
||||
(capped,),
|
||||
).fetchall()
|
||||
]
|
||||
except sqlite3.Error as exc:
|
||||
return _incomplete(f"control-plane database read failed: {exc}")
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
return {
|
||||
"sessions": sessions,
|
||||
"leases": leases,
|
||||
"terminal_lock": None,
|
||||
"prior_recovery_attempts": [],
|
||||
"inventory_complete": not incomplete,
|
||||
"incomplete_reasons": incomplete,
|
||||
}
|
||||
|
||||
|
||||
# --- Composition ------------------------------------------------------------
|
||||
|
||||
|
||||
def build_restart_class_views(viewer_role: str | None) -> tuple[RestartClassView, ...]:
|
||||
"""Render the #663 class matrix, marking what this viewer may request."""
|
||||
|
||||
normalized = str(viewer_role or "").strip().lower()
|
||||
views: list[RestartClassView] = []
|
||||
for policy in restart_coordinator.RESTART_CLASS_POLICIES.values():
|
||||
views.append(
|
||||
RestartClassView(
|
||||
restart_class=policy.restart_class.value,
|
||||
required_permission=policy.required_permission,
|
||||
expected_blast_radius=policy.expected_blast_radius,
|
||||
drain_requirement=policy.drain_requirement,
|
||||
full_drain_required=policy.full_drain_required,
|
||||
approval_requirement=policy.approval_requirement,
|
||||
request_roles=tuple(policy.request_roles),
|
||||
execution_roles=tuple(policy.execution_roles),
|
||||
viewer_may_request=normalized in policy.request_roles,
|
||||
viewer_may_execute=normalized in policy.execution_roles,
|
||||
)
|
||||
)
|
||||
return tuple(views)
|
||||
|
||||
|
||||
def build_action_authorizations(
|
||||
principal: console_authz.Principal | None,
|
||||
) -> tuple[ActionAuthorization, ...]:
|
||||
"""Authorization state for the approval controls, asked as execution.
|
||||
|
||||
``for_execution=True`` is deliberate. Asking without it answers "is this
|
||||
principal senior enough", which is not the question an operator looking at a
|
||||
control needs answered; asking with it answers "would this run", and while
|
||||
the console is in Phase 1 the honest answer is no.
|
||||
"""
|
||||
|
||||
results: list[ActionAuthorization] = []
|
||||
for action_id in REPORTED_ACTIONS:
|
||||
action = console_authz.get_action(action_id)
|
||||
decision = console_authz.authorize(action_id, principal, for_execution=True)
|
||||
results.append(
|
||||
ActionAuthorization(
|
||||
action_id=action_id,
|
||||
summary=action.summary if action else "",
|
||||
required_role=(
|
||||
action.minimum_role if action else console_authz.OPERATOR
|
||||
),
|
||||
allowed=bool(decision.allowed),
|
||||
execution_enabled=bool(decision.execution_enabled),
|
||||
reason_code=str(decision.reason_code or ""),
|
||||
detail=str(decision.detail or ""),
|
||||
)
|
||||
)
|
||||
return tuple(results)
|
||||
|
||||
|
||||
def viewer_is_privileged(principal: console_authz.Principal | None) -> bool:
|
||||
"""True when the viewer holds at least the operator role."""
|
||||
|
||||
who = principal if principal is not None else console_authz.ANONYMOUS
|
||||
if not who.authenticated:
|
||||
return False
|
||||
return who.rank >= console_authz.ROLE_ORDER.index(console_authz.OPERATOR)
|
||||
|
||||
|
||||
def load_impact_report(
|
||||
*,
|
||||
principal: console_authz.Principal | None = None,
|
||||
restart_class: str = restart_coordinator.RestartClass.FULL_MCP_RESTART.value,
|
||||
db_path: str | None = None,
|
||||
limit: int = 200,
|
||||
read_inventory: Callable[..., Mapping[str, Any]] | None = None,
|
||||
now: datetime | None = None,
|
||||
) -> tuple[dict[str, Any] | None, SourceStatus]:
|
||||
"""Evaluate the blast radius for *restart_class*, always dry-run."""
|
||||
|
||||
reader = read_inventory or read_control_plane_inventory
|
||||
try:
|
||||
inventory = dict(reader(db_path=db_path, limit=limit))
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return None, SourceStatus(
|
||||
"impact",
|
||||
STATUS_UNAVAILABLE,
|
||||
f"control-plane inventory failed: {type(exc).__name__}: {exc}",
|
||||
)
|
||||
|
||||
who = principal if principal is not None else console_authz.ANONYMOUS
|
||||
viewer_role = str(who.role or "").strip().lower()
|
||||
try:
|
||||
report = restart_coordinator.evaluate_restart_impact(
|
||||
inventory,
|
||||
now=now,
|
||||
dry_run=True,
|
||||
restart_class=restart_class,
|
||||
requester_role=viewer_role,
|
||||
requester_permissions=restart_coordinator.permissions_for_role(
|
||||
viewer_role
|
||||
),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return None, SourceStatus(
|
||||
"impact",
|
||||
STATUS_UNAVAILABLE,
|
||||
f"impact evaluation failed: {type(exc).__name__}: {exc}",
|
||||
)
|
||||
|
||||
payload = scrub(report.as_dict())
|
||||
detail = ""
|
||||
if not report.inventory_complete:
|
||||
detail = "; ".join(report.incomplete_reasons) or "inventory incomplete"
|
||||
return payload, SourceStatus("impact", STATUS_OK, detail)
|
||||
|
||||
|
||||
def load_drain_status(
|
||||
*,
|
||||
proof: Mapping[str, Any] | None = None,
|
||||
now: datetime | None = None,
|
||||
expected_impact_fingerprint: str | None = None,
|
||||
) -> tuple[dict[str, Any] | None, SourceStatus]:
|
||||
"""Verify a supplied drain proof read-only and report the verdict.
|
||||
|
||||
No proof supplied is not a failure and not a pass: it is reported as the
|
||||
absence of a proof, which is exactly what the #661 gate would deny on.
|
||||
"""
|
||||
|
||||
if proof is None:
|
||||
return None, SourceStatus(
|
||||
"drain",
|
||||
STATUS_UNAVAILABLE,
|
||||
"no drain proof supplied; the #661 gate denies a restart without a "
|
||||
"valid unexpired clean proof",
|
||||
)
|
||||
try:
|
||||
verified = drain_proof.verify_drain_proof(
|
||||
proof,
|
||||
now=now,
|
||||
expected_impact_fingerprint=expected_impact_fingerprint,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return None, SourceStatus(
|
||||
"drain",
|
||||
STATUS_UNAVAILABLE,
|
||||
f"drain proof verification failed: {type(exc).__name__}: {exc}",
|
||||
)
|
||||
return scrub(verified.as_dict()), SourceStatus("drain", STATUS_OK)
|
||||
|
||||
|
||||
def load_reconcile_status(
|
||||
*,
|
||||
load_proof: Callable[[], Any] | None = None,
|
||||
) -> tuple[dict[str, Any] | None, SourceStatus]:
|
||||
"""Report the most recent post-restart completion proof (#662)."""
|
||||
|
||||
if load_proof is None:
|
||||
return None, SourceStatus(
|
||||
"reconcile",
|
||||
STATUS_UNAVAILABLE,
|
||||
"no post-restart completion proof source is wired into this view",
|
||||
)
|
||||
try:
|
||||
proof = load_proof()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return None, SourceStatus(
|
||||
"reconcile",
|
||||
STATUS_UNAVAILABLE,
|
||||
f"reconcile proof unavailable: {type(exc).__name__}: {exc}",
|
||||
)
|
||||
if proof is None:
|
||||
return None, SourceStatus(
|
||||
"reconcile",
|
||||
STATUS_UNAVAILABLE,
|
||||
"no post-restart reconcile has been recorded",
|
||||
)
|
||||
payload = proof.as_dict() if hasattr(proof, "as_dict") else dict(proof)
|
||||
return scrub(payload), SourceStatus("reconcile", STATUS_OK)
|
||||
|
||||
|
||||
def load_restart_console_snapshot(
|
||||
*,
|
||||
principal: console_authz.Principal | None = None,
|
||||
restart_class: str = restart_coordinator.RestartClass.FULL_MCP_RESTART.value,
|
||||
db_path: str | None = None,
|
||||
limit: int = 200,
|
||||
drain_proof_payload: Mapping[str, Any] | None = None,
|
||||
read_inventory: Callable[..., Mapping[str, Any]] | None = None,
|
||||
load_reconcile_proof: Callable[[], Any] | None = None,
|
||||
now: datetime | None = None,
|
||||
) -> RestartConsoleSnapshot:
|
||||
"""Compose the read-only restart console snapshot."""
|
||||
|
||||
who = principal if principal is not None else console_authz.ANONYMOUS
|
||||
moment = now or _utc_now()
|
||||
|
||||
impact, impact_source = load_impact_report(
|
||||
principal=who,
|
||||
restart_class=restart_class,
|
||||
db_path=db_path,
|
||||
limit=limit,
|
||||
read_inventory=read_inventory,
|
||||
now=moment,
|
||||
)
|
||||
fingerprint = None
|
||||
if impact is not None:
|
||||
try:
|
||||
fingerprint = drain_proof.impact_fingerprint(impact)
|
||||
except Exception: # noqa: BLE001
|
||||
fingerprint = None
|
||||
|
||||
drain, drain_source = load_drain_status(
|
||||
proof=drain_proof_payload,
|
||||
now=moment,
|
||||
expected_impact_fingerprint=fingerprint,
|
||||
)
|
||||
reconcile, reconcile_source = load_reconcile_status(
|
||||
load_proof=load_reconcile_proof
|
||||
)
|
||||
|
||||
notes: list[str] = [
|
||||
"This surface is read-only: it evaluates and displays, and performs no "
|
||||
"restart, reload, drain, approval, or process action.",
|
||||
]
|
||||
if not impact_source.available:
|
||||
notes.append(
|
||||
"Impact preview unavailable — a restart decision must not be made "
|
||||
"from this page while the blast radius is unknown."
|
||||
)
|
||||
|
||||
return RestartConsoleSnapshot(
|
||||
generated_at=moment.isoformat(),
|
||||
viewer_role=str(who.role or "anonymous"),
|
||||
viewer_authenticated=bool(who.authenticated),
|
||||
read_only=True,
|
||||
impact=impact,
|
||||
impact_source=impact_source,
|
||||
drain=drain,
|
||||
drain_source=drain_source,
|
||||
reconcile=reconcile,
|
||||
reconcile_source=reconcile_source,
|
||||
restart_classes=build_restart_class_views(who.role),
|
||||
authorizations=build_action_authorizations(who),
|
||||
break_glass=BreakGlassSurface(
|
||||
available=False,
|
||||
issue=BREAK_GLASS_ISSUE,
|
||||
reason=BREAK_GLASS_PENDING_REASON,
|
||||
viewer_is_privileged=viewer_is_privileged(who),
|
||||
),
|
||||
notes=tuple(notes),
|
||||
)
|
||||
@@ -0,0 +1,299 @@
|
||||
"""HTML views for the read-only restart console (#667).
|
||||
|
||||
Every interpolated value passes through :func:`_esc`. Values that can carry a
|
||||
filesystem path or free-form operator text additionally pass through
|
||||
:func:`webui.inventory.scrub_text`, which redacts credential-shaped tokens
|
||||
*inside* a string rather than only at its start.
|
||||
|
||||
The page renders state and never offers a control that would mutate anything:
|
||||
the approval and break-glass panels report authorization and availability, and
|
||||
there is no form, button, or endpoint behind them.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
|
||||
from webui.inventory import scrub_text
|
||||
from webui.restart_console import RestartConsoleSnapshot, SourceStatus
|
||||
|
||||
|
||||
def _esc(value: object) -> str:
|
||||
"""Escape any value for HTML text or a quoted attribute."""
|
||||
if value is None:
|
||||
return ""
|
||||
return html.escape(str(value), quote=True)
|
||||
|
||||
|
||||
def _esc_text(value: object) -> str:
|
||||
"""Escape free-form text after redacting secrets embedded inside it."""
|
||||
if value is None:
|
||||
return ""
|
||||
return _esc(scrub_text(str(value)))
|
||||
|
||||
|
||||
def _bool_badge(
|
||||
value: bool, *, true_label: str = "yes", false_label: str = "no"
|
||||
) -> str:
|
||||
css = "badge-ok" if value else "badge-blocked"
|
||||
label = true_label if value else false_label
|
||||
return f'<span class="badge {css}">{_esc(label)}</span>'
|
||||
|
||||
|
||||
def _source_badge(source: SourceStatus) -> str:
|
||||
css = "badge-ok" if source.available else "badge-blocked"
|
||||
badge = f'<span class="badge {css}">{_esc(source.status)}</span>'
|
||||
if source.detail:
|
||||
badge += f' <span class="muted">{_esc_text(source.detail)}</span>'
|
||||
return badge
|
||||
|
||||
|
||||
def _notes_block(snapshot: RestartConsoleSnapshot) -> str:
|
||||
if not snapshot.notes:
|
||||
return ""
|
||||
items = "".join(f"<li>{_esc_text(note)}</li>" for note in snapshot.notes)
|
||||
return f"<ul class='reasons'>{items}</ul>"
|
||||
|
||||
|
||||
def _impact_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
head = (
|
||||
"<section class='health-card'>"
|
||||
f"<h3>Impact preview {_source_badge(snapshot.impact_source)}</h3>"
|
||||
)
|
||||
impact = snapshot.impact
|
||||
if impact is None:
|
||||
return (
|
||||
head
|
||||
+ "<p class='muted'>No impact preview is available, so the blast "
|
||||
"radius of a restart is unknown. Treat this as unsafe.</p></section>"
|
||||
)
|
||||
|
||||
counts = impact.get("counts") or {}
|
||||
verdict = str(impact.get("verdict") or "unknown")
|
||||
verdict_css = "badge-ok" if verdict == "safe" else "badge-blocked"
|
||||
rows = "".join(
|
||||
f"<tr><th>{_esc(key.replace('_', ' '))}</th><td>{_esc(value)}</td></tr>"
|
||||
for key, value in sorted(counts.items())
|
||||
)
|
||||
reasons = "".join(
|
||||
f"<li>{_esc_text(reason)}</li>" for reason in (impact.get("reasons") or [])
|
||||
)
|
||||
incomplete = ""
|
||||
if not impact.get("inventory_complete", False):
|
||||
detail = "; ".join(str(r) for r in (impact.get("incomplete_reasons") or []))
|
||||
incomplete = (
|
||||
"<p class='error'><strong>Inventory incomplete:</strong> "
|
||||
f"{_esc_text(detail or 'unspecified')}. The coordinator fails "
|
||||
"closed on an incomplete inventory.</p>"
|
||||
)
|
||||
|
||||
sessions = impact.get("affected_sessions") or []
|
||||
session_rows = "".join(
|
||||
"<tr>"
|
||||
f"<td><code>{_esc(s.get('session_id'))}</code></td>"
|
||||
f"<td>{_esc(s.get('role'))}</td>"
|
||||
f"<td>{_esc(s.get('pid'))}</td>"
|
||||
f"<td>{_bool_badge(bool(s.get('live')), true_label='live', false_label='idle')}</td>"
|
||||
f"<td>{_bool_badge(not s.get('heartbeat_stale'), true_label='fresh', false_label='stale')}</td>"
|
||||
"</tr>"
|
||||
for s in sessions[:50]
|
||||
)
|
||||
session_table = (
|
||||
"<h4>Sessions a restart would terminate</h4>"
|
||||
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||
"<th>Session</th><th>Role</th><th>PID</th><th>State</th>"
|
||||
"<th>Heartbeat</th></tr></thead><tbody>"
|
||||
f"{session_rows}</tbody></table></div>"
|
||||
if session_rows
|
||||
else "<p class='muted'>No affected sessions reported.</p>"
|
||||
)
|
||||
truncated = (
|
||||
f"<p class='muted'>Showing the first 50 of {_esc(len(sessions))} "
|
||||
"affected sessions.</p>"
|
||||
if len(sessions) > 50
|
||||
else ""
|
||||
)
|
||||
|
||||
return (
|
||||
head
|
||||
+ "<p class='health-headline'>Verdict "
|
||||
f"<span class='badge {verdict_css}'>{_esc(verdict)}</span> · "
|
||||
f"blast radius <code>{_esc(impact.get('blast_radius'))}</code> · "
|
||||
f"class <code>{_esc(impact.get('restart_class'))}</code></p>"
|
||||
+ incomplete
|
||||
+ (f"<ul class='reasons'>{reasons}</ul>" if reasons else "")
|
||||
+ (f"<table class='registry'><tbody>{rows}</tbody></table>" if rows else "")
|
||||
+ session_table
|
||||
+ truncated
|
||||
+ "</section>"
|
||||
)
|
||||
|
||||
|
||||
def _drain_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
head = (
|
||||
"<section class='health-card'>"
|
||||
f"<h3>Drain proof {_source_badge(snapshot.drain_source)}</h3>"
|
||||
)
|
||||
drain = snapshot.drain
|
||||
if drain is None:
|
||||
return (
|
||||
head
|
||||
+ "<p class='muted'>No drain proof has been presented to this view. "
|
||||
"The #661 gate authorizes a restart only against a valid, unexpired, "
|
||||
"clean proof, so the absence of one is a denial, not a pass.</p>"
|
||||
"</section>"
|
||||
)
|
||||
reasons = "".join(
|
||||
f"<li>{_esc_text(reason)}</li>" for reason in (drain.get("reasons") or [])
|
||||
)
|
||||
return (
|
||||
head
|
||||
+ "<table class='registry'><tbody>"
|
||||
f"<tr><th>Valid</th><td>{_bool_badge(bool(drain.get('valid')))}</td></tr>"
|
||||
f"<tr><th>Clean</th><td>{_bool_badge(bool(drain.get('clean')))}</td></tr>"
|
||||
f"<tr><th>Expired</th><td>{_bool_badge(not drain.get('expired'), true_label='no', false_label='yes')}</td></tr>"
|
||||
f"<tr><th>Tampered</th><td>{_bool_badge(not drain.get('tampered'), true_label='no', false_label='yes')}</td></tr>"
|
||||
f"<tr><th>Proof id</th><td><code>{_esc(drain.get('proof_id'))}</code></td></tr>"
|
||||
"</tbody></table>"
|
||||
+ (f"<ul class='reasons'>{reasons}</ul>" if reasons else "")
|
||||
+ "</section>"
|
||||
)
|
||||
|
||||
|
||||
def _reconcile_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
head = (
|
||||
"<section class='health-card'>"
|
||||
f"<h3>Post-restart reconcile {_source_badge(snapshot.reconcile_source)}</h3>"
|
||||
)
|
||||
proof = snapshot.reconcile
|
||||
if proof is None:
|
||||
return (
|
||||
head
|
||||
+ "<p class='muted'>No post-restart completion proof is recorded. "
|
||||
"Until one is, the last restart's recovery state is unproven.</p>"
|
||||
"</section>"
|
||||
)
|
||||
items = "".join(
|
||||
"<tr>"
|
||||
f"<td>{_esc(item.get('dimension'))}</td>"
|
||||
f"<td>{_esc(item.get('status'))}</td>"
|
||||
f"<td>{_esc_text(item.get('summary'))}</td>"
|
||||
f"<td>{_bool_badge(not item.get('follow_up_required'), true_label='no', false_label='yes')}</td>"
|
||||
"</tr>"
|
||||
for item in (proof.get("items") or [])
|
||||
)
|
||||
return (
|
||||
head
|
||||
+ "<p class='health-headline'>Status "
|
||||
f"<code>{_esc(proof.get('overall_status'))}</code> · mode "
|
||||
f"<code>{_esc(proof.get('mode'))}</code> · resolved "
|
||||
f"{_esc(proof.get('resolved_count'))} · unresolved "
|
||||
f"{_esc(proof.get('unresolved_count'))}</p>"
|
||||
+ (
|
||||
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||
"<th>Dimension</th><th>Status</th><th>Summary</th>"
|
||||
"<th>Follow-up required</th></tr></thead><tbody>"
|
||||
f"{items}</tbody></table></div>"
|
||||
if items
|
||||
else "<p class='muted'>No reconcile dimensions reported.</p>"
|
||||
)
|
||||
+ "</section>"
|
||||
)
|
||||
|
||||
|
||||
def _class_matrix_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
rows = "".join(
|
||||
"<tr>"
|
||||
f"<td><code>{_esc(view.restart_class)}</code></td>"
|
||||
f"<td><code>{_esc(view.required_permission)}</code></td>"
|
||||
f"<td>{_esc(view.expected_blast_radius)}</td>"
|
||||
f"<td>{_esc(view.drain_requirement)}</td>"
|
||||
f"<td>{_esc(view.approval_requirement)}</td>"
|
||||
f"<td>{_bool_badge(view.viewer_may_request)}</td>"
|
||||
f"<td>{_bool_badge(view.viewer_may_execute)}</td>"
|
||||
"</tr>"
|
||||
for view in snapshot.restart_classes
|
||||
)
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Restart classes</h3>"
|
||||
"<p class='muted'>The least-privilege matrix each restart request is "
|
||||
"resolved against. “You may request” and “you may "
|
||||
"execute” are computed for the current viewer role, not for a "
|
||||
"generic operator.</p>"
|
||||
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||
"<th>Class</th><th>Permission</th><th>Blast radius</th>"
|
||||
"<th>Drain</th><th>Approval</th><th>You may request</th>"
|
||||
"<th>You may execute</th></tr></thead><tbody>"
|
||||
f"{rows}</tbody></table></div>"
|
||||
"</section>"
|
||||
)
|
||||
|
||||
|
||||
def _approval_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
rows = "".join(
|
||||
"<tr>"
|
||||
f"<td><code>{_esc(a.action_id)}</code></td>"
|
||||
f"<td>{_esc(a.required_role)}</td>"
|
||||
f"<td>{_bool_badge(a.allowed)}</td>"
|
||||
f"<td>{_bool_badge(a.execution_enabled)}</td>"
|
||||
f"<td><code>{_esc(a.reason_code)}</code></td>"
|
||||
f"<td>{_esc_text(a.detail)}</td>"
|
||||
"</tr>"
|
||||
for a in snapshot.authorizations
|
||||
)
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Approval controls</h3>"
|
||||
"<p class='muted'>Authorization is probed the way execution would probe "
|
||||
"it, so “execution enabled” answers whether the action would "
|
||||
"actually run — not merely whether this role outranks the requirement. "
|
||||
"No control on this page performs the action.</p>"
|
||||
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||
"<th>Action</th><th>Required role</th><th>Authorized</th>"
|
||||
"<th>Execution enabled</th><th>Reason</th><th>Detail</th>"
|
||||
"</tr></thead><tbody>"
|
||||
f"{rows}</tbody></table></div>"
|
||||
"</section>"
|
||||
)
|
||||
|
||||
|
||||
def _break_glass_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
bg = snapshot.break_glass
|
||||
if not bg.viewer_is_privileged:
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Break-glass</h3>"
|
||||
"<p class='muted'>Break-glass status is visible to operator-class "
|
||||
"roles only. Your role does not carry that authority, so no "
|
||||
"emergency surface is shown.</p>"
|
||||
"</section>"
|
||||
)
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Break-glass "
|
||||
f"{_bool_badge(bg.available, true_label='available', false_label='unavailable')}"
|
||||
"</h3>"
|
||||
f"<p class='muted'>{_esc_text(bg.reason)}</p>"
|
||||
f"<p class='meta'>Tracked by issue #{_esc(bg.issue)}.</p>"
|
||||
"</section>"
|
||||
)
|
||||
|
||||
|
||||
def render_restart_console_page(snapshot: RestartConsoleSnapshot) -> str:
|
||||
"""Render the whole read-only restart console body."""
|
||||
|
||||
return (
|
||||
"<h2>Restart status and impact</h2>"
|
||||
f"<p class='meta'>Generated <code>{_esc(snapshot.generated_at)}</code> · "
|
||||
f"viewer role <code>{_esc(snapshot.viewer_role)}</code> · "
|
||||
f"authenticated {_bool_badge(snapshot.viewer_authenticated)} · "
|
||||
f"read-only {_bool_badge(snapshot.read_only)}</p>"
|
||||
+ _notes_block(snapshot)
|
||||
+ _impact_section(snapshot)
|
||||
+ _drain_section(snapshot)
|
||||
+ _reconcile_section(snapshot)
|
||||
+ _class_matrix_section(snapshot)
|
||||
+ _approval_section(snapshot)
|
||||
+ _break_glass_section(snapshot)
|
||||
)
|
||||
Reference in New Issue
Block a user