Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f0c6255d7d | ||
|
|
d7ad2838ec | ||
|
|
c6d68dbc7b | ||
|
|
c83a10d7c2 | ||
|
|
ca22c326a4 | ||
|
|
3bbe6df6c7 | ||
|
|
71031c812e | ||
|
|
e43ddd3cbe | ||
|
|
a64ba08e27 | ||
|
|
26f54851d1 | ||
|
|
f02a2dc030 | ||
|
|
77d808e7d4 | ||
|
|
bb8c3a537b | ||
|
|
04ae3532cc | ||
|
|
7bb5ff4719 | ||
|
|
5b7ceefa9a | ||
|
|
4a2fae8495 | ||
|
|
461e1dac78 | ||
|
|
9c69bfcd80 | ||
|
|
6010f4295b | ||
|
|
9b8e315b49 | ||
|
|
9a01543477 | ||
|
|
e91b94db56 | ||
|
|
4f06d30e07 | ||
|
|
211890f361 | ||
|
|
1c88b87ec5 | ||
|
|
b993ad1c64 | ||
|
|
76f293eb28 | ||
|
|
d0006e9f71 | ||
|
|
54559aebc3 | ||
|
|
715863799f | ||
|
|
6da68fffb8 | ||
|
|
daf7ed4c2b | ||
|
|
8598537a35 | ||
|
|
53ce1b1a5e | ||
|
|
220361ad94 | ||
|
|
433f66add8 | ||
|
|
6e6ca94338 | ||
|
|
d5d121a21b | ||
|
|
1ca2b50406 | ||
|
|
9bc021e9c0 | ||
|
|
2f4dec8323 | ||
|
|
3a9d634c17 | ||
|
|
a81db75402 | ||
|
|
930dc24632 | ||
|
|
2068bae341 | ||
|
|
7af40fb5ff | ||
|
|
619f679077 | ||
|
|
9517834913 | ||
|
|
824c42f7e3 | ||
|
|
578c44b685 | ||
|
|
3b68d15593 | ||
|
|
41622c5985 | ||
|
|
a4c73766f4 | ||
|
|
9f686253eb | ||
|
|
b2e28428a4 | ||
|
|
95e4aae287 | ||
|
|
301c78de20 | ||
|
|
dac40ab9b3 | ||
|
|
ccde9e8f11 | ||
|
|
1948d3dc21 | ||
|
|
714190e02a | ||
|
|
069a9af7e6 | ||
|
|
1cbbde0089 | ||
|
|
0a78da39e5 |
+212
-33
@@ -23,8 +23,10 @@ import json
|
||||
import os
|
||||
import uuid
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
import maintenance_drain
|
||||
from control_plane_db import (
|
||||
ControlPlaneDB,
|
||||
ControlPlaneError,
|
||||
@@ -133,10 +135,15 @@ ROLE_ACTIONS: dict[str, tuple[tuple[str, ...], tuple[str, ...]]] = {
|
||||
|
||||
|
||||
# Body phrases that prove an issue is an implementation container, not a
|
||||
# unit of direct author work (#844). Matched case-insensitively against the
|
||||
# issue body. Title alone is never sufficient (ordinary issues may mention
|
||||
# "epic" incidentally).
|
||||
# unit of direct author work (#844 / #854). Matched case-insensitively against
|
||||
# the issue body. Title alone is never sufficient (ordinary issues may mention
|
||||
# "epic", "roadmap", "vision", or "umbrella" incidentally).
|
||||
#
|
||||
# #854 extends the #844 marker set so product-vision (#652), phased-roadmap
|
||||
# (#653), and umbrella (#655) coordination records — which do not use the word
|
||||
# "epic" — are classified with the same semantic exclusion as epic containers.
|
||||
_CHILD_ONLY_BODY_MARKERS: tuple[str, ...] = (
|
||||
# Epic / child-only (#844, live #631)
|
||||
"implementation is delivered via child issues only",
|
||||
"implementation is delivered through child issues only",
|
||||
"implementation is delivered via child issues",
|
||||
@@ -150,9 +157,26 @@ _CHILD_ONLY_BODY_MARKERS: tuple[str, ...] = (
|
||||
"coordination container",
|
||||
"child-only container",
|
||||
"implementation is delegated to child",
|
||||
# Vision / roadmap / umbrella coordination (#854, live #652/#653/#655).
|
||||
# Prefer authoritative non-implementation / child-only scope language over
|
||||
# bare words like "roadmap" so ordinary implementable issues that mention
|
||||
# a parent vision or roadmap stay eligible.
|
||||
"do not implement features on this issue",
|
||||
"implementing features on this roadmap issue",
|
||||
"implementation is via linked children only",
|
||||
"no product feature claimed complete on this issue alone",
|
||||
"phased delivery roadmap and epic sequencing",
|
||||
"this issue is the enduring source of truth",
|
||||
"enduring source of truth for the",
|
||||
"canonical product vision — enduring source of truth",
|
||||
"canonical product vision - enduring source of truth",
|
||||
"state: vision-active",
|
||||
"state: roadmap-active",
|
||||
)
|
||||
|
||||
# Explicit epic / umbrella labels (structured evidence preferred over title).
|
||||
# Explicit epic / umbrella / vision / roadmap labels (structured evidence
|
||||
# preferred over title). Tracker alone is *not* included — ordinary issues
|
||||
# may carry a tracker label without being non-implementable containers.
|
||||
_EPIC_LABELS: frozenset[str] = frozenset(
|
||||
{
|
||||
"type:epic",
|
||||
@@ -161,6 +185,16 @@ _EPIC_LABELS: frozenset[str] = frozenset(
|
||||
"scope:epic",
|
||||
"type:umbrella",
|
||||
"umbrella",
|
||||
"kind:umbrella",
|
||||
"scope:umbrella",
|
||||
"type:vision",
|
||||
"vision",
|
||||
"kind:vision",
|
||||
"scope:vision",
|
||||
"type:roadmap",
|
||||
"roadmap",
|
||||
"kind:roadmap",
|
||||
"scope:roadmap",
|
||||
}
|
||||
)
|
||||
|
||||
@@ -222,16 +256,45 @@ class WorkCandidate:
|
||||
}
|
||||
|
||||
|
||||
def _title_container_prefix(title_l: str) -> str | None:
|
||||
"""Return a coordination-title prefix token if *title_l* uses one (#854).
|
||||
|
||||
Title prefixes alone never exclude; they only corroborate body/label
|
||||
evidence. Ordinary issues may say "roadmap" or "vision" mid-title.
|
||||
"""
|
||||
for prefix, token in (
|
||||
("epic:", "title_epic_prefix"),
|
||||
("epic ", "title_epic_prefix"),
|
||||
("umbrella:", "title_umbrella_prefix"),
|
||||
("umbrella ", "title_umbrella_prefix"),
|
||||
("roadmap:", "title_roadmap_prefix"),
|
||||
("roadmap ", "title_roadmap_prefix"),
|
||||
("product vision:", "title_vision_prefix"),
|
||||
("product vision ", "title_vision_prefix"),
|
||||
("vision:", "title_vision_prefix"),
|
||||
("vision ", "title_vision_prefix"),
|
||||
):
|
||||
if title_l.startswith(prefix):
|
||||
return token
|
||||
return None
|
||||
|
||||
|
||||
def classify_epic_or_child_only_container(
|
||||
c: WorkCandidate,
|
||||
) -> tuple[bool, str | None]:
|
||||
"""Return whether *c* is an epic / child-only implementation container (#844).
|
||||
"""Return whether *c* is a non-implementable coordination container (#844/#854).
|
||||
|
||||
Exclusion uses structured evidence first (labels, body scope language).
|
||||
A bare title containing the word "epic" is **not** enough — ordinary
|
||||
implementable issues may mention epics incidentally. A title that is
|
||||
explicitly prefixed ``Epic:`` only counts when the body also proves
|
||||
child-only / no-direct-implementation scope (or an epic label is present).
|
||||
A bare title containing the words "epic", "roadmap", "vision", or
|
||||
"umbrella" is **not** enough — ordinary implementable issues may mention
|
||||
those terms incidentally. Explicit title prefixes (``Epic:``, ``Roadmap:``,
|
||||
``Product vision:``, ``Umbrella:``) only count when the body also proves
|
||||
child-only / no-direct-implementation scope (or a container label is
|
||||
present).
|
||||
|
||||
Covers epic, product-vision, phased-roadmap, umbrella, and child-only
|
||||
records so the allocator never assigns coordination containers as direct
|
||||
author work.
|
||||
|
||||
PRs are never classified as containers here (they already have a head).
|
||||
"""
|
||||
@@ -245,25 +308,28 @@ def classify_epic_or_child_only_container(
|
||||
title_l = title.lower()
|
||||
|
||||
body_hits = [m for m in _CHILD_ONLY_BODY_MARKERS if m in body_l]
|
||||
title_epic_prefix = title_l.startswith("epic:") or title_l.startswith("epic ")
|
||||
title_prefix = _title_container_prefix(title_l)
|
||||
|
||||
if epic_label:
|
||||
detail = f"label={epic_label[0]}"
|
||||
if body_hits:
|
||||
detail = f"{detail}; body_marker={body_hits[0]!r}"
|
||||
if title_prefix:
|
||||
detail = f"{title_prefix}; {detail}"
|
||||
return True, detail
|
||||
|
||||
if body_hits:
|
||||
# Body proves child-only / umbrella scope. Title "Epic:" is corroborating
|
||||
# but not required — containers without the word still exclude.
|
||||
# Body proves child-only / vision / roadmap / umbrella scope. Title
|
||||
# prefixes are corroborating but not required — containers without the
|
||||
# title word still exclude.
|
||||
detail = f"body_marker={body_hits[0]!r}"
|
||||
if title_epic_prefix:
|
||||
detail = f"title_epic_prefix; {detail}"
|
||||
if title_prefix:
|
||||
detail = f"{title_prefix}; {detail}"
|
||||
return True, detail
|
||||
|
||||
# Title-only "Epic:" without body scope evidence is insufficient (#844 AC:
|
||||
# eligibility does not rely solely on the word "Epic" in a title).
|
||||
# Similarly, incidental "epic" mid-title without markers stays eligible.
|
||||
# Title-only coordination prefix without body scope evidence is
|
||||
# insufficient (#844/#854 AC: eligibility does not rely solely on a title
|
||||
# word). Incidental mid-title mentions without markers stay eligible.
|
||||
return False, None
|
||||
|
||||
|
||||
@@ -674,6 +740,46 @@ def normalize_exclude_issue_numbers(
|
||||
return sorted(out)
|
||||
|
||||
|
||||
def _claim_expires_at(claim: Any) -> datetime | None:
|
||||
"""Parse a claim's ``expires_at``, or ``None`` when it is absent/malformed."""
|
||||
if not isinstance(claim, Mapping):
|
||||
return None
|
||||
text = str(claim.get("expires_at") or "").strip()
|
||||
if not text:
|
||||
return None
|
||||
if text.endswith("Z"):
|
||||
text = text[:-1] + "+00:00"
|
||||
try:
|
||||
parsed = datetime.fromisoformat(text)
|
||||
except ValueError:
|
||||
return None
|
||||
if parsed.tzinfo is None:
|
||||
parsed = parsed.replace(tzinfo=timezone.utc)
|
||||
return parsed.astimezone(timezone.utc)
|
||||
|
||||
|
||||
def _drop_expired_claims(
|
||||
claims: Mapping[tuple[str, int], dict[str, Any]],
|
||||
*,
|
||||
now: datetime | None = None,
|
||||
) -> dict[tuple[str, int], dict[str, Any]]:
|
||||
"""Claims minus those whose lease has already expired (#643).
|
||||
|
||||
The read-only mirror of ``expire_stale_leases``: the sweep marks such rows
|
||||
``expired`` so they stop being returned as claims, and this reaches the same
|
||||
view without writing. A claim with no parseable ``expires_at`` is **kept** —
|
||||
an unreadable expiry is not evidence that work is free.
|
||||
"""
|
||||
moment = now or datetime.now(timezone.utc)
|
||||
kept: dict[tuple[str, int], dict[str, Any]] = {}
|
||||
for key, claim in (claims or {}).items():
|
||||
expires_at = _claim_expires_at(claim)
|
||||
if expires_at is not None and expires_at <= moment:
|
||||
continue
|
||||
kept[key] = claim
|
||||
return kept
|
||||
|
||||
|
||||
def candidate_set_fingerprint(
|
||||
candidates: Sequence[WorkCandidate],
|
||||
*,
|
||||
@@ -762,12 +868,22 @@ def allocate_next_work(
|
||||
exclude_issue_numbers: Sequence[int] | None = None,
|
||||
expected_candidate_set_fingerprint: str | None = None,
|
||||
allocation_mode: str | None = None,
|
||||
side_effect_free: bool = False,
|
||||
) -> dict[str, Any]:
|
||||
"""Select and optionally reserve the next work unit via control-plane DB.
|
||||
|
||||
*apply=False* (default): dry-run selection only — no lease/assignment.
|
||||
*apply=True*: atomic ``assign_and_lease`` for the selected candidate.
|
||||
|
||||
*side_effect_free* (#643): a dry run that writes **nothing** to the
|
||||
control-plane DB. A plain ``apply=False`` still registered a session row and
|
||||
swept stale leases globally, so a caller advertising a read-only preview was
|
||||
mutating on every call. Under this flag both writes are suppressed and stale
|
||||
leases are instead filtered out of the claim map in memory, which yields the
|
||||
same selection the sweep would have produced without persisting anything.
|
||||
Incompatible with *apply* — the combination fails closed rather than
|
||||
silently reserving.
|
||||
|
||||
*allocation_mode* (#840): ``cross_role`` (default for controller) inspects
|
||||
the complete queue and returns one authoritative selection naming the
|
||||
required downstream role/profile/action. ``role_scoped`` keeps prior
|
||||
@@ -821,41 +937,98 @@ def allocate_next_work(
|
||||
"allocation_mode": (allocation_mode or "").strip() or None,
|
||||
}
|
||||
|
||||
session_id = (session_id or "").strip() or f"alloc-{uuid.uuid4().hex[:12]}"
|
||||
# #659 AC2: while maintenance drain is active, no new work is assigned —
|
||||
# for dry-run and apply alike, so a preview can never be read as evidence
|
||||
# that work was assignable during the drain. Checked before session
|
||||
# registration so a drained allocator leaves no new state behind.
|
||||
try:
|
||||
db.upsert_session(
|
||||
session_id=session_id,
|
||||
role=role_norm,
|
||||
profile=profile_name,
|
||||
pid=os.getpid(),
|
||||
controller_instance_id=controller_instance_id,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — surface structured
|
||||
drain_record = db.read_maintenance_drain(remote=remote, org=org, repo=repo)
|
||||
except Exception as exc: # noqa: BLE001 — unreadable drain state fails closed
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"reasons": [
|
||||
f"failed to register session in control-plane DB: {exc} "
|
||||
"(fail closed, #613)"
|
||||
f"maintenance-drain state lookup failed: {exc} (fail closed, #659)"
|
||||
],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
|
||||
# Expire stale leases globally before selection.
|
||||
try:
|
||||
db.expire_stale_leases()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
drain_decision = maintenance_drain.classify_assignment(drain_record)
|
||||
if not drain_decision["assignment_allowed"]:
|
||||
return {
|
||||
"success": True,
|
||||
"outcome": OUTCOME_WAIT,
|
||||
"apply": apply,
|
||||
"role": role_norm,
|
||||
"allocation_mode": mode,
|
||||
"remote": remote,
|
||||
"org": org,
|
||||
"repo": repo,
|
||||
"selected": None,
|
||||
"reasons": list(drain_decision["reasons"]),
|
||||
"reason_code": drain_decision["reason_code"],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
"maintenance_drain": maintenance_drain.status_payload(
|
||||
drain_record, remote=remote, org=org, repo=repo
|
||||
),
|
||||
}
|
||||
|
||||
# A side-effect-free run may never reserve: reserving is a write, and the
|
||||
# flag is the caller's assertion that this call writes nothing (#643).
|
||||
if side_effect_free and apply:
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"reasons": [f"lease expiry failed: {exc} (fail closed)"],
|
||||
"apply": True,
|
||||
"reasons": [
|
||||
"side_effect_free is incompatible with apply=True; an "
|
||||
"assignment is a write (fail closed, #643)"
|
||||
],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
|
||||
session_id = (session_id or "").strip() or f"alloc-{uuid.uuid4().hex[:12]}"
|
||||
if not side_effect_free:
|
||||
try:
|
||||
db.upsert_session(
|
||||
session_id=session_id,
|
||||
role=role_norm,
|
||||
profile=profile_name,
|
||||
pid=os.getpid(),
|
||||
controller_instance_id=controller_instance_id,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — surface structured
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"reasons": [
|
||||
f"failed to register session in control-plane DB: {exc} "
|
||||
"(fail closed, #613)"
|
||||
],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
|
||||
# Expire stale leases globally before selection.
|
||||
try:
|
||||
db.expire_stale_leases()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"reasons": [f"lease expiry failed: {exc} (fail closed)"],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
|
||||
terminal = None
|
||||
try:
|
||||
terminal = db.get_active_terminal_lock(remote=remote, org=org, repo=repo)
|
||||
@@ -889,6 +1062,12 @@ def allocate_next_work(
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
if side_effect_free:
|
||||
# ``list_active_claims`` filters on status alone, so without the
|
||||
# global sweep an already-expired lease would still read as a live
|
||||
# claim and the preview would report work as taken that is free.
|
||||
# Drop those in memory: same view the sweep produces, no write.
|
||||
claims = _drop_expired_claims(claims)
|
||||
|
||||
try:
|
||||
exclude_nums = normalize_exclude_issue_numbers(exclude_issue_numbers)
|
||||
|
||||
+194
-1
@@ -31,8 +31,9 @@ from typing import Any, Iterator, Sequence
|
||||
|
||||
import dependency_graph
|
||||
import gitea_audit
|
||||
import maintenance_drain
|
||||
|
||||
SCHEMA_VERSION = 5
|
||||
SCHEMA_VERSION = 6
|
||||
|
||||
# Assignable work kinds only — raw monitoring incidents are never work items.
|
||||
WORK_KINDS = frozenset({"issue", "pr"})
|
||||
@@ -239,6 +240,31 @@ CREATE INDEX IF NOT EXISTS idx_session_checkpoints_session
|
||||
CREATE INDEX IF NOT EXISTS idx_session_checkpoints_work
|
||||
ON session_checkpoints(remote, org, repo, work_kind, work_number);
|
||||
|
||||
-- Graceful maintenance-drain state (#659). One current row per repository
|
||||
-- scope — drain is a *state*, not a history, so entering and exiting update
|
||||
-- the same row and every transition is audited to ``events``. Creating the
|
||||
-- table is the v5->v6 migration: additive, idempotent, and it never touches
|
||||
-- prior tables. ``state`` is CHECK-constrained so an unknown value can never
|
||||
-- be written and later read as "not draining".
|
||||
CREATE TABLE IF NOT EXISTS maintenance_drain (
|
||||
drain_id TEXT PRIMARY KEY,
|
||||
remote TEXT NOT NULL,
|
||||
org TEXT NOT NULL,
|
||||
repo TEXT NOT NULL,
|
||||
state TEXT NOT NULL DEFAULT 'inactive'
|
||||
CHECK (state IN ('inactive', 'draining')),
|
||||
reason TEXT NOT NULL DEFAULT '',
|
||||
requested_by TEXT NOT NULL DEFAULT '',
|
||||
requested_by_profile TEXT NOT NULL DEFAULT '',
|
||||
session_id TEXT NOT NULL DEFAULT '',
|
||||
entered_at TEXT NOT NULL DEFAULT '',
|
||||
exited_at TEXT NOT NULL DEFAULT '',
|
||||
drain_schema_version INTEGER NOT NULL DEFAULT 6,
|
||||
created_at TEXT NOT NULL,
|
||||
updated_at TEXT NOT NULL,
|
||||
UNIQUE (remote, org, repo)
|
||||
);
|
||||
|
||||
-- Model usage, token cost, latency, and performance events (#651)
|
||||
CREATE TABLE IF NOT EXISTS usage_events (
|
||||
usage_id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
@@ -3025,3 +3051,170 @@ class ControlPlaneDB:
|
||||
"live_lease_id": None if live_lease_id is None else str(live_lease_id),
|
||||
"reconcile_action": "reconcile_required" if stale else "safe_to_resume",
|
||||
}
|
||||
|
||||
# ── Maintenance drain (#659) ─────────────────────────────────────────────
|
||||
|
||||
@staticmethod
|
||||
def _maintenance_drain_row(row: sqlite3.Row | None) -> dict[str, Any] | None:
|
||||
"""Convert a ``maintenance_drain`` row to a plain record."""
|
||||
if row is None:
|
||||
return None
|
||||
return {key: row[key] for key in row.keys()}
|
||||
|
||||
def read_maintenance_drain(
|
||||
self, *, remote: str, org: str, repo: str
|
||||
) -> dict[str, Any] | None:
|
||||
"""Return the current drain record for a scope, or None if never set.
|
||||
|
||||
None and a stored ``inactive`` row mean the same thing to callers —
|
||||
``maintenance_drain.is_draining`` treats both as not draining — so the
|
||||
read never has to invent a record to answer the gate.
|
||||
"""
|
||||
with self._tx(immediate=False) as conn:
|
||||
row = conn.execute(
|
||||
"""
|
||||
SELECT * FROM maintenance_drain
|
||||
WHERE remote = ? AND org = ? AND repo = ?
|
||||
""",
|
||||
(str(remote or ""), str(org or ""), str(repo or "")),
|
||||
).fetchone()
|
||||
return self._maintenance_drain_row(row)
|
||||
|
||||
def set_maintenance_drain(
|
||||
self,
|
||||
*,
|
||||
remote: str,
|
||||
org: str,
|
||||
repo: str,
|
||||
state: str,
|
||||
reason: str = "",
|
||||
requested_by: str = "",
|
||||
requested_by_profile: str = "",
|
||||
session_id: str = "",
|
||||
) -> dict[str, Any]:
|
||||
"""Enter or exit maintenance drain for one repository scope (AC1).
|
||||
|
||||
The state transition is audited to ``events`` — entering and exiting
|
||||
are exactly the moments an operator has to be able to reconstruct
|
||||
later. Re-entering an already-draining scope is idempotent: it refreshes
|
||||
the reason/owner metadata, keeps the original ``entered_at``, and
|
||||
records no duplicate transition event.
|
||||
|
||||
Capability authorization happens above this layer (the drain tasks
|
||||
carry a non-``gitea.*`` permission in the task capability map); the DB
|
||||
records who asked and why, and never grants the right itself.
|
||||
"""
|
||||
state_norm = maintenance_drain.normalize_state(state)
|
||||
raw = {
|
||||
"reason": str(reason or ""),
|
||||
"requested_by": str(requested_by or ""),
|
||||
"requested_by_profile": str(requested_by_profile or ""),
|
||||
"session_id": str(session_id or ""),
|
||||
}
|
||||
clean = gitea_audit.redact(raw)
|
||||
remote_s, org_s, repo_s = str(remote or ""), str(org or ""), str(repo or "")
|
||||
now_s = _ts()
|
||||
|
||||
with self._tx() as conn:
|
||||
existing = conn.execute(
|
||||
"""
|
||||
SELECT * FROM maintenance_drain
|
||||
WHERE remote = ? AND org = ? AND repo = ?
|
||||
""",
|
||||
(remote_s, org_s, repo_s),
|
||||
).fetchone()
|
||||
|
||||
prior_state = (
|
||||
maintenance_drain.normalize_state(existing["state"])
|
||||
if existing is not None
|
||||
else maintenance_drain.STATE_INACTIVE
|
||||
)
|
||||
transitioned = prior_state != state_norm
|
||||
|
||||
prior_entered = (
|
||||
str(existing["entered_at"] or "") if existing is not None else ""
|
||||
)
|
||||
prior_exited = (
|
||||
str(existing["exited_at"] or "") if existing is not None else ""
|
||||
)
|
||||
if state_norm == maintenance_drain.STATE_DRAINING:
|
||||
# A re-entry keeps the original entry time (the drain never
|
||||
# stopped); a fresh entry stamps now and clears the old exit.
|
||||
entered_at = prior_entered if (not transitioned and prior_entered) else now_s
|
||||
exited_at = ""
|
||||
else:
|
||||
entered_at = prior_entered
|
||||
exited_at = now_s if (transitioned or not prior_exited) else prior_exited
|
||||
|
||||
if existing is None:
|
||||
drain_id = uuid.uuid4().hex
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO maintenance_drain(
|
||||
drain_id, remote, org, repo, state, reason,
|
||||
requested_by, requested_by_profile, session_id,
|
||||
entered_at, exited_at, drain_schema_version,
|
||||
created_at, updated_at
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
drain_id, remote_s, org_s, repo_s, state_norm,
|
||||
clean["reason"], clean["requested_by"],
|
||||
clean["requested_by_profile"], clean["session_id"],
|
||||
entered_at, exited_at,
|
||||
maintenance_drain.DRAIN_SCHEMA_VERSION, now_s, now_s,
|
||||
),
|
||||
)
|
||||
else:
|
||||
drain_id = str(existing["drain_id"])
|
||||
conn.execute(
|
||||
"""
|
||||
UPDATE maintenance_drain
|
||||
SET state = ?, reason = ?, requested_by = ?,
|
||||
requested_by_profile = ?, session_id = ?,
|
||||
entered_at = ?, exited_at = ?,
|
||||
drain_schema_version = ?, updated_at = ?
|
||||
WHERE drain_id = ?
|
||||
""",
|
||||
(
|
||||
state_norm, clean["reason"], clean["requested_by"],
|
||||
clean["requested_by_profile"], clean["session_id"],
|
||||
entered_at, exited_at,
|
||||
maintenance_drain.DRAIN_SCHEMA_VERSION, now_s, drain_id,
|
||||
),
|
||||
)
|
||||
|
||||
if transitioned:
|
||||
event_type = (
|
||||
"maintenance_drain_enter"
|
||||
if state_norm == maintenance_drain.STATE_DRAINING
|
||||
else "maintenance_drain_exit"
|
||||
)
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO events(work_item_id, event_type, message, created_at)
|
||||
VALUES (NULL, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
event_type,
|
||||
f"drain {drain_id} scope {remote_s}/{org_s}/{repo_s} "
|
||||
f"{prior_state} -> {state_norm} by "
|
||||
f"{clean['requested_by'] or '(unknown)'} "
|
||||
f"({clean['requested_by_profile'] or 'no profile'}); "
|
||||
f"reason: {clean['reason'] or '(none)'}",
|
||||
now_s,
|
||||
),
|
||||
)
|
||||
|
||||
row = conn.execute(
|
||||
"SELECT * FROM maintenance_drain WHERE drain_id = ?", (drain_id,)
|
||||
).fetchone()
|
||||
|
||||
record = self._maintenance_drain_row(row) or {}
|
||||
return {
|
||||
"record": record,
|
||||
"drain_id": drain_id,
|
||||
"state": state_norm,
|
||||
"prior_state": prior_state,
|
||||
"transitioned": transitioned,
|
||||
}
|
||||
|
||||
@@ -0,0 +1,167 @@
|
||||
# ADR: High-availability and rolling-restart architecture for Gitea MCP control plane
|
||||
|
||||
- **Status:** Proposed (Design ADR under [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668))
|
||||
- **Date:** 2026-07-25
|
||||
- **Tracking Issue:** [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668)
|
||||
- **Policy Version:** `mcp-ha-rolling-restart/v1`
|
||||
- **Related:**
|
||||
- Parent: [#655](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/655) — Governed MCP restart coordination and zero-disruption recovery
|
||||
- Governance Policy: [#656](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/656) / `docs/architecture/mcp-restart-governance.md`
|
||||
- Control-Plane DB Substrate: [#613](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/613) / `docs/architecture/control-plane-db-substrate.md`
|
||||
- Runtime Policy: [#615](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/615) / `docs/architecture/mcp-stable-control-runtime-policy-adr.md`
|
||||
- Product Vision: [#652](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/652) (Phase 5 Maturity)
|
||||
- Delivery Roadmap: [#653](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/653)
|
||||
|
||||
---
|
||||
|
||||
## 1. Context & Problem Statement
|
||||
|
||||
The Gitea MCP server operates as the authoritative **control plane** for managing issues, Pull Requests, code mutations, formal reviews, and workflow reconciliations. Under single-process governance ([#656](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/656)), process restarts are strictly controlled using pre-flight checks, drain phases, and operator approvals.
|
||||
|
||||
However, a single-instance control plane inherently presents fundamental constraints:
|
||||
|
||||
1. **Downtime during updates:** Even a perfectly executed single-process drain requires a window where incoming client requests must be paused or rejected while the server binary or python environment reloads.
|
||||
2. **Single point of failure:** Infrastructure issues, process crashes, or unhandled host-level terminations immediately disconnect active LLM sessions and leave transient workflows incomplete.
|
||||
3. **Multi-agent concurrency bottlenecks:** High volumes of concurrent multi-LLM tasks put all lock management, lease allocation, and Gitea API interactions through a single process event loop.
|
||||
|
||||
To achieve true zero-disruption operation and seamless rolling deployments without stopping active work, the system requires a high-availability (HA), multi-instance MCP architecture.
|
||||
|
||||
---
|
||||
|
||||
## 2. Architectural Principles & Non-Goals
|
||||
|
||||
### 2.1 Core Architectural Principles
|
||||
* **Gitea as Canonical Work SoT:** Gitea remains the ultimate System of Record (SoT) for issue states, pull requests, labels, and audit comments. The MCP control plane does not duplicate domain entities.
|
||||
* **Control-Plane DB as Multi-Instance State Substrate:** The control-plane SQLite/durable database ([#613](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/613)) acts as the single source of truth for workflow leases, session tokens, assignment records, and lock fences across all MCP nodes.
|
||||
* **Stateless Worker Nodes:** MCP role server processes (`gitea-author`, `gitea-reviewer`, `gitea-merger`, `gitea-reconciler`, `gitea-controller`) maintain no unique in-memory state; any node can handle any request given a valid session resume token.
|
||||
* **Fail-Closed Split-Brain Defense:** In any network partition or quorum loss scenario, nodes must fail closed rather than risk double-mutations or conflicting Gitea states.
|
||||
|
||||
### 2.2 Non-Goals
|
||||
* **Replacing Gitea:** We do not replace Gitea issue/PR tracking with an independent database.
|
||||
* **Immediate Multi-Node Cluster Execution in v1:** This ADR defines the target architecture and phased roadmap; immediate implementation occurs incrementally post-[#655] v1.
|
||||
|
||||
---
|
||||
|
||||
## 3. High-Availability & Rolling-Restart Architecture
|
||||
|
||||
### 3.1 Architecture Overview
|
||||
|
||||
```
|
||||
+----------------------------+
|
||||
| LLM Clients / IDE Sessions |
|
||||
+--------------+-------------+
|
||||
|
|
||||
v
|
||||
+----------------------------+
|
||||
| HA Proxy / Router |
|
||||
| (Health-based & Affinity) |
|
||||
+------+--------------+------+
|
||||
| |
|
||||
+--------------+ +--------------+
|
||||
v v
|
||||
+--------------------+ +--------------------+
|
||||
| MCP Instance Node A| | MCP Instance Node B|
|
||||
| (Version N) | | (Version N+1) |
|
||||
+---------+----------+ +---------+----------+
|
||||
| |
|
||||
+----------------------+----------------------+
|
||||
|
|
||||
v
|
||||
+----------------------------+
|
||||
| Control-Plane DB Substrate|
|
||||
| (Shared Lease & Locks) |
|
||||
+--------------+-------------+
|
||||
|
|
||||
v
|
||||
+----------------------------+
|
||||
| Gitea API |
|
||||
+----------------------------+
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### 3.2 Key System Components
|
||||
|
||||
#### A. Multiple MCP Instance Cohorts
|
||||
* The control plane runs across $N \ge 2$ redundant process nodes.
|
||||
* Dual-namespace deployment allows running the old version (Node A) alongside a updated version (Node B) during rolling upgrades.
|
||||
|
||||
#### B. Shared Durable Session Storage & Resume Tokens
|
||||
* Session context, preflight verification proofs, and capability resolution states are stored in the shared control-plane database.
|
||||
* Client requests carry an explicit `session_id` and `resume_token`. If an MCP instance restarts or a request routes to a different instance, the target node validates the token against the database without requiring full session re-initialization.
|
||||
|
||||
#### C. Shared Lease Authority & Fencing Counters
|
||||
* Workflow leases (`gitea_allocate_next_work`, `gitea_adopt_workflow_lease`) use monotonic fencing tokens (`lease_generation_id`).
|
||||
* When Node B acquires or renews a lease, it increments the generation counter. Any delayed or out-of-order write attempt from Node A using an older generation token is rejected by database constraints.
|
||||
|
||||
#### D. Leader Election & Coordinated Drain
|
||||
* Node clusters elect a primary coordinator node for administrative background tasks (such as stale lease cleanup or incident Watchdogs).
|
||||
* During a rolling deployment:
|
||||
1. Node B (new version) is launched and registers as healthy.
|
||||
2. Router directs new session creations to Node B.
|
||||
3. Node A enters `MAINTENANCE_DRAIN` status ([#659](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/659)), completing in-flight mutations while refusing new tasks.
|
||||
4. Once all active sessions migrate or complete, Node A shuts down cleanly.
|
||||
|
||||
#### E. Idempotent Mutations & Failover Safety
|
||||
* All state-changing tool executions (PR creation, review submission, merge operations, label changes) carry a deterministic `idempotency_key`.
|
||||
* If a network connection flaps or a node fails mid-mutation, the re-issued request with the same `idempotency_key` is recognized by the control-plane substrate, returning the existing recorded result without repeating side effects on Gitea.
|
||||
|
||||
#### F. Schema Version Compatibility
|
||||
* Database migrations follow non-breaking additive patterns.
|
||||
* During rolling upgrades where Node A (Version $N$) and Node B (Version $N+1$) run concurrently, both versions operate against the shared schema without structural conflicts.
|
||||
|
||||
---
|
||||
|
||||
## 4. Split-Brain & Failure Behavior
|
||||
|
||||
### 4.1 Split-Brain Risk Scenarios & Mitigation
|
||||
|
||||
| Scenario | Risk | Mitigation Strategy |
|
||||
|---|---|---|
|
||||
| **Network Partition between Nodes** | Both Node A and Node B attempt to process operations for the same issue/PR. | **Generation Fencing:** Lease renewal requires updating the DB generation counter. The node isolated from the DB fails closed immediately. |
|
||||
| **Stale Node Recovery** | Node A recovers after a long pause and executes a queued mutation. | **Lease Expiry & TTL Fencing:** Transactions verify that `expires_at > NOW()` within the atomic SQLite transaction boundaries. |
|
||||
| **Database Connection Loss** | Node loses access to shared control-plane DB substrate. | **Strict Fail-Closed:** The node immediately marks all task capabilities as `blocked` and rejects mutation tools until DB connectivity is re-established. |
|
||||
|
||||
---
|
||||
|
||||
## 5. Phased Implementation Milestones
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
M1[Milestone 1: Shared Control-Plane DB Schema & Resume Tokens] --> M2[Milestone 2: Idempotent Mutation Layer]
|
||||
M2 --> M3[Milestone 3: Health Routing & Standby Failover]
|
||||
M3 --> M4[Milestone 4: Active-Active Rolling Deployment & Auto-Drain]
|
||||
```
|
||||
|
||||
### Milestone 1: Shared Control-Plane DB Schema & Resume Tokens (Post-#655)
|
||||
* Extend [#613] Control-Plane DB schema to store multi-instance node heartbeat records and session resume tokens.
|
||||
* Enable session lookup across instances via `session_id`.
|
||||
|
||||
### Milestone 2: Idempotent Mutation Layer & Lease Fencing
|
||||
* Add mandatory `idempotency_key` tracking to all Gitea mutation tools.
|
||||
* Implement monotonic lease fencing counters in `gitea_allocate_next_work` and `gitea_adopt_workflow_lease`.
|
||||
|
||||
### Milestone 3: Health-Based Routing & Active-Passive Standby
|
||||
* Introduce lightweight proxy/router capable of checking node health endpoints.
|
||||
* Implement active-standby failover where standby node automatically assumes work if active node fails health checks.
|
||||
|
||||
### Milestone 4: Active-Active Horizontal Deployment & Rolling Upgrade Automation
|
||||
* Enable true active-active multi-instance execution.
|
||||
* Integrate automated zero-downtime rolling upgrades coordinated with `gitea_request_mcp_restart` maintenance drain.
|
||||
|
||||
---
|
||||
|
||||
## 6. Observability & Audit Requirements
|
||||
|
||||
High-availability control plane operations must expose clear telemetry and audit trails:
|
||||
|
||||
* **Node Registry Telemetry:** Active nodes, version numbers, uptime, and heartbeat timestamps reported via `gitea_get_runtime_context`.
|
||||
* **Lease Fencing Metrics:** Tracking lease acquire latency, fence rejection counts, and lease handoff durations.
|
||||
* **Failover & Re-route Audit Logs:** Durable logging of session migrations between nodes, drain initiation, and process retirement events.
|
||||
|
||||
---
|
||||
|
||||
## 7. Tradeoffs & Accepted Risks
|
||||
|
||||
* **Increased Architectural Complexity:** Moving from a single process to a multi-instance control plane requires robust DB locking, proxy routing, and migration governance.
|
||||
* **Database Dependency:** The control-plane database substrate becomes a critical shared dependency for multi-node deployments. High availability for the underlying SQLite file system / DB must be guaranteed.
|
||||
@@ -0,0 +1,83 @@
|
||||
# Incident #670: bare direct-to-master commit `2fa97c26` (retroactive audit)
|
||||
|
||||
Status: verified; disposition recommendation: **accept as-is, no revert** (final
|
||||
disposition owned by controller per issue #670).
|
||||
|
||||
## Summary
|
||||
|
||||
Commit `2fa97c26fbda555a1a83930ca5fdcea9d8e47b50`
|
||||
(`fix(mcp): load dotenv relative to project root`) landed on `prgs/master`
|
||||
as a single-parent commit with no PR wrapper and no review record, bypassing
|
||||
the sanctioned issue → branch → PR → review → merge workflow. It was
|
||||
discovered during the PR #654 post-merge audit. PR #654 itself merged
|
||||
cleanly via the Gitea API and did **not** introduce this commit.
|
||||
|
||||
## Verification evidence (acceptance criteria 1–3)
|
||||
|
||||
- **AC1 — present on `prgs/master`: yes.**
|
||||
`git merge-base --is-ancestor 2fa97c26fbda555a1a83930ca5fdcea9d8e47b50 prgs/master` → true.
|
||||
- **AC2 — no PR or review record: confirmed.**
|
||||
The commit is a single-parent, non-merge commit sitting directly on
|
||||
first-parent master between the #629 merge (`5ab5fe85`) and the #654
|
||||
merge (`ec903b0d`). A PR landing on master produces a merge commit (or a
|
||||
PR-linked head); neither exists here. The controller audit at issue-create
|
||||
time also found no PR wrapper and no review record for this SHA.
|
||||
- **AC3 — changed files and diff summary: confirmed.**
|
||||
`gitea_auth.py | 5 +++--` (+3/−2). Single parent
|
||||
`5ab5fe8583c07134d55dadf09381aecb67df246e`. The change moves
|
||||
`PROJECT_ROOT` derivation above `load_dotenv()` and loads
|
||||
`.env` relative to the project root instead of the process CWD.
|
||||
|
||||
## AC4 — why no immediate revert
|
||||
|
||||
- The dotenv fix is intentional and required for correct runtime behavior:
|
||||
without it, `load_dotenv()` resolves `.env` against the process working
|
||||
directory, which breaks MCP server launches whose CWD is not the project
|
||||
root.
|
||||
- The change is small (+3/−2), self-contained in `gitea_auth.py`, and has
|
||||
been running on master without incident since 2026-07-10.
|
||||
- Reverting would re-introduce a real bug to remove a provenance defect —
|
||||
the wrong trade. Provenance is repaired retroactively by this document,
|
||||
issue #670, and the hardening landed under #671.
|
||||
- If the controller later judges the change unsafe, a separate
|
||||
revert/repair issue is the sanctioned path (issue #670, recommended
|
||||
disposition option 4).
|
||||
|
||||
## AC5 — workflow-hardening linkage
|
||||
|
||||
Prevention already landed: **issue #671** (closed)
|
||||
*“Block direct pushes to stable branches from MCP workflow sessions”*,
|
||||
implemented by commit `5933d87647656643a67a50331c4c7b06ea751dad`
|
||||
(`feat(guard): block direct stable-branch pushes from MCP workflow sessions`).
|
||||
|
||||
Shipped guardrails include:
|
||||
|
||||
- `gitea_record_stable_branch_push_attempt` — classifies proposed commands
|
||||
for direct stable-branch push intent (`git push <remote> master`,
|
||||
refspecs, `HEAD:master`, `--force`, dry-run intent, `:master` delete),
|
||||
plus root/control-checkout local commits not carried by an issue branch,
|
||||
and writes a durable `stable_branch_contamination` marker.
|
||||
- `gitea_audit_stable_branch_contamination` — reconciler-only audit/clear
|
||||
path; a contaminated worker session cannot self-clear.
|
||||
- Review/merge/close/completion mutations fail closed while a
|
||||
contamination marker is active.
|
||||
|
||||
## AC6 — PR #654 was not the source
|
||||
|
||||
- `2fa97c26` is the **first parent** of the #654 merge commit
|
||||
`ec903b0d619e7a27d24aed272a890f4e5d381411`; it predates the #654 merge.
|
||||
- First-parent history `5ab5fe8..ec903b0`:
|
||||
`2fa97c2 fix(mcp): load dotenv relative to project root` followed by
|
||||
`ec903b0 Merge pull request 'feat: lifecycle role/hazard labels ... (#603)' (#654)`.
|
||||
- The #654 merger audit confirmed `ec903b0d` was a valid Gitea-API merge,
|
||||
the `git push prgs master` attempt during that run was a no-op, and the
|
||||
net change `2fa97c2..ec903b0` contained only the reviewed #603
|
||||
lifecycle-label files.
|
||||
- Conclusion: #654 merged reviewed content only; the unauthorized-path
|
||||
defect is solely the earlier bare commit `2fa97c26`.
|
||||
|
||||
## Explicit non-actions (unchanged by this audit)
|
||||
|
||||
- No revert of `2fa97c26`.
|
||||
- No force-push or history rewrite.
|
||||
- No master mutation from the audit session.
|
||||
@@ -0,0 +1,45 @@
|
||||
# MCP maintenance-drain mode (#659)
|
||||
|
||||
Graceful **maintenance drain** stops new work assignment and defers non-allowlisted
|
||||
mutations so sessions can finish critical handoffs and checkpoint before a
|
||||
restart. It is **not** a restart authorization: the drain *proof* and apply gate
|
||||
remain #661.
|
||||
|
||||
## State
|
||||
|
||||
Per repository scope (`remote`/`org`/`repo`) in the control-plane DB table
|
||||
`maintenance_drain` (schema v6):
|
||||
|
||||
| State | Meaning |
|
||||
|-------|---------|
|
||||
| `inactive` | Normal operation (also: no row) |
|
||||
| `draining` | Assignment stopped; non-allowlisted mutations deferred |
|
||||
|
||||
Enter/exit transitions are audited as `maintenance_drain_enter` /
|
||||
`maintenance_drain_exit` events.
|
||||
|
||||
## Tools
|
||||
|
||||
| Tool | Permission | Effect |
|
||||
|------|------------|--------|
|
||||
| `gitea_maintenance_drain_status` | `gitea.read` | Observe drain (every session) |
|
||||
| `gitea_enter_maintenance_drain` | `runtime.maintenance_drain` | Enter drain (capability-gated) |
|
||||
| `gitea_exit_maintenance_drain` | `runtime.maintenance_drain` | Exit drain |
|
||||
|
||||
`runtime.maintenance_drain` is intentionally **not** a `gitea.*` op, so ordinary
|
||||
author profiles cannot enter drain by accident.
|
||||
|
||||
## Enforcement
|
||||
|
||||
1. **Allocator** (`allocate_next_work`): while draining, returns `outcome=wait`
|
||||
with `reason_code=maintenance_drain_assignment_stopped` for dry-run and apply.
|
||||
2. **Mutation preflight** (`verify_preflight_purity`): non-allowlisted mutation
|
||||
tasks raise `MaintenanceDrainError` with a typed next action.
|
||||
3. **Allowlist** (safety only): heartbeats, lease release/abandon, session
|
||||
checkpoints, enter/exit drain. Reads always work.
|
||||
|
||||
## Restart relationship
|
||||
|
||||
Drain mode prepares the blast radius. Restart apply still requires a clean
|
||||
`DrainProof` (#661) or authorized break-glass. Status payloads never claim
|
||||
restart permission.
|
||||
@@ -0,0 +1,94 @@
|
||||
# MCP scoped recovery playbook (#669)
|
||||
|
||||
**Parent:** [#655](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/655)
|
||||
**Vision / roadmap:** [#652](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/652) · [#653](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/653)
|
||||
**Class matrix:** [#663](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/663) · `docs/mcp-restart-classes.md`
|
||||
**Coordinator:** [#658](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/658) · `restart_coordinator.py`
|
||||
**Audit lineage:** [#665](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/665)
|
||||
|
||||
## Decision
|
||||
|
||||
Full-server MCP reset is a **last resort**. Prefer the narrowest recovery that
|
||||
can clear the symptom. The coordinator **refuses** `rolling_mcp_restart`,
|
||||
`full_mcp_restart`, and `host_restart` unless:
|
||||
|
||||
1. The inventory carries a prior **attempt log** of at least one *insufficient*
|
||||
narrower recovery, **or**
|
||||
2. **Break-glass** is authorized
|
||||
(`request_break_glass` + `GITEA_BREAKGLASS_RESTART_AUTHORIZATION`).
|
||||
|
||||
Break-glass still never bypasses the #663 class matrix (role/permission).
|
||||
|
||||
## Ladder (narrow → broad)
|
||||
|
||||
| Rank | Action | Self-service | Implementation / delegation |
|
||||
|---:|---|---|---|
|
||||
| 0 | `client_reconnect` | yes | Host auto-reconnect / client reconnect · [#584](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/584) · `docs/mcp-namespace-eof-recovery.md` |
|
||||
| 1 | `capability_refresh` | yes | `gitea_resolve_task_capability` + `gitea_whoami` · [#610](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/610) · [#685](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/685) |
|
||||
| 2 | `session_reconnect` | yes | Runtime rebind + explicit `worktree_path` · [#543](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/543) · [#618](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/618) |
|
||||
| 3 | `configuration_reload` | no | Class `configuration_reload` · console reload · [#642](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/642) |
|
||||
| 4 | `lease_recovery` | no | Lock/lease recovery paths · [#702](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/702) · [#753](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/753) · [#790](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/790) |
|
||||
| 5 | `worker_restart` | no | Class `worker_restart` · #663 |
|
||||
| 6 | `role_runtime_restart` | no | Class `role_runtime_restart` · console restart · #642/#663 |
|
||||
| 7 | `connector_restart` | no | Class `connector_restart` · #663 |
|
||||
| 8 | `rolling_mcp_restart` | no | Class `rolling_mcp_restart` · design [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668) · **attempt log required** |
|
||||
| 9 | `full_mcp_restart` | no | Class `full_mcp_restart` · **attempt log required** |
|
||||
| 10 | `host_restart` | no | Class `host_restart` · **attempt log required** |
|
||||
|
||||
Machine-readable source of truth: `recovery_playbook.RECOVERY_LADDER` and
|
||||
`recovery_playbook.ladder_document()`.
|
||||
|
||||
## Attempt log shape
|
||||
|
||||
Each prior attempt is a mapping:
|
||||
|
||||
```json
|
||||
{
|
||||
"action": "client_reconnect",
|
||||
"outcome": "insufficient",
|
||||
"reason": "transport still closed after IDE reconnect",
|
||||
"actor": "prgs-controller-12345",
|
||||
"recorded_at": "2026-07-25T21:00:00+00:00"
|
||||
}
|
||||
```
|
||||
|
||||
Outcomes that count toward escalation: `failed`, `insufficient`, `denied`,
|
||||
`unresolved`, `timeout`, `error`.
|
||||
|
||||
Pass attempts into the coordinator via inventory
|
||||
`prior_recovery_attempts` or the MCP tool argument
|
||||
`prior_recovery_attempts_json` on `gitea_request_mcp_restart`.
|
||||
|
||||
Helper: `recovery_playbook.build_attempt_record(...)`.
|
||||
|
||||
## Symptom → first rung
|
||||
|
||||
`recovery_playbook.recommend_actions(symptoms=[...])` maps symptoms such as
|
||||
`transport_eof`, `stale_capability`, `stale_lease`, `daemon_corrupt` to the
|
||||
narrowest recommended action, then walks the ladder. Soft recommendations
|
||||
never replace the hard gate on broad restarts.
|
||||
|
||||
## Enforcement points
|
||||
|
||||
1. **`recovery_playbook.assess_escalation`** — pure gate.
|
||||
2. **`restart_coordinator.evaluate_restart_impact`** — when `restart_class` is
|
||||
set (policy-enforced path), broad classes require the gate; report fields
|
||||
`attempt_log_satisfied`, `playbook_escalation`, `break_glass`.
|
||||
3. **`gitea_request_mcp_restart`** — accepts attempt JSON and env-authorized
|
||||
break-glass; never restarts a process.
|
||||
|
||||
## Metrics
|
||||
|
||||
`recovery_playbook.recovery_metrics(attempts)` reports the fraction of
|
||||
successful recoveries that avoided full/host restart
|
||||
(`fraction_avoided_full_restart`).
|
||||
|
||||
## Non-goals
|
||||
|
||||
* HA multi-instance execution (#668 design only here).
|
||||
* Normalizing `pkill` (#630 contamination stays forbidden).
|
||||
* Silent mutation of leases or processes from the playbook itself.
|
||||
|
||||
## Manual process kills
|
||||
|
||||
Remain forbidden and contaminating (#630). The playbook never recommends them.
|
||||
@@ -0,0 +1,53 @@
|
||||
# MCP restart classes and blast-radius permissions (#663)
|
||||
|
||||
This is the machine-enforced class matrix used by
|
||||
`restart_coordinator.RESTART_CLASS_POLICIES`. It implements the narrower-first
|
||||
recovery ladder from #655 and the authorization policy from #656, using the
|
||||
path inventory from #657 and the impact coordinator from #658. Product and
|
||||
delivery lineage: vision #652 and roadmap #653.
|
||||
|
||||
Unknown class names are denied. The coordinator requires both the class
|
||||
permission and an eligible request role. Approval gates are additional: a
|
||||
caller cannot turn a request permission into execution authority.
|
||||
|
||||
| Restart class | Required permission | Expected blast radius | Drain requirement | Approval requirement | Audit requirement | Recovery behavior |
|
||||
|---|---|---|---|---|---|---|
|
||||
| `client_reconnect` | `mcp.reconnect.client` | none | none | self service | class, actor, client namespace, reason, outcome | Reconnect only the caller's client transport. No daemon or peer work changes. |
|
||||
| `session_reconnect` | `mcp.reconnect.session` | low | requesting-session safe point | self service | class, actor, session, reason, outcome | Rebind identity, capability, and workspace state for one session. |
|
||||
| `worker_restart` | `mcp.restart.worker.request` | low | target worker | controller approval + automated gates | class, actor, worker, approval, scoped drain, outcome | Restart one worker after its own leases and mutations drain. |
|
||||
| `role_runtime_restart` | `mcp.restart.role_runtime.request` | medium | target role runtime | controller approval + automated gates | class, actor, role namespace, approval, scoped drain, outcome | Restart and re-probe one role runtime; unrelated roles remain available. |
|
||||
| `connector_restart` | `mcp.restart.connector.request` | medium | target connector | controller approval + automated gates | class, actor, connector, approval, scoped drain, outcome | Restart one connector while unrelated runtimes remain available. |
|
||||
| `configuration_reload` | `mcp.reload.configuration.request` | low | mutation quiesce | controller approval + automated gates | class, actor, configuration revision, approval, outcome | Gracefully reload configuration without replacing the daemon. |
|
||||
| `rolling_mcp_restart` | `mcp.restart.rolling.request` | medium | one instance at a time | controller approval + automated gates | class, actor, instance order, approval, per-instance drains, outcome | Drain, restart, verify, and restore each instance before advancing. |
|
||||
| `full_mcp_restart` | `mcp.restart.full.request` | high | all sessions and mutations | controller approval + automated gates | class, actor, full impact, approval, full drain proof, outcome | Replace the complete MCP runtime only after a verified full drain. |
|
||||
| `host_restart` | `mcp.restart.host.request` | high | all host work | controller approval + infrastructure operator | class, actor, host/change or incident id, approval, full drain proof, outcome | Hand off to infrastructure ownership and reconcile every runtime afterward. |
|
||||
|
||||
## Drain boundary
|
||||
|
||||
Only `full_mcp_restart` and `host_restart` set `full_drain_required=true`.
|
||||
Reconnects and configuration reloads do not disrupt peer sessions. Worker,
|
||||
role-runtime, and connector restarts evaluate only their explicitly named
|
||||
target. Rolling restart drains one instance at a time. Missing required target
|
||||
scope denies the request rather than silently widening it to a full restart.
|
||||
|
||||
## Permission and approval boundary
|
||||
|
||||
Author, reviewer, merger, and reconciler roles may self-request reconnects and
|
||||
request scoped worker/role/connector/reload recovery. They cannot request
|
||||
rolling, full, or host restart classes. Controller/operator/admin roles may
|
||||
request the broader classes, while execution remains operator/admin-owned.
|
||||
Controller approval is independently required for every class above a session
|
||||
reconnect. Host restart additionally requires infrastructure-operator proof.
|
||||
|
||||
The MCP request tool derives class permissions from its authenticated runtime
|
||||
role. It does not accept caller-supplied permissions. Controller and operator
|
||||
authorization are read from the already-running daemon environment, never
|
||||
from a request argument.
|
||||
|
||||
## Audit and failure behavior
|
||||
|
||||
Every impact audit and every console restart/reload audit includes a
|
||||
`restart_class` field. The impact audit also includes the exact
|
||||
`required_permission`. Unknown classes, missing permissions, ineligible roles,
|
||||
missing approval, missing scoped targets, and incomplete inventory all deny
|
||||
fail closed. Manual process kills remain forbidden and contaminating (#630).
|
||||
@@ -6,11 +6,23 @@ console (#642 / #652) can see the blast radius *before* concurrent LLM work is
|
||||
disrupted. Uncoordinated restarts destroy in-flight author/reviewer/merger work
|
||||
and give operators no way to see what they are about to break.
|
||||
|
||||
This lands the coordinator + impact DTO + a dry-run MCP tool. It is the single
|
||||
This lands the coordinator + impact DTO + the MCP tool. It is the single
|
||||
sanctioned entry point for restart evaluation post-#657 (which inventoried the
|
||||
restart/reload/kill paths). The **mutative apply** path — actually performing a
|
||||
restart — is a later child gated by a drain proof and is explicitly out of
|
||||
scope here.
|
||||
restart/reload/kill paths).
|
||||
|
||||
The **drain-proof hard gate now executes inside this tool** (#661, via PR #882):
|
||||
an apply request (`dry_run=False`) is evaluated against a drain proof here and
|
||||
denied when that proof is missing, expired, unclean, tampered with, or stale.
|
||||
It is no longer a separate child operation. What remains a later child is only
|
||||
the **execution** step — actually stopping and restoring a process. This tool
|
||||
still never restarts anything: `apply_supported` is always `false` and
|
||||
`restart_performed` is always `false`.
|
||||
|
||||
The coordinator now routes every request through the restart-class policy
|
||||
matrix defined for #663. See
|
||||
[`mcp-restart-classes.md`](./mcp-restart-classes.md) for permissions, expected
|
||||
blast radius, scoped drain and approval requirements, audit fields, and
|
||||
recovery behavior for all nine classes.
|
||||
|
||||
## Components
|
||||
|
||||
@@ -19,7 +31,8 @@ scope here.
|
||||
| `restart_coordinator.evaluate_restart_impact` | `restart_coordinator.py` | Pure classification: inventory → impact report DTO. No I/O, no restart. |
|
||||
| `RestartImpactReport` / `SessionImpact` / `LeaseImpact` | `restart_coordinator.py` | Console-facing DTO (`.as_dict()` is JSON-serializable). |
|
||||
| `ControlPlaneDB.list_sessions` | `control_plane_db.py` | Read-only session inventory (the process-level unit a restart kills). |
|
||||
| `gitea_request_mcp_restart` | `gitea_mcp_server.py` | MCP tool: gathers inventory from the #613 DB, calls the coordinator, returns the report. Dry-run only. |
|
||||
| `gitea_request_mcp_restart` | `gitea_mcp_server.py` | MCP tool: gathers inventory from the #613 DB, calls the coordinator, returns the report, and on `dry_run=False` runs the #661 drain-proof hard gate. Never restarts a process. |
|
||||
| `drain_proof.gate_apply_restart` | `drain_proof.py` | The #661 hard gate: verifies a drain proof against the current impact fingerprint, or records an authorized break-glass bypass. |
|
||||
|
||||
## Dimensions evaluated
|
||||
|
||||
@@ -77,17 +90,71 @@ authorization is present.
|
||||
```text
|
||||
gitea_request_mcp_restart(remote, host, org, repo,
|
||||
dry_run=True, request_override=False,
|
||||
session_id=None, limit=200)
|
||||
session_id=None, limit=200,
|
||||
restart_class="full_mcp_restart",
|
||||
target_session_id=None, target_role=None,
|
||||
target_connector=None,
|
||||
drain_proof_json=None,
|
||||
request_break_glass=False,
|
||||
prior_recovery_attempts_json=None)
|
||||
```
|
||||
|
||||
Read-only, dry-run, and it **never restarts anything**. `apply_supported` is
|
||||
always `false`; passing `dry_run=False` performs no restart and reports that
|
||||
apply is gated by a drain proof (a separate child).
|
||||
It **never restarts anything**: `apply_supported` is always `false` and
|
||||
`restart_performed` is always `false`.
|
||||
|
||||
`prior_recovery_attempts_json` (#669) is an optional JSON array of prior
|
||||
narrow recovery attempts. Rolling / full / host classes require at least one
|
||||
*insufficient* narrower attempt (or authorized break-glass). See
|
||||
`docs/mcp-recovery-playbook.md`.
|
||||
|
||||
### Dry-run versus apply
|
||||
|
||||
| Call | Behavior |
|
||||
|------|----------|
|
||||
| `dry_run=True` (default) | Read-only impact preview. No drain proof is required or consulted. |
|
||||
| `dry_run=False` | The #661 drain-proof hard gate runs **in this tool**. The outcome is reported under `apply_gate` / `apply_authorized`; a denial also returns a durable `incident` descriptor. Still no restart. |
|
||||
|
||||
### Authorization ordering
|
||||
|
||||
An apply requires **both** authorizations, and they are independent:
|
||||
|
||||
1. **Restart-class authorization** (#663 / #669) — the requester's role and
|
||||
permissions must allow the requested class, the class's approval requirement
|
||||
must be satisfied, any target-scoped class must name its target, and broad
|
||||
classes must satisfy the recovery-playbook attempt-log gate. Failing any of
|
||||
these makes `allow_restart` `false`.
|
||||
2. **Drain-proof gate** (#661) — a valid, unexpired, clean proof bound to the
|
||||
current impact fingerprint, or an authorized break-glass.
|
||||
|
||||
`apply_authorized` is the conjunction: `gate.allow and allow_restart`. A clean
|
||||
drain proof therefore cannot override a class or requester-role denial, and a
|
||||
denied class never reports an authorized apply. `apply_gate` carries
|
||||
`drain_gate_allow` and `restart_class_authorized` so a denial is attributable to
|
||||
the authorization that produced it.
|
||||
|
||||
### Break-glass
|
||||
|
||||
Break-glass bypasses the **drain proof only** — never the restart-class matrix
|
||||
(role/permission). Separately, authorized break-glass also satisfies the #669
|
||||
attempt-log requirement for broad restarts (rolling/full/host), because that
|
||||
gate is not a class-matrix permission check.
|
||||
It is honoured solely when `request_break_glass` is set *and* the environment
|
||||
carries `GITEA_BREAKGLASS_RESTART_AUTHORIZATION`; like operator override, the
|
||||
tool argument expresses caller intent and cannot be self-asserted by a worker
|
||||
session. `break_glass_requested` and `break_glass_authorized` are both reported,
|
||||
so a bypass is never silent.
|
||||
|
||||
### Fail closed on apply
|
||||
|
||||
A missing, malformed, expired, unclean, tampered, or fingerprint-stale drain
|
||||
proof denies the apply and returns an `incident` descriptor. An unknown restart
|
||||
class denies before any of this. Ambiguity always denies.
|
||||
|
||||
## Audit
|
||||
|
||||
Every evaluation carries an `audit_record` (event, coordinator version, verdict,
|
||||
allow decision, blast radius, counts, timestamp) so restart decisions are
|
||||
restart class, required permission, allow decision, blast radius, counts,
|
||||
timestamp) so restart decisions are
|
||||
auditable. No secrets flow through the coordinator — session ids, pids, and
|
||||
profiles are operational metadata only.
|
||||
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
# Sanctioned Recovery Playbooks & Controls (Phase 2 #644)
|
||||
|
||||
## Overview
|
||||
|
||||
Stale runtimes, worktree binding mismatches, and un-reconciled merged branches previously required expert manual shell recovery. Manual process kills (`pkill -f mcp_server.py`) are strictly forbidden and classified as runtime contamination ([#630](sanctioned-restart-controls.md)).
|
||||
|
||||
Phase 2 introduces **sanctioned recovery playbooks and controls** into the Web Console:
|
||||
- **Diagnose**: Surface stale runtimes, worktree binding errors, contamination markers, and worktree anomalies via health & inventory APIs.
|
||||
- **Preview**: Render mutation ledgers and exact confirmation phrases for recovery playbooks.
|
||||
- **Confirm & Apply**: Execute sanctioned recovery actions through gated, audited paths.
|
||||
- **Verify**: Revalidate control-plane state post-recovery before claiming clean status.
|
||||
|
||||
---
|
||||
|
||||
## Recovery Playbook Taxonomy
|
||||
|
||||
| Playbook ID | Action ID | Minimum Role | Target / Scope | Description |
|
||||
|---|---|---|---|---|
|
||||
| `clear_stale_binding` | `system.clear_stale_binding` | Operator | Active worktree binding | Clear provably missing or superseded `GITEA_ACTIVE_WORKTREE` binding ([#702](../stale_binding_recovery.py)). |
|
||||
| `rebind_session_worktree` | `system.rebind_session_worktree` | Operator | Session worktree | Rebind or synchronize session worktree to verified lease worktree ([#864](../dirty_same_claimant_session_rebind.py)). |
|
||||
| `reconcile_cleanups` | `system.reconcile_cleanups` | Controller | Worktree hygiene | Execute reconciler cleanup preview and apply for merged/superseded PR branches. |
|
||||
| `sanctioned_restart` | `system.restart_namespace` | Admin | MCP Namespace | Restart MCP daemon gracefully via host supervisor ([#642](sanctioned-restart-controls.md)). |
|
||||
|
||||
---
|
||||
|
||||
## Wizard Workflow (Diagnose → Preview → Confirm → Verify)
|
||||
|
||||
### 1. Diagnose (`GET /api/v1/system/recovery/diagnose`)
|
||||
Runs control-plane diagnostics:
|
||||
- **Stale Runtime**: Mismatch between running daemon HEAD, local checkout HEAD, and remote-tracking HEAD.
|
||||
- **Worktree Binding**: Missing path (`provably_stale_missing_path`), unverified inherited binding (`unverified_inherited`), or superseded binding (`superseded_by_session_lease`).
|
||||
- **Contamination**: Checks for live contamination markers from unmanaged process kills.
|
||||
- **Worktree Anomalies**: Scans `branches/` directory for un-reconciled cleanups or missing preserved worktrees.
|
||||
|
||||
Returns `RecoveryDiagnosis` with eligible playbooks.
|
||||
|
||||
### 2. Preview (`POST /api/v1/system/recovery/preview`)
|
||||
Takes `playbook_id` and optional `target`/`params`.
|
||||
Returns:
|
||||
- **Mutation Ledger**: Step-by-step sequence of actions.
|
||||
- **Confirmation Phrase**: Exact phrase required to authorize execution (e.g., `confirm clear_stale_binding`).
|
||||
- **Authorization Decision**: RBAC check against the operator's principal.
|
||||
|
||||
### 3. Apply (`POST /api/v1/system/recovery/apply`)
|
||||
Requires `playbook_id` and matching `confirmation` phrase. Gates run in this order, and each fails closed before anything is mutated:
|
||||
|
||||
1. **RBAC and execution phase** (`console_authz.authorize(..., for_execution=True)`). The phase branch only applies when `for_execution` is set. While `ACTIVE_PHASE` is `1`, every phase-2 recovery action is refused with `phase_not_active`, so no recovery playbook writes yet. Preview reports the same decision under `execution_authorization` / `execution_blocked_reason`.
|
||||
2. **Confirmation phrase** (`confirmation_matches`).
|
||||
3. **Contamination rules** ([#630](sanctioned-restart-controls.md)): the live marker is read from the session inventory and assessed under the gated task key `console_recovery_apply`. A contaminated runtime must be cleared through the reconciler cleanup playbook, which is the one playbook exempted from this gate because it is the designated remedy. The marker is also forwarded to `sanctioned_restart.execute_restart`, so a restart cannot launder a contaminated runtime.
|
||||
|
||||
Apply then executes the sanctioned recovery logic against the **live** process environment — not a copy — and records an audit entry in `console_audit`. A playbook that leaves the binding unchanged reports `performed: false`; `binding_before`, `binding_after`, and `binding_changed` are returned so a no-op cannot read as success.
|
||||
|
||||
Apply does **not** enforce master parity. Parity is reported by Diagnose ([#610](../master_parity_gate.py)) as evidence for the operator; it is not a precondition of this endpoint.
|
||||
|
||||
### 4. Verify (`POST /api/v1/system/recovery/verify`)
|
||||
Re-evaluates control-plane diagnostics post-recovery and **reports** `clean`, `stale_runtime_clean`, `binding_clean`, `binding_classification`, and `contamination_clean`. It reports; it does not assert or block. State is read fresh rather than from the mapping a mutation just wrote. An `unverified_inherited` binding is reported as not clean, because unproven is not clean.
|
||||
|
||||
---
|
||||
|
||||
## Safety & Governance Principles
|
||||
|
||||
1. **No Manual `pkill`**: Direct process killing remains forbidden and is recorded as contamination.
|
||||
2. **Auditability**: Every recovery preview and execution is logged in the console audit trail.
|
||||
3. **Master Parity & Dual Control**: High-privilege recovery actions require controller/admin roles and explicit confirmation phrases.
|
||||
@@ -94,6 +94,10 @@ already define, and a regression test asserts each mapping matches.
|
||||
| `record_analytics_usage` | operator | gated_write | `runtime.record_analytics_usage` | Yes | No | No | 2 |
|
||||
| `system.reload_namespace` | controller | privileged | `runtime.reload_namespace` | Yes | No | No | 2 |
|
||||
| `system.restart_namespace` | admin | destructive | `runtime.restart_namespace` | Yes | **Yes** | **Yes** | 2 |
|
||||
| `system.clear_stale_binding` | operator | gated_write | `gitea.read` | Yes | No | No | 2 |
|
||||
| `system.rebind_session_worktree` | operator | gated_write | `gitea.read` | Yes | No | No | 2 |
|
||||
| `system.reconcile_cleanups` | controller | privileged | `gitea.pr.close` | Yes | No | No | 2 |
|
||||
| `initiate_workflow` | operator | gated_write | `gitea.read` | Yes | No | No | 2 |
|
||||
|
||||
**Dual control** means the acting principal may not be the sole authority: a
|
||||
second distinct principal must confirm. **Break-glass** means the action is
|
||||
@@ -112,6 +116,12 @@ by the console — both hand off to a host supervisor, and neither exposes a raw
|
||||
process kill. See
|
||||
[`sanctioned-restart-controls.md`](sanctioned-restart-controls.md) (#642).
|
||||
|
||||
`initiate_workflow` (#643) is operator-class because its outcome is a *claim*,
|
||||
not a Gitea verdict. Requesting reviewer or merger work reserves that work
|
||||
through the allocator; it does not grant the right to approve or merge, which
|
||||
stays with the MCP role profile and its own capability gates. See
|
||||
[`webui-requests.md`](webui-requests.md).
|
||||
|
||||
### Authorization decision
|
||||
|
||||
`authorize(action_id, principal, for_execution=False)` returns a decision
|
||||
@@ -126,9 +136,24 @@ record and **denies by default**. The deny reasons are closed and enumerated:
|
||||
| `phase_not_active` | Execution requested for an action whose phase is not open. |
|
||||
| `allowed_preview_only` | Authorized — preview only, execution still disabled. |
|
||||
|
||||
There is no implicit allow branch. Even the allow result reports
|
||||
`execution_enabled: false` while the console is in Phase 1, so no caller can
|
||||
read an allow as permission to mutate.
|
||||
There is no implicit allow branch.
|
||||
|
||||
`execution_enabled` on the decision reports whether the action has a live
|
||||
execution path at all, and is computed by `execution_wired(action)`. There are
|
||||
exactly two ways to be wired:
|
||||
|
||||
1. the action's `phase` is at or below `ACTIVE_PHASE`; or
|
||||
2. the action declares an `execution_env_flag` **and** that variable is set.
|
||||
|
||||
Every action that declares no flag therefore reports `execution_enabled: false`
|
||||
while the console is in Phase 1, so no caller can read an allow as permission
|
||||
to mutate. The per-action flag exists because raising `ACTIVE_PHASE` would
|
||||
enable execution for every action of that phase at once, including ones whose
|
||||
execution path is not implemented. One implemented action goes live on its own
|
||||
flag instead of dragging its unimplemented phase-mates with it.
|
||||
|
||||
`initiate_workflow` is the only action that currently declares a flag
|
||||
(`WEBUI_REQUESTS_EXECUTION`), and it stays denied until an operator sets it.
|
||||
|
||||
## Secret redaction
|
||||
|
||||
@@ -235,13 +260,22 @@ second one. The integration points are already wired and observable:
|
||||
instead of adding a parallel check.
|
||||
- **`GET /api/console/security-model`** publishes the RBAC matrix, redaction
|
||||
policy, and audit policy as JSON for operators and tests.
|
||||
- **`POST /api/v1/requests/preview` and `.../apply`** (#643) are the first
|
||||
actions to use this model for a real execution path. Preview always returns a
|
||||
decision and an audited `previewed` record; apply requires `confirm=true`,
|
||||
emits `succeeded` or `denied`, and reserves work only through the allocator.
|
||||
See [`webui-requests.md`](webui-requests.md).
|
||||
|
||||
To open Phase 2, a child issue must: raise `ACTIVE_PHASE`, implement the
|
||||
confirmation and dual-control flow the matrix already declares, emit a
|
||||
`succeeded` or `failed` record alongside the `gitea_audit` mutation record, and
|
||||
keep `viewer` unable to reach any of it. Turning on execution without the
|
||||
confirmation flow contradicts a declared requirement and is a review failure,
|
||||
not a shortcut.
|
||||
A Phase 2 action must: use `execution_wired` rather than a private enable flag,
|
||||
implement the confirmation and dual-control flow the matrix already declares,
|
||||
emit a `succeeded` or `failed` record alongside the `gitea_audit` mutation
|
||||
record, and keep `viewer` unable to reach any of it. Turning on execution
|
||||
without the confirmation flow contradicts a declared requirement and is a
|
||||
review failure, not a shortcut.
|
||||
|
||||
Raising `ACTIVE_PHASE` remains the way to open a whole phase at once, and is
|
||||
deliberately *not* what #643 did: an action-scoped opt-in cannot enable an
|
||||
action whose execution path nobody wrote.
|
||||
|
||||
## Local-dev mode
|
||||
|
||||
@@ -294,6 +328,7 @@ Until Phase 2 wires it, probe protection rests on network placement alone, as
|
||||
| `WEBUI_ROLE_MAP` | unset | JSON subject → role map |
|
||||
| `WEBUI_REQUIRE_PROBE_AUTH` | unset | Require auth for non-public probes |
|
||||
| `WEBUI_CONSOLE_AUDIT_LOG` | unset | Append-only audit sink path |
|
||||
| `WEBUI_REQUESTS_EXECUTION` | unset | Opt in to `initiate_workflow` execution (#643) |
|
||||
|
||||
All are read server-side only. None is ever rendered into a page or returned by
|
||||
an API.
|
||||
|
||||
+138
-6
@@ -57,6 +57,8 @@ status, onboarding checklist state, and the fail-closed error payloads (#635).
|
||||
| `/system-health` | System-health dashboard — readiness, version/uptime, dependencies, MCP namespaces, stale-runtime parity (#639) |
|
||||
| `/queue` | Live PR and issue queue dashboard (#429) |
|
||||
| `/api/queue` | JSON queue export with pagination metadata |
|
||||
| `/traffic` | Workflow traffic-control view — runnable, leased, blocked, needs-controller, terminal-complete (#640) |
|
||||
| `/api/traffic` | JSON traffic-control export with state classifications and next safe role actions |
|
||||
| `/projects` | Project registry list with status and onboarding progress (#427, #635) |
|
||||
| `/projects/{id}` | Project detail + onboarding checklist |
|
||||
| `/api/v1/projects` | Versioned JSON registry export (#635) |
|
||||
@@ -75,7 +77,11 @@ status, onboarding checklist state, and the fail-closed error payloads (#635).
|
||||
| `/api/actions/{id}/preview` | Mutation ledger preview (GET, read-only) |
|
||||
| `/leases` | Lease and collision visibility (#433) |
|
||||
| `/api/leases` | JSON lease/collision export |
|
||||
| `/sessions` | Phase 1 shell stub — session inventory (backed by #636) |
|
||||
| `/sessions` | Runtime and session view (#641) — health + inventory sessions/namespaces/worktrees |
|
||||
| `/api/sessions` | JSON export for the runtime/session view |
|
||||
| `/api/v1/sessions` | Versioned alias of `/api/sessions` |
|
||||
| `/gitea` | Gitea issue↔PR linkage console (#645) — both directions, with the evidence for each edge |
|
||||
| `/api/v1/gitea/linkage` | JSON linkage export; `502` when the read could not be answered |
|
||||
| `/inventory` | Phase 1 shell stub — unified inventory (backed by #636) |
|
||||
| `/timeline` | Phase 1 shell stub — workflow event timeline |
|
||||
| `/policy` | Phase 1 shell stub — capability/role policy placeholder |
|
||||
@@ -85,6 +91,35 @@ Most routes are GET-only. POST/PUT/PATCH/DELETE return `405` with
|
||||
`read-only-mvp`, except `/audit` and `/api/audit` which accept POST for
|
||||
local validator preview only (no Gitea mutations, no server-side storage).
|
||||
|
||||
### Traffic-control state vocabulary (#640)
|
||||
|
||||
The traffic view classifies each open issue/PR into exactly one bucket:
|
||||
|
||||
| Bucket | Meaning | Operator implication |
|
||||
|--------|---------|----------------------|
|
||||
| **runnable** | No active lease, no block reason, safe for its expected role | Next role may start work |
|
||||
| **leased** | Active author claim or reviewer PR lease | Do not stomp; wait or adopt via role tools |
|
||||
| **blocked** | Dependency, missing head pin, conflict, or unmet dependency | Author remediation first |
|
||||
| **needs_controller** | Contaminated, controller-only diagnosis, or `status:blocked` | Controller only |
|
||||
| **terminal_complete** | Reconciler / terminal-lock territory | Reconciler cleanup path |
|
||||
|
||||
`status:blocked` items route to **needs_controller**, not **blocked**:
|
||||
`expected_role_for_candidate` sends them to the controller, and the blocker
|
||||
reason renders in either bucket.
|
||||
|
||||
**Live path contracts (do not invent):**
|
||||
|
||||
- PR head pins come from `QueueItem.signals["head_sha"]` (full SHA). Display
|
||||
`extra["head_sha"]` is truncated and must never be used for routing.
|
||||
- Reviewer leases are keyed as `(pr, pr_number)` only — never via a linked
|
||||
`issue_number` on the same lease marker.
|
||||
- Issue claims come from `claim_inventory["entries"]`
|
||||
(`issue_claim_heartbeat.build_claim_inventory`). There is no `active_claims`
|
||||
key.
|
||||
- Queue display badges are only: `blocked`, `claimed`, `duplicate`, `stale`,
|
||||
`in-review`, `open`. Review verdicts (`request-changes`, `approved`) are
|
||||
**not** queue badges; traffic does not invent them from the queue loader.
|
||||
|
||||
## System health API (#634)
|
||||
|
||||
`GET /api/v1/system/health` is the structured, read-only health surface for
|
||||
@@ -253,11 +288,108 @@ The header carries two read-only status badges — an **environment** badge
|
||||
a **mode: read-only** badge — plus a **Docs** link to this document. No
|
||||
privileged action controls are present in the Phase 1 shell.
|
||||
|
||||
Not-yet-implemented surfaces (`/sessions`, `/inventory`, `/timeline`,
|
||||
`/policy`, `/insights`) resolve to graceful read-only stub pages instead of
|
||||
404s; their backing views land in later child issues of #631 (the inventory
|
||||
surfaces are backed by #636). Mutating methods on stub routes still fail closed
|
||||
with `read-only-mvp`.
|
||||
Not-yet-implemented surfaces (`/inventory`, `/timeline`, `/policy`,
|
||||
`/insights`) resolve to graceful read-only stub pages instead of 404s; their
|
||||
backing views land in later child issues of #631 (the inventory surfaces are
|
||||
backed by #636). Mutating methods on stub routes still fail closed with
|
||||
`read-only-mvp`.
|
||||
|
||||
### Runtime and sessions (#641)
|
||||
|
||||
`/sessions` is a live Phase 1 read-only view that composes:
|
||||
|
||||
* runtime health from `#430` (profile, role, identity, master parity, stale warning)
|
||||
* control-plane sessions / leases and filesystem locks / worktrees / namespaces from `#636`
|
||||
* durable contamination markers when detectable (`#630` runtime recovery, `#671` stable-branch push)
|
||||
|
||||
It surfaces stale indicators (dead PID, expired lease) and never silences an
|
||||
active contamination marker. Recovery links point only at sanctioned
|
||||
reconnect/operator restart docs (`docs/mcp-namespace-eof-recovery.md`,
|
||||
`docs/mcp-namespace-health.md`, `docs/mcp-restart-path-inventory.md`, this
|
||||
document). The page does **not** restart, kill, or take over sessions; manual
|
||||
`pkill` of MCP daemons is contamination, not recovery.
|
||||
|
||||
Honesty rules specific to this view:
|
||||
|
||||
* **Ownership columns never assert absence they cannot prove.** When the
|
||||
`leases` or `locks` section is degraded or unavailable, the Leases and
|
||||
Worktree-binding cells render `unknown (inventory <status>)` with an
|
||||
*authority unproven* badge instead of `none` / `unbound`, and a caveat names
|
||||
the unreadable sections. A worktree binding is correlated through lease work
|
||||
numbers, so it is unproven when *either* section fails to read.
|
||||
`/api/sessions` carries the same facts as `ownership_authority_complete`,
|
||||
`ownership_section_status`, and per-row `lease_authority` /
|
||||
`worktree_authority`, so a JSON consumer can tell "holds none" from "could
|
||||
not be read".
|
||||
* **Contamination text is redacted at the display boundary.** Marker payloads
|
||||
(`command_summary`, `reason_class`, `session_id`, `role`) are
|
||||
operator-supplied free text that does not arrive through inventory scrubbing,
|
||||
so they pass through `webui.inventory.scrub_text`, which collapses `$HOME` and
|
||||
redacts credential-shaped tokens and URL userinfo *anywhere* in the string.
|
||||
The write-time redactor is a narrow denylist and is not relied on. The field
|
||||
itself is kept — it is the `#630` evidence naming which daemon was killed.
|
||||
|
||||
## Gitea issue/PR linkage (#645)
|
||||
|
||||
`/gitea` is the Phase 3 read-only linkage console: which PR carries which issue,
|
||||
which issues are claimed by more than one PR, and what the latest Canonical
|
||||
Thread Handoff on a thread said. Gitea remains the source of truth — this
|
||||
surface reads it and never writes to it. There is no issue/PR editor, no review,
|
||||
and no merge control.
|
||||
|
||||
Query parameters (all optional):
|
||||
|
||||
| Parameter | Meaning |
|
||||
|-----------|---------|
|
||||
| `project` | Registry project id to scope the read (default: first registry entry) |
|
||||
| `state` | `open` (default) or `all`; `all` widens the window to merged/closed items, where a landed edge lives |
|
||||
| `issue=N` / `pr=N` | Focus one thread and load *its* latest canonical handoff |
|
||||
|
||||
`GET /api/v1/gitea/linkage` returns the same model as JSON
|
||||
(`schema_version: 1`). It answers `502` when the read could not be answered, so
|
||||
an automated consumer cannot mistake a fail-closed payload for "no links exist".
|
||||
The HTML page always answers `200` and renders the reason instead — an operator
|
||||
view must show why a read failed rather than withhold the page.
|
||||
|
||||
### How an edge is found
|
||||
|
||||
Each edge carries the evidence that produced it, strongest first:
|
||||
|
||||
| Evidence | Meaning |
|
||||
|----------|---------|
|
||||
| `closes_keyword` | The PR title or body declares `closes/fixes/resolves #N`. Gitea itself acts on this keyword. |
|
||||
| `branch_marker` | The PR head branch carries the canonical `(fix\|feat\|docs\|chore)/issue-N-…` marker minted by the issue lock. |
|
||||
| `body_reference` | The PR body mentions `#N` with no closing keyword. A mention is not a claim to close. |
|
||||
|
||||
Only closing and branch-marker edges populate the **issue → PR** direction: a
|
||||
bare mention is a cross-link, and counting it as ownership would invent
|
||||
contested issues out of ordinary references. The mention stays visible on the
|
||||
**PR → issue** side, labelled as such. A PR whose two strongest edges tie is
|
||||
flagged `ambiguous`; an issue claimed by two PRs is flagged `contested`.
|
||||
|
||||
### Honesty rules specific to this view
|
||||
|
||||
* **A partial read never reads as an absence.** Linkage is a claim about the
|
||||
loaded window only. When pagination did not complete, every empty edge cell
|
||||
renders `none found (partial inventory)` rather than `none`, and the JSON
|
||||
carries `inventory_complete: false` plus per-row `links_authoritative: false`.
|
||||
* **A failed read renders no table at all.** Missing credentials, an unknown
|
||||
project, or a fetch error produce `ok: false` with a reason. An empty linkage
|
||||
table would assert that no issue is linked to any PR, which such a read is not
|
||||
in a position to claim.
|
||||
* **Handoffs are loaded, never assumed.** CTH comments are thread-scoped, so
|
||||
only the focused issue or PR has its comments fetched. Every other row reports
|
||||
`not_loaded` with the reason; a thread whose comments *were* loaded and carried
|
||||
no CTH says exactly that. A comment-source failure degrades the handoff alone —
|
||||
the linkage tables still render.
|
||||
* **Unrecognised handoff headings are reported, not republished.** A `## CTH:`
|
||||
heading outside `CTH_TYPES` renders as `unrecognized`.
|
||||
* **Redaction precedes display.** Titles, labels, handoff fields, and error
|
||||
reasons pass through `webui.console_redaction` before serialization, and the
|
||||
page HTML-escapes everything it renders.
|
||||
* **Deep links are opt-in.** A link out to the Gitea web UI appears only when
|
||||
`GITEA_MCP_REVEAL_ENDPOINTS=1` is set server-side, matching how the MCP tools
|
||||
gate URL exposure. Item numbers stay usable without it.
|
||||
|
||||
## System-health dashboard (#639)
|
||||
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
# Web Console: Notifications & Human-Attention Routing (#648)
|
||||
|
||||
- **Status:** Phase 3 Live
|
||||
- **Tracking Issue:** [#648](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/648)
|
||||
- **Parent Epic:** [#631](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/631)
|
||||
- **Attention Boundary Reference:** [#628](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/628)
|
||||
|
||||
---
|
||||
|
||||
## 1. Overview
|
||||
|
||||
The **Notifications & Human-Attention Console** (`/notifications`, `/api/v1/notifications`) provides intelligent event classification and human-attention routing for autonomous workflow operations.
|
||||
|
||||
To prevent alert fatigue while ensuring critical escalation boundaries are never missed, events are classified into three distinct **Attention Classes**:
|
||||
|
||||
1. **`human-required`** (Urgent Escalation Boundary):
|
||||
- Items requiring immediate human intervention or business decisions.
|
||||
- Triggers: Auth failures, hard stops, irrecoverable state, decision locks, failed report validations, critical probe errors.
|
||||
- Display: Highlighted in red (`badge-blocked`) with a `HUMAN REQUIRED` badge.
|
||||
|
||||
2. **`operator`** (Operational Inbox):
|
||||
- Items requiring controller or operator review/triage during routine execution.
|
||||
- Triggers: Blocked PRs (merge conflicts), stale leases, duplicate PRs on issues, unassigned ready work.
|
||||
- Display: Displayed in orange/yellow (`badge-claimed`).
|
||||
|
||||
3. **`routine`** (Background Workflow Transitions):
|
||||
- Normal, healthy workflow transitions and state progressions.
|
||||
- Triggers: Active PRs/issues in standard state, clean branch creation, routine heartbeats.
|
||||
- Display: Filtered out of default inbox views to eliminate notification spam; viewable on demand via the "Routine" or "All" tab.
|
||||
|
||||
---
|
||||
|
||||
## 2. API Endpoints
|
||||
|
||||
### `GET /api/v1/notifications`
|
||||
*Compatibility Alias:* `GET /api/notifications`
|
||||
|
||||
#### Query Parameters:
|
||||
- `project_id` (optional): Filter notifications by project ID.
|
||||
- `attention_class` (optional): `inbox` (default: human-required + operator), `human-required`, `operator`, `routine`, `all`.
|
||||
|
||||
#### Example JSON Response:
|
||||
```json
|
||||
{
|
||||
"project_id": "gitea-tools",
|
||||
"repo_label": "Scaled-Tech-Consulting/Gitea-Tools",
|
||||
"human_required_count": 0,
|
||||
"operator_count": 2,
|
||||
"routine_count": 5,
|
||||
"total_count": 7,
|
||||
"fetch_error": null,
|
||||
"inbox_items": [
|
||||
{
|
||||
"id": "notif-pr-block-742",
|
||||
"attention_class": "operator",
|
||||
"category": "blocker",
|
||||
"title": "Blocked PR #742",
|
||||
"summary": "PR #742 requires merge conflict resolution.",
|
||||
"work_kind": "pr",
|
||||
"work_number": 742,
|
||||
"project_id": "gitea-tools",
|
||||
"repo_label": "Scaled-Tech-Consulting/Gitea-Tools",
|
||||
"created_at": "2026-07-25T16:39:47Z",
|
||||
"deep_link": "/traffic",
|
||||
"requires_human": false,
|
||||
"extra": {}
|
||||
}
|
||||
],
|
||||
"all_items": [...]
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. UI Navigation
|
||||
|
||||
- Access via the **Traffic** navigation menu: **Traffic → Notifications**.
|
||||
- The main view displays:
|
||||
- **Metrics Summary Bar**: Highlighting counts for Human Required, Operator Inbox, and Routine items.
|
||||
- **Attention Filter Tabs**: Toggle between Inbox (Human + Operator), Human Required, Operator, Routine, and All.
|
||||
- **Structured Event Table**: Displays category, title, summary, work item links, and timestamps.
|
||||
@@ -0,0 +1,160 @@
|
||||
# Web console requests: intent preview and workflow initiation (#643)
|
||||
|
||||
**Phase 2. Preview is always live and always read-only. Initiation is wired but
|
||||
denied until an operator opts in.**
|
||||
|
||||
Before this surface, starting role work meant pasting a prompt into a terminal
|
||||
and trusting the operator to have checked the allocator first. Nothing enforced
|
||||
that check, so two sessions could reach for the same issue and each believe it
|
||||
was theirs. This page replaces the paste with a *request*: a desired role, an
|
||||
issue or PR, and a stated intent, answered by an authorization decision and —
|
||||
on confirmation — an exclusive assignment from the allocator.
|
||||
|
||||
| Concern | Module |
|
||||
|---------|--------|
|
||||
| Request model, preview, initiation | `webui/request_service.py` |
|
||||
| Form and preview rendering | `webui/request_views.py` |
|
||||
| Authorization | `webui/console_authz.py` (`initiate_workflow`) |
|
||||
| Audit | `webui/console_audit.py` |
|
||||
| Ownership substrate | `allocator_service.py` + `control_plane_db.py` |
|
||||
|
||||
## Surfaces
|
||||
|
||||
| Path | Method | Purpose |
|
||||
|------|--------|---------|
|
||||
| `/requests` | GET | Request form |
|
||||
| `/requests` | POST | Render an intent preview. **Never assigns.** |
|
||||
| `/api/v1/requests/preview` | POST | Intent preview as JSON |
|
||||
| `/api/v1/requests/apply` | POST | Initiate — confirmed, audited, allocator-owned |
|
||||
|
||||
The HTML form has no initiate button on purpose. Initiating requires a
|
||||
confirmed POST to `/api/v1/requests/apply`, so a stray form submission cannot
|
||||
reserve work as a side effect.
|
||||
|
||||
## The request
|
||||
|
||||
```json
|
||||
{
|
||||
"desired_role": "author",
|
||||
"work_kind": "issue",
|
||||
"work_number": 643,
|
||||
"intent_summary": "implement request preview and initiation",
|
||||
"remote": "prgs",
|
||||
"org": "Scaled-Tech-Consulting",
|
||||
"repo": "Gitea-Tools",
|
||||
"expected_head_sha": null
|
||||
}
|
||||
```
|
||||
|
||||
`desired_role` is one of `author`, `reviewer`, `merger`, `reconciler`,
|
||||
`controller`. `work_kind` is `issue` or `pr`. `remote`/`org`/`repo` default to
|
||||
the first project in the registry when omitted; when neither the request nor
|
||||
the registry resolves them, the request is rejected rather than pointed at some
|
||||
other repository. `intent_summary` is required — it is what the audit record
|
||||
states as the reason — and is truncated to 500 characters.
|
||||
|
||||
Parsing rejects rather than corrects. An unknown role, an unknown work kind, a
|
||||
non-positive number, or a missing intent each return `400` with a `reason_code`
|
||||
and the offending `field`.
|
||||
|
||||
## Preview
|
||||
|
||||
Five checks, each with its own verdict, reason code, and detail:
|
||||
|
||||
| Check | Passes when |
|
||||
|-------|-------------|
|
||||
| `authorization` | The console principal holds `operator` or above |
|
||||
| `capability` | The desired role maps to a declared profile and MCP namespace |
|
||||
| `lease_availability` | No active claim holds the work unit |
|
||||
| `next_safe_action` | The allocator would independently select this exact work unit |
|
||||
| `head_pin` | PR work resolves to a head SHA, and a supplied SHA still matches |
|
||||
|
||||
A preview also returns the role's `allowed_actions` and `prohibited_actions`
|
||||
(from `allocator_service.ROLE_ACTIONS`), the `required_profile` and
|
||||
`required_namespace` the work must run under, and a `correlation_id` that ties
|
||||
the preview to its audit record and to any assignment that follows.
|
||||
|
||||
Preview is read-only in the strict sense: it calls the allocator with
|
||||
`apply=false` and writes nothing but an audit line. An unauthorized principal
|
||||
never reaches the allocator or the control-plane DB at all, so a denial cannot
|
||||
be used to enumerate the queue.
|
||||
|
||||
## Initiation
|
||||
|
||||
`POST /api/v1/requests/apply` refuses in this order, and every refusal returns
|
||||
before any assignment is attempted:
|
||||
|
||||
| Condition | Outcome | Status |
|
||||
|-----------|---------|--------|
|
||||
| Unparseable request | `invalid_request` | 400 |
|
||||
| Not authorized, or execution not wired | `denied` | 403 |
|
||||
| `confirm` not set | `denied` / `confirmation_required` | 409 |
|
||||
| Work unit already claimed | `blocked` / `duplicate_assignment` | 409 |
|
||||
| Allocator would select other work | `wait` / `not_next_safe_work` | 409 |
|
||||
| Allocator declines on apply | `blocked` or `wait` | 409 |
|
||||
| Evidence unavailable | `wait` / `evidence_unavailable` | 503 |
|
||||
| Assigned | `assigned_work` | 201 |
|
||||
|
||||
A success returns the assignment plus a `handoff` block naming the profile, the
|
||||
namespace, and the actions that stay forbidden — enough for the operator to
|
||||
continue in the right MCP namespace without guessing.
|
||||
|
||||
### Why apply runs the allocator twice
|
||||
|
||||
The allocator is the only source of exclusive ownership (#600 / #613), and it
|
||||
selects work; it does not take orders. So `apply` runs a dry-run first and
|
||||
proceeds only when the allocator would independently pick the requested work
|
||||
unit. If it would not, the request reports `wait` and mutates nothing.
|
||||
|
||||
A request is therefore a *confirmation* of the allocator's decision, never an
|
||||
override of it. The apply call carries the dry-run's
|
||||
`candidate_set_fingerprint` as a CAS pin (#776), so a queue that changed
|
||||
between the two calls fails closed rather than assigning against a stale view.
|
||||
The result is checked again on the way out: an assignment naming a different
|
||||
work unit is not read as success.
|
||||
|
||||
### Fail-closed defaults
|
||||
|
||||
- An unreadable control-plane DB denies. It is never treated as "nothing holds
|
||||
this work unit".
|
||||
- An incomplete queue inventory denies (#758). Ranking a partial candidate set
|
||||
can select the wrong work.
|
||||
- An allocator that raises denies.
|
||||
- PR work with no resolvable head SHA denies; a supplied SHA that no longer
|
||||
matches denies with `head_moved`.
|
||||
|
||||
## Enabling initiation
|
||||
|
||||
Execution is wired off. Set `WEBUI_REQUESTS_EXECUTION=1` to enable it for the
|
||||
`initiate_workflow` action only — see
|
||||
[`webui-authz-audit.md`](webui-authz-audit.md) for why this is an
|
||||
action-scoped flag rather than a phase bump. With the variable unset, `apply`
|
||||
returns `403` with `reason_code: unauthorized` no matter who asks.
|
||||
|
||||
Enabling execution does **not** enable approvals or merges. Those are phase 3
|
||||
console actions and remain forbidden in every path here; the console reserves
|
||||
work and hands off, and the MCP role profile enforces what that role may then
|
||||
do.
|
||||
|
||||
## Audit
|
||||
|
||||
Every preview and every apply emits a console audit record (schema in
|
||||
[`webui-authz-audit.md`](webui-authz-audit.md)):
|
||||
|
||||
| Event | `result` |
|
||||
|-------|----------|
|
||||
| Preview | `previewed` |
|
||||
| Refusal at any stage | `denied` |
|
||||
| Assignment created | `succeeded` |
|
||||
|
||||
`correlation.request_id` carries the request's `correlation_id`, and a
|
||||
successful record's `metadata` carries `assignment_id` and `lease_id`, so an
|
||||
assignment can be traced back to the intent that produced it. The operator's
|
||||
`intent_summary` travels in `metadata` and passes through the standard
|
||||
redaction pass before persistence like every other field.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- No browser-initiated approve or merge, in this phase or any other.
|
||||
- No bypass of allocator exclusive ownership; no self-selection of work.
|
||||
- No auto-start from raw monitoring incidents (#612 stays downstream).
|
||||
@@ -0,0 +1,102 @@
|
||||
# Web Console: restart status, impact preview, and approval state (#667)
|
||||
|
||||
Phase 1 of the console restart surface. It consumes the #655 coordinator
|
||||
substrate and displays it. It performs no restart, reload, drain, approval, or
|
||||
process action, and it registers no write endpoint.
|
||||
|
||||
Issue #667's rollout is explicit — *status views first, write approval after the
|
||||
backend gates are green* — and this change delivers only the status half.
|
||||
|
||||
## Surfaces
|
||||
|
||||
| Path | Method | Purpose |
|
||||
|------|--------|---------|
|
||||
| `/runtime/restart` | GET | Restart status page |
|
||||
| `/api/v1/system/restart/status` | GET | Same snapshot as JSON |
|
||||
|
||||
Both accept an optional `restart_class` query parameter (default
|
||||
`full_mcp_restart`). An unrecognised class is not an error: the coordinator
|
||||
resolves it as unknown and fails closed, and the page shows the resulting deny.
|
||||
|
||||
Neither path accepts `POST`; a write attempt returns `405`, and a test asserts
|
||||
it.
|
||||
|
||||
## What it shows
|
||||
|
||||
* **Impact preview (#658)** — verdict, blast radius, affected sessions, leases,
|
||||
critical sections, mutations, and the counts behind them, evaluated
|
||||
`dry_run=True` against live control-plane state.
|
||||
* **Drain proof (#661)** — verification of a supplied proof: valid, clean,
|
||||
expired, tampered, and the reasons behind a refusal.
|
||||
* **Post-restart reconcile (#662)** — the most recent completion proof, its
|
||||
overall status, and which dimensions still require follow-up.
|
||||
* **Restart classes (#663)** — the least-privilege matrix, with *you may
|
||||
request* and *you may execute* computed for the viewing role rather than for a
|
||||
generic operator.
|
||||
* **Approval controls (#633)** — the authorization state of
|
||||
`system.restart_namespace` and `system.reload_namespace`.
|
||||
* **Break-glass (#664)** — declared and marked unavailable; see below.
|
||||
|
||||
## Three rules this surface holds itself to
|
||||
|
||||
A status page that is wrong is worse than one that is missing, because an
|
||||
operator acts on it. Three properties are enforced by tests, and each was
|
||||
verified by reverting the guard and watching a test fail.
|
||||
|
||||
### An unreadable source reports unavailable, never green
|
||||
|
||||
Every source carries its own `SourceStatus`. Nothing substitutes a default,
|
||||
placeholder, or self-comparison for a reading that failed. An unreadable
|
||||
control-plane database yields `inventory_complete: false`, which the coordinator
|
||||
itself turns into a fail-closed verdict, and the page says the blast radius is
|
||||
unknown rather than showing an empty affected-sessions table.
|
||||
|
||||
An absent drain proof is reported as absent — not as a pass. The #661 gate
|
||||
authorizes a restart only against a valid, unexpired, clean proof, so no proof
|
||||
is precisely the state that gate denies on.
|
||||
|
||||
### Authorization is asked the way execution would ask it
|
||||
|
||||
Every probe passes `for_execution=True`.
|
||||
|
||||
Asked without it, an admin is `allowed` for `system.restart_namespace`. On a
|
||||
control surface that reads as a live button. Asked the way an execution attempt
|
||||
would ask, the same principal is refused `phase_not_active`, because the console
|
||||
is in Phase 1 and the action is Phase 2. This surface reports the second answer.
|
||||
|
||||
`execution_enabled` is therefore `false` for every action and every role today,
|
||||
and a test asserts that across the whole role matrix.
|
||||
|
||||
### The control-plane database is opened read-only
|
||||
|
||||
`ControlPlaneDB()` creates directories and runs migrations on construction — a
|
||||
write. This surface never constructs one. It opens the sqlite file with
|
||||
`mode=ro`, exactly as `webui/inventory.py` does, and treats a missing file as
|
||||
missing authority rather than as an empty inventory.
|
||||
|
||||
The test that protects this points at a path inside a directory that already
|
||||
exists, so a read-write `connect` would really create the file. A nested
|
||||
missing-directory path would have passed for the wrong reason.
|
||||
|
||||
## Break-glass is declared, not offered
|
||||
|
||||
The break-glass workflow (#664) is not available on this branch's base. The
|
||||
panel is rendered to operator-class roles as **unavailable**, naming the issue
|
||||
that tracks it. It is not silently omitted, because an operator who has been
|
||||
told a governance path exists needs to see that it is not wired here; and it is
|
||||
not rendered as a control, because there is nothing behind it.
|
||||
|
||||
Unprivileged viewers see only a note that the surface is operator-class.
|
||||
|
||||
## Redaction and escaping
|
||||
|
||||
Every interpolated value passes through `_esc` (`html.escape(..., quote=True)`).
|
||||
Free-form text and anything that can carry a filesystem path additionally passes
|
||||
through `webui.inventory.scrub_text`, which redacts credential-shaped tokens
|
||||
inside a string rather than only at its start. The impact payload is passed
|
||||
through `webui.inventory.scrub` before rendering.
|
||||
|
||||
## Linkage
|
||||
|
||||
Parent #655 · extends #642 · consumes #658, #661, #662, #663 · RBAC #633 ·
|
||||
console #631 · vision #652 · roadmap #653 · break-glass #664.
|
||||
+1015
File diff suppressed because it is too large
Load Diff
+676
-80
@@ -1472,6 +1472,9 @@ def verify_preflight_purity(
|
||||
# contaminated by manual MCP daemon process killing (reconciler-exempt).
|
||||
_enforce_runtime_recovery_contamination_gate(task, remote)
|
||||
|
||||
# #659 AC3: defer non-allowlisted mutations while maintenance drain is active.
|
||||
_enforce_maintenance_drain_gate(task, remote=remote, org=org, repo=repo)
|
||||
|
||||
ctx = _resolve_namespace_mutation_context(worktree_path)
|
||||
workspace = ctx["workspace_path"]
|
||||
canonical_root = ctx["canonical_repo_root"]
|
||||
@@ -2004,6 +2007,55 @@ def _enforce_runtime_recovery_contamination_gate(
|
||||
)
|
||||
|
||||
|
||||
def _enforce_maintenance_drain_gate(
|
||||
task: str | None,
|
||||
remote: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
) -> None:
|
||||
"""#659 AC3: defer non-allowlisted mutations while drain is active.
|
||||
|
||||
The single mutation chokepoint already used by every gated task, so drain
|
||||
coverage cannot drift per-tool. Allowlisted safety operations (heartbeat,
|
||||
release/abandon, checkpoint, drain exit) pass through so an in-flight
|
||||
session can still finish and hand off; everything else is deferred with a
|
||||
typed blocker. Unreadable drain state fails closed — a drain that cannot be
|
||||
read is not evidence that no drain is running.
|
||||
"""
|
||||
if _preflight_in_test_mode() and not os.environ.get(
|
||||
"GITEA_TEST_FORCE_MAINTENANCE_DRAIN"
|
||||
):
|
||||
return
|
||||
if maintenance_drain.is_allowlisted_task(task):
|
||||
return
|
||||
|
||||
try:
|
||||
_h, o, r = _resolve(remote, None, org, repo)
|
||||
except Exception: # noqa: BLE001 — scope resolution is best-effort here
|
||||
o, r = (org or ""), (repo or "")
|
||||
|
||||
db, errs = _control_plane_db_or_error()
|
||||
if db is None:
|
||||
raise RuntimeError(
|
||||
"maintenance-drain state could not be read: "
|
||||
f"{'; '.join(errs) or 'control-plane DB unavailable'} (fail closed, #659)"
|
||||
)
|
||||
try:
|
||||
record = db.read_maintenance_drain(remote=remote or "", org=o, repo=r)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
raise RuntimeError(
|
||||
f"maintenance-drain state could not be read: {_redact(str(exc))} "
|
||||
"(fail closed, #659)"
|
||||
) from exc
|
||||
|
||||
decision = maintenance_drain.classify_mutation(task, record)
|
||||
if not decision["allowed"]:
|
||||
raise maintenance_drain.MaintenanceDrainError(
|
||||
maintenance_drain.format_drain_block_error(decision),
|
||||
decision=decision,
|
||||
)
|
||||
|
||||
|
||||
def _enforce_stable_branch_contamination_gate(
|
||||
task: str | None,
|
||||
remote: str | None = None,
|
||||
@@ -2064,10 +2116,12 @@ import allocator_service # noqa: E402
|
||||
import allocator_dependencies # noqa: E402
|
||||
import dependency_graph # noqa: E402 # #784 durable dependency edges
|
||||
import control_plane_db # noqa: E402
|
||||
import maintenance_drain # noqa: E402 # #659 graceful maintenance-drain mode
|
||||
import lease_lifecycle # noqa: E402
|
||||
import lease_policy # noqa: E402
|
||||
import workflow_dashboard # noqa: E402 # #605 live queue/lease dashboard
|
||||
import restart_coordinator # noqa: E402 # #658 MCP restart coordinator/impact
|
||||
import drain_proof # noqa: E402 # #661 pre-restart drain proof and hard gate
|
||||
import incident_bridge # noqa: E402
|
||||
import sentry_observability # noqa: E402 (#606 optional Sentry observability)
|
||||
import sentry_incident_bridge # noqa: E402 (#607 Sentry→Gitea incident bridge)
|
||||
@@ -9551,15 +9605,13 @@ def gitea_edit_pr(
|
||||
if closing:
|
||||
gate_reasons = _profile_operation_gate("gitea.pr.close")
|
||||
if gate_reasons:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"pr_number": pr_number,
|
||||
"requested_state": "closed",
|
||||
"required_permission": "gitea.pr.close",
|
||||
"reasons": gate_reasons,
|
||||
"permission_report": _permission_block_report("gitea.pr.close"),
|
||||
}
|
||||
return _build_operation_gate_refusal(
|
||||
"gitea.pr.close",
|
||||
gate_reasons,
|
||||
pr_number=pr_number,
|
||||
requested_state="closed",
|
||||
required_permission="gitea.pr.close",
|
||||
)
|
||||
|
||||
h, o, r = _resolve(remote, host, org, repo)
|
||||
auth = _auth(h)
|
||||
@@ -13822,13 +13874,17 @@ def gitea_view_issue(
|
||||
|
||||
def _permission_block_report(required_operation: str,
|
||||
identity: str | None = None) -> dict:
|
||||
"""Structured, LLM-safe explanation of a permission denial (#142).
|
||||
"""Structured, LLM-safe explanation of a permission denial (#142, #897).
|
||||
|
||||
Built only after a gate has already refused; it adds guidance to the
|
||||
refusal and never widens any permission, performs network I/O, or
|
||||
raises (fail-soft: degrades to a minimal fail-closed report). Names
|
||||
configured profiles only — never auth references, tokens, endpoint
|
||||
URLs, or keychain IDs.
|
||||
|
||||
#897: never fabricate a missing permission when the active profile
|
||||
already allows the operation. That path is a diagnostic defect (the
|
||||
refusal was not a permission denial), not a cue to switch profiles.
|
||||
"""
|
||||
report = {
|
||||
"requested_operation": required_operation,
|
||||
@@ -13840,6 +13896,7 @@ def _permission_block_report(required_operation: str,
|
||||
"matching_configured_profiles": [],
|
||||
"runtime_switching_supported": False,
|
||||
"different_mcp_namespace_required": True,
|
||||
"diagnostic_defect": False,
|
||||
"exact_safe_next_action": (
|
||||
"Ask the operator to fix GITEA_MCP_CONFIG/GITEA_MCP_PROFILE; "
|
||||
"the active profile could not be resolved (fail closed)."),
|
||||
@@ -13854,6 +13911,32 @@ def _permission_block_report(required_operation: str,
|
||||
report["active_allowed_operations"] = (
|
||||
profile.get("allowed_operations") or [])
|
||||
|
||||
# #897: fail closed as a diagnostic defect when the active profile
|
||||
# already holds the operation — callers must not invent "missing".
|
||||
try:
|
||||
holds, _hold_reason = gitea_config.check_operation(
|
||||
required_operation,
|
||||
profile.get("allowed_operations") or [],
|
||||
profile.get("forbidden_operations") or [],
|
||||
)
|
||||
except Exception:
|
||||
holds = False
|
||||
if holds:
|
||||
report["missing_permission"] = None
|
||||
report["required_permission"] = required_operation
|
||||
report["diagnostic_defect"] = True
|
||||
report["different_mcp_namespace_required"] = False
|
||||
report["exact_safe_next_action"] = (
|
||||
"Diagnostic defect: the active profile already allows "
|
||||
f"{required_operation}. This is not a permission denial — "
|
||||
"inspect blocker_kind / reasons (stale-runtime or runtime-mode). "
|
||||
"Do not call gitea_activate_profile or switch MCP sessions."
|
||||
)
|
||||
report["matching_configured_profiles"] = [
|
||||
p for p in [profile.get("profile_name")] if p
|
||||
]
|
||||
return report
|
||||
|
||||
matching = []
|
||||
try:
|
||||
config = gitea_config.load_config() or {}
|
||||
@@ -13902,6 +13985,205 @@ def _permission_block_report(required_operation: str,
|
||||
return report
|
||||
|
||||
|
||||
def _reason_is_stale_runtime(reason: str) -> bool:
|
||||
"""True when *reason* is a master-parity / stale-daemon refusal (#897)."""
|
||||
r = (reason or "").lower()
|
||||
if not r:
|
||||
return False
|
||||
if "stale relative to live master" in r:
|
||||
return True
|
||||
if "server code is stale" in r:
|
||||
return True
|
||||
if "daemon is stale" in r:
|
||||
return True
|
||||
if "started at commit" in r and "workspace master is now" in r:
|
||||
return True
|
||||
if "mcp server started at" in r and "stale" in r:
|
||||
return True
|
||||
if "restart the server to load the current capability gates" in r:
|
||||
return True
|
||||
if "restart/reconnect before mutating" in r:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _reason_is_runtime_mode(reason: str) -> bool:
|
||||
"""True when *reason* is a stable-control / runtime-mode refusal (#897)."""
|
||||
r = (reason or "").lower()
|
||||
if not r:
|
||||
return False
|
||||
if _reason_is_stale_runtime(reason):
|
||||
return False
|
||||
if "runtime mode could not be assessed" in r:
|
||||
return True
|
||||
if "runtime mode is" in r:
|
||||
return True
|
||||
if "stable control runtime" in r:
|
||||
return True
|
||||
if "dev-test" in r and ("runtime" in r or "production" in r):
|
||||
return True
|
||||
if "development worktree" in r or "dev worktree" in r:
|
||||
return True
|
||||
if "launched from a 'branches/" in r or "launched from a \"branches/" in r:
|
||||
return True
|
||||
if "process-root / active-workspace alignment" in r:
|
||||
return True
|
||||
if "namespace" in r and "reproof" in r:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _reason_is_permission(reason: str) -> bool:
|
||||
"""True when *reason* is a genuine profile-permission denial (#897)."""
|
||||
r = (reason or "").lower()
|
||||
if not r:
|
||||
return False
|
||||
if _reason_is_stale_runtime(reason) or _reason_is_runtime_mode(reason):
|
||||
return False
|
||||
if "profile could not be resolved" in r:
|
||||
return True
|
||||
if "profile has no configured allowed operations" in r:
|
||||
return True
|
||||
if "profile forbids" in r:
|
||||
return True
|
||||
if "profile is not allowed to" in r:
|
||||
return True
|
||||
if "unrecognized forbidden operation" in r:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _classify_operation_gate_reasons(reasons: list[str]) -> dict:
|
||||
"""Partition gate reasons into stale / runtime-mode / permission (#897)."""
|
||||
stale: list[str] = []
|
||||
runtime_mode: list[str] = []
|
||||
permission: list[str] = []
|
||||
other: list[str] = []
|
||||
for reason in reasons or []:
|
||||
if _reason_is_stale_runtime(reason):
|
||||
stale.append(reason)
|
||||
elif _reason_is_runtime_mode(reason):
|
||||
runtime_mode.append(reason)
|
||||
elif _reason_is_permission(reason):
|
||||
permission.append(reason)
|
||||
else:
|
||||
other.append(reason)
|
||||
return {
|
||||
"stale_runtime": stale,
|
||||
"runtime_mode": runtime_mode,
|
||||
"permission": permission,
|
||||
"other": other,
|
||||
}
|
||||
|
||||
|
||||
def _stale_runtime_reconnect_action() -> str:
|
||||
"""Sanctioned recovery for a stale daemon — reconnect only (#685/#897)."""
|
||||
return (
|
||||
"Reconnect the IDE/client MCP session so the server reloads at the "
|
||||
"current master head. Do not call gitea_activate_profile or switch "
|
||||
"MCP role sessions — profile switching does not clear a stale daemon."
|
||||
)
|
||||
|
||||
|
||||
def _build_operation_gate_refusal(
|
||||
required_operation: str,
|
||||
reasons: list[str],
|
||||
**extra_fields,
|
||||
) -> dict:
|
||||
"""Structured gate refusal with typed blockers (#897).
|
||||
|
||||
Stale-runtime and runtime-mode refusals never attach a
|
||||
``permission_report`` and never recommend profile switching. True
|
||||
permission denials still get ``permission_report``. When both apply,
|
||||
causes are reported separately under distinct fields.
|
||||
"""
|
||||
classified = _classify_operation_gate_reasons(reasons)
|
||||
stale = classified["stale_runtime"]
|
||||
runtime_mode = classified["runtime_mode"]
|
||||
permission = classified["permission"]
|
||||
other = classified["other"]
|
||||
|
||||
blocked: dict = {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": list(reasons),
|
||||
"mutation_performed": False,
|
||||
"session_context_audit": session_ctx.mutation_context_audit_fields(),
|
||||
"gate_reason_classes": {
|
||||
"stale_runtime": list(stale),
|
||||
"runtime_mode": list(runtime_mode),
|
||||
"permission": list(permission),
|
||||
"other": list(other),
|
||||
},
|
||||
}
|
||||
|
||||
if stale:
|
||||
parity = _current_master_parity()
|
||||
blocked["blocker_kind"] = "runtime_reconnect_required"
|
||||
blocked["restart_required"] = True
|
||||
blocked["stop_required"] = True
|
||||
blocked["startup_head"] = parity.get("startup_head")
|
||||
blocked["current_head"] = parity.get("current_head")
|
||||
blocked["daemon_start_head"] = (
|
||||
parity.get("daemon_start_head") or parity.get("startup_head")
|
||||
)
|
||||
blocked["local_head"] = (
|
||||
parity.get("local_head") or parity.get("current_head")
|
||||
)
|
||||
blocked["live_remote_head"] = parity.get("live_remote_head")
|
||||
blocked["live_stale"] = bool(parity.get("live_stale"))
|
||||
blocked["live_known"] = bool(parity.get("live_known"))
|
||||
blocked["exact_safe_next_action"] = _stale_runtime_reconnect_action()
|
||||
if permission or other:
|
||||
blocked["permission_block_reasons"] = list(permission) + list(other)
|
||||
blocked["stale_runtime_reasons"] = list(stale)
|
||||
# Never attach permission_report for a staleness refusal.
|
||||
blocked.update(extra_fields)
|
||||
return blocked
|
||||
|
||||
if runtime_mode:
|
||||
blocked["blocker_kind"] = "runtime_mode_blocked"
|
||||
blocked["restart_required"] = False
|
||||
blocked["stop_required"] = True
|
||||
blocked["exact_safe_next_action"] = (
|
||||
"Real workflow mutations run only on the promoted stable control "
|
||||
"runtime. Promote/reload the stable runtime; do not call "
|
||||
"gitea_activate_profile or switch MCP role sessions to clear a "
|
||||
"runtime-mode block."
|
||||
)
|
||||
if permission or other:
|
||||
blocked["permission_block_reasons"] = list(permission) + list(other)
|
||||
blocked["runtime_mode_reasons"] = list(runtime_mode)
|
||||
blocked.update(extra_fields)
|
||||
return blocked
|
||||
|
||||
# Pure permission (or unclassified-as-permission) denial.
|
||||
blocked["blocker_kind"] = "permission_denied"
|
||||
blocked["permission_report"] = _permission_block_report(required_operation)
|
||||
blocked.update(extra_fields)
|
||||
return blocked
|
||||
|
||||
|
||||
def _permission_report_for_gate_reasons(
|
||||
required_operation: str,
|
||||
reasons: list[str] | None,
|
||||
) -> dict | None:
|
||||
"""Attach ``permission_report`` only for true permission denials (#897).
|
||||
|
||||
Call sites that historically always attached a permission report after
|
||||
``_profile_operation_gate`` should use this so stale/runtime refusals
|
||||
do not emit a fabricated missing-permission payload.
|
||||
"""
|
||||
if not reasons:
|
||||
return None
|
||||
classified = _classify_operation_gate_reasons(reasons)
|
||||
if classified["stale_runtime"] or classified["runtime_mode"]:
|
||||
return None
|
||||
if not (classified["permission"] or classified["other"]):
|
||||
return None
|
||||
return _permission_block_report(required_operation)
|
||||
|
||||
|
||||
def _role_for_operation(op: str) -> str | None:
|
||||
# Normalize op first
|
||||
try:
|
||||
@@ -14078,7 +14360,7 @@ def _master_parity_block(op: str) -> list[str]:
|
||||
|
||||
|
||||
def _profile_operation_gate(op: str) -> list[str]:
|
||||
"""Profile permission check for a single gated operation (#126, #216, #420).
|
||||
"""Profile permission check for a single gated operation (#126, #216, #420, #897).
|
||||
|
||||
Issue discussion comments are gated separately from the gitea.pr.*
|
||||
review/merge family: listing requires ``gitea.read``, creating requires
|
||||
@@ -14091,21 +14373,26 @@ def _profile_operation_gate(op: str) -> list[str]:
|
||||
capability gate that has since been merged, and when the runtime itself is
|
||||
not the promoted stable control runtime (#615) -- a dev/test or unknown
|
||||
runtime holds production credentials but has not been promoted.
|
||||
|
||||
#897: collect *all* independent refusal classes (stale, runtime-mode,
|
||||
permission) rather than short-circuiting after the first. Callers that
|
||||
only need a boolean still treat any non-empty list as blocked; typed
|
||||
consumers (``_build_operation_gate_refusal``) can separate causes.
|
||||
"""
|
||||
stale_reasons = _master_parity_block(op)
|
||||
if stale_reasons:
|
||||
return stale_reasons
|
||||
runtime_reasons = _runtime_mode_block(op)
|
||||
if runtime_reasons:
|
||||
return runtime_reasons
|
||||
reasons: list[str] = []
|
||||
reasons.extend(_master_parity_block(op))
|
||||
reasons.extend(_runtime_mode_block(op))
|
||||
try:
|
||||
profile = get_profile()
|
||||
except Exception as exc:
|
||||
return [f"profile could not be resolved (fail closed): {_redact(str(exc))}"]
|
||||
reasons.append(
|
||||
f"profile could not be resolved (fail closed): {_redact(str(exc))}"
|
||||
)
|
||||
return reasons
|
||||
op_ok, op_reason = gitea_config.check_operation(
|
||||
op, profile["allowed_operations"], profile["forbidden_operations"])
|
||||
if op_ok:
|
||||
return []
|
||||
return reasons
|
||||
|
||||
if _try_auto_switch_for_operation(op):
|
||||
try:
|
||||
@@ -14113,17 +14400,26 @@ def _profile_operation_gate(op: str) -> list[str]:
|
||||
op_ok, op_reason = gitea_config.check_operation(
|
||||
op, profile["allowed_operations"], profile["forbidden_operations"])
|
||||
if op_ok:
|
||||
return []
|
||||
return reasons
|
||||
except Exception as exc:
|
||||
return [f"profile could not be resolved (fail closed): {_redact(str(exc))}"]
|
||||
reasons.append(
|
||||
f"profile could not be resolved (fail closed): {_redact(str(exc))}"
|
||||
)
|
||||
return reasons
|
||||
|
||||
if op_reason == "no-allowed-operations":
|
||||
return ["profile has no configured allowed operations (fail closed)"]
|
||||
if op_reason == "forbidden":
|
||||
return [f"profile forbids '{op}'"]
|
||||
if op_reason == "invalid-forbidden-entry":
|
||||
return ["profile has an unrecognized forbidden operation entry (fail closed)"]
|
||||
return [f"profile is not allowed to {op}"]
|
||||
reasons.append(
|
||||
"profile has no configured allowed operations (fail closed)"
|
||||
)
|
||||
elif op_reason == "forbidden":
|
||||
reasons.append(f"profile forbids '{op}'")
|
||||
elif op_reason == "invalid-forbidden-entry":
|
||||
reasons.append(
|
||||
"profile has an unrecognized forbidden operation entry (fail closed)"
|
||||
)
|
||||
else:
|
||||
reasons.append(f"profile is not allowed to {op}")
|
||||
return reasons
|
||||
|
||||
|
||||
def _mutation_config_authority_block(required_operation: str) -> dict | None:
|
||||
@@ -14343,10 +14639,14 @@ def _session_context_mutation_block(
|
||||
|
||||
|
||||
def _profile_permission_block(required_operation: str, **extra_fields) -> dict | None:
|
||||
"""Structured permission denial for gated tools (#69, #142).
|
||||
"""Structured operation-gate denial for gated tools (#69, #142, #897).
|
||||
|
||||
Returns a block dict when the active profile forbids *required_operation*,
|
||||
or ``None`` when the gate passes. Never performs network I/O.
|
||||
the daemon is stale, or the runtime mode is not mutation-safe — or
|
||||
``None`` when the gate passes. Never performs network I/O.
|
||||
|
||||
#897: stale-runtime and runtime-mode refusals are typed
|
||||
(``blocker_kind``) and never carry a ``permission_report``.
|
||||
"""
|
||||
req_role = "reviewer" if any(required_operation.startswith(p) for p in (
|
||||
"gitea.pr.approve", "gitea.pr.merge", "gitea.pr.request_changes", "gitea.pr.review"
|
||||
@@ -14356,15 +14656,9 @@ def _profile_permission_block(required_operation: str, **extra_fields) -> dict |
|
||||
|
||||
reasons = _profile_operation_gate(required_operation)
|
||||
if reasons:
|
||||
blocked = {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": reasons,
|
||||
"permission_report": _permission_block_report(required_operation),
|
||||
"session_context_audit": session_ctx.mutation_context_audit_fields(),
|
||||
}
|
||||
blocked.update(extra_fields)
|
||||
return blocked
|
||||
return _build_operation_gate_refusal(
|
||||
required_operation, reasons, **extra_fields
|
||||
)
|
||||
|
||||
auth_block = _mutation_config_authority_block(required_operation)
|
||||
if auth_block is not None:
|
||||
@@ -14493,20 +14787,14 @@ def gitea_acquire_reviewer_pr_lease(
|
||||
"""Acquire a per-PR reviewer lease before review/merge mutations (#407)."""
|
||||
read_block = _profile_operation_gate("gitea.read")
|
||||
if read_block:
|
||||
return {
|
||||
"success": False,
|
||||
"acquired": False,
|
||||
"reasons": read_block,
|
||||
"permission_report": _permission_block_report("gitea.read"),
|
||||
}
|
||||
return _build_operation_gate_refusal(
|
||||
"gitea.read", read_block, acquired=False
|
||||
)
|
||||
comment_block = _profile_operation_gate("gitea.pr.comment")
|
||||
if comment_block:
|
||||
return {
|
||||
"success": False,
|
||||
"acquired": False,
|
||||
"reasons": comment_block,
|
||||
"permission_report": _permission_block_report("gitea.pr.comment"),
|
||||
}
|
||||
return _build_operation_gate_refusal(
|
||||
"gitea.pr.comment", comment_block, acquired=False
|
||||
)
|
||||
|
||||
# task=acquire_reviewer_pr_lease so verify_preflight_purity runs shared #604
|
||||
# anti-stomp for the declared lease-acquire mutation inventory entry.
|
||||
@@ -14622,20 +14910,14 @@ def gitea_acquire_merger_pr_lease(
|
||||
"""
|
||||
read_block = _profile_operation_gate("gitea.read")
|
||||
if read_block:
|
||||
return {
|
||||
"success": False,
|
||||
"acquired": False,
|
||||
"reasons": read_block,
|
||||
"permission_report": _permission_block_report("gitea.read"),
|
||||
}
|
||||
return _build_operation_gate_refusal(
|
||||
"gitea.read", read_block, acquired=False
|
||||
)
|
||||
comment_block = _profile_operation_gate("gitea.pr.comment")
|
||||
if comment_block:
|
||||
return {
|
||||
"success": False,
|
||||
"acquired": False,
|
||||
"reasons": comment_block,
|
||||
"permission_report": _permission_block_report("gitea.pr.comment"),
|
||||
}
|
||||
return _build_operation_gate_refusal(
|
||||
"gitea.pr.comment", comment_block, acquired=False
|
||||
)
|
||||
merge_block = _profile_operation_gate("gitea.pr.merge")
|
||||
if merge_block:
|
||||
return {
|
||||
@@ -19374,14 +19656,12 @@ def gitea_update_pr_branch_by_merge(
|
||||
# Permission: author branch push / PR mutation surface.
|
||||
push_block = _profile_operation_gate("gitea.branch.push")
|
||||
if push_block:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"mutation_allowed": False,
|
||||
"reasons": push_block,
|
||||
"permission_report": _permission_block_report("gitea.branch.push"),
|
||||
"role_kind": role,
|
||||
}
|
||||
return _build_operation_gate_refusal(
|
||||
"gitea.branch.push",
|
||||
push_block,
|
||||
mutation_allowed=False,
|
||||
role_kind=role,
|
||||
)
|
||||
|
||||
if role != "author":
|
||||
pre = pr_sync_status.assess_update_pr_branch_preflight(
|
||||
@@ -22332,6 +22612,193 @@ def gitea_workflow_dashboard(
|
||||
return payload
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_maintenance_drain_status(
|
||||
remote: str = "dadeschools",
|
||||
host: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
) -> dict:
|
||||
"""Read-only: current maintenance-drain state for a repository scope (#659 AC4).
|
||||
|
||||
Every session must be able to observe drain so it can stop creating new work
|
||||
and finish only allowlisted safety operations. Never mutates; never restarts.
|
||||
"""
|
||||
read_block = _profile_operation_gate("gitea.read")
|
||||
if read_block:
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": read_block,
|
||||
"permission_report": _permission_block_report("gitea.read"),
|
||||
}
|
||||
try:
|
||||
_h, o, r = _resolve(remote, host, org, repo)
|
||||
except ValueError as exc:
|
||||
return {"success": False, "read_only": True, "reasons": [str(exc)]}
|
||||
db, errs = _control_plane_db_or_error()
|
||||
if db is None:
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": errs or ["control-plane DB unavailable"],
|
||||
"maintenance_drain": maintenance_drain.status_payload(
|
||||
None, remote=remote, org=o, repo=r
|
||||
),
|
||||
}
|
||||
try:
|
||||
record = db.read_maintenance_drain(remote=remote, org=o, repo=r)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": [f"drain state unreadable: {_redact(str(exc))}"],
|
||||
}
|
||||
payload = maintenance_drain.status_payload(
|
||||
record, remote=remote, org=o, repo=r
|
||||
)
|
||||
return {"success": True, "read_only": True, "maintenance_drain": payload}
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_enter_maintenance_drain(
|
||||
reason: str = "",
|
||||
remote: str = "dadeschools",
|
||||
host: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
session_id: str | None = None,
|
||||
) -> dict:
|
||||
"""Enter graceful maintenance-drain mode for a repository scope (#659 AC1).
|
||||
|
||||
Stops new assignment and defers non-allowlisted mutations until exit. Requires
|
||||
``runtime.maintenance_drain`` (controller/lifecycle capability — not granted
|
||||
by ordinary Gitea author profiles). Audited in the control-plane event log.
|
||||
"""
|
||||
cap_block = _profile_operation_gate("runtime.maintenance_drain")
|
||||
if cap_block:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": cap_block,
|
||||
"permission_report": _permission_block_report(
|
||||
"runtime.maintenance_drain"
|
||||
),
|
||||
}
|
||||
try:
|
||||
_h, o, r = _resolve(remote, host, org, repo)
|
||||
except ValueError as exc:
|
||||
return {"success": False, "performed": False, "reasons": [str(exc)]}
|
||||
db, errs = _control_plane_db_or_error()
|
||||
if db is None:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": errs or ["control-plane DB unavailable"],
|
||||
}
|
||||
profile = get_profile() or {}
|
||||
try:
|
||||
result = db.set_maintenance_drain(
|
||||
remote=remote,
|
||||
org=o,
|
||||
repo=r,
|
||||
state=maintenance_drain.STATE_DRAINING,
|
||||
reason=reason or "operator-entered maintenance drain",
|
||||
requested_by=str(
|
||||
(profile.get("identity") or {}).get("username")
|
||||
or profile.get("expected_username")
|
||||
or ""
|
||||
),
|
||||
requested_by_profile=str(profile.get("profile_name") or ""),
|
||||
session_id=str(session_id or ""),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": [f"enter drain failed: {_redact(str(exc))}"],
|
||||
}
|
||||
record = result.get("record") or {}
|
||||
return {
|
||||
"success": True,
|
||||
"performed": True,
|
||||
"transitioned": bool(result.get("transitioned")),
|
||||
"state": result.get("state"),
|
||||
"prior_state": result.get("prior_state"),
|
||||
"drain_id": result.get("drain_id"),
|
||||
"maintenance_drain": maintenance_drain.status_payload(
|
||||
record, remote=remote, org=o, repo=r
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_exit_maintenance_drain(
|
||||
reason: str = "",
|
||||
remote: str = "dadeschools",
|
||||
host: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
session_id: str | None = None,
|
||||
) -> dict:
|
||||
"""Exit graceful maintenance-drain mode (#659 AC1). Restores assignment and mutations."""
|
||||
cap_block = _profile_operation_gate("runtime.maintenance_drain")
|
||||
if cap_block:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": cap_block,
|
||||
"permission_report": _permission_block_report(
|
||||
"runtime.maintenance_drain"
|
||||
),
|
||||
}
|
||||
try:
|
||||
_h, o, r = _resolve(remote, host, org, repo)
|
||||
except ValueError as exc:
|
||||
return {"success": False, "performed": False, "reasons": [str(exc)]}
|
||||
db, errs = _control_plane_db_or_error()
|
||||
if db is None:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": errs or ["control-plane DB unavailable"],
|
||||
}
|
||||
profile = get_profile() or {}
|
||||
try:
|
||||
result = db.set_maintenance_drain(
|
||||
remote=remote,
|
||||
org=o,
|
||||
repo=r,
|
||||
state=maintenance_drain.STATE_INACTIVE,
|
||||
reason=reason or "operator-exited maintenance drain",
|
||||
requested_by=str(
|
||||
(profile.get("identity") or {}).get("username")
|
||||
or profile.get("expected_username")
|
||||
or ""
|
||||
),
|
||||
requested_by_profile=str(profile.get("profile_name") or ""),
|
||||
session_id=str(session_id or ""),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": [f"exit drain failed: {_redact(str(exc))}"],
|
||||
}
|
||||
record = result.get("record") or {}
|
||||
return {
|
||||
"success": True,
|
||||
"performed": True,
|
||||
"transitioned": bool(result.get("transitioned")),
|
||||
"state": result.get("state"),
|
||||
"prior_state": result.get("prior_state"),
|
||||
"drain_id": result.get("drain_id"),
|
||||
"maintenance_drain": maintenance_drain.status_payload(
|
||||
record, remote=remote, org=o, repo=r
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_request_mcp_restart(
|
||||
remote: str = "dadeschools",
|
||||
@@ -22342,19 +22809,47 @@ def gitea_request_mcp_restart(
|
||||
request_override: bool = False,
|
||||
session_id: str | None = None,
|
||||
limit: int = 200,
|
||||
restart_class: str = "full_mcp_restart",
|
||||
target_session_id: str | None = None,
|
||||
target_role: str | None = None,
|
||||
target_connector: str | None = None,
|
||||
drain_proof_json: str | None = None,
|
||||
request_break_glass: bool = False,
|
||||
prior_recovery_attempts_json: str | None = None,
|
||||
) -> dict:
|
||||
"""Evaluate a proposed MCP restart and return an impact preview (#658).
|
||||
"""Evaluate a proposed MCP restart and return an impact preview (#658/#669).
|
||||
|
||||
Central restart coordinator: gathers live control-plane state (sessions,
|
||||
Central restart coordinator: resolves the requested restart class, gathers
|
||||
live control-plane state (sessions,
|
||||
leases/locks, in-flight issue/PR work, mutations, worktrees) and returns a
|
||||
blast-radius impact report with a ``safe`` / ``unsafe`` / ``override``
|
||||
verdict, so the console (#642/#652) and operators can see what a restart
|
||||
would disrupt *before* any concurrent LLM work is destroyed.
|
||||
|
||||
This tool is **dry-run and never restarts anything.** The mutative apply
|
||||
path is a separate child gated by a drain proof (non-goal here); calling
|
||||
with ``dry_run=False`` still performs no restart and reports that apply is
|
||||
not yet available.
|
||||
This tool **never restarts a process.** In dry-run (the default) it returns
|
||||
only the impact preview. With ``dry_run=False`` it enforces the #661 hard
|
||||
gate: the apply request must present a valid, unexpired, clean drain proof
|
||||
(``drain_proof_json``) or it is denied and a durable incident descriptor is
|
||||
returned under ``incident``. Break-glass is the only bypass and is honoured
|
||||
only when ``request_break_glass`` is set *and* the environment carries
|
||||
``GITEA_BREAKGLASS_RESTART_AUTHORIZATION``. Even an authorized gate performs
|
||||
no restart here; actual execution is a further child. The gate outcome is
|
||||
reported under ``apply_gate``.
|
||||
|
||||
``apply_authorized`` requires **both** authorizations to pass: the #661 drain
|
||||
gate *and* the #663 restart-class matrix (``allow_restart``). They are
|
||||
independent — the drain gate proves the blast radius was drained and knows
|
||||
nothing about whether this requester may request this class — so a class the
|
||||
matrix denied never reports an authorized apply. Break-glass bypasses the
|
||||
drain proof and, when env-authorized, the #669 attempt-log requirement for
|
||||
broad restarts; it never bypasses the class matrix. ``apply_gate`` carries
|
||||
``drain_gate_allow`` and ``restart_class_authorized`` so a denial is
|
||||
attributable to the authorization that produced it.
|
||||
|
||||
``prior_recovery_attempts_json`` (#669) is an optional JSON array of prior
|
||||
narrow recovery attempts ``{action, outcome, reason, ...}``. Rolling / full
|
||||
/ host restart classes require at least one *insufficient* narrower attempt
|
||||
unless break-glass is authorized.
|
||||
|
||||
Operator override authority is read from the process environment
|
||||
(``GITEA_OPERATOR_RESTART_OVERRIDE_AUTHORIZATION``), never self-asserted by
|
||||
@@ -22439,6 +22934,9 @@ def gitea_request_mcp_restart(
|
||||
|
||||
profile = get_profile()
|
||||
profile_name = (profile.get("profile_name") or "").strip() or "session"
|
||||
requester_role = (
|
||||
profile.get("role_kind") or profile.get("role") or ""
|
||||
).strip().lower()
|
||||
sid = (session_id or "").strip() or f"{profile_name}-{os.getpid()}"
|
||||
|
||||
# Override authority is read from the environment only — a worker session
|
||||
@@ -22448,6 +22946,39 @@ def gitea_request_mcp_restart(
|
||||
(os.environ.get("GITEA_OPERATOR_RESTART_OVERRIDE_AUTHORIZATION") or "").strip()
|
||||
)
|
||||
operator_override = bool(request_override and operator_authorized)
|
||||
controller_approved = bool(
|
||||
(
|
||||
os.environ.get("GITEA_CONTROLLER_RESTART_APPROVAL_AUTHORIZATION")
|
||||
or ""
|
||||
).strip()
|
||||
)
|
||||
requester_permissions = restart_coordinator.permissions_for_role(
|
||||
requester_role
|
||||
)
|
||||
|
||||
prior_recovery_attempts: list[dict] = []
|
||||
if prior_recovery_attempts_json:
|
||||
try:
|
||||
parsed_attempts = json.loads(prior_recovery_attempts_json)
|
||||
if isinstance(parsed_attempts, list):
|
||||
prior_recovery_attempts = [
|
||||
dict(a) for a in parsed_attempts if isinstance(a, dict)
|
||||
]
|
||||
else:
|
||||
incomplete_reasons.append(
|
||||
"prior_recovery_attempts_json must be a JSON array (#669)"
|
||||
)
|
||||
inventory_complete = False
|
||||
except (ValueError, TypeError) as exc:
|
||||
incomplete_reasons.append(
|
||||
f"invalid prior_recovery_attempts_json: {_redact(str(exc))}"
|
||||
)
|
||||
inventory_complete = False
|
||||
|
||||
break_glass_authorized = bool(
|
||||
(os.environ.get("GITEA_BREAKGLASS_RESTART_AUTHORIZATION") or "").strip()
|
||||
)
|
||||
break_glass = bool(request_break_glass and break_glass_authorized)
|
||||
|
||||
inventory = {
|
||||
"sessions": sessions,
|
||||
@@ -22455,6 +22986,7 @@ def gitea_request_mcp_restart(
|
||||
"terminal_lock": terminal_lock,
|
||||
"inventory_complete": inventory_complete,
|
||||
"incomplete_reasons": incomplete_reasons,
|
||||
"prior_recovery_attempts": prior_recovery_attempts,
|
||||
}
|
||||
|
||||
report = restart_coordinator.evaluate_restart_impact(
|
||||
@@ -22462,6 +22994,15 @@ def gitea_request_mcp_restart(
|
||||
operator_override=operator_override,
|
||||
requesting_session_id=sid,
|
||||
dry_run=True, # coordinator is always analysis-only (#658)
|
||||
restart_class=restart_class,
|
||||
requester_role=requester_role,
|
||||
requester_permissions=requester_permissions,
|
||||
controller_approved=controller_approved,
|
||||
operator_authorized=operator_authorized,
|
||||
target_session_id=target_session_id,
|
||||
target_role=target_role,
|
||||
target_connector=target_connector,
|
||||
break_glass=break_glass,
|
||||
)
|
||||
|
||||
payload = report.as_dict()
|
||||
@@ -22473,12 +23014,67 @@ def gitea_request_mcp_restart(
|
||||
payload["requesting_session_id"] = sid
|
||||
payload["operator_override_requested"] = bool(request_override)
|
||||
payload["operator_override_authorized"] = operator_authorized
|
||||
payload["controller_approval_authorized"] = controller_approved
|
||||
payload["requester_role"] = requester_role
|
||||
payload["requester_permissions"] = list(requester_permissions)
|
||||
# Actual restart execution remains a further child; this tool never restarts
|
||||
# a process. What #661 adds is the *hard gate*: an apply request (dry_run
|
||||
# False) must present a valid, unexpired, clean drain proof, or it is denied
|
||||
# and a durable incident is raised. Break-glass is the only bypass and its
|
||||
# authorization is read from the environment, never self-asserted.
|
||||
payload["apply_supported"] = False
|
||||
if not dry_run:
|
||||
payload["reasons"] = list(payload.get("reasons") or []) + [
|
||||
"apply requested but not supported: sanctioned restart apply is "
|
||||
"gated by a drain proof (separate child); no restart performed (#658)"
|
||||
]
|
||||
proof_obj: dict | None = None
|
||||
proof_parse_error: str | None = None
|
||||
if drain_proof_json:
|
||||
try:
|
||||
parsed = json.loads(drain_proof_json)
|
||||
proof_obj = parsed if isinstance(parsed, dict) else None
|
||||
if proof_obj is None:
|
||||
proof_parse_error = "drain_proof_json is not a JSON object"
|
||||
except (ValueError, TypeError) as exc:
|
||||
proof_parse_error = f"invalid drain_proof_json: {_redact(str(exc))}"
|
||||
|
||||
expected_fp = drain_proof.impact_fingerprint(report.as_dict())
|
||||
gate = drain_proof.gate_apply_restart(
|
||||
proof=proof_obj,
|
||||
break_glass=break_glass,
|
||||
expected_impact_fingerprint=expected_fp,
|
||||
requesting_session_id=sid,
|
||||
)
|
||||
gate_payload = gate.as_dict()
|
||||
if proof_parse_error and not break_glass:
|
||||
gate_payload["reasons"] = [proof_parse_error] + list(
|
||||
gate_payload.get("reasons") or []
|
||||
)
|
||||
# The #663 restart-class matrix and the #661 drain gate are two
|
||||
# independent authorizations, and an apply requires BOTH. ``gate.allow``
|
||||
# proves only that the blast radius was drained — or that break-glass
|
||||
# was authorized — and knows nothing about whether this requester may
|
||||
# request this class at all. Conjoining them keeps a class the matrix
|
||||
# denied from ever reporting an authorized apply, and keeps break-glass
|
||||
# scoped to what it is for: bypassing the drain proof, never the
|
||||
# least-privilege class matrix.
|
||||
restart_class_authorized = bool(report.allow_restart)
|
||||
gate_payload["drain_gate_allow"] = bool(gate.allow)
|
||||
gate_payload["restart_class_authorized"] = restart_class_authorized
|
||||
if not restart_class_authorized:
|
||||
gate_payload["reasons"] = list(gate_payload.get("reasons") or []) + [
|
||||
"restart class authorization denied; apply denied regardless of "
|
||||
"drain proof or break-glass (fail closed, #663)",
|
||||
*(report.authorization_reasons or []),
|
||||
]
|
||||
payload["apply_gate"] = gate_payload
|
||||
payload["apply_authorized"] = bool(gate.allow and restart_class_authorized)
|
||||
payload["break_glass_requested"] = bool(request_break_glass)
|
||||
payload["break_glass_authorized"] = break_glass_authorized
|
||||
# Even an authorized gate performs no restart here: execution is a later
|
||||
# child. The gate proves the apply path *would* be permitted.
|
||||
payload["reasons"] = list(payload.get("reasons") or []) + list(
|
||||
gate_payload.get("reasons") or []
|
||||
)
|
||||
if not gate.allow and gate.incident is not None:
|
||||
payload["incident"] = gate.incident
|
||||
return payload
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,281 @@
|
||||
"""Graceful MCP maintenance-drain mode (#659).
|
||||
|
||||
Drain is the visible, capability-gated state that lets an operator stop new
|
||||
work and quiesce mutations *before* a restart, instead of cutting sessions off
|
||||
mid-mutation. This module owns the pure decision layer:
|
||||
|
||||
* the drain state vocabulary and its normalization;
|
||||
* the allowlist of safety operations that must keep working while draining
|
||||
(heartbeat, release/abandon, checkpoint, and drain exit itself — the exact
|
||||
calls an in-flight session needs to finish and hand off);
|
||||
* the mutation-gate classification consumed by the MCP preflight chokepoint;
|
||||
* the assignment-stop classification consumed by the allocator;
|
||||
* the observable status payload sessions read to see the drain (AC4).
|
||||
|
||||
Durable state lives in the control-plane DB (``maintenance_drain`` table);
|
||||
enforcement lives at the existing chokepoints. Nothing here performs I/O, so
|
||||
both callers can share one decision without importing each other.
|
||||
|
||||
Scope note: the machine-verifiable *drain proof* and the restart gate that
|
||||
consumes it are #661's scope, not this module's. Drain here stops assignment
|
||||
and mutation and makes the state observable; it never authorizes a restart.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Mapping
|
||||
|
||||
# ── State vocabulary ──────────────────────────────────────────────────────────
|
||||
|
||||
STATE_INACTIVE = "inactive"
|
||||
STATE_DRAINING = "draining"
|
||||
DRAIN_STATES = frozenset({STATE_INACTIVE, STATE_DRAINING})
|
||||
|
||||
# Typed blocker code surfaced to clients (never a bare string at call sites).
|
||||
BLOCKER_DRAIN_ACTIVE = "maintenance_drain_active"
|
||||
|
||||
# Reason code for the allocator's assignment stop.
|
||||
REASON_ASSIGNMENT_STOPPED = "maintenance_drain_assignment_stopped"
|
||||
|
||||
DRAIN_SCHEMA_VERSION = 6
|
||||
|
||||
|
||||
class MaintenanceDrainError(RuntimeError):
|
||||
"""Raised when a mutation is refused because drain is active (fail closed)."""
|
||||
|
||||
def __init__(self, message: str, *, decision: Mapping[str, Any] | None = None):
|
||||
super().__init__(message)
|
||||
self.decision = dict(decision or {})
|
||||
self.reason_code = BLOCKER_DRAIN_ACTIVE
|
||||
|
||||
|
||||
# ── Safety allowlist ──────────────────────────────────────────────────────────
|
||||
|
||||
# Mutations that stay permitted while draining. Every entry is a *quiesce*
|
||||
# operation: it either proves an in-flight task is still alive, hands its claim
|
||||
# back, records the durable state a restart needs, or ends the drain. Nothing
|
||||
# that creates new work, new branches, new PRs, or new review/merge verdicts is
|
||||
# on this list — that is the whole point of the drain.
|
||||
ALLOWLISTED_DRAIN_TASKS: frozenset[str] = frozenset(
|
||||
{
|
||||
# Liveness of work already in flight.
|
||||
"heartbeat_issue_lock",
|
||||
"heartbeat_reviewer_pr_lease",
|
||||
"post_heartbeat",
|
||||
# Handing claims back so nothing is stranded across the restart.
|
||||
"release_workflow_lease",
|
||||
"release_reviewer_pr_lease",
|
||||
"release_merger_pr_lease",
|
||||
"abandon_workflow_lease",
|
||||
# Durable recovery state (#660) must be writable *during* drain.
|
||||
"write_session_checkpoint",
|
||||
"checkpoint_session",
|
||||
# The drain controls themselves — exit must never be self-blocked.
|
||||
"enter_maintenance_drain",
|
||||
"exit_maintenance_drain",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def normalize_task(task: str | None) -> str:
|
||||
"""Normalize a task name, tolerating the ``gitea_`` tool-name prefix."""
|
||||
name = str(task or "").strip()
|
||||
if name.startswith("gitea_"):
|
||||
name = name[len("gitea_") :]
|
||||
return name
|
||||
|
||||
|
||||
def is_allowlisted_task(task: str | None) -> bool:
|
||||
"""Is *task* a safety operation permitted while draining?"""
|
||||
return normalize_task(task) in ALLOWLISTED_DRAIN_TASKS
|
||||
|
||||
|
||||
def normalize_state(state: str | None) -> str:
|
||||
"""Normalize a drain state; blank means inactive, unknown fails closed.
|
||||
|
||||
Blank normalizes to ``inactive`` (no drain record = not draining), but an
|
||||
unrecognized non-blank value raises: silently treating ``"drainig"`` as
|
||||
inactive would disable the gate.
|
||||
"""
|
||||
value = str(state or "").strip().lower()
|
||||
if not value:
|
||||
return STATE_INACTIVE
|
||||
if value not in DRAIN_STATES:
|
||||
raise MaintenanceDrainError(
|
||||
f"unknown maintenance-drain state {value!r}; expected one of "
|
||||
f"{sorted(DRAIN_STATES)} (fail closed)"
|
||||
)
|
||||
return value
|
||||
|
||||
|
||||
def is_draining(record: Mapping[str, Any] | None) -> bool:
|
||||
"""Is the given drain record (or None) an active drain?"""
|
||||
if not record:
|
||||
return False
|
||||
return normalize_state(record.get("state")) == STATE_DRAINING
|
||||
|
||||
|
||||
# ── Decisions ─────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def classify_mutation(
|
||||
task: str | None,
|
||||
record: Mapping[str, Any] | None,
|
||||
) -> dict[str, Any]:
|
||||
"""Decide whether *task* may mutate under the given drain record.
|
||||
|
||||
Returns a decision dict with ``allowed``/``deferred`` and, when refused, a
|
||||
typed ``reason_code`` plus the one exact next action the caller may take.
|
||||
Deferred (not failed): the operation is legal again after drain exits, so
|
||||
the caller is told to wait rather than to retry a different way.
|
||||
"""
|
||||
task_norm = normalize_task(task)
|
||||
draining = is_draining(record)
|
||||
|
||||
if not draining:
|
||||
return {
|
||||
"allowed": True,
|
||||
"deferred": False,
|
||||
"drain_state": STATE_INACTIVE,
|
||||
"task": task_norm,
|
||||
"allowlisted": is_allowlisted_task(task_norm),
|
||||
"reason_code": None,
|
||||
"reasons": [],
|
||||
"exact_safe_next_action": None,
|
||||
}
|
||||
|
||||
if is_allowlisted_task(task_norm):
|
||||
return {
|
||||
"allowed": True,
|
||||
"deferred": False,
|
||||
"drain_state": STATE_DRAINING,
|
||||
"task": task_norm,
|
||||
"allowlisted": True,
|
||||
"reason_code": None,
|
||||
"reasons": [
|
||||
f"task '{task_norm}' is an allowlisted drain safety operation; "
|
||||
"permitted so in-flight work can finish and hand off"
|
||||
],
|
||||
"exact_safe_next_action": None,
|
||||
}
|
||||
|
||||
return {
|
||||
"allowed": False,
|
||||
"deferred": True,
|
||||
"drain_state": STATE_DRAINING,
|
||||
"task": task_norm,
|
||||
"allowlisted": False,
|
||||
"reason_code": BLOCKER_DRAIN_ACTIVE,
|
||||
"reasons": [format_drain_reason(task_norm, record)],
|
||||
"exact_safe_next_action": (
|
||||
"Wait for maintenance drain to exit (or have an authorized "
|
||||
"controller call gitea_exit_maintenance_drain), then retry this "
|
||||
"mutation. Reads and gitea_maintenance_drain_status stay available."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def classify_assignment(record: Mapping[str, Any] | None) -> dict[str, Any]:
|
||||
"""Decide whether the allocator may assign new work (AC2)."""
|
||||
if not is_draining(record):
|
||||
return {
|
||||
"assignment_allowed": True,
|
||||
"drain_state": STATE_INACTIVE,
|
||||
"reason_code": None,
|
||||
"reasons": [],
|
||||
}
|
||||
return {
|
||||
"assignment_allowed": False,
|
||||
"drain_state": STATE_DRAINING,
|
||||
"reason_code": REASON_ASSIGNMENT_STOPPED,
|
||||
"reasons": [
|
||||
"maintenance drain is active: new work assignment is stopped and "
|
||||
"no lease was created (fail closed, #659)" + _scope_suffix(record)
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def format_drain_reason(task: str | None, record: Mapping[str, Any] | None) -> str:
|
||||
"""Human-readable refusal line for a drained mutation."""
|
||||
task_norm = normalize_task(task) or "(unnamed task)"
|
||||
return (
|
||||
f"maintenance drain is active: mutation '{task_norm}' is deferred; only "
|
||||
"allowlisted drain safety operations "
|
||||
f"({', '.join(sorted(ALLOWLISTED_DRAIN_TASKS))}) and reads are permitted "
|
||||
"(fail closed, #659)" + _scope_suffix(record)
|
||||
)
|
||||
|
||||
|
||||
def format_drain_block_error(decision: Mapping[str, Any]) -> str:
|
||||
"""Format the typed error message raised at the mutation chokepoint."""
|
||||
reasons = list(decision.get("reasons") or [])
|
||||
head = reasons[0] if reasons else "maintenance drain is active (fail closed)"
|
||||
action = decision.get("exact_safe_next_action")
|
||||
return f"{head}. Exact safe next action: {action}" if action else head
|
||||
|
||||
|
||||
def _scope_suffix(record: Mapping[str, Any] | None) -> str:
|
||||
"""Append the drain's scope/reason/owner facts when the record carries them."""
|
||||
if not record:
|
||||
return ""
|
||||
bits: list[str] = []
|
||||
scope = "/".join(
|
||||
str(record.get(key) or "") for key in ("remote", "org", "repo")
|
||||
).strip("/")
|
||||
if scope:
|
||||
bits.append(f"scope {scope}")
|
||||
if record.get("reason"):
|
||||
bits.append(f"reason: {record['reason']}")
|
||||
if record.get("requested_by"):
|
||||
bits.append(f"entered by {record['requested_by']}")
|
||||
if record.get("entered_at"):
|
||||
bits.append(f"at {record['entered_at']}")
|
||||
return f" ({'; '.join(bits)})" if bits else ""
|
||||
|
||||
|
||||
# ── Observability (AC4) ───────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def status_payload(
|
||||
record: Mapping[str, Any] | None,
|
||||
*,
|
||||
remote: str = "",
|
||||
org: str = "",
|
||||
repo: str = "",
|
||||
) -> dict[str, Any]:
|
||||
"""Build the session-observable drain status payload.
|
||||
|
||||
Always answers, including when no drain record exists: an absent record is
|
||||
a definitive "not draining", not an unknown.
|
||||
"""
|
||||
draining = is_draining(record)
|
||||
rec: Mapping[str, Any] = record or {}
|
||||
return {
|
||||
"drain_state": STATE_DRAINING if draining else STATE_INACTIVE,
|
||||
"draining": draining,
|
||||
"remote": str(rec.get("remote") or "") or remote,
|
||||
"org": str(rec.get("org") or "") or org,
|
||||
"repo": str(rec.get("repo") or "") or repo,
|
||||
"reason": str(rec.get("reason") or ""),
|
||||
"requested_by": str(rec.get("requested_by") or ""),
|
||||
"requested_by_profile": str(rec.get("requested_by_profile") or ""),
|
||||
"session_id": str(rec.get("session_id") or ""),
|
||||
"entered_at": str(rec.get("entered_at") or ""),
|
||||
"exited_at": str(rec.get("exited_at") or ""),
|
||||
"assignment_stopped": draining,
|
||||
"mutations_deferred": draining,
|
||||
"allowlisted_tasks": sorted(ALLOWLISTED_DRAIN_TASKS),
|
||||
"reads_permitted": True,
|
||||
"record_present": bool(record),
|
||||
"schema_version": DRAIN_SCHEMA_VERSION,
|
||||
"drain_proof_scope": (
|
||||
"drain proof and the restart gate that consumes it are #661 scope; "
|
||||
"this status never authorizes a restart"
|
||||
),
|
||||
"safe_next_action": (
|
||||
"Wait for drain to exit before retrying deferred mutations; "
|
||||
"allowlisted safety operations and reads remain available."
|
||||
if draining
|
||||
else "None; maintenance drain is not active."
|
||||
),
|
||||
}
|
||||
@@ -0,0 +1,583 @@
|
||||
"""Scoped MCP recovery playbook (#669).
|
||||
|
||||
Operational recovery must prefer the *narrowest* action that can fix the
|
||||
symptom. Full MCP / host restarts are last-resort rungs on a documented
|
||||
ladder; the coordinator refuses those rungs unless a prior attempt log
|
||||
shows narrower recoveries already failed (or break-glass is authorized).
|
||||
|
||||
This module is pure classification and recommendation:
|
||||
|
||||
* No network, filesystem, or process I/O.
|
||||
* Never restarts anything.
|
||||
* Narrow recovery *execution* is delegated to existing tools/docs (linked
|
||||
per rung) — the playbook records which rung to try next and whether
|
||||
escalation to a broad restart is allowed.
|
||||
|
||||
Design lineage: umbrella #655, class matrix #663, coordinator #658,
|
||||
auto-reconnect #584, stale-runtime #610, contamination #630, audit #665.
|
||||
Vision #652 / roadmap #653.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from enum import Enum
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
PLAYBOOK_VERSION = "1.0.0-issue-669"
|
||||
|
||||
# Attempt outcomes that count as "tried and insufficient" for escalation.
|
||||
INSUFFICIENT_OUTCOMES = frozenset(
|
||||
{
|
||||
"failed",
|
||||
"insufficient",
|
||||
"denied",
|
||||
"unresolved",
|
||||
"timeout",
|
||||
"error",
|
||||
}
|
||||
)
|
||||
|
||||
# Break-glass / operator override still records that the ladder was skipped.
|
||||
OUTCOME_BREAK_GLASS = "break_glass"
|
||||
OUTCOME_SUCCESS = "success"
|
||||
OUTCOME_SKIPPED = "skipped"
|
||||
|
||||
|
||||
class RecoveryAction(str, Enum):
|
||||
"""Ordered recovery ladder (narrow → broad)."""
|
||||
|
||||
CLIENT_RECONNECT = "client_reconnect"
|
||||
CAPABILITY_REFRESH = "capability_refresh"
|
||||
SESSION_RECONNECT = "session_reconnect"
|
||||
CONFIGURATION_RELOAD = "configuration_reload"
|
||||
LEASE_RECOVERY = "lease_recovery"
|
||||
WORKER_RESTART = "worker_restart"
|
||||
ROLE_RUNTIME_RESTART = "role_runtime_restart"
|
||||
CONNECTOR_RESTART = "connector_restart"
|
||||
ROLLING_MCP_RESTART = "rolling_mcp_restart"
|
||||
FULL_MCP_RESTART = "full_mcp_restart"
|
||||
HOST_RESTART = "host_restart"
|
||||
|
||||
|
||||
# Classes that require a prior narrow-attempt log (unless break-glass).
|
||||
BROAD_RESTART_ACTIONS: frozenset[RecoveryAction] = frozenset(
|
||||
{
|
||||
RecoveryAction.ROLLING_MCP_RESTART,
|
||||
RecoveryAction.FULL_MCP_RESTART,
|
||||
RecoveryAction.HOST_RESTART,
|
||||
}
|
||||
)
|
||||
|
||||
# Map #663 restart_class strings onto playbook actions.
|
||||
RESTART_CLASS_TO_ACTION: dict[str, RecoveryAction] = {
|
||||
"client_reconnect": RecoveryAction.CLIENT_RECONNECT,
|
||||
"session_reconnect": RecoveryAction.SESSION_RECONNECT,
|
||||
"configuration_reload": RecoveryAction.CONFIGURATION_RELOAD,
|
||||
"worker_restart": RecoveryAction.WORKER_RESTART,
|
||||
"role_runtime_restart": RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||
"connector_restart": RecoveryAction.CONNECTOR_RESTART,
|
||||
"rolling_mcp_restart": RecoveryAction.ROLLING_MCP_RESTART,
|
||||
"full_mcp_restart": RecoveryAction.FULL_MCP_RESTART,
|
||||
"host_restart": RecoveryAction.HOST_RESTART,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RecoveryRung:
|
||||
"""One rung on the recovery ladder."""
|
||||
|
||||
action: RecoveryAction
|
||||
rank: int
|
||||
summary: str
|
||||
# Existing implementation or explicit delegation target.
|
||||
implementation: str
|
||||
issue_links: tuple[str, ...]
|
||||
self_service: bool
|
||||
# Restart-class permission when this rung is requested via coordinator.
|
||||
restart_class: str | None = None
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"action": self.action.value,
|
||||
"rank": self.rank,
|
||||
"summary": self.summary,
|
||||
"implementation": self.implementation,
|
||||
"issue_links": list(self.issue_links),
|
||||
"self_service": self.self_service,
|
||||
"restart_class": self.restart_class,
|
||||
}
|
||||
|
||||
|
||||
# Canonical ladder. Rank 0 is narrowest.
|
||||
RECOVERY_LADDER: tuple[RecoveryRung, ...] = (
|
||||
RecoveryRung(
|
||||
RecoveryAction.CLIENT_RECONNECT,
|
||||
0,
|
||||
"Reconnect the IDE/client MCP transport (EOF / transport flap).",
|
||||
"Host auto-reconnect or explicit client reconnect; "
|
||||
"docs/mcp-namespace-eof-recovery.md",
|
||||
("#584", "#655"),
|
||||
True,
|
||||
"client_reconnect",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.CAPABILITY_REFRESH,
|
||||
1,
|
||||
"Re-resolve task capability and clear stale permission context.",
|
||||
"Delegated: gitea_resolve_task_capability + gitea_whoami "
|
||||
"(no process change).",
|
||||
("#610", "#685", "#655"),
|
||||
True,
|
||||
None,
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.SESSION_RECONNECT,
|
||||
2,
|
||||
"Rebind identity, workspace, and namespace for one session.",
|
||||
"Delegated: gitea_get_runtime_context + explicit worktree_path "
|
||||
"rebind (#618); docs/mcp-namespace-health.md",
|
||||
("#543", "#618", "#655"),
|
||||
True,
|
||||
"session_reconnect",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.CONFIGURATION_RELOAD,
|
||||
3,
|
||||
"Gracefully reload configuration without replacing the daemon.",
|
||||
"restart_coordinator class configuration_reload; console "
|
||||
"system.reload_namespace (#642).",
|
||||
("#642", "#663", "#655"),
|
||||
False,
|
||||
"configuration_reload",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.LEASE_RECOVERY,
|
||||
4,
|
||||
"Recover or rebind stale leases/locks without a process restart.",
|
||||
"Delegated: issue lock recovery / lease lifecycle paths "
|
||||
"(#702, #753, #790).",
|
||||
("#702", "#753", "#790", "#655"),
|
||||
False,
|
||||
None,
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.WORKER_RESTART,
|
||||
5,
|
||||
"Restart one worker after its own lease and mutation scope drains.",
|
||||
"restart_coordinator class worker_restart (#663).",
|
||||
("#663", "#655"),
|
||||
False,
|
||||
"worker_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||
6,
|
||||
"Restart one role runtime and re-probe that namespace only.",
|
||||
"restart_coordinator class role_runtime_restart; console "
|
||||
"system.restart_namespace (#642).",
|
||||
("#642", "#663", "#655"),
|
||||
False,
|
||||
"role_runtime_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.CONNECTOR_RESTART,
|
||||
7,
|
||||
"Restart one connector while unrelated runtimes stay available.",
|
||||
"restart_coordinator class connector_restart (#663).",
|
||||
("#663", "#655"),
|
||||
False,
|
||||
"connector_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.ROLLING_MCP_RESTART,
|
||||
8,
|
||||
"Drain/restart/verify one instance at a time (HA path).",
|
||||
"restart_coordinator class rolling_mcp_restart; design #668.",
|
||||
("#668", "#663", "#655"),
|
||||
False,
|
||||
"rolling_mcp_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.FULL_MCP_RESTART,
|
||||
9,
|
||||
"Full stable-control MCP process restart after verified full drain.",
|
||||
"restart_coordinator class full_mcp_restart; requires attempt log "
|
||||
"unless break-glass (#669).",
|
||||
("#658", "#661", "#663", "#669", "#655"),
|
||||
False,
|
||||
"full_mcp_restart",
|
||||
),
|
||||
RecoveryRung(
|
||||
RecoveryAction.HOST_RESTART,
|
||||
10,
|
||||
"Host/infrastructure restart — broadest last-resort action.",
|
||||
"restart_coordinator class host_restart; operator-owned.",
|
||||
("#663", "#669", "#655"),
|
||||
False,
|
||||
"host_restart",
|
||||
),
|
||||
)
|
||||
|
||||
_LADDER_BY_ACTION: dict[RecoveryAction, RecoveryRung] = {
|
||||
rung.action: rung for rung in RECOVERY_LADDER
|
||||
}
|
||||
|
||||
# Symptom tokens → preferred first rung (decision tree, #663 lineage).
|
||||
SYMPTOM_TO_FIRST_ACTION: dict[str, RecoveryAction] = {
|
||||
"transport_eof": RecoveryAction.CLIENT_RECONNECT,
|
||||
"client_closing_eof": RecoveryAction.CLIENT_RECONNECT,
|
||||
"transport_flap": RecoveryAction.CLIENT_RECONNECT,
|
||||
"namespace_disconnected": RecoveryAction.CLIENT_RECONNECT,
|
||||
"stale_capability": RecoveryAction.CAPABILITY_REFRESH,
|
||||
"permission_stale": RecoveryAction.CAPABILITY_REFRESH,
|
||||
"runtime_reconnect_required": RecoveryAction.CAPABILITY_REFRESH,
|
||||
"stale_runtime": RecoveryAction.SESSION_RECONNECT,
|
||||
"worktree_unbound": RecoveryAction.SESSION_RECONNECT,
|
||||
"namespace_unhealthy": RecoveryAction.SESSION_RECONNECT,
|
||||
"config_drift": RecoveryAction.CONFIGURATION_RELOAD,
|
||||
"profile_misbound": RecoveryAction.CONFIGURATION_RELOAD,
|
||||
"stale_lease": RecoveryAction.LEASE_RECOVERY,
|
||||
"dead_pid_lock": RecoveryAction.LEASE_RECOVERY,
|
||||
"orphan_worktree": RecoveryAction.LEASE_RECOVERY,
|
||||
"single_worker_stuck": RecoveryAction.WORKER_RESTART,
|
||||
"role_runtime_dead": RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||
"connector_dead": RecoveryAction.CONNECTOR_RESTART,
|
||||
"ha_instance_unhealthy": RecoveryAction.ROLLING_MCP_RESTART,
|
||||
"daemon_corrupt": RecoveryAction.FULL_MCP_RESTART,
|
||||
"full_process_deadlock": RecoveryAction.FULL_MCP_RESTART,
|
||||
"host_unresponsive": RecoveryAction.HOST_RESTART,
|
||||
}
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def resolve_action(value: RecoveryAction | str) -> RecoveryAction:
|
||||
"""Resolve a recovery action or fail closed for unknown values."""
|
||||
if isinstance(value, RecoveryAction):
|
||||
return value
|
||||
text = str(value or "").strip()
|
||||
# Accept #663 restart_class aliases.
|
||||
if text in RESTART_CLASS_TO_ACTION:
|
||||
return RESTART_CLASS_TO_ACTION[text]
|
||||
try:
|
||||
return RecoveryAction(text)
|
||||
except ValueError as exc:
|
||||
raise ValueError(
|
||||
f"unknown recovery action {value!r}; deny (fail closed, #669)"
|
||||
) from exc
|
||||
|
||||
|
||||
def ladder_rank(action: RecoveryAction | str) -> int:
|
||||
resolved = resolve_action(action)
|
||||
return _LADDER_BY_ACTION[resolved].rank
|
||||
|
||||
|
||||
def rung_for(action: RecoveryAction | str) -> RecoveryRung:
|
||||
return _LADDER_BY_ACTION[resolve_action(action)]
|
||||
|
||||
|
||||
def normalize_attempt(raw: Mapping[str, Any]) -> dict[str, Any] | None:
|
||||
"""Normalize one prior-recovery attempt record; return None if unusable."""
|
||||
if not isinstance(raw, Mapping):
|
||||
return None
|
||||
action_raw = raw.get("action") or raw.get("recovery_action") or raw.get(
|
||||
"restart_class"
|
||||
)
|
||||
if not action_raw:
|
||||
return None
|
||||
try:
|
||||
action = resolve_action(str(action_raw))
|
||||
except ValueError:
|
||||
return None
|
||||
outcome = str(
|
||||
raw.get("outcome") or raw.get("status") or raw.get("result") or ""
|
||||
).strip().lower()
|
||||
if not outcome:
|
||||
return None
|
||||
recorded_at = raw.get("recorded_at") or raw.get("at") or raw.get("timestamp")
|
||||
reason = str(raw.get("reason") or raw.get("detail") or "").strip()
|
||||
actor = str(raw.get("actor") or raw.get("session_id") or "").strip()
|
||||
return {
|
||||
"action": action.value,
|
||||
"outcome": outcome,
|
||||
"reason": reason,
|
||||
"actor": actor,
|
||||
"recorded_at": recorded_at,
|
||||
"rank": ladder_rank(action),
|
||||
"raw": dict(raw),
|
||||
}
|
||||
|
||||
|
||||
def normalize_attempt_log(
|
||||
attempts: Sequence[Mapping[str, Any]] | None,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Return usable attempt records in ladder order."""
|
||||
out: list[dict[str, Any]] = []
|
||||
for raw in attempts or ():
|
||||
norm = normalize_attempt(raw)
|
||||
if norm is not None:
|
||||
out.append(norm)
|
||||
out.sort(key=lambda a: (a["rank"], str(a.get("recorded_at") or "")))
|
||||
return out
|
||||
|
||||
|
||||
def narrower_insufficient_attempts(
|
||||
attempts: Sequence[Mapping[str, Any]] | None,
|
||||
*,
|
||||
requested: RecoveryAction | str,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Return prior attempts narrower than *requested* that were insufficient."""
|
||||
target_rank = ladder_rank(requested)
|
||||
usable = []
|
||||
for attempt in normalize_attempt_log(attempts):
|
||||
if attempt["rank"] >= target_rank:
|
||||
continue
|
||||
if attempt["outcome"] in INSUFFICIENT_OUTCOMES:
|
||||
usable.append(attempt)
|
||||
return usable
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EscalationAssessment:
|
||||
"""Whether a requested broad recovery may proceed given the attempt log."""
|
||||
|
||||
requested_action: str
|
||||
allowed: bool
|
||||
require_attempt_log: bool
|
||||
break_glass: bool
|
||||
reasons: list[str] = field(default_factory=list)
|
||||
qualifying_attempts: list[dict[str, Any]] = field(default_factory=list)
|
||||
recommended_next: list[dict[str, Any]] = field(default_factory=list)
|
||||
playbook_version: str = PLAYBOOK_VERSION
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"playbook_version": self.playbook_version,
|
||||
"requested_action": self.requested_action,
|
||||
"allowed": self.allowed,
|
||||
"require_attempt_log": self.require_attempt_log,
|
||||
"break_glass": self.break_glass,
|
||||
"reasons": list(self.reasons),
|
||||
"qualifying_attempts": list(self.qualifying_attempts),
|
||||
"recommended_next": list(self.recommended_next),
|
||||
}
|
||||
|
||||
|
||||
def assess_escalation(
|
||||
requested: RecoveryAction | str,
|
||||
*,
|
||||
prior_recovery_attempts: Sequence[Mapping[str, Any]] | None = None,
|
||||
break_glass: bool = False,
|
||||
) -> EscalationAssessment:
|
||||
"""Gate broad restarts on a prior narrow-attempt log (#669 AC3).
|
||||
|
||||
Narrow / mid-ladder actions do not require a prior attempt log.
|
||||
``full_mcp_restart``, ``host_restart``, and ``rolling_mcp_restart``
|
||||
require at least one *insufficient* narrower attempt unless
|
||||
``break_glass`` is true.
|
||||
"""
|
||||
action = resolve_action(requested)
|
||||
require_log = action in BROAD_RESTART_ACTIONS
|
||||
reasons: list[str] = []
|
||||
qualifying = narrower_insufficient_attempts(
|
||||
prior_recovery_attempts, requested=action
|
||||
)
|
||||
|
||||
if not require_log:
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=True,
|
||||
require_attempt_log=False,
|
||||
break_glass=bool(break_glass),
|
||||
reasons=["narrow recovery; attempt log not required"],
|
||||
qualifying_attempts=qualifying,
|
||||
recommended_next=[],
|
||||
)
|
||||
|
||||
if break_glass:
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=True,
|
||||
require_attempt_log=True,
|
||||
break_glass=True,
|
||||
reasons=[
|
||||
"break-glass authorized; broad restart permitted without "
|
||||
"narrow-attempt log (#669)"
|
||||
],
|
||||
qualifying_attempts=qualifying,
|
||||
recommended_next=[],
|
||||
)
|
||||
|
||||
if qualifying:
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=True,
|
||||
require_attempt_log=True,
|
||||
break_glass=False,
|
||||
reasons=[
|
||||
f"{len(qualifying)} narrower recovery attempt(s) recorded as "
|
||||
"insufficient; escalation permitted"
|
||||
],
|
||||
qualifying_attempts=qualifying,
|
||||
recommended_next=[],
|
||||
)
|
||||
|
||||
# Deny: recommend the next untried narrow rung(s).
|
||||
recommended = recommend_actions(
|
||||
symptoms=(),
|
||||
prior_recovery_attempts=prior_recovery_attempts,
|
||||
max_actions=3,
|
||||
)
|
||||
reasons.append(
|
||||
f"{action.value} requires a prior attempt log of insufficient "
|
||||
"narrower recoveries (or break-glass); none found — deny (fail "
|
||||
"closed, #669)"
|
||||
)
|
||||
return EscalationAssessment(
|
||||
requested_action=action.value,
|
||||
allowed=False,
|
||||
require_attempt_log=True,
|
||||
break_glass=False,
|
||||
reasons=reasons,
|
||||
qualifying_attempts=[],
|
||||
recommended_next=recommended.get("recommended_actions") or [],
|
||||
)
|
||||
|
||||
|
||||
def recommend_actions(
|
||||
*,
|
||||
symptoms: Sequence[str] = (),
|
||||
prior_recovery_attempts: Sequence[Mapping[str, Any]] | None = None,
|
||||
max_actions: int = 5,
|
||||
) -> dict[str, Any]:
|
||||
"""Return ordered recommended recovery actions for the given symptoms.
|
||||
|
||||
Soft mode (rollout): recommendations only — callers decide whether to
|
||||
hard-gate. Hard mode for broad restarts is :func:`assess_escalation`.
|
||||
"""
|
||||
attempted_success = {
|
||||
a["action"]
|
||||
for a in normalize_attempt_log(prior_recovery_attempts)
|
||||
if a["outcome"] == OUTCOME_SUCCESS
|
||||
}
|
||||
attempted_any = {
|
||||
a["action"] for a in normalize_attempt_log(prior_recovery_attempts)
|
||||
}
|
||||
|
||||
first_actions: list[RecoveryAction] = []
|
||||
for symptom in symptoms:
|
||||
key = str(symptom or "").strip().lower().replace(" ", "_").replace("-", "_")
|
||||
mapped = SYMPTOM_TO_FIRST_ACTION.get(key)
|
||||
if mapped is not None and mapped not in first_actions:
|
||||
first_actions.append(mapped)
|
||||
|
||||
# Default entry: client reconnect then walk the ladder.
|
||||
if not first_actions:
|
||||
first_actions = [RecoveryAction.CLIENT_RECONNECT]
|
||||
|
||||
recommended: list[dict[str, Any]] = []
|
||||
seen: set[str] = set()
|
||||
min_rank = min(ladder_rank(a) for a in first_actions)
|
||||
|
||||
for rung in RECOVERY_LADDER:
|
||||
if rung.rank < min_rank:
|
||||
continue
|
||||
if rung.action.value in attempted_success:
|
||||
continue
|
||||
if rung.action.value in seen:
|
||||
continue
|
||||
# Prefer rungs not yet attempted; still list previously-failed ones
|
||||
# only if nothing else remains.
|
||||
entry = rung.as_dict()
|
||||
entry["already_attempted"] = rung.action.value in attempted_any
|
||||
recommended.append(entry)
|
||||
seen.add(rung.action.value)
|
||||
if len(recommended) >= max(1, int(max_actions)):
|
||||
break
|
||||
|
||||
return {
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
"symptoms": [str(s) for s in symptoms],
|
||||
"recommended_actions": recommended,
|
||||
"ladder": [r.as_dict() for r in RECOVERY_LADDER],
|
||||
"read_only": True,
|
||||
"hard_gate_note": (
|
||||
"Broad restarts (rolling/full/host) still require "
|
||||
"assess_escalation / coordinator attempt-log enforcement."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def build_attempt_record(
|
||||
action: RecoveryAction | str,
|
||||
*,
|
||||
outcome: str,
|
||||
reason: str = "",
|
||||
actor: str = "",
|
||||
recorded_at: str | None = None,
|
||||
extra: Mapping[str, Any] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build a durable-shaped attempt log entry for inventory/audit (#665)."""
|
||||
resolved = resolve_action(action)
|
||||
record = {
|
||||
"action": resolved.value,
|
||||
"outcome": str(outcome or "").strip().lower(),
|
||||
"reason": str(reason or "").strip(),
|
||||
"actor": str(actor or "").strip(),
|
||||
"recorded_at": recorded_at or _utc_now().isoformat(),
|
||||
"rank": ladder_rank(resolved),
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
}
|
||||
if extra:
|
||||
record["extra"] = dict(extra)
|
||||
return record
|
||||
|
||||
|
||||
def recovery_metrics(
|
||||
attempts: Sequence[Mapping[str, Any]] | None,
|
||||
) -> dict[str, Any]:
|
||||
"""Compute the fraction of recoveries that avoided full/host restart.
|
||||
|
||||
A recovery *episode* is approximated as one attempt with
|
||||
``outcome=success``. Successes on non-broad rungs count as avoided full
|
||||
restart; successes on full/host count as full-restart recoveries.
|
||||
"""
|
||||
norms = normalize_attempt_log(attempts)
|
||||
successes = [a for a in norms if a["outcome"] == OUTCOME_SUCCESS]
|
||||
broad_success = [
|
||||
a
|
||||
for a in successes
|
||||
if resolve_action(a["action"])
|
||||
in {RecoveryAction.FULL_MCP_RESTART, RecoveryAction.HOST_RESTART}
|
||||
]
|
||||
avoided = [a for a in successes if a not in broad_success]
|
||||
total = len(successes)
|
||||
fraction_avoided = (len(avoided) / total) if total else None
|
||||
return {
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
"attempts_total": len(norms),
|
||||
"successes_total": total,
|
||||
"successes_avoided_full_restart": len(avoided),
|
||||
"successes_full_or_host_restart": len(broad_success),
|
||||
"fraction_avoided_full_restart": fraction_avoided,
|
||||
"insufficient_attempts": sum(
|
||||
1 for a in norms if a["outcome"] in INSUFFICIENT_OUTCOMES
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def ladder_document() -> dict[str, Any]:
|
||||
"""Machine-readable ladder for docs/tools inventory."""
|
||||
return {
|
||||
"playbook_version": PLAYBOOK_VERSION,
|
||||
"parent_issues": ["#655", "#652", "#653"],
|
||||
"enforcement_issue": "#669",
|
||||
"ladder": [r.as_dict() for r in RECOVERY_LADDER],
|
||||
"broad_restart_actions": [a.value for a in sorted(BROAD_RESTART_ACTIONS, key=lambda x: x.value)],
|
||||
"insufficient_outcomes": sorted(INSUFFICIENT_OUTCOMES),
|
||||
"symptom_map": {k: v.value for k, v in sorted(SYMPTOM_TO_FIRST_ACTION.items())},
|
||||
}
|
||||
+415
-15
@@ -1,4 +1,4 @@
|
||||
"""MCP restart coordinator and impact analysis (#658).
|
||||
"""MCP restart coordinator and impact analysis (#658 / #669).
|
||||
|
||||
Before any sanctioned MCP restart, a central coordinator must evaluate the
|
||||
live control-plane state — active sessions, leases/locks, in-flight issue/PR
|
||||
@@ -16,6 +16,9 @@ Design rules (mirrors the read-only posture of ``workflow_dashboard`` /
|
||||
a mutative apply path is a later child gated by a drain proof (non-goal here).
|
||||
* **Fail closed.** If the inventory is not explicitly complete, the verdict is
|
||||
``unsafe`` / deny — an incomplete evaluation must never green-light a restart.
|
||||
* **Narrow-first (#669).** Broad classes (rolling / full / host) require a
|
||||
prior attempt log of insufficient narrower recoveries unless break-glass is
|
||||
authorized. See :mod:`recovery_playbook`.
|
||||
* **No secrets.** Session ids, pids, and profiles are operational metadata, not
|
||||
credentials; nothing secret flows through this module.
|
||||
|
||||
@@ -28,11 +31,13 @@ from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from enum import Enum
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
import lease_lifecycle
|
||||
import recovery_playbook
|
||||
|
||||
COORDINATOR_VERSION = "1.0.0-issue-658"
|
||||
COORDINATOR_VERSION = "1.2.0-issue-669"
|
||||
|
||||
# Restart verdicts. Exactly the three the acceptance criteria name.
|
||||
VERDICT_SAFE = "safe"
|
||||
@@ -54,6 +59,194 @@ LEASE_FRESHNESS_LIVE = "active"
|
||||
DEFAULT_SESSION_HEARTBEAT_STALE_SECONDS = 900
|
||||
|
||||
|
||||
class RestartClass(str, Enum):
|
||||
"""The only restart/recovery classes accepted by the coordinator."""
|
||||
|
||||
CLIENT_RECONNECT = "client_reconnect"
|
||||
SESSION_RECONNECT = "session_reconnect"
|
||||
WORKER_RESTART = "worker_restart"
|
||||
ROLE_RUNTIME_RESTART = "role_runtime_restart"
|
||||
CONNECTOR_RESTART = "connector_restart"
|
||||
CONFIGURATION_RELOAD = "configuration_reload"
|
||||
ROLLING_MCP_RESTART = "rolling_mcp_restart"
|
||||
FULL_MCP_RESTART = "full_mcp_restart"
|
||||
HOST_RESTART = "host_restart"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RestartClassPolicy:
|
||||
"""Least-privilege policy for one :class:`RestartClass`."""
|
||||
|
||||
restart_class: RestartClass
|
||||
required_permission: str
|
||||
expected_blast_radius: str
|
||||
drain_requirement: str
|
||||
full_drain_required: bool
|
||||
approval_requirement: str
|
||||
audit_requirement: str
|
||||
recovery_behavior: str
|
||||
request_roles: tuple[str, ...]
|
||||
execution_roles: tuple[str, ...]
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"restart_class": self.restart_class.value,
|
||||
"required_permission": self.required_permission,
|
||||
"expected_blast_radius": self.expected_blast_radius,
|
||||
"drain_requirement": self.drain_requirement,
|
||||
"full_drain_required": self.full_drain_required,
|
||||
"approval_requirement": self.approval_requirement,
|
||||
"audit_requirement": self.audit_requirement,
|
||||
"recovery_behavior": self.recovery_behavior,
|
||||
"request_roles": list(self.request_roles),
|
||||
"execution_roles": list(self.execution_roles),
|
||||
}
|
||||
|
||||
|
||||
WORKER_ROLES = ("author", "reviewer", "merger", "reconciler")
|
||||
CONTROL_ROLES = ("controller", "operator", "admin")
|
||||
ALL_REQUEST_ROLES = WORKER_ROLES + CONTROL_ROLES
|
||||
|
||||
RESTART_CLASS_POLICIES: dict[RestartClass, RestartClassPolicy] = {
|
||||
RestartClass.CLIENT_RECONNECT: RestartClassPolicy(
|
||||
RestartClass.CLIENT_RECONNECT,
|
||||
"mcp.reconnect.client",
|
||||
BLAST_NONE,
|
||||
"none",
|
||||
False,
|
||||
"self_service",
|
||||
"record class, actor, client namespace, reason, and outcome",
|
||||
"Reconnect only the caller's client transport; no daemon or peer session changes.",
|
||||
ALL_REQUEST_ROLES,
|
||||
ALL_REQUEST_ROLES,
|
||||
),
|
||||
RestartClass.SESSION_RECONNECT: RestartClassPolicy(
|
||||
RestartClass.SESSION_RECONNECT,
|
||||
"mcp.reconnect.session",
|
||||
BLAST_LOW,
|
||||
"requesting_session_safe_point",
|
||||
False,
|
||||
"self_service",
|
||||
"record class, actor, session id, reason, and outcome",
|
||||
"Rebind identity, capability, and workspace state for one session.",
|
||||
ALL_REQUEST_ROLES,
|
||||
ALL_REQUEST_ROLES,
|
||||
),
|
||||
RestartClass.WORKER_RESTART: RestartClassPolicy(
|
||||
RestartClass.WORKER_RESTART,
|
||||
"mcp.restart.worker.request",
|
||||
BLAST_LOW,
|
||||
"target_worker",
|
||||
False,
|
||||
"controller_approval_and_automated_gates",
|
||||
"record class, actor, target worker, approval, drain proof, and outcome",
|
||||
"Restart one worker after its own lease and mutation scope is drained.",
|
||||
ALL_REQUEST_ROLES,
|
||||
("operator", "admin"),
|
||||
),
|
||||
RestartClass.ROLE_RUNTIME_RESTART: RestartClassPolicy(
|
||||
RestartClass.ROLE_RUNTIME_RESTART,
|
||||
"mcp.restart.role_runtime.request",
|
||||
BLAST_MEDIUM,
|
||||
"target_role_runtime",
|
||||
False,
|
||||
"controller_approval_and_automated_gates",
|
||||
"record class, actor, role namespace, approval, drain proof, and outcome",
|
||||
"Restart only the selected role runtime and then re-probe that namespace.",
|
||||
ALL_REQUEST_ROLES,
|
||||
("operator", "admin"),
|
||||
),
|
||||
RestartClass.CONNECTOR_RESTART: RestartClassPolicy(
|
||||
RestartClass.CONNECTOR_RESTART,
|
||||
"mcp.restart.connector.request",
|
||||
BLAST_MEDIUM,
|
||||
"target_connector",
|
||||
False,
|
||||
"controller_approval_and_automated_gates",
|
||||
"record class, actor, connector id, approval, drain proof, and outcome",
|
||||
"Restart one connector while unrelated role runtimes remain available.",
|
||||
ALL_REQUEST_ROLES,
|
||||
("operator", "admin"),
|
||||
),
|
||||
RestartClass.CONFIGURATION_RELOAD: RestartClassPolicy(
|
||||
RestartClass.CONFIGURATION_RELOAD,
|
||||
"mcp.reload.configuration.request",
|
||||
BLAST_LOW,
|
||||
"mutation_quiesce",
|
||||
False,
|
||||
"controller_approval_and_automated_gates",
|
||||
"record class, actor, configuration revision, approval, and outcome",
|
||||
"Gracefully reload configuration without replacing the daemon process.",
|
||||
ALL_REQUEST_ROLES,
|
||||
("operator", "admin"),
|
||||
),
|
||||
RestartClass.ROLLING_MCP_RESTART: RestartClassPolicy(
|
||||
RestartClass.ROLLING_MCP_RESTART,
|
||||
"mcp.restart.rolling.request",
|
||||
BLAST_MEDIUM,
|
||||
"one_instance_at_a_time",
|
||||
False,
|
||||
"controller_approval_and_automated_gates",
|
||||
"record class, actor, instance order, approval, per-instance drains, and outcome",
|
||||
"Drain, restart, verify, and restore one instance before advancing to the next.",
|
||||
CONTROL_ROLES,
|
||||
("operator", "admin"),
|
||||
),
|
||||
RestartClass.FULL_MCP_RESTART: RestartClassPolicy(
|
||||
RestartClass.FULL_MCP_RESTART,
|
||||
"mcp.restart.full.request",
|
||||
BLAST_HIGH,
|
||||
"all_sessions_and_mutations",
|
||||
True,
|
||||
"controller_approval_and_automated_gates",
|
||||
"record class, actor, full impact report, approval, drain proof, and outcome",
|
||||
"Stop and restore the complete MCP runtime only after a verified full drain.",
|
||||
CONTROL_ROLES,
|
||||
("operator", "admin"),
|
||||
),
|
||||
RestartClass.HOST_RESTART: RestartClassPolicy(
|
||||
RestartClass.HOST_RESTART,
|
||||
"mcp.restart.host.request",
|
||||
BLAST_HIGH,
|
||||
"all_host_work",
|
||||
True,
|
||||
"controller_approval_plus_infrastructure_operator",
|
||||
"record class, actor, host, incident or change id, approval, drain proof, and outcome",
|
||||
"Hand off to infrastructure ownership; reconcile every runtime after the host returns.",
|
||||
("controller", "operator", "admin"),
|
||||
("operator", "admin"),
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def resolve_restart_class(value: RestartClass | str) -> RestartClass:
|
||||
"""Resolve a restart class or fail closed for an unknown value."""
|
||||
|
||||
if isinstance(value, RestartClass):
|
||||
return value
|
||||
try:
|
||||
return RestartClass(str(value).strip())
|
||||
except ValueError as exc:
|
||||
raise ValueError(f"unknown restart class {value!r}; deny (fail closed)") from exc
|
||||
|
||||
|
||||
def restart_class_policy(value: RestartClass | str) -> RestartClassPolicy:
|
||||
"""Return the canonical policy for *value*."""
|
||||
|
||||
return RESTART_CLASS_POLICIES[resolve_restart_class(value)]
|
||||
|
||||
|
||||
def permissions_for_role(role: str | None) -> tuple[str, ...]:
|
||||
"""Return request permissions granted to a workflow role by this policy."""
|
||||
|
||||
normalized = str(role or "").strip().lower()
|
||||
return tuple(
|
||||
policy.required_permission
|
||||
for policy in RESTART_CLASS_POLICIES.values()
|
||||
if normalized in policy.request_roles
|
||||
)
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
@@ -75,6 +268,7 @@ class SessionImpact:
|
||||
heartbeat_stale: bool
|
||||
is_requester: bool
|
||||
live: bool
|
||||
connector: str | None = None
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
@@ -87,6 +281,7 @@ class SessionImpact:
|
||||
"heartbeat_stale": self.heartbeat_stale,
|
||||
"is_requester": self.is_requester,
|
||||
"live": self.live,
|
||||
"connector": self.connector,
|
||||
}
|
||||
|
||||
|
||||
@@ -105,6 +300,7 @@ class LeaseImpact:
|
||||
disruptive: bool
|
||||
is_mutation: bool
|
||||
is_critical_section: bool
|
||||
connector: str | None = None
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
@@ -119,6 +315,7 @@ class LeaseImpact:
|
||||
"disruptive": self.disruptive,
|
||||
"is_mutation": self.is_mutation,
|
||||
"is_critical_section": self.is_critical_section,
|
||||
"connector": self.connector,
|
||||
}
|
||||
|
||||
|
||||
@@ -127,6 +324,13 @@ class RestartImpactReport:
|
||||
"""Impact preview DTO returned to the console / operator (#642/#652)."""
|
||||
|
||||
coordinator_version: str
|
||||
restart_class: str
|
||||
restart_policy: dict[str, Any]
|
||||
policy_enforced: bool
|
||||
permission_authorized: bool
|
||||
role_authorized: bool
|
||||
approval_satisfied: bool
|
||||
authorization_reasons: list[str]
|
||||
evaluated_at: str
|
||||
dry_run: bool
|
||||
restart_performed: bool
|
||||
@@ -149,10 +353,21 @@ class RestartImpactReport:
|
||||
counts: dict[str, int]
|
||||
audit_record: dict[str, Any]
|
||||
incomplete_reasons: list[str] = field(default_factory=list)
|
||||
# #669 playbook escalation gate (attempt-log enforcement).
|
||||
playbook_escalation: dict[str, Any] = field(default_factory=dict)
|
||||
attempt_log_satisfied: bool = True
|
||||
break_glass: bool = False
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"coordinator_version": self.coordinator_version,
|
||||
"restart_class": self.restart_class,
|
||||
"restart_policy": dict(self.restart_policy),
|
||||
"policy_enforced": self.policy_enforced,
|
||||
"permission_authorized": self.permission_authorized,
|
||||
"role_authorized": self.role_authorized,
|
||||
"approval_satisfied": self.approval_satisfied,
|
||||
"authorization_reasons": list(self.authorization_reasons),
|
||||
"evaluated_at": self.evaluated_at,
|
||||
"dry_run": self.dry_run,
|
||||
"restart_performed": self.restart_performed,
|
||||
@@ -175,6 +390,9 @@ class RestartImpactReport:
|
||||
"prior_recovery_attempts": list(self.prior_recovery_attempts),
|
||||
"counts": dict(self.counts),
|
||||
"audit_record": dict(self.audit_record),
|
||||
"playbook_escalation": dict(self.playbook_escalation),
|
||||
"attempt_log_satisfied": self.attempt_log_satisfied,
|
||||
"break_glass": self.break_glass,
|
||||
}
|
||||
|
||||
|
||||
@@ -206,6 +424,7 @@ def _classify_session(
|
||||
requesting_session_id and session_id == requesting_session_id
|
||||
),
|
||||
live=live,
|
||||
connector=(str(row.get("connector") or "").strip() or None),
|
||||
)
|
||||
|
||||
|
||||
@@ -258,6 +477,7 @@ def _classify_lease(row: Mapping[str, Any]) -> LeaseImpact:
|
||||
disruptive=disruptive,
|
||||
is_mutation=is_mutation,
|
||||
is_critical_section=disruptive,
|
||||
connector=(str(row.get("connector") or "").strip() or None),
|
||||
)
|
||||
|
||||
|
||||
@@ -279,6 +499,15 @@ def evaluate_restart_impact(
|
||||
requesting_session_id: str | None = None,
|
||||
dry_run: bool = True,
|
||||
session_heartbeat_stale_seconds: int = DEFAULT_SESSION_HEARTBEAT_STALE_SECONDS,
|
||||
restart_class: RestartClass | str | None = None,
|
||||
requester_role: str | None = None,
|
||||
requester_permissions: Sequence[str] | None = None,
|
||||
controller_approved: bool = False,
|
||||
operator_authorized: bool = False,
|
||||
target_session_id: str | None = None,
|
||||
target_role: str | None = None,
|
||||
target_connector: str | None = None,
|
||||
break_glass: bool = False,
|
||||
) -> RestartImpactReport:
|
||||
"""Evaluate a proposed MCP restart and return an impact preview.
|
||||
|
||||
@@ -301,6 +530,61 @@ def evaluate_restart_impact(
|
||||
"""
|
||||
moment = now or _utc_now()
|
||||
reasons: list[str] = []
|
||||
authorization_reasons: list[str] = []
|
||||
|
||||
# ``None`` preserves the pre-#663 impact-only API for callers that have not
|
||||
# yet been migrated. All MCP requests pass an explicit class and therefore
|
||||
# take the fail-closed policy path.
|
||||
policy_enforced = restart_class is not None
|
||||
try:
|
||||
resolved_class = resolve_restart_class(
|
||||
restart_class or RestartClass.FULL_MCP_RESTART
|
||||
)
|
||||
policy = RESTART_CLASS_POLICIES[resolved_class]
|
||||
unknown_class = False
|
||||
except ValueError as exc:
|
||||
resolved_class = None
|
||||
policy = None
|
||||
unknown_class = True
|
||||
authorization_reasons.append(str(exc))
|
||||
|
||||
normalized_role = str(requester_role or "").strip().lower()
|
||||
granted = {str(p).strip() for p in (requester_permissions or ())}
|
||||
if policy_enforced and policy is not None:
|
||||
permission_authorized = policy.required_permission in granted
|
||||
role_authorized = normalized_role in policy.request_roles
|
||||
if not permission_authorized:
|
||||
authorization_reasons.append(
|
||||
f"missing required permission {policy.required_permission!r}"
|
||||
)
|
||||
if not role_authorized:
|
||||
authorization_reasons.append(
|
||||
f"role {normalized_role or 'unknown'!r} may not request "
|
||||
f"{policy.restart_class.value}"
|
||||
)
|
||||
elif unknown_class:
|
||||
permission_authorized = False
|
||||
role_authorized = False
|
||||
else:
|
||||
permission_authorized = True
|
||||
role_authorized = True
|
||||
|
||||
if policy_enforced and policy is not None:
|
||||
approval = policy.approval_requirement
|
||||
if approval == "self_service":
|
||||
approval_satisfied = True
|
||||
elif approval == "controller_approval_plus_infrastructure_operator":
|
||||
approval_satisfied = bool(controller_approved and operator_authorized)
|
||||
else:
|
||||
approval_satisfied = bool(controller_approved)
|
||||
if not approval_satisfied:
|
||||
authorization_reasons.append(
|
||||
f"approval requirement not satisfied: {approval}"
|
||||
)
|
||||
elif unknown_class:
|
||||
approval_satisfied = False
|
||||
else:
|
||||
approval_satisfied = True
|
||||
|
||||
inventory_complete = bool(inventory.get("inventory_complete", False))
|
||||
incomplete_reasons = [str(r) for r in (inventory.get("incomplete_reasons") or [])]
|
||||
@@ -312,6 +596,30 @@ def evaluate_restart_impact(
|
||||
dict(a) for a in (inventory.get("prior_recovery_attempts") or [])
|
||||
]
|
||||
|
||||
# #669: broad restarts require a prior narrow-attempt log unless break-glass.
|
||||
playbook_escalation: dict[str, Any] = {}
|
||||
attempt_log_satisfied = True
|
||||
if policy_enforced and resolved_class is not None:
|
||||
try:
|
||||
escalation = recovery_playbook.assess_escalation(
|
||||
resolved_class.value,
|
||||
prior_recovery_attempts=prior_recovery_attempts,
|
||||
break_glass=bool(break_glass),
|
||||
)
|
||||
playbook_escalation = escalation.as_dict()
|
||||
attempt_log_satisfied = bool(escalation.allowed)
|
||||
if not attempt_log_satisfied:
|
||||
authorization_reasons.extend(list(escalation.reasons))
|
||||
except ValueError as exc:
|
||||
# Unknown mapping should never happen for enum values; fail closed.
|
||||
attempt_log_satisfied = False
|
||||
playbook_escalation = {
|
||||
"allowed": False,
|
||||
"reasons": [str(exc)],
|
||||
"playbook_version": recovery_playbook.PLAYBOOK_VERSION,
|
||||
}
|
||||
authorization_reasons.append(str(exc))
|
||||
|
||||
session_impacts = [
|
||||
_classify_session(
|
||||
s,
|
||||
@@ -323,15 +631,67 @@ def evaluate_restart_impact(
|
||||
]
|
||||
lease_impacts = [_classify_lease(l) for l in leases_raw]
|
||||
|
||||
# Only *other* live sessions and live leases constitute blast radius: a
|
||||
# restart that would kill only the requesting session with no other work in
|
||||
# flight is safe.
|
||||
# Route impact through the selected class. Narrow classes never inherit a
|
||||
# full-runtime drain merely because unrelated work exists.
|
||||
target_complete = True
|
||||
if resolved_class in {
|
||||
RestartClass.CLIENT_RECONNECT,
|
||||
RestartClass.SESSION_RECONNECT,
|
||||
RestartClass.CONFIGURATION_RELOAD,
|
||||
}:
|
||||
scoped_sessions: list[SessionImpact] = []
|
||||
scoped_leases: list[LeaseImpact] = []
|
||||
elif resolved_class == RestartClass.WORKER_RESTART:
|
||||
selected_session = (target_session_id or "").strip()
|
||||
target_complete = bool(selected_session)
|
||||
scoped_sessions = [
|
||||
s for s in session_impacts if s.session_id == selected_session
|
||||
]
|
||||
scoped_leases = [
|
||||
l for l in lease_impacts if l.session_id == selected_session
|
||||
]
|
||||
elif resolved_class == RestartClass.ROLE_RUNTIME_RESTART:
|
||||
selected_role = (target_role or "").strip().lower()
|
||||
target_complete = bool(selected_role)
|
||||
scoped_sessions = [
|
||||
s for s in session_impacts if str(s.role or "").lower() == selected_role
|
||||
]
|
||||
scoped_leases = [
|
||||
l for l in lease_impacts if str(l.role or "").lower() == selected_role
|
||||
]
|
||||
elif resolved_class == RestartClass.CONNECTOR_RESTART:
|
||||
selected_connector = (target_connector or "").strip()
|
||||
target_complete = bool(selected_connector)
|
||||
scoped_sessions = [
|
||||
s for s in session_impacts if s.connector == selected_connector
|
||||
]
|
||||
scoped_leases = [
|
||||
l for l in lease_impacts if l.connector == selected_connector
|
||||
]
|
||||
else:
|
||||
scoped_sessions = list(session_impacts)
|
||||
scoped_leases = list(lease_impacts)
|
||||
|
||||
if policy_enforced and not target_complete:
|
||||
authorization_reasons.append(
|
||||
f"target required for {resolved_class.value if resolved_class else 'unknown class'}"
|
||||
)
|
||||
|
||||
other_live_sessions = [
|
||||
s for s in session_impacts if s.live and not s.is_requester
|
||||
s for s in scoped_sessions if s.live and not s.is_requester
|
||||
]
|
||||
disruptive_leases = [l for l in lease_impacts if l.disruptive]
|
||||
critical_sections = [l for l in lease_impacts if l.is_critical_section]
|
||||
mutations = [l for l in lease_impacts if l.is_mutation]
|
||||
disruptive_leases = [l for l in scoped_leases if l.disruptive]
|
||||
critical_sections = [l for l in scoped_leases if l.is_critical_section]
|
||||
mutations = [l for l in scoped_leases if l.is_mutation]
|
||||
terminal_lock_in_scope = (
|
||||
terminal_lock
|
||||
if resolved_class
|
||||
not in {
|
||||
RestartClass.CLIENT_RECONNECT,
|
||||
RestartClass.SESSION_RECONNECT,
|
||||
}
|
||||
else None
|
||||
)
|
||||
|
||||
affected_issues = sorted(
|
||||
{
|
||||
@@ -348,9 +708,25 @@ def evaluate_restart_impact(
|
||||
}
|
||||
)
|
||||
|
||||
disruptive = bool(disruptive_leases or other_live_sessions or terminal_lock)
|
||||
disruptive = bool(
|
||||
disruptive_leases or other_live_sessions or terminal_lock_in_scope
|
||||
)
|
||||
|
||||
if not inventory_complete:
|
||||
authorization_ok = bool(
|
||||
not unknown_class
|
||||
and permission_authorized
|
||||
and role_authorized
|
||||
and approval_satisfied
|
||||
and target_complete
|
||||
and attempt_log_satisfied
|
||||
)
|
||||
|
||||
if policy_enforced and not authorization_ok:
|
||||
verdict = VERDICT_UNSAFE
|
||||
allow_restart = False
|
||||
reasons.append("restart class authorization denied (fail closed)")
|
||||
reasons.extend(authorization_reasons)
|
||||
elif not inventory_complete:
|
||||
verdict = VERDICT_UNSAFE
|
||||
allow_restart = False
|
||||
reasons.append(
|
||||
@@ -381,7 +757,7 @@ def evaluate_restart_impact(
|
||||
f"{len(critical_sections)} critical section(s) in flight "
|
||||
"(active lease with a live owner)"
|
||||
)
|
||||
if terminal_lock:
|
||||
if terminal_lock_in_scope:
|
||||
reasons.append("active terminal (merge) lock present")
|
||||
|
||||
override_would_allow = bool(inventory_complete and disruptive)
|
||||
@@ -406,11 +782,18 @@ def evaluate_restart_impact(
|
||||
"affected_issues": len(affected_issues),
|
||||
"affected_prs": len(affected_prs),
|
||||
"prior_recovery_attempts": len(prior_recovery_attempts),
|
||||
"attempt_log_satisfied": attempt_log_satisfied,
|
||||
}
|
||||
|
||||
audit_record = {
|
||||
"event": "restart_impact_evaluated",
|
||||
"coordinator_version": COORDINATOR_VERSION,
|
||||
"restart_class": (
|
||||
resolved_class.value if resolved_class else str(restart_class or "")
|
||||
),
|
||||
"required_permission": (
|
||||
policy.required_permission if policy is not None else None
|
||||
),
|
||||
"evaluated_at": moment.isoformat(),
|
||||
"dry_run": dry_run,
|
||||
"operator_override": bool(operator_override),
|
||||
@@ -420,10 +803,22 @@ def evaluate_restart_impact(
|
||||
"allow_restart": allow_restart,
|
||||
"blast_radius": blast_radius,
|
||||
"counts": counts,
|
||||
"attempt_log_satisfied": attempt_log_satisfied,
|
||||
"break_glass": bool(break_glass),
|
||||
"playbook_version": recovery_playbook.PLAYBOOK_VERSION,
|
||||
}
|
||||
|
||||
return RestartImpactReport(
|
||||
coordinator_version=COORDINATOR_VERSION,
|
||||
restart_class=(
|
||||
resolved_class.value if resolved_class else str(restart_class or "")
|
||||
),
|
||||
restart_policy=policy.as_dict() if policy is not None else {},
|
||||
policy_enforced=policy_enforced,
|
||||
permission_authorized=permission_authorized,
|
||||
role_authorized=role_authorized,
|
||||
approval_satisfied=approval_satisfied,
|
||||
authorization_reasons=authorization_reasons,
|
||||
evaluated_at=moment.isoformat(),
|
||||
dry_run=dry_run,
|
||||
restart_performed=False,
|
||||
@@ -440,12 +835,17 @@ def evaluate_restart_impact(
|
||||
affected_issues=affected_issues,
|
||||
affected_prs=affected_prs,
|
||||
mutations=mutations,
|
||||
terminal_lock=dict(terminal_lock)
|
||||
if isinstance(terminal_lock, Mapping)
|
||||
else terminal_lock,
|
||||
terminal_lock=(
|
||||
dict(terminal_lock_in_scope)
|
||||
if isinstance(terminal_lock_in_scope, Mapping)
|
||||
else terminal_lock_in_scope
|
||||
),
|
||||
ack_state=ack_state,
|
||||
prior_recovery_attempts=prior_recovery_attempts,
|
||||
counts=counts,
|
||||
audit_record=audit_record,
|
||||
incomplete_reasons=incomplete_reasons,
|
||||
playbook_escalation=playbook_escalation,
|
||||
attempt_log_satisfied=attempt_log_satisfied,
|
||||
break_glass=bool(break_glass),
|
||||
)
|
||||
|
||||
@@ -63,6 +63,11 @@ CONTAMINATION_GATED_TASKS = frozenset({
|
||||
"merge_pr",
|
||||
"delete_branch",
|
||||
"complete_issue",
|
||||
# Web console recovery playbooks that write (#644). These mutate runtime
|
||||
# binding and process state, so a live contamination marker must block them
|
||||
# exactly as it blocks the Gitea-side mutations above. The reconciler
|
||||
# cleanup playbook is the designated remedy and is exempted by its caller.
|
||||
"console_recovery_apply",
|
||||
})
|
||||
|
||||
CONTAMINATION_KIND = "stable_branch_push"
|
||||
|
||||
@@ -142,6 +142,23 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
# #644: Phase 2 Web Console recovery tasks.
|
||||
"clear_stale_binding": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
"rebind_session_worktree": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
# The console playbook orchestrates gitea_reconcile_merged_cleanups, whose
|
||||
# own gate is gitea.read (matching the existing reconcile_merged_cleanups
|
||||
# entry). Declaring a stricter permission here stated a second, conflicting
|
||||
# authority for one operation.
|
||||
"reconcile_cleanups": {
|
||||
"permission": "gitea.read",
|
||||
"role": "reconciler",
|
||||
},
|
||||
# PR synchronization lifecycle: assess is read-only (any role with gitea.read);
|
||||
# update-by-merge is author-only and mutates the PR head via Gitea API.
|
||||
"assess_pr_sync_status": {
|
||||
@@ -397,6 +414,24 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"role": "controller",
|
||||
},
|
||||
|
||||
# #659 maintenance drain. Same reasoning as the lifecycle controls above:
|
||||
# entering/exiting drain quiesces a whole namespace, so it carries a
|
||||
# non-``gitea.*`` permission that no configured Gitea profile satisfies by
|
||||
# accident (AC1 — capability-gated and audited). Reading drain state is
|
||||
# ordinary read authority: every session must be able to see the drain (AC4).
|
||||
"enter_maintenance_drain": {
|
||||
"permission": "runtime.maintenance_drain",
|
||||
"role": "controller",
|
||||
},
|
||||
"exit_maintenance_drain": {
|
||||
"permission": "runtime.maintenance_drain",
|
||||
"role": "controller",
|
||||
},
|
||||
"maintenance_drain_status": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
|
||||
# #601 first-class lease lifecycle — inspect/list need read; mutations gate on
|
||||
# ownership in the control-plane DB (not a separate Gitea write permission).
|
||||
"list_workflow_leases": {
|
||||
|
||||
@@ -7,6 +7,7 @@ import tempfile
|
||||
import threading
|
||||
import unittest
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from allocator_service import (
|
||||
OUTCOME_ASSIGNED,
|
||||
@@ -15,6 +16,7 @@ from allocator_service import (
|
||||
OUTCOME_PREVIEW,
|
||||
OUTCOME_WAIT,
|
||||
WorkCandidate,
|
||||
_drop_expired_claims,
|
||||
allocate_next_work,
|
||||
candidate_from_dict,
|
||||
classify_skip,
|
||||
@@ -362,5 +364,161 @@ class AllocatorServiceTest(unittest.TestCase):
|
||||
self.assertIn("unavailable", res["reasons"][0].lower())
|
||||
|
||||
|
||||
class SideEffectFreeAllocationTest(unittest.TestCase):
|
||||
"""``side_effect_free`` dry runs write nothing to the control plane (#643).
|
||||
|
||||
A plain ``apply=False`` still called ``upsert_session`` and
|
||||
``expire_stale_leases`` before the apply branch was consulted, so a caller
|
||||
advertising a read-only preview mutated on every call — one unreferenced
|
||||
session row per preview, plus a global lease sweep.
|
||||
"""
|
||||
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.db = ControlPlaneDB(os.path.join(self._tmp.name, "cp.sqlite3"))
|
||||
|
||||
def tearDown(self) -> None:
|
||||
self._tmp.cleanup()
|
||||
|
||||
def _alloc(self, **kwargs):
|
||||
defaults = dict(
|
||||
db=self.db,
|
||||
session_id="s-preview",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
candidates=[
|
||||
WorkCandidate(kind="issue", number=643, labels=("status:ready",))
|
||||
],
|
||||
apply=False,
|
||||
profile_name="prgs-author",
|
||||
username="jcwalker3",
|
||||
)
|
||||
defaults.update(kwargs)
|
||||
return allocate_next_work(**defaults)
|
||||
|
||||
def _session_ids(self) -> set[str]:
|
||||
return {str(r.get("session_id")) for r in self.db.list_sessions()}
|
||||
|
||||
def test_side_effect_free_preview_writes_no_session_row(self):
|
||||
before = self._session_ids()
|
||||
result = self._alloc(side_effect_free=True)
|
||||
self.assertEqual(result["outcome"], OUTCOME_PREVIEW)
|
||||
self.assertEqual(self._session_ids(), before)
|
||||
self.assertNotIn("s-preview", self._session_ids())
|
||||
|
||||
def test_plain_dry_run_still_registers_a_session(self):
|
||||
# The default is unchanged for every existing caller.
|
||||
self._alloc()
|
||||
self.assertIn("s-preview", self._session_ids())
|
||||
|
||||
def test_repeated_previews_do_not_accumulate_rows(self):
|
||||
for index in range(5):
|
||||
self._alloc(side_effect_free=True, session_id=f"s-{index}")
|
||||
self.assertEqual(self._session_ids(), set())
|
||||
|
||||
def test_side_effect_free_does_not_sweep_stale_leases(self):
|
||||
self.db.upsert_session(session_id="owner", role="author", pid=1)
|
||||
assigned = self.db.assign_and_lease(
|
||||
session_id="owner",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
kind="issue",
|
||||
number=999,
|
||||
lease_ttl_seconds=-60, # already expired
|
||||
)
|
||||
self.assertEqual(assigned.outcome, "assigned")
|
||||
|
||||
self._alloc(side_effect_free=True)
|
||||
|
||||
# The expired row is still 'active' in the DB: nothing swept it.
|
||||
statuses = {
|
||||
r["lease_id"]: r["status"]
|
||||
for r in self.db.list_leases(
|
||||
remote="prgs", org="org", repo="repo",
|
||||
statuses=("active", "expired"),
|
||||
)
|
||||
}
|
||||
self.assertEqual(statuses.get(assigned.lease_id), "active")
|
||||
|
||||
def test_expired_claims_are_filtered_in_memory_so_work_stays_selectable(self):
|
||||
"""The read-only mirror of the sweep: expired claims must not block."""
|
||||
self.db.upsert_session(session_id="owner", role="author", pid=1)
|
||||
self.db.assign_and_lease(
|
||||
session_id="owner",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
kind="issue",
|
||||
number=643,
|
||||
lease_ttl_seconds=-60, # expired: must not withhold #643
|
||||
)
|
||||
result = self._alloc(side_effect_free=True)
|
||||
self.assertEqual(result["outcome"], OUTCOME_PREVIEW)
|
||||
self.assertEqual(result["selected"]["number"], 643)
|
||||
|
||||
def test_a_live_claim_still_withholds_the_work(self):
|
||||
self.db.upsert_session(session_id="owner", role="author", pid=1)
|
||||
self.db.assign_and_lease(
|
||||
session_id="owner",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
kind="issue",
|
||||
number=643,
|
||||
lease_ttl_seconds=3600,
|
||||
)
|
||||
result = self._alloc(side_effect_free=True)
|
||||
self.assertNotEqual(result["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertNotEqual((result.get("selected") or {}).get("number"), 643)
|
||||
|
||||
def test_side_effect_free_with_apply_fails_closed(self):
|
||||
result = self._alloc(side_effect_free=True, apply=True)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["outcome"], OUTCOME_NO_SAFE)
|
||||
self.assertIsNone(result["assignment"])
|
||||
self.assertIn("incompatible with apply", result["reasons"][0])
|
||||
# And it reserved nothing.
|
||||
self.assertEqual(
|
||||
self.db.list_leases(remote="prgs", org="org", repo="repo"), []
|
||||
)
|
||||
|
||||
|
||||
class DropExpiredClaimsTest(unittest.TestCase):
|
||||
"""The in-memory expiry filter behind side-effect-free previews (#643)."""
|
||||
|
||||
def test_unparseable_expiry_is_kept_rather_than_assumed_free(self):
|
||||
claims = {
|
||||
("issue", 1): {"lease_id": "l1", "expires_at": "not-a-date"},
|
||||
("issue", 2): {"lease_id": "l2"},
|
||||
("issue", 3): {"lease_id": "l3", "expires_at": None},
|
||||
}
|
||||
self.assertEqual(_drop_expired_claims(claims), claims)
|
||||
|
||||
def test_expired_dropped_and_future_kept(self):
|
||||
now = datetime(2026, 7, 25, 12, 0, tzinfo=timezone.utc)
|
||||
claims = {
|
||||
("issue", 1): {"expires_at": "2026-07-25T11:59:59+00:00"},
|
||||
("issue", 2): {"expires_at": "2026-07-25T12:00:01+00:00"},
|
||||
("issue", 3): {"expires_at": "2026-07-25T12:00:00+00:00"}, # boundary
|
||||
}
|
||||
kept = _drop_expired_claims(claims, now=now)
|
||||
self.assertEqual(set(kept), {("issue", 2)})
|
||||
|
||||
def test_naive_and_zulu_timestamps_are_treated_as_utc(self):
|
||||
now = datetime(2026, 7, 25, 12, 0, tzinfo=timezone.utc)
|
||||
claims = {
|
||||
("issue", 1): {"expires_at": "2026-07-25T11:00:00"}, # naive, past
|
||||
("issue", 2): {"expires_at": "2026-07-25T13:00:00Z"}, # zulu, future
|
||||
}
|
||||
kept = _drop_expired_claims(claims, now=now)
|
||||
self.assertEqual(set(kept), {("issue", 2)})
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -0,0 +1,907 @@
|
||||
"""Tests for the pre-restart drain proof and hard gate (#661).
|
||||
|
||||
Covers the acceptance criteria:
|
||||
|
||||
1. Restart apply without a proof fails closed.
|
||||
2. A successful drain produces a verifiable proof.
|
||||
3. An open unsafe mutation makes the proof fail (multi-session fixture).
|
||||
4. Pass / fail / expired verification paths.
|
||||
|
||||
Plus the security posture: forged/tampered proofs are rejected, break-glass is
|
||||
the only bypass and is never silent, a stale blast-radius fingerprint rejects a
|
||||
proof, and no per-process secret ever leaks into a serialized artifact.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
import drain_proof as dp
|
||||
import restart_coordinator as rc
|
||||
|
||||
|
||||
NOW = datetime(2026, 7, 24, 6, 0, 0, tzinfo=timezone.utc)
|
||||
SECRET = b"unit-test-drain-proof-secret-0123456789abcdef"
|
||||
|
||||
|
||||
def _live_pid() -> int:
|
||||
return os.getpid()
|
||||
|
||||
|
||||
def _clean_drain_state() -> dict:
|
||||
"""Every drain action succeeded, no sessions outstanding."""
|
||||
|
||||
return {
|
||||
"assignments_stopped": True,
|
||||
"checkpoints_complete": True,
|
||||
"handoffs_verified": True,
|
||||
"leases_handled": True,
|
||||
"acks": {}, # no other live sessions to acknowledge
|
||||
"ack_timeout_policy_applied": False,
|
||||
}
|
||||
|
||||
|
||||
def _safe_report() -> dict:
|
||||
"""Impact report with no other live work: a restart here is safe."""
|
||||
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": [], "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id="prgs-controller-1-req",
|
||||
)
|
||||
return report.as_dict()
|
||||
|
||||
|
||||
def _unsafe_mutation_report() -> dict:
|
||||
"""Multi-session report: a second session holds a live author mutation."""
|
||||
|
||||
sessions = [
|
||||
{
|
||||
"session_id": "prgs-controller-1-req",
|
||||
"role": "controller",
|
||||
"profile": "prgs-controller",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
{
|
||||
"session_id": "prgs-author-99",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
]
|
||||
leases = [
|
||||
{
|
||||
"lease_id": "lease-mut",
|
||||
"session_id": "prgs-author-99",
|
||||
"role": "author",
|
||||
"phase": "implementing",
|
||||
"work_kind": "issue",
|
||||
"work_number": 661,
|
||||
"worktree_path": "branches/issue-661",
|
||||
"freshness": {"freshness": "active"},
|
||||
}
|
||||
]
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": sessions, "leases": leases, "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id="prgs-controller-1-req",
|
||||
)
|
||||
return report.as_dict()
|
||||
|
||||
|
||||
class BuildDrainProofTests(unittest.TestCase):
|
||||
def test_clean_drain_produces_verifiable_clean_proof(self):
|
||||
"""AC#2: a successful drain produces a verifiable proof."""
|
||||
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
requesting_session_id="prgs-controller-1-req",
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
)
|
||||
self.assertTrue(proof.clean)
|
||||
self.assertEqual(proof.failed_checks, [])
|
||||
self.assertEqual(
|
||||
{c.name for c in proof.checks}, set(dp.REQUIRED_CHECKS)
|
||||
)
|
||||
result = dp.verify_drain_proof(
|
||||
proof.as_dict(), now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertTrue(result.valid, result.reasons)
|
||||
self.assertFalse(result.expired)
|
||||
self.assertFalse(result.tampered)
|
||||
|
||||
def test_open_mutation_makes_proof_unclean(self):
|
||||
"""AC#3: an unsafe mutation still in flight fails the proof."""
|
||||
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_unsafe_mutation_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertIn(dp.CHECK_NO_INFLIGHT_MUTATIONS, proof.failed_checks)
|
||||
# Leases-handled also fails: the report still shows a disruptive lease.
|
||||
self.assertIn(dp.CHECK_LEASES_HANDLED, proof.failed_checks)
|
||||
result = dp.verify_drain_proof(proof.as_dict(), now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
|
||||
def test_incomplete_inventory_fails_no_mutations_check(self):
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report={"inventory_complete": False},
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertIn(dp.CHECK_NO_INFLIGHT_MUTATIONS, proof.failed_checks)
|
||||
|
||||
def test_missing_checkpoint_flag_fails_closed(self):
|
||||
state = _clean_drain_state()
|
||||
del state["checkpoints_complete"]
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertIn(dp.CHECK_CHECKPOINTS_COMPLETE, proof.failed_checks)
|
||||
|
||||
def test_non_true_flags_fail_closed(self):
|
||||
"""A truthy-but-not-True value (e.g. the string 'yes') must not pass."""
|
||||
|
||||
state = _clean_drain_state()
|
||||
state["assignments_stopped"] = "yes"
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertIn(dp.CHECK_ASSIGNMENTS_STOPPED, proof.failed_checks)
|
||||
|
||||
def test_ack_timeout_policy_satisfies_ack_check(self):
|
||||
state = _clean_drain_state()
|
||||
state["acks"] = {"prgs-author-99": "pending"}
|
||||
state["ack_timeout_policy_applied"] = True
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
names = {c.name: c.passed for c in proof.checks}
|
||||
self.assertTrue(names[dp.CHECK_ACKS_OR_TIMEOUT])
|
||||
|
||||
def test_outstanding_acks_without_timeout_fail(self):
|
||||
state = _clean_drain_state()
|
||||
state["acks"] = {"prgs-author-99": "pending"}
|
||||
state["ack_timeout_policy_applied"] = False
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertIn(dp.CHECK_ACKS_OR_TIMEOUT, proof.failed_checks)
|
||||
|
||||
def test_all_acked_satisfies_ack_check(self):
|
||||
state = _clean_drain_state()
|
||||
state["acks"] = {"prgs-author-99": "acked", "prgs-author-2": "acknowledged"}
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(), drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
names = {c.name: c.passed for c in proof.checks}
|
||||
self.assertTrue(names[dp.CHECK_ACKS_OR_TIMEOUT])
|
||||
|
||||
|
||||
class VerifyDrainProofTests(unittest.TestCase):
|
||||
def _clean_proof_dict(self) -> dict:
|
||||
return dp.build_drain_proof(
|
||||
impact_report=_safe_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
|
||||
def test_missing_proof_is_invalid(self):
|
||||
result = dp.verify_drain_proof(None, now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertIsNone(result.proof_id)
|
||||
|
||||
def test_expired_proof_is_invalid(self):
|
||||
"""AC#4: an expired proof fails verification."""
|
||||
|
||||
proof = self._clean_proof_dict()
|
||||
later = NOW + timedelta(seconds=dp.DEFAULT_PROOF_TTL_SECONDS + 1)
|
||||
result = dp.verify_drain_proof(proof, now=later, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertTrue(result.expired)
|
||||
|
||||
def test_proof_valid_just_before_expiry(self):
|
||||
proof = self._clean_proof_dict()
|
||||
almost = NOW + timedelta(seconds=dp.DEFAULT_PROOF_TTL_SECONDS - 1)
|
||||
result = dp.verify_drain_proof(proof, now=almost, secret=SECRET)
|
||||
self.assertTrue(result.valid, result.reasons)
|
||||
|
||||
def test_wrong_secret_rejected(self):
|
||||
"""A proof minted in a prior process (different secret) will not verify."""
|
||||
|
||||
proof = self._clean_proof_dict()
|
||||
result = dp.verify_drain_proof(proof, now=NOW, secret=b"other-secret")
|
||||
self.assertFalse(result.valid)
|
||||
self.assertTrue(result.tampered)
|
||||
|
||||
def test_flipping_clean_flag_is_detected(self):
|
||||
"""Forging clean=True on an unclean proof breaks the signature."""
|
||||
|
||||
unclean = dp.build_drain_proof(
|
||||
impact_report=_unsafe_mutation_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
self.assertFalse(unclean["clean"])
|
||||
unclean["clean"] = True # forge
|
||||
result = dp.verify_drain_proof(unclean, now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertTrue(result.tampered)
|
||||
|
||||
def test_tampering_a_check_is_detected(self):
|
||||
unclean = dp.build_drain_proof(
|
||||
impact_report=_unsafe_mutation_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
for c in unclean["checks"]:
|
||||
if c["name"] == dp.CHECK_NO_INFLIGHT_MUTATIONS:
|
||||
c["passed"] = True # forge the failing check to pass
|
||||
result = dp.verify_drain_proof(unclean, now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertTrue(result.tampered)
|
||||
|
||||
def test_missing_required_check_rejected(self):
|
||||
proof = self._clean_proof_dict()
|
||||
proof["checks"] = [
|
||||
c for c in proof["checks"] if c["name"] != dp.CHECK_HANDOFFS_OK
|
||||
]
|
||||
result = dp.verify_drain_proof(proof, now=NOW, secret=SECRET)
|
||||
self.assertFalse(result.valid)
|
||||
|
||||
def test_stale_fingerprint_rejected(self):
|
||||
proof = self._clean_proof_dict()
|
||||
result = dp.verify_drain_proof(
|
||||
proof,
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint="deadbeef",
|
||||
)
|
||||
self.assertFalse(result.valid)
|
||||
|
||||
def test_matching_fingerprint_accepted(self):
|
||||
report = _safe_report()
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=report,
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
fp = dp.impact_fingerprint(report)
|
||||
result = dp.verify_drain_proof(
|
||||
proof, now=NOW, secret=SECRET, expected_impact_fingerprint=fp
|
||||
)
|
||||
self.assertTrue(result.valid, result.reasons)
|
||||
|
||||
|
||||
class GateApplyRestartTests(unittest.TestCase):
|
||||
def _clean_proof_dict(self) -> dict:
|
||||
return dp.build_drain_proof(
|
||||
impact_report=_safe_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
|
||||
def test_apply_without_proof_denied(self):
|
||||
"""AC#1: restart apply without a proof fails closed + raises incident."""
|
||||
|
||||
decision = dp.gate_apply_restart(proof=None, now=NOW, secret=SECRET)
|
||||
self.assertFalse(decision.allow)
|
||||
self.assertEqual(decision.verdict, dp.GATE_DENY)
|
||||
self.assertIsNotNone(decision.incident)
|
||||
self.assertEqual(
|
||||
decision.incident["kind"], "restart_drain_gate_denied"
|
||||
)
|
||||
|
||||
def test_apply_with_valid_proof_allowed(self):
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=self._clean_proof_dict(), now=NOW, secret=SECRET
|
||||
)
|
||||
self.assertTrue(decision.allow)
|
||||
self.assertEqual(decision.verdict, dp.GATE_ALLOW)
|
||||
self.assertIsNone(decision.incident)
|
||||
|
||||
def test_apply_with_expired_proof_denied_with_incident(self):
|
||||
later = NOW + timedelta(seconds=dp.DEFAULT_PROOF_TTL_SECONDS + 5)
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=self._clean_proof_dict(), now=later, secret=SECRET
|
||||
)
|
||||
self.assertFalse(decision.allow)
|
||||
self.assertIsNotNone(decision.incident)
|
||||
|
||||
def test_apply_with_unclean_proof_denied(self):
|
||||
"""AC#3 at the gate: an unsafe-mutation proof is denied."""
|
||||
|
||||
unclean = dp.build_drain_proof(
|
||||
impact_report=_unsafe_mutation_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
).as_dict()
|
||||
decision = dp.gate_apply_restart(proof=unclean, now=NOW, secret=SECRET)
|
||||
self.assertFalse(decision.allow)
|
||||
self.assertIsNotNone(decision.incident)
|
||||
|
||||
def test_break_glass_allows_without_proof_but_records_bypass(self):
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=None, now=NOW, secret=SECRET, break_glass=True
|
||||
)
|
||||
self.assertTrue(decision.allow)
|
||||
self.assertEqual(decision.verdict, dp.GATE_BREAK_GLASS)
|
||||
self.assertTrue(decision.break_glass)
|
||||
self.assertIsNone(decision.incident)
|
||||
self.assertTrue(decision.audit_record["break_glass"])
|
||||
|
||||
def test_denied_gate_carries_stale_fingerprint_reason(self):
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=self._clean_proof_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint="not-the-fingerprint",
|
||||
)
|
||||
self.assertFalse(decision.allow)
|
||||
|
||||
|
||||
class SecretHygieneTests(unittest.TestCase):
|
||||
def test_secret_never_serialized(self):
|
||||
proof = dp.build_drain_proof(
|
||||
impact_report=_safe_report(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
)
|
||||
blob = dp._canonical(proof.as_dict())
|
||||
self.assertNotIn(SECRET.decode(), blob)
|
||||
# The signature is a hex digest, not the raw secret.
|
||||
self.assertNotIn(SECRET.hex(), blob)
|
||||
|
||||
def test_incident_descriptor_has_no_secret(self):
|
||||
decision = dp.gate_apply_restart(proof=None, now=NOW, secret=SECRET)
|
||||
blob = dp._canonical(decision.incident)
|
||||
self.assertNotIn(SECRET.decode(), blob)
|
||||
|
||||
|
||||
def _drained_report_with_live_sessions(count: int) -> dict:
|
||||
"""Report with ``count`` other live sessions but nothing in flight.
|
||||
|
||||
Every other checklist item passes against this report, so a failure
|
||||
isolates the acknowledgement check rather than tripping on mutations.
|
||||
"""
|
||||
|
||||
sessions = [
|
||||
{
|
||||
"session_id": "prgs-controller-1-req",
|
||||
"role": "controller",
|
||||
"profile": "prgs-controller",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
}
|
||||
]
|
||||
for index in range(count):
|
||||
sessions.append(
|
||||
{
|
||||
"session_id": f"prgs-author-{index}",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
}
|
||||
)
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": sessions, "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id="prgs-controller-1-req",
|
||||
)
|
||||
return report.as_dict()
|
||||
|
||||
|
||||
class AcknowledgementFailClosedTests(unittest.TestCase):
|
||||
"""Acknowledgement evidence must fail closed unless explicitly verified.
|
||||
|
||||
Regression cover for the reviewed fail-open on PR #882: an absent ``acks``
|
||||
key collapsed to ``{}`` and was read as "no other live sessions required to
|
||||
acknowledge", so a proof minted clean and the restart gate allowed while the
|
||||
impact report still showed other live sessions.
|
||||
"""
|
||||
|
||||
def _state(self, **overrides) -> dict:
|
||||
state = _clean_drain_state()
|
||||
state.pop("acks", None)
|
||||
state["ack_timeout_policy_applied"] = False
|
||||
state.update(overrides)
|
||||
return state
|
||||
|
||||
def _acks_check(self, proof) -> dp.DrainCheck:
|
||||
return next(c for c in proof.checks if c.name == dp.CHECK_ACKS_OR_TIMEOUT)
|
||||
|
||||
def _build(self, report: dict, state: dict):
|
||||
return dp.build_drain_proof(
|
||||
impact_report=report, drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
|
||||
def assertAcksFailClosed(self, report: dict, state: dict) -> None:
|
||||
proof = self._build(report, state)
|
||||
self.assertFalse(self._acks_check(proof).passed)
|
||||
self.assertIn(dp.CHECK_ACKS_OR_TIMEOUT, proof.failed_checks)
|
||||
self.assertFalse(proof.clean)
|
||||
|
||||
# --- missing / null / empty / malformed ------------------------------
|
||||
|
||||
def test_missing_acks_key_with_live_sessions_fails_closed(self):
|
||||
"""The exact reviewed defect: absent key, three other live sessions."""
|
||||
report = _drained_report_with_live_sessions(3)
|
||||
self.assertEqual(report["counts"]["sessions_live_other"], 3)
|
||||
state = self._state()
|
||||
self.assertNotIn("acks", state)
|
||||
proof = self._build(report, state)
|
||||
check = self._acks_check(proof)
|
||||
self.assertFalse(check.passed)
|
||||
self.assertNotIn("no other live sessions", check.detail)
|
||||
self.assertIn("fail closed", check.detail)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertEqual(proof.failed_checks, [dp.CHECK_ACKS_OR_TIMEOUT])
|
||||
|
||||
def test_none_acks_with_live_sessions_fails_closed(self):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(2), self._state(acks=None)
|
||||
)
|
||||
|
||||
def test_empty_acks_with_live_sessions_fails_closed(self):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(1), self._state(acks={})
|
||||
)
|
||||
|
||||
def test_malformed_acks_fail_closed(self):
|
||||
for malformed in ([], "ack", 7, ("ack",), True):
|
||||
with self.subTest(malformed=malformed):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(1),
|
||||
self._state(acks=malformed),
|
||||
)
|
||||
|
||||
# --- stale / unproven values -----------------------------------------
|
||||
|
||||
def test_stale_or_unproven_ack_values_fail_closed(self):
|
||||
for value in ("pending", "stale", "unknown", "", None, True, 1, NOW):
|
||||
with self.subTest(value=value):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(1),
|
||||
self._state(acks={"prgs-author-0": value}),
|
||||
)
|
||||
|
||||
def test_partial_coverage_fails_closed(self):
|
||||
"""Fewer acknowledgements than the report's live-session count."""
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(3),
|
||||
self._state(acks={"prgs-author-0": "ack"}),
|
||||
)
|
||||
|
||||
def test_one_unacked_entry_among_many_fails_closed(self):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(2),
|
||||
self._state(acks={"prgs-author-0": "ack", "prgs-author-1": "pending"}),
|
||||
)
|
||||
|
||||
def test_unproven_live_session_count_fails_closed(self):
|
||||
"""A missing or malformed count cannot prove nobody had to acknowledge."""
|
||||
malformed_counts = (
|
||||
None,
|
||||
{},
|
||||
{"sessions_live_other": None},
|
||||
{"sessions_live_other": "3"},
|
||||
{"sessions_live_other": -1},
|
||||
{"sessions_live_other": True},
|
||||
)
|
||||
for counts in malformed_counts:
|
||||
with self.subTest(counts=counts):
|
||||
report = _drained_report_with_live_sessions(0)
|
||||
if counts is None:
|
||||
report.pop("counts", None)
|
||||
else:
|
||||
report["counts"] = counts
|
||||
self.assertAcksFailClosed(report, self._state())
|
||||
|
||||
# --- valid evidence still passes -------------------------------------
|
||||
|
||||
def test_complete_valid_acks_pass(self):
|
||||
report = _drained_report_with_live_sessions(2)
|
||||
state = self._state(
|
||||
acks={"prgs-author-0": "ack", "prgs-author-1": "acknowledged"}
|
||||
)
|
||||
proof = self._build(report, state)
|
||||
self.assertTrue(self._acks_check(proof).passed)
|
||||
self.assertTrue(proof.clean)
|
||||
self.assertEqual(proof.failed_checks, [])
|
||||
|
||||
def test_no_other_live_sessions_still_passes(self):
|
||||
"""Intended behavior retained: zero live sessions needs no acks."""
|
||||
report = _drained_report_with_live_sessions(0)
|
||||
self.assertEqual(report["counts"]["sessions_live_other"], 0)
|
||||
proof = self._build(report, self._state())
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertIn("sessions_live_other=0", check.detail)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
# --- timeout policy cannot become a second fail-open ------------------
|
||||
|
||||
def test_unproven_timeout_policy_cannot_open_the_gate(self):
|
||||
for value in (None, "true", "yes", 1, "True", [], {}):
|
||||
with self.subTest(value=value):
|
||||
self.assertAcksFailClosed(
|
||||
_drained_report_with_live_sessions(2),
|
||||
self._state(ack_timeout_policy_applied=value),
|
||||
)
|
||||
|
||||
def test_explicit_timeout_policy_permits(self):
|
||||
proof = self._build(
|
||||
_drained_report_with_live_sessions(2),
|
||||
self._state(ack_timeout_policy_applied=True),
|
||||
)
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertIn("timeout policy", check.detail)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
# --- the gate itself must deny ---------------------------------------
|
||||
|
||||
def test_failed_ack_check_denies_the_restart_gate(self):
|
||||
report = _drained_report_with_live_sessions(3)
|
||||
proof = self._build(report, self._state())
|
||||
self.assertFalse(proof.clean)
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=proof.as_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint=dp.impact_fingerprint(report),
|
||||
)
|
||||
self.assertFalse(decision.allow)
|
||||
self.assertEqual(decision.verdict, dp.GATE_DENY)
|
||||
self.assertIsNotNone(decision.incident)
|
||||
|
||||
def test_unclean_ack_proof_fails_verification(self):
|
||||
report = _drained_report_with_live_sessions(3)
|
||||
proof = self._build(report, self._state())
|
||||
result = dp.verify_drain_proof(
|
||||
proof.as_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint=dp.impact_fingerprint(report),
|
||||
)
|
||||
self.assertFalse(result.valid)
|
||||
self.assertFalse(result.clean)
|
||||
|
||||
|
||||
def _identity_report(*, requester: str, others: tuple[str, ...]) -> dict:
|
||||
"""Report with explicitly named requester and other live sessions.
|
||||
|
||||
Unlike :func:`_drained_report_with_live_sessions`, the session ids are
|
||||
chosen by the caller so a test can supply acknowledgements for the *wrong*
|
||||
identities while keeping the count correct.
|
||||
"""
|
||||
|
||||
sessions = [
|
||||
{
|
||||
"session_id": requester,
|
||||
"role": "controller",
|
||||
"profile": "prgs-controller",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
}
|
||||
]
|
||||
for session_id in others:
|
||||
sessions.append(
|
||||
{
|
||||
"session_id": session_id,
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
}
|
||||
)
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": sessions, "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id=requester,
|
||||
)
|
||||
return report.as_dict()
|
||||
|
||||
|
||||
class AcknowledgementIdentityBindingTests(unittest.TestCase):
|
||||
"""Acknowledgement coverage must be bound to session identity, not counted.
|
||||
|
||||
Regression cover for the second reviewed fail-open on PR #882 (review 582,
|
||||
blocker B1): coverage compared ``acked_count`` against
|
||||
``counts.sessions_live_other``, so acknowledgements supplied for the
|
||||
requesting session and for ids that do not exist satisfied the obligations
|
||||
of the live sessions that never answered. The required identities are
|
||||
carried by the report itself — ``ack_state`` keys and ``affected_sessions``
|
||||
filtered on ``live and not is_requester`` — and only an acknowledgement
|
||||
keyed by one of those ids may count for it.
|
||||
"""
|
||||
|
||||
def _state(self, **overrides) -> dict:
|
||||
state = _clean_drain_state()
|
||||
state.pop("acks", None)
|
||||
state["ack_timeout_policy_applied"] = False
|
||||
state.update(overrides)
|
||||
return state
|
||||
|
||||
def _acks_check(self, proof) -> dp.DrainCheck:
|
||||
return next(c for c in proof.checks if c.name == dp.CHECK_ACKS_OR_TIMEOUT)
|
||||
|
||||
def _build(self, report: dict, state: dict):
|
||||
return dp.build_drain_proof(
|
||||
impact_report=report, drain_state=state, now=NOW, secret=SECRET
|
||||
)
|
||||
|
||||
def assertAcksFailClosed(self, report: dict, state: dict) -> dp.DrainCheck:
|
||||
"""Failure must propagate through the check, the proof, and the gate."""
|
||||
|
||||
proof = self._build(report, state)
|
||||
check = self._acks_check(proof)
|
||||
self.assertFalse(check.passed)
|
||||
self.assertFalse(proof.clean)
|
||||
self.assertIn(dp.CHECK_ACKS_OR_TIMEOUT, proof.failed_checks)
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=proof.as_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint=dp.impact_fingerprint(report),
|
||||
)
|
||||
self.assertEqual(decision.verdict, dp.GATE_DENY)
|
||||
self.assertFalse(decision.allow)
|
||||
return check
|
||||
|
||||
# --- the reviewer's exact reproduction --------------------------------
|
||||
|
||||
def test_requester_plus_unknown_id_cannot_satisfy_two_live_sessions(self):
|
||||
"""Review 582 B1 verbatim: requester + a nonexistent session.
|
||||
|
||||
``sessions_live_other=2`` with ``ack_state`` naming ``other-0`` and
|
||||
``other-1``; the drain state supplies an acknowledgement from the
|
||||
requesting session itself and from a session that does not exist. The
|
||||
count matches, the identities do not.
|
||||
"""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
self.assertEqual(report["counts"]["sessions_live_other"], 2)
|
||||
self.assertEqual(
|
||||
report["ack_state"], {"other-0": "pending", "other-1": "pending"}
|
||||
)
|
||||
state = self._state(acks={"req": "ack", "totally-bogus-session": "ack"})
|
||||
check = self.assertAcksFailClosed(report, state)
|
||||
self.assertIn("other-0", check.detail)
|
||||
self.assertIn("other-1", check.detail)
|
||||
self.assertIn("fail closed", check.detail)
|
||||
|
||||
# --- wrong / unknown / requester identities ---------------------------
|
||||
|
||||
def test_sufficient_count_of_wrong_ids_fails_closed(self):
|
||||
"""Right cardinality, wrong identities: two acks, neither required."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
state = self._state(acks={"ghost-a": "ack", "ghost-b": "ack"})
|
||||
check = self.assertAcksFailClosed(report, state)
|
||||
self.assertIn("do not count", check.detail)
|
||||
|
||||
def test_more_acks_than_required_still_fails_on_wrong_ids(self):
|
||||
"""Coverage cannot be bought with volume: five acks, none required."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
state = self._state(acks={f"ghost-{i}": "acknowledged" for i in range(5)})
|
||||
self.assertAcksFailClosed(report, state)
|
||||
|
||||
def test_partial_identity_match_fails_closed(self):
|
||||
"""One required id acknowledged, the rest padded with unknown ids."""
|
||||
|
||||
report = _identity_report(
|
||||
requester="req", others=("other-0", "other-1", "other-2")
|
||||
)
|
||||
state = self._state(
|
||||
acks={"other-0": "ack", "ghost-1": "ack", "ghost-2": "ack"}
|
||||
)
|
||||
check = self.assertAcksFailClosed(report, state)
|
||||
self.assertIn("other-1", check.detail)
|
||||
self.assertIn("other-2", check.detail)
|
||||
|
||||
def test_requester_ack_never_satisfies_another_sessions_obligation(self):
|
||||
"""The requester is excluded from the required set and stays excluded."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
requester_rows = [s for s in report["affected_sessions"] if s["is_requester"]]
|
||||
self.assertEqual([s["session_id"] for s in requester_rows], ["req"])
|
||||
self.assertNotIn("req", report["ack_state"])
|
||||
check = self.assertAcksFailClosed(report, self._state(acks={"req": "ack"}))
|
||||
self.assertIn("other-0", check.detail)
|
||||
|
||||
def test_fabricated_ids_do_not_count_toward_coverage(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
for bogus in ("", " ", "other-0 extra", "OTHER-0", "other-01", "0"):
|
||||
with self.subTest(bogus=bogus):
|
||||
self.assertAcksFailClosed(report, self._state(acks={bogus: "ack"}))
|
||||
|
||||
# --- per-session state must be explicitly valid ------------------------
|
||||
|
||||
def test_unproven_per_session_states_fail_closed(self):
|
||||
"""A required id present but not explicitly acknowledged fails closed."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
for value in ("pending", "stale", "unknown", "", None, True, 1, NOW):
|
||||
with self.subTest(value=value):
|
||||
self.assertAcksFailClosed(
|
||||
report,
|
||||
self._state(acks={"other-0": "ack", "other-1": value}),
|
||||
)
|
||||
|
||||
def test_report_ack_state_placeholder_is_never_read_as_an_ack(self):
|
||||
"""``ack_state`` values are the report's own placeholders, not evidence."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report["ack_state"] = {"other-0": "ack"}
|
||||
self.assertAcksFailClosed(report, self._state())
|
||||
|
||||
# --- missing / malformed / contradictory identity evidence -------------
|
||||
|
||||
def test_missing_identity_evidence_fails_closed(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report.pop("ack_state", None)
|
||||
report.pop("affected_sessions", None)
|
||||
check = self.assertAcksFailClosed(report, self._state(acks={"other-0": "ack"}))
|
||||
self.assertIn("no session-identity evidence", check.detail)
|
||||
|
||||
def test_malformed_ack_state_fails_closed(self):
|
||||
for malformed in ([], "other-0", 7, None, ("other-0",)):
|
||||
with self.subTest(malformed=malformed):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report["ack_state"] = malformed
|
||||
self.assertAcksFailClosed(
|
||||
report, self._state(acks={"other-0": "ack"})
|
||||
)
|
||||
|
||||
def test_non_string_ack_state_key_fails_closed(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report["ack_state"] = {7: "pending"}
|
||||
self.assertAcksFailClosed(report, self._state(acks={"other-0": "ack"}))
|
||||
|
||||
def test_malformed_affected_sessions_fails_closed(self):
|
||||
for malformed in ("sessions", 7, {"session_id": "other-0"}, [None], [7]):
|
||||
with self.subTest(malformed=malformed):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report.pop("ack_state", None)
|
||||
report["affected_sessions"] = malformed
|
||||
self.assertAcksFailClosed(
|
||||
report, self._state(acks={"other-0": "ack"})
|
||||
)
|
||||
|
||||
def test_affected_sessions_without_explicit_booleans_fails_closed(self):
|
||||
"""``live``/``is_requester`` must be real booleans, never inferred."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report.pop("ack_state", None)
|
||||
for row in report["affected_sessions"]:
|
||||
if row["session_id"] == "other-0":
|
||||
row["is_requester"] = "false"
|
||||
self.assertAcksFailClosed(report, self._state(acks={"other-0": "ack"}))
|
||||
|
||||
def test_affected_sessions_missing_live_flag_fails_closed(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report.pop("ack_state", None)
|
||||
for row in report["affected_sessions"]:
|
||||
row.pop("live", None)
|
||||
self.assertAcksFailClosed(report, self._state(acks={"other-0": "ack"}))
|
||||
|
||||
def test_contradictory_ack_state_and_affected_sessions_fails_closed(self):
|
||||
"""Both views present and disagreeing is unresolvable, not a tie-break."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
report["ack_state"] = {"other-0": "pending", "other-9": "pending"}
|
||||
check = self.assertAcksFailClosed(
|
||||
report, self._state(acks={"other-0": "ack", "other-9": "ack"})
|
||||
)
|
||||
self.assertIn("contradicts itself", check.detail)
|
||||
|
||||
def test_identity_count_mismatch_fails_closed(self):
|
||||
"""Identity evidence that cannot be reconciled with the count denies."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
report["counts"] = dict(report["counts"], sessions_live_other=1)
|
||||
check = self.assertAcksFailClosed(
|
||||
report, self._state(acks={"other-0": "ack", "other-1": "ack"})
|
||||
)
|
||||
self.assertIn("cannot be reconciled", check.detail)
|
||||
|
||||
def test_broken_identity_evidence_outranks_timeout_policy(self):
|
||||
"""The sanctioned timeout path cannot paper over an unreadable report."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
report["ack_state"] = "not-a-mapping"
|
||||
self.assertAcksFailClosed(report, self._state(ack_timeout_policy_applied=True))
|
||||
|
||||
# --- legitimate success is preserved -----------------------------------
|
||||
|
||||
def test_every_required_session_acknowledged_passes(self):
|
||||
report = _identity_report(
|
||||
requester="req", others=("other-0", "other-1", "other-2")
|
||||
)
|
||||
state = self._state(
|
||||
acks={
|
||||
"other-0": "ack",
|
||||
"other-1": "acked",
|
||||
"other-2": "acknowledged",
|
||||
}
|
||||
)
|
||||
proof = self._build(report, state)
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertTrue(proof.clean)
|
||||
self.assertEqual(proof.failed_checks, [])
|
||||
self.assertIn("acknowledged by identity", check.detail)
|
||||
decision = dp.gate_apply_restart(
|
||||
proof=proof.as_dict(),
|
||||
now=NOW,
|
||||
secret=SECRET,
|
||||
expected_impact_fingerprint=dp.impact_fingerprint(report),
|
||||
)
|
||||
self.assertEqual(decision.verdict, dp.GATE_ALLOW)
|
||||
self.assertTrue(decision.allow)
|
||||
|
||||
def test_required_session_ack_tolerates_surrounding_whitespace(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
proof = self._build(report, self._state(acks={" other-0 ": " ACK "}))
|
||||
self.assertTrue(self._acks_check(proof).passed)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
def test_no_other_live_sessions_still_passes_with_identity_evidence(self):
|
||||
report = _identity_report(requester="req", others=())
|
||||
self.assertEqual(report["counts"]["sessions_live_other"], 0)
|
||||
self.assertEqual(report["ack_state"], {})
|
||||
proof = self._build(report, self._state())
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertIn("sessions_live_other=0", check.detail)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
def test_explicit_timeout_policy_retains_intended_behavior(self):
|
||||
"""Valid, correctly typed timeout evidence still permits the check."""
|
||||
|
||||
report = _identity_report(requester="req", others=("other-0", "other-1"))
|
||||
proof = self._build(report, self._state(ack_timeout_policy_applied=True))
|
||||
check = self._acks_check(proof)
|
||||
self.assertTrue(check.passed)
|
||||
self.assertIn("timeout policy", check.detail)
|
||||
self.assertTrue(proof.clean)
|
||||
|
||||
def test_timeout_policy_still_strictly_typed_under_identity_binding(self):
|
||||
report = _identity_report(requester="req", others=("other-0",))
|
||||
for value in (None, "true", "True", 1, [], {}):
|
||||
with self.subTest(value=value):
|
||||
self.assertAcksFailClosed(
|
||||
report, self._state(ack_timeout_policy_applied=value)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,378 @@
|
||||
"""Allocator semantic container exclusion for vision/roadmap/umbrella (#854).
|
||||
|
||||
#844 excluded epic / child-only containers (live #631) but product-vision
|
||||
(#652), phased-roadmap (#653), and umbrella (#655) coordination records still
|
||||
ranked as implementable work. This module is the live-equivalent canary:
|
||||
|
||||
* #631 / #652 / #653 / #655-shaped records are all excluded in one inventory.
|
||||
* Independently executable children remain eligible and can be selected.
|
||||
* Ordinary issues that merely mention vision / roadmap / umbrella stay eligible.
|
||||
* Excluded containers never receive assignments or workflow leases.
|
||||
* Structured skip reason ``epic_or_child_only_container`` is reported.
|
||||
* Candidate-set fingerprint remains stable after exclusions.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
import unittest
|
||||
|
||||
from allocator_service import (
|
||||
OUTCOME_ASSIGNED,
|
||||
OUTCOME_PREVIEW,
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER,
|
||||
WorkCandidate,
|
||||
allocate_next_work,
|
||||
candidate_set_fingerprint,
|
||||
classify_epic_or_child_only_container,
|
||||
)
|
||||
from control_plane_db import ControlPlaneDB
|
||||
|
||||
REMOTE = "prgs"
|
||||
ORG = "Scaled-Tech-Consulting"
|
||||
REPO = "Gitea-Tools"
|
||||
|
||||
# Minimal bodies mirroring live coordination records (not full issue text).
|
||||
_EPIC_631_BODY = """
|
||||
## Scope (umbrella)
|
||||
|
||||
This epic owns the **product roadmap and linkage** for the Web Console.
|
||||
Implementation is delivered via child issues only.
|
||||
|
||||
## Explicit non-goals
|
||||
|
||||
* Do not implement product features in this epic issue itself.
|
||||
* No product feature implementation is claimed complete solely on this epic.
|
||||
"""
|
||||
|
||||
_VISION_652_BODY = """
|
||||
## Canonical product vision — enduring source of truth
|
||||
|
||||
**This issue is the enduring source of truth for the MCP Control Plane Web Console product vision.**
|
||||
|
||||
## Implementation linkage
|
||||
|
||||
* **Do not implement features on this issue.**
|
||||
* Sequencing: roadmap issue + #631 children.
|
||||
|
||||
## Canonical issue state
|
||||
|
||||
```text
|
||||
STATE: vision-active
|
||||
WHO_IS_NEXT: controller (triage/ordering) / author (implementation of linked children only)
|
||||
```
|
||||
"""
|
||||
|
||||
_ROADMAP_653_BODY = """
|
||||
## Purpose
|
||||
|
||||
This issue is the **phased delivery roadmap and epic sequencing** for the MCP Control Plane Web Console.
|
||||
|
||||
## Non-goals
|
||||
|
||||
* Implementing features on this roadmap issue.
|
||||
* Deleting vision items by omitting them from phases without #652 change log.
|
||||
|
||||
## Canonical issue state
|
||||
|
||||
```text
|
||||
STATE: roadmap-active
|
||||
WHO_IS_NEXT: author
|
||||
```
|
||||
"""
|
||||
|
||||
_UMBRELLA_655_BODY = """
|
||||
## Scope (umbrella)
|
||||
|
||||
This issue owns the **canonical restart-governance program**. Implementation is via linked children only.
|
||||
|
||||
## Acceptance criteria (umbrella)
|
||||
|
||||
6. No product feature claimed complete on this issue alone.
|
||||
"""
|
||||
|
||||
_CHILD_BODY = """
|
||||
## Problem
|
||||
|
||||
Operators need a workflow-event timeline model for Phase 1.
|
||||
|
||||
## Acceptance criteria
|
||||
|
||||
- [ ] Timeline model API exists
|
||||
"""
|
||||
|
||||
|
||||
def _issue(
|
||||
number: int,
|
||||
*,
|
||||
title: str = "",
|
||||
body: str = "",
|
||||
labels: tuple[str, ...] = ("status:ready", "type:feature"),
|
||||
priority: int = 20,
|
||||
) -> WorkCandidate:
|
||||
return WorkCandidate(
|
||||
kind="issue",
|
||||
number=number,
|
||||
state="open",
|
||||
labels=labels,
|
||||
title=title or f"issue {number}",
|
||||
body=body,
|
||||
priority=priority,
|
||||
)
|
||||
|
||||
|
||||
def _live_shaped_containers() -> list[WorkCandidate]:
|
||||
return [
|
||||
_issue(
|
||||
631,
|
||||
title="Epic: MCP Control Plane Web Console",
|
||||
body=_EPIC_631_BODY,
|
||||
),
|
||||
_issue(
|
||||
652,
|
||||
title="Product vision: MCP Control Plane Web Console (canonical)",
|
||||
body=_VISION_652_BODY,
|
||||
),
|
||||
_issue(
|
||||
653,
|
||||
title="Roadmap: MCP Control Plane Web Console (phased delivery)",
|
||||
body=_ROADMAP_653_BODY,
|
||||
),
|
||||
_issue(
|
||||
655,
|
||||
title="Umbrella: Governed MCP restart coordination and zero-disruption recovery",
|
||||
body=_UMBRELLA_655_BODY,
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
class ClassifySemanticContainersTest(unittest.TestCase):
|
||||
def test_652_vision_is_container(self) -> None:
|
||||
c = _issue(
|
||||
652,
|
||||
title="Product vision: MCP Control Plane Web Console (canonical)",
|
||||
body=_VISION_652_BODY,
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIsNotNone(detail)
|
||||
self.assertIn("body_marker", detail or "")
|
||||
|
||||
def test_653_roadmap_is_container(self) -> None:
|
||||
c = _issue(
|
||||
653,
|
||||
title="Roadmap: MCP Control Plane Web Console (phased delivery)",
|
||||
body=_ROADMAP_653_BODY,
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIn("body_marker", detail or "")
|
||||
|
||||
def test_655_umbrella_is_container(self) -> None:
|
||||
c = _issue(
|
||||
655,
|
||||
title="Umbrella: Governed MCP restart coordination and zero-disruption recovery",
|
||||
body=_UMBRELLA_655_BODY,
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIn("body_marker", detail or "")
|
||||
|
||||
def test_631_still_container_after_854(self) -> None:
|
||||
c = _issue(
|
||||
631,
|
||||
title="Epic: MCP Control Plane Web Console",
|
||||
body=_EPIC_631_BODY,
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIn("body_marker", detail or "")
|
||||
|
||||
def test_incidental_vision_roadmap_umbrella_words_not_container(self) -> None:
|
||||
cases = (
|
||||
(
|
||||
"Document vision handoff conventions",
|
||||
"Update docs so implementable issues that mention a vision "
|
||||
"remain independently executable.",
|
||||
),
|
||||
(
|
||||
"Clarify roadmap sequencing notes",
|
||||
"Write a short note about how the roadmap issue relates to children.",
|
||||
),
|
||||
(
|
||||
"Umbrella recovery checklist for authors",
|
||||
"Authors should still implement the concrete recovery fix here.",
|
||||
),
|
||||
(
|
||||
"Product vision wording in the help text",
|
||||
"Fix a typo in the operator-facing help string that says product vision.",
|
||||
),
|
||||
)
|
||||
for title, body in cases:
|
||||
with self.subTest(title=title):
|
||||
c = _issue(900, title=title, body=body)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertFalse(is_c)
|
||||
self.assertIsNone(detail)
|
||||
|
||||
def test_title_prefix_alone_not_container(self) -> None:
|
||||
for title in (
|
||||
"Epic: something mentioned only in title",
|
||||
"Roadmap: title only without body scope",
|
||||
"Product vision: title only without body scope",
|
||||
"Umbrella: title only without body scope",
|
||||
):
|
||||
with self.subTest(title=title):
|
||||
c = _issue(
|
||||
901,
|
||||
title=title,
|
||||
body="Implement a concrete fix for the allocator skip list.",
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertFalse(is_c)
|
||||
self.assertIsNone(detail)
|
||||
|
||||
def test_roadmap_label_alone_is_container(self) -> None:
|
||||
c = _issue(
|
||||
902,
|
||||
title="Console delivery sequencing",
|
||||
body="Track phased delivery only.",
|
||||
labels=("status:ready", "type:roadmap"),
|
||||
)
|
||||
is_c, detail = classify_epic_or_child_only_container(c)
|
||||
self.assertTrue(is_c)
|
||||
self.assertIn("type:roadmap", detail or "")
|
||||
|
||||
def test_child_referencing_parent_policy_stays_eligible(self) -> None:
|
||||
"""Children may quote parent policy without becoming containers."""
|
||||
c = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=(
|
||||
_CHILD_BODY
|
||||
+ "\n\nParent #652 says do not implement on the vision issue; "
|
||||
"this child is the implementable unit."
|
||||
),
|
||||
)
|
||||
is_c, _ = classify_epic_or_child_only_container(c)
|
||||
self.assertFalse(is_c)
|
||||
|
||||
|
||||
class AllocateSemanticContainerExclusionTest(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.db = ControlPlaneDB(os.path.join(self._tmp.name, "cp.sqlite3"))
|
||||
|
||||
def _alloc(self, candidates, **kwargs):
|
||||
defaults = dict(
|
||||
session_id="sess-854",
|
||||
role="author",
|
||||
remote=REMOTE,
|
||||
org=ORG,
|
||||
repo=REPO,
|
||||
profile_name="prgs-author",
|
||||
username="jcwalker3",
|
||||
claims={},
|
||||
apply=False,
|
||||
)
|
||||
defaults.update(kwargs)
|
||||
return allocate_next_work(self.db, candidates=candidates, **defaults)
|
||||
|
||||
def test_live_equivalent_canary_excludes_all_containers_selects_child(self) -> None:
|
||||
containers = _live_shaped_containers()
|
||||
child = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=_CHILD_BODY,
|
||||
)
|
||||
inventory = containers + [child]
|
||||
res = self._alloc(inventory, apply=False)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["outcome"], OUTCOME_PREVIEW)
|
||||
self.assertEqual(res["selected"]["number"], 637)
|
||||
|
||||
skipped = {s["number"]: s for s in res["skipped"]}
|
||||
for number in (631, 652, 653, 655):
|
||||
self.assertIn(number, skipped, res["skipped"])
|
||||
self.assertEqual(
|
||||
skipped[number]["reason_code"],
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER,
|
||||
)
|
||||
self.assertIn(
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER, skipped[number]["reason"]
|
||||
)
|
||||
|
||||
def test_containers_cannot_receive_assignment_or_lease(self) -> None:
|
||||
containers = _live_shaped_containers()
|
||||
res = self._alloc(containers, apply=True)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertNotEqual(res["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertIsNone(res.get("assignment"))
|
||||
self.assertIsNone(res.get("selected"))
|
||||
skipped = {s["number"]: s for s in res["skipped"]}
|
||||
for number in (631, 652, 653, 655):
|
||||
self.assertEqual(
|
||||
skipped[number]["reason_code"],
|
||||
SKIP_EPIC_OR_CHILD_ONLY_CONTAINER,
|
||||
)
|
||||
|
||||
leases = []
|
||||
if hasattr(self.db, "list_active_leases"):
|
||||
leases = self.db.list_active_leases(
|
||||
remote=REMOTE, org=ORG, repo=REPO
|
||||
)
|
||||
if not leases and hasattr(self.db, "list_leases"):
|
||||
leases = self.db.list_leases(remote=REMOTE, org=ORG, repo=REPO)
|
||||
for lease in leases or []:
|
||||
work_number = (
|
||||
lease.get("work_number") if isinstance(lease, dict) else None
|
||||
)
|
||||
self.assertNotIn(work_number, {631, 652, 653, 655})
|
||||
|
||||
def test_apply_selects_child_not_container(self) -> None:
|
||||
containers = _live_shaped_containers()
|
||||
child = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=_CHILD_BODY,
|
||||
)
|
||||
res = self._alloc(containers + [child], apply=True)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["outcome"], OUTCOME_ASSIGNED)
|
||||
self.assertEqual(res["selected"]["number"], 637)
|
||||
self.assertEqual(res["assignment"]["work_number"], 637)
|
||||
|
||||
def test_fingerprint_stable_with_containers_present(self) -> None:
|
||||
containers = _live_shaped_containers()
|
||||
child = _issue(
|
||||
637,
|
||||
title="Web Console: Workflow-event timeline model (Phase 1)",
|
||||
body=_CHILD_BODY,
|
||||
)
|
||||
inventory = containers + [child]
|
||||
fp_before = candidate_set_fingerprint(inventory)
|
||||
res = self._alloc(inventory, apply=False)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["selected"]["number"], 637)
|
||||
# Allocator reports the same CAS fingerprint for the full candidate set.
|
||||
reported = res.get("candidate_set_fingerprint")
|
||||
self.assertEqual(reported, fp_before)
|
||||
# Re-fingerprint of the same inventory is byte-stable.
|
||||
self.assertEqual(candidate_set_fingerprint(inventory), fp_before)
|
||||
|
||||
def test_incidental_mentions_remain_eligible(self) -> None:
|
||||
ordinary = _issue(
|
||||
700,
|
||||
title="Document roadmap handoff conventions",
|
||||
body="Write runbook text about vision vs roadmap vs child issues.",
|
||||
)
|
||||
res = self._alloc([ordinary], apply=False)
|
||||
self.assertTrue(res["success"], res)
|
||||
self.assertEqual(res["selected"]["number"], 700)
|
||||
self.assertEqual(res["skipped"], [])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,395 @@
|
||||
"""``apply_authorized`` requires BOTH authorizations (#886 review blocker B1).
|
||||
|
||||
The #663 restart-class matrix and the #661 drain-proof hard gate are independent
|
||||
authorizations that first coexisted when PR #882 landed on master and PR #886
|
||||
merged it into the restart-class branch. The union preserved both, but the apply
|
||||
decision consulted only the drain gate::
|
||||
|
||||
payload["apply_authorized"] = gate.allow # pre-fix
|
||||
|
||||
so a clean drain proof — or an authorized break-glass, which needs no proof at
|
||||
all — reported ``apply_authorized: True`` for a restart class the least-privilege
|
||||
matrix had just denied, in the same payload that carried
|
||||
``allow_restart: False`` and "role 'author' may not request full_mcp_restart".
|
||||
|
||||
These tests pin the conjunction and the properties that must survive it. They
|
||||
exercise the real MCP tool, which previously had no test coverage at all — that
|
||||
absence is why the defect shipped.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
import drain_proof
|
||||
import gitea_mcp_server as srv
|
||||
|
||||
CONTROLLER_APPROVAL_ENV = "GITEA_CONTROLLER_RESTART_APPROVAL_AUTHORIZATION"
|
||||
BREAK_GLASS_ENV = "GITEA_BREAKGLASS_RESTART_AUTHORIZATION"
|
||||
|
||||
# A quiet control plane: nothing live, so the blast radius never masks the
|
||||
# authorization outcome under test.
|
||||
QUIET_SESSIONS: list[dict] = []
|
||||
QUIET_LEASES: list[dict] = []
|
||||
|
||||
# #669: broad restarts need a prior narrow-attempt log (unless break-glass).
|
||||
PRIOR_NARROW_ATTEMPTS_JSON = json.dumps(
|
||||
[
|
||||
{
|
||||
"action": "client_reconnect",
|
||||
"outcome": "insufficient",
|
||||
"reason": "still flapping after reconnect",
|
||||
}
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
class _FakeDB:
|
||||
"""Minimal control-plane DB stand-in for the restart inventory."""
|
||||
|
||||
def __init__(self, sessions=QUIET_SESSIONS, terminal=None):
|
||||
self._sessions = list(sessions)
|
||||
self._terminal = terminal
|
||||
|
||||
def list_sessions(self, statuses=None, limit=None):
|
||||
return list(self._sessions)
|
||||
|
||||
def get_active_terminal_lock(self, remote=None, org=None, repo=None):
|
||||
return self._terminal
|
||||
|
||||
|
||||
def _profile(role: str) -> dict:
|
||||
return {"profile_name": f"prgs-{role}", "role_kind": role, "role": role}
|
||||
|
||||
|
||||
class _RestartToolHarness(unittest.TestCase):
|
||||
"""Drives the real ``gitea_request_mcp_restart`` with a stubbed inventory."""
|
||||
|
||||
def _call(self, *, role: str, env: dict | None = None, **kwargs) -> dict:
|
||||
environ = {k: v for k, v in os.environ.items()
|
||||
if k not in (CONTROLLER_APPROVAL_ENV, BREAK_GLASS_ENV)}
|
||||
environ.update(env or {})
|
||||
with patch.object(srv, "_profile_operation_gate", return_value=None), \
|
||||
patch.object(srv, "_resolve",
|
||||
return_value=("gitea.prgs.cc",
|
||||
"Scaled-Tech-Consulting",
|
||||
"Gitea-Tools")), \
|
||||
patch.object(srv, "get_profile", return_value=_profile(role)), \
|
||||
patch.object(srv, "_control_plane_db_or_error",
|
||||
return_value=(_FakeDB(), [])), \
|
||||
patch.object(srv.lease_lifecycle, "list_active_leases",
|
||||
return_value={"leases": list(QUIET_LEASES)}), \
|
||||
patch.dict(os.environ, environ, clear=True):
|
||||
return srv.gitea_request_mcp_restart(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
session_id="probe-session",
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
def _clean_proof_for(self, preview: dict) -> str:
|
||||
"""Mint a genuinely clean, signature-valid proof bound to *preview*.
|
||||
|
||||
Built from the tool's own dry-run report, so the fingerprint matches and
|
||||
the proof is rejected for authorization reasons only — never because it
|
||||
was stale or forged.
|
||||
"""
|
||||
proof = drain_proof.build_drain_proof(
|
||||
impact_report=preview,
|
||||
drain_state={
|
||||
"assignments_stopped": True,
|
||||
"checkpoints_complete": True,
|
||||
"handoffs_verified": True,
|
||||
"leases_handled": True,
|
||||
"acks": {},
|
||||
},
|
||||
requesting_session_id="probe-session",
|
||||
)
|
||||
self.assertTrue(proof.clean, "harness must mint a clean proof")
|
||||
return json.dumps(proof.as_dict())
|
||||
|
||||
|
||||
class TestConjunction(_RestartToolHarness):
|
||||
"""AC1/AC2 — the two authorizations are ANDed, in both directions."""
|
||||
|
||||
def test_gate_allow_with_class_denied_yields_apply_authorized_false(self):
|
||||
# An author may not request full_mcp_restart (CONTROL_ROLES only).
|
||||
preview = self._call(role="author", restart_class="full_mcp_restart")
|
||||
self.assertFalse(preview["allow_restart"])
|
||||
|
||||
result = self._call(
|
||||
role="author",
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
)
|
||||
|
||||
self.assertTrue(result["apply_gate"]["drain_gate_allow"],
|
||||
"drain gate itself should have allowed this proof")
|
||||
self.assertFalse(result["apply_gate"]["restart_class_authorized"])
|
||||
self.assertFalse(result["apply_authorized"],
|
||||
"a clean proof must not authorize a denied class")
|
||||
self.assertFalse(result["allow_restart"])
|
||||
|
||||
def test_gate_allow_with_class_allowed_can_yield_apply_authorized_true(self):
|
||||
preview = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
self.assertTrue(preview["allow_restart"],
|
||||
"operator + controller approval must authorize the class")
|
||||
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
|
||||
self.assertTrue(result["apply_gate"]["drain_gate_allow"])
|
||||
self.assertTrue(result["apply_gate"]["restart_class_authorized"])
|
||||
self.assertTrue(result["apply_authorized"],
|
||||
"both authorizations pass; apply must be authorized")
|
||||
|
||||
def test_denial_is_attributable_to_the_authorization_that_caused_it(self):
|
||||
preview = self._call(role="author", restart_class="full_mcp_restart")
|
||||
result = self._call(
|
||||
role="author",
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
)
|
||||
blob = " ".join(result["apply_gate"]["reasons"]).lower()
|
||||
self.assertIn("restart class authorization denied", blob)
|
||||
self.assertIn("full_mcp_restart", blob)
|
||||
|
||||
|
||||
class TestProofCannotOverrideAuthorization(_RestartToolHarness):
|
||||
"""AC3 — a clean proof never overrides a class or requester-role denial."""
|
||||
|
||||
def test_clean_proof_cannot_override_role_denial(self):
|
||||
for role in ("author", "reviewer", "merger", "reconciler"):
|
||||
with self.subTest(role=role):
|
||||
preview = self._call(role=role, restart_class="full_mcp_restart")
|
||||
result = self._call(
|
||||
role=role,
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
)
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
|
||||
def test_clean_proof_cannot_override_missing_controller_approval(self):
|
||||
# Correct role, but the class demands controller approval and the
|
||||
# environment carries none.
|
||||
preview = self._call(role="operator", restart_class="full_mcp_restart")
|
||||
self.assertFalse(preview["allow_restart"])
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
)
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
|
||||
def test_clean_proof_cannot_override_unknown_class(self):
|
||||
preview = self._call(role="operator", restart_class="not_a_real_class",
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"})
|
||||
self.assertFalse(preview["allow_restart"])
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="not_a_real_class",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"},
|
||||
)
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
|
||||
def test_clean_proof_cannot_override_missing_scope_target(self):
|
||||
# worker_restart without target_session_id fails closed on scoping.
|
||||
preview = self._call(role="operator", restart_class="worker_restart",
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"})
|
||||
self.assertFalse(preview["allow_restart"])
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="worker_restart",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"},
|
||||
)
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
|
||||
|
||||
class TestBreakGlassDoesNotCollapseTheMatrix(_RestartToolHarness):
|
||||
"""AC4 — break-glass bypasses the drain proof only, never the class matrix."""
|
||||
|
||||
def test_break_glass_does_not_authorize_a_denied_class(self):
|
||||
result = self._call(
|
||||
role="author",
|
||||
restart_class="host_restart",
|
||||
dry_run=False,
|
||||
request_break_glass=True,
|
||||
env={BREAK_GLASS_ENV: "operator-issued"},
|
||||
)
|
||||
self.assertTrue(result["break_glass_authorized"])
|
||||
self.assertTrue(result["apply_gate"]["drain_gate_allow"],
|
||||
"break-glass does satisfy the drain gate")
|
||||
self.assertFalse(result["apply_gate"]["restart_class_authorized"])
|
||||
self.assertFalse(result["apply_authorized"],
|
||||
"break-glass must not collapse the class matrix")
|
||||
|
||||
def test_break_glass_across_every_worker_role_and_restricted_class(self):
|
||||
for role in ("author", "reviewer", "merger", "reconciler"):
|
||||
for klass in ("rolling_mcp_restart", "full_mcp_restart",
|
||||
"host_restart"):
|
||||
with self.subTest(role=role, restart_class=klass):
|
||||
result = self._call(
|
||||
role=role,
|
||||
restart_class=klass,
|
||||
dry_run=False,
|
||||
request_break_glass=True,
|
||||
env={BREAK_GLASS_ENV: "operator-issued"},
|
||||
)
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
|
||||
def test_break_glass_still_works_when_the_class_is_authorized(self):
|
||||
# Break-glass keeps its purpose: skipping the drain proof for a caller
|
||||
# the matrix does allow.
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
request_break_glass=True,
|
||||
env={BREAK_GLASS_ENV: "operator-issued",
|
||||
CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
self.assertTrue(result["apply_authorized"])
|
||||
self.assertEqual(result["apply_gate"]["verdict"], "break_glass")
|
||||
|
||||
def test_break_glass_is_not_self_assertable(self):
|
||||
# Requested but no environment authorization -> no bypass, and the
|
||||
# unproven apply is denied.
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
request_break_glass=True,
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
self.assertTrue(result["break_glass_requested"])
|
||||
self.assertFalse(result["break_glass_authorized"])
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
self.assertIn("incident", result)
|
||||
|
||||
|
||||
class TestRestrictedClassesStayDenied(_RestartToolHarness):
|
||||
"""AC5 — restricted classes remain denied to unauthorized requesters."""
|
||||
|
||||
def test_restricted_classes_denied_for_worker_roles(self):
|
||||
for role in ("author", "reviewer", "merger", "reconciler"):
|
||||
for klass in ("rolling_mcp_restart", "full_mcp_restart",
|
||||
"host_restart"):
|
||||
with self.subTest(role=role, restart_class=klass):
|
||||
preview = self._call(
|
||||
role=role,
|
||||
restart_class=klass,
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"},
|
||||
)
|
||||
self.assertFalse(preview["allow_restart"])
|
||||
self.assertFalse(preview["permission_authorized"])
|
||||
self.assertFalse(preview["role_authorized"])
|
||||
|
||||
def test_host_restart_needs_controller_and_infrastructure_operator(self):
|
||||
# controller approval alone is not enough for host_restart.
|
||||
preview = self._call(role="controller", restart_class="host_restart",
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"})
|
||||
self.assertFalse(preview["approval_satisfied"])
|
||||
self.assertFalse(preview["allow_restart"])
|
||||
|
||||
|
||||
class TestExistingPathsStillWork(_RestartToolHarness):
|
||||
"""AC6 — valid scoped and unscoped restart paths are unaffected."""
|
||||
|
||||
def test_dry_run_never_reports_apply_authorization(self):
|
||||
result = self._call(role="operator", restart_class="full_mcp_restart",
|
||||
env={CONTROLLER_APPROVAL_ENV: "yes"})
|
||||
self.assertNotIn("apply_authorized", result)
|
||||
self.assertNotIn("apply_gate", result)
|
||||
self.assertFalse(result["apply_supported"])
|
||||
self.assertFalse(result["restart_performed"])
|
||||
|
||||
def test_self_service_unscoped_classes_authorize_for_every_role(self):
|
||||
for role in ("author", "reviewer", "merger", "reconciler",
|
||||
"controller", "operator", "admin"):
|
||||
for klass in ("client_reconnect", "session_reconnect"):
|
||||
with self.subTest(role=role, restart_class=klass):
|
||||
preview = self._call(role=role, restart_class=klass)
|
||||
self.assertTrue(preview["allow_restart"])
|
||||
|
||||
def test_scoped_class_with_target_authorizes_and_applies(self):
|
||||
env = {CONTROLLER_APPROVAL_ENV: "operator-approved"}
|
||||
preview = self._call(role="operator", restart_class="worker_restart",
|
||||
target_session_id="worker-1", env=env)
|
||||
self.assertTrue(preview["allow_restart"])
|
||||
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="worker_restart",
|
||||
target_session_id="worker-1",
|
||||
dry_run=False,
|
||||
drain_proof_json=self._clean_proof_for(preview),
|
||||
env=env,
|
||||
)
|
||||
self.assertTrue(result["apply_authorized"])
|
||||
|
||||
def test_apply_still_denies_without_any_proof(self):
|
||||
# The #661 hard gate is untouched by the conjunction.
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||
dry_run=False,
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
self.assertFalse(result["apply_gate"]["drain_gate_allow"])
|
||||
self.assertTrue(result["apply_gate"]["restart_class_authorized"])
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
self.assertEqual(result["incident"]["kind"], "restart_drain_gate_denied")
|
||||
|
||||
def test_apply_denies_on_malformed_proof(self):
|
||||
result = self._call(
|
||||
role="operator",
|
||||
restart_class="full_mcp_restart",
|
||||
dry_run=False,
|
||||
drain_proof_json="{not valid json",
|
||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||
)
|
||||
self.assertFalse(result["apply_authorized"])
|
||||
self.assertTrue(any("invalid drain_proof_json" in reason
|
||||
for reason in result["apply_gate"]["reasons"]))
|
||||
|
||||
def test_tool_never_restarts_on_any_path(self):
|
||||
for kwargs in (
|
||||
{"restart_class": "client_reconnect"},
|
||||
{"restart_class": "full_mcp_restart", "dry_run": False},
|
||||
{"restart_class": "host_restart", "dry_run": False,
|
||||
"request_break_glass": True},
|
||||
):
|
||||
with self.subTest(**kwargs):
|
||||
result = self._call(role="operator", env={
|
||||
CONTROLLER_APPROVAL_ENV: "yes", BREAK_GLASS_ENV: "yes"},
|
||||
**kwargs)
|
||||
self.assertFalse(result["restart_performed"])
|
||||
self.assertFalse(result["apply_supported"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,453 @@
|
||||
"""#897: stale-runtime / runtime-mode refusals must not look like permission denials.
|
||||
|
||||
Acceptance criteria (issue #897):
|
||||
|
||||
* Stale-runtime and runtime-mode refusals are typed distinctly from
|
||||
profile-permission refusals (distinct ``blocker_kind``).
|
||||
* A refusal caused by staleness or runtime mode never emits a
|
||||
``permission_report`` and never names a permission the active profile holds.
|
||||
* ``_permission_block_report`` verifies the active profile actually lacks the
|
||||
operation before reporting it missing.
|
||||
* A stale-runtime refusal reports reconnect-only recovery and never recommends
|
||||
``gitea_activate_profile`` or an MCP session switch.
|
||||
* The blocker payload states the observed heads (parity fields).
|
||||
* Matrix across author / reviewer / merger / reconciler profiles.
|
||||
* Regression: ``gitea_create_issue`` on a stale daemon under ``prgs-author``
|
||||
never returns ``missing_permission: gitea.issue.create``.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
sys.path.insert(0, str(__import__("pathlib").Path(__file__).resolve().parent.parent))
|
||||
|
||||
import gitea_config # noqa: E402
|
||||
import gitea_mcp_server as mcp_server # noqa: E402
|
||||
|
||||
SHA_START = "7af40fb5ff7debd5e9165fe97d9c7c279358e175"
|
||||
SHA_LIVE = "2f4dec832327513118f2fe92b74da25d124a01cb"
|
||||
|
||||
ROLE_MATRIX = (
|
||||
(
|
||||
"prgs-author",
|
||||
"author",
|
||||
"gitea.issue.create",
|
||||
[
|
||||
"gitea.read",
|
||||
"gitea.issue.create",
|
||||
"gitea.issue.comment",
|
||||
"gitea.issue.close",
|
||||
"gitea.branch.create",
|
||||
"gitea.branch.push",
|
||||
"gitea.pr.create",
|
||||
"gitea.pr.comment",
|
||||
"gitea.repo.commit",
|
||||
],
|
||||
["gitea.pr.approve", "gitea.pr.merge", "gitea.pr.request_changes"],
|
||||
"gitea.pr.merge", # forbidden op for pure-permission case
|
||||
),
|
||||
(
|
||||
"prgs-reviewer",
|
||||
"reviewer",
|
||||
"gitea.pr.review",
|
||||
[
|
||||
"gitea.read",
|
||||
"gitea.pr.review",
|
||||
"gitea.pr.approve",
|
||||
"gitea.pr.request_changes",
|
||||
"gitea.pr.comment",
|
||||
"gitea.issue.comment",
|
||||
],
|
||||
["gitea.branch.push", "gitea.pr.create"],
|
||||
"gitea.branch.push",
|
||||
),
|
||||
(
|
||||
"prgs-merger",
|
||||
"merger",
|
||||
"gitea.pr.merge",
|
||||
[
|
||||
"gitea.read",
|
||||
"gitea.pr.merge",
|
||||
"gitea.pr.comment",
|
||||
"gitea.issue.comment",
|
||||
],
|
||||
["gitea.pr.approve", "gitea.branch.push", "gitea.pr.create"],
|
||||
"gitea.branch.push",
|
||||
),
|
||||
(
|
||||
"prgs-reconciler",
|
||||
"reconciler",
|
||||
"gitea.branch.delete",
|
||||
[
|
||||
"gitea.read",
|
||||
"gitea.branch.delete",
|
||||
"gitea.pr.comment",
|
||||
"gitea.issue.comment",
|
||||
"gitea.pr.close",
|
||||
"gitea.issue.close",
|
||||
],
|
||||
["gitea.pr.approve", "gitea.pr.merge"],
|
||||
"gitea.pr.merge",
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _profile(name: str, role: str, allowed: list[str], forbidden: list[str]) -> dict:
|
||||
return {
|
||||
"profile_name": name,
|
||||
"role": role,
|
||||
"role_kind": role,
|
||||
"allowed_operations": list(allowed),
|
||||
"forbidden_operations": list(forbidden),
|
||||
"identity": "test-user",
|
||||
}
|
||||
|
||||
|
||||
def _config(profiles: dict) -> dict:
|
||||
return {
|
||||
"version": 2,
|
||||
"profiles": {
|
||||
name: {
|
||||
"role": p["role"],
|
||||
"allowed_operations": p["allowed_operations"],
|
||||
"forbidden_operations": p["forbidden_operations"],
|
||||
}
|
||||
for name, p in profiles.items()
|
||||
},
|
||||
"rules": {"allow_runtime_switching": True},
|
||||
}
|
||||
|
||||
|
||||
class Issue897Helpers(unittest.TestCase):
|
||||
def test_classify_stale_reason_strings(self):
|
||||
stale = (
|
||||
f"live remote master is {SHA_LIVE[:12]} but the MCP server started "
|
||||
f"at {SHA_START[:12]}; the daemon is stale relative to live master "
|
||||
"-- restart/reconnect before mutating"
|
||||
)
|
||||
classified = mcp_server._classify_operation_gate_reasons([stale])
|
||||
self.assertEqual(classified["stale_runtime"], [stale])
|
||||
self.assertEqual(classified["permission"], [])
|
||||
self.assertEqual(classified["runtime_mode"], [])
|
||||
|
||||
def test_classify_permission_reason(self):
|
||||
reason = "profile is not allowed to gitea.pr.merge"
|
||||
classified = mcp_server._classify_operation_gate_reasons([reason])
|
||||
self.assertEqual(classified["permission"], [reason])
|
||||
self.assertEqual(classified["stale_runtime"], [])
|
||||
|
||||
def test_classify_runtime_mode_reason(self):
|
||||
reason = (
|
||||
"runtime mode is 'dev-test' and the mutation targets the "
|
||||
"production repository; dev/test runtimes must not mutate real "
|
||||
"issues or PRs (ADR: stable control runtime vs dev runtime)"
|
||||
)
|
||||
classified = mcp_server._classify_operation_gate_reasons([reason])
|
||||
self.assertEqual(classified["runtime_mode"], [reason])
|
||||
self.assertEqual(classified["stale_runtime"], [])
|
||||
|
||||
|
||||
class Issue897PermissionBlockReport(unittest.TestCase):
|
||||
def test_holds_op_is_diagnostic_defect_not_missing_permission(self):
|
||||
profile = _profile(
|
||||
"prgs-author",
|
||||
"author",
|
||||
["gitea.read", "gitea.issue.create", "gitea.issue.comment"],
|
||||
[],
|
||||
)
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server.gitea_config, "load_config", return_value=_config({"prgs-author": profile})
|
||||
), patch.object(
|
||||
mcp_server.gitea_config, "is_runtime_switching_enabled", return_value=True
|
||||
):
|
||||
report = mcp_server._permission_block_report("gitea.issue.create")
|
||||
self.assertTrue(report.get("diagnostic_defect"), report)
|
||||
self.assertIsNone(report.get("missing_permission"), report)
|
||||
action = (report.get("exact_safe_next_action") or "").lower()
|
||||
# Must not *recommend* profile switching; mentioning the forbidden
|
||||
# action in a "do not call" instruction is fine.
|
||||
self.assertNotIn("call gitea_activate_profile with", action)
|
||||
self.assertNotIn("switch to the author mcp session", action)
|
||||
self.assertNotIn("switch to the reviewer mcp session", action)
|
||||
self.assertIn("diagnostic defect", action)
|
||||
|
||||
def test_true_missing_permission_still_reports(self):
|
||||
profile = _profile(
|
||||
"prgs-author",
|
||||
"author",
|
||||
["gitea.read", "gitea.issue.create"],
|
||||
["gitea.pr.merge"],
|
||||
)
|
||||
reviewer = _profile(
|
||||
"prgs-reviewer",
|
||||
"reviewer",
|
||||
["gitea.read", "gitea.pr.merge", "gitea.pr.approve"],
|
||||
[],
|
||||
)
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server.gitea_config,
|
||||
"load_config",
|
||||
return_value=_config({"prgs-author": profile, "prgs-reviewer": reviewer}),
|
||||
), patch.object(
|
||||
mcp_server.gitea_config, "is_runtime_switching_enabled", return_value=True
|
||||
):
|
||||
report = mcp_server._permission_block_report("gitea.pr.merge")
|
||||
self.assertFalse(report.get("diagnostic_defect"), report)
|
||||
self.assertEqual(report.get("missing_permission"), "gitea.pr.merge")
|
||||
self.assertIn("prgs-reviewer", report.get("matching_configured_profiles") or [])
|
||||
|
||||
|
||||
class Issue897GateRefusalMatrix(unittest.TestCase):
|
||||
def _stale_parity(self) -> dict:
|
||||
return {
|
||||
"in_parity": True,
|
||||
"stale": False,
|
||||
"restart_required": True,
|
||||
"determinable": True,
|
||||
"startup_head": SHA_START,
|
||||
"current_head": SHA_START,
|
||||
"daemon_start_head": SHA_START,
|
||||
"local_head": SHA_START,
|
||||
"live_remote_head": SHA_LIVE,
|
||||
"live_known": True,
|
||||
"live_stale": True,
|
||||
"mutation_safe": False,
|
||||
"reasons": [
|
||||
f"live remote master is {SHA_LIVE[:12]} but the MCP server "
|
||||
f"started at {SHA_START[:12]}; the daemon is stale relative "
|
||||
"to live master -- restart/reconnect before mutating"
|
||||
],
|
||||
}
|
||||
|
||||
def test_stale_plus_permitted_op_all_roles(self):
|
||||
for name, role, permitted_op, allowed, forbidden, _forbidden_op in ROLE_MATRIX:
|
||||
with self.subTest(profile=name, op=permitted_op):
|
||||
profile = _profile(name, role, allowed, forbidden)
|
||||
parity = self._stale_parity()
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server, "_current_master_parity", return_value=parity
|
||||
), patch.object(
|
||||
mcp_server, "_master_parity_block", return_value=list(parity["reasons"])
|
||||
), patch.object(
|
||||
mcp_server, "_runtime_mode_block", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server, "_ensure_matching_profile", return_value=None
|
||||
), patch.object(
|
||||
mcp_server.session_ctx,
|
||||
"mutation_context_audit_fields",
|
||||
return_value={"session_profile": name},
|
||||
):
|
||||
blocked = mcp_server._profile_permission_block(permitted_op)
|
||||
self.assertIsNotNone(blocked, name)
|
||||
assert blocked is not None
|
||||
self.assertEqual(
|
||||
blocked.get("blocker_kind"),
|
||||
"runtime_reconnect_required",
|
||||
blocked,
|
||||
)
|
||||
self.assertNotIn("permission_report", blocked, blocked)
|
||||
self.assertTrue(blocked.get("restart_required"), blocked)
|
||||
self.assertEqual(blocked.get("startup_head"), SHA_START, blocked)
|
||||
self.assertEqual(blocked.get("live_remote_head"), SHA_LIVE, blocked)
|
||||
action = (blocked.get("exact_safe_next_action") or "").lower()
|
||||
self.assertIn("reconnect", action)
|
||||
self.assertNotIn("call gitea_activate_profile with", action)
|
||||
self.assertNotIn("switch to the author mcp session", action)
|
||||
self.assertNotIn("switch to the reviewer mcp session", action)
|
||||
|
||||
def test_fresh_plus_forbidden_op_all_roles(self):
|
||||
for name, role, _permitted, allowed, forbidden, forbidden_op in ROLE_MATRIX:
|
||||
with self.subTest(profile=name, op=forbidden_op):
|
||||
profile = _profile(name, role, allowed, forbidden)
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server, "_master_parity_block", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server, "_runtime_mode_block", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server, "_ensure_matching_profile", return_value=None
|
||||
), patch.object(
|
||||
mcp_server.session_ctx,
|
||||
"mutation_context_audit_fields",
|
||||
return_value={"session_profile": name},
|
||||
), patch.object(
|
||||
mcp_server.gitea_config,
|
||||
"load_config",
|
||||
return_value=_config({name: profile}),
|
||||
), patch.object(
|
||||
mcp_server.gitea_config, "is_runtime_switching_enabled", return_value=False
|
||||
):
|
||||
blocked = mcp_server._profile_permission_block(forbidden_op)
|
||||
self.assertIsNotNone(blocked, name)
|
||||
assert blocked is not None
|
||||
self.assertEqual(blocked.get("blocker_kind"), "permission_denied", blocked)
|
||||
self.assertIn("permission_report", blocked, blocked)
|
||||
report = blocked["permission_report"]
|
||||
self.assertEqual(report.get("missing_permission"), forbidden_op, report)
|
||||
self.assertFalse(report.get("diagnostic_defect"), report)
|
||||
# No runtime reconnect fields for pure permission denial
|
||||
self.assertNotEqual(
|
||||
blocked.get("blocker_kind"), "runtime_reconnect_required"
|
||||
)
|
||||
|
||||
def test_stale_plus_forbidden_op_both_causes_separated(self):
|
||||
for name, role, _permitted, allowed, forbidden, forbidden_op in ROLE_MATRIX:
|
||||
with self.subTest(profile=name, op=forbidden_op):
|
||||
profile = _profile(name, role, allowed, forbidden)
|
||||
parity = self._stale_parity()
|
||||
stale_reason = parity["reasons"][0]
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server, "_current_master_parity", return_value=parity
|
||||
), patch.object(
|
||||
mcp_server, "_master_parity_block", return_value=[stale_reason]
|
||||
), patch.object(
|
||||
mcp_server, "_runtime_mode_block", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server, "_ensure_matching_profile", return_value=None
|
||||
), patch.object(
|
||||
mcp_server.session_ctx,
|
||||
"mutation_context_audit_fields",
|
||||
return_value={"session_profile": name},
|
||||
):
|
||||
# Gate collects both classes; force permission reason too.
|
||||
with patch.object(
|
||||
mcp_server,
|
||||
"_profile_operation_gate",
|
||||
return_value=[
|
||||
stale_reason,
|
||||
f"profile is not allowed to {forbidden_op}",
|
||||
],
|
||||
):
|
||||
blocked = mcp_server._profile_permission_block(forbidden_op)
|
||||
self.assertIsNotNone(blocked)
|
||||
assert blocked is not None
|
||||
self.assertEqual(
|
||||
blocked.get("blocker_kind"), "runtime_reconnect_required", blocked
|
||||
)
|
||||
self.assertNotIn("permission_report", blocked, blocked)
|
||||
self.assertIn("permission_block_reasons", blocked, blocked)
|
||||
self.assertIn("stale_runtime_reasons", blocked, blocked)
|
||||
classes = blocked.get("gate_reason_classes") or {}
|
||||
self.assertTrue(classes.get("stale_runtime"), classes)
|
||||
self.assertTrue(classes.get("permission"), classes)
|
||||
|
||||
def test_runtime_mode_block_no_permission_report(self):
|
||||
profile = _profile(
|
||||
"prgs-author",
|
||||
"author",
|
||||
["gitea.read", "gitea.issue.create"],
|
||||
[],
|
||||
)
|
||||
runtime_reason = (
|
||||
"runtime mode is 'dev-test' and the mutation targets the "
|
||||
"production repository; dev/test runtimes must not mutate real "
|
||||
"issues or PRs (ADR: stable control runtime vs dev runtime)"
|
||||
)
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server, "_master_parity_block", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server, "_runtime_mode_block", return_value=[runtime_reason]
|
||||
), patch.object(
|
||||
mcp_server, "_ensure_matching_profile", return_value=None
|
||||
), patch.object(
|
||||
mcp_server.session_ctx,
|
||||
"mutation_context_audit_fields",
|
||||
return_value={"session_profile": "prgs-author"},
|
||||
):
|
||||
blocked = mcp_server._profile_permission_block("gitea.issue.create")
|
||||
self.assertIsNotNone(blocked)
|
||||
assert blocked is not None
|
||||
self.assertEqual(blocked.get("blocker_kind"), "runtime_mode_blocked", blocked)
|
||||
self.assertNotIn("permission_report", blocked, blocked)
|
||||
action = (blocked.get("exact_safe_next_action") or "").lower()
|
||||
self.assertNotIn("call gitea_activate_profile with", action)
|
||||
self.assertIn("stable control runtime", action)
|
||||
|
||||
|
||||
class Issue897CreateIssueRegression(unittest.TestCase):
|
||||
def test_create_issue_stale_daemon_never_missing_issue_create(self):
|
||||
"""Regression AC: stale prgs-author create_issue must not claim missing create."""
|
||||
profile = _profile(
|
||||
"prgs-author",
|
||||
"author",
|
||||
[
|
||||
"gitea.read",
|
||||
"gitea.issue.create",
|
||||
"gitea.issue.comment",
|
||||
"gitea.branch.create",
|
||||
"gitea.branch.push",
|
||||
"gitea.pr.create",
|
||||
"gitea.pr.comment",
|
||||
"gitea.repo.commit",
|
||||
],
|
||||
[],
|
||||
)
|
||||
stale_reason = (
|
||||
f"live remote master is {SHA_LIVE[:12]} but the MCP server started "
|
||||
f"at {SHA_START[:12]}; the daemon is stale relative to live master "
|
||||
"-- restart/reconnect before mutating"
|
||||
)
|
||||
parity = {
|
||||
"in_parity": True,
|
||||
"stale": False,
|
||||
"restart_required": True,
|
||||
"determinable": True,
|
||||
"startup_head": SHA_START,
|
||||
"current_head": SHA_START,
|
||||
"daemon_start_head": SHA_START,
|
||||
"local_head": SHA_START,
|
||||
"live_remote_head": SHA_LIVE,
|
||||
"live_known": True,
|
||||
"live_stale": True,
|
||||
"mutation_safe": False,
|
||||
"reasons": [stale_reason],
|
||||
}
|
||||
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile), patch.object(
|
||||
mcp_server, "_current_master_parity", return_value=parity
|
||||
), patch.object(
|
||||
mcp_server, "_master_parity_block", return_value=[stale_reason]
|
||||
), patch.object(
|
||||
mcp_server, "_runtime_mode_block", return_value=[]
|
||||
), patch.object(
|
||||
mcp_server, "_ensure_matching_profile", return_value=None
|
||||
), patch.object(
|
||||
mcp_server.session_ctx,
|
||||
"mutation_context_audit_fields",
|
||||
return_value={"session_profile": "prgs-author"},
|
||||
), patch.object(
|
||||
mcp_server, "_mutation_config_authority_block", return_value=None
|
||||
), patch.object(
|
||||
mcp_server, "_session_context_mutation_block", return_value=None
|
||||
):
|
||||
blocked = mcp_server._profile_permission_block(
|
||||
"gitea.issue.create", remote="prgs"
|
||||
)
|
||||
|
||||
self.assertIsNotNone(blocked)
|
||||
assert blocked is not None
|
||||
self.assertEqual(blocked.get("blocker_kind"), "runtime_reconnect_required")
|
||||
self.assertNotIn("permission_report", blocked)
|
||||
# Even if a caller still built a raw report, holds-check must not claim missing.
|
||||
with patch.object(mcp_server, "get_profile", return_value=profile):
|
||||
raw = mcp_server._permission_block_report("gitea.issue.create")
|
||||
self.assertIsNone(raw.get("missing_permission"), raw)
|
||||
self.assertNotEqual(raw.get("missing_permission"), "gitea.issue.create")
|
||||
|
||||
def test_permission_report_for_gate_reasons_skips_stale(self):
|
||||
stale = (
|
||||
f"live remote master is {SHA_LIVE[:12]} but the MCP server started "
|
||||
f"at {SHA_START[:12]}; the daemon is stale relative to live master "
|
||||
"-- restart/reconnect before mutating"
|
||||
)
|
||||
self.assertIsNone(
|
||||
mcp_server._permission_report_for_gate_reasons(
|
||||
"gitea.issue.comment", [stale]
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,237 @@
|
||||
"""Tests for graceful MCP maintenance-drain mode (#659).
|
||||
|
||||
Acceptance coverage:
|
||||
|
||||
1. Enter/exit is durable and audited (DB substrate).
|
||||
2. New work assignment stops during drain (allocator WAIT).
|
||||
3. Mutations deferred except allowlisted safety ops.
|
||||
4. Sessions can observe drain state.
|
||||
5. Fail-closed on unreadable drain state.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
import maintenance_drain
|
||||
from control_plane_db import ControlPlaneDB
|
||||
from allocator_service import WorkCandidate, allocate_next_work, OUTCOME_WAIT
|
||||
|
||||
|
||||
class TestDrainDecisions(unittest.TestCase):
|
||||
def test_inactive_allows_mutations_and_assignment(self):
|
||||
decision = maintenance_drain.classify_mutation("create_pr", None)
|
||||
self.assertTrue(decision["allowed"])
|
||||
self.assertFalse(decision["deferred"])
|
||||
assign = maintenance_drain.classify_assignment(None)
|
||||
self.assertTrue(assign["assignment_allowed"])
|
||||
|
||||
def test_draining_defers_non_allowlisted_mutation(self):
|
||||
record = {"state": "draining", "remote": "prgs", "org": "o", "repo": "r"}
|
||||
decision = maintenance_drain.classify_mutation("create_pr", record)
|
||||
self.assertFalse(decision["allowed"])
|
||||
self.assertTrue(decision["deferred"])
|
||||
self.assertEqual(decision["reason_code"], maintenance_drain.BLOCKER_DRAIN_ACTIVE)
|
||||
self.assertIn("create_pr", decision["reasons"][0])
|
||||
|
||||
def test_allowlisted_safety_ops_pass_during_drain(self):
|
||||
record = {"state": "draining"}
|
||||
for task in (
|
||||
"heartbeat_issue_lock",
|
||||
"gitea_release_reviewer_pr_lease",
|
||||
"write_session_checkpoint",
|
||||
"exit_maintenance_drain",
|
||||
):
|
||||
with self.subTest(task=task):
|
||||
decision = maintenance_drain.classify_mutation(task, record)
|
||||
self.assertTrue(decision["allowed"], decision)
|
||||
|
||||
def test_assignment_stopped_during_drain(self):
|
||||
record = {"state": "draining", "reason": "upgrade"}
|
||||
decision = maintenance_drain.classify_assignment(record)
|
||||
self.assertFalse(decision["assignment_allowed"])
|
||||
self.assertEqual(
|
||||
decision["reason_code"], maintenance_drain.REASON_ASSIGNMENT_STOPPED
|
||||
)
|
||||
|
||||
def test_unknown_state_fails_closed(self):
|
||||
with self.assertRaises(maintenance_drain.MaintenanceDrainError):
|
||||
maintenance_drain.normalize_state("drainig")
|
||||
|
||||
def test_status_payload_always_answers(self):
|
||||
inactive = maintenance_drain.status_payload(None, remote="prgs", org="o", repo="r")
|
||||
self.assertFalse(inactive["draining"])
|
||||
self.assertTrue(inactive["reads_permitted"])
|
||||
active = maintenance_drain.status_payload(
|
||||
{"state": "draining", "reason": "reboot", "requested_by": "ops"},
|
||||
remote="prgs",
|
||||
org="o",
|
||||
repo="r",
|
||||
)
|
||||
self.assertTrue(active["draining"])
|
||||
self.assertTrue(active["assignment_stopped"])
|
||||
self.assertTrue(active["mutations_deferred"])
|
||||
self.assertIn("heartbeat_issue_lock", active["allowlisted_tasks"])
|
||||
|
||||
|
||||
class TestDrainDB(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.db = ControlPlaneDB(db_path=str(Path(self._tmp.name) / "cp.sqlite3"))
|
||||
|
||||
def test_enter_exit_idempotent_and_audited(self):
|
||||
first = self.db.set_maintenance_drain(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
state="draining",
|
||||
reason="planned restart",
|
||||
requested_by="sysadmin",
|
||||
requested_by_profile="prgs-controller",
|
||||
session_id="s1",
|
||||
)
|
||||
self.assertTrue(first["transitioned"])
|
||||
self.assertEqual(first["state"], "draining")
|
||||
self.assertTrue(maintenance_drain.is_draining(first["record"]))
|
||||
|
||||
again = self.db.set_maintenance_drain(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
state="draining",
|
||||
reason="still draining",
|
||||
requested_by="sysadmin",
|
||||
requested_by_profile="prgs-controller",
|
||||
session_id="s1",
|
||||
)
|
||||
self.assertFalse(again["transitioned"])
|
||||
self.assertEqual(again["record"]["entered_at"], first["record"]["entered_at"])
|
||||
|
||||
exited = self.db.set_maintenance_drain(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
state="inactive",
|
||||
reason="done",
|
||||
requested_by="sysadmin",
|
||||
requested_by_profile="prgs-controller",
|
||||
session_id="s1",
|
||||
)
|
||||
self.assertTrue(exited["transitioned"])
|
||||
self.assertFalse(maintenance_drain.is_draining(exited["record"]))
|
||||
self.assertTrue(exited["record"]["exited_at"])
|
||||
|
||||
# Events recorded for transitions only (enter + exit).
|
||||
with self.db._tx(immediate=False) as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT event_type FROM events WHERE event_type LIKE 'maintenance_drain_%' "
|
||||
"ORDER BY event_id"
|
||||
).fetchall()
|
||||
types = [r[0] for r in rows]
|
||||
self.assertEqual(types, ["maintenance_drain_enter", "maintenance_drain_exit"])
|
||||
|
||||
def test_read_missing_is_none_not_error(self):
|
||||
self.assertIsNone(
|
||||
self.db.read_maintenance_drain(remote="prgs", org="o", repo="r")
|
||||
)
|
||||
|
||||
|
||||
class TestAllocatorStopsDuringDrain(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.db = ControlPlaneDB(db_path=str(Path(self._tmp.name) / "cp.sqlite3"))
|
||||
|
||||
def test_allocate_returns_wait_while_draining(self):
|
||||
self.db.set_maintenance_drain(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
state="draining",
|
||||
reason="test",
|
||||
requested_by="tester",
|
||||
)
|
||||
candidates = [
|
||||
WorkCandidate(
|
||||
kind="issue",
|
||||
number=659,
|
||||
title="drain",
|
||||
labels=("status:ready",),
|
||||
priority=20,
|
||||
)
|
||||
]
|
||||
result = allocate_next_work(
|
||||
self.db,
|
||||
role="author",
|
||||
session_id="test-session",
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
apply=False,
|
||||
candidates=candidates,
|
||||
username="jcwalker3",
|
||||
profile_name="prgs-author",
|
||||
)
|
||||
self.assertEqual(result["outcome"], OUTCOME_WAIT)
|
||||
self.assertIsNone(result.get("selected"))
|
||||
self.assertEqual(
|
||||
result.get("reason_code"),
|
||||
maintenance_drain.REASON_ASSIGNMENT_STOPPED,
|
||||
)
|
||||
self.assertTrue(result["maintenance_drain"]["draining"])
|
||||
|
||||
def test_allocate_works_when_inactive(self):
|
||||
candidates = [
|
||||
WorkCandidate(
|
||||
kind="issue",
|
||||
number=659,
|
||||
title="drain",
|
||||
labels=("status:ready",),
|
||||
priority=20,
|
||||
)
|
||||
]
|
||||
result = allocate_next_work(
|
||||
self.db,
|
||||
role="author",
|
||||
session_id="test-session-2",
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
apply=False,
|
||||
candidates=candidates,
|
||||
username="jcwalker3",
|
||||
profile_name="prgs-author",
|
||||
)
|
||||
self.assertNotEqual(
|
||||
result.get("reason_code"),
|
||||
maintenance_drain.REASON_ASSIGNMENT_STOPPED,
|
||||
)
|
||||
|
||||
|
||||
class TestCapabilityMap(unittest.TestCase):
|
||||
def test_drain_tasks_mapped(self):
|
||||
import task_capability_map as tcm
|
||||
|
||||
self.assertEqual(
|
||||
tcm.required_permission("enter_maintenance_drain"),
|
||||
"runtime.maintenance_drain",
|
||||
)
|
||||
self.assertEqual(
|
||||
tcm.required_permission("exit_maintenance_drain"),
|
||||
"runtime.maintenance_drain",
|
||||
)
|
||||
self.assertEqual(
|
||||
tcm.required_permission("maintenance_drain_status"),
|
||||
"gitea.read",
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -105,3 +105,120 @@ def test_cross_links_do_not_embed_secrets():
|
||||
text = _read(path)
|
||||
for marker in ("ghp_", "BEGIN PRIVATE KEY", "Authorization: Bearer"):
|
||||
assert marker not in text, f"{path} contains {marker!r}"
|
||||
|
||||
|
||||
# --- Coordinator doc stays in lock-step with the tool (#886 review blocker B2) --
|
||||
#
|
||||
# PR #882 moved the #661 drain-proof hard gate *into* gitea_request_mcp_restart,
|
||||
# but the coordinator document still described the proof as "a separate child"
|
||||
# and omitted both new parameters. Nothing referenced that document, so nothing
|
||||
# caught the drift. These tests bind the prose to the real signature.
|
||||
|
||||
COORDINATOR_DOC = REPO_ROOT / "docs" / "mcp-restart-coordinator.md"
|
||||
|
||||
# Affirmative claims that were accurate before #661 landed and are now false.
|
||||
# Matched against whitespace-normalized text so re-wrapping cannot hide them.
|
||||
# Deliberately not the bare phrase "a separate child": the corrected prose uses
|
||||
# it in a negation ("no longer a separate child operation"), and a guard that
|
||||
# forbids naming the old behaviour would block explaining that it changed.
|
||||
STALE_PRE_661_PHRASES = (
|
||||
"gated by a drain proof (a separate child)",
|
||||
"is a later child gated by a drain proof",
|
||||
"mutative apply path is explicitly out of scope",
|
||||
"apply is gated by a drain proof (a separate child)",
|
||||
)
|
||||
|
||||
|
||||
def _documented_signature_block() -> str:
|
||||
"""The fenced signature block for the tool, as published in the doc."""
|
||||
text = _read(COORDINATOR_DOC)
|
||||
marker = "gitea_request_mcp_restart("
|
||||
start = text.index(marker)
|
||||
end = text.index("```", start)
|
||||
return text[start:end]
|
||||
|
||||
|
||||
def test_documented_signature_matches_the_real_tool_signature():
|
||||
import inspect
|
||||
|
||||
import gitea_mcp_server
|
||||
|
||||
block = _documented_signature_block()
|
||||
real = inspect.signature(gitea_mcp_server.gitea_request_mcp_restart)
|
||||
for name in real.parameters:
|
||||
assert name in block, (
|
||||
f"docs/mcp-restart-coordinator.md documents no {name!r} parameter; "
|
||||
"the published signature has drifted from the tool"
|
||||
)
|
||||
|
||||
|
||||
def test_drain_proof_and_break_glass_parameters_are_documented():
|
||||
block = _documented_signature_block()
|
||||
for name in ("drain_proof_json", "request_break_glass"):
|
||||
assert name in block, f"signature block missing {name}"
|
||||
|
||||
|
||||
def test_restart_class_and_target_scoping_parameters_survive():
|
||||
block = _documented_signature_block()
|
||||
for name in ("restart_class", "target_session_id", "target_role",
|
||||
"target_connector"):
|
||||
assert name in block, f"signature block lost #663 parameter {name}"
|
||||
|
||||
|
||||
def test_gate_is_documented_as_executing_inside_this_tool():
|
||||
lower = _read(COORDINATOR_DOC).lower()
|
||||
assert "inside this tool" in lower, (
|
||||
"the coordinator doc must state that the drain-proof gate executes in "
|
||||
"gitea_request_mcp_restart, not in a later child"
|
||||
)
|
||||
assert "no longer a separate child operation" in lower
|
||||
|
||||
|
||||
def test_stale_pre_661_wording_cannot_return():
|
||||
normalized = " ".join(_read(COORDINATOR_DOC).split()).lower()
|
||||
for phrase in STALE_PRE_661_PHRASES:
|
||||
assert phrase not in normalized, (
|
||||
f"stale pre-#661 wording returned to the coordinator doc: {phrase!r}"
|
||||
)
|
||||
|
||||
|
||||
def test_dry_run_versus_apply_behavior_is_documented():
|
||||
lower = _read(COORDINATOR_DOC).lower()
|
||||
assert "dry_run=true" in lower and "dry_run=false" in lower
|
||||
assert "apply_supported" in lower and "restart_performed" in lower
|
||||
assert "never restarts anything" in lower
|
||||
|
||||
|
||||
def test_authorization_ordering_and_conjunction_are_documented():
|
||||
text = _read(COORDINATOR_DOC)
|
||||
lower = text.lower()
|
||||
assert "authorization ordering" in lower
|
||||
assert "allow_restart" in text
|
||||
assert "apply_authorized" in text
|
||||
# The conjunction itself, and the attribution fields behind it.
|
||||
assert "gate.allow and allow_restart" in text
|
||||
for field in ("drain_gate_allow", "restart_class_authorized"):
|
||||
assert field in text, f"doc omits apply_gate.{field}"
|
||||
|
||||
|
||||
def test_break_glass_scope_is_documented_as_drain_proof_only():
|
||||
text = _read(COORDINATOR_DOC)
|
||||
lower = text.lower()
|
||||
assert "break-glass" in lower
|
||||
assert "drain proof only" in lower, (
|
||||
"doc must state break-glass never bypasses the restart-class matrix"
|
||||
)
|
||||
assert "GITEA_BREAKGLASS_RESTART_AUTHORIZATION" in text
|
||||
|
||||
|
||||
def test_fail_closed_on_apply_is_documented():
|
||||
lower = _read(COORDINATOR_DOC).lower()
|
||||
assert "fail closed" in lower
|
||||
for condition in ("expired", "unclean", "tampered", "stale"):
|
||||
assert condition in lower, f"fail-closed list omits {condition!r}"
|
||||
|
||||
|
||||
def test_coordinator_doc_embeds_no_secrets():
|
||||
text = _read(COORDINATOR_DOC)
|
||||
for marker in ("ghp_", "BEGIN PRIVATE KEY", "Authorization: Bearer"):
|
||||
assert marker not in text, f"{COORDINATOR_DOC} contains {marker!r}"
|
||||
|
||||
@@ -0,0 +1,217 @@
|
||||
"""Unit tests for the scoped recovery playbook (#669)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import recovery_playbook as rp
|
||||
import restart_coordinator as rc
|
||||
|
||||
|
||||
def test_ladder_covers_eleven_ordered_rungs():
|
||||
ranks = [r.rank for r in rp.RECOVERY_LADDER]
|
||||
assert ranks == list(range(len(rp.RECOVERY_LADDER)))
|
||||
assert len(rp.RECOVERY_LADDER) == 11
|
||||
assert rp.RECOVERY_LADDER[0].action is rp.RecoveryAction.CLIENT_RECONNECT
|
||||
assert rp.RECOVERY_LADDER[-1].action is rp.RecoveryAction.HOST_RESTART
|
||||
|
||||
|
||||
def test_ladder_document_links_parent_issues():
|
||||
doc = rp.ladder_document()
|
||||
assert "#655" in doc["parent_issues"]
|
||||
assert "#652" in doc["parent_issues"]
|
||||
assert "#653" in doc["parent_issues"]
|
||||
assert doc["enforcement_issue"] == "#669"
|
||||
assert "full_mcp_restart" in doc["broad_restart_actions"]
|
||||
|
||||
|
||||
def test_recommend_transport_eof_starts_at_client_reconnect():
|
||||
plan = rp.recommend_actions(symptoms=["transport_eof"])
|
||||
assert plan["recommended_actions"][0]["action"] == "client_reconnect"
|
||||
assert plan["recommended_actions"][0]["issue_links"]
|
||||
|
||||
|
||||
def test_recommend_skips_successful_prior_attempts():
|
||||
attempts = [
|
||||
rp.build_attempt_record(
|
||||
"client_reconnect", outcome="success", reason="reconnected"
|
||||
)
|
||||
]
|
||||
plan = rp.recommend_actions(
|
||||
symptoms=["transport_eof"], prior_recovery_attempts=attempts
|
||||
)
|
||||
actions = [a["action"] for a in plan["recommended_actions"]]
|
||||
assert "client_reconnect" not in actions
|
||||
assert actions[0] == "capability_refresh"
|
||||
|
||||
|
||||
def test_escalation_denied_without_attempt_log():
|
||||
result = rp.assess_escalation("full_mcp_restart", prior_recovery_attempts=[])
|
||||
assert result.allowed is False
|
||||
assert result.require_attempt_log is True
|
||||
assert any("#669" in r for r in result.reasons)
|
||||
assert result.recommended_next # soft recommendations still provided
|
||||
|
||||
|
||||
def test_escalation_allowed_after_insufficient_narrower():
|
||||
attempts = [
|
||||
rp.build_attempt_record(
|
||||
"client_reconnect",
|
||||
outcome="insufficient",
|
||||
reason="still flapping",
|
||||
),
|
||||
rp.build_attempt_record(
|
||||
"session_reconnect",
|
||||
outcome="failed",
|
||||
reason="namespace still dead",
|
||||
),
|
||||
]
|
||||
result = rp.assess_escalation(
|
||||
"full_mcp_restart", prior_recovery_attempts=attempts
|
||||
)
|
||||
assert result.allowed is True
|
||||
assert len(result.qualifying_attempts) == 2
|
||||
|
||||
|
||||
def test_escalation_break_glass_bypasses_attempt_log():
|
||||
result = rp.assess_escalation(
|
||||
"host_restart", prior_recovery_attempts=[], break_glass=True
|
||||
)
|
||||
assert result.allowed is True
|
||||
assert result.break_glass is True
|
||||
|
||||
|
||||
def test_narrow_action_does_not_require_attempt_log():
|
||||
result = rp.assess_escalation(
|
||||
"client_reconnect", prior_recovery_attempts=[]
|
||||
)
|
||||
assert result.allowed is True
|
||||
assert result.require_attempt_log is False
|
||||
|
||||
|
||||
def test_same_rank_attempt_does_not_qualify_for_escalation():
|
||||
attempts = [
|
||||
rp.build_attempt_record(
|
||||
"full_mcp_restart", outcome="failed", reason="already failed full"
|
||||
)
|
||||
]
|
||||
result = rp.assess_escalation(
|
||||
"full_mcp_restart", prior_recovery_attempts=attempts
|
||||
)
|
||||
assert result.allowed is False
|
||||
|
||||
|
||||
def test_success_outcome_does_not_qualify_for_escalation():
|
||||
attempts = [
|
||||
rp.build_attempt_record(
|
||||
"client_reconnect", outcome="success", reason="fixed"
|
||||
)
|
||||
]
|
||||
result = rp.assess_escalation(
|
||||
"full_mcp_restart", prior_recovery_attempts=attempts
|
||||
)
|
||||
assert result.allowed is False
|
||||
|
||||
|
||||
def test_recovery_metrics_fraction_avoided():
|
||||
attempts = [
|
||||
rp.build_attempt_record("client_reconnect", outcome="success"),
|
||||
rp.build_attempt_record("session_reconnect", outcome="success"),
|
||||
rp.build_attempt_record("full_mcp_restart", outcome="success"),
|
||||
]
|
||||
metrics = rp.recovery_metrics(attempts)
|
||||
assert metrics["successes_total"] == 3
|
||||
assert metrics["successes_avoided_full_restart"] == 2
|
||||
assert metrics["successes_full_or_host_restart"] == 1
|
||||
assert abs(metrics["fraction_avoided_full_restart"] - (2 / 3)) < 1e-9
|
||||
|
||||
|
||||
def test_coordinator_denies_full_restart_without_attempt_log():
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"prior_recovery_attempts": [],
|
||||
}
|
||||
report = rc.evaluate_restart_impact(
|
||||
inv,
|
||||
restart_class=rc.RestartClass.FULL_MCP_RESTART,
|
||||
requester_role="controller",
|
||||
requester_permissions=rc.permissions_for_role("controller"),
|
||||
controller_approved=True,
|
||||
operator_authorized=True,
|
||||
)
|
||||
assert report.allow_restart is False
|
||||
assert report.attempt_log_satisfied is False
|
||||
assert report.verdict == rc.VERDICT_UNSAFE
|
||||
blob = " ".join(report.reasons + report.authorization_reasons)
|
||||
assert "#669" in blob or "attempt log" in blob
|
||||
|
||||
|
||||
def test_coordinator_allows_full_restart_with_attempt_log():
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"prior_recovery_attempts": [
|
||||
{
|
||||
"action": "client_reconnect",
|
||||
"outcome": "insufficient",
|
||||
"reason": "still broken",
|
||||
}
|
||||
],
|
||||
}
|
||||
report = rc.evaluate_restart_impact(
|
||||
inv,
|
||||
restart_class=rc.RestartClass.FULL_MCP_RESTART,
|
||||
requester_role="controller",
|
||||
requester_permissions=rc.permissions_for_role("controller"),
|
||||
controller_approved=True,
|
||||
operator_authorized=True,
|
||||
)
|
||||
assert report.attempt_log_satisfied is True
|
||||
assert report.allow_restart is True
|
||||
assert report.verdict == rc.VERDICT_SAFE
|
||||
|
||||
|
||||
def test_coordinator_break_glass_allows_without_log():
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"prior_recovery_attempts": [],
|
||||
}
|
||||
report = rc.evaluate_restart_impact(
|
||||
inv,
|
||||
restart_class=rc.RestartClass.FULL_MCP_RESTART,
|
||||
requester_role="controller",
|
||||
requester_permissions=rc.permissions_for_role("controller"),
|
||||
controller_approved=True,
|
||||
operator_authorized=True,
|
||||
break_glass=True,
|
||||
)
|
||||
assert report.break_glass is True
|
||||
assert report.attempt_log_satisfied is True
|
||||
assert report.allow_restart is True
|
||||
|
||||
|
||||
def test_coordinator_client_reconnect_unaffected():
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"prior_recovery_attempts": [],
|
||||
}
|
||||
report = rc.evaluate_restart_impact(
|
||||
inv,
|
||||
restart_class=rc.RestartClass.CLIENT_RECONNECT,
|
||||
requester_role="author",
|
||||
requester_permissions=rc.permissions_for_role("author"),
|
||||
)
|
||||
assert report.attempt_log_satisfied is True
|
||||
assert report.allow_restart is True
|
||||
|
||||
|
||||
def test_restart_class_alias_accepted():
|
||||
assert (
|
||||
rp.resolve_action("full_mcp_restart")
|
||||
is rp.RecoveryAction.FULL_MCP_RESTART
|
||||
)
|
||||
@@ -0,0 +1,232 @@
|
||||
"""Permission, drain, routing, and audit matrix for restart classes (#663)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from datetime import datetime, timezone
|
||||
|
||||
import restart_coordinator as rc
|
||||
|
||||
NOW = datetime(2026, 7, 24, 20, 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
def _inventory() -> dict:
|
||||
return {
|
||||
"inventory_complete": True,
|
||||
"sessions": [
|
||||
{
|
||||
"session_id": "requester",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": os.getpid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
{
|
||||
"session_id": "reviewer",
|
||||
"role": "reviewer",
|
||||
"profile": "prgs-reviewer",
|
||||
"pid": os.getpid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
],
|
||||
"leases": [
|
||||
{
|
||||
"lease_id": "review-lease",
|
||||
"session_id": "reviewer",
|
||||
"role": "reviewer",
|
||||
"phase": "reviewing",
|
||||
"work_kind": "pr",
|
||||
"work_number": 900,
|
||||
"worktree_path": "/tmp/review-900",
|
||||
"freshness": {"freshness": "active"},
|
||||
}
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def _evaluate(
|
||||
restart_class: rc.RestartClass,
|
||||
*,
|
||||
role: str = "controller",
|
||||
permissions: tuple[str, ...] | None = None,
|
||||
approved: bool = True,
|
||||
operator: bool = True,
|
||||
**targets,
|
||||
):
|
||||
return rc.evaluate_restart_impact(
|
||||
_inventory(),
|
||||
now=NOW,
|
||||
requesting_session_id="requester",
|
||||
restart_class=restart_class,
|
||||
requester_role=role,
|
||||
requester_permissions=(
|
||||
permissions if permissions is not None
|
||||
else rc.permissions_for_role(role)
|
||||
),
|
||||
controller_approved=approved,
|
||||
operator_authorized=operator,
|
||||
**targets,
|
||||
)
|
||||
|
||||
|
||||
def test_policy_table_covers_exactly_all_nine_classes():
|
||||
assert set(rc.RESTART_CLASS_POLICIES) == set(rc.RestartClass)
|
||||
assert len(rc.RESTART_CLASS_POLICIES) == 9
|
||||
for restart_class, policy in rc.RESTART_CLASS_POLICIES.items():
|
||||
assert policy.restart_class is restart_class
|
||||
assert policy.required_permission
|
||||
assert policy.expected_blast_radius in {
|
||||
rc.BLAST_NONE, rc.BLAST_LOW, rc.BLAST_MEDIUM, rc.BLAST_HIGH
|
||||
}
|
||||
assert policy.drain_requirement
|
||||
assert policy.approval_requirement
|
||||
assert policy.audit_requirement
|
||||
assert policy.recovery_behavior
|
||||
|
||||
|
||||
def test_permission_matrix_allows_each_class_with_exact_permission():
|
||||
targets = {
|
||||
rc.RestartClass.WORKER_RESTART: {"target_session_id": "reviewer"},
|
||||
rc.RestartClass.ROLE_RUNTIME_RESTART: {"target_role": "reviewer"},
|
||||
rc.RestartClass.CONNECTOR_RESTART: {"target_connector": "github"},
|
||||
}
|
||||
for restart_class, policy in rc.RESTART_CLASS_POLICIES.items():
|
||||
report = _evaluate(
|
||||
restart_class,
|
||||
permissions=(policy.required_permission,),
|
||||
**targets.get(restart_class, {}),
|
||||
)
|
||||
assert report.permission_authorized, restart_class
|
||||
assert report.role_authorized, restart_class
|
||||
assert report.approval_satisfied, restart_class
|
||||
assert report.audit_record["restart_class"] == restart_class.value
|
||||
assert (
|
||||
report.audit_record["required_permission"]
|
||||
== policy.required_permission
|
||||
)
|
||||
|
||||
|
||||
def test_missing_or_nearby_permission_denies():
|
||||
report = _evaluate(
|
||||
rc.RestartClass.ROLE_RUNTIME_RESTART,
|
||||
permissions=("mcp.restart.worker.request",),
|
||||
target_role="reviewer",
|
||||
)
|
||||
assert report.verdict == rc.VERDICT_UNSAFE
|
||||
assert not report.allow_restart
|
||||
assert not report.permission_authorized
|
||||
assert any("missing required permission" in r for r in report.reasons)
|
||||
|
||||
|
||||
def test_unknown_restart_class_denies_fail_closed():
|
||||
report = rc.evaluate_restart_impact(
|
||||
_inventory(),
|
||||
now=NOW,
|
||||
restart_class="surprise_reboot",
|
||||
requester_role="admin",
|
||||
requester_permissions=("mcp.restart.host.request",),
|
||||
controller_approved=True,
|
||||
operator_authorized=True,
|
||||
)
|
||||
assert report.verdict == rc.VERDICT_UNSAFE
|
||||
assert not report.allow_restart
|
||||
assert report.restart_policy == {}
|
||||
assert any("unknown restart class" in r for r in report.reasons)
|
||||
|
||||
|
||||
def test_worker_roles_cannot_request_full_or_host_restart():
|
||||
for role in rc.WORKER_ROLES:
|
||||
granted = rc.permissions_for_role(role)
|
||||
assert "mcp.restart.full.request" not in granted
|
||||
assert "mcp.restart.host.request" not in granted
|
||||
report = _evaluate(
|
||||
rc.RestartClass.FULL_MCP_RESTART,
|
||||
role=role,
|
||||
permissions=granted,
|
||||
)
|
||||
assert not report.role_authorized
|
||||
assert not report.allow_restart
|
||||
|
||||
|
||||
def test_controller_approval_is_independent_of_permission():
|
||||
report = _evaluate(
|
||||
rc.RestartClass.WORKER_RESTART,
|
||||
approved=False,
|
||||
target_session_id="reviewer",
|
||||
)
|
||||
assert report.permission_authorized
|
||||
assert not report.approval_satisfied
|
||||
assert not report.allow_restart
|
||||
|
||||
|
||||
def test_narrow_classes_do_not_inherit_full_drain_or_peer_lease_block():
|
||||
for restart_class in (
|
||||
rc.RestartClass.CLIENT_RECONNECT,
|
||||
rc.RestartClass.SESSION_RECONNECT,
|
||||
rc.RestartClass.CONFIGURATION_RELOAD,
|
||||
):
|
||||
report = _evaluate(restart_class)
|
||||
assert not report.restart_policy["full_drain_required"]
|
||||
assert report.counts["leases_disruptive"] == 0
|
||||
assert report.counts["sessions_live_other"] == 0
|
||||
assert report.counts["critical_sections"] == 0
|
||||
assert report.counts["mutations"] == 0
|
||||
assert report.allow_restart, (restart_class, report.reasons)
|
||||
|
||||
|
||||
def test_client_reconnect_does_not_wait_for_unrelated_terminal_lock():
|
||||
inventory = _inventory()
|
||||
inventory["terminal_lock"] = {"terminal_pr": 901}
|
||||
report = rc.evaluate_restart_impact(
|
||||
inventory,
|
||||
now=NOW,
|
||||
requesting_session_id="requester",
|
||||
restart_class=rc.RestartClass.CLIENT_RECONNECT,
|
||||
requester_role="author",
|
||||
requester_permissions=rc.permissions_for_role("author"),
|
||||
)
|
||||
assert report.allow_restart
|
||||
assert report.terminal_lock is None
|
||||
|
||||
|
||||
def test_scoped_restart_only_counts_named_target():
|
||||
report = _evaluate(
|
||||
rc.RestartClass.ROLE_RUNTIME_RESTART,
|
||||
target_role="author",
|
||||
)
|
||||
assert report.counts["leases_disruptive"] == 0
|
||||
assert report.affected_prs == []
|
||||
assert report.allow_restart
|
||||
|
||||
reviewer = _evaluate(
|
||||
rc.RestartClass.ROLE_RUNTIME_RESTART,
|
||||
target_role="reviewer",
|
||||
)
|
||||
assert reviewer.counts["leases_disruptive"] == 1
|
||||
assert reviewer.affected_prs == [900]
|
||||
assert not reviewer.allow_restart
|
||||
|
||||
|
||||
def test_missing_scoped_target_denies_instead_of_widening():
|
||||
for restart_class in (
|
||||
rc.RestartClass.WORKER_RESTART,
|
||||
rc.RestartClass.ROLE_RUNTIME_RESTART,
|
||||
rc.RestartClass.CONNECTOR_RESTART,
|
||||
):
|
||||
report = _evaluate(restart_class)
|
||||
assert not report.allow_restart
|
||||
assert any("target required" in r for r in report.reasons)
|
||||
|
||||
|
||||
def test_only_full_and_host_classes_require_full_drain():
|
||||
requiring_full = {
|
||||
restart_class
|
||||
for restart_class, policy in rc.RESTART_CLASS_POLICIES.items()
|
||||
if policy.full_drain_required
|
||||
}
|
||||
assert requiring_full == {
|
||||
rc.RestartClass.FULL_MCP_RESTART,
|
||||
rc.RestartClass.HOST_RESTART,
|
||||
}
|
||||
@@ -0,0 +1,506 @@
|
||||
"""Unit and integration tests for Phase 2 Web Console recovery controls (#644)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import types
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
import merged_cleanup_reconcile
|
||||
import runtime_recovery_guard
|
||||
import stable_branch_push_guard
|
||||
import stale_binding_recovery
|
||||
from webui import console_authz, console_recovery, system_health
|
||||
from webui.app import create_app
|
||||
|
||||
|
||||
class TestConsoleRecovery(unittest.TestCase):
|
||||
|
||||
def test_diagnose_recovery_healthy(self) -> None:
|
||||
diag = console_recovery.diagnose_recovery()
|
||||
self.assertIn(diag.status, {console_recovery.STATUS_HEALTHY, console_recovery.STATUS_ACTION_REQUIRED})
|
||||
self.assertIsInstance(diag.playbooks, tuple)
|
||||
self.assertGreaterEqual(len(diag.playbooks), 4)
|
||||
|
||||
playbook_ids = {pb.playbook_id for pb in diag.playbooks}
|
||||
self.assertIn(console_recovery.PLAYBOOK_CLEAR_STALE_BINDING, playbook_ids)
|
||||
self.assertIn(console_recovery.PLAYBOOK_REBIND_SESSION, playbook_ids)
|
||||
self.assertIn(console_recovery.PLAYBOOK_RECONCILE_CLEANUPS, playbook_ids)
|
||||
self.assertIn(console_recovery.PLAYBOOK_SANCTIONED_RESTART, playbook_ids)
|
||||
|
||||
def test_confirmation_phrase_generation_and_matching(self) -> None:
|
||||
phrase = console_recovery.confirmation_phrase("clear_stale_binding")
|
||||
self.assertEqual(phrase, "confirm clear_stale_binding")
|
||||
self.assertTrue(console_recovery.confirmation_matches("clear_stale_binding", "confirm clear_stale_binding"))
|
||||
self.assertFalse(console_recovery.confirmation_matches("clear_stale_binding", "wrong phrase"))
|
||||
|
||||
phrase_target = console_recovery.confirmation_phrase("sanctioned_restart", "gitea-author")
|
||||
self.assertEqual(phrase_target, "confirm sanctioned_restart gitea-author")
|
||||
self.assertTrue(console_recovery.confirmation_matches("sanctioned_restart", "confirm sanctioned_restart gitea-author", "gitea-author"))
|
||||
|
||||
def test_build_recovery_preview(self) -> None:
|
||||
principal = console_authz.Principal("[email protected]", console_authz.OPERATOR, console_authz.IDENTITY_LOCAL_DEV, True)
|
||||
preview = console_recovery.build_recovery_preview(
|
||||
playbook_id=console_recovery.PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
target="test-worktree",
|
||||
principal=principal,
|
||||
)
|
||||
self.assertEqual(preview["playbook_id"], console_recovery.PLAYBOOK_CLEAR_STALE_BINDING)
|
||||
self.assertEqual(preview["action_id"], console_recovery.ACTION_CLEAR_STALE_BINDING)
|
||||
self.assertEqual(preview["confirmation_phrase"], "confirm clear_stale_binding test-worktree")
|
||||
self.assertTrue(len(preview["mutation_ledger"]) >= 3)
|
||||
self.assertTrue(preview["authorization"]["allowed"])
|
||||
|
||||
def test_build_recovery_preview_unknown_playbook(self) -> None:
|
||||
preview = console_recovery.build_recovery_preview("unknown_playbook")
|
||||
self.assertFalse(preview.get("allowed"))
|
||||
self.assertEqual(preview.get("error"), "unknown_playbook")
|
||||
|
||||
def test_execute_recovery_playbook_confirmation_mismatch(self) -> None:
|
||||
# Authorization is checked before confirmation, so the phase gate has to
|
||||
# pass for this test to reach the branch it is about.
|
||||
principal = console_authz.Principal("[email protected]", console_authz.OPERATOR, console_authz.IDENTITY_LOCAL_DEV, True)
|
||||
with self._phase_two_enabled():
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
confirmation="invalid confirmation",
|
||||
principal=principal,
|
||||
)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertFalse(result["allowed"])
|
||||
self.assertEqual(result["error"], "confirmation_mismatch")
|
||||
|
||||
def test_execute_recovery_playbook_unauthorized(self) -> None:
|
||||
# Anonymous principal has viewer role -> should be denied
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
confirmation="confirm clear_stale_binding",
|
||||
principal=console_authz.ANONYMOUS,
|
||||
)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertFalse(result["allowed"])
|
||||
self.assertEqual(result["error"], console_authz.DENY_UNAUTHENTICATED)
|
||||
|
||||
def test_execute_refuses_phase_two_write_while_console_is_phase_one(self) -> None:
|
||||
"""B1: the apply path must arm the phase gate, not skip it.
|
||||
|
||||
``authorize`` only applies the phase branch when ``for_execution=True``.
|
||||
The apply path used the default, so an operator executed a phase-2 write
|
||||
while ``ACTIVE_PHASE`` was 1.
|
||||
"""
|
||||
self.assertGreater(
|
||||
console_authz.get_action(console_recovery.ACTION_CLEAR_STALE_BINDING).phase,
|
||||
console_authz.ACTIVE_PHASE,
|
||||
"fixture assumes the recovery actions are ahead of the active phase",
|
||||
)
|
||||
principal = console_authz.Principal(
|
||||
"[email protected]", console_authz.OPERATOR, console_authz.IDENTITY_LOCAL_DEV, True
|
||||
)
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_CLEAR_STALE_BINDING
|
||||
)
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
confirmation=phrase,
|
||||
principal=principal,
|
||||
)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertFalse(result["allowed"])
|
||||
self.assertEqual(result["error"], console_authz.DENY_PHASE_NOT_ACTIVE)
|
||||
|
||||
def test_preview_execution_enabled_matches_the_execution_decision(self) -> None:
|
||||
"""B1: preview must not report a bare False it cannot explain."""
|
||||
principal = console_authz.Principal(
|
||||
"[email protected]", console_authz.OPERATOR, console_authz.IDENTITY_LOCAL_DEV, True
|
||||
)
|
||||
preview = console_recovery.build_recovery_preview(
|
||||
playbook_id=console_recovery.PLAYBOOK_REBIND_SESSION,
|
||||
target="branches/feat-issue-644",
|
||||
principal=principal,
|
||||
)
|
||||
self.assertFalse(preview["execution_enabled"])
|
||||
self.assertEqual(
|
||||
preview["execution_blocked_reason"], console_authz.DENY_PHASE_NOT_ACTIVE
|
||||
)
|
||||
self.assertFalse(preview["execution_authorization"]["allowed"])
|
||||
# The preview (non-execution) decision still allows, by role.
|
||||
self.assertTrue(preview["authorization"]["allowed"])
|
||||
|
||||
def _phase_two_enabled(self):
|
||||
"""Raise ACTIVE_PHASE so the execution branches are reachable in tests."""
|
||||
return patch.object(console_authz, "ACTIVE_PHASE", 2)
|
||||
|
||||
def _operator(self) -> console_authz.Principal:
|
||||
return console_authz.Principal(
|
||||
"[email protected]", console_authz.OPERATOR, console_authz.IDENTITY_LOCAL_DEV, True
|
||||
)
|
||||
|
||||
def test_rebind_mutates_the_live_environment_not_a_copy(self) -> None:
|
||||
"""B2: the playbook must change the mapping it claims to have changed."""
|
||||
live_env = {stale_binding_recovery.ACTIVE_WORKTREE_ENV: "branches/stale-old"}
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_REBIND_SESSION, "branches/feat-issue-644"
|
||||
)
|
||||
with self._phase_two_enabled():
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_REBIND_SESSION,
|
||||
confirmation=phrase,
|
||||
target="branches/feat-issue-644",
|
||||
principal=self._operator(),
|
||||
env=live_env,
|
||||
)
|
||||
self.assertTrue(result["success"])
|
||||
self.assertEqual(
|
||||
live_env[stale_binding_recovery.ACTIVE_WORKTREE_ENV],
|
||||
"branches/feat-issue-644",
|
||||
"rebind reported success without changing the caller's environment",
|
||||
)
|
||||
self.assertTrue(result["applied_result"]["binding_changed"])
|
||||
self.assertEqual(result["applied_result"]["binding_before"], "branches/stale-old")
|
||||
self.assertEqual(
|
||||
result["applied_result"]["binding_after"], "branches/feat-issue-644"
|
||||
)
|
||||
|
||||
def test_clear_stale_binding_reports_failure_when_nothing_changed(self) -> None:
|
||||
"""B2: a no-op recovery must never be reported as success."""
|
||||
live_env: dict[str, str] = {}
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_CLEAR_STALE_BINDING
|
||||
)
|
||||
with self._phase_two_enabled():
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
confirmation=phrase,
|
||||
principal=self._operator(),
|
||||
env=live_env,
|
||||
)
|
||||
self.assertFalse(
|
||||
result["success"],
|
||||
"a clear that changed no binding must not report success",
|
||||
)
|
||||
self.assertFalse(result["applied_result"]["binding_changed"])
|
||||
|
||||
def test_clear_stale_binding_clears_the_live_binding(self) -> None:
|
||||
"""B2: the sanctioned clear must reach the caller's environment."""
|
||||
missing = "/nonexistent/branches/deleted-worktree"
|
||||
live_env = {stale_binding_recovery.ACTIVE_WORKTREE_ENV: missing}
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_CLEAR_STALE_BINDING
|
||||
)
|
||||
with self._phase_two_enabled():
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
confirmation=phrase,
|
||||
principal=self._operator(),
|
||||
env=live_env,
|
||||
)
|
||||
if result["success"]:
|
||||
self.assertNotIn(stale_binding_recovery.ACTIVE_WORKTREE_ENV, live_env)
|
||||
self.assertEqual(result["applied_result"]["binding_before"], missing)
|
||||
self.assertIsNone(result["applied_result"]["binding_after"])
|
||||
else:
|
||||
# Fail closed is acceptable; reporting a clear that did not happen
|
||||
# is not. This is the invariant the blocker was about.
|
||||
self.assertFalse(result["applied_result"]["binding_changed"])
|
||||
self.assertEqual(
|
||||
live_env.get(stale_binding_recovery.ACTIVE_WORKTREE_ENV), missing
|
||||
)
|
||||
|
||||
def test_reconcile_playbook_calls_an_entry_point_that_exists(self) -> None:
|
||||
"""B3: the previous call named a function absent from the module."""
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_RECONCILE_CLEANUPS
|
||||
)
|
||||
fake_server = types.SimpleNamespace(
|
||||
gitea_reconcile_merged_cleanups=lambda **kwargs: {
|
||||
"success": True,
|
||||
"entries": [{"issue_number": 100}],
|
||||
}
|
||||
)
|
||||
with self._phase_two_enabled(), patch.dict(
|
||||
sys.modules, {"gitea_mcp_server": fake_server}
|
||||
):
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_RECONCILE_CLEANUPS,
|
||||
confirmation=phrase,
|
||||
principal=console_authz.Principal(
|
||||
"[email protected]",
|
||||
console_authz.ADMIN,
|
||||
console_authz.IDENTITY_LOCAL_DEV,
|
||||
True,
|
||||
),
|
||||
)
|
||||
self.assertTrue(result["success"], result.get("applied_result"))
|
||||
self.assertNotIn("error_type", result["applied_result"])
|
||||
self.assertEqual(result["applied_result"]["reconciled_count"], 1)
|
||||
|
||||
def test_reconcile_entry_point_exists_on_the_real_module(self) -> None:
|
||||
"""B3 regression: guard the symbol itself, not just the call shape."""
|
||||
import gitea_mcp_server
|
||||
|
||||
self.assertTrue(
|
||||
hasattr(gitea_mcp_server, "gitea_reconcile_merged_cleanups"),
|
||||
"console recovery depends on this reconciler entry point",
|
||||
)
|
||||
self.assertFalse(
|
||||
hasattr(merged_cleanup_reconcile, "reconcile_merged_cleanups"),
|
||||
"if this module grows the orchestrator, point the playbook back at it",
|
||||
)
|
||||
|
||||
def test_contamination_gate_blocks_a_writing_playbook(self) -> None:
|
||||
"""B4: a live marker plus a gated task key must actually block."""
|
||||
marker = {
|
||||
"kind": "manual_daemon_kill",
|
||||
"reason_class": "manual_daemon_kill",
|
||||
"command_summary": "pkill -f gitea_mcp_server",
|
||||
"active": True,
|
||||
}
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_REBIND_SESSION, "branches/feat-issue-644"
|
||||
)
|
||||
live_env = {stale_binding_recovery.ACTIVE_WORKTREE_ENV: "branches/stale-old"}
|
||||
with self._phase_two_enabled(), patch.object(
|
||||
console_recovery, "load_active_contamination_marker", return_value=marker
|
||||
):
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_REBIND_SESSION,
|
||||
confirmation=phrase,
|
||||
target="branches/feat-issue-644",
|
||||
principal=self._operator(),
|
||||
env=live_env,
|
||||
)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["error"], "contaminated_runtime")
|
||||
self.assertEqual(
|
||||
live_env[stale_binding_recovery.ACTIVE_WORKTREE_ENV],
|
||||
"branches/stale-old",
|
||||
"a blocked playbook must not have mutated anything",
|
||||
)
|
||||
|
||||
def test_contamination_gate_exempts_the_reconciler_remedy(self) -> None:
|
||||
"""B4: the designated remedy must stay reachable while contaminated."""
|
||||
marker = {
|
||||
"kind": "manual_daemon_kill",
|
||||
"reason_class": "manual_daemon_kill",
|
||||
"command_summary": "pkill -f gitea_mcp_server",
|
||||
"active": True,
|
||||
}
|
||||
phrase = console_recovery.confirmation_phrase(
|
||||
console_recovery.PLAYBOOK_RECONCILE_CLEANUPS
|
||||
)
|
||||
fake_server = types.SimpleNamespace(
|
||||
gitea_reconcile_merged_cleanups=lambda **kwargs: {
|
||||
"success": True,
|
||||
"entries": [],
|
||||
}
|
||||
)
|
||||
with self._phase_two_enabled(), patch.object(
|
||||
console_recovery, "load_active_contamination_marker", return_value=marker
|
||||
), patch.dict(sys.modules, {"gitea_mcp_server": fake_server}):
|
||||
result = console_recovery.execute_recovery_playbook(
|
||||
playbook_id=console_recovery.PLAYBOOK_RECONCILE_CLEANUPS,
|
||||
confirmation=phrase,
|
||||
principal=console_authz.Principal(
|
||||
"[email protected]",
|
||||
console_authz.ADMIN,
|
||||
console_authz.IDENTITY_LOCAL_DEV,
|
||||
True,
|
||||
),
|
||||
)
|
||||
self.assertNotEqual(result.get("error"), "contaminated_runtime")
|
||||
|
||||
def test_gated_task_key_is_actually_gated(self) -> None:
|
||||
"""B4: the console action id was never a member of the gated set."""
|
||||
self.assertIn(
|
||||
console_recovery.CONTAMINATION_GATED_TASK,
|
||||
stable_branch_push_guard.CONTAMINATION_GATED_TASKS,
|
||||
)
|
||||
self.assertNotIn(
|
||||
console_recovery.ACTION_CLEAR_STALE_BINDING,
|
||||
stable_branch_push_guard.CONTAMINATION_GATED_TASKS,
|
||||
)
|
||||
|
||||
def test_diagnosis_reads_the_key_the_gate_returns(self) -> None:
|
||||
"""B4: ``contaminated`` is a key assess_contamination_gate never returns."""
|
||||
gate = runtime_recovery_guard.assess_contamination_gate(
|
||||
None, task=console_recovery.CONTAMINATION_GATED_TASK, actual_role="operator"
|
||||
)
|
||||
self.assertNotIn("contaminated", gate)
|
||||
self.assertIn("block", gate)
|
||||
|
||||
def test_contaminated_runtime_is_reported_unclean(self) -> None:
|
||||
"""B4: verify_post_recovery reported contamination_clean unconditionally."""
|
||||
marker = {
|
||||
"kind": "manual_daemon_kill",
|
||||
"reason_class": "manual_daemon_kill",
|
||||
"command_summary": "pkill -f gitea_mcp_server",
|
||||
"active": True,
|
||||
}
|
||||
with patch.object(
|
||||
console_recovery, "load_active_contamination_marker", return_value=marker
|
||||
):
|
||||
verification = console_recovery.verify_post_recovery()
|
||||
diag = console_recovery.diagnose_recovery()
|
||||
self.assertFalse(verification["contamination_clean"])
|
||||
self.assertFalse(verification["clean"])
|
||||
self.assertEqual(diag.status, console_recovery.STATUS_BLOCKED_CONTAMINATION)
|
||||
|
||||
def test_master_parity_baseline_is_not_the_head_it_is_compared_against(self) -> None:
|
||||
"""B5: capture_startup_parity was fed the head it was then compared to."""
|
||||
stale = system_health.StaleRuntime(
|
||||
daemon_head="a" * 40,
|
||||
checkout_head="b" * 40,
|
||||
remote_head="b" * 40,
|
||||
stale=True,
|
||||
determinable=True,
|
||||
mutation_safe=False,
|
||||
reasons=("daemon is behind the checkout",),
|
||||
)
|
||||
with patch.object(system_health, "assess_stale_runtime", return_value=stale):
|
||||
diag = console_recovery.diagnose_recovery()
|
||||
parity = diag.master_parity
|
||||
self.assertEqual(parity["startup_head"], "a" * 40)
|
||||
self.assertEqual(parity["current_head"], "b" * 40)
|
||||
self.assertNotEqual(parity["startup_head"], parity["current_head"])
|
||||
self.assertFalse(parity["in_parity"])
|
||||
|
||||
def test_master_parity_carries_the_live_remote_dimension(self) -> None:
|
||||
"""B5: live_remote_head was never passed, dropping the #610 dimension."""
|
||||
stale = system_health.StaleRuntime(
|
||||
daemon_head="c" * 40,
|
||||
checkout_head="c" * 40,
|
||||
remote_head="d" * 40,
|
||||
stale=False,
|
||||
determinable=True,
|
||||
mutation_safe=False,
|
||||
reasons=(),
|
||||
)
|
||||
with patch.object(system_health, "assess_stale_runtime", return_value=stale):
|
||||
diag = console_recovery.diagnose_recovery()
|
||||
self.assertEqual(diag.master_parity.get("live_remote_head"), "d" * 40)
|
||||
|
||||
def test_verify_post_recovery(self) -> None:
|
||||
verification = console_recovery.verify_post_recovery()
|
||||
self.assertIn("clean", verification)
|
||||
self.assertIn("status", verification)
|
||||
self.assertIn("reasons", verification)
|
||||
|
||||
def test_unverified_inherited_binding_is_not_reported_clean(self) -> None:
|
||||
"""B2: ``not clear_eligible`` also read clean for unproven bindings."""
|
||||
binding = {
|
||||
"classification": stale_binding_recovery.CLASSIFICATION_UNVERIFIED_INHERITED,
|
||||
"clear_eligible": False,
|
||||
}
|
||||
diag = console_recovery.diagnose_recovery()
|
||||
patched = console_recovery.RecoveryDiagnosis(
|
||||
status=diag.status,
|
||||
clean=diag.clean,
|
||||
stale_runtime=diag.stale_runtime,
|
||||
master_parity=diag.master_parity,
|
||||
stale_binding=binding,
|
||||
contamination=diag.contamination,
|
||||
worktree_anomalies=diag.worktree_anomalies,
|
||||
playbooks=diag.playbooks,
|
||||
reasons=diag.reasons,
|
||||
)
|
||||
with patch.object(console_recovery, "diagnose_recovery", return_value=patched):
|
||||
verification = console_recovery.verify_post_recovery()
|
||||
self.assertFalse(verification["binding_clean"])
|
||||
self.assertEqual(
|
||||
verification["binding_classification"],
|
||||
stale_binding_recovery.CLASSIFICATION_UNVERIFIED_INHERITED,
|
||||
)
|
||||
|
||||
|
||||
class TestConsoleRecoveryApi(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self.app = create_app()
|
||||
self.client = TestClient(self.app)
|
||||
|
||||
def test_api_recovery_diagnose(self) -> None:
|
||||
res = self.client.get("/api/v1/system/recovery/diagnose")
|
||||
self.assertEqual(res.status_code, 200)
|
||||
data = res.json()
|
||||
self.assertIn("status", data)
|
||||
self.assertIn("clean", data)
|
||||
self.assertIn("playbooks", data)
|
||||
self.assertTrue(len(data["playbooks"]) >= 4)
|
||||
|
||||
def test_api_recovery_preview(self) -> None:
|
||||
res = self.client.post(
|
||||
"/api/v1/system/recovery/preview",
|
||||
json={"playbook_id": "clear_stale_binding", "target": "active"},
|
||||
)
|
||||
self.assertEqual(res.status_code, 200)
|
||||
data = res.json()
|
||||
self.assertEqual(data["playbook_id"], "clear_stale_binding")
|
||||
self.assertEqual(data["confirmation_phrase"], "confirm clear_stale_binding active")
|
||||
self.assertIn("mutation_ledger", data)
|
||||
|
||||
def test_api_recovery_apply_denied_without_auth(self) -> None:
|
||||
res = self.client.post(
|
||||
"/api/v1/system/recovery/apply",
|
||||
json={"playbook_id": "clear_stale_binding", "confirmation": "confirm clear_stale_binding"},
|
||||
)
|
||||
self.assertEqual(res.status_code, 400)
|
||||
data = res.json()
|
||||
self.assertFalse(data["success"])
|
||||
self.assertFalse(data["allowed"])
|
||||
|
||||
def test_api_recovery_apply_refuses_phase_two_write_with_dev_auth(self) -> None:
|
||||
"""B1: this previously asserted the phase-gate bypass as intended.
|
||||
|
||||
An authenticated operator posting a valid confirmation still must not
|
||||
execute a phase-2 write while the console is in phase 1. The refusal is
|
||||
the contract; a 200 here means the gate is not armed.
|
||||
"""
|
||||
env = {
|
||||
"WEBUI_AUTH_MODE": "local_dev",
|
||||
"WEBUI_DEV_SUBJECT": "[email protected]",
|
||||
"WEBUI_DEV_ROLE": "operator",
|
||||
}
|
||||
before = os.environ.get("GITEA_ACTIVE_WORKTREE")
|
||||
with patch.dict(os.environ, env):
|
||||
res = self.client.post(
|
||||
"/api/v1/system/recovery/apply",
|
||||
json={
|
||||
"playbook_id": "rebind_session_worktree",
|
||||
"target": "branches/feat-issue-644",
|
||||
"confirmation": "confirm rebind_session_worktree branches/feat-issue-644",
|
||||
},
|
||||
)
|
||||
self.assertEqual(res.status_code, 400)
|
||||
data = res.json()
|
||||
self.assertFalse(data["success"])
|
||||
self.assertFalse(data["allowed"])
|
||||
self.assertEqual(data["error"], console_authz.DENY_PHASE_NOT_ACTIVE)
|
||||
self.assertEqual(
|
||||
os.environ.get("GITEA_ACTIVE_WORKTREE"),
|
||||
before,
|
||||
"a refused apply must not have rebound the live process environment",
|
||||
)
|
||||
|
||||
def test_api_recovery_preview_reports_why_execution_is_disabled(self) -> None:
|
||||
res = self.client.post(
|
||||
"/api/v1/system/recovery/preview",
|
||||
json={"playbook_id": "rebind_session_worktree", "target": "active"},
|
||||
)
|
||||
self.assertEqual(res.status_code, 200)
|
||||
data = res.json()
|
||||
self.assertFalse(data["execution_enabled"])
|
||||
self.assertIn("execution_authorization", data)
|
||||
|
||||
def test_api_recovery_verify(self) -> None:
|
||||
res = self.client.get("/api/v1/system/recovery/verify")
|
||||
self.assertEqual(res.status_code, 200)
|
||||
data = res.json()
|
||||
self.assertIn("clean", data)
|
||||
self.assertIn("status", data)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,511 @@
|
||||
"""Tests for the Gitea issue↔PR linkage console (#645, Phase 3).
|
||||
|
||||
Covers the acceptance criteria of the issue:
|
||||
|
||||
* AC1 — issue↔PR linkage is visible for the selected project/repo, in both
|
||||
directions, with the evidence that produced each edge.
|
||||
* AC2 — the latest canonical handoff (CTH) is summarized for a focused thread.
|
||||
* AC3 — an external Gitea link appears only under the admin reveal opt-in.
|
||||
* AC4 — every case is driven by mocked Gitea payloads; no network.
|
||||
|
||||
Plus the invariants this console must not violate: a partial or failed read is
|
||||
never rendered as "no link exists", an unfetched thread is never rendered as
|
||||
"no handoff", redaction happens before display, and the surface stays read-only.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from tests.webui_testclient import TestClient
|
||||
|
||||
from canonical_thread_handoff import format_cth_body
|
||||
from webui.app import create_app
|
||||
from webui.linkage_loader import (
|
||||
EVIDENCE_BRANCH,
|
||||
EVIDENCE_CLOSES,
|
||||
EVIDENCE_REFERENCE,
|
||||
HANDOFF_LOADED,
|
||||
HANDOFF_NOT_LOADED,
|
||||
HANDOFF_UNAVAILABLE,
|
||||
LinkageSnapshot,
|
||||
load_linkage_snapshot,
|
||||
resolve_linkage,
|
||||
resolve_pr_links,
|
||||
snapshot_to_dict,
|
||||
summarize_handoff,
|
||||
)
|
||||
from webui.linkage_views import render_linkage_page
|
||||
from webui.nav import nav_hrefs
|
||||
from webui.queue_loader import PaginationMeta
|
||||
|
||||
|
||||
def _pagination(*, complete: bool = True, count: int = 0) -> PaginationMeta:
|
||||
return PaginationMeta(
|
||||
page=1,
|
||||
per_page=50,
|
||||
returned_count=count,
|
||||
has_more=not complete,
|
||||
is_final_page=complete,
|
||||
inventory_complete=complete,
|
||||
pages_fetched=1,
|
||||
)
|
||||
|
||||
|
||||
def _pr(
|
||||
number: int,
|
||||
*,
|
||||
title: str = "",
|
||||
body: str = "",
|
||||
head: str = "",
|
||||
state: str = "open",
|
||||
labels: tuple[str, ...] = (),
|
||||
) -> dict:
|
||||
return {
|
||||
"number": number,
|
||||
"title": title or f"pr {number}",
|
||||
"body": body,
|
||||
"state": state,
|
||||
"head": {"ref": head},
|
||||
"labels": [{"name": name} for name in labels],
|
||||
}
|
||||
|
||||
|
||||
def _issue(
|
||||
number: int,
|
||||
*,
|
||||
title: str = "",
|
||||
state: str = "open",
|
||||
labels: tuple[str, ...] = (),
|
||||
) -> dict:
|
||||
return {
|
||||
"number": number,
|
||||
"title": title or f"issue {number}",
|
||||
"state": state,
|
||||
"labels": [{"name": name} for name in labels],
|
||||
}
|
||||
|
||||
|
||||
def _fetcher(items: list[dict], *, complete: bool = True):
|
||||
def _fetch(*_args, **_kwargs):
|
||||
return items, _pagination(complete=complete, count=len(items))
|
||||
|
||||
return _fetch
|
||||
|
||||
|
||||
def _load(
|
||||
issues: list[dict],
|
||||
prs: list[dict],
|
||||
*,
|
||||
complete: bool = True,
|
||||
**kwargs,
|
||||
) -> LinkageSnapshot:
|
||||
return load_linkage_snapshot(
|
||||
fetch_prs=_fetcher(prs, complete=complete),
|
||||
fetch_issues=_fetcher(issues, complete=complete),
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
|
||||
def _cth(comment_id: int, *, created_at: str, status: str, next_owner: str) -> dict:
|
||||
return {
|
||||
"id": comment_id,
|
||||
"created_at": created_at,
|
||||
"user": {"login": "jcwalker3"},
|
||||
"body": format_cth_body(
|
||||
cth_type="Author Handoff",
|
||||
status=status,
|
||||
next_owner=next_owner,
|
||||
current_blocker="none",
|
||||
decision="implemented",
|
||||
proof="full suite green",
|
||||
next_action="review PR",
|
||||
ready_to_paste_prompt="Review PR #902 now.",
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
class TestLinkageEvidence(unittest.TestCase):
|
||||
"""AC1 — every edge records how it was found, and keeps all candidates."""
|
||||
|
||||
def test_closes_keyword_in_body_is_strongest_evidence(self):
|
||||
links = resolve_pr_links(_pr(902, body="Closes #643"))
|
||||
self.assertEqual([link.issue_number for link in links], [643])
|
||||
self.assertEqual(links[0].evidence, (EVIDENCE_CLOSES,))
|
||||
self.assertTrue(links[0].closes)
|
||||
|
||||
def test_closes_keyword_in_title_counts(self):
|
||||
links = resolve_pr_links(_pr(902, title="feat(webui): preview (Closes #643)"))
|
||||
self.assertEqual(links[0].evidence, (EVIDENCE_CLOSES,))
|
||||
|
||||
def test_canonical_branch_marker_links_without_a_keyword(self):
|
||||
links = resolve_pr_links(_pr(902, head="feat/issue-643-request-preview"))
|
||||
self.assertEqual([link.issue_number for link in links], [643])
|
||||
self.assertEqual(links[0].evidence, (EVIDENCE_BRANCH,))
|
||||
self.assertFalse(links[0].closes)
|
||||
|
||||
def test_non_canonical_branch_is_not_treated_as_a_marker(self):
|
||||
self.assertEqual(resolve_pr_links(_pr(902, head="issue-643-preview")), ())
|
||||
|
||||
def test_bare_mention_is_recorded_as_the_weakest_evidence(self):
|
||||
links = resolve_pr_links(_pr(902, body="context in #643"))
|
||||
self.assertEqual(links[0].evidence, (EVIDENCE_REFERENCE,))
|
||||
self.assertFalse(links[0].closes)
|
||||
|
||||
def test_several_evidence_kinds_merge_onto_one_edge(self):
|
||||
links = resolve_pr_links(
|
||||
_pr(902, body="Closes #643 — see #643", head="feat/issue-643-preview")
|
||||
)
|
||||
self.assertEqual(len(links), 1)
|
||||
self.assertEqual(
|
||||
links[0].evidence,
|
||||
(EVIDENCE_CLOSES, EVIDENCE_BRANCH, EVIDENCE_REFERENCE),
|
||||
)
|
||||
|
||||
def test_stronger_evidence_sorts_first(self):
|
||||
links = resolve_pr_links(_pr(902, body="Closes #643, related #700"))
|
||||
self.assertEqual([link.issue_number for link in links], [643, 700])
|
||||
|
||||
def test_self_reference_is_not_linkage(self):
|
||||
links = resolve_pr_links(_pr(902, body="supersedes #902"))
|
||||
self.assertEqual(links, ())
|
||||
|
||||
def test_every_candidate_is_kept_never_collapsed_to_a_guess(self):
|
||||
links = resolve_pr_links(_pr(902, body="Closes #643\nCloses #644"))
|
||||
self.assertEqual([link.issue_number for link in links], [643, 644])
|
||||
|
||||
|
||||
class TestLinkageIndex(unittest.TestCase):
|
||||
def test_issue_direction_ignores_mention_only_edges(self):
|
||||
index = resolve_linkage([_pr(902, body="context in #643")])
|
||||
self.assertIsNone(index.issue_prs.get(643))
|
||||
self.assertEqual(index.pr_links[902][0].evidence, (EVIDENCE_REFERENCE,))
|
||||
|
||||
def test_contested_issue_is_reported_when_two_prs_claim_it(self):
|
||||
index = resolve_linkage(
|
||||
[_pr(902, body="Closes #643"), _pr(903, head="feat/issue-643-again")]
|
||||
)
|
||||
self.assertEqual(index.contested_issues(), (643,))
|
||||
self.assertEqual(index.issue_prs[643], (902, 903))
|
||||
|
||||
def test_single_claim_is_not_contested(self):
|
||||
index = resolve_linkage([_pr(902, body="Closes #643")])
|
||||
self.assertEqual(index.contested_issues(), ())
|
||||
|
||||
def test_ambiguous_when_two_issues_tie_at_the_strongest_evidence(self):
|
||||
index = resolve_linkage([_pr(902, body="Closes #643\nCloses #644")])
|
||||
self.assertTrue(index.ambiguous(902))
|
||||
|
||||
def test_weaker_candidate_alongside_a_stronger_one_is_not_ambiguous(self):
|
||||
index = resolve_linkage([_pr(902, body="Closes #643, see #700")])
|
||||
self.assertFalse(index.ambiguous(902))
|
||||
self.assertEqual(index.primary_issue(902).issue_number, 643)
|
||||
|
||||
def test_malformed_pr_row_is_skipped_not_raised_on(self):
|
||||
index = resolve_linkage([{"title": "no number"}, _pr(902, body="Closes #643")])
|
||||
self.assertEqual(sorted(index.pr_links), [902])
|
||||
|
||||
|
||||
class TestLinkageSnapshot(unittest.TestCase):
|
||||
"""AC1 — linkage is visible per project/repo, in both directions."""
|
||||
|
||||
def test_both_directions_are_populated(self):
|
||||
snapshot = _load([_issue(643)], [_pr(902, body="Closes #643")])
|
||||
self.assertTrue(snapshot.ok)
|
||||
self.assertEqual([node.number for node in snapshot.issues], [643])
|
||||
self.assertEqual(snapshot.issues[0].linked_prs, (902,))
|
||||
self.assertEqual(snapshot.prs[0].links[0].issue_number, 643)
|
||||
|
||||
def test_repo_scope_comes_from_the_registry_project(self):
|
||||
snapshot = _load([], [])
|
||||
self.assertIn("/", snapshot.repo_label)
|
||||
self.assertTrue(snapshot.project_id)
|
||||
|
||||
def test_unknown_project_fails_closed_with_a_reason(self):
|
||||
snapshot = _load([_issue(643)], [], project_id="no-such-project")
|
||||
self.assertFalse(snapshot.ok)
|
||||
self.assertIn("not found in registry", snapshot.fetch_error)
|
||||
self.assertEqual(snapshot.issues, ())
|
||||
|
||||
def test_orphan_pr_is_identifiable(self):
|
||||
snapshot = _load([], [_pr(902), _pr(903, body="Closes #643")])
|
||||
self.assertEqual([node.number for node in snapshot.orphan_prs], [902])
|
||||
|
||||
def test_state_scope_defaults_to_open_and_is_reported(self):
|
||||
self.assertEqual(_load([], []).state_scope, "open")
|
||||
self.assertEqual(_load([], [], state="all").state_scope, "all")
|
||||
|
||||
def test_unsupported_state_falls_back_to_open(self):
|
||||
self.assertEqual(_load([], [], state="../etc").state_scope, "open")
|
||||
|
||||
def test_state_is_passed_through_to_the_fetchers(self):
|
||||
seen: list[str] = []
|
||||
|
||||
def _fetch(*_args, **kwargs):
|
||||
seen.append(kwargs.get("state", ""))
|
||||
return [], _pagination()
|
||||
|
||||
load_linkage_snapshot(state="all", fetch_prs=_fetch, fetch_issues=_fetch)
|
||||
self.assertEqual(seen, ["all", "all"])
|
||||
|
||||
|
||||
class TestPartialInventoryIsNotAnAbsenceClaim(unittest.TestCase):
|
||||
"""An empty edge list from a partial read must never read as 'no link'."""
|
||||
|
||||
def test_incomplete_pagination_marks_links_non_authoritative(self):
|
||||
snapshot = _load([_issue(643)], [], complete=False)
|
||||
self.assertFalse(snapshot.inventory_complete)
|
||||
self.assertFalse(snapshot.issues[0].links_authoritative)
|
||||
|
||||
def test_complete_pagination_marks_links_authoritative(self):
|
||||
snapshot = _load([_issue(643)], [], complete=True)
|
||||
self.assertTrue(snapshot.inventory_complete)
|
||||
self.assertTrue(snapshot.issues[0].links_authoritative)
|
||||
|
||||
def test_partial_window_renders_a_qualified_empty_cell(self):
|
||||
html = render_linkage_page(_load([_issue(643)], [], complete=False))
|
||||
self.assertIn("none found (partial inventory)", html)
|
||||
|
||||
def test_complete_window_renders_a_plain_none(self):
|
||||
html = render_linkage_page(_load([_issue(643)], [], complete=True))
|
||||
self.assertNotIn("partial inventory", html)
|
||||
self.assertIn(">none<", html)
|
||||
|
||||
def test_missing_credentials_fail_closed_without_a_table(self):
|
||||
with mock.patch(
|
||||
"webui.linkage_loader._offline_test_mode", return_value=False
|
||||
), mock.patch("webui.linkage_loader.get_auth_header", return_value=""):
|
||||
snapshot = load_linkage_snapshot()
|
||||
self.assertFalse(snapshot.ok)
|
||||
self.assertIn("credentials unavailable", snapshot.fetch_error)
|
||||
html = render_linkage_page(snapshot)
|
||||
self.assertIn("Linkage unavailable", html)
|
||||
self.assertNotIn("Issues → pull requests", html)
|
||||
|
||||
def test_fetch_failure_is_reported_not_raised(self):
|
||||
def _boom(*_args, **_kwargs):
|
||||
raise RuntimeError("gitea 502")
|
||||
|
||||
snapshot = load_linkage_snapshot(fetch_prs=_boom, fetch_issues=_boom)
|
||||
self.assertFalse(snapshot.ok)
|
||||
self.assertIn("Gitea fetch failed", snapshot.fetch_error)
|
||||
|
||||
|
||||
class TestHandoffSummary(unittest.TestCase):
|
||||
"""AC2 — the latest canonical handoff is summarized for a focused thread."""
|
||||
|
||||
def test_latest_cth_wins(self):
|
||||
summary = summarize_handoff([
|
||||
_cth(1, created_at="2026-07-24T10:00:00Z", status="in progress",
|
||||
next_owner="author"),
|
||||
_cth(2, created_at="2026-07-25T10:00:00Z", status="PR-open",
|
||||
next_owner="reviewer"),
|
||||
])
|
||||
self.assertEqual(summary.comment_id, 2)
|
||||
self.assertEqual(summary.status, "PR-open")
|
||||
self.assertEqual(summary.next_owner, "reviewer")
|
||||
self.assertTrue(summary.cth_type_known)
|
||||
|
||||
def test_thread_without_a_cth_summarizes_to_none(self):
|
||||
self.assertIsNone(summarize_handoff([{"id": 1, "body": "ordinary comment"}]))
|
||||
|
||||
def test_unknown_heading_is_reported_not_republished(self):
|
||||
summary = summarize_handoff([
|
||||
{
|
||||
"id": 5,
|
||||
"created_at": "2026-07-25T10:00:00Z",
|
||||
"user": {"login": "someone"},
|
||||
"body": "<!-- cth:v1 -->\n## CTH: Totally Made Up\n\nStatus: odd\n",
|
||||
}
|
||||
])
|
||||
self.assertFalse(summary.cth_type_known)
|
||||
self.assertEqual(summary.cth_type, "unrecognized")
|
||||
self.assertNotIn("Totally Made Up", json.dumps(summary.to_dict()))
|
||||
|
||||
def test_focused_pr_loads_its_handoff(self):
|
||||
snapshot = _load(
|
||||
[_issue(643)],
|
||||
[_pr(902, body="Closes #643")],
|
||||
pr=902,
|
||||
comment_source=lambda kind, number: [
|
||||
_cth(2, created_at="2026-07-25T10:00:00Z", status="PR-open",
|
||||
next_owner="reviewer")
|
||||
],
|
||||
)
|
||||
self.assertEqual(snapshot.handoff_status.state, HANDOFF_LOADED)
|
||||
self.assertEqual(snapshot.focus, ("pr", 902))
|
||||
self.assertEqual(snapshot.prs[0].handoff.status, "PR-open")
|
||||
|
||||
def test_unfocused_rows_report_not_loaded_never_none(self):
|
||||
snapshot = _load(
|
||||
[_issue(643)],
|
||||
[_pr(902, body="Closes #643"), _pr(903)],
|
||||
pr=902,
|
||||
comment_source=lambda kind, number: [],
|
||||
)
|
||||
other = next(node for node in snapshot.prs if node.number == 903)
|
||||
self.assertIsNone(other.handoff)
|
||||
self.assertEqual(other.handoff_status.state, HANDOFF_NOT_LOADED)
|
||||
self.assertIn("not loaded", render_linkage_page(snapshot))
|
||||
|
||||
def test_no_focus_means_no_thread_is_claimed_handoff_free(self):
|
||||
snapshot = _load([_issue(643)], [])
|
||||
self.assertEqual(snapshot.handoff_status.state, HANDOFF_NOT_LOADED)
|
||||
self.assertIn("thread-scoped", snapshot.handoff_status.reason)
|
||||
|
||||
def test_comment_source_failure_degrades_only_the_handoff(self):
|
||||
def _boom(_kind, _number):
|
||||
raise RuntimeError("comments 500")
|
||||
|
||||
snapshot = _load(
|
||||
[_issue(643)], [_pr(902, body="Closes #643")], pr=902, comment_source=_boom
|
||||
)
|
||||
self.assertTrue(snapshot.ok)
|
||||
self.assertEqual(snapshot.handoff_status.state, HANDOFF_UNAVAILABLE)
|
||||
self.assertEqual(snapshot.issues[0].linked_prs, (902,))
|
||||
self.assertIn("unavailable", render_linkage_page(snapshot))
|
||||
|
||||
def test_loaded_thread_with_no_cth_says_so_explicitly(self):
|
||||
snapshot = _load(
|
||||
[_issue(643)],
|
||||
[_pr(902, body="Closes #643")],
|
||||
pr=902,
|
||||
comment_source=lambda kind, number: [{"id": 1, "body": "hi"}],
|
||||
)
|
||||
self.assertIn(
|
||||
"no Canonical Thread Handoff comment found", render_linkage_page(snapshot)
|
||||
)
|
||||
|
||||
|
||||
class TestDeepLinks(unittest.TestCase):
|
||||
"""AC3 — an external Gitea link is emitted only when permitted."""
|
||||
|
||||
def test_deep_links_are_withheld_by_default(self):
|
||||
with mock.patch.dict(os.environ, {"GITEA_MCP_REVEAL_ENDPOINTS": ""}):
|
||||
snapshot = _load([_issue(643)], [])
|
||||
html = render_linkage_page(snapshot)
|
||||
self.assertFalse(snapshot.deep_links_enabled)
|
||||
self.assertIsNone(snapshot.issues[0].deep_link)
|
||||
self.assertIn("Gitea deep links are withheld", html)
|
||||
|
||||
def test_reveal_opt_in_emits_the_link(self):
|
||||
with mock.patch.dict(os.environ, {"GITEA_MCP_REVEAL_ENDPOINTS": "1"}):
|
||||
snapshot = _load([_issue(643)], [_pr(902, body="Closes #643")])
|
||||
html = render_linkage_page(snapshot)
|
||||
self.assertTrue(snapshot.deep_links_enabled)
|
||||
self.assertIn("/issues/643", snapshot.issues[0].deep_link)
|
||||
self.assertIn("/pulls/902", snapshot.prs[0].deep_link)
|
||||
self.assertIn(f'href="{snapshot.issues[0].deep_link}"', html)
|
||||
|
||||
|
||||
class TestRedactionBoundary(unittest.TestCase):
|
||||
def test_secret_shaped_title_is_redacted_before_display(self):
|
||||
snapshot = _load(
|
||||
[_issue(643, title="token=ghp_thisisnotarealsecretvalue0001")], []
|
||||
)
|
||||
payload = json.dumps(snapshot_to_dict(snapshot))
|
||||
self.assertNotIn("ghp_thisisnotarealsecretvalue0001", payload)
|
||||
self.assertNotIn(
|
||||
"ghp_thisisnotarealsecretvalue0001", render_linkage_page(snapshot)
|
||||
)
|
||||
|
||||
def test_handoff_fields_are_redacted(self):
|
||||
comment = _cth(
|
||||
2, created_at="2026-07-25T10:00:00Z", status="ok", next_owner="reviewer"
|
||||
)
|
||||
comment["body"] += "\nDecision: password=hunter2hunter2\n"
|
||||
snapshot = _load(
|
||||
[_issue(643)],
|
||||
[_pr(902, body="Closes #643")],
|
||||
pr=902,
|
||||
comment_source=lambda kind, number: [comment],
|
||||
)
|
||||
self.assertNotIn("hunter2hunter2", json.dumps(snapshot_to_dict(snapshot)))
|
||||
self.assertNotIn("hunter2hunter2", render_linkage_page(snapshot))
|
||||
|
||||
def test_html_escapes_markup_in_a_title(self):
|
||||
snapshot = _load([_issue(643, title="<script>alert(1)</script>")], [])
|
||||
html = render_linkage_page(snapshot)
|
||||
self.assertNotIn("<script>alert(1)</script>", html)
|
||||
self.assertIn("<script>", html)
|
||||
|
||||
|
||||
class TestLinkageRoutes(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.snapshot = _load(
|
||||
[_issue(643, labels=("status:ready",))],
|
||||
[_pr(902, body="Closes #643", labels=("status:pr-open",))],
|
||||
)
|
||||
self.client = TestClient(create_app())
|
||||
|
||||
def test_page_renders_both_tables(self):
|
||||
with mock.patch("webui.app.load_linkage_snapshot", return_value=self.snapshot):
|
||||
response = self.client.get("/gitea")
|
||||
self.assertEqual(response.status_code, 200)
|
||||
self.assertIn("Issues → pull requests", response.text)
|
||||
self.assertIn("Pull requests → issues", response.text)
|
||||
self.assertIn("#643", response.text)
|
||||
|
||||
def test_api_exports_the_same_model(self):
|
||||
with mock.patch("webui.app.load_linkage_snapshot", return_value=self.snapshot):
|
||||
response = self.client.get("/api/v1/gitea/linkage")
|
||||
self.assertEqual(response.status_code, 200)
|
||||
payload = response.json()
|
||||
self.assertTrue(payload["ok"])
|
||||
self.assertEqual(payload["issues"][0]["linked_prs"], [902])
|
||||
self.assertEqual(payload["prs"][0]["links"][0]["issue_number"], 643)
|
||||
self.assertEqual(payload["schema_version"], 1)
|
||||
|
||||
def test_api_declares_the_evidence_vocabulary(self):
|
||||
with mock.patch("webui.app.load_linkage_snapshot", return_value=self.snapshot):
|
||||
payload = self.client.get("/api/v1/gitea/linkage").json()
|
||||
names = {entry["name"] for entry in payload["evidence_kinds"]}
|
||||
self.assertEqual(names, {EVIDENCE_CLOSES, EVIDENCE_BRANCH, EVIDENCE_REFERENCE})
|
||||
|
||||
def test_api_fails_closed_with_a_non_200_when_the_read_failed(self):
|
||||
failed = _load([], [], project_id="no-such-project")
|
||||
with mock.patch("webui.app.load_linkage_snapshot", return_value=failed):
|
||||
response = self.client.get("/api/v1/gitea/linkage")
|
||||
self.assertEqual(response.status_code, 502)
|
||||
self.assertFalse(response.json()["ok"])
|
||||
|
||||
def test_page_still_renders_when_the_read_failed(self):
|
||||
failed = _load([], [], project_id="no-such-project")
|
||||
with mock.patch("webui.app.load_linkage_snapshot", return_value=failed):
|
||||
response = self.client.get("/gitea")
|
||||
self.assertEqual(response.status_code, 200)
|
||||
self.assertIn("Linkage unavailable", response.text)
|
||||
|
||||
def test_query_parameters_reach_the_loader(self):
|
||||
with mock.patch(
|
||||
"webui.app.load_linkage_snapshot", return_value=self.snapshot
|
||||
) as loader:
|
||||
self.client.get("/gitea?project=gitea-tools&state=all&pr=902")
|
||||
loader.assert_called_once()
|
||||
args, kwargs = loader.call_args
|
||||
self.assertEqual(args[0], "gitea-tools")
|
||||
self.assertEqual(kwargs["state"], "all")
|
||||
self.assertEqual(kwargs["pr"], 902)
|
||||
self.assertIsNone(kwargs["issue"])
|
||||
|
||||
def test_surface_stays_read_only(self):
|
||||
for path in ("/gitea", "/api/v1/gitea/linkage"):
|
||||
with self.subTest(path=path):
|
||||
self.assertEqual(self.client.post(path).status_code, 405)
|
||||
|
||||
def test_nav_exposes_the_linkage_page_as_live(self):
|
||||
self.assertIn("/gitea", nav_hrefs())
|
||||
home = self.client.get("/").text
|
||||
self.assertIn('href="/gitea"', home)
|
||||
self.assertIn(">Gitea<", home)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,465 @@
|
||||
"""Unit tests for Phase 3 Notifications and Human-Attention Console (#648)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from webui.app import create_app
|
||||
from webui.notifications import (
|
||||
ATTENTION_HUMAN_REQUIRED,
|
||||
ATTENTION_OPERATOR,
|
||||
ATTENTION_ROUTINE,
|
||||
CATEGORY_AUTH,
|
||||
CATEGORY_BLOCKER,
|
||||
CATEGORY_LEASE,
|
||||
CATEGORY_SYSTEM,
|
||||
CATEGORY_VALIDATION,
|
||||
CATEGORY_WORKFLOW,
|
||||
NotificationItem,
|
||||
NotificationSnapshot,
|
||||
classify_attention_event,
|
||||
load_notifications_snapshot,
|
||||
snapshot_to_dict,
|
||||
)
|
||||
from webui.notification_views import render_notifications_page
|
||||
from webui.project_registry import load_registry
|
||||
from webui.queue_loader import QueueItem, QueueSnapshot
|
||||
from webui.lease_loader import CollisionWarning, LeaseSnapshot
|
||||
from webui.system_health import DependencyProbe, SystemHealthSnapshot, VersionInfo, StaleRuntime
|
||||
|
||||
|
||||
def test_classify_attention_event_rules():
|
||||
# 1. Critical escalation boundaries -> human-required
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_AUTH, "Auth error", "Unauthorized access attempt", is_auth_failure=True
|
||||
)
|
||||
assert att_cls == ATTENTION_HUMAN_REQUIRED
|
||||
assert req_human is True
|
||||
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_SYSTEM, "Hard stop", "Hard stop triggered", is_hard_stop=True
|
||||
)
|
||||
assert att_cls == ATTENTION_HUMAN_REQUIRED
|
||||
assert req_human is True
|
||||
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_VALIDATION, "Validation Error", "Report validation failed", is_validation_failure=True
|
||||
)
|
||||
assert att_cls == ATTENTION_HUMAN_REQUIRED
|
||||
assert req_human is True
|
||||
|
||||
# 2. Operational issues -> operator
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_BLOCKER, "PR Blocked", "Merge conflict detected", is_blocker=True
|
||||
)
|
||||
assert att_cls == ATTENTION_OPERATOR
|
||||
assert req_human is False
|
||||
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_LEASE, "Lease Expired", "Session lease expired", is_stale=True
|
||||
)
|
||||
assert att_cls == ATTENTION_OPERATOR
|
||||
assert req_human is False
|
||||
|
||||
# 3. Routine workflow transitions -> routine
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_WORKFLOW, "PR Active", "PR in review"
|
||||
)
|
||||
assert att_cls == ATTENTION_ROUTINE
|
||||
assert req_human is False
|
||||
|
||||
|
||||
def test_notification_snapshot_aggregation():
|
||||
reg = load_registry()
|
||||
proj_id = reg.projects[0].id if reg.projects else "gitea-tools"
|
||||
|
||||
mock_queue = QueueSnapshot(
|
||||
project_id=proj_id,
|
||||
repo_label="org/repo",
|
||||
prs=(
|
||||
QueueItem(
|
||||
number=101,
|
||||
title="Blocked PR",
|
||||
badges=("blocked",),
|
||||
extra={},
|
||||
),
|
||||
QueueItem(
|
||||
number=102,
|
||||
title="Normal PR",
|
||||
badges=("in-review",),
|
||||
extra={},
|
||||
),
|
||||
),
|
||||
issues=(),
|
||||
pr_pagination=None,
|
||||
issue_pagination=None,
|
||||
)
|
||||
|
||||
mock_leases = LeaseSnapshot(
|
||||
project_id=proj_id,
|
||||
repo_label="org/repo",
|
||||
issue_lock=None,
|
||||
claim_inventory={},
|
||||
reviewer_leases=(
|
||||
{
|
||||
"pr_number": 101,
|
||||
"status": "expired",
|
||||
"is_expired": True,
|
||||
},
|
||||
),
|
||||
duplicate_prs=(
|
||||
CollisionWarning(
|
||||
kind="duplicate_pr",
|
||||
message="Multiple open PRs for issue #101",
|
||||
issue_number=101,
|
||||
pr_numbers=(101, 103),
|
||||
),
|
||||
),
|
||||
duplicate_branches=(),
|
||||
collision_history=(),
|
||||
fetch_error=None,
|
||||
)
|
||||
|
||||
mock_version = VersionInfo(
|
||||
git_sha="abc1234",
|
||||
git_describe="v1.0.0",
|
||||
control_plane_schema_version=1,
|
||||
python_version="3.11",
|
||||
known=True,
|
||||
)
|
||||
|
||||
mock_stale = StaleRuntime(
|
||||
daemon_head="abc1234",
|
||||
checkout_head="abc1234",
|
||||
remote_head="abc1234",
|
||||
stale=False,
|
||||
determinable=True,
|
||||
mutation_safe=True,
|
||||
reasons=(),
|
||||
)
|
||||
|
||||
mock_health = SystemHealthSnapshot(
|
||||
status="degraded",
|
||||
ready=False,
|
||||
readiness_complete=True,
|
||||
readiness_reasons=("Auth failure",),
|
||||
service="webui",
|
||||
mode="test",
|
||||
version=mock_version,
|
||||
started_at="2026-07-25T00:00:00Z",
|
||||
uptime_seconds=100.0,
|
||||
timestamp="2026-07-25T00:00:00Z",
|
||||
deep_probes_requested=True,
|
||||
dependencies=(
|
||||
DependencyProbe(
|
||||
name="auth_service",
|
||||
kind="auth",
|
||||
status="unauthorized",
|
||||
detail="Token expired",
|
||||
required=True,
|
||||
),
|
||||
),
|
||||
mcp_namespaces=(),
|
||||
stale_runtime=mock_stale,
|
||||
probe_errors=(),
|
||||
)
|
||||
|
||||
snapshot = load_notifications_snapshot(
|
||||
proj_id,
|
||||
load_queue=lambda _id: mock_queue,
|
||||
load_leases=lambda **_kwargs: mock_leases,
|
||||
load_health=lambda **_kwargs: mock_health,
|
||||
)
|
||||
|
||||
assert snapshot.project_id == proj_id
|
||||
assert snapshot.total_count == 5
|
||||
assert snapshot.human_required_count >= 1 # auth probe failure
|
||||
assert snapshot.operator_count >= 3 # blocked PR + expired lease + duplicate PR collision
|
||||
assert snapshot.routine_count >= 1 # normal PR
|
||||
|
||||
# Inbox items should include operator and human-required items only
|
||||
inbox_classes = {item.attention_class for item in snapshot.inbox_items}
|
||||
assert ATTENTION_ROUTINE not in inbox_classes
|
||||
assert ATTENTION_OPERATOR in inbox_classes
|
||||
assert ATTENTION_HUMAN_REQUIRED in inbox_classes
|
||||
|
||||
|
||||
def test_snapshot_to_dict_and_redaction():
|
||||
item = NotificationItem(
|
||||
id="notif-1",
|
||||
attention_class=ATTENTION_HUMAN_REQUIRED,
|
||||
category=CATEGORY_AUTH,
|
||||
title="Auth Error",
|
||||
summary="Failed auth header: Bearer secret_token_12345",
|
||||
work_kind="system",
|
||||
work_number=None,
|
||||
project_id="test-proj",
|
||||
repo_label="org/repo",
|
||||
created_at="2026-07-25T16:00:00Z",
|
||||
requires_human=True,
|
||||
)
|
||||
snap = NotificationSnapshot(
|
||||
project_id="test-proj",
|
||||
repo_label="org/repo",
|
||||
items=(item,),
|
||||
human_required_count=1,
|
||||
operator_count=0,
|
||||
routine_count=0,
|
||||
total_count=1,
|
||||
)
|
||||
|
||||
data = snapshot_to_dict(snap)
|
||||
assert data["project_id"] == "test-proj"
|
||||
assert data["human_required_count"] == 1
|
||||
assert len(data["inbox_items"]) == 1
|
||||
|
||||
# Redaction test
|
||||
summary = data["inbox_items"][0]["summary"]
|
||||
assert "secret_token_12345" not in summary
|
||||
assert "<redacted>" in summary or "Bearer" in summary
|
||||
|
||||
|
||||
def test_notifications_html_views():
|
||||
item = NotificationItem(
|
||||
id="notif-1",
|
||||
attention_class=ATTENTION_HUMAN_REQUIRED,
|
||||
category=CATEGORY_AUTH,
|
||||
title="Critical Auth Failure",
|
||||
summary="Auth failure details",
|
||||
work_kind="issue",
|
||||
work_number=42,
|
||||
project_id="test-proj",
|
||||
repo_label="org/repo",
|
||||
created_at="2026-07-25T16:00:00Z",
|
||||
requires_human=True,
|
||||
)
|
||||
snap = NotificationSnapshot(
|
||||
project_id="test-proj",
|
||||
repo_label="org/repo",
|
||||
items=(item,),
|
||||
human_required_count=1,
|
||||
operator_count=0,
|
||||
routine_count=0,
|
||||
total_count=1,
|
||||
)
|
||||
|
||||
html = render_notifications_page(snap, filter_class="inbox")
|
||||
assert "Notifications & Attention Inbox" in html or "Notifications & Attention Inbox" in html
|
||||
assert "Critical Auth Failure" in html
|
||||
assert "HUMAN REQUIRED" in html
|
||||
assert "Human Required" in html
|
||||
|
||||
|
||||
def test_notifications_app_routes():
|
||||
app = create_app()
|
||||
client = TestClient(app)
|
||||
|
||||
# 1. HTML Route
|
||||
res = client.get("/notifications")
|
||||
assert res.status_code == 200
|
||||
assert "Notifications" in res.text
|
||||
assert "Attention Inbox" in res.text
|
||||
|
||||
# 2. API Route /api/v1/notifications
|
||||
res_api = client.get("/api/v1/notifications")
|
||||
assert res_api.status_code == 200
|
||||
json_data = res_api.json()
|
||||
assert "human_required_count" in json_data
|
||||
assert "operator_count" in json_data
|
||||
assert "routine_count" in json_data
|
||||
assert "inbox_items" in json_data
|
||||
|
||||
# 3. Compatibility Alias /api/notifications
|
||||
res_alias = client.get("/api/notifications")
|
||||
assert res_alias.status_code == 200
|
||||
assert res_alias.json()["project_id"] == json_data["project_id"]
|
||||
|
||||
|
||||
def test_classify_ignores_human_authored_title_and_summary_keywords():
|
||||
"""B1: keywords in human-authored titles must not escalate routine work (#905)."""
|
||||
# Routine transition whose title/summary mention critical-boundary words
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_WORKFLOW,
|
||||
"record irrecoverable decision lock provenance",
|
||||
"PR #999 'record irrecoverable decision lock provenance' is in routine state in-review.",
|
||||
)
|
||||
assert att_cls == ATTENTION_ROUTINE
|
||||
assert req_human is False
|
||||
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_WORKFLOW,
|
||||
"fix unauthorized token path",
|
||||
"Issue #1 'fix unauthorized token path' state: claimed. hard stop docs only.",
|
||||
)
|
||||
assert att_cls == ATTENTION_ROUTINE
|
||||
assert req_human is False
|
||||
|
||||
# Structured flags still escalate (machine-driven)
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_SYSTEM,
|
||||
"anything",
|
||||
"anything with hard stop in text",
|
||||
is_hard_stop=True,
|
||||
)
|
||||
assert att_cls == ATTENTION_HUMAN_REQUIRED
|
||||
assert req_human is True
|
||||
|
||||
|
||||
def test_notification_ids_are_unique_across_probe_errors_and_collisions():
|
||||
"""B2: published notification ids must be unique within a snapshot (#905)."""
|
||||
reg = load_registry()
|
||||
proj_id = reg.projects[0].id if reg.projects else "gitea-tools"
|
||||
|
||||
mock_queue = QueueSnapshot(
|
||||
project_id=proj_id,
|
||||
repo_label="org/repo",
|
||||
prs=(),
|
||||
issues=(),
|
||||
pr_pagination=None,
|
||||
issue_pagination=None,
|
||||
)
|
||||
mock_leases = LeaseSnapshot(
|
||||
project_id=proj_id,
|
||||
repo_label="org/repo",
|
||||
issue_lock=None,
|
||||
claim_inventory={},
|
||||
reviewer_leases=(),
|
||||
duplicate_prs=(
|
||||
CollisionWarning(
|
||||
kind="duplicate_pr",
|
||||
message="Multiple open PRs for issue #10",
|
||||
issue_number=10,
|
||||
pr_numbers=(10, 11),
|
||||
),
|
||||
CollisionWarning(
|
||||
kind="duplicate_branch",
|
||||
message="Another collision without issue",
|
||||
issue_number=None,
|
||||
pr_numbers=(12, 13),
|
||||
),
|
||||
CollisionWarning(
|
||||
kind="duplicate_pr",
|
||||
message="Second issue collision",
|
||||
issue_number=10,
|
||||
pr_numbers=(14, 15),
|
||||
),
|
||||
),
|
||||
duplicate_branches=(),
|
||||
collision_history=(),
|
||||
fetch_error=None,
|
||||
)
|
||||
mock_version = VersionInfo(
|
||||
git_sha="abc1234",
|
||||
git_describe="v1.0.0",
|
||||
control_plane_schema_version=1,
|
||||
python_version="3.11",
|
||||
known=True,
|
||||
)
|
||||
mock_stale = StaleRuntime(
|
||||
daemon_head="abc1234",
|
||||
checkout_head="abc1234",
|
||||
remote_head="abc1234",
|
||||
stale=False,
|
||||
determinable=True,
|
||||
mutation_safe=True,
|
||||
reasons=(),
|
||||
)
|
||||
mock_health = SystemHealthSnapshot(
|
||||
status="degraded",
|
||||
ready=False,
|
||||
readiness_complete=True,
|
||||
readiness_reasons=(),
|
||||
service="webui",
|
||||
mode="test",
|
||||
version=mock_version,
|
||||
started_at="2026-07-25T00:00:00Z",
|
||||
uptime_seconds=100.0,
|
||||
timestamp="2026-07-25T00:00:00Z",
|
||||
deep_probes_requested=True,
|
||||
dependencies=(),
|
||||
mcp_namespaces=(),
|
||||
stale_runtime=mock_stale,
|
||||
probe_errors=("error alpha", "error beta"),
|
||||
)
|
||||
|
||||
snapshot = load_notifications_snapshot(
|
||||
proj_id,
|
||||
load_queue=lambda _id: mock_queue,
|
||||
load_leases=lambda **_kwargs: mock_leases,
|
||||
load_health=lambda **_kwargs: mock_health,
|
||||
)
|
||||
ids = [item.id for item in snapshot.items]
|
||||
assert len(ids) == len(set(ids)), f"duplicate notification ids: {ids}"
|
||||
assert any(i.startswith(f"notif-sys-err-{proj_id}-") for i in ids)
|
||||
assert any(i.startswith("notif-collision-") for i in ids)
|
||||
|
||||
|
||||
def test_probe_errors_do_not_set_fetch_error():
|
||||
"""B3: probe_errors must not be reported as fetch_error (#905)."""
|
||||
reg = load_registry()
|
||||
proj_id = reg.projects[0].id if reg.projects else "gitea-tools"
|
||||
|
||||
mock_queue = QueueSnapshot(
|
||||
project_id=proj_id,
|
||||
repo_label="org/repo",
|
||||
prs=(),
|
||||
issues=(),
|
||||
pr_pagination=None,
|
||||
issue_pagination=None,
|
||||
fetch_error=None,
|
||||
)
|
||||
mock_leases = LeaseSnapshot(
|
||||
project_id=proj_id,
|
||||
repo_label="org/repo",
|
||||
issue_lock=None,
|
||||
claim_inventory={},
|
||||
reviewer_leases=(),
|
||||
duplicate_prs=(),
|
||||
duplicate_branches=(),
|
||||
collision_history=(),
|
||||
fetch_error=None,
|
||||
)
|
||||
mock_version = VersionInfo(
|
||||
git_sha="abc1234",
|
||||
git_describe="v1.0.0",
|
||||
control_plane_schema_version=1,
|
||||
python_version="3.11",
|
||||
known=True,
|
||||
)
|
||||
mock_stale = StaleRuntime(
|
||||
daemon_head="abc1234",
|
||||
checkout_head="abc1234",
|
||||
remote_head="abc1234",
|
||||
stale=False,
|
||||
determinable=True,
|
||||
mutation_safe=True,
|
||||
reasons=(),
|
||||
)
|
||||
mock_health = SystemHealthSnapshot(
|
||||
status="degraded",
|
||||
ready=False,
|
||||
readiness_complete=True,
|
||||
readiness_reasons=(),
|
||||
service="webui",
|
||||
mode="test",
|
||||
version=mock_version,
|
||||
started_at="2026-07-25T00:00:00Z",
|
||||
uptime_seconds=100.0,
|
||||
timestamp="2026-07-25T00:00:00Z",
|
||||
deep_probes_requested=True,
|
||||
dependencies=(),
|
||||
mcp_namespaces=(),
|
||||
stale_runtime=mock_stale,
|
||||
probe_errors=("probe blew up",),
|
||||
)
|
||||
|
||||
snapshot = load_notifications_snapshot(
|
||||
proj_id,
|
||||
load_queue=lambda _id: mock_queue,
|
||||
load_leases=lambda **_kwargs: mock_leases,
|
||||
load_health=lambda **_kwargs: mock_health,
|
||||
)
|
||||
assert snapshot.fetch_error is None
|
||||
# probe errors still appear as items
|
||||
assert any("probe blew up" in item.summary for item in snapshot.items)
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,452 @@
|
||||
"""Read-only restart console: views, gates, and honesty rules (#667).
|
||||
|
||||
The console consumes the #655 substrate. These tests hold it to the three
|
||||
properties that make a status surface trustworthy:
|
||||
|
||||
* an unreadable source is reported unavailable, never rendered as green;
|
||||
* authorization is probed the way execution would probe it, so an allow is
|
||||
never shown for something that could not run;
|
||||
* the surface performs no mutation, including no write to the control-plane DB.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sqlite3
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from starlette.testclient import TestClient # noqa: E402
|
||||
|
||||
import restart_coordinator # noqa: E402
|
||||
from webui import console_authz, restart_console, restart_views # noqa: E402
|
||||
from webui.app import create_app # noqa: E402
|
||||
|
||||
NOW = datetime(2026, 7, 25, 21, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
def _principal(role: str) -> console_authz.Principal:
|
||||
return console_authz.Principal(
|
||||
subject="[email protected]",
|
||||
role=role,
|
||||
identity_source=console_authz.IDENTITY_LOCAL_DEV,
|
||||
authenticated=True,
|
||||
)
|
||||
|
||||
|
||||
def _inventory(*, complete: bool = True, sessions=(), leases=()):
|
||||
def _read(**_kwargs):
|
||||
return {
|
||||
"sessions": list(sessions),
|
||||
"leases": list(leases),
|
||||
"terminal_lock": None,
|
||||
"prior_recovery_attempts": [],
|
||||
"inventory_complete": complete,
|
||||
"incomplete_reasons": (
|
||||
[] if complete else ["fixture: inventory withheld"]
|
||||
),
|
||||
}
|
||||
|
||||
return _read
|
||||
|
||||
|
||||
def _live_session(session_id: str = "prgs-author-1234-abcd") -> dict:
|
||||
return {
|
||||
"session_id": session_id,
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": os.getpid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": (NOW - timedelta(seconds=30)).isoformat(),
|
||||
}
|
||||
|
||||
|
||||
def drain_proof_fixture() -> dict:
|
||||
"""A structurally complete but unsigned drain proof."""
|
||||
return {
|
||||
"version": "drain-proof/v1",
|
||||
"proof_id": "deadbeef" * 8,
|
||||
"clean": True,
|
||||
"issued_at": (NOW - timedelta(minutes=1)).isoformat(),
|
||||
"expires_at": (NOW + timedelta(minutes=5)).isoformat(),
|
||||
"requesting_session_id": "s-live",
|
||||
"impact_fingerprint": "f" * 64,
|
||||
"checks": [],
|
||||
"failed_checks": [],
|
||||
}
|
||||
|
||||
|
||||
class RestartClassMatrixTest(unittest.TestCase):
|
||||
def test_every_policy_class_is_rendered(self) -> None:
|
||||
views = restart_console.build_restart_class_views("operator")
|
||||
self.assertEqual(len(views), len(restart_coordinator.RESTART_CLASS_POLICIES))
|
||||
|
||||
def test_viewer_capability_is_role_scoped_not_generic(self) -> None:
|
||||
"""A worker role must not be shown as able to request a full restart."""
|
||||
author = {
|
||||
v.restart_class: v
|
||||
for v in restart_console.build_restart_class_views("author")
|
||||
}
|
||||
operator = {
|
||||
v.restart_class: v
|
||||
for v in restart_console.build_restart_class_views("operator")
|
||||
}
|
||||
full = restart_coordinator.RestartClass.FULL_MCP_RESTART.value
|
||||
|
||||
self.assertFalse(author[full].viewer_may_request)
|
||||
self.assertFalse(author[full].viewer_may_execute)
|
||||
self.assertTrue(operator[full].viewer_may_request)
|
||||
self.assertTrue(operator[full].viewer_may_execute)
|
||||
|
||||
def test_unknown_role_may_do_nothing(self) -> None:
|
||||
views = restart_console.build_restart_class_views("not-a-role")
|
||||
self.assertTrue(all(not v.viewer_may_request for v in views))
|
||||
self.assertTrue(all(not v.viewer_may_execute for v in views))
|
||||
|
||||
|
||||
class AuthorizationProbeTest(unittest.TestCase):
|
||||
def test_probe_asks_for_execution_so_phase_gate_is_reported(self) -> None:
|
||||
"""An admin clears the role bar and still cannot execute in Phase 1.
|
||||
|
||||
This is the case that distinguishes the two probes. Asked without
|
||||
``for_execution`` an admin is *allowed* for ``system.restart_namespace``,
|
||||
which on a control surface reads as a live button. Asked the way
|
||||
execution asks, the same principal is refused ``phase_not_active``. The
|
||||
console must report the second answer.
|
||||
"""
|
||||
by_id = {
|
||||
a.action_id: a
|
||||
for a in restart_console.build_action_authorizations(
|
||||
_principal(console_authz.ADMIN)
|
||||
)
|
||||
}
|
||||
restart = by_id["system.restart_namespace"]
|
||||
|
||||
self.assertFalse(restart.execution_enabled)
|
||||
self.assertEqual(restart.reason_code, console_authz.DENY_PHASE_NOT_ACTIVE)
|
||||
|
||||
permissive = console_authz.authorize(
|
||||
"system.restart_namespace", _principal(console_authz.ADMIN)
|
||||
)
|
||||
self.assertTrue(
|
||||
permissive.allowed,
|
||||
"guard precondition: without for_execution an admin is allowed, "
|
||||
"which is exactly why the console must not probe that way",
|
||||
)
|
||||
|
||||
def test_operator_is_refused_the_admin_only_restart_action(self) -> None:
|
||||
"""Role refusal precedes the phase gate and is reported as such."""
|
||||
by_id = {
|
||||
a.action_id: a
|
||||
for a in restart_console.build_action_authorizations(
|
||||
_principal(console_authz.OPERATOR)
|
||||
)
|
||||
}
|
||||
self.assertEqual(
|
||||
by_id["system.restart_namespace"].reason_code,
|
||||
console_authz.DENY_INSUFFICIENT_ROLE,
|
||||
)
|
||||
|
||||
def test_anonymous_is_denied_unauthenticated(self) -> None:
|
||||
by_id = {
|
||||
a.action_id: a for a in restart_console.build_action_authorizations(None)
|
||||
}
|
||||
self.assertEqual(
|
||||
by_id["system.restart_namespace"].reason_code,
|
||||
console_authz.DENY_UNAUTHENTICATED,
|
||||
)
|
||||
|
||||
def test_no_authorization_ever_reports_execution_enabled(self) -> None:
|
||||
for role in (
|
||||
console_authz.VIEWER,
|
||||
console_authz.OPERATOR,
|
||||
console_authz.CONTROLLER,
|
||||
console_authz.ADMIN,
|
||||
):
|
||||
for auth in restart_console.build_action_authorizations(_principal(role)):
|
||||
self.assertFalse(
|
||||
auth.execution_enabled,
|
||||
f"{role} reported execution_enabled for {auth.action_id}",
|
||||
)
|
||||
|
||||
|
||||
class ImpactPreviewTest(unittest.TestCase):
|
||||
def test_impact_renders_from_coordinator_dto(self) -> None:
|
||||
impact, source = restart_console.load_impact_report(
|
||||
principal=_principal(console_authz.OPERATOR),
|
||||
read_inventory=_inventory(sessions=[_live_session()]),
|
||||
now=NOW,
|
||||
)
|
||||
self.assertTrue(source.available)
|
||||
self.assertIsNotNone(impact)
|
||||
self.assertEqual(
|
||||
impact["restart_class"],
|
||||
restart_coordinator.RestartClass.FULL_MCP_RESTART.value,
|
||||
)
|
||||
self.assertIn("verdict", impact)
|
||||
self.assertFalse(impact["restart_performed"])
|
||||
self.assertTrue(impact["dry_run"])
|
||||
|
||||
def test_incomplete_inventory_is_surfaced_and_denies(self) -> None:
|
||||
impact, source = restart_console.load_impact_report(
|
||||
principal=_principal(console_authz.OPERATOR),
|
||||
read_inventory=_inventory(complete=False),
|
||||
now=NOW,
|
||||
)
|
||||
self.assertFalse(impact["inventory_complete"])
|
||||
self.assertFalse(impact["allow_restart"])
|
||||
self.assertTrue(source.detail, "incomplete inventory must explain itself")
|
||||
|
||||
def test_inventory_reader_failure_is_unavailable_not_empty(self) -> None:
|
||||
"""A reader that raises must not be rendered as 'no sessions affected'."""
|
||||
|
||||
def _boom(**_kwargs):
|
||||
raise RuntimeError("control-plane unreachable")
|
||||
|
||||
impact, source = restart_console.load_impact_report(
|
||||
principal=_principal(console_authz.OPERATOR),
|
||||
read_inventory=_boom,
|
||||
now=NOW,
|
||||
)
|
||||
self.assertIsNone(impact)
|
||||
self.assertFalse(source.available)
|
||||
self.assertIn("control-plane unreachable", source.detail)
|
||||
|
||||
|
||||
class ControlPlaneReadTest(unittest.TestCase):
|
||||
def test_missing_database_is_incomplete_not_empty(self) -> None:
|
||||
inventory = restart_console.read_control_plane_inventory(
|
||||
db_path="/nonexistent/control-plane.sqlite3"
|
||||
)
|
||||
self.assertFalse(inventory["inventory_complete"])
|
||||
self.assertEqual(inventory["sessions"], [])
|
||||
self.assertTrue(inventory["incomplete_reasons"])
|
||||
|
||||
def test_reader_never_creates_the_database(self) -> None:
|
||||
"""Reading status must not bring a control-plane DB into existence.
|
||||
|
||||
The path deliberately sits in a directory that already exists: a
|
||||
read-write ``sqlite3.connect`` would happily create the file there, so
|
||||
this fails if the reader ever stops opening the database ``mode=ro``.
|
||||
A nested-missing-directory path would pass for the wrong reason,
|
||||
because sqlite cannot create the parent directory either way.
|
||||
"""
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
path = os.path.join(tmp, "control_plane.sqlite3")
|
||||
self.assertTrue(os.path.isdir(os.path.dirname(path)))
|
||||
|
||||
inventory = restart_console.read_control_plane_inventory(db_path=path)
|
||||
|
||||
self.assertFalse(
|
||||
os.path.exists(path),
|
||||
"reading restart status created a control-plane database",
|
||||
)
|
||||
self.assertFalse(inventory["inventory_complete"])
|
||||
|
||||
def test_reads_active_sessions_from_a_real_database(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
path = os.path.join(tmp, "cp.sqlite3")
|
||||
conn = sqlite3.connect(path)
|
||||
conn.execute(
|
||||
"CREATE TABLE sessions (session_id TEXT, role TEXT, profile TEXT,"
|
||||
" pid INTEGER, status TEXT, last_heartbeat_at TEXT)"
|
||||
)
|
||||
conn.execute(
|
||||
"CREATE TABLE work_items (work_item_id INTEGER, kind TEXT,"
|
||||
" number INTEGER)"
|
||||
)
|
||||
conn.execute(
|
||||
"CREATE TABLE leases (lease_id TEXT, session_id TEXT, role TEXT,"
|
||||
" phase TEXT, status TEXT, worktree_path TEXT,"
|
||||
" work_item_id INTEGER, expires_at TEXT)"
|
||||
)
|
||||
conn.execute(
|
||||
"INSERT INTO sessions VALUES (?,?,?,?,?,?)",
|
||||
("s-live", "author", "prgs-author", 4242, "active", NOW.isoformat()),
|
||||
)
|
||||
conn.execute(
|
||||
"INSERT INTO sessions VALUES (?,?,?,?,?,?)",
|
||||
("s-done", "author", "prgs-author", 11, "closed", NOW.isoformat()),
|
||||
)
|
||||
conn.execute("INSERT INTO work_items VALUES (1, 'issue', 667)")
|
||||
conn.execute(
|
||||
"INSERT INTO leases VALUES (?,?,?,?,?,?,?,?)",
|
||||
(
|
||||
"l-1",
|
||||
"s-live",
|
||||
"author",
|
||||
"allocated",
|
||||
"active",
|
||||
None,
|
||||
1,
|
||||
NOW.isoformat(),
|
||||
),
|
||||
)
|
||||
conn.commit()
|
||||
conn.close()
|
||||
|
||||
inventory = restart_console.read_control_plane_inventory(db_path=path)
|
||||
|
||||
self.assertTrue(inventory["inventory_complete"])
|
||||
self.assertEqual([s["session_id"] for s in inventory["sessions"]], ["s-live"])
|
||||
self.assertEqual(inventory["leases"][0]["work_number"], 667)
|
||||
|
||||
|
||||
class DrainAndReconcileTest(unittest.TestCase):
|
||||
def test_absent_drain_proof_is_not_a_pass(self) -> None:
|
||||
drain, source = restart_console.load_drain_status(proof=None, now=NOW)
|
||||
self.assertIsNone(drain)
|
||||
self.assertFalse(source.available)
|
||||
self.assertIn("denies", source.detail)
|
||||
|
||||
def test_tampered_drain_proof_is_reported_invalid(self) -> None:
|
||||
proof = drain_proof_fixture()
|
||||
proof["clean"] = True
|
||||
proof["proof_id"] = "0" * 64
|
||||
drain, source = restart_console.load_drain_status(proof=proof, now=NOW)
|
||||
self.assertTrue(source.available)
|
||||
self.assertFalse(drain["valid"])
|
||||
|
||||
def test_absent_reconcile_proof_is_unavailable(self) -> None:
|
||||
reconcile, source = restart_console.load_reconcile_status(load_proof=None)
|
||||
self.assertIsNone(reconcile)
|
||||
self.assertFalse(source.available)
|
||||
|
||||
def test_reconcile_proof_is_rendered_when_supplied(self) -> None:
|
||||
payload = {
|
||||
"overall_status": "degraded",
|
||||
"mode": "log_only",
|
||||
"resolved_count": 3,
|
||||
"unresolved_count": 2,
|
||||
"items": [
|
||||
{
|
||||
"dimension": "leases",
|
||||
"status": "unresolved",
|
||||
"summary": "2 orphaned leases",
|
||||
"follow_up_required": True,
|
||||
}
|
||||
],
|
||||
}
|
||||
reconcile, source = restart_console.load_reconcile_status(
|
||||
load_proof=lambda: payload
|
||||
)
|
||||
self.assertTrue(source.available)
|
||||
self.assertEqual(reconcile["unresolved_count"], 2)
|
||||
|
||||
|
||||
class RenderingTest(unittest.TestCase):
|
||||
def _snapshot(self, **kwargs):
|
||||
params = {
|
||||
"principal": _principal(console_authz.OPERATOR),
|
||||
"read_inventory": _inventory(sessions=[_live_session()]),
|
||||
"now": NOW,
|
||||
}
|
||||
params.update(kwargs)
|
||||
return restart_console.load_restart_console_snapshot(**params)
|
||||
|
||||
def test_page_renders_every_section(self) -> None:
|
||||
html = restart_views.render_restart_console_page(self._snapshot())
|
||||
for heading in (
|
||||
"Impact preview",
|
||||
"Drain proof",
|
||||
"Post-restart reconcile",
|
||||
"Restart classes",
|
||||
"Approval controls",
|
||||
"Break-glass",
|
||||
):
|
||||
self.assertIn(heading, html)
|
||||
|
||||
def test_hostile_session_id_is_escaped(self) -> None:
|
||||
hostile = "<script>alert('x')</script>"
|
||||
html = restart_views.render_restart_console_page(
|
||||
self._snapshot(read_inventory=_inventory(sessions=[_live_session(hostile)]))
|
||||
)
|
||||
self.assertNotIn("<script>alert", html)
|
||||
self.assertIn("<script>", html)
|
||||
|
||||
def test_unavailable_impact_says_unsafe_rather_than_clean(self) -> None:
|
||||
def _boom(**_kwargs):
|
||||
raise RuntimeError("nope")
|
||||
|
||||
snapshot = self._snapshot(read_inventory=_boom)
|
||||
html = restart_views.render_restart_console_page(snapshot)
|
||||
self.assertIn("blast radius of a restart is unknown", html)
|
||||
self.assertIn("unavailable", html)
|
||||
|
||||
def test_break_glass_is_hidden_from_unprivileged_viewers(self) -> None:
|
||||
viewer_html = restart_views.render_restart_console_page(
|
||||
self._snapshot(principal=_principal(console_authz.VIEWER))
|
||||
)
|
||||
self.assertIn("visible to operator-class", viewer_html)
|
||||
self.assertNotIn(
|
||||
f"#{restart_console.BREAK_GLASS_ISSUE}", viewer_html
|
||||
)
|
||||
|
||||
def test_break_glass_shown_to_operator_is_marked_unavailable(self) -> None:
|
||||
html = restart_views.render_restart_console_page(self._snapshot())
|
||||
self.assertIn("unavailable", html)
|
||||
self.assertIn(f"#{restart_console.BREAK_GLASS_ISSUE}", html)
|
||||
|
||||
def test_snapshot_always_declares_itself_read_only(self) -> None:
|
||||
self.assertTrue(self._snapshot().read_only)
|
||||
|
||||
|
||||
class RestartConsoleRouteTest(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self.client = TestClient(create_app())
|
||||
|
||||
def test_page_route_renders(self) -> None:
|
||||
res = self.client.get("/runtime/restart")
|
||||
self.assertEqual(res.status_code, 200)
|
||||
self.assertIn("Restart status and impact", res.text)
|
||||
|
||||
def test_api_route_exports_snapshot(self) -> None:
|
||||
res = self.client.get("/api/v1/system/restart/status")
|
||||
self.assertEqual(res.status_code, 200)
|
||||
payload = res.json()
|
||||
self.assertTrue(payload["read_only"])
|
||||
self.assertEqual(payload["links"]["issue"], 667)
|
||||
self.assertEqual(
|
||||
len(payload["restart_classes"]),
|
||||
len(restart_coordinator.RESTART_CLASS_POLICIES),
|
||||
)
|
||||
|
||||
def test_restart_class_is_selectable(self) -> None:
|
||||
res = self.client.get(
|
||||
"/api/v1/system/restart/status?restart_class=client_reconnect"
|
||||
)
|
||||
self.assertEqual(res.status_code, 200)
|
||||
self.assertEqual(res.json()["impact"]["restart_class"], "client_reconnect")
|
||||
|
||||
def test_unknown_restart_class_fails_closed(self) -> None:
|
||||
res = self.client.get(
|
||||
"/api/v1/system/restart/status?restart_class=obliterate-everything"
|
||||
)
|
||||
self.assertEqual(res.status_code, 200)
|
||||
impact = res.json()["impact"]
|
||||
self.assertFalse(impact["allow_restart"])
|
||||
|
||||
def test_anonymous_api_reader_gets_no_execution_grant(self) -> None:
|
||||
payload = self.client.get("/api/v1/system/restart/status").json()
|
||||
self.assertFalse(payload["break_glass"]["available"])
|
||||
for auth in payload["authorizations"]:
|
||||
self.assertFalse(auth["execution_enabled"])
|
||||
|
||||
def test_route_is_registered_in_nav(self) -> None:
|
||||
from webui.nav import nav_hrefs
|
||||
|
||||
self.assertIn("/runtime/restart", nav_hrefs())
|
||||
|
||||
def test_no_write_method_is_exposed(self) -> None:
|
||||
"""The surface is read-only: nothing accepts a POST."""
|
||||
for path in ("/runtime/restart", "/api/v1/system/restart/status"):
|
||||
self.assertEqual(self.client.post(path).status_code, 405, path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -444,6 +444,12 @@ class TestAuditEmission(unittest.TestCase):
|
||||
)
|
||||
self.assertEqual(record["target"]["namespace"], NAMESPACE)
|
||||
self.assertEqual(record["target"]["mode"], "restart")
|
||||
self.assertEqual(
|
||||
record["target"]["restart_class"], "role_runtime_restart"
|
||||
)
|
||||
self.assertEqual(
|
||||
record["metadata"]["restart_class"], "role_runtime_restart"
|
||||
)
|
||||
self.assertEqual(record["result"], console_audit.RESULT_ALLOWED)
|
||||
self.assertEqual(record["actor"]["subject"], "[email protected]")
|
||||
self.assertFalse(record["metadata"]["process_kill_executed"])
|
||||
|
||||
@@ -0,0 +1,739 @@
|
||||
"""Tests for the Runtime and session view (Phase 1, #641).
|
||||
|
||||
Covers clean and stale session rendering, contamination marker surfacing,
|
||||
worktree binding display, sanctioned recovery links (no pkill), nav/live
|
||||
status, and the JSON API export.
|
||||
|
||||
Also pins the two invariants a reviewer found violated at head a81db754:
|
||||
degraded ownership sections must render as *unknown* rather than as an
|
||||
affirmative "none"/"unbound", and contamination payload text must be redacted
|
||||
at the display boundary rather than trusted from the write-time denylist.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from tests.webui_testclient import TestClient
|
||||
|
||||
from webui.app import create_app
|
||||
from webui.inventory import (
|
||||
AUTHORITY_CONTROL_PLANE_DB,
|
||||
AUTHORITY_FILESYSTEM,
|
||||
InventorySection,
|
||||
InventorySnapshot,
|
||||
STATUS_DEGRADED,
|
||||
STATUS_OK,
|
||||
STATUS_UNAVAILABLE,
|
||||
)
|
||||
from webui.nav import NAV_GROUPS, STUB_PAGES, iter_nav_items
|
||||
from webui.runtime_health import FileHash, RuntimeSnapshot
|
||||
from webui.session_loader import (
|
||||
ContaminationMarker,
|
||||
SessionRow,
|
||||
SessionViewSnapshot,
|
||||
_build_session_rows,
|
||||
_inspect_contamination,
|
||||
load_session_view_snapshot,
|
||||
snapshot_to_dict,
|
||||
)
|
||||
from webui.session_views import render_sessions_page
|
||||
|
||||
|
||||
def _runtime(
|
||||
*,
|
||||
stale: str | None = None,
|
||||
profile: str = "prgs-author",
|
||||
role: str = "author",
|
||||
) -> RuntimeSnapshot:
|
||||
return RuntimeSnapshot(
|
||||
project_id="gitea-tools",
|
||||
repo_root="/tmp/repo",
|
||||
remote="prgs",
|
||||
host="gitea.prgs.cc",
|
||||
profile_name=profile,
|
||||
role_kind=role,
|
||||
config_model="v2-contexts",
|
||||
profile_mode="dynamic-profile",
|
||||
profile_source="config file profile",
|
||||
authenticated_username="jcwalker3",
|
||||
identity_error=None,
|
||||
repo_sha="a" * 40,
|
||||
remote_master_sha="a" * 40,
|
||||
commits_behind_master=0,
|
||||
stale_runtime_warning=stale,
|
||||
shell_health={"shell_use_allowed": True, "consecutive_spawn_failures": 0},
|
||||
workflow_hashes=(
|
||||
FileHash(label="SKILL.md", path="skills/llm-project-workflow/SKILL.md", sha256="abc"),
|
||||
),
|
||||
schema_hashes=(),
|
||||
restart_guidance="docs/mcp-namespace-eof-recovery.md",
|
||||
fetch_error=None,
|
||||
)
|
||||
|
||||
|
||||
def _inventory(
|
||||
*,
|
||||
sessions: tuple[dict, ...] = (),
|
||||
leases: tuple[dict, ...] = (),
|
||||
locks: tuple[dict, ...] = (),
|
||||
worktrees: tuple[dict, ...] = (),
|
||||
namespaces: tuple[dict, ...] = (),
|
||||
statuses: dict[str, str] | None = None,
|
||||
) -> InventorySnapshot:
|
||||
"""Build a snapshot; ``statuses`` degrades named sections (default all ok)."""
|
||||
status_of = statuses or {}
|
||||
|
||||
def _status(name: str) -> str:
|
||||
return status_of.get(name, STATUS_OK)
|
||||
|
||||
sections = (
|
||||
InventorySection(
|
||||
name="sessions",
|
||||
authority=AUTHORITY_CONTROL_PLANE_DB,
|
||||
status=_status("sessions"),
|
||||
items=sessions,
|
||||
),
|
||||
InventorySection(
|
||||
name="leases",
|
||||
authority=AUTHORITY_CONTROL_PLANE_DB,
|
||||
status=_status("leases"),
|
||||
items=leases,
|
||||
),
|
||||
InventorySection(
|
||||
name="locks",
|
||||
authority=AUTHORITY_FILESYSTEM,
|
||||
status=_status("locks"),
|
||||
items=locks,
|
||||
),
|
||||
InventorySection(
|
||||
name="worktrees",
|
||||
authority=AUTHORITY_FILESYSTEM,
|
||||
status=_status("worktrees"),
|
||||
items=worktrees,
|
||||
),
|
||||
InventorySection(
|
||||
name="namespaces",
|
||||
authority=AUTHORITY_FILESYSTEM,
|
||||
status=_status("namespaces"),
|
||||
items=namespaces
|
||||
or (
|
||||
{
|
||||
"profile_name": "prgs-author",
|
||||
"role": "author",
|
||||
"mcp_namespace": "gitea-author",
|
||||
"capability_summary": {
|
||||
"can_author": True,
|
||||
"can_review": False,
|
||||
"can_merge": False,
|
||||
},
|
||||
"active": True,
|
||||
},
|
||||
),
|
||||
reason="only the profile serving this web process is observable",
|
||||
),
|
||||
)
|
||||
index = {section.name: section for section in sections}
|
||||
return InventorySnapshot(
|
||||
generated_at="2026-07-25T00:00:00+00:00",
|
||||
sections=sections,
|
||||
collisions=(),
|
||||
correlations=(),
|
||||
scan_ms=1.0,
|
||||
_section_index=index,
|
||||
)
|
||||
|
||||
|
||||
def _clean_session() -> dict:
|
||||
return {
|
||||
"session_id": "prgs-author-111-clean",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"namespace": "gitea-author",
|
||||
"pid": 1111,
|
||||
"pid_alive": True,
|
||||
"status": "active",
|
||||
"started_at": "2026-07-25T00:00:00Z",
|
||||
"last_heartbeat_at": "2026-07-25T01:00:00Z",
|
||||
}
|
||||
|
||||
|
||||
def _stale_session() -> dict:
|
||||
return {
|
||||
"session_id": "prgs-author-222-stale",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"namespace": "gitea-author",
|
||||
"pid": 2222,
|
||||
"pid_alive": False,
|
||||
"status": "active",
|
||||
"started_at": "2026-07-24T00:00:00Z",
|
||||
"last_heartbeat_at": "2026-07-24T01:00:00Z",
|
||||
}
|
||||
|
||||
|
||||
class TestBuildSessionRows(unittest.TestCase):
|
||||
def test_clean_session_has_no_stale_or_contamination_flags(self):
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(),),
|
||||
leases=(
|
||||
{
|
||||
"lease_id": "lease-clean",
|
||||
"session_id": "prgs-author-111-clean",
|
||||
"status": "active",
|
||||
"expired": False,
|
||||
"work_kind": "issue",
|
||||
"work_number": 641,
|
||||
},
|
||||
),
|
||||
locks=(
|
||||
{
|
||||
"issue_number": 641,
|
||||
"branch_name": "feat/issue-641-runtime-session-view",
|
||||
"worktree_path": "~/Development/Gitea-Tools/branches/feat-issue-641",
|
||||
"live": True,
|
||||
},
|
||||
),
|
||||
)
|
||||
rows = _build_session_rows(inventory, contamination=())
|
||||
self.assertEqual(len(rows), 1)
|
||||
row = rows[0]
|
||||
self.assertEqual(row.session_id, "prgs-author-111-clean")
|
||||
self.assertEqual(row.role, "author")
|
||||
self.assertEqual(row.namespace, "gitea-author")
|
||||
self.assertEqual(row.pid_alive, True)
|
||||
self.assertEqual(row.lease_ids, ("lease-clean",))
|
||||
self.assertEqual(row.work_refs, ("issue#641",))
|
||||
self.assertTrue(row.worktree_paths)
|
||||
self.assertEqual(row.stale_flags, ())
|
||||
self.assertEqual(row.contamination_flags, ())
|
||||
|
||||
def test_stale_session_flags_dead_pid(self):
|
||||
inventory = _inventory(sessions=(_stale_session(),))
|
||||
rows = _build_session_rows(inventory, contamination=())
|
||||
self.assertEqual(rows[0].stale_flags, ("pid-dead",))
|
||||
|
||||
def test_contamination_marker_binds_to_session(self):
|
||||
inventory = _inventory(sessions=(_clean_session(),))
|
||||
marker = ContaminationMarker(
|
||||
kind="runtime_recovery_contamination",
|
||||
on_disk=True,
|
||||
has_payload=True,
|
||||
summary="manual daemon kill",
|
||||
reason_class="manual_daemon_kill",
|
||||
session_id="prgs-author-111-clean",
|
||||
role="author",
|
||||
command_summary="pkill -f mcp_server.py",
|
||||
cleared=False,
|
||||
)
|
||||
rows = _build_session_rows(inventory, contamination=(marker,))
|
||||
self.assertIn("runtime_recovery_contamination", rows[0].contamination_flags)
|
||||
|
||||
def test_process_wide_contamination_surfaces_on_all_sessions(self):
|
||||
inventory = _inventory(sessions=(_clean_session(), _stale_session()))
|
||||
marker = ContaminationMarker(
|
||||
kind="stable_branch_contamination",
|
||||
on_disk=True,
|
||||
has_payload=True,
|
||||
summary="direct master push attempt",
|
||||
reason_class="stable_branch_push",
|
||||
session_id=None,
|
||||
cleared=False,
|
||||
)
|
||||
rows = _build_session_rows(inventory, contamination=(marker,))
|
||||
self.assertEqual(len(rows), 2)
|
||||
for row in rows:
|
||||
self.assertTrue(
|
||||
any("stable_branch_contamination" in f for f in row.contamination_flags)
|
||||
)
|
||||
|
||||
|
||||
class TestRenderSessionsPage(unittest.TestCase):
|
||||
def _snapshot(
|
||||
self,
|
||||
*,
|
||||
sessions: tuple[dict, ...],
|
||||
contamination: tuple[ContaminationMarker, ...] = (),
|
||||
stale_runtime: str | None = None,
|
||||
) -> SessionViewSnapshot:
|
||||
inventory = _inventory(
|
||||
sessions=sessions,
|
||||
leases=(
|
||||
{
|
||||
"lease_id": "lease-1",
|
||||
"session_id": sessions[0]["session_id"] if sessions else "",
|
||||
"status": "active",
|
||||
"expired": False,
|
||||
"work_kind": "issue",
|
||||
"work_number": 641,
|
||||
},
|
||||
)
|
||||
if sessions
|
||||
else (),
|
||||
locks=(
|
||||
{
|
||||
"issue_number": 641,
|
||||
"worktree_path": "branches/feat-issue-641",
|
||||
},
|
||||
)
|
||||
if sessions
|
||||
else (),
|
||||
worktrees=(
|
||||
{
|
||||
"rel_path": "branches/feat-issue-641",
|
||||
"branch": "feat/issue-641-runtime-session-view",
|
||||
"classification": "active_issue_work",
|
||||
"registered_worktree": True,
|
||||
"dirty": False,
|
||||
},
|
||||
),
|
||||
)
|
||||
rows = _build_session_rows(inventory, contamination)
|
||||
return SessionViewSnapshot(
|
||||
runtime=_runtime(stale=stale_runtime),
|
||||
inventory=inventory,
|
||||
sessions=rows,
|
||||
contamination_markers=contamination,
|
||||
)
|
||||
|
||||
def test_clean_session_render(self):
|
||||
html = render_sessions_page(self._snapshot(sessions=(_clean_session(),)))
|
||||
self.assertIn("Runtime and sessions", html)
|
||||
self.assertIn("prgs-author-111-clean", html)
|
||||
self.assertIn("gitea-author", html)
|
||||
self.assertIn("branches/feat-issue-641", html)
|
||||
self.assertIn("Sanctioned recovery", html)
|
||||
self.assertIn("docs/mcp-namespace-eof-recovery.md", html)
|
||||
# Recovery section must name reconnect and forbid manual kill.
|
||||
recovery_idx = html.lower().find("sanctioned recovery")
|
||||
self.assertGreaterEqual(recovery_idx, 0)
|
||||
recovery = html[recovery_idx:].lower()
|
||||
self.assertIn("reconnect", recovery)
|
||||
self.assertIn("contamination", recovery)
|
||||
self.assertIn("not recovery", recovery)
|
||||
self.assertNotIn("run pkill", recovery)
|
||||
self.assertNotIn("killall", recovery)
|
||||
|
||||
def test_stale_session_render(self):
|
||||
html = render_sessions_page(self._snapshot(sessions=(_stale_session(),)))
|
||||
self.assertIn("prgs-author-222-stale", html)
|
||||
self.assertIn("pid-dead", html)
|
||||
self.assertIn("badge-stale", html)
|
||||
|
||||
def test_contamination_render_is_not_silent(self):
|
||||
marker = ContaminationMarker(
|
||||
kind="runtime_recovery_contamination",
|
||||
on_disk=True,
|
||||
has_payload=True,
|
||||
summary="manual kill",
|
||||
reason_class="manual_daemon_kill",
|
||||
session_id="prgs-author-111-clean",
|
||||
command_summary="pkill -f mcp_server.py",
|
||||
cleared=False,
|
||||
)
|
||||
html = render_sessions_page(
|
||||
self._snapshot(sessions=(_clean_session(),), contamination=(marker,))
|
||||
)
|
||||
self.assertIn("Contamination markers", html)
|
||||
self.assertIn("runtime_recovery_contamination", html)
|
||||
self.assertIn("ACTIVE", html)
|
||||
self.assertIn("badge-blocked", html)
|
||||
|
||||
def test_stale_runtime_banner(self):
|
||||
html = render_sessions_page(
|
||||
self._snapshot(
|
||||
sessions=(_clean_session(),),
|
||||
stale_runtime="server behind master by 3 commits",
|
||||
)
|
||||
)
|
||||
self.assertIn("Stale runtime", html)
|
||||
self.assertIn("server behind master", html)
|
||||
|
||||
|
||||
class TestSessionLoaderComposition(unittest.TestCase):
|
||||
def test_load_with_injected_sources(self):
|
||||
inventory = _inventory(sessions=(_clean_session(), _stale_session()))
|
||||
snap = load_session_view_snapshot(
|
||||
load_runtime=lambda: _runtime(),
|
||||
load_inventory=lambda: inventory,
|
||||
inspect_contamination=lambda **_k: {
|
||||
"on_disk": False,
|
||||
"has_payload": False,
|
||||
"summary": "absent",
|
||||
},
|
||||
load_contamination_payload=lambda **_k: None,
|
||||
)
|
||||
self.assertEqual(len(snap.sessions), 2)
|
||||
self.assertEqual(snap.stale_session_count, 1)
|
||||
self.assertEqual(snap.contaminated_session_count, 0)
|
||||
data = snapshot_to_dict(snap)
|
||||
self.assertEqual(data["view"], "runtime-sessions")
|
||||
self.assertEqual(data["issue"], 641)
|
||||
self.assertTrue(data["read_only"])
|
||||
self.assertEqual(data["session_counts"]["total"], 2)
|
||||
self.assertEqual(data["session_counts"]["stale"], 1)
|
||||
self.assertIn("recovery_docs", data)
|
||||
|
||||
|
||||
class TestSessionsRoutes(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.client = TestClient(create_app())
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(), _stale_session()),
|
||||
leases=(
|
||||
{
|
||||
"lease_id": "lease-x",
|
||||
"session_id": "prgs-author-111-clean",
|
||||
"status": "active",
|
||||
"expired": False,
|
||||
"work_kind": "issue",
|
||||
"work_number": 641,
|
||||
},
|
||||
),
|
||||
locks=(
|
||||
{
|
||||
"issue_number": 641,
|
||||
"worktree_path": "branches/feat-issue-641",
|
||||
},
|
||||
),
|
||||
worktrees=(
|
||||
{
|
||||
"rel_path": "branches/feat-issue-641",
|
||||
"branch": "feat/issue-641-runtime-session-view",
|
||||
"classification": "active_issue_work",
|
||||
"registered_worktree": True,
|
||||
"dirty": False,
|
||||
},
|
||||
),
|
||||
)
|
||||
rows = _build_session_rows(inventory, contamination=())
|
||||
self.snapshot = SessionViewSnapshot(
|
||||
runtime=_runtime(stale="stale for test"),
|
||||
inventory=inventory,
|
||||
sessions=rows,
|
||||
contamination_markers=(),
|
||||
)
|
||||
self._patch = mock.patch(
|
||||
"webui.app.load_session_view_snapshot",
|
||||
return_value=self.snapshot,
|
||||
)
|
||||
self._patch.start()
|
||||
|
||||
def tearDown(self):
|
||||
self._patch.stop()
|
||||
|
||||
def test_sessions_page_live(self):
|
||||
response = self.client.get("/sessions")
|
||||
self.assertEqual(response.status_code, 200)
|
||||
self.assertIn("Runtime and sessions", response.text)
|
||||
self.assertIn("prgs-author-111-clean", response.text)
|
||||
self.assertIn("prgs-author-222-stale", response.text)
|
||||
self.assertIn("pid-dead", response.text)
|
||||
self.assertIn("Sanctioned recovery", response.text)
|
||||
self.assertNotIn("Phase 1 shell placeholder", response.text)
|
||||
self.assertNotIn("child issue of #425", response.text.lower())
|
||||
|
||||
def test_api_sessions_json(self):
|
||||
for path in ("/api/sessions", "/api/v1/sessions"):
|
||||
response = self.client.get(path)
|
||||
self.assertEqual(response.status_code, 200, path)
|
||||
data = response.json()
|
||||
self.assertEqual(data["view"], "runtime-sessions")
|
||||
self.assertEqual(data["session_counts"]["total"], 2)
|
||||
self.assertEqual(data["session_counts"]["stale"], 1)
|
||||
self.assertTrue(data["read_only"])
|
||||
self.assertEqual(data["mutations"], [])
|
||||
|
||||
def test_nav_marks_sessions_live(self):
|
||||
sessions_items = [
|
||||
item for item in iter_nav_items() if item.href == "/sessions"
|
||||
]
|
||||
self.assertEqual(len(sessions_items), 1)
|
||||
self.assertEqual(sessions_items[0].status, "live")
|
||||
self.assertNotIn("/sessions", STUB_PAGES)
|
||||
# Home page should not mark Sessions as stub.
|
||||
home = self.client.get("/")
|
||||
self.assertEqual(home.status_code, 200)
|
||||
self.assertIn('href="/sessions"', home.text)
|
||||
# Stub marker only appears next to remaining stub destinations.
|
||||
self.assertNotIn(
|
||||
'href="/sessions">Sessions</a> <span class="muted">(stub)</span>',
|
||||
home.text,
|
||||
)
|
||||
|
||||
|
||||
class TestDegradedOwnershipAuthority(unittest.TestCase):
|
||||
"""B1: a section that could not be read must never render as absence."""
|
||||
|
||||
def _snapshot(self, inventory: InventorySnapshot) -> SessionViewSnapshot:
|
||||
return SessionViewSnapshot(
|
||||
runtime=_runtime(),
|
||||
inventory=inventory,
|
||||
sessions=_build_session_rows(inventory, contamination=()),
|
||||
contamination_markers=(),
|
||||
)
|
||||
|
||||
def test_unavailable_leases_mark_row_authority_unproven(self):
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(),),
|
||||
statuses={"leases": STATUS_UNAVAILABLE},
|
||||
)
|
||||
row = _build_session_rows(inventory, contamination=())[0]
|
||||
self.assertEqual(row.lease_ids, ())
|
||||
self.assertEqual(row.lease_authority, STATUS_UNAVAILABLE)
|
||||
# Worktree binding is correlated through lease work numbers, so it
|
||||
# inherits the unreadable lease section.
|
||||
self.assertEqual(row.worktree_authority, STATUS_UNAVAILABLE)
|
||||
self.assertFalse(row.ownership_authority_complete)
|
||||
|
||||
def test_readable_locks_are_not_reported_unbound_when_leases_degrade(self):
|
||||
# The narrow variant: locks hold a real worktree_path and read cleanly,
|
||||
# but the lease section that supplies the correlating work number does
|
||||
# not. The row must say unknown, not "unbound".
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(),),
|
||||
locks=(
|
||||
{
|
||||
"issue_number": 641,
|
||||
"worktree_path": "branches/feat-issue-641",
|
||||
},
|
||||
),
|
||||
statuses={"leases": STATUS_DEGRADED},
|
||||
)
|
||||
row = _build_session_rows(inventory, contamination=())[0]
|
||||
self.assertEqual(row.worktree_paths, ())
|
||||
self.assertEqual(row.worktree_authority, STATUS_DEGRADED)
|
||||
self.assertFalse(row.ownership_authority_complete)
|
||||
|
||||
def test_degraded_render_says_unknown_not_none_or_unbound(self):
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(),),
|
||||
statuses={"leases": STATUS_UNAVAILABLE, "locks": STATUS_UNAVAILABLE},
|
||||
)
|
||||
html = render_sessions_page(self._snapshot(inventory))
|
||||
self.assertIn("unknown (inventory unavailable)", html)
|
||||
self.assertIn("authority unproven", html)
|
||||
self.assertIn("Ownership authority incomplete", html)
|
||||
# The affirmative-absence strings must be gone from the row entirely.
|
||||
self.assertNotIn(">none<", html)
|
||||
self.assertNotIn(">unbound<", html)
|
||||
|
||||
def test_clean_inventory_still_renders_affirmative_absence(self):
|
||||
# Guards against over-correcting B1 into "everything is unknown".
|
||||
inventory = _inventory(sessions=(_clean_session(),))
|
||||
html = render_sessions_page(self._snapshot(inventory))
|
||||
self.assertIn(">none<", html)
|
||||
self.assertIn(">unbound<", html)
|
||||
# The column legend mentions "unknown (inventory …)" as static copy, so
|
||||
# assert on the per-row marker and the concrete statuses instead.
|
||||
self.assertNotIn("authority unproven", html)
|
||||
self.assertNotIn("unknown (inventory unavailable)", html)
|
||||
self.assertNotIn("unknown (inventory degraded)", html)
|
||||
self.assertNotIn("Ownership authority incomplete", html)
|
||||
|
||||
def test_json_export_carries_snapshot_and_per_row_authority(self):
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(),),
|
||||
statuses={"locks": STATUS_UNAVAILABLE},
|
||||
)
|
||||
data = snapshot_to_dict(self._snapshot(inventory))
|
||||
self.assertFalse(data["ownership_authority_complete"])
|
||||
self.assertEqual(
|
||||
data["ownership_section_status"]["locks"], STATUS_UNAVAILABLE
|
||||
)
|
||||
self.assertEqual(data["ownership_section_status"]["leases"], STATUS_OK)
|
||||
self.assertIn("unknown, not unowned", data["ownership_note"])
|
||||
|
||||
row = data["sessions"][0]
|
||||
self.assertTrue(row["lease_authority_complete"])
|
||||
self.assertFalse(row["worktree_authority_complete"])
|
||||
self.assertEqual(row["worktree_authority"], STATUS_UNAVAILABLE)
|
||||
self.assertFalse(row["ownership_authority_complete"])
|
||||
self.assertIn("unknown, not unowned", row["ownership_note"])
|
||||
|
||||
def test_json_export_is_affirmative_when_every_source_reads(self):
|
||||
inventory = _inventory(sessions=(_clean_session(),))
|
||||
data = snapshot_to_dict(self._snapshot(inventory))
|
||||
self.assertTrue(data["ownership_authority_complete"])
|
||||
self.assertTrue(data["sessions"][0]["ownership_authority_complete"])
|
||||
|
||||
def test_missing_session_list_is_not_reported_as_no_sessions(self):
|
||||
inventory = _inventory(statuses={"sessions": STATUS_UNAVAILABLE})
|
||||
html = render_sessions_page(self._snapshot(inventory))
|
||||
self.assertIn("could not be read", html)
|
||||
self.assertIn("not evidence that no sessions exist", html)
|
||||
|
||||
def test_expired_lease_flags_row_as_stale(self):
|
||||
inventory = _inventory(
|
||||
sessions=(_clean_session(),),
|
||||
leases=(
|
||||
{
|
||||
"lease_id": "lease-expired-1",
|
||||
"session_id": "prgs-author-111-clean",
|
||||
"status": "active",
|
||||
"expired": True,
|
||||
"work_kind": "issue",
|
||||
"work_number": 641,
|
||||
},
|
||||
),
|
||||
)
|
||||
row = _build_session_rows(inventory, contamination=())[0]
|
||||
self.assertIn("lease-expired", row.stale_flags)
|
||||
self.assertIn("active-lease-past-expiry", row.stale_flags)
|
||||
|
||||
|
||||
class TestContaminationRedaction(unittest.TestCase):
|
||||
"""B2: marker payload text is redacted at the display boundary."""
|
||||
|
||||
def _marker(self, payload: dict) -> ContaminationMarker:
|
||||
return _inspect_contamination(
|
||||
"runtime_recovery_contamination",
|
||||
remote="prgs",
|
||||
inspect=lambda **_k: {
|
||||
"on_disk": True,
|
||||
"has_payload": True,
|
||||
"summary": "",
|
||||
},
|
||||
load=lambda **_k: payload,
|
||||
)
|
||||
|
||||
def test_home_paths_are_collapsed(self):
|
||||
home = os.path.expanduser("~")
|
||||
marker = self._marker(
|
||||
{"command_summary": f"pkill -f {home}/Development/Gitea-Tools/x.py"}
|
||||
)
|
||||
self.assertNotIn(home, marker.command_summary)
|
||||
self.assertIn("~/Development/Gitea-Tools/x.py", marker.command_summary)
|
||||
|
||||
def test_secrets_missed_by_the_write_time_denylist_are_redacted(self):
|
||||
# Each of these was verified in review to survive
|
||||
# stable_branch_push_guard.redact_command untouched.
|
||||
cases = (
|
||||
("curl -H 'X-Api-Key: SUPERSECRET123' https://example.invalid", "SUPERSECRET123"),
|
||||
("cmd --password hunter2 origin master", "hunter2"),
|
||||
("PRIVATE_KEY=abc123 python deploy.py", "abc123"),
|
||||
("fetch https://user:[email protected]/x.git", "user:pw"),
|
||||
)
|
||||
for raw, secret in cases:
|
||||
with self.subTest(raw=raw):
|
||||
marker = self._marker({"command_summary": raw})
|
||||
self.assertNotIn(secret, marker.command_summary)
|
||||
self.assertIn("[redacted]", marker.command_summary)
|
||||
|
||||
def test_command_summary_is_redacted_not_removed(self):
|
||||
# It is legitimate #630 evidence: the operator must still see which
|
||||
# daemon was killed.
|
||||
marker = self._marker(
|
||||
{
|
||||
"command_summary": "pkill -f gitea_mcp_server.py",
|
||||
"reason_class": "manual_daemon_kill",
|
||||
"session_id": "prgs-author-111-clean",
|
||||
"role": "author",
|
||||
}
|
||||
)
|
||||
self.assertIn("pkill -f gitea_mcp_server.py", marker.command_summary)
|
||||
self.assertEqual(marker.reason_class, "manual_daemon_kill")
|
||||
self.assertEqual(marker.session_id, "prgs-author-111-clean")
|
||||
self.assertEqual(marker.role, "author")
|
||||
|
||||
def test_rendered_page_exposes_no_home_path_from_a_marker(self):
|
||||
home = os.path.expanduser("~")
|
||||
marker = self._marker(
|
||||
{
|
||||
"command_summary": f"pkill -f {home}/Development/Gitea-Tools/x.py",
|
||||
"reason_class": "manual_daemon_kill",
|
||||
}
|
||||
)
|
||||
inventory = _inventory(sessions=(_clean_session(),))
|
||||
html = render_sessions_page(
|
||||
SessionViewSnapshot(
|
||||
runtime=_runtime(),
|
||||
inventory=inventory,
|
||||
sessions=_build_session_rows(inventory, (marker,)),
|
||||
contamination_markers=(marker,),
|
||||
)
|
||||
)
|
||||
self.assertIn("Contamination markers", html)
|
||||
self.assertNotIn(home, html)
|
||||
|
||||
|
||||
class TestSessionsPageEscaping(unittest.TestCase):
|
||||
"""Hostile values from every rendered source stay inert (N2)."""
|
||||
|
||||
HOSTILE = '<script>alert("xss")</script>'
|
||||
|
||||
def test_hostile_session_and_marker_values_are_escaped(self):
|
||||
session = dict(_clean_session())
|
||||
session["session_id"] = f"sid-{self.HOSTILE}"
|
||||
session["role"] = self.HOSTILE
|
||||
session["profile"] = self.HOSTILE
|
||||
session["namespace"] = self.HOSTILE
|
||||
session["status"] = self.HOSTILE
|
||||
inventory = _inventory(
|
||||
sessions=(session,),
|
||||
leases=(
|
||||
{
|
||||
"lease_id": self.HOSTILE,
|
||||
"session_id": session["session_id"],
|
||||
"status": "active",
|
||||
"expired": False,
|
||||
"work_kind": self.HOSTILE,
|
||||
"work_number": 641,
|
||||
},
|
||||
),
|
||||
locks=(
|
||||
{
|
||||
"issue_number": 641,
|
||||
"worktree_path": self.HOSTILE,
|
||||
},
|
||||
),
|
||||
)
|
||||
marker = ContaminationMarker(
|
||||
kind="runtime_recovery_contamination",
|
||||
on_disk=True,
|
||||
has_payload=True,
|
||||
summary=self.HOSTILE,
|
||||
reason_class=self.HOSTILE,
|
||||
session_id=session["session_id"],
|
||||
role=self.HOSTILE,
|
||||
command_summary=self.HOSTILE,
|
||||
cleared=False,
|
||||
)
|
||||
html = render_sessions_page(
|
||||
SessionViewSnapshot(
|
||||
runtime=_runtime(),
|
||||
inventory=inventory,
|
||||
sessions=_build_session_rows(inventory, (marker,)),
|
||||
contamination_markers=(marker,),
|
||||
)
|
||||
)
|
||||
self.assertNotIn("<script>", html)
|
||||
self.assertNotIn('alert("xss")', html)
|
||||
self.assertIn("<script>", html)
|
||||
|
||||
def test_hostile_values_in_a_degraded_render_are_escaped(self):
|
||||
inventory = _inventory(
|
||||
sessions=(dict(_clean_session(), session_id=f"sid-{self.HOSTILE}"),),
|
||||
statuses={"leases": STATUS_UNAVAILABLE, "locks": STATUS_DEGRADED},
|
||||
)
|
||||
html = render_sessions_page(
|
||||
SessionViewSnapshot(
|
||||
runtime=_runtime(),
|
||||
inventory=inventory,
|
||||
sessions=_build_session_rows(inventory, contamination=()),
|
||||
contamination_markers=(),
|
||||
)
|
||||
)
|
||||
self.assertNotIn("<script>", html)
|
||||
self.assertIn("<script>", html)
|
||||
self.assertIn("unknown (inventory", html)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,408 @@
|
||||
"""Tests for web UI workflow traffic-control view (#640)."""
|
||||
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from starlette.testclient import TestClient
|
||||
|
||||
from webui.app import create_app
|
||||
from webui.traffic_loader import (
|
||||
TrafficItem,
|
||||
TrafficSnapshot,
|
||||
load_traffic_snapshot,
|
||||
snapshot_to_dict,
|
||||
)
|
||||
from webui.traffic_views import render_traffic_page
|
||||
from allocator_service import WorkCandidate
|
||||
|
||||
|
||||
class TestTrafficClassification(unittest.TestCase):
|
||||
def test_runnable_candidate_classification(self):
|
||||
cand = WorkCandidate(
|
||||
kind="issue",
|
||||
number=640,
|
||||
state="open",
|
||||
labels=("status:ready",),
|
||||
title="Web Console: Workflow traffic-control view (Phase 1)",
|
||||
priority=20,
|
||||
)
|
||||
snap = load_traffic_snapshot(candidates=[cand])
|
||||
self.assertEqual(len(snap.runnable), 1)
|
||||
self.assertEqual(snap.runnable[0].number, 640)
|
||||
self.assertTrue(snap.runnable[0].is_safe)
|
||||
self.assertEqual(snap.runnable[0].traffic_state, "runnable")
|
||||
|
||||
def test_blocked_dependency_candidate_classification(self):
|
||||
cand = WorkCandidate(
|
||||
kind="issue",
|
||||
number=643,
|
||||
state="open",
|
||||
labels=("status:ready",),
|
||||
title="Web Console: Requests & intent preview (Phase 2)",
|
||||
priority=20,
|
||||
dependency_unmet=True,
|
||||
dependency_reason="issue#643 depends on unresolved issue(s) #640; they are not closed",
|
||||
)
|
||||
snap = load_traffic_snapshot(candidates=[cand])
|
||||
self.assertEqual(len(snap.blocked), 1)
|
||||
self.assertEqual(snap.blocked[0].number, 643)
|
||||
self.assertFalse(snap.blocked[0].is_safe)
|
||||
self.assertEqual(snap.blocked[0].traffic_state, "blocked")
|
||||
self.assertIn("depends on unresolved issue(s) #640", snap.blocked[0].block_reason)
|
||||
|
||||
def test_leased_candidate_classification(self):
|
||||
cand = WorkCandidate(
|
||||
kind="issue",
|
||||
number=640,
|
||||
state="open",
|
||||
labels=("status:in-progress",),
|
||||
title="Web Console: Workflow traffic-control view (Phase 1)",
|
||||
priority=20,
|
||||
)
|
||||
lease = {
|
||||
"kind": "issue",
|
||||
"number": 640,
|
||||
"session_id": "prgs-author-12345",
|
||||
"role": "author",
|
||||
"status": "active",
|
||||
}
|
||||
snap = load_traffic_snapshot(candidates=[cand], leases=[lease])
|
||||
self.assertEqual(len(snap.leased), 1)
|
||||
self.assertEqual(snap.leased[0].number, 640)
|
||||
self.assertEqual(snap.leased[0].traffic_state, "leased")
|
||||
self.assertIsNotNone(snap.leased[0].lease_info)
|
||||
|
||||
def test_needs_controller_candidate_classification(self):
|
||||
cand = WorkCandidate(
|
||||
kind="issue",
|
||||
number=700,
|
||||
state="open",
|
||||
labels=("status:blocked",),
|
||||
title="Controller intervention needed",
|
||||
priority=10,
|
||||
blocked=True,
|
||||
)
|
||||
snap = load_traffic_snapshot(candidates=[cand])
|
||||
self.assertEqual(len(snap.needs_controller), 1)
|
||||
self.assertEqual(snap.needs_controller[0].number, 700)
|
||||
|
||||
|
||||
class TestTrafficLoader(unittest.TestCase):
|
||||
def test_snapshot_to_dict_export(self):
|
||||
cand = WorkCandidate(
|
||||
kind="issue",
|
||||
number=640,
|
||||
state="open",
|
||||
labels=("status:ready",),
|
||||
title="Traffic control test",
|
||||
priority=20,
|
||||
)
|
||||
snap = load_traffic_snapshot(candidates=[cand])
|
||||
data = snapshot_to_dict(snap)
|
||||
self.assertEqual(data["project_id"], "gitea-tools")
|
||||
self.assertEqual(len(data["runnable"]), 1)
|
||||
self.assertTrue(data["inventory_complete"])
|
||||
|
||||
def test_fail_closed_error_handling(self):
|
||||
with mock.patch("webui.traffic_loader.load_queue_snapshot", side_effect=RuntimeError("Gitea connection failed")):
|
||||
snap = load_traffic_snapshot()
|
||||
self.assertIsNotNone(snap.fetch_error)
|
||||
self.assertIn("Failed to load traffic state", snap.fetch_error)
|
||||
self.assertEqual(len(snap.runnable), 0)
|
||||
self.assertFalse(snap.inventory_complete)
|
||||
|
||||
|
||||
class TestTrafficLivePath(unittest.TestCase):
|
||||
"""Live path tests: inject QueueSnapshot + LeaseSnapshot (no candidates=).
|
||||
|
||||
Covers the production ``load_traffic_snapshot()`` branch that ``/traffic``
|
||||
and ``/api/traffic`` actually execute (#640 B1–B5).
|
||||
"""
|
||||
|
||||
FULL_SHA = "069a9af7e6aa2c2994e07199d1b0814819457017"
|
||||
|
||||
def _queue(
|
||||
self,
|
||||
*,
|
||||
prs=(),
|
||||
issues=(),
|
||||
):
|
||||
from webui.queue_loader import QueueSnapshot
|
||||
|
||||
return QueueSnapshot(
|
||||
project_id="gitea-tools",
|
||||
repo_label="Scaled-Tech-Consulting/Gitea-Tools",
|
||||
prs=tuple(prs),
|
||||
issues=tuple(issues),
|
||||
pr_pagination=None,
|
||||
issue_pagination=None,
|
||||
fetch_error=None,
|
||||
)
|
||||
|
||||
def _lease(
|
||||
self,
|
||||
*,
|
||||
claim_inventory=None,
|
||||
reviewer_leases=(),
|
||||
):
|
||||
from webui.lease_loader import LeaseSnapshot
|
||||
|
||||
return LeaseSnapshot(
|
||||
project_id="gitea-tools",
|
||||
repo_label="Scaled-Tech-Consulting/Gitea-Tools",
|
||||
issue_lock=None,
|
||||
claim_inventory=claim_inventory or {"entries": [], "counts": {}},
|
||||
reviewer_leases=tuple(reviewer_leases),
|
||||
duplicate_prs=(),
|
||||
duplicate_branches=(),
|
||||
collision_history=(),
|
||||
fetch_error=None,
|
||||
)
|
||||
|
||||
def test_live_pr_uses_full_head_sha_and_is_runnable(self):
|
||||
from webui.queue_loader import QueueItem
|
||||
|
||||
pr = QueueItem(
|
||||
number=885,
|
||||
title="traffic control",
|
||||
badges=("in-review",),
|
||||
extra={"head_sha": self.FULL_SHA[:12], "linked_issue": "640"},
|
||||
signals={
|
||||
"head_sha": self.FULL_SHA,
|
||||
"mergeable": True,
|
||||
"labels": (),
|
||||
"linked_issue": 640,
|
||||
},
|
||||
)
|
||||
q = self._queue(prs=[pr])
|
||||
l = self._lease()
|
||||
snap = load_traffic_snapshot(
|
||||
fetch_queue_snapshot=lambda: q,
|
||||
fetch_lease_snapshot=lambda: l,
|
||||
)
|
||||
self.assertIsNone(snap.fetch_error)
|
||||
self.assertEqual(len(snap.runnable), 1)
|
||||
item = snap.runnable[0]
|
||||
self.assertEqual(item.kind, "pr")
|
||||
self.assertEqual(item.number, 885)
|
||||
self.assertEqual(item.head_sha, self.FULL_SHA)
|
||||
self.assertNotEqual(item.head_sha, self.FULL_SHA[:12])
|
||||
self.assertIsNone(item.block_reason)
|
||||
self.assertEqual(len(snap.blocked), 0)
|
||||
|
||||
def test_live_pr_without_head_sha_is_blocked(self):
|
||||
from webui.queue_loader import QueueItem
|
||||
|
||||
pr = QueueItem(
|
||||
number=1,
|
||||
title="missing pin",
|
||||
badges=("open",),
|
||||
extra={"head_sha": ""},
|
||||
signals={"head_sha": "", "mergeable": True, "labels": ()},
|
||||
)
|
||||
snap = load_traffic_snapshot(
|
||||
fetch_queue_snapshot=lambda: self._queue(prs=[pr]),
|
||||
fetch_lease_snapshot=lambda: self._lease(),
|
||||
)
|
||||
self.assertEqual(len(snap.blocked) + len(snap.needs_controller), 1)
|
||||
item = (snap.blocked or snap.needs_controller)[0]
|
||||
self.assertIn("missing head_sha", (item.block_reason or "").lower())
|
||||
|
||||
def test_reviewer_lease_keys_by_pr_not_linked_issue(self):
|
||||
from webui.queue_loader import QueueItem
|
||||
|
||||
pr = QueueItem(
|
||||
number=885,
|
||||
title="leased pr",
|
||||
badges=("in-review",),
|
||||
extra={"head_sha": self.FULL_SHA[:12]},
|
||||
signals={"head_sha": self.FULL_SHA, "mergeable": True, "labels": ()},
|
||||
)
|
||||
issue = QueueItem(
|
||||
number=640,
|
||||
title="linked issue",
|
||||
badges=("open",),
|
||||
extra={},
|
||||
signals={"labels": ()},
|
||||
)
|
||||
# Marker-shaped record: has both pr_number and issue_number; must
|
||||
# attach to the PR only (B2).
|
||||
reviewer_lease = {
|
||||
"pr_number": 885,
|
||||
"issue_number": 640,
|
||||
"phase": "validating",
|
||||
"reviewer_identity": "sysadmin",
|
||||
"session_id": "review-sess-1",
|
||||
}
|
||||
snap = load_traffic_snapshot(
|
||||
fetch_queue_snapshot=lambda: self._queue(prs=[pr], issues=[issue]),
|
||||
fetch_lease_snapshot=lambda: self._lease(reviewer_leases=[reviewer_lease]),
|
||||
)
|
||||
leased_prs = [i for i in snap.leased if i.kind == "pr" and i.number == 885]
|
||||
self.assertEqual(len(leased_prs), 1)
|
||||
self.assertEqual(leased_prs[0].lease_info.get("pr_number"), 885)
|
||||
# Issue 640 must not inherit the reviewer lease just because issue_number
|
||||
# is present on the marker.
|
||||
for item in list(snap.leased) + list(snap.runnable) + list(snap.blocked):
|
||||
if item.kind == "issue" and item.number == 640:
|
||||
self.assertIsNone(
|
||||
item.lease_info,
|
||||
"reviewer lease must not attach to linked issue #640",
|
||||
)
|
||||
break
|
||||
else:
|
||||
self.fail("expected issue #640 in traffic snapshot")
|
||||
|
||||
def test_claim_inventory_entries_key_marks_issue_leased(self):
|
||||
from webui.queue_loader import QueueItem
|
||||
|
||||
issue = QueueItem(
|
||||
number=640,
|
||||
title="claimed issue",
|
||||
badges=("claimed",),
|
||||
extra={},
|
||||
signals={"labels": ("status:in-progress",)},
|
||||
)
|
||||
inventory = {
|
||||
"entries": [
|
||||
{
|
||||
"issue_number": 640,
|
||||
"status": "active",
|
||||
"latest_heartbeat": {"session_id": "author-sess-9"},
|
||||
"reasons": ["claim has structured heartbeat proof"],
|
||||
}
|
||||
],
|
||||
"counts": {"active": 1},
|
||||
"in_progress_total": 1,
|
||||
}
|
||||
snap = load_traffic_snapshot(
|
||||
fetch_queue_snapshot=lambda: self._queue(issues=[issue]),
|
||||
fetch_lease_snapshot=lambda: self._lease(claim_inventory=inventory),
|
||||
)
|
||||
leased_issues = [i for i in snap.leased if i.kind == "issue" and i.number == 640]
|
||||
self.assertEqual(len(leased_issues), 1)
|
||||
self.assertEqual(leased_issues[0].traffic_state, "leased")
|
||||
|
||||
def test_active_claims_key_is_ignored(self):
|
||||
"""B3 regression: fictional ``active_claims`` must not create lease_info."""
|
||||
from webui.queue_loader import QueueItem
|
||||
|
||||
issue = QueueItem(
|
||||
number=640,
|
||||
title="open issue",
|
||||
badges=("open",),
|
||||
extra={},
|
||||
signals={"labels": ()},
|
||||
)
|
||||
# Only the broken key — must NOT produce lease_info. Entries-less
|
||||
# inventory is empty (entries is the real claim_inventory key).
|
||||
inventory = {
|
||||
"active_claims": [
|
||||
{
|
||||
"kind": "issue",
|
||||
"number": 640,
|
||||
"issue_number": 640,
|
||||
"status": "active",
|
||||
},
|
||||
],
|
||||
"counts": {},
|
||||
}
|
||||
snap = load_traffic_snapshot(
|
||||
fetch_queue_snapshot=lambda: self._queue(issues=[issue]),
|
||||
fetch_lease_snapshot=lambda: self._lease(claim_inventory=inventory),
|
||||
)
|
||||
items = [
|
||||
i
|
||||
for i in (
|
||||
list(snap.runnable)
|
||||
+ list(snap.leased)
|
||||
+ list(snap.blocked)
|
||||
+ list(snap.needs_controller)
|
||||
)
|
||||
if i.kind == "issue" and i.number == 640
|
||||
]
|
||||
self.assertEqual(len(items), 1)
|
||||
self.assertIsNone(
|
||||
items[0].lease_info,
|
||||
"active_claims is not a real inventory key; entries-only",
|
||||
)
|
||||
|
||||
|
||||
class TestTrafficRoutesAndRendering(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.client = TestClient(create_app())
|
||||
|
||||
def test_traffic_html_page_rendering(self):
|
||||
cand1 = WorkCandidate(
|
||||
kind="issue",
|
||||
number=640,
|
||||
state="open",
|
||||
labels=("status:ready",),
|
||||
title="Traffic View Implementation",
|
||||
priority=20,
|
||||
)
|
||||
cand2 = WorkCandidate(
|
||||
kind="issue",
|
||||
number=643,
|
||||
state="open",
|
||||
labels=("status:ready",),
|
||||
title="Dependent Feature",
|
||||
priority=20,
|
||||
dependency_unmet=True,
|
||||
dependency_reason="issue#643 depends on unresolved issue(s) #640; they are not closed",
|
||||
)
|
||||
snap = load_traffic_snapshot(candidates=[cand1, cand2])
|
||||
with mock.patch("webui.app.load_traffic_snapshot", return_value=snap):
|
||||
response = self.client.get("/traffic")
|
||||
|
||||
self.assertEqual(response.status_code, 200)
|
||||
self.assertIn("Workflow Traffic Control", response.text)
|
||||
self.assertIn("1. Runnable Lanes", response.text)
|
||||
self.assertIn("3. Blocked Items", response.text)
|
||||
self.assertIn("Traffic View Implementation", response.text)
|
||||
self.assertIn("depends on unresolved issue(s) #640", response.text)
|
||||
|
||||
def test_api_traffic_json_route(self):
|
||||
cand = WorkCandidate(
|
||||
kind="issue",
|
||||
number=640,
|
||||
state="open",
|
||||
labels=("status:ready",),
|
||||
title="Traffic View API Test",
|
||||
priority=20,
|
||||
)
|
||||
snap = load_traffic_snapshot(candidates=[cand])
|
||||
with mock.patch("webui.app.load_traffic_snapshot", return_value=snap):
|
||||
response = self.client.get("/api/traffic")
|
||||
|
||||
self.assertEqual(response.status_code, 200)
|
||||
data = response.json()
|
||||
self.assertEqual(data["project_id"], "gitea-tools")
|
||||
self.assertEqual(len(data["runnable"]), 1)
|
||||
self.assertEqual(data["runnable"][0]["number"], 640)
|
||||
|
||||
def test_render_traffic_fail_closed_page(self):
|
||||
snap = TrafficSnapshot(
|
||||
project_id="gitea-tools",
|
||||
repo_label="Scaled-Tech-Consulting/Gitea-Tools",
|
||||
runnable=(),
|
||||
leased=(),
|
||||
blocked=(),
|
||||
needs_controller=(),
|
||||
terminal_complete=(),
|
||||
next_roles=(),
|
||||
fetch_error="Gitea credentials unavailable for gitea.prgs.cc",
|
||||
inventory_complete=False,
|
||||
)
|
||||
html = render_traffic_page(snap)
|
||||
self.assertIn("Traffic data unavailable", html)
|
||||
self.assertIn("Fail closed", html)
|
||||
self.assertNotIn("1. Runnable Lanes", html)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
+336
@@ -42,10 +42,25 @@ from webui.lease_loader import load_lease_snapshot, snapshot_to_dict as lease_sn
|
||||
from webui.lease_views import render_leases_page
|
||||
from webui.queue_loader import load_queue_snapshot, snapshot_to_dict as queue_snapshot_to_dict
|
||||
from webui.queue_views import render_queue_page
|
||||
from webui.traffic_loader import load_traffic_snapshot, snapshot_to_dict as traffic_snapshot_to_dict
|
||||
from webui.traffic_views import render_traffic_page
|
||||
from webui.worktree_scanner import load_hygiene_snapshot, snapshot_to_dict as worktree_snapshot_to_dict
|
||||
from webui.worktree_views import render_worktrees_page
|
||||
from webui.runtime_health import load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict
|
||||
import restart_coordinator
|
||||
from webui.restart_console import load_restart_console_snapshot
|
||||
from webui.restart_views import render_restart_console_page
|
||||
from webui.runtime_views import render_runtime_page
|
||||
from webui.session_loader import (
|
||||
load_session_view_snapshot,
|
||||
snapshot_to_dict as session_view_snapshot_to_dict,
|
||||
)
|
||||
from webui.session_views import render_sessions_page
|
||||
from webui.linkage_loader import (
|
||||
load_linkage_snapshot,
|
||||
snapshot_to_dict as linkage_snapshot_to_dict,
|
||||
)
|
||||
from webui.linkage_views import render_linkage_page
|
||||
from webui.inventory import (
|
||||
SECTION_NAMES as _INVENTORY_SECTIONS,
|
||||
load_inventory_snapshot,
|
||||
@@ -65,6 +80,13 @@ from webui.system_health import (
|
||||
snapshot_to_dict as system_health_to_dict,
|
||||
)
|
||||
from webui.system_health_views import render_system_health_page
|
||||
from webui.notifications import (
|
||||
load_notifications_snapshot,
|
||||
snapshot_to_dict as notifications_snapshot_to_dict,
|
||||
)
|
||||
from webui.notification_views import render_notifications_page
|
||||
from webui import request_service
|
||||
from webui.request_views import render_requests_page
|
||||
|
||||
_READ_ONLY_METHODS = frozenset({"GET", "HEAD", "OPTIONS"})
|
||||
_AUDIT_MUTATION_PATHS = frozenset({"/audit", "/api/audit"})
|
||||
@@ -80,10 +102,12 @@ def _stub_page(title: str, description: str) -> HTMLResponse:
|
||||
|
||||
|
||||
_LEGACY_PAGES = (
|
||||
("/traffic", "Traffic", "workflow traffic-control view (#640)"),
|
||||
("/queue", "Queue", "live PR and issue dashboard (#429)"),
|
||||
("/projects", "Projects", "registry and onboarding (#427)"),
|
||||
("/prompts", "Prompts", "canonical workflow prompt library (#428)"),
|
||||
("/runtime", "Runtime", "MCP health and stale-runtime detection (#430)"),
|
||||
("/sessions", "Sessions", "runtime and session view (#641)"),
|
||||
("/audit", "Audit", "final-report paste and validator preview (#431)"),
|
||||
("/worktrees", "Worktrees", "branch hygiene dashboard (#432)"),
|
||||
("/leases", "Leases", "collision and lease visibility (#433)"),
|
||||
@@ -191,6 +215,84 @@ async def system_health(request: Request) -> HTMLResponse:
|
||||
)
|
||||
|
||||
|
||||
async def api_recovery_diagnose(_request: Request) -> JSONResponse:
|
||||
from webui.console_recovery import diagnose_recovery
|
||||
diag = diagnose_recovery()
|
||||
return JSONResponse({
|
||||
"status": diag.status,
|
||||
"clean": diag.clean,
|
||||
"stale_runtime": diag.stale_runtime,
|
||||
"master_parity": diag.master_parity,
|
||||
"stale_binding": diag.stale_binding,
|
||||
"contamination": diag.contamination,
|
||||
"worktree_anomalies": list(diag.worktree_anomalies),
|
||||
"reasons": list(diag.reasons),
|
||||
"playbooks": [
|
||||
{
|
||||
"playbook_id": pb.playbook_id,
|
||||
"label": pb.label,
|
||||
"action_id": pb.action_id,
|
||||
"description": pb.description,
|
||||
"eligible": pb.eligible,
|
||||
"requires_confirmation": pb.requires_confirmation,
|
||||
"reason": pb.reason,
|
||||
"params_schema": pb.params_schema,
|
||||
}
|
||||
for pb in diag.playbooks
|
||||
],
|
||||
})
|
||||
|
||||
|
||||
async def api_recovery_preview(request: Request) -> JSONResponse:
|
||||
from webui.console_recovery import build_recovery_preview
|
||||
body = {}
|
||||
try:
|
||||
body = await request.json()
|
||||
except Exception:
|
||||
pass
|
||||
playbook_id = body.get("playbook_id") or request.query_params.get("playbook_id") or ""
|
||||
target = body.get("target") or request.query_params.get("target")
|
||||
principal = resolve_principal(request.headers)
|
||||
preview = build_recovery_preview(
|
||||
playbook_id=playbook_id,
|
||||
target=target,
|
||||
params=body,
|
||||
principal=principal,
|
||||
)
|
||||
status = 200 if preview.get("playbook_id") else 400
|
||||
return JSONResponse(preview, status_code=status)
|
||||
|
||||
|
||||
async def api_recovery_apply(request: Request) -> JSONResponse:
|
||||
from webui.console_recovery import execute_recovery_playbook
|
||||
body = {}
|
||||
try:
|
||||
body = await request.json()
|
||||
except Exception:
|
||||
pass
|
||||
playbook_id = body.get("playbook_id", "")
|
||||
confirmation = body.get("confirmation", "")
|
||||
target = body.get("target")
|
||||
principal = resolve_principal(request.headers)
|
||||
request_id = getattr(request.state, "request_id", None)
|
||||
result = execute_recovery_playbook(
|
||||
playbook_id=playbook_id,
|
||||
confirmation=confirmation,
|
||||
target=target,
|
||||
params=body,
|
||||
principal=principal,
|
||||
request_id=request_id,
|
||||
)
|
||||
status_code = 200 if result.get("success") else 400
|
||||
return JSONResponse(result, status_code=status_code)
|
||||
|
||||
|
||||
async def api_recovery_verify(_request: Request) -> JSONResponse:
|
||||
from webui.console_recovery import verify_post_recovery
|
||||
verification = verify_post_recovery()
|
||||
return JSONResponse(verification, status_code=200)
|
||||
|
||||
|
||||
async def queue(_request: Request) -> HTMLResponse:
|
||||
snapshot = load_queue_snapshot()
|
||||
return HTMLResponse(render_page(title="Queue", body_html=render_queue_page(snapshot)))
|
||||
@@ -200,6 +302,15 @@ async def api_queue(_request: Request) -> JSONResponse:
|
||||
return JSONResponse(queue_snapshot_to_dict(load_queue_snapshot()))
|
||||
|
||||
|
||||
async def traffic(_request: Request) -> HTMLResponse:
|
||||
snapshot = load_traffic_snapshot()
|
||||
return HTMLResponse(render_traffic_page(snapshot))
|
||||
|
||||
|
||||
async def api_traffic(_request: Request) -> JSONResponse:
|
||||
return JSONResponse(traffic_snapshot_to_dict(load_traffic_snapshot()))
|
||||
|
||||
|
||||
def _load_project_registry() -> tuple[ProjectRegistry | None, RegistryError | None]:
|
||||
"""Load the registry, converting validation failure into a fail-closed pair."""
|
||||
try:
|
||||
@@ -313,6 +424,79 @@ async def api_runtime(_request: Request) -> JSONResponse:
|
||||
return JSONResponse(runtime_snapshot_to_dict(load_runtime_snapshot()))
|
||||
|
||||
|
||||
def _restart_console_snapshot(request: Request):
|
||||
"""Build the read-only restart snapshot for the requesting principal (#667)."""
|
||||
principal = resolve_principal(request.headers)
|
||||
restart_class = (
|
||||
request.query_params.get("restart_class")
|
||||
or restart_coordinator.RestartClass.FULL_MCP_RESTART.value
|
||||
)
|
||||
return load_restart_console_snapshot(
|
||||
principal=principal, restart_class=restart_class
|
||||
)
|
||||
|
||||
|
||||
async def restart_console_page(request: Request) -> HTMLResponse:
|
||||
"""Restart status, impact preview, and approval state (#667). Read-only."""
|
||||
snapshot = _restart_console_snapshot(request)
|
||||
return HTMLResponse(
|
||||
render_page(
|
||||
title="Restart", body_html=render_restart_console_page(snapshot)
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
async def api_restart_status(request: Request) -> JSONResponse:
|
||||
"""JSON export of the read-only restart console snapshot (#667)."""
|
||||
return JSONResponse(_restart_console_snapshot(request).as_dict())
|
||||
|
||||
|
||||
async def sessions(_request: Request) -> HTMLResponse:
|
||||
"""Runtime and session view (#641) — read-only composition of health + inventory."""
|
||||
snapshot = load_session_view_snapshot()
|
||||
return HTMLResponse(render_sessions_page(snapshot))
|
||||
|
||||
|
||||
async def api_sessions(_request: Request) -> JSONResponse:
|
||||
"""JSON export for the runtime/session view (#641)."""
|
||||
return JSONResponse(session_view_snapshot_to_dict(load_session_view_snapshot()))
|
||||
|
||||
|
||||
def _linkage_snapshot(request: Request):
|
||||
"""Load one linkage snapshot from the request's scope and focus parameters."""
|
||||
return load_linkage_snapshot(
|
||||
request.query_params.get("project") or None,
|
||||
state=request.query_params.get("state"),
|
||||
issue=_query_int(request, "issue"),
|
||||
pr=_query_int(request, "pr"),
|
||||
)
|
||||
|
||||
|
||||
async def gitea_linkage(request: Request) -> HTMLResponse:
|
||||
"""Gitea issue↔PR linkage console (#645) — read-only.
|
||||
|
||||
Always 200, including on a failed read: this is an operator view, and it
|
||||
must render *why* linkage could not be loaded rather than withhold the page.
|
||||
The snapshot itself carries ``ok=False`` and the page refuses to draw a
|
||||
linkage table it cannot stand behind.
|
||||
"""
|
||||
return HTMLResponse(render_linkage_page(_linkage_snapshot(request)))
|
||||
|
||||
|
||||
async def api_v1_gitea_linkage(request: Request) -> JSONResponse:
|
||||
"""JSON export of the issue↔PR linkage model (#645).
|
||||
|
||||
Unlike the HTML view, the API answers with 502 when the snapshot could not
|
||||
be loaded, so an automated consumer cannot read a fail-closed payload as a
|
||||
successful "no links exist" result.
|
||||
"""
|
||||
snapshot = _linkage_snapshot(request)
|
||||
return JSONResponse(
|
||||
linkage_snapshot_to_dict(snapshot),
|
||||
status_code=200 if snapshot.ok else 502,
|
||||
)
|
||||
|
||||
|
||||
async def _parse_audit_form(request: Request) -> tuple[str, str | None]:
|
||||
if request.method == "GET":
|
||||
return "", None
|
||||
@@ -710,6 +894,125 @@ async def api_v1_analytics_ingest(request: Request) -> JSONResponse:
|
||||
)
|
||||
|
||||
|
||||
async def notifications_route(request: Request) -> HTMLResponse:
|
||||
project_id = request.query_params.get("project_id")
|
||||
attention_class = request.query_params.get("attention_class") or "inbox"
|
||||
snap = load_notifications_snapshot(project_id)
|
||||
html = render_notifications_page(
|
||||
snap, filter_class=attention_class, filter_project=project_id
|
||||
)
|
||||
return HTMLResponse(html)
|
||||
|
||||
|
||||
async def api_notifications(request: Request) -> JSONResponse:
|
||||
project_id = request.query_params.get("project_id")
|
||||
snap = load_notifications_snapshot(project_id)
|
||||
data = notifications_snapshot_to_dict(snap)
|
||||
return JSONResponse(data)
|
||||
|
||||
def _default_request_scope() -> dict[str, str]:
|
||||
"""Resolve remote/org/repo from the project registry for request forms.
|
||||
|
||||
Returns an empty mapping when the registry cannot be read, which makes
|
||||
``parse_request`` reject a request that did not name its own scope rather
|
||||
than letting it default to some other repository.
|
||||
"""
|
||||
from webui.queue_loader import _host_from_url # host normalisation helper
|
||||
|
||||
registry, error = _load_project_registry()
|
||||
if error is not None or not registry.projects:
|
||||
return {}
|
||||
project = registry.projects[0]
|
||||
host = _host_from_url(project.remote_host)
|
||||
return {
|
||||
"remote": _derive_remote(host),
|
||||
"org": project.gitea_owner or "",
|
||||
"repo": project.repo_name or "",
|
||||
}
|
||||
|
||||
|
||||
async def _request_payload(request: Request) -> dict[str, object]:
|
||||
"""Read a request body as JSON or form-encoded. Never raises."""
|
||||
content_type = (request.headers.get("content-type") or "").lower()
|
||||
if "application/json" in content_type:
|
||||
try:
|
||||
body = await request.json()
|
||||
except Exception:
|
||||
return {}
|
||||
return dict(body) if isinstance(body, dict) else {}
|
||||
try:
|
||||
form = await request.form()
|
||||
except Exception:
|
||||
return {}
|
||||
return {key: form[key] for key in form}
|
||||
|
||||
|
||||
async def requests_page(request: Request) -> HTMLResponse:
|
||||
"""Operator request form and intent preview (#643).
|
||||
|
||||
POST here only ever *previews*. Initiation is a separate confirmed call to
|
||||
``/api/v1/requests/apply`` so that submitting this form cannot reserve
|
||||
work as a side effect.
|
||||
"""
|
||||
submitted: dict[str, object] = {}
|
||||
preview = None
|
||||
error = None
|
||||
if request.method == "POST":
|
||||
submitted = await _request_payload(request)
|
||||
work_request, error = request_service.parse_request(
|
||||
submitted, default_scope=_default_request_scope()
|
||||
)
|
||||
if work_request is not None:
|
||||
preview = request_service.preview_request(
|
||||
work_request,
|
||||
principal=resolve_principal(headers=dict(request.headers)),
|
||||
)
|
||||
return HTMLResponse(
|
||||
render_requests_page(
|
||||
preview=preview, error=error, submitted=submitted
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
async def api_v1_request_preview(request: Request) -> JSONResponse:
|
||||
"""Dry-run authorization and intent preview for a work request (#643)."""
|
||||
payload = await _request_payload(request)
|
||||
work_request, error = request_service.parse_request(
|
||||
payload, default_scope=_default_request_scope()
|
||||
)
|
||||
if work_request is None:
|
||||
return JSONResponse(error.to_dict(), status_code=400)
|
||||
preview = request_service.preview_request(
|
||||
work_request,
|
||||
principal=resolve_principal(headers=dict(request.headers)),
|
||||
)
|
||||
return JSONResponse(
|
||||
preview.to_dict(), status_code=200 if preview.authorized else 403
|
||||
)
|
||||
|
||||
|
||||
async def api_v1_request_apply(request: Request) -> JSONResponse:
|
||||
"""Initiate a previewed work request through the allocator (#643).
|
||||
|
||||
Fail-closed at every step: unauthorized, unconfirmed, not-next-safe, and
|
||||
already-claimed all return without attempting an assignment.
|
||||
"""
|
||||
payload = await _request_payload(request)
|
||||
work_request, error = request_service.parse_request(
|
||||
payload, default_scope=_default_request_scope()
|
||||
)
|
||||
if work_request is None:
|
||||
return JSONResponse(error.to_dict(), status_code=400)
|
||||
confirm = _truthy_flag(str(payload.get("confirm") or ""))
|
||||
result = request_service.apply_request(
|
||||
work_request,
|
||||
principal=resolve_principal(headers=dict(request.headers)),
|
||||
confirm=confirm,
|
||||
)
|
||||
status = int(result.pop("status_code", 403))
|
||||
return JSONResponse(result, status_code=status)
|
||||
|
||||
|
||||
async def method_not_allowed(request: Request, _exc: Exception) -> Response:
|
||||
path = request.url.path
|
||||
if path in _AUDIT_MUTATION_PATHS and request.method == "POST":
|
||||
@@ -736,6 +1039,11 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
Route("/system-health", system_health, methods=["GET"]),
|
||||
Route("/queue", queue, methods=["GET"]),
|
||||
Route("/api/queue", api_queue, methods=["GET"]),
|
||||
Route("/traffic", traffic, methods=["GET"]),
|
||||
Route("/api/traffic", api_traffic, methods=["GET"]),
|
||||
Route("/notifications", notifications_route, methods=["GET"]),
|
||||
Route("/api/notifications", api_notifications, methods=["GET"]),
|
||||
Route("/api/v1/notifications", api_notifications, methods=["GET"]),
|
||||
Route("/projects", projects, methods=["GET"]),
|
||||
Route("/projects/{project_id}", project_detail, methods=["GET"]),
|
||||
Route("/api/projects", api_projects, methods=["GET"]),
|
||||
@@ -750,7 +1058,19 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
Route("/api/prompts", api_prompts, methods=["GET"]),
|
||||
Route("/runtime", runtime, methods=["GET"]),
|
||||
Route("/api/runtime", api_runtime, methods=["GET"]),
|
||||
# #667 read-only restart status / impact preview / approval state.
|
||||
Route("/runtime/restart", restart_console_page, methods=["GET"]),
|
||||
Route(
|
||||
"/api/v1/system/restart/status",
|
||||
api_restart_status,
|
||||
methods=["GET"],
|
||||
),
|
||||
Route("/sessions", sessions, methods=["GET"]),
|
||||
Route("/api/sessions", api_sessions, methods=["GET"]),
|
||||
Route("/api/v1/sessions", api_sessions, methods=["GET"]),
|
||||
Route("/api/v1/timeline", api_v1_timeline, methods=["GET"]),
|
||||
Route("/gitea", gitea_linkage, methods=["GET"]),
|
||||
Route("/api/v1/gitea/linkage", api_v1_gitea_linkage, methods=["GET"]),
|
||||
Route("/analytics", analytics, methods=["GET"]),
|
||||
Route("/api/analytics", api_v1_analytics, methods=["GET"]),
|
||||
Route("/api/v1/analytics", api_v1_analytics, methods=["GET"]),
|
||||
@@ -772,6 +1092,17 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
api_action_attempt,
|
||||
methods=["POST"],
|
||||
),
|
||||
Route("/requests", requests_page, methods=["GET", "POST"]),
|
||||
Route(
|
||||
"/api/v1/requests/preview",
|
||||
api_v1_request_preview,
|
||||
methods=["POST"],
|
||||
),
|
||||
Route(
|
||||
"/api/v1/requests/apply",
|
||||
api_v1_request_apply,
|
||||
methods=["POST"],
|
||||
),
|
||||
Route("/api/leases", api_leases, methods=["GET"]),
|
||||
Route("/api/v1/inventory", api_inventory, methods=["GET"]),
|
||||
Route(
|
||||
@@ -784,6 +1115,11 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
||||
api_console_security_model,
|
||||
methods=["GET"],
|
||||
),
|
||||
# #644 Phase 2 Recovery API routes
|
||||
Route("/api/v1/system/recovery/diagnose", api_recovery_diagnose, methods=["GET"]),
|
||||
Route("/api/v1/system/recovery/preview", api_recovery_preview, methods=["POST", "GET"]),
|
||||
Route("/api/v1/system/recovery/apply", api_recovery_apply, methods=["POST"]),
|
||||
Route("/api/v1/system/recovery/verify", api_recovery_verify, methods=["POST", "GET"]),
|
||||
*[
|
||||
Route(path, phase_stub, methods=["GET"])
|
||||
for path in STUB_PAGES
|
||||
|
||||
+105
-8
@@ -115,6 +115,12 @@ class ConsoleAction:
|
||||
break_glass: bool
|
||||
phase: int
|
||||
summary: str
|
||||
# Opt-in switch for an action whose execution path is genuinely wired
|
||||
# ahead of its phase becoming globally active (#643). Naming a variable
|
||||
# here enables nothing on its own: the variable must also be set in the
|
||||
# environment. An action that leaves this ``None`` can only execute once
|
||||
# ACTIVE_PHASE reaches its phase, exactly as before.
|
||||
execution_env_flag: str | None = None
|
||||
|
||||
@property
|
||||
def mcp_permission(self) -> str:
|
||||
@@ -277,6 +283,61 @@ _ACTION_SPECS: tuple[ConsoleAction, ...] = (
|
||||
phase=2,
|
||||
summary="Restart one MCP namespace via the host supervisor.",
|
||||
),
|
||||
# #644: Phase 2 recovery controls & playbooks.
|
||||
ConsoleAction(
|
||||
action_id="system.clear_stale_binding",
|
||||
task_key="clear_stale_binding",
|
||||
action_class=CLASS_WRITE,
|
||||
minimum_role=OPERATOR,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary="Clear provably stale or superseded GITEA_ACTIVE_WORKTREE binding.",
|
||||
),
|
||||
ConsoleAction(
|
||||
action_id="system.rebind_session_worktree",
|
||||
task_key="rebind_session_worktree",
|
||||
action_class=CLASS_WRITE,
|
||||
minimum_role=OPERATOR,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary="Rebind session worktree context to verified lease worktree.",
|
||||
),
|
||||
ConsoleAction(
|
||||
action_id="system.reconcile_cleanups",
|
||||
task_key="reconcile_cleanups",
|
||||
action_class=CLASS_PRIVILEGED,
|
||||
minimum_role=CONTROLLER,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary="Run reconciler cleanup for merged or superseded PR branches.",
|
||||
),
|
||||
# #643: submit a work request — desired role, issue/PR, intent — and let
|
||||
# the allocator reserve it. This is the one Phase 2 action whose execution
|
||||
# path is actually implemented (``webui.request_service``), so it carries
|
||||
# the opt-in flag; it stays denied until an operator sets that variable.
|
||||
# Authority is operator-class because the outcome is a claim, not a Gitea
|
||||
# verdict: initiating reviewer or merger *work* does not grant the right
|
||||
# to approve or merge, which stays with the MCP role profile.
|
||||
ConsoleAction(
|
||||
action_id="initiate_workflow",
|
||||
task_key="allocate_next_work",
|
||||
action_class=CLASS_WRITE,
|
||||
minimum_role=OPERATOR,
|
||||
requires_confirmation=True,
|
||||
dual_control=False,
|
||||
break_glass=False,
|
||||
phase=2,
|
||||
summary=(
|
||||
"Preview and initiate allocator-owned workflow work for a role."
|
||||
),
|
||||
execution_env_flag="WEBUI_REQUESTS_EXECUTION",
|
||||
),
|
||||
)
|
||||
|
||||
ACTIONS: dict[str, ConsoleAction] = {a.action_id: a for a in _ACTION_SPECS}
|
||||
@@ -430,6 +491,33 @@ ALLOW_PREVIEW = "allowed_preview_only"
|
||||
# gated on this model landing; nothing here enables it.
|
||||
ACTIVE_PHASE = 1
|
||||
|
||||
_TRUTHY = frozenset({"1", "true", "yes", "on"})
|
||||
|
||||
|
||||
def execution_wired(
|
||||
action: ConsoleAction | None, env: dict[str, str] | None = None
|
||||
) -> bool:
|
||||
"""Whether *action* has a live execution path right now.
|
||||
|
||||
Two ways to be wired, and only two. The action's phase is active, or the
|
||||
action declares an opt-in environment variable *and* that variable is set.
|
||||
Everything else — including every action that never declares a flag — is
|
||||
unwired, so the default across the registry stays deny.
|
||||
|
||||
Bumping ``ACTIVE_PHASE`` would enable execution for every action of that
|
||||
phase at once. The per-action flag exists so a single implemented action
|
||||
can go live without dragging its unimplemented phase-mates with it.
|
||||
"""
|
||||
if action is None:
|
||||
return False
|
||||
if action.phase <= ACTIVE_PHASE:
|
||||
return True
|
||||
flag = (action.execution_env_flag or "").strip()
|
||||
if not flag:
|
||||
return False
|
||||
source = env if env is not None else os.environ
|
||||
return (source.get(flag) or "").strip().lower() in _TRUTHY
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class AuthorizationDecision:
|
||||
@@ -469,16 +557,19 @@ def authorize(
|
||||
principal: Principal | None = None,
|
||||
*,
|
||||
for_execution: bool = False,
|
||||
env: dict[str, str] | None = None,
|
||||
) -> AuthorizationDecision:
|
||||
"""Decide whether *principal* may invoke *action_id*. Deny by default.
|
||||
|
||||
``for_execution`` distinguishes a read-only preview from a real invocation.
|
||||
Even an allowed decision reports ``execution_enabled=False`` while the
|
||||
console is in Phase 1, so no caller can read an allow as permission to
|
||||
mutate.
|
||||
``execution_enabled`` reports whether the action has a live execution path
|
||||
at all (:func:`execution_wired`) — for every action without an explicit
|
||||
opt-in flag that stays ``False`` while the console is in Phase 1, so no
|
||||
caller can read an allow as permission to mutate.
|
||||
"""
|
||||
who = principal if principal is not None else ANONYMOUS
|
||||
action = get_action(action_id)
|
||||
wired = execution_wired(action, env)
|
||||
|
||||
if action is None:
|
||||
return AuthorizationDecision(
|
||||
@@ -497,7 +588,7 @@ def authorize(
|
||||
"requires_confirmation": action.requires_confirmation,
|
||||
"dual_control": action.dual_control,
|
||||
"break_glass": action.break_glass,
|
||||
"execution_enabled": False,
|
||||
"execution_enabled": wired,
|
||||
}
|
||||
|
||||
if not who.authenticated:
|
||||
@@ -530,13 +621,19 @@ def authorize(
|
||||
**base,
|
||||
)
|
||||
|
||||
if for_execution and action.phase > ACTIVE_PHASE:
|
||||
if for_execution and not wired:
|
||||
return AuthorizationDecision(
|
||||
allowed=False,
|
||||
reason_code=DENY_PHASE_NOT_ACTIVE,
|
||||
detail=(
|
||||
f"Action {action_id!r} belongs to phase {action.phase}; the "
|
||||
f"console is in phase {ACTIVE_PHASE}. Execution is not wired."
|
||||
f"console is in phase {ACTIVE_PHASE}"
|
||||
+ (
|
||||
f" and {action.execution_env_flag} is not set"
|
||||
if action.execution_env_flag
|
||||
else ""
|
||||
)
|
||||
+ ". Execution is not wired."
|
||||
),
|
||||
**base,
|
||||
)
|
||||
@@ -545,8 +642,8 @@ def authorize(
|
||||
allowed=True,
|
||||
reason_code=ALLOW_PREVIEW,
|
||||
detail=(
|
||||
"Principal holds the required role. Preview only — execution "
|
||||
"remains disabled until the Phase 2 action framework ships."
|
||||
"Principal holds the required role. Execution proceeds only for an "
|
||||
"action with a wired execution path; everything else is preview."
|
||||
),
|
||||
**base,
|
||||
)
|
||||
|
||||
@@ -0,0 +1,721 @@
|
||||
"""Web Console Phase 2 Recovery Controls & Playbooks (#644).
|
||||
|
||||
Provides canonical recovery controls for the web console:
|
||||
1. Diagnosis: Surfaces stale runtimes, worktree binding errors, contamination markers,
|
||||
and un-reconciled cleanups.
|
||||
2. Gated Actions & Playbooks: Guided recovery (rebind session worktree, clear stale
|
||||
binding, trigger reconciler cleanups, sanctioned restart).
|
||||
3. RBAC, Contamination (#630), and Master Parity (#610) integration.
|
||||
4. Audit trail via ``console_audit`` and mandatory post-recovery revalidation.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import master_parity_gate
|
||||
import runtime_recovery_guard
|
||||
import stale_binding_recovery
|
||||
from webui import console_audit, console_authz, sanctioned_restart, system_health, worktree_scanner
|
||||
|
||||
# --- Recovery Statuses ------------------------------------------------------
|
||||
STATUS_HEALTHY = "healthy"
|
||||
STATUS_ACTION_REQUIRED = "action_required"
|
||||
STATUS_BLOCKED_CONTAMINATION = "blocked_contamination"
|
||||
STATUS_RECONNECT_REQUIRED = "reconnect_required"
|
||||
|
||||
# --- Playbook Identifiers ---------------------------------------------------
|
||||
PLAYBOOK_CLEAR_STALE_BINDING = "clear_stale_binding"
|
||||
PLAYBOOK_REBIND_SESSION = "rebind_session_worktree"
|
||||
PLAYBOOK_RECONCILE_CLEANUPS = "reconcile_cleanups"
|
||||
PLAYBOOK_SANCTIONED_RESTART = "sanctioned_restart"
|
||||
|
||||
KNOWN_PLAYBOOKS: tuple[str, ...] = (
|
||||
PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
PLAYBOOK_REBIND_SESSION,
|
||||
PLAYBOOK_RECONCILE_CLEANUPS,
|
||||
PLAYBOOK_SANCTIONED_RESTART,
|
||||
)
|
||||
|
||||
# --- Console Action Mapping -------------------------------------------------
|
||||
ACTION_CLEAR_STALE_BINDING = "system.clear_stale_binding"
|
||||
ACTION_REBIND_SESSION = "system.rebind_session_worktree"
|
||||
ACTION_RECONCILE_CLEANUPS = "system.reconcile_cleanups"
|
||||
|
||||
PLAYBOOK_ACTIONS: dict[str, str] = {
|
||||
PLAYBOOK_CLEAR_STALE_BINDING: ACTION_CLEAR_STALE_BINDING,
|
||||
PLAYBOOK_REBIND_SESSION: ACTION_REBIND_SESSION,
|
||||
PLAYBOOK_RECONCILE_CLEANUPS: ACTION_RECONCILE_CLEANUPS,
|
||||
PLAYBOOK_SANCTIONED_RESTART: sanctioned_restart.ACTION_RESTART_NAMESPACE,
|
||||
}
|
||||
|
||||
#: Task key handed to :func:`runtime_recovery_guard.assess_contamination_gate`.
|
||||
#: A console *action id* is not a task name and is not a member of
|
||||
#: ``CONTAMINATION_GATED_TASKS``, so passing one left the #630 gate inert. Every
|
||||
#: writing recovery playbook shares this one gated task key; the reconciler
|
||||
#: cleanup playbook is exempted separately because it is the designated remedy.
|
||||
CONTAMINATION_GATED_TASK = "console_recovery_apply"
|
||||
|
||||
#: Remote whose contamination markers govern this console. Markers are written
|
||||
#: per remote, so reading the wrong one reports a contaminated runtime clean.
|
||||
REMOTE_ENV = "WEBUI_GITEA_REMOTE"
|
||||
DEFAULT_REMOTE = "prgs"
|
||||
|
||||
|
||||
def _console_remote(env: dict[str, str] | None = None) -> str:
|
||||
env_map = env if env is not None else os.environ
|
||||
return (env_map.get(REMOTE_ENV) or "").strip() or DEFAULT_REMOTE
|
||||
|
||||
|
||||
def load_active_contamination_marker(
|
||||
remote: str | None = None, env: dict[str, str] | None = None
|
||||
) -> dict[str, Any] | None:
|
||||
"""Return the live #630 contamination marker payload, or ``None``.
|
||||
|
||||
The gate is only meaningful when it is fed a real marker: with
|
||||
``marker=None`` :func:`assess_contamination_gate` returns ``block: False``
|
||||
on its first statement. The #641 session inventory already reads the durable
|
||||
markers, so reuse that reader rather than adding a second source of truth.
|
||||
Never raises into a diagnosis or execution path.
|
||||
"""
|
||||
try:
|
||||
from webui import session_loader
|
||||
except Exception: # noqa: BLE001 — never break recovery on an import problem
|
||||
return None
|
||||
try:
|
||||
markers = session_loader._load_contamination_markers(
|
||||
remote=remote or _console_remote(env)
|
||||
)
|
||||
except Exception: # noqa: BLE001 — fail soft; the caller degrades to no marker
|
||||
return None
|
||||
for marker in markers:
|
||||
payload = marker.to_dict()
|
||||
if payload.get("active"):
|
||||
return payload
|
||||
return None
|
||||
|
||||
|
||||
def _active_binding(env_map: Any) -> str | None:
|
||||
"""Read the live worktree binding so a no-op recovery cannot report success."""
|
||||
value = env_map.get(stale_binding_recovery.ACTIVE_WORKTREE_ENV)
|
||||
return value if value else None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RecoveryLedgerEntry:
|
||||
"""One planned recovery step displayed before execution."""
|
||||
|
||||
sequence: int
|
||||
step: str
|
||||
summary: str
|
||||
executes_process_kill: bool = False
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class PlaybookDescriptor:
|
||||
"""Structured recovery playbook option returned during diagnosis."""
|
||||
|
||||
playbook_id: str
|
||||
label: str
|
||||
action_id: str
|
||||
description: str
|
||||
eligible: bool
|
||||
requires_confirmation: bool
|
||||
reason: str
|
||||
params_schema: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RecoveryDiagnosis:
|
||||
"""Complete diagnostic snapshot of control-plane recovery needs."""
|
||||
|
||||
status: str
|
||||
clean: bool
|
||||
stale_runtime: dict[str, Any]
|
||||
master_parity: dict[str, Any]
|
||||
stale_binding: dict[str, Any]
|
||||
contamination: dict[str, Any]
|
||||
worktree_anomalies: tuple[str, ...]
|
||||
playbooks: tuple[PlaybookDescriptor, ...]
|
||||
reasons: tuple[str, ...]
|
||||
|
||||
|
||||
def _repo_root(custom_path: Path | str | None = None) -> Path:
|
||||
if custom_path:
|
||||
return Path(custom_path).resolve()
|
||||
override = (os.environ.get("WEBUI_REPO_ROOT") or "").strip()
|
||||
if override:
|
||||
return Path(override).resolve()
|
||||
return Path(__file__).resolve().parent.parent
|
||||
|
||||
|
||||
def confirmation_phrase(playbook_id: str, target: str | None = None) -> str:
|
||||
"""Construct exact confirmation phrase required for a recovery playbook."""
|
||||
clean_target = (target or "").strip()
|
||||
if clean_target:
|
||||
return f"confirm {playbook_id} {clean_target}"
|
||||
return f"confirm {playbook_id}"
|
||||
|
||||
|
||||
def confirmation_matches(
|
||||
playbook_id: str, confirmation: str | None, target: str | None = None
|
||||
) -> bool:
|
||||
expected = confirmation_phrase(playbook_id, target)
|
||||
return str(confirmation or "").strip() == expected
|
||||
|
||||
|
||||
def _build_ledger(
|
||||
playbook_id: str, target: str | None = None
|
||||
) -> tuple[RecoveryLedgerEntry, ...]:
|
||||
if playbook_id == PLAYBOOK_CLEAR_STALE_BINDING:
|
||||
return (
|
||||
RecoveryLedgerEntry(1, "quiesce", "Stop admitting new gated mutations."),
|
||||
RecoveryLedgerEntry(
|
||||
2,
|
||||
"clear_env",
|
||||
f"Remove stale env binding GITEA_ACTIVE_WORKTREE ({target or 'active'}).",
|
||||
),
|
||||
RecoveryLedgerEntry(
|
||||
3, "audit", "Record clear_stale_binding event in console audit log."
|
||||
),
|
||||
RecoveryLedgerEntry(
|
||||
4, "revalidate", "Re-run diagnosis to verify clean binding state."
|
||||
),
|
||||
)
|
||||
if playbook_id == PLAYBOOK_REBIND_SESSION:
|
||||
return (
|
||||
RecoveryLedgerEntry(1, "quiesce", "Stop admitting new gated mutations."),
|
||||
RecoveryLedgerEntry(
|
||||
2,
|
||||
"rebind_worktree",
|
||||
f"Rebind session worktree context safely to {target or 'target worktree'}.",
|
||||
),
|
||||
RecoveryLedgerEntry(
|
||||
3, "audit", "Record rebind_session_worktree event in console audit log."
|
||||
),
|
||||
RecoveryLedgerEntry(
|
||||
4, "revalidate", "Re-run diagnosis to verify worktree binding state."
|
||||
),
|
||||
)
|
||||
if playbook_id == PLAYBOOK_RECONCILE_CLEANUPS:
|
||||
return (
|
||||
RecoveryLedgerEntry(1, "quiesce", "Stop admitting new gated mutations."),
|
||||
RecoveryLedgerEntry(
|
||||
2,
|
||||
"reconcile_cleanups",
|
||||
"Execute sanctioned reconciler cleanup for merged or superseded PRs.",
|
||||
),
|
||||
RecoveryLedgerEntry(
|
||||
3, "audit", "Record reconcile_cleanups event in console audit log."
|
||||
),
|
||||
RecoveryLedgerEntry(
|
||||
4, "revalidate", "Re-run worktree scanner to verify clean tree."
|
||||
),
|
||||
)
|
||||
if playbook_id == PLAYBOOK_SANCTIONED_RESTART:
|
||||
restart_ledger = sanctioned_restart._mutation_ledger(
|
||||
target or "gitea-author", sanctioned_restart.MODE_RESTART
|
||||
)
|
||||
return tuple(
|
||||
RecoveryLedgerEntry(
|
||||
sequence=e.sequence,
|
||||
step=e.step,
|
||||
summary=e.summary,
|
||||
executes_process_kill=e.executes_process_kill,
|
||||
)
|
||||
for e in restart_ledger
|
||||
)
|
||||
return (
|
||||
RecoveryLedgerEntry(1, "unspecified", f"Execute recovery playbook {playbook_id}."),
|
||||
)
|
||||
|
||||
|
||||
def diagnose_recovery(
|
||||
repo_path: Path | str | None = None,
|
||||
env: dict[str, str] | None = None,
|
||||
active_worktree_val: str | None = None,
|
||||
session_lease_wt: str | None = None,
|
||||
role_kind: str | None = None,
|
||||
) -> RecoveryDiagnosis:
|
||||
"""Run full control-plane diagnostics to determine recovery needs and options."""
|
||||
root = _repo_root(repo_path)
|
||||
source_env = dict(env) if env is not None else dict(os.environ)
|
||||
reasons: list[str] = []
|
||||
|
||||
# 1. Stale runtime assessment
|
||||
stale_runtime_obj = system_health.assess_stale_runtime(root)
|
||||
stale_runtime_dict = {
|
||||
"daemon_head": stale_runtime_obj.daemon_head,
|
||||
"checkout_head": stale_runtime_obj.checkout_head,
|
||||
"remote_head": stale_runtime_obj.remote_head,
|
||||
"stale": stale_runtime_obj.stale,
|
||||
"determinable": stale_runtime_obj.determinable,
|
||||
"mutation_safe": stale_runtime_obj.mutation_safe,
|
||||
"reasons": list(stale_runtime_obj.reasons),
|
||||
}
|
||||
if stale_runtime_obj.stale:
|
||||
reasons.append("Runtime HEAD disagrees with checkout/remote HEAD.")
|
||||
|
||||
# 2. Master parity assessment
|
||||
#
|
||||
# The baseline is the commit the *running process* started at, which is what
|
||||
# the parity gate is about. Capturing it from ``checkout_head`` and then
|
||||
# comparing it against that same value made ``in_parity`` structurally
|
||||
# incapable of being false. ``live_remote_head`` restores the #610
|
||||
# live-remote dimension, which was previously dropped.
|
||||
checkout_head = stale_runtime_obj.checkout_head
|
||||
startup_dict = master_parity_gate.capture_startup_parity(
|
||||
str(root), head=stale_runtime_obj.daemon_head
|
||||
)
|
||||
parity_dict = master_parity_gate.assess_master_parity(
|
||||
startup_dict, checkout_head, stale_runtime_obj.remote_head
|
||||
)
|
||||
if not parity_dict.get("in_parity", True):
|
||||
reasons.append("Repository is not in master parity.")
|
||||
|
||||
# 3. Worktree binding classification
|
||||
boot_bindings = stale_binding_recovery.snapshot_boot_bindings(source_env)
|
||||
active_val = (
|
||||
active_worktree_val
|
||||
if active_worktree_val is not None
|
||||
else source_env.get(stale_binding_recovery.ACTIVE_WORKTREE_ENV)
|
||||
)
|
||||
boot_inherited = bool(boot_bindings.get("active_worktree") and active_val == boot_bindings.get("active_worktree"))
|
||||
|
||||
path_exists = None
|
||||
if active_val:
|
||||
path_exists = os.path.exists(os.path.realpath(active_val))
|
||||
|
||||
binding_class = stale_binding_recovery.classify_active_worktree_binding(
|
||||
active_value=active_val,
|
||||
session_lease_worktree=session_lease_wt,
|
||||
boot_inherited=boot_inherited,
|
||||
path_exists=path_exists,
|
||||
role_kind=role_kind,
|
||||
)
|
||||
|
||||
if binding_class.get("clear_eligible"):
|
||||
reasons.append(
|
||||
f"Active worktree binding is stale ({binding_class.get('classification')})."
|
||||
)
|
||||
elif binding_class.get("classification") == stale_binding_recovery.CLASSIFICATION_UNVERIFIED_INHERITED:
|
||||
reasons.append("Inherited worktree binding is unverified.")
|
||||
|
||||
# 4. Contamination assessment (#630)
|
||||
#
|
||||
# A real marker and a task key the gate actually gates on: with marker=None
|
||||
# the gate short-circuits to ``block: False``, and with a console action id
|
||||
# the task is outside CONTAMINATION_GATED_TASKS, so it could never block.
|
||||
contamination_marker = load_active_contamination_marker(env=source_env)
|
||||
contamination_dict = runtime_recovery_guard.assess_contamination_gate(
|
||||
contamination_marker,
|
||||
task=CONTAMINATION_GATED_TASK,
|
||||
actual_role=role_kind,
|
||||
)
|
||||
contaminated = bool(contamination_dict.get("block"))
|
||||
if contaminated:
|
||||
reasons.append("Runtime is contaminated by manual process kill (#630).")
|
||||
|
||||
# 5. Worktree scanner hygiene & anomalies
|
||||
hygiene = worktree_scanner.load_hygiene_snapshot(project_root=str(root))
|
||||
worktree_anomalies = hygiene.anomalies
|
||||
|
||||
# Determine status & eligible playbooks
|
||||
playbooks: list[PlaybookDescriptor] = []
|
||||
|
||||
# Playbook 1: Clear Stale Binding
|
||||
clear_eligible = bool(binding_class.get("clear_eligible"))
|
||||
playbooks.append(
|
||||
PlaybookDescriptor(
|
||||
playbook_id=PLAYBOOK_CLEAR_STALE_BINDING,
|
||||
label="Clear Stale Worktree Binding",
|
||||
action_id=ACTION_CLEAR_STALE_BINDING,
|
||||
description="Clear provably stale or superseded GITEA_ACTIVE_WORKTREE environment binding.",
|
||||
eligible=clear_eligible,
|
||||
requires_confirmation=True,
|
||||
reason=(
|
||||
f"Binding classified as {binding_class.get('classification')}; clear is authorized."
|
||||
if clear_eligible
|
||||
else "Active worktree binding is clean, corroborated, or absent."
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
# Playbook 2: Rebind Session Worktree
|
||||
rebind_eligible = bool(
|
||||
active_val
|
||||
or binding_class.get("classification") == stale_binding_recovery.CLASSIFICATION_UNVERIFIED_INHERITED
|
||||
)
|
||||
playbooks.append(
|
||||
PlaybookDescriptor(
|
||||
playbook_id=PLAYBOOK_REBIND_SESSION,
|
||||
label="Rebind Session Worktree",
|
||||
action_id=ACTION_REBIND_SESSION,
|
||||
description="Rebind or synchronize session worktree binding safely with active lease.",
|
||||
eligible=rebind_eligible,
|
||||
requires_confirmation=True,
|
||||
reason=(
|
||||
"Session worktree binding can be rebound to verified lease worktree."
|
||||
if rebind_eligible
|
||||
else "Session worktree is properly bound."
|
||||
),
|
||||
params_schema={"target_worktree": "string"},
|
||||
)
|
||||
)
|
||||
|
||||
# Playbook 3: Reconcile Cleanups
|
||||
reconcile_eligible = bool(hygiene.anomalies or any(e.classification in {"stale-clean", "detached-review"} for e in hygiene.entries))
|
||||
playbooks.append(
|
||||
PlaybookDescriptor(
|
||||
playbook_id=PLAYBOOK_RECONCILE_CLEANUPS,
|
||||
label="Trigger Reconciler Cleanups",
|
||||
action_id=ACTION_RECONCILE_CLEANUPS,
|
||||
description="Run sanctioned reconciler cleanup preview and apply for merged/superseded PR branches.",
|
||||
eligible=reconcile_eligible,
|
||||
requires_confirmation=True,
|
||||
reason=(
|
||||
f"Worktree hygiene scanner detected {len(hygiene.anomalies)} anomalies and cleanups needed."
|
||||
if reconcile_eligible
|
||||
else "No reconciler cleanups pending."
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
# Playbook 4: Sanctioned Restart
|
||||
restart_eligible = bool(stale_runtime_obj.stale or contaminated)
|
||||
playbooks.append(
|
||||
PlaybookDescriptor(
|
||||
playbook_id=PLAYBOOK_SANCTIONED_RESTART,
|
||||
label="Sanctioned MCP Restart",
|
||||
action_id=sanctioned_restart.ACTION_RESTART_NAMESPACE,
|
||||
description="Restart MCP daemon via configured host supervisor without manual process kill.",
|
||||
eligible=restart_eligible,
|
||||
requires_confirmation=True,
|
||||
reason=(
|
||||
"Stale runtime or contamination detected; host supervisor restart available."
|
||||
if restart_eligible
|
||||
else "Runtime is healthy and clean."
|
||||
),
|
||||
params_schema={"namespace": "string", "mode": "restart|reload"},
|
||||
)
|
||||
)
|
||||
|
||||
clean = not reasons and not contaminated
|
||||
if contaminated:
|
||||
status = STATUS_BLOCKED_CONTAMINATION
|
||||
elif reasons:
|
||||
status = STATUS_ACTION_REQUIRED
|
||||
else:
|
||||
status = STATUS_HEALTHY
|
||||
|
||||
return RecoveryDiagnosis(
|
||||
status=status,
|
||||
clean=clean,
|
||||
stale_runtime=stale_runtime_dict,
|
||||
master_parity=parity_dict,
|
||||
stale_binding=binding_class,
|
||||
contamination=contamination_dict,
|
||||
worktree_anomalies=tuple(worktree_anomalies),
|
||||
playbooks=tuple(playbooks),
|
||||
reasons=tuple(reasons),
|
||||
)
|
||||
|
||||
|
||||
def build_recovery_preview(
|
||||
playbook_id: str,
|
||||
target: str | None = None,
|
||||
params: dict[str, Any] | None = None,
|
||||
principal: console_authz.Principal | None = None,
|
||||
env: dict[str, str] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Generate dry-run preview & mutation ledger for a recovery playbook."""
|
||||
if playbook_id not in KNOWN_PLAYBOOKS:
|
||||
return {
|
||||
"allowed": False,
|
||||
"error": "unknown_playbook",
|
||||
"detail": f"Playbook {playbook_id!r} is not a registered recovery playbook.",
|
||||
}
|
||||
|
||||
action_id = PLAYBOOK_ACTIONS[playbook_id]
|
||||
action = console_authz.get_action(action_id)
|
||||
decision = console_authz.authorize(action_id, principal)
|
||||
# Preview and apply must answer the same question. ``execution_enabled`` was
|
||||
# a hardcoded False beside an authorization decision taken without
|
||||
# ``for_execution``, so the preview could not tell an operator *why*
|
||||
# execution was disabled — and the apply path did not ask at all.
|
||||
execution_decision = console_authz.authorize(
|
||||
action_id, principal, for_execution=True
|
||||
)
|
||||
phrase = confirmation_phrase(playbook_id, target)
|
||||
ledger = _build_ledger(playbook_id, target)
|
||||
|
||||
return {
|
||||
"playbook_id": playbook_id,
|
||||
"action_id": action_id,
|
||||
"target": target,
|
||||
"required_role": action.minimum_role if action else console_authz.OPERATOR,
|
||||
"required_permission": action.mcp_permission if action else "gitea.read",
|
||||
"requires_confirmation": True,
|
||||
"confirmation_phrase": phrase,
|
||||
"mutation_ledger": [asdict(entry) for entry in ledger],
|
||||
"authorization": decision.to_dict(),
|
||||
"execution_authorization": execution_decision.to_dict(),
|
||||
"params": dict(params or {}),
|
||||
"execution_enabled": bool(execution_decision.allowed),
|
||||
"execution_blocked_reason": (
|
||||
None if execution_decision.allowed else execution_decision.reason_code
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def execute_recovery_playbook(
|
||||
playbook_id: str,
|
||||
confirmation: str | None = None,
|
||||
target: str | None = None,
|
||||
params: dict[str, Any] | None = None,
|
||||
principal: console_authz.Principal | None = None,
|
||||
env: dict[str, str] | None = None,
|
||||
request_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Gated execution of a recovery playbook with audit logging and revalidation."""
|
||||
if playbook_id not in KNOWN_PLAYBOOKS:
|
||||
return {
|
||||
"success": False,
|
||||
"allowed": False,
|
||||
"error": "unknown_playbook",
|
||||
"detail": f"Playbook {playbook_id!r} is not known.",
|
||||
}
|
||||
|
||||
action_id = PLAYBOOK_ACTIONS[playbook_id]
|
||||
# The mapping the playbooks actually mutate. ``dict(os.environ)`` produced a
|
||||
# throwaway copy: every env playbook wrote to it, verified against it, and
|
||||
# left the running daemon bound to the value it claimed to have fixed.
|
||||
mutation_env: Any = env if env is not None else os.environ
|
||||
|
||||
# 1. Authorization check — ``for_execution=True`` is what arms the phase
|
||||
# gate (console_authz.authorize only applies it in that branch). Without it
|
||||
# a phase-2 write executed while the console was in phase 1.
|
||||
decision = console_authz.authorize(action_id, principal, for_execution=True)
|
||||
if not decision.allowed:
|
||||
console_audit.record_event(
|
||||
action_id=action_id,
|
||||
result=console_audit.RESULT_DENIED,
|
||||
principal=principal,
|
||||
target={"playbook_id": playbook_id, "target": target},
|
||||
reason_code=decision.reason_code,
|
||||
detail=decision.detail,
|
||||
request_id=request_id,
|
||||
session_id=session_id,
|
||||
)
|
||||
return {
|
||||
"success": False,
|
||||
"allowed": False,
|
||||
"error": decision.reason_code,
|
||||
"detail": decision.detail,
|
||||
}
|
||||
|
||||
# 2. Confirmation phrase check
|
||||
if not confirmation_matches(playbook_id, confirmation, target):
|
||||
expected = confirmation_phrase(playbook_id, target)
|
||||
detail = f"Confirmation phrase mismatch. Expected: {expected!r}"
|
||||
console_audit.record_event(
|
||||
action_id=action_id,
|
||||
result=console_audit.RESULT_DENIED,
|
||||
principal=principal,
|
||||
target={"playbook_id": playbook_id, "target": target},
|
||||
reason_code="confirmation_mismatch",
|
||||
detail=detail,
|
||||
request_id=request_id,
|
||||
session_id=session_id,
|
||||
)
|
||||
return {
|
||||
"success": False,
|
||||
"allowed": False,
|
||||
"error": "confirmation_mismatch",
|
||||
"detail": detail,
|
||||
"expected_confirmation_phrase": expected,
|
||||
}
|
||||
|
||||
# 3. Contamination rule (#630) check
|
||||
role_str = principal.role if principal else None
|
||||
contamination_marker = load_active_contamination_marker(env=env)
|
||||
contam = runtime_recovery_guard.assess_contamination_gate(
|
||||
contamination_marker,
|
||||
task=CONTAMINATION_GATED_TASK,
|
||||
actual_role=role_str,
|
||||
)
|
||||
if contam.get("block"):
|
||||
if playbook_id != PLAYBOOK_RECONCILE_CLEANUPS:
|
||||
detail = "Runtime is contaminated by a manual process kill (#630). Run reconciler cleanup playbook first."
|
||||
console_audit.record_event(
|
||||
action_id=action_id,
|
||||
result=console_audit.RESULT_DENIED,
|
||||
principal=principal,
|
||||
target={"playbook_id": playbook_id, "target": target},
|
||||
reason_code="contaminated_runtime",
|
||||
detail=detail,
|
||||
request_id=request_id,
|
||||
session_id=session_id,
|
||||
)
|
||||
return {
|
||||
"success": False,
|
||||
"allowed": False,
|
||||
"error": "contaminated_runtime",
|
||||
"detail": detail,
|
||||
}
|
||||
|
||||
# 4. Execute playbook action
|
||||
applied_result: dict[str, Any] = {"performed": False}
|
||||
if playbook_id == PLAYBOOK_CLEAR_STALE_BINDING:
|
||||
binding_before = _active_binding(mutation_env)
|
||||
diagnosis = diagnose_recovery(env=env)
|
||||
plan = stale_binding_recovery.plan_recovery(diagnosis.stale_binding)
|
||||
applied_result = stale_binding_recovery.apply_recovery(plan, env=mutation_env)
|
||||
binding_after = _active_binding(mutation_env)
|
||||
applied_result = {
|
||||
**applied_result,
|
||||
"binding_before": binding_before,
|
||||
"binding_after": binding_after,
|
||||
"binding_changed": binding_before != binding_after,
|
||||
}
|
||||
# A clear that did not clear is not a success, whatever the plan said.
|
||||
if not applied_result["binding_changed"]:
|
||||
applied_result["performed"] = False
|
||||
applied_result.setdefault("reasons", []).append(
|
||||
"clear_stale_binding did not change the live worktree binding"
|
||||
)
|
||||
elif playbook_id == PLAYBOOK_REBIND_SESSION:
|
||||
target_wt = target or (params or {}).get("target_worktree")
|
||||
if target_wt:
|
||||
binding_before = _active_binding(mutation_env)
|
||||
mutation_env[stale_binding_recovery.ACTIVE_WORKTREE_ENV] = target_wt
|
||||
binding_after = _active_binding(mutation_env)
|
||||
applied_result = {
|
||||
"performed": binding_after == target_wt,
|
||||
"rebound_worktree": target_wt,
|
||||
"binding_before": binding_before,
|
||||
"binding_after": binding_after,
|
||||
"binding_changed": binding_before != binding_after,
|
||||
"cleared_stale": binding_before != binding_after,
|
||||
}
|
||||
if binding_after != target_wt:
|
||||
applied_result["reasons"] = [
|
||||
"rebind_session_worktree did not take effect on the live "
|
||||
"environment"
|
||||
]
|
||||
else:
|
||||
applied_result = {
|
||||
"performed": False,
|
||||
"reason": "No target_worktree specified for rebind.",
|
||||
}
|
||||
elif playbook_id == PLAYBOOK_RECONCILE_CLEANUPS:
|
||||
# ``merged_cleanup_reconcile`` exposes the building blocks only; the
|
||||
# orchestrator is the MCP tool. The previous call named a function that
|
||||
# does not exist, and a bare ``except`` turned the AttributeError into a
|
||||
# generic failure, so this playbook could never succeed. Imported lazily
|
||||
# because the MCP server module is large and binds FastMCP at import.
|
||||
try:
|
||||
import gitea_mcp_server
|
||||
|
||||
snapshot = gitea_mcp_server.gitea_reconcile_merged_cleanups(
|
||||
dry_run=False,
|
||||
execute_confirmed=True,
|
||||
remote=(params or {}).get("remote") or _console_remote(env),
|
||||
org=(params or {}).get("org"),
|
||||
repo=(params or {}).get("repo"),
|
||||
)
|
||||
performed_reconcile = bool(snapshot.get("success"))
|
||||
applied_result = {
|
||||
"performed": performed_reconcile,
|
||||
"reconciled_count": len(snapshot.get("entries") or []),
|
||||
"snapshot": snapshot,
|
||||
}
|
||||
if not performed_reconcile:
|
||||
applied_result["reasons"] = list(snapshot.get("reasons") or [])
|
||||
except Exception as exc: # noqa: BLE001 — surfaced with its type
|
||||
applied_result = {
|
||||
"performed": False,
|
||||
"error": str(exc),
|
||||
"error_type": type(exc).__name__,
|
||||
}
|
||||
elif playbook_id == PLAYBOOK_SANCTIONED_RESTART:
|
||||
ns = target or (params or {}).get("namespace", "gitea-author")
|
||||
md = (params or {}).get("mode", sanctioned_restart.MODE_RESTART)
|
||||
restart_res = sanctioned_restart.execute_restart(
|
||||
namespace=ns,
|
||||
mode=md,
|
||||
principal=principal,
|
||||
confirmation=f"{md} {ns}",
|
||||
# Without the marker the stricter guard at sanctioned_restart.py:375
|
||||
# never fires and a restart can launder a contaminated runtime.
|
||||
contamination_marker=contamination_marker,
|
||||
env=mutation_env,
|
||||
request_id=request_id,
|
||||
session_id=session_id,
|
||||
)
|
||||
applied_result = restart_res
|
||||
|
||||
# ``allowed`` is not ``performed``: execute_restart documents that success is
|
||||
# False in both directions because the host supervisor still has to act.
|
||||
performed = bool(applied_result.get("performed"))
|
||||
|
||||
# 5. Record Audit Log
|
||||
audit_record = console_audit.record_event(
|
||||
action_id=action_id,
|
||||
result=console_audit.RESULT_ALLOWED if performed else console_audit.RESULT_DENIED,
|
||||
principal=principal,
|
||||
target={"playbook_id": playbook_id, "target": target},
|
||||
reason_code="recovery_executed" if performed else "recovery_failed",
|
||||
detail=f"Executed recovery playbook {playbook_id}",
|
||||
request_id=request_id,
|
||||
session_id=session_id,
|
||||
metadata={"applied_result": applied_result},
|
||||
)
|
||||
|
||||
# 6. Post-recovery verification recheck.
|
||||
#
|
||||
# Re-read state rather than re-reading the mapping the mutation just wrote:
|
||||
# verifying the mutated copy confirmed changes that never reached the
|
||||
# process. Passing ``env`` through means a caller-supplied mapping is the
|
||||
# live one for that caller, and ``None`` re-reads ``os.environ`` fresh.
|
||||
post_verification = verify_post_recovery(env=env)
|
||||
|
||||
return {
|
||||
"success": performed,
|
||||
"allowed": True,
|
||||
"playbook_id": playbook_id,
|
||||
"action_id": action_id,
|
||||
"applied_result": applied_result,
|
||||
"audit": audit_record,
|
||||
"post_recovery_verification": post_verification,
|
||||
}
|
||||
|
||||
|
||||
def verify_post_recovery(
|
||||
repo_path: Path | str | None = None, env: dict[str, str] | None = None
|
||||
) -> dict[str, Any]:
|
||||
"""Revalidate control-plane state post-recovery before clean status."""
|
||||
diag = diagnose_recovery(repo_path, env)
|
||||
classification = diag.stale_binding.get("classification")
|
||||
# ``not clear_eligible`` also reads clean for every binding recovery is not
|
||||
# allowed to touch — an unverified inherited binding is unproven, not clean.
|
||||
binding_clean = (
|
||||
not diag.stale_binding.get("clear_eligible")
|
||||
and classification != stale_binding_recovery.CLASSIFICATION_UNVERIFIED_INHERITED
|
||||
)
|
||||
return {
|
||||
"clean": diag.clean,
|
||||
"status": diag.status,
|
||||
"stale_runtime_clean": not diag.stale_runtime.get("stale"),
|
||||
"binding_clean": binding_clean,
|
||||
"binding_classification": classification,
|
||||
# The gate returns ``block``; it has never returned ``contaminated``, so
|
||||
# reading that key reported every runtime clean unconditionally.
|
||||
"contamination_clean": not diag.contamination.get("block"),
|
||||
"anomalies_count": len(diag.worktree_anomalies),
|
||||
"reasons": list(diag.reasons),
|
||||
}
|
||||
@@ -178,6 +178,13 @@ def build_action_registry() -> ActionRegistry:
|
||||
("system.restart_namespace", "Restart MCP namespace",
|
||||
"restart_namespace", "host.supervisor_restart",
|
||||
"Restart one MCP namespace via the host supervisor."),
|
||||
# #644: Phase 2 recovery playbooks & controls.
|
||||
("system.clear_stale_binding", "Clear stale binding", "clear_stale_binding",
|
||||
"console.clear_stale_binding", "Clear provably stale or superseded env binding."),
|
||||
("system.rebind_session_worktree", "Rebind session worktree", "rebind_session_worktree",
|
||||
"console.rebind_session_worktree", "Rebind session worktree to verified lease."),
|
||||
("system.reconcile_cleanups", "Reconcile cleanups", "reconcile_cleanups",
|
||||
"console.reconcile_cleanups", "Run reconciler cleanup for merged or superseded PRs."),
|
||||
)
|
||||
actions = tuple(
|
||||
GatedAction(
|
||||
|
||||
@@ -76,6 +76,35 @@ _CREDENTIAL_KEY_RE = re.compile(
|
||||
)
|
||||
_REDACTED = "[redacted]"
|
||||
|
||||
#: Credential-shaped *name* as it appears inside a free-form command line. This
|
||||
#: is deliberately broader than :data:`_CREDENTIAL_KEY_RE` — it also matches a
|
||||
#: bare ``key`` component, so ``PRIVATE_KEY=`` is caught. Over-redacting a
|
||||
#: displayed string is safe; under-redacting one is not.
|
||||
_TEXT_CREDENTIAL_NAME = (
|
||||
r"[A-Za-z0-9_.\-]*"
|
||||
r"(?:token|secret|password|passwd|key|authorization|bearer|credential)"
|
||||
r"[A-Za-z0-9_.\-]*"
|
||||
)
|
||||
#: A value following such a name: single-quoted, double-quoted, or bare. The
|
||||
#: bare form stops at a quote so an enclosing quote survives the redaction.
|
||||
_TEXT_CREDENTIAL_VALUE = r"'[^']*'|\"[^\"]*\"|[^\s'\"]+"
|
||||
|
||||
_TEXT_CREDENTIAL_FLAG_RE = re.compile(
|
||||
rf"(?P<key>(?<![\w\-])--?{_TEXT_CREDENTIAL_NAME})"
|
||||
rf"(?P<sep>[=\s]+)"
|
||||
rf"(?P<value>{_TEXT_CREDENTIAL_VALUE})",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_TEXT_CREDENTIAL_ASSIGN_RE = re.compile(
|
||||
rf"(?P<key>(?<![\w\-]){_TEXT_CREDENTIAL_NAME})"
|
||||
rf"(?P<sep>\s*[:=]\s*)"
|
||||
rf"(?P<value>{_TEXT_CREDENTIAL_VALUE})",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
_TEXT_URL_USERINFO_RE = re.compile(
|
||||
r"(?P<scheme>\b[A-Za-z][A-Za-z0-9+.\-]*://)[^\s/@]+@"
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class InventorySection:
|
||||
@@ -225,6 +254,42 @@ def scrub(value: Any, *, key: str | None = None) -> Any:
|
||||
return repr(value)
|
||||
|
||||
|
||||
def collapse_home(text: str) -> str:
|
||||
"""Collapse every ``$HOME`` occurrence *inside* a string, not just a prefix."""
|
||||
home = os.path.expanduser("~")
|
||||
if not home or home == "/":
|
||||
return text
|
||||
return text.replace(home, "~")
|
||||
|
||||
|
||||
def scrub_text(value: Any) -> Any:
|
||||
"""Redact a free-form text blob such as a recorded command line.
|
||||
|
||||
:func:`scrub` keys off structured field *names* and whole-value prefixes,
|
||||
which is right for inventory records but blind to a secret embedded in the
|
||||
middle of a sentence. This collapses ``$HOME`` and redacts credential-shaped
|
||||
tokens and URL userinfo *anywhere* in the string, so operator-supplied text
|
||||
rendered verbatim — contamination ``command_summary`` (#630) above all — is
|
||||
held to the same standard as every other field on the page.
|
||||
|
||||
Returns ``None`` unchanged so callers can keep "absent" distinct from "".
|
||||
"""
|
||||
if value is None:
|
||||
return None
|
||||
text = value if isinstance(value, str) else str(value)
|
||||
text = collapse_home(text)
|
||||
text = _TEXT_URL_USERINFO_RE.sub(
|
||||
lambda m: f"{m.group('scheme')}{_REDACTED}@", text
|
||||
)
|
||||
text = _TEXT_CREDENTIAL_FLAG_RE.sub(
|
||||
lambda m: f"{m.group('key')}{m.group('sep')}{_REDACTED}", text
|
||||
)
|
||||
text = _TEXT_CREDENTIAL_ASSIGN_RE.sub(
|
||||
lambda m: f"{m.group('key')}{m.group('sep')}{_REDACTED}", text
|
||||
)
|
||||
return text
|
||||
|
||||
|
||||
# ── control-plane database (read-only) ───────────────────────────────────────
|
||||
|
||||
|
||||
|
||||
@@ -182,10 +182,17 @@ def _extract_reviewer_leases(
|
||||
parsed = parse_reviewer_lease_comment(comment.get("body") or "")
|
||||
if not parsed:
|
||||
continue
|
||||
subject_pr = parsed.get("pr_number") or pr_number
|
||||
leases.append(
|
||||
{
|
||||
**parsed,
|
||||
"pr_number": parsed.get("pr_number") or pr_number,
|
||||
"pr_number": subject_pr,
|
||||
# The lease subject is the PR, never the linked issue: a
|
||||
# reviewer lease on PR #N must not be attributed to issue #N
|
||||
# or to the issue that PR closes (#640).
|
||||
"kind": "pr",
|
||||
"number": subject_pr,
|
||||
"role": "reviewer",
|
||||
"comment_id": comment.get("id"),
|
||||
"author": (comment.get("user") or {}).get("login"),
|
||||
"created_at": comment.get("created_at"),
|
||||
|
||||
@@ -0,0 +1,771 @@
|
||||
"""Gitea issue↔PR linkage model for the console (#645, Phase 3).
|
||||
|
||||
Operators lose context between an issue and the PR that closes it: which PR
|
||||
carries which issue, whether two PRs claim the same issue, and what the latest
|
||||
canonical handoff on that thread said. The evidence exists in Gitea, but only
|
||||
as free text scattered across PR titles, bodies, and branch names.
|
||||
|
||||
This module resolves that linkage into one read-only model:
|
||||
|
||||
* :func:`resolve_linkage` is a pure function from raw Gitea issue/PR payloads to
|
||||
a :class:`LinkageIndex`. It records *how* each edge was found (a ``Closes #N``
|
||||
keyword, the canonical ``feat/issue-N-…`` branch marker, or a bare ``#N``
|
||||
body reference) and never collapses several candidates into one silent guess.
|
||||
* :func:`load_linkage_snapshot` scopes that index to a registry project and
|
||||
optionally attaches the latest Canonical Thread Handoff (CTH) summary for one
|
||||
focused issue or PR.
|
||||
|
||||
Design rules, matching the rest of the console:
|
||||
|
||||
- **Read-only.** Gitea is read through the shared authenticated helpers. No
|
||||
endpoint here mutates anything, and no write action is registered.
|
||||
- **Qualified absence.** Linkage is a claim about a *loaded* window of Gitea.
|
||||
When pagination did not complete, when credentials were unavailable, or when
|
||||
only open items were fetched, the snapshot says so and every "no linked PR"
|
||||
is marked non-authoritative. An empty edge list from a partial read is not
|
||||
evidence that no link exists.
|
||||
- **Handoff is loaded, never assumed.** CTH comments are thread-scoped, so they
|
||||
are fetched only for an explicitly focused issue or PR. Every other row
|
||||
reports ``not_loaded`` rather than rendering as "no handoff".
|
||||
- **Redaction at the boundary.** Titles, labels, handoff fields, and error
|
||||
reasons are free text from Gitea and cross :mod:`webui.console_redaction`
|
||||
before they leave this module.
|
||||
- **Deep links are opt-in.** A link to the Gitea web UI is emitted only under
|
||||
the ``GITEA_MCP_REVEAL_ENDPOINTS`` admin opt-in, exactly as the MCP tools
|
||||
gate their own URL exposure.
|
||||
|
||||
Non-goals (from the issue): no issue/PR editor, no browser review or merge, no
|
||||
reimplementation of Gitea search.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Callable, Iterable, Sequence
|
||||
|
||||
from gitea_auth import api_fetch_page, get_auth_header, gitea_url, repo_api_url
|
||||
|
||||
from webui import console_redaction
|
||||
from webui.project_registry import ProjectRecord, load_registry
|
||||
from webui.queue_loader import (
|
||||
PaginationMeta,
|
||||
_fetch_issues,
|
||||
_fetch_prs,
|
||||
_host_from_url,
|
||||
)
|
||||
|
||||
#: Version of the serialized linkage contract. Bump on any breaking change.
|
||||
LINKAGE_SCHEMA_VERSION = 1
|
||||
|
||||
# --- Linkage evidence -------------------------------------------------------
|
||||
# Ordered strongest to weakest. The strength ordering is what makes an
|
||||
# ambiguous PR detectable: two candidates at the same strength are a genuine
|
||||
# ambiguity, while a weaker candidate alongside a stronger one is not.
|
||||
EVIDENCE_CLOSES = "closes_keyword"
|
||||
EVIDENCE_BRANCH = "branch_marker"
|
||||
EVIDENCE_REFERENCE = "body_reference"
|
||||
|
||||
EVIDENCE_ORDER: tuple[str, ...] = (
|
||||
EVIDENCE_CLOSES,
|
||||
EVIDENCE_BRANCH,
|
||||
EVIDENCE_REFERENCE,
|
||||
)
|
||||
_EVIDENCE_RANK = {name: rank for rank, name in enumerate(EVIDENCE_ORDER)}
|
||||
|
||||
EVIDENCE_DESCRIPTIONS: dict[str, str] = {
|
||||
EVIDENCE_CLOSES: (
|
||||
"the PR title or body declares 'closes/fixes/resolves #N' — Gitea itself "
|
||||
"acts on this keyword, so it is the strongest available evidence"
|
||||
),
|
||||
EVIDENCE_BRANCH: (
|
||||
"the PR head branch carries the canonical issue marker "
|
||||
"'(fix|feat|docs|chore)/issue-N-…' minted by the issue lock"
|
||||
),
|
||||
EVIDENCE_REFERENCE: (
|
||||
"the PR body mentions '#N' without a closing keyword; a mention is not "
|
||||
"a claim that the PR closes that issue"
|
||||
),
|
||||
}
|
||||
|
||||
_CLOSES_RE = re.compile(r"(?:closes|fixes|resolves)\s+#(\d+)", re.IGNORECASE)
|
||||
_REFERENCE_RE = re.compile(r"#(\d+)")
|
||||
_BRANCH_MARKER_RE = re.compile(
|
||||
r"^(?:fix|feat|docs|chore)/issue-(\d+)(?:[-/]|$)", re.IGNORECASE
|
||||
)
|
||||
|
||||
# Handoff-source states. ``not_loaded`` is deliberately distinct from "none
|
||||
# found": a row whose comments were never fetched proves nothing about whether
|
||||
# a handoff exists on that thread.
|
||||
HANDOFF_NOT_LOADED = "not_loaded"
|
||||
HANDOFF_LOADED = "loaded"
|
||||
HANDOFF_UNAVAILABLE = "unavailable"
|
||||
|
||||
# Which item states were fetched. Linkage claims are scoped to this window.
|
||||
STATE_OPEN = "open"
|
||||
STATE_ALL = "all"
|
||||
_SUPPORTED_STATES = (STATE_OPEN, STATE_ALL)
|
||||
|
||||
|
||||
def _redact(value: Any) -> Any:
|
||||
"""Redact one free-text field, failing closed to the placeholder."""
|
||||
if value is None:
|
||||
return None
|
||||
return console_redaction.redact_text(str(value))
|
||||
|
||||
|
||||
def deep_links_enabled(env: dict[str, str] | None = None) -> bool:
|
||||
"""Whether Gitea web-UI deep links may be emitted (admin/debug opt-in)."""
|
||||
source = env if env is not None else os.environ
|
||||
return (source.get("GITEA_MCP_REVEAL_ENDPOINTS") or "").strip().lower() in {
|
||||
"1",
|
||||
"true",
|
||||
"yes",
|
||||
"on",
|
||||
}
|
||||
|
||||
|
||||
def _deep_link(host: str, org: str, repo: str, kind: str, number: int) -> str | None:
|
||||
"""Build a Gitea web link for one item, or None when reveal is not enabled."""
|
||||
if not deep_links_enabled() or not (host and org and repo):
|
||||
return None
|
||||
segment = "pulls" if kind == "pr" else "issues"
|
||||
try:
|
||||
return gitea_url(host, f"/{org}/{repo}/{segment}/{int(number)}")
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
# --- Pure linkage resolution -------------------------------------------------
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class IssueLink:
|
||||
"""One resolved edge from a PR to an issue, with the evidence that found it."""
|
||||
|
||||
issue_number: int
|
||||
evidence: tuple[str, ...]
|
||||
|
||||
@property
|
||||
def strength(self) -> int:
|
||||
"""Rank of the strongest evidence backing this edge (lower is stronger)."""
|
||||
return min(
|
||||
(_EVIDENCE_RANK.get(name, len(EVIDENCE_ORDER)) for name in self.evidence),
|
||||
default=len(EVIDENCE_ORDER),
|
||||
)
|
||||
|
||||
@property
|
||||
def closes(self) -> bool:
|
||||
"""True only when the PR *declares* it closes the issue."""
|
||||
return EVIDENCE_CLOSES in self.evidence
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"issue_number": self.issue_number,
|
||||
"evidence": list(self.evidence),
|
||||
"closes": self.closes,
|
||||
}
|
||||
|
||||
|
||||
def _sorted_links(links: Iterable[IssueLink]) -> tuple[IssueLink, ...]:
|
||||
return tuple(sorted(links, key=lambda link: (link.strength, link.issue_number)))
|
||||
|
||||
|
||||
def resolve_pr_links(pr: dict[str, Any]) -> tuple[IssueLink, ...]:
|
||||
"""Resolve every issue a PR points at, strongest evidence first.
|
||||
|
||||
Every candidate is kept. Collapsing to a single "linked issue" is what makes
|
||||
a mislinked or double-claimed PR invisible, so the caller decides what to do
|
||||
with several candidates rather than being handed one guess.
|
||||
"""
|
||||
found: dict[int, set[str]] = {}
|
||||
|
||||
def _add(number: Any, evidence: str) -> None:
|
||||
try:
|
||||
issue_number = int(number)
|
||||
except (TypeError, ValueError):
|
||||
return
|
||||
if issue_number <= 0:
|
||||
return
|
||||
found.setdefault(issue_number, set()).add(evidence)
|
||||
|
||||
title = str(pr.get("title") or "")
|
||||
body = str(pr.get("body") or "")
|
||||
for text in (title, body):
|
||||
for match in _CLOSES_RE.finditer(text):
|
||||
_add(match.group(1), EVIDENCE_CLOSES)
|
||||
|
||||
head_ref = str((pr.get("head") or {}).get("ref") or "")
|
||||
branch_match = _BRANCH_MARKER_RE.match(head_ref.strip())
|
||||
if branch_match:
|
||||
_add(branch_match.group(1), EVIDENCE_BRANCH)
|
||||
|
||||
# The ``#N`` inside "Closes #N" is the *same* textual occurrence as the
|
||||
# closing keyword, not a second, independent mention. Blanking the closing
|
||||
# phrases first keeps "mention" meaning what the legend says it means: a
|
||||
# reference the PR made without claiming to close anything.
|
||||
for match in _REFERENCE_RE.finditer(_CLOSES_RE.sub(" ", body)):
|
||||
_add(match.group(1), EVIDENCE_REFERENCE)
|
||||
|
||||
# A PR's own number appearing in its body is self-reference, not linkage.
|
||||
try:
|
||||
found.pop(int(pr.get("number")), None)
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
|
||||
return _sorted_links(
|
||||
IssueLink(
|
||||
issue_number=number,
|
||||
evidence=tuple(name for name in EVIDENCE_ORDER if name in evidence),
|
||||
)
|
||||
for number, evidence in found.items()
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class LinkageIndex:
|
||||
"""Resolved linkage over one loaded window of issues and PRs."""
|
||||
|
||||
pr_links: dict[int, tuple[IssueLink, ...]]
|
||||
issue_prs: dict[int, tuple[int, ...]]
|
||||
|
||||
def primary_issue(self, pr_number: int) -> IssueLink | None:
|
||||
"""The strongest edge for a PR, or None when it points at no issue."""
|
||||
links = self.pr_links.get(int(pr_number)) or ()
|
||||
return links[0] if links else None
|
||||
|
||||
def ambiguous(self, pr_number: int) -> bool:
|
||||
"""True when two or more issues tie at the PR's strongest evidence."""
|
||||
links = self.pr_links.get(int(pr_number)) or ()
|
||||
if len(links) < 2:
|
||||
return False
|
||||
best = links[0].strength
|
||||
return sum(1 for link in links if link.strength == best) > 1
|
||||
|
||||
def contested_issues(self) -> tuple[int, ...]:
|
||||
"""Issues claimed by more than one PR in the loaded window."""
|
||||
return tuple(
|
||||
number for number, prs in sorted(self.issue_prs.items()) if len(prs) > 1
|
||||
)
|
||||
|
||||
|
||||
def resolve_linkage(prs: Sequence[dict[str, Any]]) -> LinkageIndex:
|
||||
"""Build the bidirectional linkage index for a loaded window of PRs.
|
||||
|
||||
Only *closing* and *branch-marker* edges populate the issue→PR direction: a
|
||||
bare ``#N`` mention is a reference, and treating it as "this PR is the work
|
||||
for issue N" would invent contested issues out of ordinary cross-links. The
|
||||
weaker edge stays visible on the PR→issue side, where it is labelled.
|
||||
"""
|
||||
pr_links: dict[int, tuple[IssueLink, ...]] = {}
|
||||
issue_prs: dict[int, list[int]] = {}
|
||||
for pr in prs or []:
|
||||
try:
|
||||
pr_number = int(pr["number"])
|
||||
except (KeyError, TypeError, ValueError):
|
||||
continue
|
||||
links = resolve_pr_links(pr)
|
||||
pr_links[pr_number] = links
|
||||
for link in links:
|
||||
if link.evidence == (EVIDENCE_REFERENCE,):
|
||||
continue
|
||||
bucket = issue_prs.setdefault(link.issue_number, [])
|
||||
if pr_number not in bucket:
|
||||
bucket.append(pr_number)
|
||||
return LinkageIndex(
|
||||
pr_links=pr_links,
|
||||
issue_prs={number: tuple(sorted(items)) for number, items in issue_prs.items()},
|
||||
)
|
||||
|
||||
|
||||
# --- Canonical handoff summary ----------------------------------------------
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class HandoffSummary:
|
||||
"""The latest CTH comment on one thread, redacted for display."""
|
||||
|
||||
comment_id: int | None
|
||||
created_at: str | None
|
||||
author: str | None
|
||||
cth_type: str
|
||||
cth_type_known: bool
|
||||
status: str | None
|
||||
next_owner: str | None
|
||||
current_blocker: str | None
|
||||
decision: str | None
|
||||
next_action: str | None
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"comment_id": self.comment_id,
|
||||
"created_at": self.created_at,
|
||||
"author": self.author,
|
||||
"cth_type": self.cth_type,
|
||||
"cth_type_known": self.cth_type_known,
|
||||
"status": self.status,
|
||||
"next_owner": self.next_owner,
|
||||
"current_blocker": self.current_blocker,
|
||||
"decision": self.decision,
|
||||
"next_action": self.next_action,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class HandoffStatus:
|
||||
"""Why a thread's handoff summary is present, absent, or unknown."""
|
||||
|
||||
state: str
|
||||
reason: str | None = None
|
||||
target: str | None = None
|
||||
|
||||
@property
|
||||
def loaded(self) -> bool:
|
||||
return self.state == HANDOFF_LOADED
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {"state": self.state, "reason": self.reason, "target": self.target}
|
||||
|
||||
|
||||
def summarize_handoff(comments: Sequence[dict[str, Any]]) -> HandoffSummary | None:
|
||||
"""Summarize the newest CTH comment in *comments*, or None when there is none.
|
||||
|
||||
Every field is redacted before it is returned: a handoff body is operator
|
||||
free text that regularly quotes commands, and it is rendered verbatim on the
|
||||
page this feeds.
|
||||
"""
|
||||
from canonical_thread_handoff import find_latest_cth, is_known_cth_type
|
||||
|
||||
try:
|
||||
latest = find_latest_cth(list(comments or []))
|
||||
except Exception:
|
||||
return None
|
||||
if not latest:
|
||||
return None
|
||||
fields = latest.get("fields") or {}
|
||||
cth_type = str(latest.get("cth_type") or "").strip()
|
||||
known = is_known_cth_type(cth_type)
|
||||
try:
|
||||
comment_id: int | None = int(latest.get("comment_id"))
|
||||
except (TypeError, ValueError):
|
||||
comment_id = None
|
||||
return HandoffSummary(
|
||||
comment_id=comment_id,
|
||||
created_at=_redact(latest.get("created_at")),
|
||||
author=_redact(latest.get("author")),
|
||||
# An unrecognised heading is reported as such rather than republished:
|
||||
# the heading is free text, and CTH_TYPES is the only authority for what
|
||||
# a handoff type may be.
|
||||
cth_type=cth_type if known else "unrecognized",
|
||||
cth_type_known=known,
|
||||
status=_redact(fields.get("status")),
|
||||
next_owner=_redact(fields.get("next owner")),
|
||||
current_blocker=_redact(fields.get("current blocker")),
|
||||
decision=_redact(fields.get("decision")),
|
||||
next_action=_redact(fields.get("next action")),
|
||||
)
|
||||
|
||||
|
||||
CommentSource = Callable[[str, int], list[dict[str, Any]]]
|
||||
|
||||
|
||||
def build_comment_source(host: str, org: str, repo: str) -> CommentSource | None:
|
||||
"""Build an authenticated ``(kind, number) -> comments`` fetcher, or None.
|
||||
|
||||
Returns None when the console is running in offline test mode or when no
|
||||
credential is available for *host*, so the caller reports the handoff source
|
||||
as unavailable instead of as an empty thread.
|
||||
"""
|
||||
if _offline_test_mode() or not (host and org and repo):
|
||||
return None
|
||||
auth = get_auth_header(host)
|
||||
if not auth:
|
||||
return None
|
||||
|
||||
def _fetch(kind: str, number: int) -> list[dict[str, Any]]:
|
||||
segment = "pulls" if kind == "pr" else "issues"
|
||||
url = f"{repo_api_url(host, org, repo)}/{segment}/{int(number)}/comments"
|
||||
comments: list[dict[str, Any]] = []
|
||||
page = 1
|
||||
while page <= 20:
|
||||
raw, meta = api_fetch_page(url, auth, page=page, limit=50)
|
||||
comments.extend(raw)
|
||||
if bool(meta["is_final_page"]):
|
||||
break
|
||||
page += 1
|
||||
return comments
|
||||
|
||||
return _fetch
|
||||
|
||||
|
||||
# --- Snapshot ----------------------------------------------------------------
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class LinkageNode:
|
||||
"""One issue or PR row with its resolved links and display metadata."""
|
||||
|
||||
kind: str
|
||||
number: int
|
||||
title: str
|
||||
state: str
|
||||
labels: tuple[str, ...] = ()
|
||||
links: tuple[IssueLink, ...] = ()
|
||||
linked_prs: tuple[int, ...] = ()
|
||||
ambiguous: bool = False
|
||||
contested: bool = False
|
||||
deep_link: str | None = None
|
||||
handoff: HandoffSummary | None = None
|
||||
handoff_status: HandoffStatus = HandoffStatus(HANDOFF_NOT_LOADED)
|
||||
links_authoritative: bool = True
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"kind": self.kind,
|
||||
"number": self.number,
|
||||
"title": self.title,
|
||||
"state": self.state,
|
||||
"labels": list(self.labels),
|
||||
"links": [link.to_dict() for link in self.links],
|
||||
"linked_prs": list(self.linked_prs),
|
||||
"ambiguous": self.ambiguous,
|
||||
"contested": self.contested,
|
||||
"deep_link": self.deep_link,
|
||||
"links_authoritative": self.links_authoritative,
|
||||
"handoff": self.handoff.to_dict() if self.handoff else None,
|
||||
"handoff_status": self.handoff_status.to_dict(),
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class LinkageSnapshot:
|
||||
"""One answered linkage query over a scoped window of a Gitea repo."""
|
||||
|
||||
ok: bool
|
||||
project_id: str
|
||||
repo_label: str
|
||||
host: str
|
||||
state_scope: str
|
||||
issues: tuple[LinkageNode, ...] = ()
|
||||
prs: tuple[LinkageNode, ...] = ()
|
||||
contested_issues: tuple[int, ...] = ()
|
||||
focus: tuple[str, int] | None = None
|
||||
inventory_complete: bool = False
|
||||
deep_links_enabled: bool = False
|
||||
handoff_status: HandoffStatus = HandoffStatus(HANDOFF_NOT_LOADED)
|
||||
fetch_error: str | None = None
|
||||
|
||||
@property
|
||||
def orphan_prs(self) -> tuple[LinkageNode, ...]:
|
||||
"""PRs in the loaded window that point at no issue at all."""
|
||||
return tuple(node for node in self.prs if not node.links)
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"ok": self.ok,
|
||||
"schema_version": LINKAGE_SCHEMA_VERSION,
|
||||
"project_id": self.project_id,
|
||||
"repo": self.repo_label,
|
||||
"state_scope": self.state_scope,
|
||||
"inventory_complete": self.inventory_complete,
|
||||
"deep_links_enabled": self.deep_links_enabled,
|
||||
"focus": (
|
||||
None
|
||||
if self.focus is None
|
||||
else {"kind": self.focus[0], "number": self.focus[1]}
|
||||
),
|
||||
"handoff_source": self.handoff_status.to_dict(),
|
||||
"fetch_error": self.fetch_error,
|
||||
"contested_issues": list(self.contested_issues),
|
||||
"issues": [node.to_dict() for node in self.issues],
|
||||
"prs": [node.to_dict() for node in self.prs],
|
||||
"evidence_kinds": [
|
||||
{"name": name, "description": EVIDENCE_DESCRIPTIONS[name]}
|
||||
for name in EVIDENCE_ORDER
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def snapshot_to_dict(snapshot: LinkageSnapshot) -> dict[str, Any]:
|
||||
"""JSON-serializable export for ``/api/v1/gitea/linkage``."""
|
||||
return snapshot.to_dict()
|
||||
|
||||
|
||||
def _offline_test_mode() -> bool:
|
||||
return (os.environ.get("WEBUI_TEST_OFFLINE") or "").strip().lower() in {
|
||||
"1",
|
||||
"true",
|
||||
"yes",
|
||||
}
|
||||
|
||||
|
||||
def _labels_of(item: dict[str, Any]) -> tuple[str, ...]:
|
||||
return tuple(
|
||||
str(_redact(label.get("name")))
|
||||
for label in (item.get("labels") or [])
|
||||
if label.get("name")
|
||||
)
|
||||
|
||||
|
||||
def _failed_snapshot(
|
||||
*,
|
||||
project_id: str,
|
||||
repo_label: str,
|
||||
host: str,
|
||||
state_scope: str,
|
||||
reason: str,
|
||||
) -> LinkageSnapshot:
|
||||
"""A read that could not be answered. Never an empty-and-healthy snapshot."""
|
||||
return LinkageSnapshot(
|
||||
ok=False,
|
||||
project_id=project_id,
|
||||
repo_label=repo_label,
|
||||
host=host,
|
||||
state_scope=state_scope,
|
||||
inventory_complete=False,
|
||||
deep_links_enabled=deep_links_enabled(),
|
||||
handoff_status=HandoffStatus(
|
||||
HANDOFF_UNAVAILABLE, reason="linkage inventory could not be loaded"
|
||||
),
|
||||
fetch_error=str(_redact(reason)),
|
||||
)
|
||||
|
||||
|
||||
def _resolve_project(project_id: str | None) -> ProjectRecord | None:
|
||||
registry = load_registry()
|
||||
if project_id:
|
||||
for entry in registry.projects:
|
||||
if entry.id == project_id:
|
||||
return entry
|
||||
return None
|
||||
return registry.projects[0] if registry.projects else None
|
||||
|
||||
|
||||
def _normalize_state(state: str | None) -> str:
|
||||
text = (state or STATE_OPEN).strip().lower()
|
||||
return text if text in _SUPPORTED_STATES else STATE_OPEN
|
||||
|
||||
|
||||
def load_linkage_snapshot(
|
||||
project_id: str | None = None,
|
||||
*,
|
||||
state: str | None = None,
|
||||
issue: int | None = None,
|
||||
pr: int | None = None,
|
||||
fetch_prs: Callable[..., tuple[list[dict], PaginationMeta]] | None = None,
|
||||
fetch_issues: Callable[..., tuple[list[dict], PaginationMeta]] | None = None,
|
||||
comment_source: CommentSource | None = None,
|
||||
) -> LinkageSnapshot:
|
||||
"""Load issue↔PR linkage for a registry project.
|
||||
|
||||
``issue``/``pr`` focus one thread: the focused row is the only one whose
|
||||
Canonical Thread Handoff comments are fetched, because handoff comments are
|
||||
thread-scoped and loading them for a whole queue would be one request per
|
||||
row. Every unfocused row reports its handoff as ``not_loaded``.
|
||||
"""
|
||||
state_scope = _normalize_state(state)
|
||||
try:
|
||||
project = _resolve_project(project_id)
|
||||
except Exception as exc: # registry invalid — fail closed with the reason
|
||||
return _failed_snapshot(
|
||||
project_id=project_id or "",
|
||||
repo_label="",
|
||||
host="",
|
||||
state_scope=state_scope,
|
||||
reason=f"project registry unavailable: {exc}",
|
||||
)
|
||||
|
||||
if project is None:
|
||||
return _failed_snapshot(
|
||||
project_id=project_id or "",
|
||||
repo_label="",
|
||||
host="",
|
||||
state_scope=state_scope,
|
||||
reason=(
|
||||
f"project {project_id!r} not found in registry"
|
||||
if project_id
|
||||
else "no projects registered"
|
||||
),
|
||||
)
|
||||
|
||||
host = _host_from_url(project.remote_host)
|
||||
repo_label = f"{project.gitea_owner}/{project.repo_name}"
|
||||
offline_test = _offline_test_mode()
|
||||
|
||||
def _empty_fetch(*_args, **_kwargs):
|
||||
return [], PaginationMeta(
|
||||
page=1,
|
||||
per_page=50,
|
||||
returned_count=0,
|
||||
has_more=False,
|
||||
is_final_page=True,
|
||||
# An offline stub loaded nothing; claiming a complete inventory here
|
||||
# would let the page assert that no issue has a linked PR.
|
||||
inventory_complete=False,
|
||||
pages_fetched=0,
|
||||
)
|
||||
|
||||
pr_fetch = fetch_prs or (_empty_fetch if offline_test else _fetch_prs)
|
||||
issue_fetch = fetch_issues or (_empty_fetch if offline_test else _fetch_issues)
|
||||
using_live_fetch = not offline_test and (fetch_prs is None or fetch_issues is None)
|
||||
auth = get_auth_header(host) if using_live_fetch else "test-auth"
|
||||
if using_live_fetch and not auth:
|
||||
return _failed_snapshot(
|
||||
project_id=project.id,
|
||||
repo_label=repo_label,
|
||||
host=host,
|
||||
state_scope=state_scope,
|
||||
reason=(
|
||||
f"Gitea credentials unavailable for {host}; linkage cannot be "
|
||||
"loaded (fail closed — not rendering an empty linkage table)"
|
||||
),
|
||||
)
|
||||
|
||||
try:
|
||||
raw_prs, pr_pagination = pr_fetch(
|
||||
host, project.gitea_owner, project.repo_name, auth, state=state_scope
|
||||
)
|
||||
raw_issues, issue_pagination = issue_fetch(
|
||||
host, project.gitea_owner, project.repo_name, auth, state=state_scope
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001 — operator-visible fetch failure
|
||||
return _failed_snapshot(
|
||||
project_id=project.id,
|
||||
repo_label=repo_label,
|
||||
host=host,
|
||||
state_scope=state_scope,
|
||||
reason=f"Gitea fetch failed: {exc}",
|
||||
)
|
||||
|
||||
inventory_complete = bool(
|
||||
getattr(pr_pagination, "inventory_complete", False)
|
||||
and getattr(issue_pagination, "inventory_complete", False)
|
||||
)
|
||||
|
||||
index = resolve_linkage(raw_prs)
|
||||
contested = index.contested_issues()
|
||||
|
||||
focus: tuple[str, int] | None = None
|
||||
if pr is not None:
|
||||
focus = ("pr", int(pr))
|
||||
elif issue is not None:
|
||||
focus = ("issue", int(issue))
|
||||
|
||||
unfocused_reason = (
|
||||
"canonical handoff comments are thread-scoped; focus one issue or PR "
|
||||
"to load its latest handoff"
|
||||
)
|
||||
handoff_status = HandoffStatus(HANDOFF_NOT_LOADED, reason=unfocused_reason)
|
||||
focus_handoff: HandoffSummary | None = None
|
||||
if focus is not None:
|
||||
source = comment_source
|
||||
if source is None and not offline_test:
|
||||
source = build_comment_source(host, project.gitea_owner, project.repo_name)
|
||||
target = f"{focus[0]}#{focus[1]}"
|
||||
if source is None:
|
||||
handoff_status = HandoffStatus(
|
||||
HANDOFF_UNAVAILABLE,
|
||||
reason="no authenticated comment source available for this read",
|
||||
target=target,
|
||||
)
|
||||
else:
|
||||
try:
|
||||
focus_handoff = summarize_handoff(source(focus[0], focus[1]) or [])
|
||||
handoff_status = HandoffStatus(HANDOFF_LOADED, target=target)
|
||||
except Exception as exc: # fail soft: degrade this source only
|
||||
handoff_status = HandoffStatus(
|
||||
HANDOFF_UNAVAILABLE,
|
||||
reason=str(_redact(f"handoff fetch failed: {exc}")),
|
||||
target=target,
|
||||
)
|
||||
|
||||
def _node_handoff(
|
||||
kind: str, number: int
|
||||
) -> tuple[HandoffSummary | None, HandoffStatus]:
|
||||
"""Attach the handoff only to the focused row; qualify every other row."""
|
||||
if focus == (kind, number):
|
||||
return (focus_handoff, handoff_status)
|
||||
return (
|
||||
None,
|
||||
HandoffStatus(
|
||||
HANDOFF_NOT_LOADED,
|
||||
reason=unfocused_reason if focus is None else "not the focused thread",
|
||||
),
|
||||
)
|
||||
|
||||
def _number_of(raw: dict[str, Any]) -> int | None:
|
||||
try:
|
||||
return int(raw["number"])
|
||||
except (KeyError, TypeError, ValueError):
|
||||
return None
|
||||
|
||||
def _sort_key(raw: dict[str, Any]) -> int:
|
||||
number = _number_of(raw)
|
||||
return -1 if number is None else number
|
||||
|
||||
issue_nodes: list[LinkageNode] = []
|
||||
for raw in sorted(raw_issues or [], key=_sort_key, reverse=True):
|
||||
number = _number_of(raw)
|
||||
if number is None:
|
||||
continue
|
||||
node_handoff, node_status = _node_handoff("issue", number)
|
||||
linked_prs = index.issue_prs.get(number, ())
|
||||
issue_nodes.append(
|
||||
LinkageNode(
|
||||
kind="issue",
|
||||
number=number,
|
||||
title=str(_redact(raw.get("title")) or ""),
|
||||
state=str(raw.get("state") or ""),
|
||||
labels=_labels_of(raw),
|
||||
linked_prs=linked_prs,
|
||||
contested=len(linked_prs) > 1,
|
||||
deep_link=_deep_link(
|
||||
host, project.gitea_owner, project.repo_name, "issue", number
|
||||
),
|
||||
handoff=node_handoff,
|
||||
handoff_status=node_status,
|
||||
links_authoritative=inventory_complete,
|
||||
)
|
||||
)
|
||||
|
||||
pr_nodes: list[LinkageNode] = []
|
||||
for raw in sorted(raw_prs or [], key=_sort_key, reverse=True):
|
||||
number = _number_of(raw)
|
||||
if number is None:
|
||||
continue
|
||||
node_handoff, node_status = _node_handoff("pr", number)
|
||||
links = index.pr_links.get(number, ())
|
||||
pr_nodes.append(
|
||||
LinkageNode(
|
||||
kind="pr",
|
||||
number=number,
|
||||
title=str(_redact(raw.get("title")) or ""),
|
||||
state=str(raw.get("state") or ""),
|
||||
labels=_labels_of(raw),
|
||||
links=links,
|
||||
ambiguous=index.ambiguous(number),
|
||||
contested=any(link.issue_number in contested for link in links),
|
||||
deep_link=_deep_link(
|
||||
host, project.gitea_owner, project.repo_name, "pr", number
|
||||
),
|
||||
handoff=node_handoff,
|
||||
handoff_status=node_status,
|
||||
links_authoritative=inventory_complete,
|
||||
)
|
||||
)
|
||||
|
||||
return LinkageSnapshot(
|
||||
ok=True,
|
||||
project_id=project.id,
|
||||
repo_label=repo_label,
|
||||
host=host,
|
||||
state_scope=state_scope,
|
||||
issues=tuple(issue_nodes),
|
||||
prs=tuple(pr_nodes),
|
||||
contested_issues=contested,
|
||||
focus=focus,
|
||||
inventory_complete=inventory_complete,
|
||||
deep_links_enabled=deep_links_enabled(),
|
||||
handoff_status=handoff_status,
|
||||
)
|
||||
@@ -0,0 +1,364 @@
|
||||
"""HTML views for the Gitea issue↔PR linkage console (#645, Phase 3).
|
||||
|
||||
Read-only renderer over :mod:`webui.linkage_loader`. The page's job is to make
|
||||
three things impossible to misread:
|
||||
|
||||
* **why** an edge exists — every link carries its evidence badge, so a bare
|
||||
``#N`` mention never looks like a closing claim;
|
||||
* **what was not loaded** — a partial inventory, an unfocused thread, or an
|
||||
unavailable handoff source renders as an explicit qualifier, never as an
|
||||
affirmative "none";
|
||||
* **that nothing here mutates** — there is no review, merge, or edit control,
|
||||
and the deep link out to Gitea appears only under the admin reveal opt-in.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from html import escape
|
||||
from typing import Sequence
|
||||
|
||||
from webui.layout import render_page
|
||||
from webui.linkage_loader import (
|
||||
EVIDENCE_BRANCH,
|
||||
EVIDENCE_CLOSES,
|
||||
EVIDENCE_DESCRIPTIONS,
|
||||
EVIDENCE_ORDER,
|
||||
EVIDENCE_REFERENCE,
|
||||
HANDOFF_LOADED,
|
||||
HANDOFF_NOT_LOADED,
|
||||
HandoffSummary,
|
||||
LinkageNode,
|
||||
LinkageSnapshot,
|
||||
)
|
||||
|
||||
_EVIDENCE_CSS = {
|
||||
EVIDENCE_CLOSES: "badge-health-ok",
|
||||
EVIDENCE_BRANCH: "badge-health-skipped",
|
||||
EVIDENCE_REFERENCE: "badge-health-unproven",
|
||||
}
|
||||
|
||||
_EVIDENCE_LABEL = {
|
||||
EVIDENCE_CLOSES: "closes",
|
||||
EVIDENCE_BRANCH: "branch",
|
||||
EVIDENCE_REFERENCE: "mention",
|
||||
}
|
||||
|
||||
|
||||
def _badge(text: str, css: str) -> str:
|
||||
return f'<span class="badge {css}">{escape(text)}</span>'
|
||||
|
||||
|
||||
def _labels(names: Sequence[str]) -> str:
|
||||
if not names:
|
||||
return '<span class="muted">—</span>'
|
||||
return " ".join(_badge(name, "badge-health-skipped") for name in names)
|
||||
|
||||
|
||||
def _ref(node: LinkageNode) -> str:
|
||||
"""Render an item reference, hyperlinked only when deep links are revealed."""
|
||||
label = f"#{node.number}"
|
||||
if node.deep_link:
|
||||
return f'<a href="{escape(node.deep_link)}"><code>{escape(label)}</code></a>'
|
||||
return f"<code>{escape(label)}</code>"
|
||||
|
||||
|
||||
def _scope_card(snapshot: LinkageSnapshot) -> str:
|
||||
focus = (
|
||||
"none"
|
||||
if snapshot.focus is None
|
||||
else f"{snapshot.focus[0]}#{snapshot.focus[1]}"
|
||||
)
|
||||
completeness = (
|
||||
_badge("complete", "badge-health-ok")
|
||||
if snapshot.inventory_complete
|
||||
else _badge("partial", "badge-health-degraded")
|
||||
)
|
||||
links_note = (
|
||||
"Every linkage edge below is a claim about this loaded window only."
|
||||
if snapshot.inventory_complete
|
||||
else (
|
||||
"Pagination did not complete for this window, so an empty link list "
|
||||
"means <em>none found in what was loaded</em> — not that no link exists."
|
||||
)
|
||||
)
|
||||
deep_links = (
|
||||
_badge("enabled", "badge-health-ok")
|
||||
if snapshot.deep_links_enabled
|
||||
else _badge("hidden", "badge-health-skipped")
|
||||
)
|
||||
return f"""<div class="health-card">
|
||||
<h3>Scope</h3>
|
||||
<table class="detail">
|
||||
<tr><th>Project</th><td><code>{escape(snapshot.project_id or "—")}</code></td></tr>
|
||||
<tr><th>Repository</th><td><code>{escape(snapshot.repo_label or "—")}</code></td></tr>
|
||||
<tr><th>Item state</th><td><code>{escape(snapshot.state_scope)}</code></td></tr>
|
||||
<tr><th>Focused thread</th><td><code>{escape(focus)}</code></td></tr>
|
||||
<tr><th>Inventory</th><td>{completeness}</td></tr>
|
||||
<tr><th>Gitea deep links</th><td>{deep_links}</td></tr>
|
||||
</table>
|
||||
<p class="muted">{links_note}</p>
|
||||
</div>"""
|
||||
|
||||
|
||||
def _error_card(snapshot: LinkageSnapshot) -> str:
|
||||
if snapshot.ok and not snapshot.fetch_error:
|
||||
return ""
|
||||
return (
|
||||
'<div class="health-card health-stale"><strong>Linkage unavailable:</strong> '
|
||||
f"{escape(snapshot.fetch_error or 'the linkage read did not complete')}. "
|
||||
"No linkage table is rendered: an empty table would read as "
|
||||
"<em>no issue is linked to any PR</em>, which this read cannot claim."
|
||||
"</div>"
|
||||
)
|
||||
|
||||
|
||||
def _contested_card(snapshot: LinkageSnapshot) -> str:
|
||||
if not snapshot.contested_issues:
|
||||
return ""
|
||||
refs = ", ".join(f"<code>#{number}</code>" for number in snapshot.contested_issues)
|
||||
return (
|
||||
'<div class="health-card health-stale">'
|
||||
f"<strong>Contested issues:</strong> {refs}. More than one PR in this "
|
||||
"window claims each of these — duplicate work or a superseded PR. "
|
||||
"Resolution stays in Gitea and the workflow; this console only reports it."
|
||||
"</div>"
|
||||
)
|
||||
|
||||
|
||||
def _evidence_badges(evidence: Sequence[str]) -> str:
|
||||
return " ".join(
|
||||
_badge(
|
||||
_EVIDENCE_LABEL.get(name, name),
|
||||
_EVIDENCE_CSS.get(name, "badge-health-skipped"),
|
||||
)
|
||||
for name in EVIDENCE_ORDER
|
||||
if name in evidence
|
||||
)
|
||||
|
||||
|
||||
def _handoff_inline(handoff: HandoffSummary) -> str:
|
||||
type_css = "badge-health-ok" if handoff.cth_type_known else "badge-health-degraded"
|
||||
return (
|
||||
f'{_badge(handoff.cth_type or "—", type_css)}'
|
||||
f'<div class="muted" style="font-size:0.82rem;">'
|
||||
f'{escape(handoff.status or "—")} → {escape(handoff.next_owner or "—")}</div>'
|
||||
)
|
||||
|
||||
|
||||
def _handoff_cell(node: LinkageNode) -> str:
|
||||
"""Render the handoff column, distinguishing 'none found' from 'not loaded'."""
|
||||
status = node.handoff_status
|
||||
if status.state == HANDOFF_LOADED:
|
||||
if node.handoff is None:
|
||||
return '<span class="muted">no canonical handoff on this thread</span>'
|
||||
return _handoff_inline(node.handoff)
|
||||
if status.state == HANDOFF_NOT_LOADED:
|
||||
return (
|
||||
f'{_badge("not loaded", "badge-health-skipped")}'
|
||||
f'<div class="muted" style="font-size:0.82rem;">'
|
||||
f'{escape(status.reason or "")}</div>'
|
||||
)
|
||||
return (
|
||||
f'{_badge("unavailable", "badge-health-degraded")}'
|
||||
f'<div class="muted" style="font-size:0.82rem;">'
|
||||
f'{escape(status.reason or "")}</div>'
|
||||
)
|
||||
|
||||
|
||||
def _issue_rows(snapshot: LinkageSnapshot) -> str:
|
||||
rows = []
|
||||
for node in snapshot.issues:
|
||||
if node.linked_prs:
|
||||
linked = ", ".join(f"<code>#{number}</code>" for number in node.linked_prs)
|
||||
if node.contested:
|
||||
linked += " " + _badge("contested", "badge-blocked")
|
||||
elif node.links_authoritative:
|
||||
linked = '<span class="muted">none</span>'
|
||||
else:
|
||||
# The distinction an operator needs: nothing found in a window that
|
||||
# was not fully loaded is not the same as nothing existing.
|
||||
linked = '<span class="muted">none found (partial inventory)</span>'
|
||||
rows.append(
|
||||
"<tr>"
|
||||
f"<td>{_ref(node)}</td>"
|
||||
f"<td>{escape(node.title)}</td>"
|
||||
f"<td>{escape(node.state or '—')}</td>"
|
||||
f"<td>{_labels(node.labels)}</td>"
|
||||
f"<td>{linked}</td>"
|
||||
f"<td>{_handoff_cell(node)}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
if not rows:
|
||||
return '<tr><td colspan="6" class="muted">No issues in the loaded window.</td></tr>'
|
||||
return "".join(rows)
|
||||
|
||||
|
||||
def _pr_rows(snapshot: LinkageSnapshot) -> str:
|
||||
rows = []
|
||||
for node in snapshot.prs:
|
||||
if node.links:
|
||||
linked = "".join(
|
||||
f"<div><code>#{link.issue_number}</code> "
|
||||
f"{_evidence_badges(link.evidence)}</div>"
|
||||
for link in node.links
|
||||
)
|
||||
if node.ambiguous:
|
||||
linked += _badge("ambiguous", "badge-blocked")
|
||||
if node.contested:
|
||||
linked += " " + _badge("contested", "badge-blocked")
|
||||
elif node.links_authoritative:
|
||||
linked = '<span class="muted">no issue reference</span>'
|
||||
else:
|
||||
linked = '<span class="muted">none found (partial inventory)</span>'
|
||||
rows.append(
|
||||
"<tr>"
|
||||
f"<td>{_ref(node)}</td>"
|
||||
f"<td>{escape(node.title)}</td>"
|
||||
f"<td>{escape(node.state or '—')}</td>"
|
||||
f"<td>{_labels(node.labels)}</td>"
|
||||
f"<td>{linked}</td>"
|
||||
f"<td>{_handoff_cell(node)}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
if not rows:
|
||||
return (
|
||||
'<tr><td colspan="6" class="muted">No pull requests in the loaded '
|
||||
"window.</td></tr>"
|
||||
)
|
||||
return "".join(rows)
|
||||
|
||||
|
||||
def _focus_card(snapshot: LinkageSnapshot) -> str:
|
||||
"""Render the focused thread's latest canonical handoff, when one was loaded."""
|
||||
if snapshot.focus is None:
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Canonical handoff</h3>
|
||||
<p class="muted">{escape(snapshot.handoff_status.reason or "")}
|
||||
Add <code>?issue=N</code> or <code>?pr=N</code> to load the latest
|
||||
Canonical Thread Handoff for one thread.</p>
|
||||
</div>"""
|
||||
|
||||
kind, number = snapshot.focus
|
||||
target = f"{kind} #{number}"
|
||||
if not snapshot.handoff_status.loaded:
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Canonical handoff — {escape(target)}</h3>
|
||||
<p class="muted">{_badge("unavailable", "badge-health-degraded")}
|
||||
{escape(snapshot.handoff_status.reason or "handoff source did not run")}.
|
||||
This is not evidence that the thread carries no handoff.</p>
|
||||
</div>"""
|
||||
|
||||
handoff = next(
|
||||
(
|
||||
node.handoff
|
||||
for node in (snapshot.issues + snapshot.prs)
|
||||
if node.kind == kind and node.number == number and node.handoff
|
||||
),
|
||||
None,
|
||||
)
|
||||
if handoff is None:
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Canonical handoff — {escape(target)}</h3>
|
||||
<p class="muted">Comments loaded; no Canonical Thread Handoff comment found on
|
||||
this thread.</p>
|
||||
</div>"""
|
||||
|
||||
type_css = "badge-health-ok" if handoff.cth_type_known else "badge-health-degraded"
|
||||
unknown_note = (
|
||||
""
|
||||
if handoff.cth_type_known
|
||||
else (
|
||||
'<p class="muted">The comment\'s heading is not a declared CTH type, '
|
||||
"so it is reported as unrecognized rather than republished.</p>"
|
||||
)
|
||||
)
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Canonical handoff — {escape(target)}</h3>
|
||||
<p class="meta">{_badge(handoff.cth_type or "—", type_css)}
|
||||
by <code>{escape(handoff.author or "unknown")}</code>
|
||||
at <code>{escape(handoff.created_at or "unknown")}</code></p>
|
||||
{unknown_note}
|
||||
<table class="detail">
|
||||
<tr><th>Status</th><td>{escape(handoff.status or "—")}</td></tr>
|
||||
<tr><th>Next owner</th><td>{escape(handoff.next_owner or "—")}</td></tr>
|
||||
<tr><th>Current blocker</th><td>{escape(handoff.current_blocker or "—")}</td></tr>
|
||||
<tr><th>Decision</th><td>{escape(handoff.decision or "—")}</td></tr>
|
||||
<tr><th>Next action</th><td>{escape(handoff.next_action or "—")}</td></tr>
|
||||
</table>
|
||||
<p class="muted">Full event history:
|
||||
<a href="/api/v1/timeline?{escape(kind)}={number}"><code>/api/v1/timeline</code></a></p>
|
||||
</div>"""
|
||||
|
||||
|
||||
def _legend_card(snapshot: LinkageSnapshot) -> str:
|
||||
items = "".join(
|
||||
f"<li>{_badge(_EVIDENCE_LABEL[name], _EVIDENCE_CSS[name])} — "
|
||||
f"{escape(EVIDENCE_DESCRIPTIONS[name])}</li>"
|
||||
for name in EVIDENCE_ORDER
|
||||
)
|
||||
reveal_note = (
|
||||
"Gitea deep links are shown because the "
|
||||
"<code>GITEA_MCP_REVEAL_ENDPOINTS</code> admin opt-in is set."
|
||||
if snapshot.deep_links_enabled
|
||||
else (
|
||||
"Gitea deep links are withheld. Set "
|
||||
"<code>GITEA_MCP_REVEAL_ENDPOINTS=1</code> server-side to reveal "
|
||||
"them; item numbers stay usable without them."
|
||||
)
|
||||
)
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>How an edge was found</h3>
|
||||
<ul class="reasons">{items}</ul>
|
||||
<p class="muted">{reveal_note}</p>
|
||||
<p class="muted">Read-only surface: no issue or PR editing, no review, and no merge.
|
||||
JSON export: <a href="/api/v1/gitea/linkage"><code>/api/v1/gitea/linkage</code></a></p>
|
||||
</div>"""
|
||||
|
||||
|
||||
def render_linkage_page(snapshot: LinkageSnapshot) -> str:
|
||||
"""Render the full HTML page for the Gitea linkage console."""
|
||||
if not snapshot.ok:
|
||||
return render_page(
|
||||
title="Gitea linkage",
|
||||
body_html=f"""<h2>Gitea issue and PR linkage</h2>
|
||||
<p class="meta">Phase 3 read-only linkage console (#645).</p>
|
||||
{_error_card(snapshot)}
|
||||
{_scope_card(snapshot)}""",
|
||||
)
|
||||
|
||||
body = f"""<h2>Gitea issue and PR linkage</h2>
|
||||
<p class="meta">Phase 3 read-only linkage console (#645). Gitea remains the source of
|
||||
truth; this page reads it and never writes to it.</p>
|
||||
{_scope_card(snapshot)}
|
||||
{_contested_card(snapshot)}
|
||||
|
||||
<div class="prompt-card">
|
||||
<h3>Issues → pull requests</h3>
|
||||
<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Issue</th><th>Title</th><th>State</th><th>Labels</th>
|
||||
<th>Linked PRs</th><th>Latest handoff</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>{_issue_rows(snapshot)}</tbody>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
<div class="prompt-card">
|
||||
<h3>Pull requests → issues</h3>
|
||||
<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>PR</th><th>Title</th><th>State</th><th>Labels</th>
|
||||
<th>Linked issues</th><th>Latest handoff</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>{_pr_rows(snapshot)}</tbody>
|
||||
</table>
|
||||
</div>
|
||||
|
||||
{_focus_card(snapshot)}
|
||||
{_legend_card(snapshot)}
|
||||
"""
|
||||
return render_page(title="Gitea linkage", body_html=body)
|
||||
+10
-7
@@ -6,7 +6,8 @@ destination is a GET view or a Phase 1 placeholder. No mutation links.
|
||||
|
||||
Nav groups follow the #631 Phase 1 information architecture: Health, Traffic,
|
||||
Runtime/Sessions, Projects, Inventory, Timeline, Policy (placeholder), and
|
||||
Insights (placeholder). Later-phase surfaces are declared as ``stub`` items and
|
||||
Insights (placeholder), joined by the Phase 3 Gitea linkage group (#645).
|
||||
Later-phase surfaces are declared as ``stub`` items and
|
||||
backed by ``STUB_PAGES`` so their nav links resolve to a graceful placeholder
|
||||
instead of a 404.
|
||||
"""
|
||||
@@ -41,13 +42,17 @@ NAV_GROUPS: tuple[NavGroup, ...] = (
|
||||
NavItem("/system-health", "System health"),
|
||||
)),
|
||||
NavGroup("Traffic", (
|
||||
NavItem("/traffic", "Traffic control"),
|
||||
NavItem("/queue", "Queue"),
|
||||
NavItem("/leases", "Leases"),
|
||||
NavItem("/actions", "Actions"),
|
||||
NavItem("/notifications", "Notifications"),
|
||||
NavItem("/requests", "Requests"),
|
||||
)),
|
||||
NavGroup("Runtime/Sessions", (
|
||||
NavItem("/runtime", "Runtime health"),
|
||||
NavItem("/sessions", "Sessions", "stub"),
|
||||
NavItem("/runtime/restart", "Restart status"),
|
||||
NavItem("/sessions", "Sessions"),
|
||||
)),
|
||||
NavGroup("Projects", (
|
||||
NavItem("/projects", "Projects"),
|
||||
@@ -59,6 +64,9 @@ NAV_GROUPS: tuple[NavGroup, ...] = (
|
||||
NavGroup("Timeline", (
|
||||
NavItem("/timeline", "Timeline", "stub"),
|
||||
)),
|
||||
NavGroup("Gitea", (
|
||||
NavItem("/gitea", "Issue/PR linkage"),
|
||||
)),
|
||||
NavGroup("Policy", (
|
||||
NavItem("/policy", "Policy", "stub"),
|
||||
NavItem("/prompts", "Prompts"),
|
||||
@@ -75,11 +83,6 @@ NAV_GROUPS: tuple[NavGroup, ...] = (
|
||||
# issues of epic #631. Each maps a path to (title, description). Routes are
|
||||
# registered so nav links resolve to a graceful, read-only stub page.
|
||||
STUB_PAGES: dict[str, tuple[str, str]] = {
|
||||
"/sessions": (
|
||||
"Sessions",
|
||||
"Active session, capability, and role inventory. Backed by the unified "
|
||||
"inventory API (#636) once it lands.",
|
||||
),
|
||||
"/inventory": (
|
||||
"Inventory",
|
||||
"Unified sessions, leases, locks, namespaces, and worktree inventory. "
|
||||
|
||||
@@ -0,0 +1,158 @@
|
||||
"""HTML rendering for Phase 3 Notifications and Human-Attention Console (#648)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from html import escape
|
||||
from typing import Sequence
|
||||
|
||||
from webui.layout import render_page
|
||||
from webui.notifications import (
|
||||
ATTENTION_HUMAN_REQUIRED,
|
||||
ATTENTION_OPERATOR,
|
||||
ATTENTION_ROUTINE,
|
||||
NotificationItem,
|
||||
NotificationSnapshot,
|
||||
)
|
||||
|
||||
|
||||
def _render_attention_badge(attention_class: str) -> str:
|
||||
cls = "badge"
|
||||
if attention_class == ATTENTION_HUMAN_REQUIRED:
|
||||
cls += " badge-blocked"
|
||||
elif attention_class == ATTENTION_OPERATOR:
|
||||
cls += " badge-claimed"
|
||||
else:
|
||||
cls += " muted"
|
||||
return f'<span class="{cls}">{escape(attention_class)}</span>'
|
||||
|
||||
|
||||
def _render_notification_row(item: NotificationItem) -> str:
|
||||
category_label = escape(item.category.upper())
|
||||
id_str = escape(item.id)
|
||||
title_str = escape(item.title)
|
||||
summary_str = escape(item.summary)
|
||||
att_badge = _render_attention_badge(item.attention_class)
|
||||
|
||||
work_item_html = "—"
|
||||
if item.work_number and item.work_kind:
|
||||
kind_label = escape(item.work_kind.upper())
|
||||
num_str = f"#{item.work_number}"
|
||||
link = item.deep_link or "#"
|
||||
work_item_html = f'<a href="{escape(link)}"><code>{kind_label} {num_str}</code></a>'
|
||||
|
||||
requires_human_label = (
|
||||
'<span class="badge badge-blocked" style="font-size:0.75rem;">HUMAN REQUIRED</span>'
|
||||
if item.requires_human
|
||||
else ""
|
||||
)
|
||||
|
||||
return f"""<tr>
|
||||
<td><code>{category_label}</code><br><span class="muted" style="font-size:0.75rem;">{id_str}</span></td>
|
||||
<td>
|
||||
<div><strong>{title_str}</strong> {att_badge} {requires_human_label}</div>
|
||||
<div class="muted" style="font-size:0.85rem; margin-top:0.25rem;">{summary_str}</div>
|
||||
</td>
|
||||
<td>{work_item_html}</td>
|
||||
<td><span class="muted" style="font-size:0.8rem;">{escape(item.created_at[:19])}</span></td>
|
||||
</tr>"""
|
||||
|
||||
|
||||
def _render_notifications_table(items: Sequence[NotificationItem], empty_message: str) -> str:
|
||||
if not items:
|
||||
return f'<p class="muted" style="padding:1rem 0;">{escape(empty_message)}</p>'
|
||||
|
||||
rows = "".join(_render_notification_row(item) for item in items)
|
||||
return f"""<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th style="width: 18%;">Category & ID</th>
|
||||
<th style="width: 52%;">Title & Attention Summary</th>
|
||||
<th style="width: 15%;">Work Item</th>
|
||||
<th style="width: 15%;">Time</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{rows}
|
||||
</tbody>
|
||||
</table>"""
|
||||
|
||||
|
||||
def render_notifications_page(
|
||||
snapshot: NotificationSnapshot,
|
||||
*,
|
||||
filter_class: str = "inbox",
|
||||
filter_project: str | None = None,
|
||||
) -> str:
|
||||
"""Render the notifications and attention inbox page."""
|
||||
title = "Notifications & Attention Inbox"
|
||||
|
||||
err_html = ""
|
||||
if snapshot.fetch_error:
|
||||
err_html = f'<div class="stub" style="border-color:#e53e3e; background:#fff5f5; color:#c53030; margin-bottom:1rem;"><p><strong>Fetch Warning:</strong> {escape(snapshot.fetch_error)}</p></div>'
|
||||
|
||||
# Determine items to render based on filter_class
|
||||
if filter_class == ATTENTION_HUMAN_REQUIRED:
|
||||
display_items = snapshot.human_required_items
|
||||
active_tab_title = "Human-Required Escalations"
|
||||
elif filter_class == ATTENTION_OPERATOR:
|
||||
display_items = snapshot.operator_items
|
||||
active_tab_title = "Operator Inbox Items"
|
||||
elif filter_class == ATTENTION_ROUTINE:
|
||||
display_items = snapshot.routine_items
|
||||
active_tab_title = "Routine Workflow Transitions"
|
||||
elif filter_class == "all":
|
||||
display_items = snapshot.items
|
||||
active_tab_title = "All Events (including Routine)"
|
||||
else: # "inbox" default
|
||||
display_items = snapshot.inbox_items
|
||||
active_tab_title = "Attention Inbox (Human + Operator)"
|
||||
|
||||
hr_cls = "badge-blocked" if snapshot.human_required_count > 0 else "muted"
|
||||
op_cls = "badge-claimed" if snapshot.operator_count > 0 else "muted"
|
||||
|
||||
metrics_html = f"""<div style="display:flex; gap:1rem; margin-bottom:1.5rem;">
|
||||
<div class="health-card" style="flex:1;">
|
||||
<span class="muted" style="font-size:0.85rem;">Human Required</span>
|
||||
<h2 style="margin:0.2rem 0;"><span class="badge {hr_cls}" style="font-size:1.4rem;">{snapshot.human_required_count}</span></h2>
|
||||
<p class="muted" style="font-size:0.8rem; margin:0;">Critical escalation boundary</p>
|
||||
</div>
|
||||
<div class="health-card" style="flex:1;">
|
||||
<span class="muted" style="font-size:0.85rem;">Operator Inbox</span>
|
||||
<h2 style="margin:0.2rem 0;"><span class="badge {op_cls}" style="font-size:1.4rem;">{snapshot.operator_count}</span></h2>
|
||||
<p class="muted" style="font-size:0.8rem; margin:0;">Operational items needing review</p>
|
||||
</div>
|
||||
<div class="health-card" style="flex:1;">
|
||||
<span class="muted" style="font-size:0.85rem;">Routine Transitions</span>
|
||||
<h2 style="margin:0.2rem 0;"><span class="badge muted" style="font-size:1.4rem;">{snapshot.routine_count}</span></h2>
|
||||
<p class="muted" style="font-size:0.8rem; margin:0;">Background transitions (filtered)</p>
|
||||
</div>
|
||||
</div>"""
|
||||
|
||||
# Filter navigation links
|
||||
def _tab_link(target_class: str, label: str) -> str:
|
||||
is_active = (filter_class == target_class)
|
||||
style = "font-weight:bold; border-bottom:2px solid currentColor;" if is_active else "color:#4a5568;"
|
||||
return f'<a href="/notifications?attention_class={target_class}" style="margin-right:1.25rem; text-decoration:none; padding-bottom:0.25rem; {style}">{label}</a>'
|
||||
|
||||
tabs_html = f"""<div style="margin-bottom:1.25rem; border-bottom:1px solid #e2e8f0; padding-bottom:0.5rem;">
|
||||
{_tab_link("inbox", f"Attention Inbox ({snapshot.human_required_count + snapshot.operator_count})")}
|
||||
{_tab_link("human-required", f"Human Required ({snapshot.human_required_count})")}
|
||||
{_tab_link("operator", f"Operator ({snapshot.operator_count})")}
|
||||
{_tab_link("routine", f"Routine ({snapshot.routine_count})")}
|
||||
{_tab_link("all", f"All Events ({snapshot.total_count})")}
|
||||
</div>"""
|
||||
|
||||
table_html = _render_notifications_table(
|
||||
display_items,
|
||||
f"No items match attention filter '{filter_class}'.",
|
||||
)
|
||||
|
||||
body = f"""<h2>{escape(title)}</h2>
|
||||
<p class="muted">Phase 3 console surface for human-attention routing (#648). Routine workflow transitions are filtered by default to eliminate notification fatigue.</p>
|
||||
{err_html}
|
||||
{metrics_html}
|
||||
{tabs_html}
|
||||
<h3>{escape(active_tab_title)}</h3>
|
||||
{table_html}"""
|
||||
|
||||
return render_page(title=title, body_html=body)
|
||||
@@ -0,0 +1,486 @@
|
||||
"""Notifications and human-attention routing module for Phase 3 web console (#648).
|
||||
|
||||
Defines attention classes, event classification rules, and inbox aggregation so
|
||||
operators receive direct alerts only for human-required escalation boundaries
|
||||
(#628) while routine workflow transitions remain available for pull-based review.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Callable
|
||||
|
||||
from webui import console_redaction
|
||||
from webui.project_registry import load_registry
|
||||
from webui.queue_loader import QueueSnapshot, load_queue_snapshot
|
||||
from webui.lease_loader import LeaseSnapshot, load_lease_snapshot
|
||||
from webui.system_health import SystemHealthSnapshot, load_system_health
|
||||
|
||||
# Attention class definitions (#628, #648)
|
||||
ATTENTION_ROUTINE = "routine"
|
||||
ATTENTION_OPERATOR = "operator"
|
||||
ATTENTION_HUMAN_REQUIRED = "human-required"
|
||||
|
||||
ATTENTION_CLASSES = (
|
||||
ATTENTION_ROUTINE,
|
||||
ATTENTION_OPERATOR,
|
||||
ATTENTION_HUMAN_REQUIRED,
|
||||
)
|
||||
|
||||
# Notification categories
|
||||
CATEGORY_AUTH = "auth"
|
||||
CATEGORY_BLOCKER = "blocker"
|
||||
CATEGORY_LEASE = "lease"
|
||||
CATEGORY_VALIDATION = "validation"
|
||||
CATEGORY_WORKFLOW = "workflow"
|
||||
CATEGORY_SYSTEM = "system"
|
||||
|
||||
CATEGORIES = (
|
||||
CATEGORY_AUTH,
|
||||
CATEGORY_BLOCKER,
|
||||
CATEGORY_LEASE,
|
||||
CATEGORY_VALIDATION,
|
||||
CATEGORY_WORKFLOW,
|
||||
CATEGORY_SYSTEM,
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NotificationItem:
|
||||
"""A single notification or inbox event."""
|
||||
|
||||
id: str
|
||||
attention_class: str # "routine", "operator", "human-required"
|
||||
category: str # "auth", "blocker", "lease", "validation", etc.
|
||||
title: str
|
||||
summary: str
|
||||
work_kind: str | None # "issue", "pr", "session", "system"
|
||||
work_number: int | None
|
||||
project_id: str
|
||||
repo_label: str
|
||||
created_at: str
|
||||
deep_link: str | None = None
|
||||
requires_human: bool = False
|
||||
extra: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"id": self.id,
|
||||
"attention_class": self.attention_class,
|
||||
"category": self.category,
|
||||
"title": self.title,
|
||||
"summary": console_redaction.redact_text(self.summary),
|
||||
"work_kind": self.work_kind,
|
||||
"work_number": self.work_number,
|
||||
"project_id": self.project_id,
|
||||
"repo_label": self.repo_label,
|
||||
"created_at": self.created_at,
|
||||
"deep_link": self.deep_link,
|
||||
"requires_human": self.requires_human,
|
||||
"extra": self.extra,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NotificationSnapshot:
|
||||
"""Snapshot of notifications and attention inbox state."""
|
||||
|
||||
project_id: str
|
||||
repo_label: str
|
||||
items: tuple[NotificationItem, ...]
|
||||
human_required_count: int
|
||||
operator_count: int
|
||||
routine_count: int
|
||||
total_count: int
|
||||
fetch_error: str | None = None
|
||||
|
||||
@property
|
||||
def inbox_items(self) -> tuple[NotificationItem, ...]:
|
||||
"""Items requiring operator or human attention (excluding routine)."""
|
||||
return tuple(
|
||||
item
|
||||
for item in self.items
|
||||
if item.attention_class in {ATTENTION_OPERATOR, ATTENTION_HUMAN_REQUIRED}
|
||||
)
|
||||
|
||||
@property
|
||||
def human_required_items(self) -> tuple[NotificationItem, ...]:
|
||||
return tuple(
|
||||
item for item in self.items if item.attention_class == ATTENTION_HUMAN_REQUIRED
|
||||
)
|
||||
|
||||
@property
|
||||
def operator_items(self) -> tuple[NotificationItem, ...]:
|
||||
return tuple(
|
||||
item for item in self.items if item.attention_class == ATTENTION_OPERATOR
|
||||
)
|
||||
|
||||
@property
|
||||
def routine_items(self) -> tuple[NotificationItem, ...]:
|
||||
return tuple(
|
||||
item for item in self.items if item.attention_class == ATTENTION_ROUTINE
|
||||
)
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"project_id": self.project_id,
|
||||
"repo_label": self.repo_label,
|
||||
"human_required_count": self.human_required_count,
|
||||
"operator_count": self.operator_count,
|
||||
"routine_count": self.routine_count,
|
||||
"total_count": self.total_count,
|
||||
"fetch_error": self.fetch_error,
|
||||
"inbox_items": [item.as_dict() for item in self.inbox_items],
|
||||
"all_items": [item.as_dict() for item in self.items],
|
||||
}
|
||||
|
||||
|
||||
def classify_attention_event(
|
||||
category: str,
|
||||
title: str,
|
||||
summary: str,
|
||||
*,
|
||||
is_hard_stop: bool = False,
|
||||
is_auth_failure: bool = False,
|
||||
is_irrecoverable: bool = False,
|
||||
is_decision_lock: bool = False,
|
||||
is_validation_failure: bool = False,
|
||||
is_stale: bool = False,
|
||||
is_blocker: bool = False,
|
||||
) -> tuple[str, bool]:
|
||||
"""Classify an event into an attention class and human requirement flag.
|
||||
|
||||
Rules (#628, #648):
|
||||
1. Critical boundaries (hard stop, auth failure, irrecoverable state,
|
||||
decision lock, validation failure) -> ATTENTION_HUMAN_REQUIRED (requires_human=True).
|
||||
2. Operational queues (blocker, stale lease, unassigned ready work, queue collision)
|
||||
-> ATTENTION_OPERATOR (requires_human=False).
|
||||
3. Routine state transitions (clean progression, healthy heartbeats) -> ATTENTION_ROUTINE (requires_human=False).
|
||||
|
||||
Classification uses structured flags and category only. Human-authored
|
||||
``title`` / ``summary`` text is never substring-matched for escalation
|
||||
(PR #905 review B1) — callers that need text signals must set flags from
|
||||
machine-generated status/detail fields before calling this function.
|
||||
"""
|
||||
del title, summary # kept for API stability; never used for classification
|
||||
if (
|
||||
is_hard_stop
|
||||
or is_auth_failure
|
||||
or is_irrecoverable
|
||||
or is_decision_lock
|
||||
or is_validation_failure
|
||||
or category in {CATEGORY_AUTH, CATEGORY_VALIDATION}
|
||||
):
|
||||
return ATTENTION_HUMAN_REQUIRED, True
|
||||
|
||||
if is_stale or is_blocker or category in {CATEGORY_BLOCKER, CATEGORY_LEASE}:
|
||||
return ATTENTION_OPERATOR, False
|
||||
|
||||
return ATTENTION_ROUTINE, False
|
||||
|
||||
|
||||
def load_notifications_snapshot(
|
||||
project_id: str | None = None,
|
||||
*,
|
||||
load_queue: Callable[..., QueueSnapshot] | None = None,
|
||||
load_leases: Callable[..., LeaseSnapshot] | None = None,
|
||||
load_health: Callable[..., SystemHealthSnapshot] | None = None,
|
||||
) -> NotificationSnapshot:
|
||||
"""Load and classify attention notifications across queue, leases, and system health."""
|
||||
registry = load_registry()
|
||||
project = None
|
||||
if project_id:
|
||||
for entry in registry.projects:
|
||||
if entry.id == project_id:
|
||||
project = entry
|
||||
break
|
||||
else:
|
||||
project = registry.projects[0] if registry.projects else None
|
||||
|
||||
if project is None:
|
||||
return NotificationSnapshot(
|
||||
project_id=project_id or "",
|
||||
repo_label="",
|
||||
items=(),
|
||||
human_required_count=0,
|
||||
operator_count=0,
|
||||
routine_count=0,
|
||||
total_count=0,
|
||||
fetch_error="project not found in registry",
|
||||
)
|
||||
|
||||
queue_loader_fn = load_queue or load_queue_snapshot
|
||||
lease_loader_fn = load_leases or load_lease_snapshot
|
||||
health_loader_fn = load_health or load_system_health
|
||||
|
||||
try:
|
||||
queue_snap = queue_loader_fn(project.id)
|
||||
except TypeError:
|
||||
queue_snap = queue_loader_fn(project_id=project.id)
|
||||
|
||||
try:
|
||||
lease_snap = lease_loader_fn(project_id=project.id)
|
||||
except TypeError:
|
||||
lease_snap = lease_loader_fn(project.id)
|
||||
|
||||
try:
|
||||
health_snap = health_loader_fn(project_id=project.id)
|
||||
except TypeError:
|
||||
try:
|
||||
health_snap = health_loader_fn(project.id)
|
||||
except TypeError:
|
||||
health_snap = health_loader_fn()
|
||||
|
||||
items: list[NotificationItem] = []
|
||||
now_iso = datetime.now(timezone.utc).isoformat()
|
||||
|
||||
# 1. System health alerts (highest priority)
|
||||
for err_idx, probe_err in enumerate(getattr(health_snap, "probe_errors", ())):
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_SYSTEM,
|
||||
"System Health Probe Error",
|
||||
probe_err,
|
||||
is_blocker=True,
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-sys-err-{project.id}-{err_idx}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_SYSTEM,
|
||||
title="System Health Error",
|
||||
summary=f"System health error: {probe_err}",
|
||||
work_kind="system",
|
||||
work_number=None,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link="/system",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
|
||||
for probe in getattr(health_snap, "dependencies", ()):
|
||||
if probe.status not in ("ok", "healthy"):
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_SYSTEM,
|
||||
f"Probe Failure: {probe.name}",
|
||||
probe.detail or probe.status,
|
||||
is_hard_stop=("stop" in probe.status or "fatal" in probe.status),
|
||||
is_auth_failure=("auth" in probe.name.lower() or "unauthorized" in probe.status.lower()),
|
||||
is_blocker=True,
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-probe-{probe.name}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_AUTH if "auth" in probe.name.lower() else CATEGORY_SYSTEM,
|
||||
title=f"Health Probe Alert: {probe.name}",
|
||||
summary=f"Probe '{probe.name}' reported status '{probe.status}': {probe.detail}",
|
||||
work_kind="system",
|
||||
work_number=None,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link="/system",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
|
||||
# 2. Queue items (PRs and Issues)
|
||||
for pr in queue_snap.prs:
|
||||
if "blocked" in pr.badges:
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_BLOCKER,
|
||||
f"PR #{pr.number} Blocked",
|
||||
f"PR #{pr.number} '{pr.title}' is blocked or has merge conflicts.",
|
||||
is_blocker=True,
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-pr-block-{pr.number}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_BLOCKER,
|
||||
title=f"Blocked PR #{pr.number}",
|
||||
summary=f"PR #{pr.number} ({pr.title}) requires merge conflict resolution.",
|
||||
work_kind="pr",
|
||||
work_number=pr.number,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link=f"/traffic",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
elif "stale" in pr.badges:
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_WORKFLOW,
|
||||
f"PR #{pr.number} Stale",
|
||||
f"PR #{pr.number} '{pr.title}' has had no activity for over 14 days.",
|
||||
is_stale=True,
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-pr-stale-{pr.number}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_WORKFLOW,
|
||||
title=f"Stale PR #{pr.number}",
|
||||
summary=f"PR #{pr.number} ({pr.title}) is stale.",
|
||||
work_kind="pr",
|
||||
work_number=pr.number,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link=f"/queue",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
else:
|
||||
# Routine PR transition
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_WORKFLOW,
|
||||
f"PR #{pr.number} Active",
|
||||
f"PR #{pr.number} '{pr.title}' is in routine state {', '.join(pr.badges)}.",
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-pr-routine-{pr.number}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_WORKFLOW,
|
||||
title=f"Routine PR #{pr.number}",
|
||||
summary=f"PR #{pr.number} ({pr.title}) state: {', '.join(pr.badges)}.",
|
||||
work_kind="pr",
|
||||
work_number=pr.number,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link=f"/queue",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
|
||||
for issue in queue_snap.issues:
|
||||
if "duplicate" in issue.badges:
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_BLOCKER,
|
||||
f"Issue #{issue.number} Duplicate PRs",
|
||||
f"Issue #{issue.number} has multiple linked PRs.",
|
||||
is_blocker=True,
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-issue-dup-{issue.number}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_BLOCKER,
|
||||
title=f"Duplicate PRs on Issue #{issue.number}",
|
||||
summary=f"Issue #{issue.number} ({issue.title}) linked to multiple PRs.",
|
||||
work_kind="issue",
|
||||
work_number=issue.number,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link=f"/traffic",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
elif "claimed" in issue.badges or "in-review" in issue.badges:
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_WORKFLOW,
|
||||
f"Issue #{issue.number} Active",
|
||||
f"Issue #{issue.number} '{issue.title}' in state {', '.join(issue.badges)}.",
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-issue-routine-{issue.number}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_WORKFLOW,
|
||||
title=f"Routine Issue #{issue.number}",
|
||||
summary=f"Issue #{issue.number} ({issue.title}) state: {', '.join(issue.badges)}.",
|
||||
work_kind="issue",
|
||||
work_number=issue.number,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link=f"/queue",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
|
||||
# 3. Leases / Collisions
|
||||
for lease in lease_snap.reviewer_leases:
|
||||
if lease.get("is_expired") or lease.get("status") == "expired":
|
||||
pr_num = lease.get("pr_number") or lease.get("work_item_number")
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_LEASE,
|
||||
f"Reviewer Lease Expired for PR #{pr_num}",
|
||||
f"Reviewer lease for PR #{pr_num} has expired.",
|
||||
is_stale=True,
|
||||
)
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-lease-exp-pr-{pr_num}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_LEASE,
|
||||
title=f"Expired Reviewer Lease (PR #{pr_num})",
|
||||
summary=f"Reviewer lease for PR #{pr_num} expired.",
|
||||
work_kind="pr",
|
||||
work_number=pr_num,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link="/leases",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
|
||||
for col_idx, collision in enumerate(lease_snap.duplicate_prs):
|
||||
att_cls, req_human = classify_attention_event(
|
||||
CATEGORY_BLOCKER,
|
||||
f"Duplicate PR Collision ({collision.kind})",
|
||||
collision.message,
|
||||
is_blocker=True,
|
||||
)
|
||||
issue_part = collision.issue_number if collision.issue_number is not None else "none"
|
||||
kind_part = (collision.kind or "unknown").replace(" ", "-")
|
||||
items.append(
|
||||
NotificationItem(
|
||||
id=f"notif-collision-{kind_part}-{issue_part}-{col_idx}",
|
||||
attention_class=att_cls,
|
||||
category=CATEGORY_BLOCKER,
|
||||
title=f"Collision Alert ({collision.kind})",
|
||||
summary=collision.message,
|
||||
work_kind="issue" if collision.issue_number else "pr",
|
||||
work_number=collision.issue_number,
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
created_at=now_iso,
|
||||
deep_link="/leases",
|
||||
requires_human=req_human,
|
||||
)
|
||||
)
|
||||
|
||||
human_req_count = sum(1 for i in items if i.attention_class == ATTENTION_HUMAN_REQUIRED)
|
||||
operator_count = sum(1 for i in items if i.attention_class == ATTENTION_OPERATOR)
|
||||
routine_count = sum(1 for i in items if i.attention_class == ATTENTION_ROUTINE)
|
||||
|
||||
# Fetch errors are transport/load failures only — not probe results that
|
||||
# already surface as first-class notification items (PR #905 review B3).
|
||||
fetch_err = queue_snap.fetch_error or lease_snap.fetch_error
|
||||
if isinstance(fetch_err, (tuple, list)):
|
||||
fetch_err = "; ".join(fetch_err) if fetch_err else None
|
||||
|
||||
return NotificationSnapshot(
|
||||
project_id=project.id,
|
||||
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||
items=tuple(items),
|
||||
human_required_count=human_req_count,
|
||||
operator_count=operator_count,
|
||||
routine_count=routine_count,
|
||||
total_count=len(items),
|
||||
fetch_error=fetch_err,
|
||||
)
|
||||
|
||||
|
||||
def snapshot_to_dict(snapshot: NotificationSnapshot) -> dict[str, Any]:
|
||||
"""JSON-serializable export for /api/v1/notifications."""
|
||||
return snapshot.as_dict()
|
||||
+61
-5
@@ -4,7 +4,7 @@ from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Callable
|
||||
from urllib.parse import urlparse
|
||||
@@ -31,10 +31,20 @@ class PaginationMeta:
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class QueueItem:
|
||||
"""One queue row.
|
||||
|
||||
``extra`` holds *display* strings for the queue page (values are truncated
|
||||
or humanized for rendering). ``signals`` holds the *authoritative* typed
|
||||
values taken straight from the Gitea payload, for consumers that classify
|
||||
or pin state rather than render it (#640). Never derive identity or
|
||||
concurrency decisions from ``extra``.
|
||||
"""
|
||||
|
||||
number: int
|
||||
title: str
|
||||
badges: tuple[str, ...]
|
||||
extra: dict[str, str]
|
||||
signals: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
@@ -134,21 +144,37 @@ def _format_pr_item(pr: dict, badges: tuple[str, ...]) -> QueueItem:
|
||||
"mergeable" if mergeable is True else "conflicted" if mergeable is False else "unknown"
|
||||
)
|
||||
linked = _extract_linked_issue(pr.get("title"), pr.get("body"))
|
||||
head_sha = str(head.get("sha") or "")
|
||||
labels = tuple(
|
||||
str(lb.get("name") or "") for lb in (pr.get("labels") or []) if lb.get("name")
|
||||
)
|
||||
return QueueItem(
|
||||
number=int(pr["number"]),
|
||||
title=str(pr.get("title") or ""),
|
||||
badges=badges,
|
||||
extra={
|
||||
"branch": f"{head.get('ref', '?')} → {base.get('ref', '?')}",
|
||||
"head_sha": str(head.get("sha") or "")[:12],
|
||||
# Display only — truncated. Pin against signals["head_sha"] instead.
|
||||
"head_sha": head_sha[:12],
|
||||
"mergeable": merge_label,
|
||||
"linked_issue": str(linked) if linked is not None else "",
|
||||
},
|
||||
signals={
|
||||
"head_sha": head_sha,
|
||||
"head_ref": str(head.get("ref") or ""),
|
||||
"base_ref": str(base.get("ref") or ""),
|
||||
"mergeable": mergeable if isinstance(mergeable, bool) else None,
|
||||
"labels": labels,
|
||||
"linked_issue": linked,
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def _format_issue_item(issue: dict, badges: tuple[str, ...]) -> QueueItem:
|
||||
labels = ", ".join(lb.get("name", "") for lb in issue.get("labels", []))
|
||||
label_names = tuple(
|
||||
str(lb.get("name") or "") for lb in (issue.get("labels") or []) if lb.get("name")
|
||||
)
|
||||
labels = ", ".join(label_names)
|
||||
assignee = (issue.get("assignee") or {}).get("login", "")
|
||||
return QueueItem(
|
||||
number=int(issue["number"]),
|
||||
@@ -159,6 +185,11 @@ def _format_issue_item(issue: dict, badges: tuple[str, ...]) -> QueueItem:
|
||||
"assignee": assignee or "unassigned",
|
||||
"state": str(issue.get("state") or ""),
|
||||
},
|
||||
signals={
|
||||
"labels": label_names,
|
||||
"assignee": assignee,
|
||||
"state": str(issue.get("state") or ""),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
@@ -180,6 +211,19 @@ def _pagination_from_pages(
|
||||
)
|
||||
|
||||
|
||||
_SUPPORTED_FETCH_STATES = ("open", "closed", "all")
|
||||
|
||||
|
||||
def _safe_state(state: str | None) -> str:
|
||||
"""Constrain a caller-supplied item state before it reaches a query string.
|
||||
|
||||
The value is interpolated into the Gitea URL, so an unrecognised state falls
|
||||
back to ``open`` rather than being passed through.
|
||||
"""
|
||||
text = (state or "open").strip().lower()
|
||||
return text if text in _SUPPORTED_FETCH_STATES else "open"
|
||||
|
||||
|
||||
def _fetch_prs(
|
||||
host: str,
|
||||
org: str,
|
||||
@@ -187,8 +231,15 @@ def _fetch_prs(
|
||||
auth: str,
|
||||
*,
|
||||
per_page: int = 50,
|
||||
state: str = "open",
|
||||
) -> tuple[list[dict], PaginationMeta]:
|
||||
url = f"{repo_api_url(host, org, repo)}/pulls?state=open"
|
||||
"""Fetch PRs in *state* (``open``, ``closed``, or ``all``).
|
||||
|
||||
The queue dashboard only ever wants the open window, so ``open`` stays the
|
||||
default. The linkage console (#645) widens it, because a landed issue↔PR
|
||||
edge lives on a merged PR.
|
||||
"""
|
||||
url = f"{repo_api_url(host, org, repo)}/pulls?state={_safe_state(state)}"
|
||||
all_raw: list[dict] = []
|
||||
pages_fetched = 0
|
||||
is_final = False
|
||||
@@ -217,8 +268,13 @@ def _fetch_issues(
|
||||
auth: str,
|
||||
*,
|
||||
per_page: int = 50,
|
||||
state: str = "open",
|
||||
) -> tuple[list[dict], PaginationMeta]:
|
||||
url = f"{repo_api_url(host, org, repo)}/issues?state=open&type=issues"
|
||||
"""Fetch issues in *state* (``open``, ``closed``, or ``all``); see :func:`_fetch_prs`."""
|
||||
url = (
|
||||
f"{repo_api_url(host, org, repo)}/issues"
|
||||
f"?state={_safe_state(state)}&type=issues"
|
||||
)
|
||||
all_raw: list[dict] = []
|
||||
page = 1
|
||||
pages_fetched = 0
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,164 @@
|
||||
"""HTML views for the operator request surface (#643).
|
||||
|
||||
The form is deliberately a *preview* form. It has no initiate button, because
|
||||
initiating requires a confirmed POST to ``/api/v1/requests/apply`` and a stray
|
||||
form submission must not be able to produce one by accident.
|
||||
|
||||
Nothing rendered here is trusted input: every interpolated value is escaped,
|
||||
and the page renders only values the service already produced rather than
|
||||
echoing a raw request body back.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
from webui.layout import render_page
|
||||
from webui.request_service import (
|
||||
REQUESTABLE_ROLES,
|
||||
WORK_KINDS,
|
||||
RequestError,
|
||||
RequestPreview,
|
||||
)
|
||||
|
||||
REQUESTS_PATH = "/requests"
|
||||
PREVIEW_API_PATH = "/api/v1/requests/preview"
|
||||
APPLY_API_PATH = "/api/v1/requests/apply"
|
||||
|
||||
|
||||
def _escape(text: Any) -> str:
|
||||
return html.escape(str(text if text is not None else ""), quote=True)
|
||||
|
||||
|
||||
REQUEST_PAGE_STYLES = """
|
||||
<style>
|
||||
.request-form { display: grid; gap: 0.75rem; max-width: 44rem; }
|
||||
.request-form label { display: grid; gap: 0.25rem; font-size: 0.9rem; }
|
||||
.request-check { margin: 0.35rem 0; }
|
||||
.request-check .verdict-ok { color: var(--accent); }
|
||||
.request-check .verdict-fail { color: #d14; }
|
||||
.request-prohibited code { margin-right: 0.4rem; }
|
||||
</style>
|
||||
"""
|
||||
|
||||
|
||||
def _options(values: tuple[str, ...], selected: Any) -> str:
|
||||
return "".join(
|
||||
f"<option value='{_escape(value)}'"
|
||||
+ (" selected" if selected == value else "")
|
||||
+ f">{_escape(value)}</option>"
|
||||
for value in values
|
||||
)
|
||||
|
||||
|
||||
def _form(values: dict[str, Any] | None = None) -> str:
|
||||
current = dict(values or {})
|
||||
number = current.get("work_number")
|
||||
return (
|
||||
f"<form class='request-form' method='post' action='{REQUESTS_PATH}'>"
|
||||
"<label>Desired role<select name='desired_role'>"
|
||||
f"{_options(REQUESTABLE_ROLES, current.get('desired_role'))}"
|
||||
"</select></label>"
|
||||
"<label>Work kind<select name='work_kind'>"
|
||||
f"{_options(WORK_KINDS, current.get('work_kind'))}"
|
||||
"</select></label>"
|
||||
"<label>Issue or PR number"
|
||||
"<input type='number' name='work_number' min='1' required "
|
||||
f"value='{_escape(number) if number else ''}'></label>"
|
||||
"<label>Intent summary"
|
||||
"<input type='text' name='intent_summary' maxlength='500' required "
|
||||
f"value='{_escape(current.get('intent_summary'))}'></label>"
|
||||
"<label>Expected head SHA <span class='muted'>(PR work only)</span>"
|
||||
"<input type='text' name='expected_head_sha' "
|
||||
f"value='{_escape(current.get('expected_head_sha'))}'></label>"
|
||||
"<button type='submit' class='copy-btn'>Preview request</button>"
|
||||
"<p class='muted meta'>Preview is read-only and creates no assignment. "
|
||||
f"Initiating requires a confirmed POST to <code>{APPLY_API_PATH}</code>."
|
||||
"</p>"
|
||||
"</form>"
|
||||
)
|
||||
|
||||
|
||||
def _checks_block(preview: RequestPreview) -> str:
|
||||
rows = []
|
||||
for check in preview.checks:
|
||||
verdict = "PASS" if check.ok else "FAIL"
|
||||
css = "verdict-ok" if check.ok else "verdict-fail"
|
||||
rows.append(
|
||||
"<li class='request-check'>"
|
||||
f"<span class='{css}'><strong>{verdict}</strong></span> "
|
||||
f"<code>{_escape(check.name)}</code> — {_escape(check.detail)} "
|
||||
f"<span class='muted meta'>({_escape(check.reason_code)})</span>"
|
||||
"</li>"
|
||||
)
|
||||
return "<ul>" + "".join(rows) + "</ul>"
|
||||
|
||||
|
||||
def _preview_block(preview: RequestPreview) -> str:
|
||||
verdict = "AUTHORIZED" if preview.authorized else "DENIED"
|
||||
prohibited = "".join(
|
||||
f"<code>{_escape(action)}</code>" for action in preview.prohibited_actions
|
||||
)
|
||||
request = preview.request
|
||||
evidence = json.dumps(preview.allocator_evidence, indent=2, default=str)
|
||||
return (
|
||||
"<h3>Intent preview</h3>"
|
||||
f"<p><strong>{verdict}</strong> — {_escape(preview.detail)}</p>"
|
||||
"<p class='meta'>"
|
||||
f"Role <code>{_escape(request.desired_role)}</code> · "
|
||||
f"{_escape(request.work_kind)} <code>{_escape(request.display_ref)}</code>"
|
||||
f" · profile <code>{_escape(preview.required_profile)}</code> · "
|
||||
f"namespace <code>{_escape(preview.required_namespace)}</code> · "
|
||||
f"permission <code>{_escape(preview.required_permission)}</code>"
|
||||
"</p>"
|
||||
f"<p>Intent: {_escape(request.intent_summary)}</p>"
|
||||
f"{_checks_block(preview)}"
|
||||
f"<p><strong>Next safe action:</strong> "
|
||||
f"{_escape(preview.next_safe_action)}</p>"
|
||||
"<p class='request-prohibited'><strong>Prohibited for this role:</strong> "
|
||||
+ (prohibited or "<span class='muted'>none declared</span>")
|
||||
+ "</p>"
|
||||
"<p class='muted meta'>Correlation id "
|
||||
f"<code>{_escape(preview.correlation_id)}</code></p>"
|
||||
"<details><summary>Allocator evidence</summary>"
|
||||
f"<pre class='prompt-text'>{_escape(evidence)}</pre>"
|
||||
"</details>"
|
||||
)
|
||||
|
||||
|
||||
def _error_block(error: RequestError) -> str:
|
||||
field = (
|
||||
f"<p class='meta'>Field: <code>{_escape(error.field_name)}</code></p>"
|
||||
if error.field_name
|
||||
else ""
|
||||
)
|
||||
return (
|
||||
"<h3>Request rejected</h3>"
|
||||
f"<p><strong>{_escape(error.reason_code)}</strong> — "
|
||||
f"{_escape(error.detail)}</p>{field}"
|
||||
)
|
||||
|
||||
|
||||
def render_requests_page(
|
||||
*,
|
||||
preview: RequestPreview | None = None,
|
||||
error: RequestError | None = None,
|
||||
submitted: dict[str, Any] | None = None,
|
||||
) -> str:
|
||||
"""Render the request form, plus a preview or rejection when one exists."""
|
||||
body = (
|
||||
"<h2>Requests</h2>"
|
||||
"<p>Submit a work request — desired role, issue or PR, and intent — "
|
||||
"and see whether it would be authorized before anything is reserved. "
|
||||
"Initiation goes through the allocator (#600/#613); this console never "
|
||||
"self-selects work, never approves, and never merges.</p>"
|
||||
+ _form(submitted)
|
||||
+ (_error_block(error) if error is not None else "")
|
||||
+ (_preview_block(preview) if preview is not None else "")
|
||||
+ f"<p class='meta'><a href='{PREVIEW_API_PATH}'>Preview API</a> · "
|
||||
"<a href='/api/console/security-model'>RBAC model</a></p>"
|
||||
+ REQUEST_PAGE_STYLES
|
||||
)
|
||||
return render_page(title="Requests", body_html=body)
|
||||
@@ -0,0 +1,579 @@
|
||||
"""Read-only restart status, impact preview, and approval state (#667).
|
||||
|
||||
Phase 1 of the console restart surface. It *consumes* the #655 coordinator
|
||||
substrate and renders it; it never restarts, reloads, drains, approves, or kills
|
||||
anything. There is no apply path in this module, so there is no execution gate
|
||||
here to arm incorrectly — the only writes the console could perform are the ones
|
||||
it does not implement.
|
||||
|
||||
Sources, each independently fail-soft and each reported with its own
|
||||
:class:`SourceStatus`:
|
||||
|
||||
* :mod:`restart_coordinator` — restart-class policy matrix (#663) and the
|
||||
blast-radius impact report (#658).
|
||||
* :mod:`drain_proof` — drain checklist and gate verdict (#661), verified
|
||||
read-only against a caller-supplied proof.
|
||||
* :mod:`post_restart_reconcile` — post-restart completion proof (#662).
|
||||
* :mod:`webui.console_authz` — role authorization for the approval controls
|
||||
(#633).
|
||||
|
||||
Three rules this module holds itself to, because a status surface that lies is
|
||||
worse than one that is absent:
|
||||
|
||||
**A source that could not be read is reported unavailable, never green.** No
|
||||
default, placeholder, or self-comparison is substituted for a reading that
|
||||
failed. An unreadable control-plane DB yields ``inventory_complete=False``,
|
||||
which the coordinator itself turns into a fail-closed verdict.
|
||||
|
||||
**Authorization is asked the way execution would ask it.** Every authorization
|
||||
probe passes ``for_execution=True``, so the console reports whether the action
|
||||
could actually run rather than the weaker "this principal is the right role".
|
||||
While the console is in Phase 1 that answer is ``phase_not_active`` for every
|
||||
phase-2 action, and the surface says so plainly instead of showing an allow.
|
||||
|
||||
**The database is opened read-only.** ``ControlPlaneDB()`` creates directories
|
||||
and runs migrations on construction, which is a write; this module opens the
|
||||
sqlite file with ``mode=ro`` exactly as :mod:`webui.inventory` does, and treats
|
||||
a missing file as missing authority rather than an empty inventory.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import sqlite3
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Callable, Mapping
|
||||
|
||||
import control_plane_db
|
||||
import drain_proof
|
||||
import restart_coordinator
|
||||
from webui import console_authz
|
||||
from webui.inventory import redact_path, scrub
|
||||
|
||||
# --- Source status ----------------------------------------------------------
|
||||
|
||||
STATUS_OK = "ok"
|
||||
STATUS_UNAVAILABLE = "unavailable"
|
||||
|
||||
#: Console actions whose authorization state this surface reports. Both are
|
||||
#: pre-existing #642 actions; this module adds no new console action because it
|
||||
#: performs no console action.
|
||||
REPORTED_ACTIONS: tuple[str, ...] = (
|
||||
"system.restart_namespace",
|
||||
"system.reload_namespace",
|
||||
)
|
||||
|
||||
#: The break-glass workflow (#664) is not consumed here. It is declared so the
|
||||
#: surface is honest about the gap rather than silently omitting a governance
|
||||
#: path the operator has been told exists.
|
||||
BREAK_GLASS_ISSUE = 664
|
||||
BREAK_GLASS_PENDING_REASON = (
|
||||
"The break-glass workflow (#664) is not yet available on this branch's "
|
||||
"base; no break-glass control is offered and none is implied."
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SourceStatus:
|
||||
"""Whether one backing source could be read, and why not when it could not."""
|
||||
|
||||
name: str
|
||||
status: str
|
||||
detail: str = ""
|
||||
|
||||
@property
|
||||
def available(self) -> bool:
|
||||
return self.status == STATUS_OK
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"name": self.name,
|
||||
"status": self.status,
|
||||
"available": self.available,
|
||||
"detail": self.detail,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RestartClassView:
|
||||
"""One row of the #663 restart-class matrix, scoped to the viewer's role."""
|
||||
|
||||
restart_class: str
|
||||
required_permission: str
|
||||
expected_blast_radius: str
|
||||
drain_requirement: str
|
||||
full_drain_required: bool
|
||||
approval_requirement: str
|
||||
request_roles: tuple[str, ...]
|
||||
execution_roles: tuple[str, ...]
|
||||
viewer_may_request: bool
|
||||
viewer_may_execute: bool
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"restart_class": self.restart_class,
|
||||
"required_permission": self.required_permission,
|
||||
"expected_blast_radius": self.expected_blast_radius,
|
||||
"drain_requirement": self.drain_requirement,
|
||||
"full_drain_required": self.full_drain_required,
|
||||
"approval_requirement": self.approval_requirement,
|
||||
"request_roles": list(self.request_roles),
|
||||
"execution_roles": list(self.execution_roles),
|
||||
"viewer_may_request": self.viewer_may_request,
|
||||
"viewer_may_execute": self.viewer_may_execute,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ActionAuthorization:
|
||||
"""Authorization state for one console action, asked as execution would."""
|
||||
|
||||
action_id: str
|
||||
summary: str
|
||||
required_role: str
|
||||
allowed: bool
|
||||
execution_enabled: bool
|
||||
reason_code: str
|
||||
detail: str
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"action_id": self.action_id,
|
||||
"summary": self.summary,
|
||||
"required_role": self.required_role,
|
||||
"allowed": self.allowed,
|
||||
"execution_enabled": self.execution_enabled,
|
||||
"reason_code": self.reason_code,
|
||||
"detail": self.detail,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class BreakGlassSurface:
|
||||
"""Declared-but-unavailable break-glass panel (#664 is not on this base)."""
|
||||
|
||||
available: bool
|
||||
issue: int
|
||||
reason: str
|
||||
viewer_is_privileged: bool
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"available": self.available,
|
||||
"issue": self.issue,
|
||||
"reason": self.reason,
|
||||
"viewer_is_privileged": self.viewer_is_privileged,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RestartConsoleSnapshot:
|
||||
"""Everything the read-only restart console renders."""
|
||||
|
||||
generated_at: str
|
||||
viewer_role: str
|
||||
viewer_authenticated: bool
|
||||
read_only: bool
|
||||
impact: dict[str, Any] | None
|
||||
impact_source: SourceStatus
|
||||
drain: dict[str, Any] | None
|
||||
drain_source: SourceStatus
|
||||
reconcile: dict[str, Any] | None
|
||||
reconcile_source: SourceStatus
|
||||
restart_classes: tuple[RestartClassView, ...]
|
||||
authorizations: tuple[ActionAuthorization, ...]
|
||||
break_glass: BreakGlassSurface
|
||||
notes: tuple[str, ...] = field(default_factory=tuple)
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"generated_at": self.generated_at,
|
||||
"viewer_role": self.viewer_role,
|
||||
"viewer_authenticated": self.viewer_authenticated,
|
||||
"read_only": self.read_only,
|
||||
"impact": self.impact,
|
||||
"impact_source": self.impact_source.as_dict(),
|
||||
"drain": self.drain,
|
||||
"drain_source": self.drain_source.as_dict(),
|
||||
"reconcile": self.reconcile,
|
||||
"reconcile_source": self.reconcile_source.as_dict(),
|
||||
"restart_classes": [c.as_dict() for c in self.restart_classes],
|
||||
"authorizations": [a.as_dict() for a in self.authorizations],
|
||||
"break_glass": self.break_glass.as_dict(),
|
||||
"notes": list(self.notes),
|
||||
"links": {
|
||||
"issue": 667,
|
||||
"extends": 642,
|
||||
"umbrella": 655,
|
||||
"coordinator": 658,
|
||||
"drain_proof": 661,
|
||||
"reconcile": 662,
|
||||
"restart_classes": 663,
|
||||
"break_glass": BREAK_GLASS_ISSUE,
|
||||
"vision": 652,
|
||||
"roadmap": 653,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
# --- Control-plane inventory (read-only) ------------------------------------
|
||||
|
||||
|
||||
def read_control_plane_inventory(
|
||||
*,
|
||||
db_path: str | None = None,
|
||||
limit: int = 200,
|
||||
) -> dict[str, Any]:
|
||||
"""Read sessions and leases for an impact evaluation, read-only.
|
||||
|
||||
Returns the inventory mapping
|
||||
:func:`restart_coordinator.evaluate_restart_impact` expects.
|
||||
``inventory_complete`` is True only when every read succeeded, so a partial
|
||||
read denies rather than under-reporting the blast radius.
|
||||
|
||||
The database is never created, migrated, or written: a missing file means
|
||||
the console has no session authority, which is not the same as there being
|
||||
no sessions.
|
||||
"""
|
||||
|
||||
path = (db_path or control_plane_db.default_db_path() or "").strip()
|
||||
incomplete: list[str] = []
|
||||
|
||||
def _incomplete(reason: str) -> dict[str, Any]:
|
||||
return {
|
||||
"sessions": [],
|
||||
"leases": [],
|
||||
"terminal_lock": None,
|
||||
"prior_recovery_attempts": [],
|
||||
"inventory_complete": False,
|
||||
"incomplete_reasons": [reason],
|
||||
}
|
||||
|
||||
if not path:
|
||||
return _incomplete("control-plane database path is not configured")
|
||||
if not os.path.exists(path):
|
||||
return _incomplete(
|
||||
f"control-plane database not present at {redact_path(path)}; "
|
||||
"no session or lease authority available"
|
||||
)
|
||||
|
||||
try:
|
||||
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=5)
|
||||
conn.row_factory = sqlite3.Row
|
||||
except sqlite3.Error as exc:
|
||||
return _incomplete(f"control-plane database could not be opened: {exc}")
|
||||
|
||||
sessions: list[dict[str, Any]] = []
|
||||
leases: list[dict[str, Any]] = []
|
||||
capped = max(1, int(limit))
|
||||
try:
|
||||
tables = {
|
||||
str(row[0])
|
||||
for row in conn.execute(
|
||||
"SELECT name FROM sqlite_master WHERE type = 'table'"
|
||||
).fetchall()
|
||||
}
|
||||
if "sessions" not in tables:
|
||||
incomplete.append("control-plane database has no sessions table")
|
||||
else:
|
||||
sessions = [
|
||||
dict(row)
|
||||
for row in conn.execute(
|
||||
"SELECT session_id, role, profile, pid, status,"
|
||||
" last_heartbeat_at FROM sessions"
|
||||
" WHERE status = 'active'"
|
||||
" ORDER BY last_heartbeat_at DESC LIMIT ?",
|
||||
(capped,),
|
||||
).fetchall()
|
||||
]
|
||||
|
||||
if "leases" not in tables:
|
||||
incomplete.append("control-plane database has no leases table")
|
||||
elif "work_items" not in tables:
|
||||
incomplete.append(
|
||||
"control-plane database has no work_items table; lease work "
|
||||
"identity cannot be resolved"
|
||||
)
|
||||
else:
|
||||
leases = [
|
||||
dict(row)
|
||||
for row in conn.execute(
|
||||
"SELECT l.lease_id, l.session_id, l.role, l.phase,"
|
||||
" l.status AS freshness, l.worktree_path,"
|
||||
" w.kind AS work_kind, w.number AS work_number"
|
||||
" FROM leases l"
|
||||
" JOIN work_items w ON w.work_item_id = l.work_item_id"
|
||||
" WHERE l.status = 'active'"
|
||||
" ORDER BY l.expires_at DESC LIMIT ?",
|
||||
(capped,),
|
||||
).fetchall()
|
||||
]
|
||||
except sqlite3.Error as exc:
|
||||
return _incomplete(f"control-plane database read failed: {exc}")
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
return {
|
||||
"sessions": sessions,
|
||||
"leases": leases,
|
||||
"terminal_lock": None,
|
||||
"prior_recovery_attempts": [],
|
||||
"inventory_complete": not incomplete,
|
||||
"incomplete_reasons": incomplete,
|
||||
}
|
||||
|
||||
|
||||
# --- Composition ------------------------------------------------------------
|
||||
|
||||
|
||||
def build_restart_class_views(viewer_role: str | None) -> tuple[RestartClassView, ...]:
|
||||
"""Render the #663 class matrix, marking what this viewer may request."""
|
||||
|
||||
normalized = str(viewer_role or "").strip().lower()
|
||||
views: list[RestartClassView] = []
|
||||
for policy in restart_coordinator.RESTART_CLASS_POLICIES.values():
|
||||
views.append(
|
||||
RestartClassView(
|
||||
restart_class=policy.restart_class.value,
|
||||
required_permission=policy.required_permission,
|
||||
expected_blast_radius=policy.expected_blast_radius,
|
||||
drain_requirement=policy.drain_requirement,
|
||||
full_drain_required=policy.full_drain_required,
|
||||
approval_requirement=policy.approval_requirement,
|
||||
request_roles=tuple(policy.request_roles),
|
||||
execution_roles=tuple(policy.execution_roles),
|
||||
viewer_may_request=normalized in policy.request_roles,
|
||||
viewer_may_execute=normalized in policy.execution_roles,
|
||||
)
|
||||
)
|
||||
return tuple(views)
|
||||
|
||||
|
||||
def build_action_authorizations(
|
||||
principal: console_authz.Principal | None,
|
||||
) -> tuple[ActionAuthorization, ...]:
|
||||
"""Authorization state for the approval controls, asked as execution.
|
||||
|
||||
``for_execution=True`` is deliberate. Asking without it answers "is this
|
||||
principal senior enough", which is not the question an operator looking at a
|
||||
control needs answered; asking with it answers "would this run", and while
|
||||
the console is in Phase 1 the honest answer is no.
|
||||
"""
|
||||
|
||||
results: list[ActionAuthorization] = []
|
||||
for action_id in REPORTED_ACTIONS:
|
||||
action = console_authz.get_action(action_id)
|
||||
decision = console_authz.authorize(action_id, principal, for_execution=True)
|
||||
results.append(
|
||||
ActionAuthorization(
|
||||
action_id=action_id,
|
||||
summary=action.summary if action else "",
|
||||
required_role=(
|
||||
action.minimum_role if action else console_authz.OPERATOR
|
||||
),
|
||||
allowed=bool(decision.allowed),
|
||||
execution_enabled=bool(decision.execution_enabled),
|
||||
reason_code=str(decision.reason_code or ""),
|
||||
detail=str(decision.detail or ""),
|
||||
)
|
||||
)
|
||||
return tuple(results)
|
||||
|
||||
|
||||
def viewer_is_privileged(principal: console_authz.Principal | None) -> bool:
|
||||
"""True when the viewer holds at least the operator role."""
|
||||
|
||||
who = principal if principal is not None else console_authz.ANONYMOUS
|
||||
if not who.authenticated:
|
||||
return False
|
||||
return who.rank >= console_authz.ROLE_ORDER.index(console_authz.OPERATOR)
|
||||
|
||||
|
||||
def load_impact_report(
|
||||
*,
|
||||
principal: console_authz.Principal | None = None,
|
||||
restart_class: str = restart_coordinator.RestartClass.FULL_MCP_RESTART.value,
|
||||
db_path: str | None = None,
|
||||
limit: int = 200,
|
||||
read_inventory: Callable[..., Mapping[str, Any]] | None = None,
|
||||
now: datetime | None = None,
|
||||
) -> tuple[dict[str, Any] | None, SourceStatus]:
|
||||
"""Evaluate the blast radius for *restart_class*, always dry-run."""
|
||||
|
||||
reader = read_inventory or read_control_plane_inventory
|
||||
try:
|
||||
inventory = dict(reader(db_path=db_path, limit=limit))
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return None, SourceStatus(
|
||||
"impact",
|
||||
STATUS_UNAVAILABLE,
|
||||
f"control-plane inventory failed: {type(exc).__name__}: {exc}",
|
||||
)
|
||||
|
||||
who = principal if principal is not None else console_authz.ANONYMOUS
|
||||
viewer_role = str(who.role or "").strip().lower()
|
||||
try:
|
||||
report = restart_coordinator.evaluate_restart_impact(
|
||||
inventory,
|
||||
now=now,
|
||||
dry_run=True,
|
||||
restart_class=restart_class,
|
||||
requester_role=viewer_role,
|
||||
requester_permissions=restart_coordinator.permissions_for_role(
|
||||
viewer_role
|
||||
),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return None, SourceStatus(
|
||||
"impact",
|
||||
STATUS_UNAVAILABLE,
|
||||
f"impact evaluation failed: {type(exc).__name__}: {exc}",
|
||||
)
|
||||
|
||||
payload = scrub(report.as_dict())
|
||||
detail = ""
|
||||
if not report.inventory_complete:
|
||||
detail = "; ".join(report.incomplete_reasons) or "inventory incomplete"
|
||||
return payload, SourceStatus("impact", STATUS_OK, detail)
|
||||
|
||||
|
||||
def load_drain_status(
|
||||
*,
|
||||
proof: Mapping[str, Any] | None = None,
|
||||
now: datetime | None = None,
|
||||
expected_impact_fingerprint: str | None = None,
|
||||
) -> tuple[dict[str, Any] | None, SourceStatus]:
|
||||
"""Verify a supplied drain proof read-only and report the verdict.
|
||||
|
||||
No proof supplied is not a failure and not a pass: it is reported as the
|
||||
absence of a proof, which is exactly what the #661 gate would deny on.
|
||||
"""
|
||||
|
||||
if proof is None:
|
||||
return None, SourceStatus(
|
||||
"drain",
|
||||
STATUS_UNAVAILABLE,
|
||||
"no drain proof supplied; the #661 gate denies a restart without a "
|
||||
"valid unexpired clean proof",
|
||||
)
|
||||
try:
|
||||
verified = drain_proof.verify_drain_proof(
|
||||
proof,
|
||||
now=now,
|
||||
expected_impact_fingerprint=expected_impact_fingerprint,
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return None, SourceStatus(
|
||||
"drain",
|
||||
STATUS_UNAVAILABLE,
|
||||
f"drain proof verification failed: {type(exc).__name__}: {exc}",
|
||||
)
|
||||
return scrub(verified.as_dict()), SourceStatus("drain", STATUS_OK)
|
||||
|
||||
|
||||
def load_reconcile_status(
|
||||
*,
|
||||
load_proof: Callable[[], Any] | None = None,
|
||||
) -> tuple[dict[str, Any] | None, SourceStatus]:
|
||||
"""Report the most recent post-restart completion proof (#662)."""
|
||||
|
||||
if load_proof is None:
|
||||
return None, SourceStatus(
|
||||
"reconcile",
|
||||
STATUS_UNAVAILABLE,
|
||||
"no post-restart completion proof source is wired into this view",
|
||||
)
|
||||
try:
|
||||
proof = load_proof()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return None, SourceStatus(
|
||||
"reconcile",
|
||||
STATUS_UNAVAILABLE,
|
||||
f"reconcile proof unavailable: {type(exc).__name__}: {exc}",
|
||||
)
|
||||
if proof is None:
|
||||
return None, SourceStatus(
|
||||
"reconcile",
|
||||
STATUS_UNAVAILABLE,
|
||||
"no post-restart reconcile has been recorded",
|
||||
)
|
||||
payload = proof.as_dict() if hasattr(proof, "as_dict") else dict(proof)
|
||||
return scrub(payload), SourceStatus("reconcile", STATUS_OK)
|
||||
|
||||
|
||||
def load_restart_console_snapshot(
|
||||
*,
|
||||
principal: console_authz.Principal | None = None,
|
||||
restart_class: str = restart_coordinator.RestartClass.FULL_MCP_RESTART.value,
|
||||
db_path: str | None = None,
|
||||
limit: int = 200,
|
||||
drain_proof_payload: Mapping[str, Any] | None = None,
|
||||
read_inventory: Callable[..., Mapping[str, Any]] | None = None,
|
||||
load_reconcile_proof: Callable[[], Any] | None = None,
|
||||
now: datetime | None = None,
|
||||
) -> RestartConsoleSnapshot:
|
||||
"""Compose the read-only restart console snapshot."""
|
||||
|
||||
who = principal if principal is not None else console_authz.ANONYMOUS
|
||||
moment = now or _utc_now()
|
||||
|
||||
impact, impact_source = load_impact_report(
|
||||
principal=who,
|
||||
restart_class=restart_class,
|
||||
db_path=db_path,
|
||||
limit=limit,
|
||||
read_inventory=read_inventory,
|
||||
now=moment,
|
||||
)
|
||||
fingerprint = None
|
||||
if impact is not None:
|
||||
try:
|
||||
fingerprint = drain_proof.impact_fingerprint(impact)
|
||||
except Exception: # noqa: BLE001
|
||||
fingerprint = None
|
||||
|
||||
drain, drain_source = load_drain_status(
|
||||
proof=drain_proof_payload,
|
||||
now=moment,
|
||||
expected_impact_fingerprint=fingerprint,
|
||||
)
|
||||
reconcile, reconcile_source = load_reconcile_status(
|
||||
load_proof=load_reconcile_proof
|
||||
)
|
||||
|
||||
notes: list[str] = [
|
||||
"This surface is read-only: it evaluates and displays, and performs no "
|
||||
"restart, reload, drain, approval, or process action.",
|
||||
]
|
||||
if not impact_source.available:
|
||||
notes.append(
|
||||
"Impact preview unavailable — a restart decision must not be made "
|
||||
"from this page while the blast radius is unknown."
|
||||
)
|
||||
|
||||
return RestartConsoleSnapshot(
|
||||
generated_at=moment.isoformat(),
|
||||
viewer_role=str(who.role or "anonymous"),
|
||||
viewer_authenticated=bool(who.authenticated),
|
||||
read_only=True,
|
||||
impact=impact,
|
||||
impact_source=impact_source,
|
||||
drain=drain,
|
||||
drain_source=drain_source,
|
||||
reconcile=reconcile,
|
||||
reconcile_source=reconcile_source,
|
||||
restart_classes=build_restart_class_views(who.role),
|
||||
authorizations=build_action_authorizations(who),
|
||||
break_glass=BreakGlassSurface(
|
||||
available=False,
|
||||
issue=BREAK_GLASS_ISSUE,
|
||||
reason=BREAK_GLASS_PENDING_REASON,
|
||||
viewer_is_privileged=viewer_is_privileged(who),
|
||||
),
|
||||
notes=tuple(notes),
|
||||
)
|
||||
@@ -0,0 +1,299 @@
|
||||
"""HTML views for the read-only restart console (#667).
|
||||
|
||||
Every interpolated value passes through :func:`_esc`. Values that can carry a
|
||||
filesystem path or free-form operator text additionally pass through
|
||||
:func:`webui.inventory.scrub_text`, which redacts credential-shaped tokens
|
||||
*inside* a string rather than only at its start.
|
||||
|
||||
The page renders state and never offers a control that would mutate anything:
|
||||
the approval and break-glass panels report authorization and availability, and
|
||||
there is no form, button, or endpoint behind them.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
|
||||
from webui.inventory import scrub_text
|
||||
from webui.restart_console import RestartConsoleSnapshot, SourceStatus
|
||||
|
||||
|
||||
def _esc(value: object) -> str:
|
||||
"""Escape any value for HTML text or a quoted attribute."""
|
||||
if value is None:
|
||||
return ""
|
||||
return html.escape(str(value), quote=True)
|
||||
|
||||
|
||||
def _esc_text(value: object) -> str:
|
||||
"""Escape free-form text after redacting secrets embedded inside it."""
|
||||
if value is None:
|
||||
return ""
|
||||
return _esc(scrub_text(str(value)))
|
||||
|
||||
|
||||
def _bool_badge(
|
||||
value: bool, *, true_label: str = "yes", false_label: str = "no"
|
||||
) -> str:
|
||||
css = "badge-ok" if value else "badge-blocked"
|
||||
label = true_label if value else false_label
|
||||
return f'<span class="badge {css}">{_esc(label)}</span>'
|
||||
|
||||
|
||||
def _source_badge(source: SourceStatus) -> str:
|
||||
css = "badge-ok" if source.available else "badge-blocked"
|
||||
badge = f'<span class="badge {css}">{_esc(source.status)}</span>'
|
||||
if source.detail:
|
||||
badge += f' <span class="muted">{_esc_text(source.detail)}</span>'
|
||||
return badge
|
||||
|
||||
|
||||
def _notes_block(snapshot: RestartConsoleSnapshot) -> str:
|
||||
if not snapshot.notes:
|
||||
return ""
|
||||
items = "".join(f"<li>{_esc_text(note)}</li>" for note in snapshot.notes)
|
||||
return f"<ul class='reasons'>{items}</ul>"
|
||||
|
||||
|
||||
def _impact_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
head = (
|
||||
"<section class='health-card'>"
|
||||
f"<h3>Impact preview {_source_badge(snapshot.impact_source)}</h3>"
|
||||
)
|
||||
impact = snapshot.impact
|
||||
if impact is None:
|
||||
return (
|
||||
head
|
||||
+ "<p class='muted'>No impact preview is available, so the blast "
|
||||
"radius of a restart is unknown. Treat this as unsafe.</p></section>"
|
||||
)
|
||||
|
||||
counts = impact.get("counts") or {}
|
||||
verdict = str(impact.get("verdict") or "unknown")
|
||||
verdict_css = "badge-ok" if verdict == "safe" else "badge-blocked"
|
||||
rows = "".join(
|
||||
f"<tr><th>{_esc(key.replace('_', ' '))}</th><td>{_esc(value)}</td></tr>"
|
||||
for key, value in sorted(counts.items())
|
||||
)
|
||||
reasons = "".join(
|
||||
f"<li>{_esc_text(reason)}</li>" for reason in (impact.get("reasons") or [])
|
||||
)
|
||||
incomplete = ""
|
||||
if not impact.get("inventory_complete", False):
|
||||
detail = "; ".join(str(r) for r in (impact.get("incomplete_reasons") or []))
|
||||
incomplete = (
|
||||
"<p class='error'><strong>Inventory incomplete:</strong> "
|
||||
f"{_esc_text(detail or 'unspecified')}. The coordinator fails "
|
||||
"closed on an incomplete inventory.</p>"
|
||||
)
|
||||
|
||||
sessions = impact.get("affected_sessions") or []
|
||||
session_rows = "".join(
|
||||
"<tr>"
|
||||
f"<td><code>{_esc(s.get('session_id'))}</code></td>"
|
||||
f"<td>{_esc(s.get('role'))}</td>"
|
||||
f"<td>{_esc(s.get('pid'))}</td>"
|
||||
f"<td>{_bool_badge(bool(s.get('live')), true_label='live', false_label='idle')}</td>"
|
||||
f"<td>{_bool_badge(not s.get('heartbeat_stale'), true_label='fresh', false_label='stale')}</td>"
|
||||
"</tr>"
|
||||
for s in sessions[:50]
|
||||
)
|
||||
session_table = (
|
||||
"<h4>Sessions a restart would terminate</h4>"
|
||||
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||
"<th>Session</th><th>Role</th><th>PID</th><th>State</th>"
|
||||
"<th>Heartbeat</th></tr></thead><tbody>"
|
||||
f"{session_rows}</tbody></table></div>"
|
||||
if session_rows
|
||||
else "<p class='muted'>No affected sessions reported.</p>"
|
||||
)
|
||||
truncated = (
|
||||
f"<p class='muted'>Showing the first 50 of {_esc(len(sessions))} "
|
||||
"affected sessions.</p>"
|
||||
if len(sessions) > 50
|
||||
else ""
|
||||
)
|
||||
|
||||
return (
|
||||
head
|
||||
+ "<p class='health-headline'>Verdict "
|
||||
f"<span class='badge {verdict_css}'>{_esc(verdict)}</span> · "
|
||||
f"blast radius <code>{_esc(impact.get('blast_radius'))}</code> · "
|
||||
f"class <code>{_esc(impact.get('restart_class'))}</code></p>"
|
||||
+ incomplete
|
||||
+ (f"<ul class='reasons'>{reasons}</ul>" if reasons else "")
|
||||
+ (f"<table class='registry'><tbody>{rows}</tbody></table>" if rows else "")
|
||||
+ session_table
|
||||
+ truncated
|
||||
+ "</section>"
|
||||
)
|
||||
|
||||
|
||||
def _drain_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
head = (
|
||||
"<section class='health-card'>"
|
||||
f"<h3>Drain proof {_source_badge(snapshot.drain_source)}</h3>"
|
||||
)
|
||||
drain = snapshot.drain
|
||||
if drain is None:
|
||||
return (
|
||||
head
|
||||
+ "<p class='muted'>No drain proof has been presented to this view. "
|
||||
"The #661 gate authorizes a restart only against a valid, unexpired, "
|
||||
"clean proof, so the absence of one is a denial, not a pass.</p>"
|
||||
"</section>"
|
||||
)
|
||||
reasons = "".join(
|
||||
f"<li>{_esc_text(reason)}</li>" for reason in (drain.get("reasons") or [])
|
||||
)
|
||||
return (
|
||||
head
|
||||
+ "<table class='registry'><tbody>"
|
||||
f"<tr><th>Valid</th><td>{_bool_badge(bool(drain.get('valid')))}</td></tr>"
|
||||
f"<tr><th>Clean</th><td>{_bool_badge(bool(drain.get('clean')))}</td></tr>"
|
||||
f"<tr><th>Expired</th><td>{_bool_badge(not drain.get('expired'), true_label='no', false_label='yes')}</td></tr>"
|
||||
f"<tr><th>Tampered</th><td>{_bool_badge(not drain.get('tampered'), true_label='no', false_label='yes')}</td></tr>"
|
||||
f"<tr><th>Proof id</th><td><code>{_esc(drain.get('proof_id'))}</code></td></tr>"
|
||||
"</tbody></table>"
|
||||
+ (f"<ul class='reasons'>{reasons}</ul>" if reasons else "")
|
||||
+ "</section>"
|
||||
)
|
||||
|
||||
|
||||
def _reconcile_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
head = (
|
||||
"<section class='health-card'>"
|
||||
f"<h3>Post-restart reconcile {_source_badge(snapshot.reconcile_source)}</h3>"
|
||||
)
|
||||
proof = snapshot.reconcile
|
||||
if proof is None:
|
||||
return (
|
||||
head
|
||||
+ "<p class='muted'>No post-restart completion proof is recorded. "
|
||||
"Until one is, the last restart's recovery state is unproven.</p>"
|
||||
"</section>"
|
||||
)
|
||||
items = "".join(
|
||||
"<tr>"
|
||||
f"<td>{_esc(item.get('dimension'))}</td>"
|
||||
f"<td>{_esc(item.get('status'))}</td>"
|
||||
f"<td>{_esc_text(item.get('summary'))}</td>"
|
||||
f"<td>{_bool_badge(not item.get('follow_up_required'), true_label='no', false_label='yes')}</td>"
|
||||
"</tr>"
|
||||
for item in (proof.get("items") or [])
|
||||
)
|
||||
return (
|
||||
head
|
||||
+ "<p class='health-headline'>Status "
|
||||
f"<code>{_esc(proof.get('overall_status'))}</code> · mode "
|
||||
f"<code>{_esc(proof.get('mode'))}</code> · resolved "
|
||||
f"{_esc(proof.get('resolved_count'))} · unresolved "
|
||||
f"{_esc(proof.get('unresolved_count'))}</p>"
|
||||
+ (
|
||||
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||
"<th>Dimension</th><th>Status</th><th>Summary</th>"
|
||||
"<th>Follow-up required</th></tr></thead><tbody>"
|
||||
f"{items}</tbody></table></div>"
|
||||
if items
|
||||
else "<p class='muted'>No reconcile dimensions reported.</p>"
|
||||
)
|
||||
+ "</section>"
|
||||
)
|
||||
|
||||
|
||||
def _class_matrix_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
rows = "".join(
|
||||
"<tr>"
|
||||
f"<td><code>{_esc(view.restart_class)}</code></td>"
|
||||
f"<td><code>{_esc(view.required_permission)}</code></td>"
|
||||
f"<td>{_esc(view.expected_blast_radius)}</td>"
|
||||
f"<td>{_esc(view.drain_requirement)}</td>"
|
||||
f"<td>{_esc(view.approval_requirement)}</td>"
|
||||
f"<td>{_bool_badge(view.viewer_may_request)}</td>"
|
||||
f"<td>{_bool_badge(view.viewer_may_execute)}</td>"
|
||||
"</tr>"
|
||||
for view in snapshot.restart_classes
|
||||
)
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Restart classes</h3>"
|
||||
"<p class='muted'>The least-privilege matrix each restart request is "
|
||||
"resolved against. “You may request” and “you may "
|
||||
"execute” are computed for the current viewer role, not for a "
|
||||
"generic operator.</p>"
|
||||
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||
"<th>Class</th><th>Permission</th><th>Blast radius</th>"
|
||||
"<th>Drain</th><th>Approval</th><th>You may request</th>"
|
||||
"<th>You may execute</th></tr></thead><tbody>"
|
||||
f"{rows}</tbody></table></div>"
|
||||
"</section>"
|
||||
)
|
||||
|
||||
|
||||
def _approval_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
rows = "".join(
|
||||
"<tr>"
|
||||
f"<td><code>{_esc(a.action_id)}</code></td>"
|
||||
f"<td>{_esc(a.required_role)}</td>"
|
||||
f"<td>{_bool_badge(a.allowed)}</td>"
|
||||
f"<td>{_bool_badge(a.execution_enabled)}</td>"
|
||||
f"<td><code>{_esc(a.reason_code)}</code></td>"
|
||||
f"<td>{_esc_text(a.detail)}</td>"
|
||||
"</tr>"
|
||||
for a in snapshot.authorizations
|
||||
)
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Approval controls</h3>"
|
||||
"<p class='muted'>Authorization is probed the way execution would probe "
|
||||
"it, so “execution enabled” answers whether the action would "
|
||||
"actually run — not merely whether this role outranks the requirement. "
|
||||
"No control on this page performs the action.</p>"
|
||||
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||
"<th>Action</th><th>Required role</th><th>Authorized</th>"
|
||||
"<th>Execution enabled</th><th>Reason</th><th>Detail</th>"
|
||||
"</tr></thead><tbody>"
|
||||
f"{rows}</tbody></table></div>"
|
||||
"</section>"
|
||||
)
|
||||
|
||||
|
||||
def _break_glass_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||
bg = snapshot.break_glass
|
||||
if not bg.viewer_is_privileged:
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Break-glass</h3>"
|
||||
"<p class='muted'>Break-glass status is visible to operator-class "
|
||||
"roles only. Your role does not carry that authority, so no "
|
||||
"emergency surface is shown.</p>"
|
||||
"</section>"
|
||||
)
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Break-glass "
|
||||
f"{_bool_badge(bg.available, true_label='available', false_label='unavailable')}"
|
||||
"</h3>"
|
||||
f"<p class='muted'>{_esc_text(bg.reason)}</p>"
|
||||
f"<p class='meta'>Tracked by issue #{_esc(bg.issue)}.</p>"
|
||||
"</section>"
|
||||
)
|
||||
|
||||
|
||||
def render_restart_console_page(snapshot: RestartConsoleSnapshot) -> str:
|
||||
"""Render the whole read-only restart console body."""
|
||||
|
||||
return (
|
||||
"<h2>Restart status and impact</h2>"
|
||||
f"<p class='meta'>Generated <code>{_esc(snapshot.generated_at)}</code> · "
|
||||
f"viewer role <code>{_esc(snapshot.viewer_role)}</code> · "
|
||||
f"authenticated {_bool_badge(snapshot.viewer_authenticated)} · "
|
||||
f"read-only {_bool_badge(snapshot.read_only)}</p>"
|
||||
+ _notes_block(snapshot)
|
||||
+ _impact_section(snapshot)
|
||||
+ _drain_section(snapshot)
|
||||
+ _reconcile_section(snapshot)
|
||||
+ _class_matrix_section(snapshot)
|
||||
+ _approval_section(snapshot)
|
||||
+ _break_glass_section(snapshot)
|
||||
)
|
||||
@@ -88,5 +88,7 @@ def render_runtime_page(snapshot: RuntimeSnapshot) -> str:
|
||||
"<p class='muted'>MVP is read-only — restart MCP servers from your IDE/operator "
|
||||
"workflow. Related issue: <code>#420</code>. Guidance: "
|
||||
f"<code>{html.escape(snapshot.restart_guidance)}</code></p>"
|
||||
"<p class='muted'>This page does not expose tokens or perform MCP restarts.</p>"
|
||||
)
|
||||
"<p class='muted'>This page does not expose tokens or perform MCP restarts. "
|
||||
"Correlated sessions, worktree bindings, and contamination markers: "
|
||||
"<a href='/sessions'>/sessions</a> (#641).</p>"
|
||||
)
|
||||
|
||||
@@ -38,6 +38,7 @@ from dataclasses import asdict, dataclass
|
||||
from typing import Any
|
||||
|
||||
import mcp_namespace_health
|
||||
import restart_coordinator
|
||||
import runtime_recovery_guard
|
||||
from webui import console_audit, console_authz
|
||||
|
||||
@@ -99,6 +100,14 @@ def _clean(value: Any) -> str:
|
||||
return str(value or "").strip()
|
||||
|
||||
|
||||
def restart_class_for_mode(mode: str) -> str:
|
||||
"""Map the existing namespace controls onto the #663 class taxonomy."""
|
||||
|
||||
if _clean(mode) == MODE_RELOAD:
|
||||
return restart_coordinator.RestartClass.CONFIGURATION_RELOAD.value
|
||||
return restart_coordinator.RestartClass.ROLE_RUNTIME_RESTART.value
|
||||
|
||||
|
||||
# --- Mutation ledger --------------------------------------------------------
|
||||
|
||||
|
||||
@@ -256,6 +265,7 @@ def build_restart_preview(
|
||||
|
||||
return {
|
||||
"action_id": action_id,
|
||||
"restart_class": restart_class_for_mode(md),
|
||||
"namespace": ns,
|
||||
"mode": md,
|
||||
"scope_valid": scope_error is None,
|
||||
@@ -309,6 +319,7 @@ def assess_restart_request(
|
||||
"reason_code": reason_code,
|
||||
"detail": detail,
|
||||
"action_id": action_id,
|
||||
"restart_class": restart_class_for_mode(md),
|
||||
"namespace": ns,
|
||||
"mode": md,
|
||||
"preview": preview,
|
||||
@@ -393,6 +404,7 @@ def assess_restart_request(
|
||||
"process."
|
||||
),
|
||||
"action_id": action_id,
|
||||
"restart_class": restart_class_for_mode(md),
|
||||
"namespace": ns,
|
||||
"mode": md,
|
||||
"preview": preview,
|
||||
@@ -441,7 +453,11 @@ def execute_restart(
|
||||
else console_audit.RESULT_DENIED
|
||||
),
|
||||
principal=principal,
|
||||
target={"namespace": assessment["namespace"], "mode": assessment["mode"]},
|
||||
target={
|
||||
"namespace": assessment["namespace"],
|
||||
"mode": assessment["mode"],
|
||||
"restart_class": assessment["restart_class"],
|
||||
},
|
||||
reason_code=assessment["reason_code"],
|
||||
detail=assessment["detail"],
|
||||
request_id=request_id,
|
||||
@@ -450,6 +466,7 @@ def execute_restart(
|
||||
"gates_passed": assessment["gates_passed"],
|
||||
"process_kill_executed": False,
|
||||
"post_restart_verification_required": True,
|
||||
"restart_class": assessment["restart_class"],
|
||||
},
|
||||
)
|
||||
|
||||
@@ -463,6 +480,7 @@ def execute_restart(
|
||||
"namespace": assessment["namespace"],
|
||||
"mode": assessment["mode"],
|
||||
"action_id": action_id,
|
||||
"restart_class": assessment["restart_class"],
|
||||
"process_kill_executed": False,
|
||||
"host_hook": assessment["preview"]["restart_hook"],
|
||||
"next_action": (
|
||||
|
||||
@@ -0,0 +1,517 @@
|
||||
"""Compose runtime health + inventory into a sessions/runtime view (#641).
|
||||
|
||||
Phase 1 is read-only. It correlates namespaces, sessions, capabilities,
|
||||
worktree bindings, lease ownership, stale flags, and contamination markers
|
||||
when they are detectable on disk (#630 / #671). It never restarts, kills, or
|
||||
takes over a session.
|
||||
|
||||
Sources:
|
||||
|
||||
* :mod:`webui.runtime_health` — profile, role, stale runtime, shell health.
|
||||
* :mod:`webui.inventory` — sessions, leases, locks, worktrees, namespaces
|
||||
and collision signals from the control-plane DB + filesystem.
|
||||
* :mod:`mcp_session_state` — durable contamination markers (inspect only).
|
||||
|
||||
Secrets are never read. Inventory-sourced values arrive already redacted by
|
||||
:func:`webui.inventory.scrub`; free-form marker text this module loads itself is
|
||||
put through :func:`webui.inventory.scrub_text`, which collapses ``$HOME`` and
|
||||
redacts credential-shaped tokens *inside* a string rather than only at its start.
|
||||
|
||||
Ownership columns are authority-aware. A lease or lock section that could not be
|
||||
read renders as ``unknown``, never as ``none`` or ``unbound``: a lease the reader
|
||||
could not load is not an absent lease (see
|
||||
:attr:`webui.inventory.InventorySnapshot.ownership_authority_complete`).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Callable
|
||||
|
||||
import mcp_session_state
|
||||
from webui.inventory import (
|
||||
OWNERSHIP_SECTIONS,
|
||||
STATUS_OK,
|
||||
InventorySection,
|
||||
InventorySnapshot,
|
||||
load_inventory_snapshot,
|
||||
scrub_text,
|
||||
snapshot_to_dict as inventory_snapshot_to_dict,
|
||||
)
|
||||
from webui.runtime_health import (
|
||||
RuntimeSnapshot,
|
||||
load_runtime_snapshot,
|
||||
snapshot_to_dict as runtime_snapshot_to_dict,
|
||||
)
|
||||
|
||||
# Sanctioned recovery pointers only — never pkill / killall (#630).
|
||||
SANCTIONED_RECOVERY_DOCS: tuple[dict[str, str], ...] = (
|
||||
{
|
||||
"label": "MCP namespace EOF recovery (reconnect only)",
|
||||
"path": "docs/mcp-namespace-eof-recovery.md",
|
||||
"note": "IDE/client reconnect or operator-owned restart; never kill daemons.",
|
||||
},
|
||||
{
|
||||
"label": "MCP namespace health",
|
||||
"path": "docs/mcp-namespace-health.md",
|
||||
"note": "client_namespace probe proves namespace health.",
|
||||
},
|
||||
{
|
||||
"label": "Restart path inventory",
|
||||
"path": "docs/mcp-restart-path-inventory.md",
|
||||
"note": "Catalog of sanctioned reconnect/restart paths.",
|
||||
},
|
||||
{
|
||||
"label": "Local web UI recovery",
|
||||
"path": "docs/webui-local-dev.md",
|
||||
"note": "Operator console start and documented recovery sequence.",
|
||||
},
|
||||
)
|
||||
|
||||
_CONTAMINATION_KINDS: tuple[str, ...] = (
|
||||
mcp_session_state.KIND_RUNTIME_RECOVERY_CONTAMINATION,
|
||||
mcp_session_state.KIND_STABLE_BRANCH_CONTAMINATION,
|
||||
)
|
||||
|
||||
#: Status recorded on a row when the backing inventory section is absent
|
||||
#: entirely — distinct from a section that reported itself degraded.
|
||||
AUTHORITY_MISSING = "missing"
|
||||
|
||||
|
||||
def _section_status(section: InventorySection | None) -> str:
|
||||
"""Status of an ownership section, treating an absent section as missing."""
|
||||
if section is None:
|
||||
return AUTHORITY_MISSING
|
||||
return section.status
|
||||
|
||||
|
||||
def _combined_authority(*statuses: str) -> str:
|
||||
"""Worst status of the sections a derived column depends on.
|
||||
|
||||
A column proved from two sections is only trustworthy when *both* read
|
||||
cleanly, so the first non-``ok`` status wins.
|
||||
"""
|
||||
for status in statuses:
|
||||
if status != STATUS_OK:
|
||||
return status
|
||||
return STATUS_OK
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ContaminationMarker:
|
||||
"""A detectable durable contamination marker (audit-safe summary)."""
|
||||
|
||||
kind: str
|
||||
on_disk: bool
|
||||
has_payload: bool
|
||||
summary: str
|
||||
reason_class: str | None = None
|
||||
session_id: str | None = None
|
||||
role: str | None = None
|
||||
command_summary: str | None = None
|
||||
cleared: bool = False
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"kind": self.kind,
|
||||
"on_disk": self.on_disk,
|
||||
"has_payload": self.has_payload,
|
||||
"summary": self.summary,
|
||||
"reason_class": self.reason_class,
|
||||
"session_id": self.session_id,
|
||||
"role": self.role,
|
||||
"command_summary": self.command_summary,
|
||||
"cleared": self.cleared,
|
||||
"active": self.on_disk and self.has_payload and not self.cleared,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SessionRow:
|
||||
"""One correlated session row for the sessions table."""
|
||||
|
||||
session_id: str
|
||||
role: str | None
|
||||
profile: str | None
|
||||
namespace: str | None
|
||||
pid: int | None
|
||||
pid_alive: bool | None
|
||||
status: str | None
|
||||
started_at: str | None
|
||||
last_heartbeat_at: str | None
|
||||
lease_ids: tuple[str, ...] = ()
|
||||
work_refs: tuple[str, ...] = ()
|
||||
worktree_paths: tuple[str, ...] = ()
|
||||
stale_flags: tuple[str, ...] = ()
|
||||
contamination_flags: tuple[str, ...] = ()
|
||||
#: Status of the section backing ``lease_ids``/``work_refs``. While this is
|
||||
#: not ``ok`` those tuples mean "could not be read", never "none held".
|
||||
lease_authority: str = STATUS_OK
|
||||
#: Worst status across the sections backing ``worktree_paths`` (locks are
|
||||
#: correlated through lease work numbers, so both must read cleanly).
|
||||
worktree_authority: str = STATUS_OK
|
||||
|
||||
@property
|
||||
def ownership_authority_complete(self) -> bool:
|
||||
"""True only when this row's ownership columns are provable."""
|
||||
return self.lease_authority == STATUS_OK and self.worktree_authority == STATUS_OK
|
||||
|
||||
def to_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"session_id": self.session_id,
|
||||
"role": self.role,
|
||||
"profile": self.profile,
|
||||
"namespace": self.namespace,
|
||||
"pid": self.pid,
|
||||
"pid_alive": self.pid_alive,
|
||||
"status": self.status,
|
||||
"started_at": self.started_at,
|
||||
"last_heartbeat_at": self.last_heartbeat_at,
|
||||
"lease_ids": list(self.lease_ids),
|
||||
"work_refs": list(self.work_refs),
|
||||
"worktree_paths": list(self.worktree_paths),
|
||||
"stale_flags": list(self.stale_flags),
|
||||
"contamination_flags": list(self.contamination_flags),
|
||||
"is_stale": bool(self.stale_flags),
|
||||
"is_contaminated": bool(self.contamination_flags),
|
||||
"lease_authority": self.lease_authority,
|
||||
"worktree_authority": self.worktree_authority,
|
||||
"lease_authority_complete": self.lease_authority == STATUS_OK,
|
||||
"worktree_authority_complete": self.worktree_authority == STATUS_OK,
|
||||
"ownership_authority_complete": self.ownership_authority_complete,
|
||||
"ownership_note": (
|
||||
"Lease and worktree columns are proved from sections that read "
|
||||
"cleanly."
|
||||
if self.ownership_authority_complete
|
||||
else "An ownership source could not be read; empty lease_ids or "
|
||||
"worktree_paths on this row mean unknown, not unowned."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SessionViewSnapshot:
|
||||
"""Composed runtime + session inventory view (#641)."""
|
||||
|
||||
runtime: RuntimeSnapshot
|
||||
inventory: InventorySnapshot
|
||||
sessions: tuple[SessionRow, ...]
|
||||
contamination_markers: tuple[ContaminationMarker, ...]
|
||||
recovery_docs: tuple[dict[str, str], ...] = SANCTIONED_RECOVERY_DOCS
|
||||
fetch_error: str | None = None
|
||||
|
||||
@property
|
||||
def stale_session_count(self) -> int:
|
||||
return sum(1 for row in self.sessions if row.stale_flags)
|
||||
|
||||
@property
|
||||
def contaminated_session_count(self) -> int:
|
||||
return sum(1 for row in self.sessions if row.contamination_flags)
|
||||
|
||||
@property
|
||||
def active_contamination(self) -> tuple[ContaminationMarker, ...]:
|
||||
return tuple(m for m in self.contamination_markers if m.to_dict()["active"])
|
||||
|
||||
@property
|
||||
def ownership_authority_complete(self) -> bool:
|
||||
"""False while any ownership section is degraded, absent, or unavailable."""
|
||||
return self.inventory.ownership_authority_complete
|
||||
|
||||
@property
|
||||
def ownership_section_status(self) -> dict[str, str]:
|
||||
"""Per-section status for the three ownership-bearing sections."""
|
||||
return {
|
||||
name: _section_status(self.inventory.section(name))
|
||||
for name in OWNERSHIP_SECTIONS
|
||||
}
|
||||
|
||||
|
||||
def _inspect_contamination(
|
||||
kind: str,
|
||||
*,
|
||||
remote: str | None,
|
||||
inspect: Callable[..., dict[str, Any]] | None = None,
|
||||
load: Callable[..., dict[str, Any] | None] | None = None,
|
||||
) -> ContaminationMarker:
|
||||
"""Inspect one contamination kind; never raises into the page render path."""
|
||||
inspect_fn = inspect or mcp_session_state.inspect_state_envelope
|
||||
load_fn = load or mcp_session_state.load_state
|
||||
try:
|
||||
envelope = inspect_fn(kind=kind, remote=remote)
|
||||
except Exception as exc: # noqa: BLE001 — fail soft for the dashboard
|
||||
return ContaminationMarker(
|
||||
kind=kind,
|
||||
on_disk=False,
|
||||
has_payload=False,
|
||||
summary=f"contamination inspect failed: {type(exc).__name__}",
|
||||
)
|
||||
|
||||
reason_class = None
|
||||
session_id = None
|
||||
role = None
|
||||
command_summary = None
|
||||
cleared = False
|
||||
summary = scrub_text(str(envelope.get("summary") or ""))
|
||||
|
||||
if envelope.get("on_disk") and envelope.get("has_payload"):
|
||||
try:
|
||||
payload = load_fn(kind=kind, remote=remote) or {}
|
||||
except Exception: # noqa: BLE001
|
||||
payload = {}
|
||||
if isinstance(payload, dict):
|
||||
reason_class = payload.get("reason_class")
|
||||
session_id = payload.get("session_id")
|
||||
role = payload.get("role")
|
||||
command_summary = payload.get("command_summary") or payload.get("detail")
|
||||
cleared = bool(payload.get("cleared_by_reconciler"))
|
||||
if not summary:
|
||||
summary = scrub_text(
|
||||
f"{kind}: {reason_class or 'present'}"
|
||||
+ (" (cleared)" if cleared else "")
|
||||
)
|
||||
|
||||
# Marker payloads are operator-supplied free text and are the one thing on
|
||||
# this page that does not arrive through inventory scrubbing. The write-time
|
||||
# redactor is a narrow denylist, so redact again at the display boundary:
|
||||
# it leaves $HOME paths, `-H 'X-Api-Key: …'`, `--password …`, and
|
||||
# `PRIVATE_KEY=…` intact. The field itself stays — it is #630 evidence.
|
||||
return ContaminationMarker(
|
||||
kind=kind,
|
||||
on_disk=bool(envelope.get("on_disk")),
|
||||
has_payload=bool(envelope.get("has_payload")),
|
||||
summary=summary or f"{kind}: not present",
|
||||
reason_class=scrub_text(str(reason_class)) if reason_class else None,
|
||||
session_id=scrub_text(str(session_id)) if session_id else None,
|
||||
role=scrub_text(str(role)) if role else None,
|
||||
command_summary=scrub_text(str(command_summary)) if command_summary else None,
|
||||
cleared=cleared,
|
||||
)
|
||||
|
||||
|
||||
def _load_contamination_markers(
|
||||
*,
|
||||
remote: str | None,
|
||||
inspect: Callable[..., dict[str, Any]] | None = None,
|
||||
load: Callable[..., dict[str, Any] | None] | None = None,
|
||||
) -> tuple[ContaminationMarker, ...]:
|
||||
return tuple(
|
||||
_inspect_contamination(kind, remote=remote, inspect=inspect, load=load)
|
||||
for kind in _CONTAMINATION_KINDS
|
||||
)
|
||||
|
||||
|
||||
def _build_session_rows(
|
||||
inventory: InventorySnapshot,
|
||||
contamination: tuple[ContaminationMarker, ...],
|
||||
) -> tuple[SessionRow, ...]:
|
||||
sessions_section = inventory.section("sessions")
|
||||
leases_section = inventory.section("leases")
|
||||
locks_section = inventory.section("locks")
|
||||
|
||||
# Ownership columns may only assert absence when their source read cleanly.
|
||||
lease_authority = _section_status(leases_section)
|
||||
# Locks carry claimant profile/username, not control-plane session ids, so a
|
||||
# worktree binding is correlated through lease work numbers: it depends on
|
||||
# the locks *and* the leases section.
|
||||
worktree_authority = _combined_authority(
|
||||
lease_authority, _section_status(locks_section)
|
||||
)
|
||||
|
||||
leases_by_session: dict[str, list[dict[str, Any]]] = {}
|
||||
for lease in (leases_section.items if leases_section else ()):
|
||||
sid = str(lease.get("session_id") or "")
|
||||
if sid:
|
||||
leases_by_session.setdefault(sid, []).append(lease)
|
||||
|
||||
active_markers = [m for m in contamination if m.to_dict()["active"]]
|
||||
marker_session_ids = {
|
||||
m.session_id for m in active_markers if m.session_id
|
||||
}
|
||||
|
||||
rows: list[SessionRow] = []
|
||||
for raw in sessions_section.items if sessions_section else ():
|
||||
sid = str(raw.get("session_id") or "")
|
||||
if not sid:
|
||||
continue
|
||||
session_leases = leases_by_session.get(sid, [])
|
||||
lease_ids = tuple(
|
||||
str(lease["lease_id"])
|
||||
for lease in session_leases
|
||||
if lease.get("lease_id")
|
||||
)
|
||||
work_refs: list[str] = []
|
||||
work_numbers: list[int] = []
|
||||
for lease in session_leases:
|
||||
kind = lease.get("work_kind")
|
||||
number = lease.get("work_number")
|
||||
if kind and number is not None:
|
||||
work_refs.append(f"{kind}#{number}")
|
||||
try:
|
||||
work_numbers.append(int(number))
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
|
||||
worktree_paths: list[str] = []
|
||||
for lock in (locks_section.items if locks_section else ()):
|
||||
try:
|
||||
issue_no = int(lock.get("issue_number"))
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
if issue_no in work_numbers and lock.get("worktree_path"):
|
||||
worktree_paths.append(str(lock["worktree_path"]))
|
||||
|
||||
stale_flags: list[str] = []
|
||||
if raw.get("pid_alive") is False:
|
||||
stale_flags.append("pid-dead")
|
||||
status = str(raw.get("status") or "").lower()
|
||||
if status and status not in {"active", "alive", "running", "ok"}:
|
||||
stale_flags.append(f"status:{status}")
|
||||
for lease in session_leases:
|
||||
if lease.get("expired") is True:
|
||||
stale_flags.append("lease-expired")
|
||||
if str(lease.get("status") or "").lower() == "active" and lease.get(
|
||||
"expired"
|
||||
) is True:
|
||||
stale_flags.append("active-lease-past-expiry")
|
||||
|
||||
contamination_flags: list[str] = []
|
||||
if sid in marker_session_ids:
|
||||
for marker in active_markers:
|
||||
if marker.session_id == sid:
|
||||
contamination_flags.append(marker.kind)
|
||||
# Process-wide contamination with no session binding still surfaces
|
||||
# against every live session so it cannot be silent (#630).
|
||||
for marker in active_markers:
|
||||
if not marker.session_id and marker.kind not in contamination_flags:
|
||||
contamination_flags.append(f"{marker.kind}:process-wide")
|
||||
|
||||
rows.append(
|
||||
SessionRow(
|
||||
session_id=sid,
|
||||
role=raw.get("role"),
|
||||
profile=raw.get("profile"),
|
||||
namespace=raw.get("namespace"),
|
||||
pid=raw.get("pid") if isinstance(raw.get("pid"), int) else None,
|
||||
pid_alive=raw.get("pid_alive")
|
||||
if isinstance(raw.get("pid_alive"), bool)
|
||||
else None,
|
||||
status=raw.get("status"),
|
||||
started_at=raw.get("started_at"),
|
||||
last_heartbeat_at=raw.get("last_heartbeat_at"),
|
||||
lease_ids=lease_ids,
|
||||
work_refs=tuple(work_refs),
|
||||
worktree_paths=tuple(worktree_paths),
|
||||
stale_flags=tuple(dict.fromkeys(stale_flags)),
|
||||
contamination_flags=tuple(dict.fromkeys(contamination_flags)),
|
||||
lease_authority=lease_authority,
|
||||
worktree_authority=worktree_authority,
|
||||
)
|
||||
)
|
||||
|
||||
return tuple(rows)
|
||||
|
||||
|
||||
def load_session_view_snapshot(
|
||||
*,
|
||||
load_runtime: Callable[..., RuntimeSnapshot] | None = None,
|
||||
load_inventory: Callable[..., InventorySnapshot] | None = None,
|
||||
inspect_contamination: Callable[..., dict[str, Any]] | None = None,
|
||||
load_contamination_payload: Callable[..., dict[str, Any] | None] | None = None,
|
||||
) -> SessionViewSnapshot:
|
||||
"""Build the composed sessions/runtime view. Fail-soft on partial sources."""
|
||||
runtime_loader = load_runtime or load_runtime_snapshot
|
||||
inventory_loader = load_inventory or load_inventory_snapshot
|
||||
fetch_error: str | None = None
|
||||
|
||||
try:
|
||||
runtime = runtime_loader()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
fetch_error = f"runtime snapshot failed: {type(exc).__name__}: {exc}"
|
||||
# Minimal placeholder so the page still renders inventory + recovery.
|
||||
from webui.runtime_health import RuntimeSnapshot as _RS
|
||||
|
||||
runtime = _RS(
|
||||
project_id="unknown",
|
||||
repo_root="",
|
||||
remote="",
|
||||
host="",
|
||||
profile_name="unknown",
|
||||
role_kind="unknown",
|
||||
config_model="unknown",
|
||||
profile_mode="unknown",
|
||||
profile_source="unknown",
|
||||
authenticated_username=None,
|
||||
identity_error=str(exc),
|
||||
repo_sha=None,
|
||||
remote_master_sha=None,
|
||||
commits_behind_master=None,
|
||||
stale_runtime_warning=None,
|
||||
shell_health={},
|
||||
workflow_hashes=(),
|
||||
schema_hashes=(),
|
||||
restart_guidance="docs/mcp-namespace-eof-recovery.md",
|
||||
fetch_error=str(exc),
|
||||
)
|
||||
|
||||
try:
|
||||
inventory = inventory_loader()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
msg = f"inventory snapshot failed: {type(exc).__name__}: {exc}"
|
||||
fetch_error = f"{fetch_error}; {msg}" if fetch_error else msg
|
||||
inventory = load_inventory_snapshot(
|
||||
db_path="/nonexistent-for-fail-soft",
|
||||
lock_dir="/nonexistent-for-fail-soft",
|
||||
)
|
||||
|
||||
remote = getattr(runtime, "remote", None)
|
||||
contamination = _load_contamination_markers(
|
||||
remote=remote,
|
||||
inspect=inspect_contamination,
|
||||
load=load_contamination_payload,
|
||||
)
|
||||
sessions = _build_session_rows(inventory, contamination)
|
||||
return SessionViewSnapshot(
|
||||
runtime=runtime,
|
||||
inventory=inventory,
|
||||
sessions=sessions,
|
||||
contamination_markers=contamination,
|
||||
fetch_error=fetch_error,
|
||||
)
|
||||
|
||||
|
||||
def snapshot_to_dict(snapshot: SessionViewSnapshot) -> dict[str, Any]:
|
||||
"""JSON export for ``/api/sessions`` (read-only)."""
|
||||
return {
|
||||
"api_version": "v1",
|
||||
"view": "runtime-sessions",
|
||||
"issue": 641,
|
||||
"fetch_error": snapshot.fetch_error,
|
||||
"runtime": runtime_snapshot_to_dict(snapshot.runtime),
|
||||
"inventory": inventory_snapshot_to_dict(snapshot.inventory),
|
||||
"sessions": [row.to_dict() for row in snapshot.sessions],
|
||||
"session_counts": {
|
||||
"total": len(snapshot.sessions),
|
||||
"stale": snapshot.stale_session_count,
|
||||
"contaminated": snapshot.contaminated_session_count,
|
||||
},
|
||||
"ownership_authority_complete": snapshot.ownership_authority_complete,
|
||||
"ownership_section_status": snapshot.ownership_section_status,
|
||||
"ownership_note": (
|
||||
"Every ownership source read cleanly; a session with no lease and no "
|
||||
"worktree path genuinely holds neither."
|
||||
if snapshot.ownership_authority_complete
|
||||
else "An ownership source is degraded or unavailable. Empty lease_ids "
|
||||
"and worktree_paths mean unknown, not unowned; consult each row's "
|
||||
"lease_authority and worktree_authority."
|
||||
),
|
||||
"contamination_markers": [
|
||||
marker.to_dict() for marker in snapshot.contamination_markers
|
||||
],
|
||||
"active_contamination": [
|
||||
marker.to_dict() for marker in snapshot.active_contamination
|
||||
],
|
||||
"recovery_docs": [dict(doc) for doc in snapshot.recovery_docs],
|
||||
"read_only": True,
|
||||
"phase": 1,
|
||||
"mutations": [],
|
||||
}
|
||||
@@ -0,0 +1,403 @@
|
||||
"""HTML views for the Runtime and session view (Phase 1, #641).
|
||||
|
||||
Read-only composition of runtime health (#430) and inventory sessions /
|
||||
namespaces / worktrees (#636). Surfaces stale and contamination indicators
|
||||
when detectable. Recovery links point only at sanctioned reconnect/restart
|
||||
docs — never at manual process kill (#630).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from html import escape
|
||||
from typing import Sequence
|
||||
|
||||
from webui.inventory import STATUS_OK
|
||||
from webui.layout import render_page
|
||||
from webui.session_loader import (
|
||||
ContaminationMarker,
|
||||
SessionRow,
|
||||
SessionViewSnapshot,
|
||||
)
|
||||
|
||||
|
||||
def _badge(text: str, css: str) -> str:
|
||||
return f'<span class="badge {css}">{escape(text)}</span>'
|
||||
|
||||
|
||||
def _flags(flags: Sequence[str], *, css: str) -> str:
|
||||
if not flags:
|
||||
return '<span class="muted">—</span>'
|
||||
return " ".join(_badge(flag, css) for flag in flags)
|
||||
|
||||
|
||||
def _unproven_cell(status: str) -> str:
|
||||
"""Render an ownership column whose backing inventory section failed to read.
|
||||
|
||||
Never "none" and never "unbound": an unreadable source proves nothing about
|
||||
ownership, and claiming otherwise is the exact failure the
|
||||
``ownership_authority_complete`` invariant exists to prevent.
|
||||
"""
|
||||
return (
|
||||
f'<span class="muted">unknown (inventory {escape(status)})</span><br>'
|
||||
f'{_badge("authority unproven", "badge-health-degraded")}'
|
||||
)
|
||||
|
||||
|
||||
def _runtime_banner(snapshot: SessionViewSnapshot) -> str:
|
||||
runtime = snapshot.runtime
|
||||
stale = runtime.stale_runtime_warning
|
||||
stale_html = ""
|
||||
if stale:
|
||||
stale_html = (
|
||||
f'<div class="health-card health-stale" style="margin-top:0.75rem;">'
|
||||
f"<strong>Stale runtime:</strong> {escape(stale)}</div>"
|
||||
)
|
||||
identity = runtime.authenticated_username or "unresolved"
|
||||
if runtime.identity_error:
|
||||
identity = f"unresolved ({runtime.identity_error})"
|
||||
|
||||
return f"""<div class="health-card">
|
||||
<h3>Runtime context</h3>
|
||||
<table class="detail">
|
||||
<tr><th>Profile</th><td><code>{escape(runtime.profile_name)}</code></td></tr>
|
||||
<tr><th>Role kind</th><td>{escape(runtime.role_kind)}</td></tr>
|
||||
<tr><th>Identity</th><td>{escape(str(identity))}</td></tr>
|
||||
<tr><th>Remote / host</th>
|
||||
<td><code>{escape(runtime.remote)}</code> · <code>{escape(runtime.host)}</code></td>
|
||||
</tr>
|
||||
<tr><th>Local HEAD</th>
|
||||
<td><code>{escape(runtime.repo_sha or "unknown")}</code></td>
|
||||
</tr>
|
||||
<tr><th>Remote master</th>
|
||||
<td><code>{escape(runtime.remote_master_sha or "unknown")}</code></td>
|
||||
</tr>
|
||||
<tr><th>Commits behind</th>
|
||||
<td>{escape(str(runtime.commits_behind_master if runtime.commits_behind_master is not None else "unknown"))}</td>
|
||||
</tr>
|
||||
</table>
|
||||
<p class="muted">Full runtime detail: <a href="/runtime">/runtime</a> ·
|
||||
Inventory API: <a href="/api/v1/inventory"><code>/api/v1/inventory</code></a></p>
|
||||
{stale_html}
|
||||
</div>"""
|
||||
|
||||
|
||||
def _summary_bar(snapshot: SessionViewSnapshot) -> str:
|
||||
total = len(snapshot.sessions)
|
||||
stale = snapshot.stale_session_count
|
||||
contaminated = snapshot.contaminated_session_count
|
||||
active_markers = len(snapshot.active_contamination)
|
||||
inv_status = snapshot.inventory.status
|
||||
authority_complete = snapshot.ownership_authority_complete
|
||||
authority_text = "complete" if authority_complete else "incomplete"
|
||||
authority_css = "badge-health-ok" if authority_complete else "badge-health-degraded"
|
||||
return f"""<div class="health-card" style="display:flex; flex-wrap:wrap; gap:1rem; align-items:center;">
|
||||
<div><strong>Sessions:</strong> <span class="badge badge-health-ok">{total}</span></div>
|
||||
<div><strong>Stale:</strong> <span class="badge badge-stale">{stale}</span></div>
|
||||
<div><strong>Contaminated:</strong> <span class="badge badge-blocked">{contaminated}</span></div>
|
||||
<div><strong>Active markers:</strong> <span class="badge badge-health-unproven">{active_markers}</span></div>
|
||||
<div><strong>Inventory:</strong> <span class="badge badge-health-skipped">{escape(inv_status)}</span></div>
|
||||
<div><strong>Ownership authority:</strong> <span class="badge {authority_css}">{authority_text}</span></div>
|
||||
</div>"""
|
||||
|
||||
|
||||
def _ownership_caveat(snapshot: SessionViewSnapshot) -> str:
|
||||
"""Name the unreadable ownership sections, or render nothing when all read."""
|
||||
if snapshot.ownership_authority_complete:
|
||||
return ""
|
||||
degraded = ", ".join(
|
||||
f"{name}: {status}"
|
||||
for name, status in snapshot.ownership_section_status.items()
|
||||
if status != STATUS_OK
|
||||
)
|
||||
return (
|
||||
'<div class="health-card health-stale">'
|
||||
"<strong>Ownership authority incomplete:</strong> "
|
||||
f"{escape(degraded)}. Columns marked <em>unknown</em> could not be read. "
|
||||
"No session below may be treated as holding no lease or no worktree "
|
||||
"binding — absence of evidence is not evidence of absence.</div>"
|
||||
)
|
||||
|
||||
|
||||
def _render_session_row(row: SessionRow) -> str:
|
||||
pid = "—" if row.pid is None else str(row.pid)
|
||||
pid_alive = "—" if row.pid_alive is None else ("alive" if row.pid_alive else "dead")
|
||||
pid_css = (
|
||||
"badge-health-ok"
|
||||
if row.pid_alive is True
|
||||
else ("badge-blocked" if row.pid_alive is False else "badge-health-skipped")
|
||||
)
|
||||
# An empty tuple only means "holds none" when its source read cleanly.
|
||||
if row.lease_authority != STATUS_OK:
|
||||
lease_cell = _unproven_cell(row.lease_authority)
|
||||
else:
|
||||
leases = (
|
||||
", ".join(f"<code>{escape(lid)}</code>" for lid in row.lease_ids)
|
||||
if row.lease_ids
|
||||
else '<span class="muted">none</span>'
|
||||
)
|
||||
work = (
|
||||
", ".join(escape(ref) for ref in row.work_refs)
|
||||
if row.work_refs
|
||||
else '<span class="muted">—</span>'
|
||||
)
|
||||
lease_cell = (
|
||||
f'{leases}<div class="muted" style="font-size:0.82rem; '
|
||||
f'margin-top:0.2rem;">{work}</div>'
|
||||
)
|
||||
|
||||
if row.worktree_authority != STATUS_OK:
|
||||
worktrees = _unproven_cell(row.worktree_authority)
|
||||
else:
|
||||
worktrees = (
|
||||
"<br>".join(f"<code>{escape(path)}</code>" for path in row.worktree_paths)
|
||||
if row.worktree_paths
|
||||
else '<span class="muted">unbound</span>'
|
||||
)
|
||||
return f"""<tr>
|
||||
<td><code>{escape(row.session_id)}</code></td>
|
||||
<td>
|
||||
<div><code>{escape(str(row.role or "—"))}</code> / <code>{escape(str(row.profile or "—"))}</code></div>
|
||||
<div class="muted" style="font-size:0.82rem;">ns: <code>{escape(str(row.namespace or "—"))}</code></div>
|
||||
</td>
|
||||
<td>
|
||||
<code>{escape(pid)}</code>
|
||||
{_badge(pid_alive, pid_css)}
|
||||
</td>
|
||||
<td>{escape(str(row.status or "—"))}<div class="muted" style="font-size:0.82rem;">{escape(str(row.last_heartbeat_at or ""))}</div></td>
|
||||
<td>{lease_cell}</td>
|
||||
<td>{worktrees}</td>
|
||||
<td>{_flags(row.stale_flags, css="badge-stale")}</td>
|
||||
<td>{_flags(row.contamination_flags, css="badge-blocked")}</td>
|
||||
</tr>"""
|
||||
|
||||
|
||||
def _sessions_table(snapshot: SessionViewSnapshot) -> str:
|
||||
rows: Sequence[SessionRow] = snapshot.sessions
|
||||
if not rows:
|
||||
sessions_status = snapshot.ownership_section_status.get("sessions", STATUS_OK)
|
||||
if sessions_status != STATUS_OK:
|
||||
return (
|
||||
'<p class="muted">Session inventory is '
|
||||
f"<strong>{escape(sessions_status)}</strong> — the session list "
|
||||
"could not be read. This is not evidence that no sessions "
|
||||
"exist.</p>"
|
||||
)
|
||||
return (
|
||||
'<p class="muted">No control-plane sessions recorded. Inventory may '
|
||||
"be unavailable, or no MCP workers have registered yet.</p>"
|
||||
)
|
||||
body = "".join(_render_session_row(row) for row in rows)
|
||||
return f"""<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Session</th>
|
||||
<th>Role / profile / namespace</th>
|
||||
<th>PID</th>
|
||||
<th>Status</th>
|
||||
<th>Leases / work</th>
|
||||
<th>Worktree binding</th>
|
||||
<th>Stale</th>
|
||||
<th>Contamination</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{body}
|
||||
</tbody>
|
||||
</table>"""
|
||||
|
||||
|
||||
def _namespaces_section(snapshot: SessionViewSnapshot) -> str:
|
||||
section = snapshot.inventory.section("namespaces")
|
||||
if section is None:
|
||||
return (
|
||||
'<div class="prompt-card"><h3>Namespaces</h3>'
|
||||
'<p class="muted">Namespaces section not loaded.</p></div>'
|
||||
)
|
||||
if not section.ok:
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Namespaces {_badge(section.status, "badge-health-degraded")}</h3>
|
||||
<p class="muted">{escape(section.reason or "unavailable")}</p>
|
||||
</div>"""
|
||||
|
||||
rows = []
|
||||
for item in section.items:
|
||||
caps = item.get("capability_summary") or {}
|
||||
cap_bits = ", ".join(
|
||||
name for name, ok in sorted(caps.items()) if ok
|
||||
) or "none"
|
||||
rows.append(
|
||||
"<tr>"
|
||||
f"<td><code>{escape(str(item.get('mcp_namespace') or '—'))}</code></td>"
|
||||
f"<td><code>{escape(str(item.get('profile_name') or '—'))}</code></td>"
|
||||
f"<td>{escape(str(item.get('role') or '—'))}</td>"
|
||||
f"<td>{escape(cap_bits)}</td>"
|
||||
f"<td>{'yes' if item.get('active') else 'no'}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
reason = (
|
||||
f'<p class="muted">{escape(section.reason)}</p>'
|
||||
if section.reason
|
||||
else ""
|
||||
)
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Namespaces / capabilities</h3>
|
||||
{reason}
|
||||
<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Namespace</th>
|
||||
<th>Profile</th>
|
||||
<th>Role</th>
|
||||
<th>Capabilities</th>
|
||||
<th>Active in process</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{"".join(rows) if rows else '<tr><td colspan="5" class="muted">No namespace rows.</td></tr>'}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>"""
|
||||
|
||||
|
||||
def _worktrees_section(snapshot: SessionViewSnapshot) -> str:
|
||||
section = snapshot.inventory.section("worktrees")
|
||||
if section is None:
|
||||
return ""
|
||||
if not section.ok and not section.items:
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Worktrees {_badge(section.status, "badge-health-degraded")}</h3>
|
||||
<p class="muted">{escape(section.reason or "unavailable")}</p>
|
||||
</div>"""
|
||||
|
||||
rows = []
|
||||
for item in section.items[:50]:
|
||||
rows.append(
|
||||
"<tr>"
|
||||
f"<td><code>{escape(str(item.get('rel_path') or item.get('path') or '—'))}</code></td>"
|
||||
f"<td><code>{escape(str(item.get('branch') or '—'))}</code></td>"
|
||||
f"<td>{escape(str(item.get('classification') or '—'))}</td>"
|
||||
f"<td>{'yes' if item.get('registered_worktree') else 'no'}</td>"
|
||||
f"<td>{'dirty' if item.get('dirty') else 'clean'}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
more = ""
|
||||
if len(section.items) > 50:
|
||||
more = f'<p class="muted">Showing 50 of {len(section.items)}. Full list: <a href="/worktrees">/worktrees</a>.</p>'
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Worktree bindings</h3>
|
||||
<p class="muted">Registered issue worktrees under <code>branches/</code>. Hygiene detail: <a href="/worktrees">/worktrees</a>.</p>
|
||||
<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Path</th>
|
||||
<th>Branch</th>
|
||||
<th>Classification</th>
|
||||
<th>Registered</th>
|
||||
<th>State</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{"".join(rows) if rows else '<tr><td colspan="5" class="muted">No worktrees recorded.</td></tr>'}
|
||||
</tbody>
|
||||
</table>
|
||||
{more}
|
||||
</div>"""
|
||||
|
||||
|
||||
def _contamination_section(markers: Sequence[ContaminationMarker]) -> str:
|
||||
if not markers:
|
||||
return (
|
||||
'<div class="prompt-card"><h3>Contamination markers</h3>'
|
||||
'<p class="muted">No contamination kinds inspected.</p></div>'
|
||||
)
|
||||
rows = []
|
||||
for marker in markers:
|
||||
active = marker.to_dict()["active"]
|
||||
status = "ACTIVE" if active else ("cleared" if marker.cleared else "absent")
|
||||
css = "badge-blocked" if active else "badge-health-ok"
|
||||
rows.append(
|
||||
"<tr>"
|
||||
f"<td><code>{escape(marker.kind)}</code></td>"
|
||||
f"<td>{_badge(status, css)}</td>"
|
||||
f"<td>{escape(marker.reason_class or '—')}</td>"
|
||||
f"<td><code>{escape(marker.session_id or '—')}</code></td>"
|
||||
f"<td>{escape(marker.command_summary or marker.summary)}</td>"
|
||||
"</tr>"
|
||||
)
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Contamination markers (#630 / #671)</h3>
|
||||
<p class="muted">Durable markers only — never silent when present. Clearance is reconciler-only.</p>
|
||||
<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th>Kind</th>
|
||||
<th>State</th>
|
||||
<th>Reason class</th>
|
||||
<th>Session</th>
|
||||
<th>Summary</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{"".join(rows)}
|
||||
</tbody>
|
||||
</table>
|
||||
</div>"""
|
||||
|
||||
|
||||
def _recovery_section(snapshot: SessionViewSnapshot) -> str:
|
||||
items = []
|
||||
for doc in snapshot.recovery_docs:
|
||||
items.append(
|
||||
"<li>"
|
||||
f"<code>{escape(doc['path'])}</code> — "
|
||||
f"<strong>{escape(doc['label'])}</strong>: {escape(doc['note'])}"
|
||||
"</li>"
|
||||
)
|
||||
return f"""<div class="prompt-card">
|
||||
<h3>Sanctioned recovery (read-only)</h3>
|
||||
<p class="muted">This view does <strong>not</strong> restart, kill, or take over sessions.
|
||||
Manual <code>pkill</code> / <code>kill</code> of MCP daemons is contamination (#630), not recovery.</p>
|
||||
<ul class="reasons">
|
||||
{"".join(items)}
|
||||
<li>Prefer IDE/client reconnect (<code>/mcp reconnect</code>) or an operator-owned restart recorded in the restart inventory.</li>
|
||||
</ul>
|
||||
</div>"""
|
||||
|
||||
|
||||
def render_sessions_page(snapshot: SessionViewSnapshot) -> str:
|
||||
"""Render the full HTML body for the runtime/session view."""
|
||||
error_block = ""
|
||||
if snapshot.fetch_error:
|
||||
error_block = (
|
||||
f'<div class="health-card health-stale"><strong>Partial load:</strong> '
|
||||
f"{escape(snapshot.fetch_error)}</div>"
|
||||
)
|
||||
if snapshot.runtime.fetch_error:
|
||||
error_block += (
|
||||
f'<div class="health-card health-stale"><strong>Runtime note:</strong> '
|
||||
f"{escape(snapshot.runtime.fetch_error)}</div>"
|
||||
)
|
||||
|
||||
body = f"""
|
||||
{error_block}
|
||||
{_runtime_banner(snapshot)}
|
||||
{_summary_bar(snapshot)}
|
||||
|
||||
<div class="prompt-card">
|
||||
<h3>Sessions</h3>
|
||||
<p class="muted">Control-plane sessions correlated with leases and worktree bindings.
|
||||
Stale and contamination flags are fail-soft: absence of a marker is not proof of cleanliness when inventory is degraded.
|
||||
Lease and worktree columns read <em>unknown (inventory …)</em> when their source could not be loaded.</p>
|
||||
{_ownership_caveat(snapshot)}
|
||||
{_sessions_table(snapshot)}
|
||||
</div>
|
||||
|
||||
{_namespaces_section(snapshot)}
|
||||
{_worktrees_section(snapshot)}
|
||||
{_contamination_section(snapshot.contamination_markers)}
|
||||
{_recovery_section(snapshot)}
|
||||
"""
|
||||
return render_page(title="Sessions", body_html=f"""<h2>Runtime and sessions</h2>
|
||||
<p class="meta">Phase 1 read-only view (#641). Combines runtime health (#430) with
|
||||
unified inventory sessions/namespaces/worktrees (#636). No restart or session-takeover controls.</p>
|
||||
{body}""")
|
||||
@@ -270,24 +270,53 @@ def _probe_error_card(snapshot: SystemHealthSnapshot) -> str:
|
||||
|
||||
|
||||
def _recovery_card() -> str:
|
||||
"""Sanctioned recovery pointers only — never a manual process kill (#630)."""
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Recovery</h3>"
|
||||
"<p class='muted'>This dashboard is read-only. Restart and reload "
|
||||
"controls arrive in Phase 2 (#642); until then recovery runs through "
|
||||
"the sanctioned client reconnect / operator restart path.</p>"
|
||||
"<ul class='reasons'>"
|
||||
"<li><a href='/runtime'>Runtime and session view</a> — active profile, "
|
||||
"workflow hashes, and shell health.</li>"
|
||||
"<li>Reconnect the MCP client from the IDE, then re-run the blocked "
|
||||
"cycle. Never kill the daemon process manually: unmanaged kills are "
|
||||
"recorded as runtime contamination (#630).</li>"
|
||||
"<li>See <code>docs/webui-local-dev.md</code> for the documented "
|
||||
"recovery sequence.</li>"
|
||||
"</ul>"
|
||||
"</section>"
|
||||
)
|
||||
"""Sanctioned recovery controls & playbooks (#644, Phase 2)."""
|
||||
try:
|
||||
from webui import console_recovery
|
||||
diag = console_recovery.diagnose_recovery()
|
||||
# Every other card in this file escapes at the interpolation boundary.
|
||||
# This one did not, and it is where a #630 marker's operator-supplied
|
||||
# command_summary lands once the contamination gate is wired.
|
||||
status_badge = (
|
||||
f"<span class='status-pill {_esc(diag.status)}'>{_esc(diag.status)}</span>"
|
||||
)
|
||||
playbook_lis = ""
|
||||
for pb in diag.playbooks:
|
||||
elig = "eligible" if pb.eligible else "disabled"
|
||||
playbook_lis += (
|
||||
f"<li><strong>{_esc(pb.label)}</strong> "
|
||||
f"(<code>{_esc(pb.playbook_id)}</code>) — "
|
||||
f"<span class='badge {elig}'>{elig}</span>: {_esc(pb.description)} "
|
||||
f"<em class='muted'>({_esc(pb.reason)})</em></li>"
|
||||
)
|
||||
reasons_html = ""
|
||||
if diag.reasons:
|
||||
items = "".join(f"<li>{_esc(r)}</li>" for r in diag.reasons)
|
||||
reasons_html = f"<ul class='reasons'>{items}</ul>"
|
||||
else:
|
||||
reasons_html = "<p class='clean-note'>No recovery actions currently required. Control plane is healthy.</p>"
|
||||
|
||||
return (
|
||||
"<section class='health-card recovery-card'>"
|
||||
f"<h3>Sanctioned Recovery Controls (Phase 2 #644) {status_badge}</h3>"
|
||||
"<p class='muted'>Guided recovery wizard: Diagnose → Preview → Confirm → Verify. "
|
||||
"Reconnect the MCP client from the IDE, then re-run the blocked cycle. "
|
||||
"Never kill the daemon process manually: unmanaged kills are recorded as runtime contamination (#630).</p>"
|
||||
f"{reasons_html}"
|
||||
"<h4>Available Recovery Playbooks</h4>"
|
||||
f"<ul class='playbooks-list'>{playbook_lis}</ul>"
|
||||
"<p class='meta'>APIs: <code>/api/v1/system/recovery/diagnose</code>, "
|
||||
"<code>/api/v1/system/recovery/preview</code>, <code>/api/v1/system/recovery/apply</code>, "
|
||||
"<code>/api/v1/system/recovery/verify</code>.</p>"
|
||||
"</section>"
|
||||
)
|
||||
except Exception as exc:
|
||||
return (
|
||||
"<section class='health-card'>"
|
||||
"<h3>Sanctioned Recovery Controls (Phase 2 #644)</h3>"
|
||||
f"<p class='error'>Recovery diagnostics unavailable: {_esc(exc)}</p>"
|
||||
"</section>"
|
||||
)
|
||||
|
||||
|
||||
def render_system_health_page(snapshot: SystemHealthSnapshot) -> str:
|
||||
|
||||
@@ -0,0 +1,459 @@
|
||||
"""Traffic-control view loader for Phase 1 operator web console (#640).
|
||||
|
||||
Combines queue snapshots, inventory leases, dependency graph classifications,
|
||||
and workflow dashboard rules to deliver full traffic-control visibility:
|
||||
runnable, leased (in-progress), blocked (dependency/lock), needs-controller,
|
||||
and terminal-complete candidates.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Any, Callable, Sequence
|
||||
|
||||
from webui.project_registry import find_project, load_registry
|
||||
from webui.queue_loader import load_queue_snapshot, QueueSnapshot
|
||||
from webui.lease_loader import load_lease_snapshot, LeaseSnapshot
|
||||
from workflow_dashboard import (
|
||||
DashboardSnapshot,
|
||||
QueueEntry,
|
||||
RoleNextAction,
|
||||
build_workflow_dashboard,
|
||||
DASHBOARD_ROLES,
|
||||
)
|
||||
from allocator_service import WorkCandidate
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TrafficItem:
|
||||
kind: str # "issue" or "pr"
|
||||
number: int
|
||||
title: str
|
||||
traffic_state: str # "runnable", "leased", "blocked", "needs_controller", "terminal_complete"
|
||||
expected_role: str
|
||||
safe_for_roles: tuple[str, ...]
|
||||
badges: tuple[str, ...]
|
||||
block_reason: str | None = None
|
||||
lease_info: dict[str, Any] | None = None
|
||||
head_sha: str | None = None
|
||||
|
||||
@property
|
||||
def is_safe(self) -> bool:
|
||||
return self.block_reason is None and bool(self.safe_for_roles)
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"kind": self.kind,
|
||||
"number": self.number,
|
||||
"title": self.title,
|
||||
"traffic_state": self.traffic_state,
|
||||
"expected_role": self.expected_role,
|
||||
"safe_for_roles": list(self.safe_for_roles),
|
||||
"badges": list(self.badges),
|
||||
"block_reason": self.block_reason,
|
||||
"lease_info": self.lease_info,
|
||||
"head_sha": self.head_sha,
|
||||
"is_safe": self.is_safe,
|
||||
}
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class TrafficSnapshot:
|
||||
project_id: str
|
||||
repo_label: str
|
||||
runnable: tuple[TrafficItem, ...]
|
||||
leased: tuple[TrafficItem, ...]
|
||||
blocked: tuple[TrafficItem, ...]
|
||||
needs_controller: tuple[TrafficItem, ...]
|
||||
terminal_complete: tuple[TrafficItem, ...]
|
||||
next_roles: tuple[dict[str, Any], ...]
|
||||
fetch_error: str | None = None
|
||||
inventory_complete: bool = True
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"project_id": self.project_id,
|
||||
"repo_label": self.repo_label,
|
||||
"runnable": [i.as_dict() for i in self.runnable],
|
||||
"leased": [i.as_dict() for i in self.leased],
|
||||
"blocked": [i.as_dict() for i in self.blocked],
|
||||
"needs_controller": [i.as_dict() for i in self.needs_controller],
|
||||
"terminal_complete": [i.as_dict() for i in self.terminal_complete],
|
||||
"next_roles": list(self.next_roles),
|
||||
"fetch_error": self.fetch_error,
|
||||
"inventory_complete": self.inventory_complete,
|
||||
}
|
||||
|
||||
|
||||
def _classify_traffic_item(
|
||||
entry: QueueEntry,
|
||||
*,
|
||||
lease_info: dict[str, Any] | None = None,
|
||||
) -> TrafficItem:
|
||||
"""Classify a QueueEntry into a TrafficItem with explicit traffic state."""
|
||||
badges = list(entry.badges)
|
||||
block_reason = entry.block_reason
|
||||
expected_role = entry.expected_role
|
||||
|
||||
entry_is_safe = entry.block_reason is None and bool(entry.safe_for_roles)
|
||||
# Lease state is checked first: an item that is both leased and blocked is
|
||||
# reported as leased. That is safe by construction — a leased item is never
|
||||
# placed in the runnable lane — and it keeps the operator's attention on the
|
||||
# session that currently owns the work. The blocker text still renders.
|
||||
if lease_info is not None or "in-progress" in badges or "claimed" in badges:
|
||||
state = "leased"
|
||||
elif expected_role == "reconciler" or "terminal-lock" in badges:
|
||||
state = "terminal_complete"
|
||||
elif expected_role == "controller" or "contaminated" in badges or "needs-controller" in badges:
|
||||
state = "needs_controller"
|
||||
elif (
|
||||
block_reason is not None
|
||||
or "blocked" in badges
|
||||
or "dependency-unmet" in badges
|
||||
or "blocked-by-terminal" in badges
|
||||
or "status:blocked" in badges
|
||||
):
|
||||
state = "blocked"
|
||||
elif entry_is_safe:
|
||||
state = "runnable"
|
||||
else:
|
||||
state = "needs_controller"
|
||||
|
||||
return TrafficItem(
|
||||
kind=entry.kind,
|
||||
number=entry.number,
|
||||
title=entry.title,
|
||||
traffic_state=state,
|
||||
expected_role=expected_role,
|
||||
safe_for_roles=entry.safe_for_roles,
|
||||
badges=tuple(badges),
|
||||
block_reason=block_reason,
|
||||
lease_info=lease_info,
|
||||
head_sha=entry.head_sha,
|
||||
)
|
||||
|
||||
|
||||
# Claim statuses from ``issue_claim_heartbeat.build_claim_inventory`` that mean
|
||||
# a live worker currently holds the issue. Everything else (``stale``,
|
||||
# ``phantom``, ``reclaimable``, ``not_claimed``) is reported through the
|
||||
# dashboard's stale-lease channel and is never rendered as an active lease.
|
||||
_ACTIVE_CLAIM_STATUSES = frozenset({"active", "awaiting_review"})
|
||||
|
||||
# Statuses that positively mean "not an active lease" for any lease record.
|
||||
_INACTIVE_LEASE_STATUSES = frozenset(
|
||||
{"expired", "stale", "released", "moot", "reclaimable", "phantom", "not_claimed"}
|
||||
)
|
||||
|
||||
|
||||
def _candidates_from_queue_snapshot(q_snap: QueueSnapshot) -> list[WorkCandidate]:
|
||||
"""Build allocator candidates from the queue loader's authoritative signals.
|
||||
|
||||
Display badges (``blocked``/``claimed``/``duplicate``/``stale``/
|
||||
``in-review``/``open``) are rendering hints, not routing state, so nothing
|
||||
here branches on them. Every routing field comes from
|
||||
``QueueItem.signals`` — the raw Gitea payload values.
|
||||
|
||||
The queue loader reads ``/pulls`` and ``/issues`` only; it never fetches
|
||||
review verdicts. ``request_changes_current_head`` / ``approval_on_current_head``
|
||||
are therefore left at their fail-safe ``False`` rather than being guessed
|
||||
from badges: an unproven approval must never route a PR to the merger.
|
||||
"""
|
||||
candidates: list[WorkCandidate] = []
|
||||
|
||||
for pr in q_snap.prs:
|
||||
signals = pr.signals or {}
|
||||
head_sha = str(signals.get("head_sha") or "").strip()
|
||||
mergeable = signals.get("mergeable")
|
||||
labels = tuple(str(x) for x in (signals.get("labels") or ()))
|
||||
candidates.append(
|
||||
WorkCandidate(
|
||||
kind="pr",
|
||||
number=pr.number,
|
||||
state="open",
|
||||
labels=labels,
|
||||
title=pr.title,
|
||||
# Full 40-char SHA from head.sha — never the 12-char display value.
|
||||
head_sha=head_sha or None,
|
||||
priority=5,
|
||||
mergeable=mergeable is True,
|
||||
blocked=mergeable is False or "status:blocked" in labels,
|
||||
)
|
||||
)
|
||||
|
||||
for issue in q_snap.issues:
|
||||
signals = issue.signals or {}
|
||||
labels = tuple(str(x) for x in (signals.get("labels") or ()))
|
||||
lowered = {label.lower() for label in labels}
|
||||
candidates.append(
|
||||
WorkCandidate(
|
||||
kind="issue",
|
||||
number=issue.number,
|
||||
state="open",
|
||||
labels=labels,
|
||||
title=issue.title,
|
||||
priority=20 if "status:ready" in lowered else 10,
|
||||
blocked="status:blocked" in lowered,
|
||||
# A live claim by another session is not this session's work.
|
||||
already_claimed_elsewhere="status:in-progress" in lowered,
|
||||
)
|
||||
)
|
||||
|
||||
return candidates
|
||||
|
||||
|
||||
def candidates_from_queue_snapshot(q_snap: QueueSnapshot) -> list[WorkCandidate]:
|
||||
"""Public alias for :func:`_candidates_from_queue_snapshot` (#643).
|
||||
|
||||
The request-initiation service ranks the same candidate set this view
|
||||
renders, so both must agree on how a queue row becomes a candidate. One
|
||||
construction, two callers — not two that can drift apart.
|
||||
"""
|
||||
return _candidates_from_queue_snapshot(q_snap)
|
||||
|
||||
|
||||
def _claim_lease_records(inventory: dict[str, Any] | None) -> list[dict[str, Any]]:
|
||||
"""Normalize ``build_claim_inventory`` entries into lease records.
|
||||
|
||||
The inventory contract is ``{"entries", "counts", "heartbeat_lease_minutes",
|
||||
"reclaim_after_minutes", "in_progress_total"}``. Each entry is keyed by
|
||||
``issue_number``; the subject kind is therefore always ``issue``.
|
||||
"""
|
||||
entries = (inventory or {}).get("entries") or ()
|
||||
records: list[dict[str, Any]] = []
|
||||
for entry in entries:
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
number = entry.get("issue_number")
|
||||
if number is None:
|
||||
continue
|
||||
try:
|
||||
number_int = int(number)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
heartbeat = entry.get("latest_heartbeat") or {}
|
||||
record = dict(entry)
|
||||
record.update(
|
||||
{
|
||||
"kind": "issue",
|
||||
"number": number_int,
|
||||
"role": "author",
|
||||
"lease_source": "issue-claim-heartbeat",
|
||||
}
|
||||
)
|
||||
if isinstance(heartbeat, dict):
|
||||
if heartbeat.get("session_id") and not record.get("session_id"):
|
||||
record["session_id"] = heartbeat.get("session_id")
|
||||
if heartbeat.get("author") and not record.get("author"):
|
||||
record["author"] = heartbeat.get("author")
|
||||
records.append(record)
|
||||
return records
|
||||
|
||||
|
||||
def _lease_subject(lease: dict[str, Any]) -> tuple[str, int] | None:
|
||||
"""Return the ``(kind, number)`` a lease record actually covers.
|
||||
|
||||
Fails closed: a record that does not identify exactly one subject is
|
||||
dropped rather than attributed to a guessed work item (#640 — never invent
|
||||
a lease, and never attach a PR lease to a same-numbered issue).
|
||||
"""
|
||||
kind = str(lease.get("kind") or lease.get("work_kind") or "").strip().lower()
|
||||
pr_number = lease.get("pr_number")
|
||||
issue_number = lease.get("issue_number")
|
||||
|
||||
if kind not in ("pr", "issue"):
|
||||
if pr_number is not None and issue_number is None:
|
||||
kind = "pr"
|
||||
elif issue_number is not None and pr_number is None:
|
||||
kind = "issue"
|
||||
else:
|
||||
return None
|
||||
|
||||
number = lease.get("number")
|
||||
if number is None:
|
||||
number = lease.get("work_number")
|
||||
if number is None:
|
||||
number = pr_number if kind == "pr" else issue_number
|
||||
if number is None:
|
||||
return None
|
||||
try:
|
||||
return kind, int(number)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _is_active_lease(lease: dict[str, Any]) -> bool:
|
||||
"""True when the record proves a worker currently holds the item."""
|
||||
if lease.get("stale") or lease.get("expired"):
|
||||
return False
|
||||
status = str(lease.get("status") or lease.get("lease_status") or "").strip().lower()
|
||||
if status in _INACTIVE_LEASE_STATUSES:
|
||||
return False
|
||||
if lease.get("lease_source") == "issue-claim-heartbeat":
|
||||
return status in _ACTIVE_CLAIM_STATUSES
|
||||
return True
|
||||
|
||||
|
||||
def load_traffic_snapshot(
|
||||
*,
|
||||
candidates: Sequence[WorkCandidate] | None = None,
|
||||
leases: Sequence[dict[str, Any]] | None = None,
|
||||
terminal_pr: int | None = None,
|
||||
fetch_queue_snapshot: Callable[[], QueueSnapshot] | None = None,
|
||||
fetch_lease_snapshot: Callable[[], LeaseSnapshot] | None = None,
|
||||
project_id: str = "gitea-tools",
|
||||
) -> TrafficSnapshot:
|
||||
"""Load and compute the traffic-control snapshot."""
|
||||
try:
|
||||
reg = load_registry()
|
||||
proj = find_project(reg, project_id)
|
||||
repo_label = proj.remote_repo if proj else "Scaled-Tech-Consulting/Gitea-Tools"
|
||||
except Exception:
|
||||
repo_label = "Scaled-Tech-Consulting/Gitea-Tools"
|
||||
|
||||
# Injected candidates path (pure unit testing)
|
||||
if candidates is not None:
|
||||
dashboard = build_workflow_dashboard(
|
||||
candidates=candidates,
|
||||
leases=leases,
|
||||
terminal_pr=terminal_pr,
|
||||
inventory_complete=True,
|
||||
)
|
||||
return _build_traffic_snapshot_from_dashboard(
|
||||
project_id=project_id,
|
||||
repo_label=repo_label,
|
||||
dashboard=dashboard,
|
||||
leases=leases or (),
|
||||
)
|
||||
|
||||
# Live snapshot loading
|
||||
q_loader = fetch_queue_snapshot or load_queue_snapshot
|
||||
l_loader = fetch_lease_snapshot or load_lease_snapshot
|
||||
|
||||
try:
|
||||
q_snap = q_loader()
|
||||
l_snap = l_loader()
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return TrafficSnapshot(
|
||||
project_id=project_id,
|
||||
repo_label=repo_label,
|
||||
runnable=(),
|
||||
leased=(),
|
||||
blocked=(),
|
||||
needs_controller=(),
|
||||
terminal_complete=(),
|
||||
next_roles=(),
|
||||
fetch_error=f"Failed to load traffic state: {exc}",
|
||||
inventory_complete=False,
|
||||
)
|
||||
|
||||
if q_snap.fetch_error or l_snap.fetch_error:
|
||||
err = q_snap.fetch_error or l_snap.fetch_error
|
||||
return TrafficSnapshot(
|
||||
project_id=project_id,
|
||||
repo_label=repo_label,
|
||||
runnable=(),
|
||||
leased=(),
|
||||
blocked=(),
|
||||
needs_controller=(),
|
||||
terminal_complete=(),
|
||||
next_roles=(),
|
||||
fetch_error=err,
|
||||
inventory_complete=False,
|
||||
)
|
||||
|
||||
candidate_list = _candidates_from_queue_snapshot(q_snap)
|
||||
|
||||
raw_leases: list[dict[str, Any]] = _claim_lease_records(l_snap.claim_inventory)
|
||||
for r_lease in l_snap.reviewer_leases or ():
|
||||
if not isinstance(r_lease, dict):
|
||||
continue
|
||||
# Always pin reviewer leases to the PR subject, even if a linked
|
||||
# issue_number is present on the marker (#640 B2).
|
||||
normalized = dict(r_lease)
|
||||
subject = normalized.get("pr_number") or normalized.get("number")
|
||||
if subject is None:
|
||||
continue
|
||||
try:
|
||||
pr_num = int(subject)
|
||||
except (TypeError, ValueError):
|
||||
continue
|
||||
normalized["kind"] = "pr"
|
||||
normalized["number"] = pr_num
|
||||
normalized["pr_number"] = pr_num
|
||||
normalized.setdefault("role", "reviewer")
|
||||
raw_leases.append(normalized)
|
||||
|
||||
dashboard = build_workflow_dashboard(
|
||||
candidates=candidate_list,
|
||||
leases=raw_leases,
|
||||
inventory_complete=q_snap.pr_pagination.inventory_complete if q_snap.pr_pagination else True,
|
||||
)
|
||||
|
||||
return _build_traffic_snapshot_from_dashboard(
|
||||
project_id=project_id,
|
||||
repo_label=repo_label,
|
||||
dashboard=dashboard,
|
||||
leases=raw_leases,
|
||||
)
|
||||
|
||||
|
||||
def _build_traffic_snapshot_from_dashboard(
|
||||
*,
|
||||
project_id: str,
|
||||
repo_label: str,
|
||||
dashboard: DashboardSnapshot,
|
||||
leases: Sequence[dict[str, Any]],
|
||||
) -> TrafficSnapshot:
|
||||
"""Classify dashboard entries into the 5 traffic state buckets."""
|
||||
all_entries = dashboard.open_prs + dashboard.open_issues
|
||||
|
||||
# Map each active lease onto the exact work item it covers. Records whose
|
||||
# subject cannot be determined, and claims that are stale/phantom/
|
||||
# reclaimable, are deliberately dropped instead of guessed.
|
||||
lease_map: dict[tuple[str, int], dict[str, Any]] = {}
|
||||
for lease in leases:
|
||||
if not isinstance(lease, dict) or not _is_active_lease(lease):
|
||||
continue
|
||||
subject = _lease_subject(lease)
|
||||
if subject is not None:
|
||||
lease_map[subject] = lease
|
||||
|
||||
runnable: list[TrafficItem] = []
|
||||
leased: list[TrafficItem] = []
|
||||
blocked: list[TrafficItem] = []
|
||||
needs_controller: list[TrafficItem] = []
|
||||
terminal_complete: list[TrafficItem] = []
|
||||
|
||||
for entry in all_entries:
|
||||
l_info = lease_map.get((entry.kind, entry.number))
|
||||
item = _classify_traffic_item(entry, lease_info=l_info)
|
||||
|
||||
if item.traffic_state == "leased":
|
||||
leased.append(item)
|
||||
elif item.traffic_state == "terminal_complete":
|
||||
terminal_complete.append(item)
|
||||
elif item.traffic_state == "blocked":
|
||||
blocked.append(item)
|
||||
elif item.traffic_state == "needs_controller":
|
||||
needs_controller.append(item)
|
||||
else:
|
||||
runnable.append(item)
|
||||
|
||||
next_roles = [dashboard.next_safe_by_role[r].as_dict() for r in DASHBOARD_ROLES if r in dashboard.next_safe_by_role]
|
||||
|
||||
return TrafficSnapshot(
|
||||
project_id=project_id,
|
||||
repo_label=repo_label,
|
||||
runnable=tuple(runnable),
|
||||
leased=tuple(leased),
|
||||
blocked=tuple(blocked),
|
||||
needs_controller=tuple(needs_controller),
|
||||
terminal_complete=tuple(terminal_complete),
|
||||
next_roles=tuple(next_roles),
|
||||
fetch_error=None,
|
||||
inventory_complete=dashboard.inventory_complete,
|
||||
)
|
||||
|
||||
|
||||
def snapshot_to_dict(snapshot: TrafficSnapshot) -> dict[str, Any]:
|
||||
return snapshot.as_dict()
|
||||
@@ -0,0 +1,170 @@
|
||||
"""HTML rendering for Phase 1 Traffic-Control View (#640)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from html import escape
|
||||
from typing import Sequence
|
||||
|
||||
from webui.layout import render_page
|
||||
from webui.traffic_loader import TrafficItem, TrafficSnapshot
|
||||
|
||||
|
||||
def _render_badges(badges: Sequence[str]) -> str:
|
||||
if not badges:
|
||||
return ""
|
||||
out = []
|
||||
for b in badges:
|
||||
cls = "badge"
|
||||
b_lower = b.lower()
|
||||
if "blocked" in b_lower or "unmet" in b_lower:
|
||||
cls += " badge-blocked"
|
||||
elif "claimed" in b_lower or "in-progress" in b_lower or "leased" in b_lower:
|
||||
cls += " badge-claimed"
|
||||
elif "review" in b_lower or "ready" in b_lower:
|
||||
cls += " badge-in-review"
|
||||
elif "duplicate" in b_lower:
|
||||
cls += " badge-duplicate"
|
||||
elif "stale" in b_lower:
|
||||
cls += " badge-stale"
|
||||
out.append(f'<span class="{cls}">{escape(b)}</span>')
|
||||
return f'<div class="badges">{"".join(out)}</div>'
|
||||
|
||||
|
||||
def _render_traffic_item_row(item: TrafficItem) -> str:
|
||||
kind_label = escape(item.kind.upper())
|
||||
num_str = f"#{item.number}"
|
||||
title_str = escape(item.title)
|
||||
role_str = escape(item.expected_role)
|
||||
badges_html = _render_badges(item.badges)
|
||||
|
||||
reason_html = ""
|
||||
if item.block_reason:
|
||||
reason_html = f'<div class="muted" style="font-size:0.82rem; margin-top:0.2rem;"><strong>Blocker:</strong> {escape(item.block_reason)}</div>'
|
||||
|
||||
lease_html = ""
|
||||
if item.lease_info:
|
||||
owner = escape(str(item.lease_info.get("session_id") or item.lease_info.get("reviewer_identity") or "active worker"))
|
||||
lease_html = f'<div class="muted" style="font-size:0.82rem; margin-top:0.2rem;"><strong>Lease:</strong> {owner}</div>'
|
||||
|
||||
return f"""<tr>
|
||||
<td><code>{kind_label} {num_str}</code></td>
|
||||
<td>
|
||||
<div><strong>{title_str}</strong> {badges_html}</div>
|
||||
{reason_html}
|
||||
{lease_html}
|
||||
</td>
|
||||
<td><code>{role_str}</code></td>
|
||||
</tr>"""
|
||||
|
||||
|
||||
def _render_traffic_table(items: Sequence[TrafficItem], empty_message: str) -> str:
|
||||
if not items:
|
||||
return f'<p class="muted">{escape(empty_message)}</p>'
|
||||
|
||||
rows = "".join(_render_traffic_item_row(item) for item in items)
|
||||
return f"""<table class="registry">
|
||||
<thead>
|
||||
<tr>
|
||||
<th style="width: 15%;">Item</th>
|
||||
<th style="width: 65%;">Title & Details</th>
|
||||
<th style="width: 20%;">Next Role</th>
|
||||
</tr>
|
||||
</thead>
|
||||
<tbody>
|
||||
{rows}
|
||||
</tbody>
|
||||
</table>"""
|
||||
|
||||
|
||||
def _render_next_roles(next_roles: Sequence[dict]) -> str:
|
||||
if not next_roles:
|
||||
return ""
|
||||
|
||||
cards = []
|
||||
for r in next_roles:
|
||||
role = escape(r.get("role", "unknown"))
|
||||
status = r.get("status", "idle")
|
||||
prompt = escape(r.get("prompt", ""))
|
||||
|
||||
status_cls = "badge-health-ok" if status == "safe" else ("badge-blocked" if "blocked" in status else "badge-health-skipped")
|
||||
cards.append(f"""<div class="health-card" style="margin-bottom:0.75rem;">
|
||||
<div style="display:flex; justify-content:space-between; align-items:center;">
|
||||
<h3>Role: <code>{role}</code></h3>
|
||||
<span class="badge {status_cls}">status: {escape(status)}</span>
|
||||
</div>
|
||||
<p class="meta" style="margin:0.35rem 0 0;">{prompt}</p>
|
||||
</div>""")
|
||||
|
||||
return f"""<div style="margin: 1.5rem 0;">
|
||||
<h3>Next Safe Role Actions</h3>
|
||||
{"".join(cards)}
|
||||
</div>"""
|
||||
|
||||
|
||||
def render_traffic_page(snapshot: TrafficSnapshot) -> str:
|
||||
"""Render the full HTML view for workflow traffic control."""
|
||||
if snapshot.fetch_error:
|
||||
body = f"""<h2>Workflow Traffic Control</h2>
|
||||
<p class="meta">Repository: <code>{escape(snapshot.repo_label)}</code></p>
|
||||
<div class="health-card health-stale">
|
||||
<h3>Traffic data unavailable</h3>
|
||||
<p class="health-headline">{escape(snapshot.fetch_error)}</p>
|
||||
<p class="muted">Fail closed: traffic state cannot be established cleanly. Check credentials or remote connectivity.</p>
|
||||
</div>"""
|
||||
return render_page(title="Traffic Control", body_html=body)
|
||||
|
||||
runnable_count = len(snapshot.runnable)
|
||||
leased_count = len(snapshot.leased)
|
||||
blocked_count = len(snapshot.blocked)
|
||||
controller_count = len(snapshot.needs_controller)
|
||||
terminal_count = len(snapshot.terminal_complete)
|
||||
|
||||
summary_bar = f"""<div class="health-card" style="display:flex; flex-wrap:wrap; gap:1rem; align-items:center;">
|
||||
<div><strong>Runnable:</strong> <span class="badge badge-health-ok">{runnable_count}</span></div>
|
||||
<div><strong>Leased:</strong> <span class="badge badge-claimed">{leased_count}</span></div>
|
||||
<div><strong>Blocked:</strong> <span class="badge badge-blocked">{blocked_count}</span></div>
|
||||
<div><strong>Needs Controller:</strong> <span class="badge badge-duplicate">{controller_count}</span></div>
|
||||
<div><strong>Terminal Complete:</strong> <span class="badge badge-stale">{terminal_count}</span></div>
|
||||
</div>"""
|
||||
|
||||
next_roles_html = _render_next_roles(snapshot.next_roles)
|
||||
|
||||
sections_html = f"""
|
||||
<div class="prompt-card">
|
||||
<h3>1. Runnable Lanes (Ready for Allocation)</h3>
|
||||
<p class="muted">Safe work items with no unmet dependencies or active leases. Safe for allocation.</p>
|
||||
{_render_traffic_table(snapshot.runnable, "No runnable items ready for allocation.")}
|
||||
</div>
|
||||
|
||||
<div class="prompt-card">
|
||||
<h3>2. In-Progress Work (Active Leases)</h3>
|
||||
<p class="muted">Work items currently leased and actively being worked by an assigned role session.</p>
|
||||
{_render_traffic_table(snapshot.leased, "No active leases in flight.")}
|
||||
</div>
|
||||
|
||||
<div class="prompt-card">
|
||||
<h3>3. Blocked Items (Dependencies / Locks)</h3>
|
||||
<p class="muted">Items blocked by unmet dependency issues, a missing head pin, a merge conflict, or an active terminal review lock. Items labelled status:blocked route to section 4. Never presented as safe.</p>
|
||||
{_render_traffic_table(snapshot.blocked, "No blocked items.")}
|
||||
</div>
|
||||
|
||||
<div class="prompt-card">
|
||||
<h3>4. Needs Controller Intervention</h3>
|
||||
<p class="muted">Items requiring controller routing, diagnosis, or cross-role assignment.</p>
|
||||
{_render_traffic_table(snapshot.needs_controller, "No items requiring controller intervention.")}
|
||||
</div>
|
||||
|
||||
<div class="prompt-card">
|
||||
<h3>5. Terminal / Complete Candidates</h3>
|
||||
<p class="muted">Items ready for terminal reconciliation or post-merge worktree cleanup.</p>
|
||||
{_render_traffic_table(snapshot.terminal_complete, "No terminal complete candidates.")}
|
||||
</div>
|
||||
"""
|
||||
|
||||
body = f"""<h2>Workflow Traffic Control</h2>
|
||||
<p class="meta">Repository: <code>{escape(snapshot.repo_label)}</code></p>
|
||||
{summary_bar}
|
||||
{next_roles_html}
|
||||
{sections_html}"""
|
||||
|
||||
return render_page(title="Traffic Control", body_html=body)
|
||||
Reference in New Issue
Block a user