Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
f0c6255d7d | ||
|
|
77d808e7d4 | ||
|
|
e91b94db56 |
@@ -26,6 +26,7 @@ from dataclasses import dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Mapping, Sequence
|
||||
|
||||
import maintenance_drain
|
||||
from control_plane_db import (
|
||||
ControlPlaneDB,
|
||||
ControlPlaneError,
|
||||
@@ -936,6 +937,46 @@ def allocate_next_work(
|
||||
"allocation_mode": (allocation_mode or "").strip() or None,
|
||||
}
|
||||
|
||||
# #659 AC2: while maintenance drain is active, no new work is assigned —
|
||||
# for dry-run and apply alike, so a preview can never be read as evidence
|
||||
# that work was assignable during the drain. Checked before session
|
||||
# registration so a drained allocator leaves no new state behind.
|
||||
try:
|
||||
drain_record = db.read_maintenance_drain(remote=remote, org=org, repo=repo)
|
||||
except Exception as exc: # noqa: BLE001 — unreadable drain state fails closed
|
||||
return {
|
||||
"success": False,
|
||||
"outcome": OUTCOME_NO_SAFE,
|
||||
"reasons": [
|
||||
f"maintenance-drain state lookup failed: {exc} (fail closed, #659)"
|
||||
],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
}
|
||||
|
||||
drain_decision = maintenance_drain.classify_assignment(drain_record)
|
||||
if not drain_decision["assignment_allowed"]:
|
||||
return {
|
||||
"success": True,
|
||||
"outcome": OUTCOME_WAIT,
|
||||
"apply": apply,
|
||||
"role": role_norm,
|
||||
"allocation_mode": mode,
|
||||
"remote": remote,
|
||||
"org": org,
|
||||
"repo": repo,
|
||||
"selected": None,
|
||||
"reasons": list(drain_decision["reasons"]),
|
||||
"reason_code": drain_decision["reason_code"],
|
||||
"skipped": [],
|
||||
"assignment": None,
|
||||
"substrate": "control_plane_db",
|
||||
"maintenance_drain": maintenance_drain.status_payload(
|
||||
drain_record, remote=remote, org=org, repo=repo
|
||||
),
|
||||
}
|
||||
|
||||
# A side-effect-free run may never reserve: reserving is a write, and the
|
||||
# flag is the caller's assertion that this call writes nothing (#643).
|
||||
if side_effect_free and apply:
|
||||
|
||||
+194
-1
@@ -31,8 +31,9 @@ from typing import Any, Iterator, Sequence
|
||||
|
||||
import dependency_graph
|
||||
import gitea_audit
|
||||
import maintenance_drain
|
||||
|
||||
SCHEMA_VERSION = 5
|
||||
SCHEMA_VERSION = 6
|
||||
|
||||
# Assignable work kinds only — raw monitoring incidents are never work items.
|
||||
WORK_KINDS = frozenset({"issue", "pr"})
|
||||
@@ -239,6 +240,31 @@ CREATE INDEX IF NOT EXISTS idx_session_checkpoints_session
|
||||
CREATE INDEX IF NOT EXISTS idx_session_checkpoints_work
|
||||
ON session_checkpoints(remote, org, repo, work_kind, work_number);
|
||||
|
||||
-- Graceful maintenance-drain state (#659). One current row per repository
|
||||
-- scope — drain is a *state*, not a history, so entering and exiting update
|
||||
-- the same row and every transition is audited to ``events``. Creating the
|
||||
-- table is the v5->v6 migration: additive, idempotent, and it never touches
|
||||
-- prior tables. ``state`` is CHECK-constrained so an unknown value can never
|
||||
-- be written and later read as "not draining".
|
||||
CREATE TABLE IF NOT EXISTS maintenance_drain (
|
||||
drain_id TEXT PRIMARY KEY,
|
||||
remote TEXT NOT NULL,
|
||||
org TEXT NOT NULL,
|
||||
repo TEXT NOT NULL,
|
||||
state TEXT NOT NULL DEFAULT 'inactive'
|
||||
CHECK (state IN ('inactive', 'draining')),
|
||||
reason TEXT NOT NULL DEFAULT '',
|
||||
requested_by TEXT NOT NULL DEFAULT '',
|
||||
requested_by_profile TEXT NOT NULL DEFAULT '',
|
||||
session_id TEXT NOT NULL DEFAULT '',
|
||||
entered_at TEXT NOT NULL DEFAULT '',
|
||||
exited_at TEXT NOT NULL DEFAULT '',
|
||||
drain_schema_version INTEGER NOT NULL DEFAULT 6,
|
||||
created_at TEXT NOT NULL,
|
||||
updated_at TEXT NOT NULL,
|
||||
UNIQUE (remote, org, repo)
|
||||
);
|
||||
|
||||
-- Model usage, token cost, latency, and performance events (#651)
|
||||
CREATE TABLE IF NOT EXISTS usage_events (
|
||||
usage_id INTEGER PRIMARY KEY AUTOINCREMENT,
|
||||
@@ -3025,3 +3051,170 @@ class ControlPlaneDB:
|
||||
"live_lease_id": None if live_lease_id is None else str(live_lease_id),
|
||||
"reconcile_action": "reconcile_required" if stale else "safe_to_resume",
|
||||
}
|
||||
|
||||
# ── Maintenance drain (#659) ─────────────────────────────────────────────
|
||||
|
||||
@staticmethod
|
||||
def _maintenance_drain_row(row: sqlite3.Row | None) -> dict[str, Any] | None:
|
||||
"""Convert a ``maintenance_drain`` row to a plain record."""
|
||||
if row is None:
|
||||
return None
|
||||
return {key: row[key] for key in row.keys()}
|
||||
|
||||
def read_maintenance_drain(
|
||||
self, *, remote: str, org: str, repo: str
|
||||
) -> dict[str, Any] | None:
|
||||
"""Return the current drain record for a scope, or None if never set.
|
||||
|
||||
None and a stored ``inactive`` row mean the same thing to callers —
|
||||
``maintenance_drain.is_draining`` treats both as not draining — so the
|
||||
read never has to invent a record to answer the gate.
|
||||
"""
|
||||
with self._tx(immediate=False) as conn:
|
||||
row = conn.execute(
|
||||
"""
|
||||
SELECT * FROM maintenance_drain
|
||||
WHERE remote = ? AND org = ? AND repo = ?
|
||||
""",
|
||||
(str(remote or ""), str(org or ""), str(repo or "")),
|
||||
).fetchone()
|
||||
return self._maintenance_drain_row(row)
|
||||
|
||||
def set_maintenance_drain(
|
||||
self,
|
||||
*,
|
||||
remote: str,
|
||||
org: str,
|
||||
repo: str,
|
||||
state: str,
|
||||
reason: str = "",
|
||||
requested_by: str = "",
|
||||
requested_by_profile: str = "",
|
||||
session_id: str = "",
|
||||
) -> dict[str, Any]:
|
||||
"""Enter or exit maintenance drain for one repository scope (AC1).
|
||||
|
||||
The state transition is audited to ``events`` — entering and exiting
|
||||
are exactly the moments an operator has to be able to reconstruct
|
||||
later. Re-entering an already-draining scope is idempotent: it refreshes
|
||||
the reason/owner metadata, keeps the original ``entered_at``, and
|
||||
records no duplicate transition event.
|
||||
|
||||
Capability authorization happens above this layer (the drain tasks
|
||||
carry a non-``gitea.*`` permission in the task capability map); the DB
|
||||
records who asked and why, and never grants the right itself.
|
||||
"""
|
||||
state_norm = maintenance_drain.normalize_state(state)
|
||||
raw = {
|
||||
"reason": str(reason or ""),
|
||||
"requested_by": str(requested_by or ""),
|
||||
"requested_by_profile": str(requested_by_profile or ""),
|
||||
"session_id": str(session_id or ""),
|
||||
}
|
||||
clean = gitea_audit.redact(raw)
|
||||
remote_s, org_s, repo_s = str(remote or ""), str(org or ""), str(repo or "")
|
||||
now_s = _ts()
|
||||
|
||||
with self._tx() as conn:
|
||||
existing = conn.execute(
|
||||
"""
|
||||
SELECT * FROM maintenance_drain
|
||||
WHERE remote = ? AND org = ? AND repo = ?
|
||||
""",
|
||||
(remote_s, org_s, repo_s),
|
||||
).fetchone()
|
||||
|
||||
prior_state = (
|
||||
maintenance_drain.normalize_state(existing["state"])
|
||||
if existing is not None
|
||||
else maintenance_drain.STATE_INACTIVE
|
||||
)
|
||||
transitioned = prior_state != state_norm
|
||||
|
||||
prior_entered = (
|
||||
str(existing["entered_at"] or "") if existing is not None else ""
|
||||
)
|
||||
prior_exited = (
|
||||
str(existing["exited_at"] or "") if existing is not None else ""
|
||||
)
|
||||
if state_norm == maintenance_drain.STATE_DRAINING:
|
||||
# A re-entry keeps the original entry time (the drain never
|
||||
# stopped); a fresh entry stamps now and clears the old exit.
|
||||
entered_at = prior_entered if (not transitioned and prior_entered) else now_s
|
||||
exited_at = ""
|
||||
else:
|
||||
entered_at = prior_entered
|
||||
exited_at = now_s if (transitioned or not prior_exited) else prior_exited
|
||||
|
||||
if existing is None:
|
||||
drain_id = uuid.uuid4().hex
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO maintenance_drain(
|
||||
drain_id, remote, org, repo, state, reason,
|
||||
requested_by, requested_by_profile, session_id,
|
||||
entered_at, exited_at, drain_schema_version,
|
||||
created_at, updated_at
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
drain_id, remote_s, org_s, repo_s, state_norm,
|
||||
clean["reason"], clean["requested_by"],
|
||||
clean["requested_by_profile"], clean["session_id"],
|
||||
entered_at, exited_at,
|
||||
maintenance_drain.DRAIN_SCHEMA_VERSION, now_s, now_s,
|
||||
),
|
||||
)
|
||||
else:
|
||||
drain_id = str(existing["drain_id"])
|
||||
conn.execute(
|
||||
"""
|
||||
UPDATE maintenance_drain
|
||||
SET state = ?, reason = ?, requested_by = ?,
|
||||
requested_by_profile = ?, session_id = ?,
|
||||
entered_at = ?, exited_at = ?,
|
||||
drain_schema_version = ?, updated_at = ?
|
||||
WHERE drain_id = ?
|
||||
""",
|
||||
(
|
||||
state_norm, clean["reason"], clean["requested_by"],
|
||||
clean["requested_by_profile"], clean["session_id"],
|
||||
entered_at, exited_at,
|
||||
maintenance_drain.DRAIN_SCHEMA_VERSION, now_s, drain_id,
|
||||
),
|
||||
)
|
||||
|
||||
if transitioned:
|
||||
event_type = (
|
||||
"maintenance_drain_enter"
|
||||
if state_norm == maintenance_drain.STATE_DRAINING
|
||||
else "maintenance_drain_exit"
|
||||
)
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO events(work_item_id, event_type, message, created_at)
|
||||
VALUES (NULL, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
event_type,
|
||||
f"drain {drain_id} scope {remote_s}/{org_s}/{repo_s} "
|
||||
f"{prior_state} -> {state_norm} by "
|
||||
f"{clean['requested_by'] or '(unknown)'} "
|
||||
f"({clean['requested_by_profile'] or 'no profile'}); "
|
||||
f"reason: {clean['reason'] or '(none)'}",
|
||||
now_s,
|
||||
),
|
||||
)
|
||||
|
||||
row = conn.execute(
|
||||
"SELECT * FROM maintenance_drain WHERE drain_id = ?", (drain_id,)
|
||||
).fetchone()
|
||||
|
||||
record = self._maintenance_drain_row(row) or {}
|
||||
return {
|
||||
"record": record,
|
||||
"drain_id": drain_id,
|
||||
"state": state_norm,
|
||||
"prior_state": prior_state,
|
||||
"transitioned": transitioned,
|
||||
}
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
# MCP maintenance-drain mode (#659)
|
||||
|
||||
Graceful **maintenance drain** stops new work assignment and defers non-allowlisted
|
||||
mutations so sessions can finish critical handoffs and checkpoint before a
|
||||
restart. It is **not** a restart authorization: the drain *proof* and apply gate
|
||||
remain #661.
|
||||
|
||||
## State
|
||||
|
||||
Per repository scope (`remote`/`org`/`repo`) in the control-plane DB table
|
||||
`maintenance_drain` (schema v6):
|
||||
|
||||
| State | Meaning |
|
||||
|-------|---------|
|
||||
| `inactive` | Normal operation (also: no row) |
|
||||
| `draining` | Assignment stopped; non-allowlisted mutations deferred |
|
||||
|
||||
Enter/exit transitions are audited as `maintenance_drain_enter` /
|
||||
`maintenance_drain_exit` events.
|
||||
|
||||
## Tools
|
||||
|
||||
| Tool | Permission | Effect |
|
||||
|------|------------|--------|
|
||||
| `gitea_maintenance_drain_status` | `gitea.read` | Observe drain (every session) |
|
||||
| `gitea_enter_maintenance_drain` | `runtime.maintenance_drain` | Enter drain (capability-gated) |
|
||||
| `gitea_exit_maintenance_drain` | `runtime.maintenance_drain` | Exit drain |
|
||||
|
||||
`runtime.maintenance_drain` is intentionally **not** a `gitea.*` op, so ordinary
|
||||
author profiles cannot enter drain by accident.
|
||||
|
||||
## Enforcement
|
||||
|
||||
1. **Allocator** (`allocate_next_work`): while draining, returns `outcome=wait`
|
||||
with `reason_code=maintenance_drain_assignment_stopped` for dry-run and apply.
|
||||
2. **Mutation preflight** (`verify_preflight_purity`): non-allowlisted mutation
|
||||
tasks raise `MaintenanceDrainError` with a typed next action.
|
||||
3. **Allowlist** (safety only): heartbeats, lease release/abandon, session
|
||||
checkpoints, enter/exit drain. Reads always work.
|
||||
|
||||
## Restart relationship
|
||||
|
||||
Drain mode prepares the blast radius. Restart apply still requires a clean
|
||||
`DrainProof` (#661) or authorized break-glass. Status payloads never claim
|
||||
restart permission.
|
||||
+241
-15
@@ -24,8 +24,6 @@ import subprocess
|
||||
import uuid
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Any
|
||||
import hashlib
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -1474,6 +1472,9 @@ def verify_preflight_purity(
|
||||
# contaminated by manual MCP daemon process killing (reconciler-exempt).
|
||||
_enforce_runtime_recovery_contamination_gate(task, remote)
|
||||
|
||||
# #659 AC3: defer non-allowlisted mutations while maintenance drain is active.
|
||||
_enforce_maintenance_drain_gate(task, remote=remote, org=org, repo=repo)
|
||||
|
||||
ctx = _resolve_namespace_mutation_context(worktree_path)
|
||||
workspace = ctx["workspace_path"]
|
||||
canonical_root = ctx["canonical_repo_root"]
|
||||
@@ -2006,6 +2007,55 @@ def _enforce_runtime_recovery_contamination_gate(
|
||||
)
|
||||
|
||||
|
||||
def _enforce_maintenance_drain_gate(
|
||||
task: str | None,
|
||||
remote: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
) -> None:
|
||||
"""#659 AC3: defer non-allowlisted mutations while drain is active.
|
||||
|
||||
The single mutation chokepoint already used by every gated task, so drain
|
||||
coverage cannot drift per-tool. Allowlisted safety operations (heartbeat,
|
||||
release/abandon, checkpoint, drain exit) pass through so an in-flight
|
||||
session can still finish and hand off; everything else is deferred with a
|
||||
typed blocker. Unreadable drain state fails closed — a drain that cannot be
|
||||
read is not evidence that no drain is running.
|
||||
"""
|
||||
if _preflight_in_test_mode() and not os.environ.get(
|
||||
"GITEA_TEST_FORCE_MAINTENANCE_DRAIN"
|
||||
):
|
||||
return
|
||||
if maintenance_drain.is_allowlisted_task(task):
|
||||
return
|
||||
|
||||
try:
|
||||
_h, o, r = _resolve(remote, None, org, repo)
|
||||
except Exception: # noqa: BLE001 — scope resolution is best-effort here
|
||||
o, r = (org or ""), (repo or "")
|
||||
|
||||
db, errs = _control_plane_db_or_error()
|
||||
if db is None:
|
||||
raise RuntimeError(
|
||||
"maintenance-drain state could not be read: "
|
||||
f"{'; '.join(errs) or 'control-plane DB unavailable'} (fail closed, #659)"
|
||||
)
|
||||
try:
|
||||
record = db.read_maintenance_drain(remote=remote or "", org=o, repo=r)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
raise RuntimeError(
|
||||
f"maintenance-drain state could not be read: {_redact(str(exc))} "
|
||||
"(fail closed, #659)"
|
||||
) from exc
|
||||
|
||||
decision = maintenance_drain.classify_mutation(task, record)
|
||||
if not decision["allowed"]:
|
||||
raise maintenance_drain.MaintenanceDrainError(
|
||||
maintenance_drain.format_drain_block_error(decision),
|
||||
decision=decision,
|
||||
)
|
||||
|
||||
|
||||
def _enforce_stable_branch_contamination_gate(
|
||||
task: str | None,
|
||||
remote: str | None = None,
|
||||
@@ -2066,6 +2116,7 @@ import allocator_service # noqa: E402
|
||||
import allocator_dependencies # noqa: E402
|
||||
import dependency_graph # noqa: E402 # #784 durable dependency edges
|
||||
import control_plane_db # noqa: E402
|
||||
import maintenance_drain # noqa: E402 # #659 graceful maintenance-drain mode
|
||||
import lease_lifecycle # noqa: E402
|
||||
import lease_policy # noqa: E402
|
||||
import workflow_dashboard # noqa: E402 # #605 live queue/lease dashboard
|
||||
@@ -2260,12 +2311,7 @@ def _seed_session_context(
|
||||
expected_username=expected,
|
||||
source=source,
|
||||
canonical_repository_root=canonical_root_pin,
|
||||
cohort_id=_COHORT_ID,
|
||||
startup_sha=_STARTUP_PARITY.get("startup_head"),
|
||||
endpoint=host or (profile.get("base_url") or "").strip() or None,
|
||||
config_fingerprint=_CONFIG_FINGERPRINT,
|
||||
)
|
||||
|
||||
import issue_work_duplicate_gate # noqa: E402
|
||||
import issue_workflow_labels # noqa: E402
|
||||
import terminal_pr_label_cleanup # noqa: E402 # #780 status:pr-open terminal rule
|
||||
@@ -2294,11 +2340,6 @@ import stable_control_runtime # noqa: E402
|
||||
# master has advanced past the running code and fail closed until restart.
|
||||
# Read-only operations are never blocked by staleness.
|
||||
_STARTUP_PARITY = master_parity_gate.capture_startup_parity(PROJECT_ROOT)
|
||||
_COHORT_ID: str = f"cohort-p{os.getpid()}-{_STARTUP_PARITY.get('startup_head') or 'unknown'}"
|
||||
_CONFIG_FINGERPRINT: str = hashlib.sha256(
|
||||
(PROJECT_ROOT + str(_STARTUP_PARITY.get("startup_head"))).encode("utf-8")
|
||||
).hexdigest()[:16]
|
||||
|
||||
|
||||
# Stable-control runtime facts (#615): which runtime this process serves from.
|
||||
# These are the *immutable* facts -- process root, branch, head, checkout-ness --
|
||||
@@ -14209,10 +14250,8 @@ def _current_master_parity() -> dict:
|
||||
current_head = master_parity_gate.read_git_head(PROJECT_ROOT)
|
||||
live_head = master_parity_gate.read_remote_master_head(
|
||||
PROJECT_ROOT, remote=_git_default_remote_name(PROJECT_ROOT))
|
||||
bound_context = session_ctx.get_session_context()
|
||||
return master_parity_gate.assess_master_parity(
|
||||
_STARTUP_PARITY, current_head, live_remote_head=live_head, bound_cohort=bound_context)
|
||||
|
||||
_STARTUP_PARITY, current_head, live_remote_head=live_head)
|
||||
|
||||
|
||||
def _current_runtime_mode_report(refresh: bool = False) -> dict:
|
||||
@@ -22573,6 +22612,193 @@ def gitea_workflow_dashboard(
|
||||
return payload
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_maintenance_drain_status(
|
||||
remote: str = "dadeschools",
|
||||
host: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
) -> dict:
|
||||
"""Read-only: current maintenance-drain state for a repository scope (#659 AC4).
|
||||
|
||||
Every session must be able to observe drain so it can stop creating new work
|
||||
and finish only allowlisted safety operations. Never mutates; never restarts.
|
||||
"""
|
||||
read_block = _profile_operation_gate("gitea.read")
|
||||
if read_block:
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": read_block,
|
||||
"permission_report": _permission_block_report("gitea.read"),
|
||||
}
|
||||
try:
|
||||
_h, o, r = _resolve(remote, host, org, repo)
|
||||
except ValueError as exc:
|
||||
return {"success": False, "read_only": True, "reasons": [str(exc)]}
|
||||
db, errs = _control_plane_db_or_error()
|
||||
if db is None:
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": errs or ["control-plane DB unavailable"],
|
||||
"maintenance_drain": maintenance_drain.status_payload(
|
||||
None, remote=remote, org=o, repo=r
|
||||
),
|
||||
}
|
||||
try:
|
||||
record = db.read_maintenance_drain(remote=remote, org=o, repo=r)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"read_only": True,
|
||||
"reasons": [f"drain state unreadable: {_redact(str(exc))}"],
|
||||
}
|
||||
payload = maintenance_drain.status_payload(
|
||||
record, remote=remote, org=o, repo=r
|
||||
)
|
||||
return {"success": True, "read_only": True, "maintenance_drain": payload}
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_enter_maintenance_drain(
|
||||
reason: str = "",
|
||||
remote: str = "dadeschools",
|
||||
host: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
session_id: str | None = None,
|
||||
) -> dict:
|
||||
"""Enter graceful maintenance-drain mode for a repository scope (#659 AC1).
|
||||
|
||||
Stops new assignment and defers non-allowlisted mutations until exit. Requires
|
||||
``runtime.maintenance_drain`` (controller/lifecycle capability — not granted
|
||||
by ordinary Gitea author profiles). Audited in the control-plane event log.
|
||||
"""
|
||||
cap_block = _profile_operation_gate("runtime.maintenance_drain")
|
||||
if cap_block:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": cap_block,
|
||||
"permission_report": _permission_block_report(
|
||||
"runtime.maintenance_drain"
|
||||
),
|
||||
}
|
||||
try:
|
||||
_h, o, r = _resolve(remote, host, org, repo)
|
||||
except ValueError as exc:
|
||||
return {"success": False, "performed": False, "reasons": [str(exc)]}
|
||||
db, errs = _control_plane_db_or_error()
|
||||
if db is None:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": errs or ["control-plane DB unavailable"],
|
||||
}
|
||||
profile = get_profile() or {}
|
||||
try:
|
||||
result = db.set_maintenance_drain(
|
||||
remote=remote,
|
||||
org=o,
|
||||
repo=r,
|
||||
state=maintenance_drain.STATE_DRAINING,
|
||||
reason=reason or "operator-entered maintenance drain",
|
||||
requested_by=str(
|
||||
(profile.get("identity") or {}).get("username")
|
||||
or profile.get("expected_username")
|
||||
or ""
|
||||
),
|
||||
requested_by_profile=str(profile.get("profile_name") or ""),
|
||||
session_id=str(session_id or ""),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": [f"enter drain failed: {_redact(str(exc))}"],
|
||||
}
|
||||
record = result.get("record") or {}
|
||||
return {
|
||||
"success": True,
|
||||
"performed": True,
|
||||
"transitioned": bool(result.get("transitioned")),
|
||||
"state": result.get("state"),
|
||||
"prior_state": result.get("prior_state"),
|
||||
"drain_id": result.get("drain_id"),
|
||||
"maintenance_drain": maintenance_drain.status_payload(
|
||||
record, remote=remote, org=o, repo=r
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_exit_maintenance_drain(
|
||||
reason: str = "",
|
||||
remote: str = "dadeschools",
|
||||
host: str | None = None,
|
||||
org: str | None = None,
|
||||
repo: str | None = None,
|
||||
session_id: str | None = None,
|
||||
) -> dict:
|
||||
"""Exit graceful maintenance-drain mode (#659 AC1). Restores assignment and mutations."""
|
||||
cap_block = _profile_operation_gate("runtime.maintenance_drain")
|
||||
if cap_block:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": cap_block,
|
||||
"permission_report": _permission_block_report(
|
||||
"runtime.maintenance_drain"
|
||||
),
|
||||
}
|
||||
try:
|
||||
_h, o, r = _resolve(remote, host, org, repo)
|
||||
except ValueError as exc:
|
||||
return {"success": False, "performed": False, "reasons": [str(exc)]}
|
||||
db, errs = _control_plane_db_or_error()
|
||||
if db is None:
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": errs or ["control-plane DB unavailable"],
|
||||
}
|
||||
profile = get_profile() or {}
|
||||
try:
|
||||
result = db.set_maintenance_drain(
|
||||
remote=remote,
|
||||
org=o,
|
||||
repo=r,
|
||||
state=maintenance_drain.STATE_INACTIVE,
|
||||
reason=reason or "operator-exited maintenance drain",
|
||||
requested_by=str(
|
||||
(profile.get("identity") or {}).get("username")
|
||||
or profile.get("expected_username")
|
||||
or ""
|
||||
),
|
||||
requested_by_profile=str(profile.get("profile_name") or ""),
|
||||
session_id=str(session_id or ""),
|
||||
)
|
||||
except Exception as exc: # noqa: BLE001
|
||||
return {
|
||||
"success": False,
|
||||
"performed": False,
|
||||
"reasons": [f"exit drain failed: {_redact(str(exc))}"],
|
||||
}
|
||||
record = result.get("record") or {}
|
||||
return {
|
||||
"success": True,
|
||||
"performed": True,
|
||||
"transitioned": bool(result.get("transitioned")),
|
||||
"state": result.get("state"),
|
||||
"prior_state": result.get("prior_state"),
|
||||
"drain_id": result.get("drain_id"),
|
||||
"maintenance_drain": maintenance_drain.status_payload(
|
||||
record, remote=remote, org=o, repo=r
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
@mcp.tool()
|
||||
def gitea_request_mcp_restart(
|
||||
remote: str = "dadeschools",
|
||||
|
||||
@@ -0,0 +1,281 @@
|
||||
"""Graceful MCP maintenance-drain mode (#659).
|
||||
|
||||
Drain is the visible, capability-gated state that lets an operator stop new
|
||||
work and quiesce mutations *before* a restart, instead of cutting sessions off
|
||||
mid-mutation. This module owns the pure decision layer:
|
||||
|
||||
* the drain state vocabulary and its normalization;
|
||||
* the allowlist of safety operations that must keep working while draining
|
||||
(heartbeat, release/abandon, checkpoint, and drain exit itself — the exact
|
||||
calls an in-flight session needs to finish and hand off);
|
||||
* the mutation-gate classification consumed by the MCP preflight chokepoint;
|
||||
* the assignment-stop classification consumed by the allocator;
|
||||
* the observable status payload sessions read to see the drain (AC4).
|
||||
|
||||
Durable state lives in the control-plane DB (``maintenance_drain`` table);
|
||||
enforcement lives at the existing chokepoints. Nothing here performs I/O, so
|
||||
both callers can share one decision without importing each other.
|
||||
|
||||
Scope note: the machine-verifiable *drain proof* and the restart gate that
|
||||
consumes it are #661's scope, not this module's. Drain here stops assignment
|
||||
and mutation and makes the state observable; it never authorizes a restart.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any, Mapping
|
||||
|
||||
# ── State vocabulary ──────────────────────────────────────────────────────────
|
||||
|
||||
STATE_INACTIVE = "inactive"
|
||||
STATE_DRAINING = "draining"
|
||||
DRAIN_STATES = frozenset({STATE_INACTIVE, STATE_DRAINING})
|
||||
|
||||
# Typed blocker code surfaced to clients (never a bare string at call sites).
|
||||
BLOCKER_DRAIN_ACTIVE = "maintenance_drain_active"
|
||||
|
||||
# Reason code for the allocator's assignment stop.
|
||||
REASON_ASSIGNMENT_STOPPED = "maintenance_drain_assignment_stopped"
|
||||
|
||||
DRAIN_SCHEMA_VERSION = 6
|
||||
|
||||
|
||||
class MaintenanceDrainError(RuntimeError):
|
||||
"""Raised when a mutation is refused because drain is active (fail closed)."""
|
||||
|
||||
def __init__(self, message: str, *, decision: Mapping[str, Any] | None = None):
|
||||
super().__init__(message)
|
||||
self.decision = dict(decision or {})
|
||||
self.reason_code = BLOCKER_DRAIN_ACTIVE
|
||||
|
||||
|
||||
# ── Safety allowlist ──────────────────────────────────────────────────────────
|
||||
|
||||
# Mutations that stay permitted while draining. Every entry is a *quiesce*
|
||||
# operation: it either proves an in-flight task is still alive, hands its claim
|
||||
# back, records the durable state a restart needs, or ends the drain. Nothing
|
||||
# that creates new work, new branches, new PRs, or new review/merge verdicts is
|
||||
# on this list — that is the whole point of the drain.
|
||||
ALLOWLISTED_DRAIN_TASKS: frozenset[str] = frozenset(
|
||||
{
|
||||
# Liveness of work already in flight.
|
||||
"heartbeat_issue_lock",
|
||||
"heartbeat_reviewer_pr_lease",
|
||||
"post_heartbeat",
|
||||
# Handing claims back so nothing is stranded across the restart.
|
||||
"release_workflow_lease",
|
||||
"release_reviewer_pr_lease",
|
||||
"release_merger_pr_lease",
|
||||
"abandon_workflow_lease",
|
||||
# Durable recovery state (#660) must be writable *during* drain.
|
||||
"write_session_checkpoint",
|
||||
"checkpoint_session",
|
||||
# The drain controls themselves — exit must never be self-blocked.
|
||||
"enter_maintenance_drain",
|
||||
"exit_maintenance_drain",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def normalize_task(task: str | None) -> str:
|
||||
"""Normalize a task name, tolerating the ``gitea_`` tool-name prefix."""
|
||||
name = str(task or "").strip()
|
||||
if name.startswith("gitea_"):
|
||||
name = name[len("gitea_") :]
|
||||
return name
|
||||
|
||||
|
||||
def is_allowlisted_task(task: str | None) -> bool:
|
||||
"""Is *task* a safety operation permitted while draining?"""
|
||||
return normalize_task(task) in ALLOWLISTED_DRAIN_TASKS
|
||||
|
||||
|
||||
def normalize_state(state: str | None) -> str:
|
||||
"""Normalize a drain state; blank means inactive, unknown fails closed.
|
||||
|
||||
Blank normalizes to ``inactive`` (no drain record = not draining), but an
|
||||
unrecognized non-blank value raises: silently treating ``"drainig"`` as
|
||||
inactive would disable the gate.
|
||||
"""
|
||||
value = str(state or "").strip().lower()
|
||||
if not value:
|
||||
return STATE_INACTIVE
|
||||
if value not in DRAIN_STATES:
|
||||
raise MaintenanceDrainError(
|
||||
f"unknown maintenance-drain state {value!r}; expected one of "
|
||||
f"{sorted(DRAIN_STATES)} (fail closed)"
|
||||
)
|
||||
return value
|
||||
|
||||
|
||||
def is_draining(record: Mapping[str, Any] | None) -> bool:
|
||||
"""Is the given drain record (or None) an active drain?"""
|
||||
if not record:
|
||||
return False
|
||||
return normalize_state(record.get("state")) == STATE_DRAINING
|
||||
|
||||
|
||||
# ── Decisions ─────────────────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def classify_mutation(
|
||||
task: str | None,
|
||||
record: Mapping[str, Any] | None,
|
||||
) -> dict[str, Any]:
|
||||
"""Decide whether *task* may mutate under the given drain record.
|
||||
|
||||
Returns a decision dict with ``allowed``/``deferred`` and, when refused, a
|
||||
typed ``reason_code`` plus the one exact next action the caller may take.
|
||||
Deferred (not failed): the operation is legal again after drain exits, so
|
||||
the caller is told to wait rather than to retry a different way.
|
||||
"""
|
||||
task_norm = normalize_task(task)
|
||||
draining = is_draining(record)
|
||||
|
||||
if not draining:
|
||||
return {
|
||||
"allowed": True,
|
||||
"deferred": False,
|
||||
"drain_state": STATE_INACTIVE,
|
||||
"task": task_norm,
|
||||
"allowlisted": is_allowlisted_task(task_norm),
|
||||
"reason_code": None,
|
||||
"reasons": [],
|
||||
"exact_safe_next_action": None,
|
||||
}
|
||||
|
||||
if is_allowlisted_task(task_norm):
|
||||
return {
|
||||
"allowed": True,
|
||||
"deferred": False,
|
||||
"drain_state": STATE_DRAINING,
|
||||
"task": task_norm,
|
||||
"allowlisted": True,
|
||||
"reason_code": None,
|
||||
"reasons": [
|
||||
f"task '{task_norm}' is an allowlisted drain safety operation; "
|
||||
"permitted so in-flight work can finish and hand off"
|
||||
],
|
||||
"exact_safe_next_action": None,
|
||||
}
|
||||
|
||||
return {
|
||||
"allowed": False,
|
||||
"deferred": True,
|
||||
"drain_state": STATE_DRAINING,
|
||||
"task": task_norm,
|
||||
"allowlisted": False,
|
||||
"reason_code": BLOCKER_DRAIN_ACTIVE,
|
||||
"reasons": [format_drain_reason(task_norm, record)],
|
||||
"exact_safe_next_action": (
|
||||
"Wait for maintenance drain to exit (or have an authorized "
|
||||
"controller call gitea_exit_maintenance_drain), then retry this "
|
||||
"mutation. Reads and gitea_maintenance_drain_status stay available."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def classify_assignment(record: Mapping[str, Any] | None) -> dict[str, Any]:
|
||||
"""Decide whether the allocator may assign new work (AC2)."""
|
||||
if not is_draining(record):
|
||||
return {
|
||||
"assignment_allowed": True,
|
||||
"drain_state": STATE_INACTIVE,
|
||||
"reason_code": None,
|
||||
"reasons": [],
|
||||
}
|
||||
return {
|
||||
"assignment_allowed": False,
|
||||
"drain_state": STATE_DRAINING,
|
||||
"reason_code": REASON_ASSIGNMENT_STOPPED,
|
||||
"reasons": [
|
||||
"maintenance drain is active: new work assignment is stopped and "
|
||||
"no lease was created (fail closed, #659)" + _scope_suffix(record)
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def format_drain_reason(task: str | None, record: Mapping[str, Any] | None) -> str:
|
||||
"""Human-readable refusal line for a drained mutation."""
|
||||
task_norm = normalize_task(task) or "(unnamed task)"
|
||||
return (
|
||||
f"maintenance drain is active: mutation '{task_norm}' is deferred; only "
|
||||
"allowlisted drain safety operations "
|
||||
f"({', '.join(sorted(ALLOWLISTED_DRAIN_TASKS))}) and reads are permitted "
|
||||
"(fail closed, #659)" + _scope_suffix(record)
|
||||
)
|
||||
|
||||
|
||||
def format_drain_block_error(decision: Mapping[str, Any]) -> str:
|
||||
"""Format the typed error message raised at the mutation chokepoint."""
|
||||
reasons = list(decision.get("reasons") or [])
|
||||
head = reasons[0] if reasons else "maintenance drain is active (fail closed)"
|
||||
action = decision.get("exact_safe_next_action")
|
||||
return f"{head}. Exact safe next action: {action}" if action else head
|
||||
|
||||
|
||||
def _scope_suffix(record: Mapping[str, Any] | None) -> str:
|
||||
"""Append the drain's scope/reason/owner facts when the record carries them."""
|
||||
if not record:
|
||||
return ""
|
||||
bits: list[str] = []
|
||||
scope = "/".join(
|
||||
str(record.get(key) or "") for key in ("remote", "org", "repo")
|
||||
).strip("/")
|
||||
if scope:
|
||||
bits.append(f"scope {scope}")
|
||||
if record.get("reason"):
|
||||
bits.append(f"reason: {record['reason']}")
|
||||
if record.get("requested_by"):
|
||||
bits.append(f"entered by {record['requested_by']}")
|
||||
if record.get("entered_at"):
|
||||
bits.append(f"at {record['entered_at']}")
|
||||
return f" ({'; '.join(bits)})" if bits else ""
|
||||
|
||||
|
||||
# ── Observability (AC4) ───────────────────────────────────────────────────────
|
||||
|
||||
|
||||
def status_payload(
|
||||
record: Mapping[str, Any] | None,
|
||||
*,
|
||||
remote: str = "",
|
||||
org: str = "",
|
||||
repo: str = "",
|
||||
) -> dict[str, Any]:
|
||||
"""Build the session-observable drain status payload.
|
||||
|
||||
Always answers, including when no drain record exists: an absent record is
|
||||
a definitive "not draining", not an unknown.
|
||||
"""
|
||||
draining = is_draining(record)
|
||||
rec: Mapping[str, Any] = record or {}
|
||||
return {
|
||||
"drain_state": STATE_DRAINING if draining else STATE_INACTIVE,
|
||||
"draining": draining,
|
||||
"remote": str(rec.get("remote") or "") or remote,
|
||||
"org": str(rec.get("org") or "") or org,
|
||||
"repo": str(rec.get("repo") or "") or repo,
|
||||
"reason": str(rec.get("reason") or ""),
|
||||
"requested_by": str(rec.get("requested_by") or ""),
|
||||
"requested_by_profile": str(rec.get("requested_by_profile") or ""),
|
||||
"session_id": str(rec.get("session_id") or ""),
|
||||
"entered_at": str(rec.get("entered_at") or ""),
|
||||
"exited_at": str(rec.get("exited_at") or ""),
|
||||
"assignment_stopped": draining,
|
||||
"mutations_deferred": draining,
|
||||
"allowlisted_tasks": sorted(ALLOWLISTED_DRAIN_TASKS),
|
||||
"reads_permitted": True,
|
||||
"record_present": bool(record),
|
||||
"schema_version": DRAIN_SCHEMA_VERSION,
|
||||
"drain_proof_scope": (
|
||||
"drain proof and the restart gate that consumes it are #661 scope; "
|
||||
"this status never authorizes a restart"
|
||||
),
|
||||
"safe_next_action": (
|
||||
"Wait for drain to exit before retrying deferred mutations; "
|
||||
"allowlisted safety operations and reads remain available."
|
||||
if draining
|
||||
else "None; maintenance drain is not active."
|
||||
),
|
||||
}
|
||||
+9
-70
@@ -183,7 +183,6 @@ def assess_master_parity(
|
||||
startup: dict | None,
|
||||
current_head: str | None,
|
||||
live_remote_head: str | None = None,
|
||||
bound_cohort: dict | None = None,
|
||||
) -> dict:
|
||||
"""Compare the startup baseline against the current on-disk ``HEAD``.
|
||||
|
||||
@@ -193,8 +192,7 @@ def assess_master_parity(
|
||||
could not be determined, which is not treated as stale).
|
||||
- ``stale`` -- the on-disk master has definitively advanced past the
|
||||
running process.
|
||||
- ``restart_required`` -- ``stale``, ``live_stale``, or ``cohort_stale``; the
|
||||
recovery action.
|
||||
- ``restart_required`` -- ``stale`` or ``live_stale``; the recovery action.
|
||||
- ``determinable`` -- whether both local HEADs were known well enough to
|
||||
compare.
|
||||
- ``startup_head`` / ``current_head`` / ``reasons``.
|
||||
@@ -211,69 +209,24 @@ def assess_master_parity(
|
||||
- ``live_known`` -- whether the live remote target was resolved.
|
||||
- ``live_stale`` -- the live remote master has advanced past the running
|
||||
process (daemon is behind live master) even if local parity is green.
|
||||
- ``bound_cohort`` -- metadata describing the bound MCP cohort.
|
||||
- ``cohort_parity_match`` -- whether bound cohort startup SHA matches parity.
|
||||
- ``cohort_stale`` -- bound cohort startup SHA is stale relative to parity.
|
||||
- ``mutation_safe`` -- the daemon code, local checkout, live remote target,
|
||||
and bound cohort all agree; the only state in which a mutation may rely
|
||||
on parity.
|
||||
- ``mutation_safe`` -- the daemon code, local checkout, and live remote
|
||||
target all agree; the only state in which a mutation may rely on parity.
|
||||
"""
|
||||
startup_head = (startup or {}).get("startup_head")
|
||||
reasons: list[str] = []
|
||||
|
||||
cohort_info: dict | None = None
|
||||
cohort_parity_match = True
|
||||
cohort_stale = False
|
||||
if bound_cohort:
|
||||
c_id = str(bound_cohort.get("cohort_id") or "").strip() or None
|
||||
c_pid = bound_cohort.get("pid")
|
||||
c_sha = str(
|
||||
bound_cohort.get("startup_sha")
|
||||
or bound_cohort.get("git_head")
|
||||
or ""
|
||||
).strip() or None
|
||||
c_endpoint = str(bound_cohort.get("endpoint") or "").strip() or None
|
||||
c_fingerprint = str(
|
||||
bound_cohort.get("config_fingerprint") or ""
|
||||
).strip() or None
|
||||
|
||||
cohort_info = {
|
||||
"cohort_id": c_id,
|
||||
"pid": c_pid,
|
||||
"startup_sha": c_sha,
|
||||
"endpoint": c_endpoint,
|
||||
"config_fingerprint": c_fingerprint,
|
||||
}
|
||||
|
||||
parity_ref = live_remote_head or current_head or startup_head
|
||||
if c_sha and parity_ref:
|
||||
if c_sha.lower() != parity_ref.lower():
|
||||
cohort_parity_match = False
|
||||
cohort_stale = True
|
||||
reasons.append(
|
||||
f"bound cohort startup SHA '{_short(c_sha)}' does not "
|
||||
f"match authoritative parity SHA '{_short(parity_ref)}' "
|
||||
"(stale cohort refused)"
|
||||
)
|
||||
|
||||
if startup_head is None:
|
||||
reasons.append(
|
||||
"startup commit was not captured; code parity cannot be enforced")
|
||||
return _result(True, False, False, startup_head, current_head,
|
||||
live_remote_head, False, reasons,
|
||||
bound_cohort=cohort_info,
|
||||
cohort_parity_match=cohort_parity_match,
|
||||
cohort_stale=cohort_stale)
|
||||
live_remote_head, False, reasons)
|
||||
|
||||
if current_head is None:
|
||||
reasons.append(
|
||||
"current workspace HEAD could not be read; code parity cannot be "
|
||||
"enforced")
|
||||
return _result(True, False, False, startup_head, current_head,
|
||||
live_remote_head, False, reasons,
|
||||
bound_cohort=cohort_info,
|
||||
cohort_parity_match=cohort_parity_match,
|
||||
cohort_stale=cohort_stale)
|
||||
live_remote_head, False, reasons)
|
||||
|
||||
local_in_parity = startup_head == current_head
|
||||
local_stale = not local_in_parity
|
||||
@@ -293,28 +246,18 @@ def assess_master_parity(
|
||||
|
||||
return _result(
|
||||
local_in_parity, local_stale, True, startup_head, current_head,
|
||||
live_remote_head, live_stale, reasons,
|
||||
bound_cohort=cohort_info,
|
||||
cohort_parity_match=cohort_parity_match,
|
||||
cohort_stale=cohort_stale)
|
||||
live_remote_head, live_stale, reasons)
|
||||
|
||||
|
||||
def _result(in_parity, stale, determinable, startup_head, current_head,
|
||||
live_remote_head, live_stale, reasons, bound_cohort=None,
|
||||
cohort_parity_match=True, cohort_stale=False):
|
||||
live_remote_head, live_stale, reasons):
|
||||
live_known = live_remote_head is not None
|
||||
mutation_safe = (
|
||||
determinable
|
||||
and in_parity
|
||||
and live_known
|
||||
and not live_stale
|
||||
and cohort_parity_match
|
||||
and not cohort_stale
|
||||
)
|
||||
determinable and in_parity and live_known and not live_stale)
|
||||
return {
|
||||
"in_parity": in_parity,
|
||||
"stale": stale,
|
||||
"restart_required": stale or live_stale or cohort_stale,
|
||||
"restart_required": stale or live_stale,
|
||||
"determinable": determinable,
|
||||
"startup_head": startup_head,
|
||||
"current_head": current_head,
|
||||
@@ -324,15 +267,11 @@ def _result(in_parity, stale, determinable, startup_head, current_head,
|
||||
"live_remote_head": live_remote_head,
|
||||
"live_known": live_known,
|
||||
"live_stale": live_stale,
|
||||
"cohort_parity_match": cohort_parity_match,
|
||||
"cohort_stale": cohort_stale,
|
||||
"bound_cohort": bound_cohort,
|
||||
"mutation_safe": mutation_safe,
|
||||
"reasons": list(reasons),
|
||||
}
|
||||
|
||||
|
||||
|
||||
def gate_disabled() -> bool:
|
||||
"""Whether the parity gate is disabled by env escape hatch."""
|
||||
return bool((os.environ.get(ENV_DISABLE) or "").strip())
|
||||
|
||||
+3
-45
@@ -104,8 +104,6 @@ def classify_namespace_probe(
|
||||
profile: str | None = None,
|
||||
configured: bool = True,
|
||||
probe_source: str | None = None,
|
||||
expected_parity_sha: str | None = None,
|
||||
bound_cohort: dict[str, Any] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Classify whether a required tool is callable through a live namespace.
|
||||
|
||||
@@ -137,50 +135,20 @@ def classify_namespace_probe(
|
||||
else:
|
||||
error_type = "namespace_call_failed"
|
||||
|
||||
# Extract cohort metadata
|
||||
cohort_meta = bound_cohort or probe.get("cohort") or probe.get("bound_cohort") or {}
|
||||
cohort_id = str(
|
||||
cohort_meta.get("cohort_id") or probe.get("cohort_id") or ""
|
||||
).strip() or None
|
||||
startup_sha = str(
|
||||
cohort_meta.get("startup_sha")
|
||||
or cohort_meta.get("git_head")
|
||||
or probe.get("startup_sha")
|
||||
or probe.get("git_head")
|
||||
or ""
|
||||
).strip() or None
|
||||
endpoint = str(
|
||||
cohort_meta.get("endpoint") or probe.get("endpoint") or ""
|
||||
).strip() or None
|
||||
config_fingerprint = str(
|
||||
cohort_meta.get("config_fingerprint") or probe.get("config_fingerprint") or ""
|
||||
).strip() or None
|
||||
|
||||
expected_sha = (expected_parity_sha or "").strip().lower() or None
|
||||
stale_cohort = False
|
||||
if expected_sha and startup_sha:
|
||||
if startup_sha.lower() != expected_sha:
|
||||
stale_cohort = True
|
||||
error_type = "stale_cohort_refused"
|
||||
|
||||
if not configured:
|
||||
error_type = "namespace_not_configured"
|
||||
elif registered is False:
|
||||
error_type = "tool_missing"
|
||||
elif not probe_result:
|
||||
error_type = "live_probe_missing"
|
||||
elif stale_cohort:
|
||||
error_type = "stale_cohort_refused"
|
||||
elif not probe_success and not error_type:
|
||||
error_type = "namespace_call_failed"
|
||||
|
||||
callable_live = bool(configured and probe_result and probe_success and not stale_cohort)
|
||||
callable_live = bool(configured and probe_result and probe_success)
|
||||
# Probe-path health (spawn or client). IDE-proven only for client path.
|
||||
healthy = bool(configured and registered is not False and callable_live)
|
||||
ide_namespace_proven = bool(healthy and source == PROBE_SOURCE_CLIENT)
|
||||
process_pid = process.get("pid") if isinstance(process, dict) else (
|
||||
cohort_meta.get("pid") if isinstance(cohort_meta, dict) else None
|
||||
)
|
||||
process_pid = process.get("pid") if isinstance(process, dict) else None
|
||||
profile_name = profile or (
|
||||
process.get("profile") if isinstance(process, dict) else None
|
||||
)
|
||||
@@ -193,13 +161,7 @@ def classify_namespace_probe(
|
||||
reasons.append(
|
||||
f"Required tool '{tool}' is not registered in namespace '{ns}'."
|
||||
)
|
||||
if error_type == "stale_cohort_refused":
|
||||
reasons.append(
|
||||
f"Bound cohort startup SHA '{startup_sha[:12] if startup_sha else 'unknown'}' "
|
||||
f"does not match expected parity SHA '{expected_sha[:12] if expected_sha else 'unknown'}' "
|
||||
"(stale cohort refused)."
|
||||
)
|
||||
elif error_type == "live_probe_missing":
|
||||
if error_type == "live_probe_missing":
|
||||
reasons.append(
|
||||
f"No live client invocation proof was supplied for '{ns}.{tool}'."
|
||||
)
|
||||
@@ -286,10 +248,6 @@ def classify_namespace_probe(
|
||||
"env": env_summary,
|
||||
"config_path": config_path,
|
||||
"probe_source": source,
|
||||
"cohort_id": cohort_id,
|
||||
"startup_sha": startup_sha,
|
||||
"endpoint": endpoint,
|
||||
"config_fingerprint": config_fingerprint,
|
||||
},
|
||||
"blocks_merge_workflow": blocks,
|
||||
}
|
||||
|
||||
@@ -151,16 +151,12 @@ class RestartCompletionProof:
|
||||
unresolved_count: int
|
||||
skipped_count: int
|
||||
note: str
|
||||
binding_unchanged: bool = False
|
||||
prior_reconcile_id: str | None = None
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"schema_version": self.schema_version,
|
||||
"reconcile_version": self.reconcile_version,
|
||||
"reconcile_id": self.reconcile_id,
|
||||
"binding_unchanged": self.binding_unchanged,
|
||||
"prior_reconcile_id": self.prior_reconcile_id,
|
||||
"started_at": self.started_at,
|
||||
"finished_at": self.finished_at,
|
||||
"boot_head_sha": self.boot_head_sha,
|
||||
@@ -177,7 +173,6 @@ class RestartCompletionProof:
|
||||
"skipped_count": self.skipped_count,
|
||||
"note": self.note,
|
||||
"links": {
|
||||
|
||||
"umbrella": 655,
|
||||
"vision": 652,
|
||||
"roadmap": 653,
|
||||
@@ -364,9 +359,7 @@ def reconcile_after_restart(
|
||||
now: datetime | None = None,
|
||||
mode: str = MODE_LOG_ONLY,
|
||||
reconcile_id: str | None = None,
|
||||
prior_reconcile_id: str | None = None,
|
||||
) -> RestartCompletionProof:
|
||||
|
||||
"""Classify a post-restart inventory into a completion proof (#662).
|
||||
|
||||
Parameters
|
||||
@@ -757,26 +750,12 @@ def reconcile_after_restart(
|
||||
f"Mode={mode_norm}."
|
||||
)
|
||||
|
||||
prior_id = str(
|
||||
prior_reconcile_id
|
||||
or inventory.get("prior_reconcile_id")
|
||||
or ""
|
||||
).strip() or None
|
||||
target_rec_id = str(reconcile_id or "").strip() or None
|
||||
binding_unchanged = bool(
|
||||
prior_id and target_rec_id and target_rec_id == prior_id
|
||||
)
|
||||
final_reconcile_id = target_rec_id or f"reconcile-{uuid4().hex[:12]}"
|
||||
|
||||
return RestartCompletionProof(
|
||||
schema_version=SCHEMA_VERSION,
|
||||
reconcile_version=RECONCILE_VERSION,
|
||||
reconcile_id=final_reconcile_id,
|
||||
binding_unchanged=binding_unchanged,
|
||||
prior_reconcile_id=prior_id,
|
||||
reconcile_id=(reconcile_id or f"reconcile-{uuid4().hex[:12]}"),
|
||||
started_at=_ts(started),
|
||||
finished_at=_ts(finished),
|
||||
|
||||
boot_head_sha=(
|
||||
str(inventory.get("boot_head_sha")).strip()
|
||||
if inventory.get("boot_head_sha")
|
||||
|
||||
@@ -33,10 +33,6 @@ class _SessionContext:
|
||||
source: str
|
||||
pid: int
|
||||
canonical_repository_root: str | None = None
|
||||
cohort_id: str | None = None
|
||||
startup_sha: str | None = None
|
||||
endpoint: str | None = None
|
||||
config_fingerprint: str | None = None
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
@@ -51,14 +47,9 @@ class _SessionContext:
|
||||
"source": self.source,
|
||||
"pid": self.pid,
|
||||
"canonical_repository_root": self.canonical_repository_root,
|
||||
"cohort_id": self.cohort_id,
|
||||
"startup_sha": self.startup_sha,
|
||||
"endpoint": self.endpoint,
|
||||
"config_fingerprint": self.config_fingerprint,
|
||||
}
|
||||
|
||||
|
||||
|
||||
# Process-local only — never a shared file (same rationale as mutation authority).
|
||||
# The frozen value prevents partial mutation, while the lock makes first-bind and
|
||||
# sanctioned rebind atomic across concurrent MCP calls.
|
||||
@@ -81,13 +72,6 @@ def _reset_session_context_for_testing() -> None:
|
||||
_SESSION_CONTEXT = None
|
||||
|
||||
|
||||
def clear_session_context() -> None:
|
||||
"""Purge process-session context and cohort bindings on disconnect."""
|
||||
global _SESSION_CONTEXT
|
||||
with _SESSION_CONTEXT_LOCK:
|
||||
_SESSION_CONTEXT = None
|
||||
|
||||
|
||||
def get_session_context() -> dict[str, Any] | None:
|
||||
"""Return a detached snapshot of the bound context, or None if unbound."""
|
||||
with _SESSION_CONTEXT_LOCK:
|
||||
@@ -162,10 +146,6 @@ def bind_session_context(
|
||||
expected_username: str | None = None,
|
||||
source: str = "bind",
|
||||
canonical_repository_root: str | None = None,
|
||||
cohort_id: str | None = None,
|
||||
startup_sha: str | None = None,
|
||||
endpoint: str | None = None,
|
||||
config_fingerprint: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Atomically bind/re-bind context (the explicit activation path)."""
|
||||
with _SESSION_CONTEXT_LOCK:
|
||||
@@ -180,10 +160,6 @@ def bind_session_context(
|
||||
expected_username=expected_username,
|
||||
source=source,
|
||||
canonical_repository_root=canonical_repository_root,
|
||||
cohort_id=cohort_id,
|
||||
startup_sha=startup_sha,
|
||||
endpoint=endpoint,
|
||||
config_fingerprint=config_fingerprint,
|
||||
)
|
||||
|
||||
|
||||
@@ -199,10 +175,6 @@ def _bind_session_context_unlocked(
|
||||
expected_username: str | None,
|
||||
source: str,
|
||||
canonical_repository_root: str | None = None,
|
||||
cohort_id: str | None = None,
|
||||
startup_sha: str | None = None,
|
||||
endpoint: str | None = None,
|
||||
config_fingerprint: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Store a complete immutable context while the caller holds the lock."""
|
||||
global _SESSION_CONTEXT
|
||||
@@ -218,10 +190,6 @@ def _bind_session_context_unlocked(
|
||||
source=source,
|
||||
pid=os.getpid(),
|
||||
canonical_repository_root=(canonical_repository_root or "").strip() or None,
|
||||
cohort_id=(cohort_id or "").strip() or None,
|
||||
startup_sha=(startup_sha or "").strip() or None,
|
||||
endpoint=(endpoint or "").strip() or None,
|
||||
config_fingerprint=(config_fingerprint or "").strip() or None,
|
||||
)
|
||||
return _SESSION_CONTEXT.as_dict()
|
||||
|
||||
@@ -238,10 +206,6 @@ def seed_session_context_if_unbound(
|
||||
expected_username: str | None = None,
|
||||
source: str = "seed",
|
||||
canonical_repository_root: str | None = None,
|
||||
cohort_id: str | None = None,
|
||||
startup_sha: str | None = None,
|
||||
endpoint: str | None = None,
|
||||
config_fingerprint: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Atomically bind only when this process has no current context.
|
||||
|
||||
@@ -263,15 +227,10 @@ def seed_session_context_if_unbound(
|
||||
expected_username=expected_username,
|
||||
source=source,
|
||||
canonical_repository_root=canonical_repository_root,
|
||||
cohort_id=cohort_id,
|
||||
startup_sha=startup_sha,
|
||||
endpoint=endpoint,
|
||||
config_fingerprint=config_fingerprint,
|
||||
)
|
||||
return _SESSION_CONTEXT.as_dict()
|
||||
|
||||
|
||||
|
||||
def assess_session_context(
|
||||
*,
|
||||
profile_name: str | None,
|
||||
@@ -627,10 +586,6 @@ def mutation_context_audit_fields(
|
||||
"session_identity": None,
|
||||
"session_repository": None,
|
||||
"session_org": None,
|
||||
"session_cohort_id": None,
|
||||
"session_startup_sha": None,
|
||||
"session_endpoint": None,
|
||||
"session_config_fingerprint": None,
|
||||
}
|
||||
return {
|
||||
"session_context_bound": True,
|
||||
@@ -643,14 +598,9 @@ def mutation_context_audit_fields(
|
||||
"session_role_kind": data.get("role_kind"),
|
||||
"session_context_source": data.get("source"),
|
||||
"session_canonical_repository_root": data.get("canonical_repository_root"),
|
||||
"session_cohort_id": data.get("cohort_id"),
|
||||
"session_startup_sha": data.get("startup_sha"),
|
||||
"session_endpoint": data.get("endpoint"),
|
||||
"session_config_fingerprint": data.get("config_fingerprint"),
|
||||
}
|
||||
|
||||
|
||||
|
||||
def _assessment(
|
||||
proven: bool, reasons: list[str], ctx: Mapping[str, Any] | None
|
||||
) -> dict[str, Any]:
|
||||
|
||||
@@ -414,6 +414,24 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"role": "controller",
|
||||
},
|
||||
|
||||
# #659 maintenance drain. Same reasoning as the lifecycle controls above:
|
||||
# entering/exiting drain quiesces a whole namespace, so it carries a
|
||||
# non-``gitea.*`` permission that no configured Gitea profile satisfies by
|
||||
# accident (AC1 — capability-gated and audited). Reading drain state is
|
||||
# ordinary read authority: every session must be able to see the drain (AC4).
|
||||
"enter_maintenance_drain": {
|
||||
"permission": "runtime.maintenance_drain",
|
||||
"role": "controller",
|
||||
},
|
||||
"exit_maintenance_drain": {
|
||||
"permission": "runtime.maintenance_drain",
|
||||
"role": "controller",
|
||||
},
|
||||
"maintenance_drain_status": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
|
||||
# #601 first-class lease lifecycle — inspect/list need read; mutations gate on
|
||||
# ownership in the control-plane DB (not a separate Gitea write permission).
|
||||
"list_workflow_leases": {
|
||||
|
||||
@@ -1,253 +0,0 @@
|
||||
"""Tests for Issue #689: Deterministic MCP namespace attachment.
|
||||
|
||||
Verifies cohort identity exposure, stale cohort refusal, parity matching,
|
||||
reconcile_id freshness, session context cleanup, and regression scenarios.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
import master_parity_gate
|
||||
import mcp_namespace_health
|
||||
import post_restart_reconcile
|
||||
import session_context_binding as session_ctx
|
||||
|
||||
|
||||
class TestIssue689DeterministicCohortAttachment(unittest.TestCase):
|
||||
"""Suite covering Issue #689 acceptance criteria."""
|
||||
|
||||
def setUp(self) -> None:
|
||||
session_ctx._reset_session_context_for_testing()
|
||||
|
||||
def tearDown(self) -> None:
|
||||
session_ctx._reset_session_context_for_testing()
|
||||
|
||||
def test_ac1_session_context_exposes_cohort_identity(self) -> None:
|
||||
"""AC1: Session context exposes cohort ID, startup SHA, endpoint, and config fingerprint."""
|
||||
ctx = session_ctx.bind_session_context(
|
||||
profile_name="prgs-author",
|
||||
remote="prgs",
|
||||
host="gitea.prgs.cc",
|
||||
identity="jcwalker3",
|
||||
repository="Gitea-Tools",
|
||||
org="Scaled-Tech-Consulting",
|
||||
role_kind="author",
|
||||
cohort_id="cohort-p1234-abc123456789",
|
||||
startup_sha="abc123456789def",
|
||||
endpoint="gitea.prgs.cc",
|
||||
config_fingerprint="fingerprint12345",
|
||||
)
|
||||
self.assertEqual(ctx["cohort_id"], "cohort-p1234-abc123456789")
|
||||
self.assertEqual(ctx["startup_sha"], "abc123456789def")
|
||||
self.assertEqual(ctx["endpoint"], "gitea.prgs.cc")
|
||||
self.assertEqual(ctx["config_fingerprint"], "fingerprint12345")
|
||||
|
||||
fetched = session_ctx.get_session_context()
|
||||
self.assertIsNotNone(fetched)
|
||||
self.assertEqual(fetched["cohort_id"], "cohort-p1234-abc123456789")
|
||||
self.assertEqual(fetched["startup_sha"], "abc123456789def")
|
||||
|
||||
def test_ac2_stale_cohort_refused_by_probe_classifier(self) -> None:
|
||||
"""AC2: Probe classifier refuses binding to a stale cohort as stale_cohort_refused."""
|
||||
probe_res = {
|
||||
"success": True,
|
||||
"cohort": {
|
||||
"cohort_id": "cohort-obsolete-1",
|
||||
"startup_sha": "22698c1000000000000000000000000000000000",
|
||||
"endpoint": "gitea.prgs.cc",
|
||||
"config_fingerprint": "fp-old",
|
||||
},
|
||||
}
|
||||
res = mcp_namespace_health.classify_namespace_probe(
|
||||
"gitea-author",
|
||||
probe_result=probe_res,
|
||||
probe_source="client_namespace",
|
||||
expected_parity_sha="a4c73766f4b0cc32f7c3808688eceeb6fee74335",
|
||||
)
|
||||
self.assertFalse(res["healthy"])
|
||||
self.assertFalse(res["success"])
|
||||
self.assertEqual(res["error_type"], "stale_cohort_refused")
|
||||
self.assertIn("stale cohort refused", " ".join(res["reasons"]))
|
||||
self.assertEqual(
|
||||
res["diagnostics"]["startup_sha"],
|
||||
"22698c1000000000000000000000000000000000",
|
||||
)
|
||||
|
||||
def test_ac3_reconnection_parity_matching_and_fail_closed(self) -> None:
|
||||
"""AC3: Parity gate fails closed when bound cohort startup SHA mismatches parity."""
|
||||
startup = {"startup_head": "a4c73766f4b0cc32f7c3808688eceeb6fee74335"}
|
||||
current = "a4c73766f4b0cc32f7c3808688eceeb6fee74335"
|
||||
live_remote = "a4c73766f4b0cc32f7c3808688eceeb6fee74335"
|
||||
|
||||
# Matching cohort
|
||||
matching_cohort = {
|
||||
"cohort_id": "cohort-fresh",
|
||||
"startup_sha": "a4c73766f4b0cc32f7c3808688eceeb6fee74335",
|
||||
}
|
||||
res_matching = master_parity_gate.assess_master_parity(
|
||||
startup, current, live_remote_head=live_remote, bound_cohort=matching_cohort
|
||||
)
|
||||
self.assertTrue(res_matching["cohort_parity_match"])
|
||||
self.assertFalse(res_matching["cohort_stale"])
|
||||
self.assertTrue(res_matching["mutation_safe"])
|
||||
|
||||
# Mismatched obsolete cohort
|
||||
obsolete_cohort = {
|
||||
"cohort_id": "cohort-obsolete-22698c1",
|
||||
"startup_sha": "22698c1000000000000000000000000000000000",
|
||||
}
|
||||
res_stale = master_parity_gate.assess_master_parity(
|
||||
startup, current, live_remote_head=live_remote, bound_cohort=obsolete_cohort
|
||||
)
|
||||
self.assertFalse(res_stale["cohort_parity_match"])
|
||||
self.assertTrue(res_stale["cohort_stale"])
|
||||
self.assertTrue(res_stale["restart_required"])
|
||||
self.assertFalse(res_stale["mutation_safe"])
|
||||
|
||||
def test_ac4_reconcile_id_freshness(self) -> None:
|
||||
"""AC4: Re-attachment distinguishes new reconcile_id from preserved binding."""
|
||||
inventory = {"inventory_complete": True}
|
||||
|
||||
# New attachment generates fresh reconcile_id
|
||||
proof1 = post_restart_reconcile.reconcile_after_restart(inventory)
|
||||
self.assertFalse(proof1.binding_unchanged)
|
||||
self.assertTrue(proof1.reconcile_id.startswith("reconcile-"))
|
||||
|
||||
# Preserved binding reports binding_unchanged=True
|
||||
proof2 = post_restart_reconcile.reconcile_after_restart(
|
||||
inventory,
|
||||
reconcile_id=proof1.reconcile_id,
|
||||
prior_reconcile_id=proof1.reconcile_id,
|
||||
)
|
||||
self.assertTrue(proof2.binding_unchanged)
|
||||
self.assertEqual(proof2.reconcile_id, proof1.reconcile_id)
|
||||
|
||||
# Disconnected re-attachment gets new reconcile_id
|
||||
proof3 = post_restart_reconcile.reconcile_after_restart(
|
||||
inventory,
|
||||
prior_reconcile_id=proof1.reconcile_id,
|
||||
)
|
||||
self.assertFalse(proof3.binding_unchanged)
|
||||
self.assertNotEqual(proof3.reconcile_id, proof1.reconcile_id)
|
||||
|
||||
def test_ac5_session_disconnect_clears_bindings(self) -> None:
|
||||
"""AC5: clear_session_context purges session context on disconnect."""
|
||||
session_ctx.bind_session_context(
|
||||
profile_name="prgs-author",
|
||||
remote="prgs",
|
||||
host="gitea.prgs.cc",
|
||||
identity="jcwalker3",
|
||||
cohort_id="cohort-1",
|
||||
)
|
||||
self.assertIsNotNone(session_ctx.get_session_context())
|
||||
|
||||
session_ctx.clear_session_context()
|
||||
self.assertIsNone(session_ctx.get_session_context())
|
||||
|
||||
def test_ac6_bound_cohort_in_diagnostics(self) -> None:
|
||||
"""AC6: Bound cohort identity appears in audit diagnostics."""
|
||||
session_ctx.bind_session_context(
|
||||
profile_name="prgs-author",
|
||||
remote="prgs",
|
||||
host="gitea.prgs.cc",
|
||||
identity="jcwalker3",
|
||||
cohort_id="cohort-test-99",
|
||||
startup_sha="sha99999",
|
||||
endpoint="gitea.prgs.cc",
|
||||
config_fingerprint="fp999",
|
||||
)
|
||||
audit = session_ctx.mutation_context_audit_fields()
|
||||
self.assertTrue(audit["session_context_bound"])
|
||||
self.assertEqual(audit["session_cohort_id"], "cohort-test-99")
|
||||
self.assertEqual(audit["session_startup_sha"], "sha99999")
|
||||
self.assertEqual(audit["session_endpoint"], "gitea.prgs.cc")
|
||||
self.assertEqual(audit["session_config_fingerprint"], "fp999")
|
||||
|
||||
def test_ac7_regression_n_reconnects_never_bind_to_obsolete_daemon(self) -> None:
|
||||
"""AC7: N reconnects against a daemon set containing obsolete daemons never bind obsolete ones."""
|
||||
live_master = "master-head-latest-12345"
|
||||
daemons = [
|
||||
{"id": "d1", "startup_sha": "obsolete-head-11111"},
|
||||
{"id": "d2", "startup_sha": "obsolete-head-22698c1"},
|
||||
{"id": "d3", "startup_sha": live_master},
|
||||
{"id": "d4", "startup_sha": "obsolete-head-33333"},
|
||||
]
|
||||
|
||||
for _ in range(5):
|
||||
for daemon in daemons:
|
||||
res = master_parity_gate.assess_master_parity(
|
||||
{"startup_head": live_master},
|
||||
live_master,
|
||||
live_remote_head=live_master,
|
||||
bound_cohort=daemon,
|
||||
)
|
||||
if daemon["startup_sha"] != live_master:
|
||||
self.assertFalse(res["mutation_safe"])
|
||||
self.assertTrue(res["cohort_stale"])
|
||||
else:
|
||||
self.assertTrue(res["mutation_safe"])
|
||||
self.assertFalse(res["cohort_stale"])
|
||||
|
||||
def test_ac8_regression_incident_shape_reproduction(self) -> None:
|
||||
"""AC8: Reproduce incident shape — obsolete cohort 22698c1 resident vs newer daemon."""
|
||||
live_master = "a4c73766f4b0cc32f7c3808688eceeb6fee74335"
|
||||
obsolete_cohort = {
|
||||
"cohort_id": "cohort-resident-22698c1",
|
||||
"startup_sha": "22698c1000000000000000000000000000000000",
|
||||
}
|
||||
new_cohort = {
|
||||
"cohort_id": "cohort-spawned-new",
|
||||
"startup_sha": live_master,
|
||||
}
|
||||
|
||||
# Obsolete cohort fails parity check
|
||||
obs_res = mcp_namespace_health.classify_namespace_probe(
|
||||
"gitea-author",
|
||||
probe_result={"success": True, "cohort": obsolete_cohort},
|
||||
probe_source="client_namespace",
|
||||
expected_parity_sha=live_master,
|
||||
)
|
||||
self.assertFalse(obs_res["healthy"])
|
||||
self.assertEqual(obs_res["error_type"], "stale_cohort_refused")
|
||||
|
||||
# Fresh cohort succeeds
|
||||
new_res = mcp_namespace_health.classify_namespace_probe(
|
||||
"gitea-author",
|
||||
probe_result={"success": True, "cohort": new_cohort},
|
||||
probe_source="client_namespace",
|
||||
expected_parity_sha=live_master,
|
||||
)
|
||||
self.assertTrue(new_res["healthy"])
|
||||
|
||||
def test_ac9_regression_bound_cohort_going_stale_detected(self) -> None:
|
||||
"""AC9: A bound cohort that later goes stale is detected on next attachment check."""
|
||||
initial_master = "sha-v1-initial"
|
||||
cohort = {"cohort_id": "c1", "startup_sha": initial_master}
|
||||
|
||||
# Initial state: in parity
|
||||
res1 = master_parity_gate.assess_master_parity(
|
||||
{"startup_head": initial_master},
|
||||
initial_master,
|
||||
live_remote_head=initial_master,
|
||||
bound_cohort=cohort,
|
||||
)
|
||||
self.assertTrue(res1["mutation_safe"])
|
||||
|
||||
# Master advances to sha-v2-advanced while cohort remains at sha-v1-initial
|
||||
advanced_master = "sha-v2-advanced"
|
||||
res2 = master_parity_gate.assess_master_parity(
|
||||
{"startup_head": initial_master},
|
||||
advanced_master,
|
||||
live_remote_head=advanced_master,
|
||||
bound_cohort=cohort,
|
||||
)
|
||||
self.assertFalse(res2["mutation_safe"])
|
||||
self.assertTrue(res2["restart_required"])
|
||||
self.assertTrue(res2["cohort_stale"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,237 @@
|
||||
"""Tests for graceful MCP maintenance-drain mode (#659).
|
||||
|
||||
Acceptance coverage:
|
||||
|
||||
1. Enter/exit is durable and audited (DB substrate).
|
||||
2. New work assignment stops during drain (allocator WAIT).
|
||||
3. Mutations deferred except allowlisted safety ops.
|
||||
4. Sessions can observe drain state.
|
||||
5. Fail-closed on unreadable drain state.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from unittest import mock
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
import maintenance_drain
|
||||
from control_plane_db import ControlPlaneDB
|
||||
from allocator_service import WorkCandidate, allocate_next_work, OUTCOME_WAIT
|
||||
|
||||
|
||||
class TestDrainDecisions(unittest.TestCase):
|
||||
def test_inactive_allows_mutations_and_assignment(self):
|
||||
decision = maintenance_drain.classify_mutation("create_pr", None)
|
||||
self.assertTrue(decision["allowed"])
|
||||
self.assertFalse(decision["deferred"])
|
||||
assign = maintenance_drain.classify_assignment(None)
|
||||
self.assertTrue(assign["assignment_allowed"])
|
||||
|
||||
def test_draining_defers_non_allowlisted_mutation(self):
|
||||
record = {"state": "draining", "remote": "prgs", "org": "o", "repo": "r"}
|
||||
decision = maintenance_drain.classify_mutation("create_pr", record)
|
||||
self.assertFalse(decision["allowed"])
|
||||
self.assertTrue(decision["deferred"])
|
||||
self.assertEqual(decision["reason_code"], maintenance_drain.BLOCKER_DRAIN_ACTIVE)
|
||||
self.assertIn("create_pr", decision["reasons"][0])
|
||||
|
||||
def test_allowlisted_safety_ops_pass_during_drain(self):
|
||||
record = {"state": "draining"}
|
||||
for task in (
|
||||
"heartbeat_issue_lock",
|
||||
"gitea_release_reviewer_pr_lease",
|
||||
"write_session_checkpoint",
|
||||
"exit_maintenance_drain",
|
||||
):
|
||||
with self.subTest(task=task):
|
||||
decision = maintenance_drain.classify_mutation(task, record)
|
||||
self.assertTrue(decision["allowed"], decision)
|
||||
|
||||
def test_assignment_stopped_during_drain(self):
|
||||
record = {"state": "draining", "reason": "upgrade"}
|
||||
decision = maintenance_drain.classify_assignment(record)
|
||||
self.assertFalse(decision["assignment_allowed"])
|
||||
self.assertEqual(
|
||||
decision["reason_code"], maintenance_drain.REASON_ASSIGNMENT_STOPPED
|
||||
)
|
||||
|
||||
def test_unknown_state_fails_closed(self):
|
||||
with self.assertRaises(maintenance_drain.MaintenanceDrainError):
|
||||
maintenance_drain.normalize_state("drainig")
|
||||
|
||||
def test_status_payload_always_answers(self):
|
||||
inactive = maintenance_drain.status_payload(None, remote="prgs", org="o", repo="r")
|
||||
self.assertFalse(inactive["draining"])
|
||||
self.assertTrue(inactive["reads_permitted"])
|
||||
active = maintenance_drain.status_payload(
|
||||
{"state": "draining", "reason": "reboot", "requested_by": "ops"},
|
||||
remote="prgs",
|
||||
org="o",
|
||||
repo="r",
|
||||
)
|
||||
self.assertTrue(active["draining"])
|
||||
self.assertTrue(active["assignment_stopped"])
|
||||
self.assertTrue(active["mutations_deferred"])
|
||||
self.assertIn("heartbeat_issue_lock", active["allowlisted_tasks"])
|
||||
|
||||
|
||||
class TestDrainDB(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.db = ControlPlaneDB(db_path=str(Path(self._tmp.name) / "cp.sqlite3"))
|
||||
|
||||
def test_enter_exit_idempotent_and_audited(self):
|
||||
first = self.db.set_maintenance_drain(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
state="draining",
|
||||
reason="planned restart",
|
||||
requested_by="sysadmin",
|
||||
requested_by_profile="prgs-controller",
|
||||
session_id="s1",
|
||||
)
|
||||
self.assertTrue(first["transitioned"])
|
||||
self.assertEqual(first["state"], "draining")
|
||||
self.assertTrue(maintenance_drain.is_draining(first["record"]))
|
||||
|
||||
again = self.db.set_maintenance_drain(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
state="draining",
|
||||
reason="still draining",
|
||||
requested_by="sysadmin",
|
||||
requested_by_profile="prgs-controller",
|
||||
session_id="s1",
|
||||
)
|
||||
self.assertFalse(again["transitioned"])
|
||||
self.assertEqual(again["record"]["entered_at"], first["record"]["entered_at"])
|
||||
|
||||
exited = self.db.set_maintenance_drain(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
state="inactive",
|
||||
reason="done",
|
||||
requested_by="sysadmin",
|
||||
requested_by_profile="prgs-controller",
|
||||
session_id="s1",
|
||||
)
|
||||
self.assertTrue(exited["transitioned"])
|
||||
self.assertFalse(maintenance_drain.is_draining(exited["record"]))
|
||||
self.assertTrue(exited["record"]["exited_at"])
|
||||
|
||||
# Events recorded for transitions only (enter + exit).
|
||||
with self.db._tx(immediate=False) as conn:
|
||||
rows = conn.execute(
|
||||
"SELECT event_type FROM events WHERE event_type LIKE 'maintenance_drain_%' "
|
||||
"ORDER BY event_id"
|
||||
).fetchall()
|
||||
types = [r[0] for r in rows]
|
||||
self.assertEqual(types, ["maintenance_drain_enter", "maintenance_drain_exit"])
|
||||
|
||||
def test_read_missing_is_none_not_error(self):
|
||||
self.assertIsNone(
|
||||
self.db.read_maintenance_drain(remote="prgs", org="o", repo="r")
|
||||
)
|
||||
|
||||
|
||||
class TestAllocatorStopsDuringDrain(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.addCleanup(self._tmp.cleanup)
|
||||
self.db = ControlPlaneDB(db_path=str(Path(self._tmp.name) / "cp.sqlite3"))
|
||||
|
||||
def test_allocate_returns_wait_while_draining(self):
|
||||
self.db.set_maintenance_drain(
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
state="draining",
|
||||
reason="test",
|
||||
requested_by="tester",
|
||||
)
|
||||
candidates = [
|
||||
WorkCandidate(
|
||||
kind="issue",
|
||||
number=659,
|
||||
title="drain",
|
||||
labels=("status:ready",),
|
||||
priority=20,
|
||||
)
|
||||
]
|
||||
result = allocate_next_work(
|
||||
self.db,
|
||||
role="author",
|
||||
session_id="test-session",
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
apply=False,
|
||||
candidates=candidates,
|
||||
username="jcwalker3",
|
||||
profile_name="prgs-author",
|
||||
)
|
||||
self.assertEqual(result["outcome"], OUTCOME_WAIT)
|
||||
self.assertIsNone(result.get("selected"))
|
||||
self.assertEqual(
|
||||
result.get("reason_code"),
|
||||
maintenance_drain.REASON_ASSIGNMENT_STOPPED,
|
||||
)
|
||||
self.assertTrue(result["maintenance_drain"]["draining"])
|
||||
|
||||
def test_allocate_works_when_inactive(self):
|
||||
candidates = [
|
||||
WorkCandidate(
|
||||
kind="issue",
|
||||
number=659,
|
||||
title="drain",
|
||||
labels=("status:ready",),
|
||||
priority=20,
|
||||
)
|
||||
]
|
||||
result = allocate_next_work(
|
||||
self.db,
|
||||
role="author",
|
||||
session_id="test-session-2",
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="Gitea-Tools",
|
||||
apply=False,
|
||||
candidates=candidates,
|
||||
username="jcwalker3",
|
||||
profile_name="prgs-author",
|
||||
)
|
||||
self.assertNotEqual(
|
||||
result.get("reason_code"),
|
||||
maintenance_drain.REASON_ASSIGNMENT_STOPPED,
|
||||
)
|
||||
|
||||
|
||||
class TestCapabilityMap(unittest.TestCase):
|
||||
def test_drain_tasks_mapped(self):
|
||||
import task_capability_map as tcm
|
||||
|
||||
self.assertEqual(
|
||||
tcm.required_permission("enter_maintenance_drain"),
|
||||
"runtime.maintenance_drain",
|
||||
)
|
||||
self.assertEqual(
|
||||
tcm.required_permission("exit_maintenance_drain"),
|
||||
"runtime.maintenance_drain",
|
||||
)
|
||||
self.assertEqual(
|
||||
tcm.required_permission("maintenance_drain_status"),
|
||||
"gitea.read",
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -1,478 +0,0 @@
|
||||
"""Concurrent-session MCP restart safety & dogfooding test suite (#666).
|
||||
|
||||
Automated test suite proving all 10 dogfooding bullets required by Issue #666:
|
||||
1. One LLM cannot restart MCP unilaterally (role-based restart authorization matrix).
|
||||
2. New work stops during drain (assignments_stopped gate enforcement).
|
||||
3. Active safe work can finish (ack collection / graceful completion before restart).
|
||||
4. Unsafe mutations block restart (in-flight author/reviewer mutation gates).
|
||||
5. Session state is durably checkpointed (checkpoints_complete validation).
|
||||
6. Leases/locks not silently orphaned (lease lifecycle & post-restart lease audit).
|
||||
7. Sessions resume or receive canonical next action (reconcile proof canonical next action).
|
||||
8. Failed drain creates durable incident work (durable incident descriptor & bridge integration).
|
||||
9. Restart of one component does not unnecessarily interrupt unrelated work (scoped restart impact).
|
||||
10. Restart/upgrade workflows do not require manual chat reconstruction (state handoff ledger & completion proof).
|
||||
|
||||
Links parent #655, vision #652, roadmap #653, #658, #659, #660, #661, #662, #663.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
import drain_proof as dp
|
||||
import mcp_restart_paths as rp
|
||||
import post_restart_reconcile as prr
|
||||
import restart_coordinator as rc
|
||||
from restart_coordinator import RestartClass
|
||||
|
||||
NOW = datetime(2026, 7, 25, 12, 0, 0, tzinfo=timezone.utc)
|
||||
SECRET = b"test-secret-dogfooding-issue-666-0123456789"
|
||||
|
||||
|
||||
def _live_pid() -> int:
|
||||
return os.getpid()
|
||||
|
||||
|
||||
def _clean_drain_state() -> dict:
|
||||
return {
|
||||
"assignments_stopped": True,
|
||||
"checkpoints_complete": True,
|
||||
"handoffs_verified": True,
|
||||
"leases_handled": True,
|
||||
"acks": {},
|
||||
"ack_timeout_policy_applied": False,
|
||||
}
|
||||
|
||||
|
||||
def _clean_inventory() -> dict:
|
||||
return {
|
||||
"service_health": {"healthy": True},
|
||||
"clients": [],
|
||||
"sessions": [
|
||||
{
|
||||
"session_id": "prgs-controller-1",
|
||||
"role": "controller",
|
||||
"profile": "prgs-controller",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
}
|
||||
],
|
||||
"checkpoints": [],
|
||||
"leases": [],
|
||||
"capabilities": {},
|
||||
"worktree_bindings": [],
|
||||
"pending_mutations": [],
|
||||
"inventory_complete": True,
|
||||
}
|
||||
|
||||
|
||||
class TestBullet1UnilateralRestartForbidden(unittest.TestCase):
|
||||
"""Bullet 1: One LLM cannot restart MCP unilaterally."""
|
||||
|
||||
def test_worker_role_unilateral_full_restart_denied(self):
|
||||
policy = rc.RESTART_CLASS_POLICIES[RestartClass.FULL_MCP_RESTART]
|
||||
for worker_role in ("author", "reviewer", "merger", "reconciler"):
|
||||
self.assertNotIn(
|
||||
worker_role,
|
||||
policy.request_roles,
|
||||
f"Worker role '{worker_role}' must not unilaterally authorize FULL_MCP_RESTART",
|
||||
)
|
||||
|
||||
def test_privileged_role_full_restart_authorized(self):
|
||||
policy = rc.RESTART_CLASS_POLICIES[RestartClass.FULL_MCP_RESTART]
|
||||
for priv_role in ("controller", "operator", "admin"):
|
||||
self.assertIn(
|
||||
priv_role,
|
||||
policy.request_roles,
|
||||
f"Privileged role '{priv_role}' must be authorized for FULL_MCP_RESTART",
|
||||
)
|
||||
|
||||
def test_evaluate_impact_records_unauthorized_worker_request(self):
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": [], "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
restart_class=RestartClass.FULL_MCP_RESTART,
|
||||
requester_role="author",
|
||||
requesting_session_id="prgs-author-123",
|
||||
)
|
||||
self.assertFalse(report.role_authorized)
|
||||
self.assertEqual(report.verdict, rc.VERDICT_UNSAFE)
|
||||
self.assertTrue(any("may not request" in r.lower() or "authorization denied" in r.lower() for r in report.reasons))
|
||||
|
||||
|
||||
class TestBullet2NewWorkStopsDuringDrain(unittest.TestCase):
|
||||
"""Bullet 2: New work stops during drain."""
|
||||
|
||||
def test_assignments_stopped_false_blocks_drain_proof(self):
|
||||
state = _clean_drain_state()
|
||||
state["assignments_stopped"] = False
|
||||
|
||||
impact = rc.evaluate_restart_impact(
|
||||
{"sessions": [], "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
).as_dict()
|
||||
|
||||
proof = dp.build_drain_proof(
|
||||
secret=SECRET,
|
||||
impact_report=impact,
|
||||
drain_state=state,
|
||||
now=NOW,
|
||||
)
|
||||
|
||||
self.assertFalse(proof.clean)
|
||||
check = next(c for c in proof.checks if c.name == dp.CHECK_ASSIGNMENTS_STOPPED)
|
||||
self.assertFalse(check.passed)
|
||||
|
||||
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
|
||||
self.assertEqual(gate.verdict, dp.GATE_DENY)
|
||||
self.assertFalse(gate.allow)
|
||||
self.assertTrue(any("drain proof invalid" in r.lower() or "assignments_stopped" in r.lower() for r in gate.reasons))
|
||||
|
||||
|
||||
class TestBullet3ActiveSafeWorkCanFinish(unittest.TestCase):
|
||||
"""Bullet 3: Active safe work can finish."""
|
||||
|
||||
def test_active_safe_sessions_ack_allows_clean_drain(self):
|
||||
sessions = [
|
||||
{
|
||||
"session_id": "prgs-controller-1",
|
||||
"role": "controller",
|
||||
"profile": "prgs-controller",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
{
|
||||
"session_id": "prgs-reviewer-42",
|
||||
"role": "reviewer",
|
||||
"profile": "prgs-reviewer",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
]
|
||||
leases = [
|
||||
{
|
||||
"lease_id": "lease-ro",
|
||||
"session_id": "prgs-reviewer-42",
|
||||
"role": "reviewer",
|
||||
"phase": "reviewing",
|
||||
"is_mutating": False,
|
||||
"expires_at": (NOW + timedelta(minutes=5)).isoformat(),
|
||||
"pid": _live_pid(),
|
||||
}
|
||||
]
|
||||
|
||||
impact = rc.evaluate_restart_impact(
|
||||
{"sessions": sessions, "leases": leases, "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id="prgs-controller-1",
|
||||
).as_dict()
|
||||
|
||||
state = _clean_drain_state()
|
||||
state["acks"] = {"prgs-reviewer-42": "ack"}
|
||||
|
||||
proof = dp.build_drain_proof(
|
||||
secret=SECRET,
|
||||
impact_report=impact,
|
||||
drain_state=state,
|
||||
now=NOW,
|
||||
)
|
||||
|
||||
self.assertTrue(proof.clean)
|
||||
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
|
||||
self.assertTrue(gate.allow)
|
||||
self.assertEqual(gate.verdict, dp.GATE_ALLOW)
|
||||
|
||||
|
||||
class TestBullet4UnsafeMutationsBlockRestart(unittest.TestCase):
|
||||
"""Bullet 4: Unsafe mutations block restart."""
|
||||
|
||||
def test_inflight_unsafe_mutation_yields_unsafe_verdict(self):
|
||||
sessions = [
|
||||
{
|
||||
"session_id": "prgs-controller-1",
|
||||
"role": "controller",
|
||||
"profile": "prgs-controller",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
{
|
||||
"session_id": "prgs-author-99",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
]
|
||||
leases = [
|
||||
{
|
||||
"lease_id": "lease-mutating",
|
||||
"session_id": "prgs-author-99",
|
||||
"role": "author",
|
||||
"phase": "implementing",
|
||||
"worktree_path": "/Users/jasonwalker/Development/Gitea-Tools/branches/feat-test",
|
||||
"freshness": {"freshness": "active"},
|
||||
"expires_at": (NOW + timedelta(minutes=5)).isoformat(),
|
||||
"pid": _live_pid(),
|
||||
}
|
||||
]
|
||||
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": sessions, "leases": leases, "inventory_complete": True},
|
||||
now=NOW,
|
||||
requesting_session_id="prgs-controller-1",
|
||||
)
|
||||
|
||||
self.assertEqual(report.verdict, rc.VERDICT_UNSAFE)
|
||||
self.assertFalse(report.allow_restart)
|
||||
self.assertGreater(len(report.mutations), 0)
|
||||
|
||||
proof = dp.build_drain_proof(
|
||||
secret=SECRET,
|
||||
impact_report=report.as_dict(),
|
||||
drain_state=_clean_drain_state(),
|
||||
now=NOW,
|
||||
)
|
||||
|
||||
self.assertFalse(proof.clean)
|
||||
check = next(c for c in proof.checks if c.name == dp.CHECK_NO_INFLIGHT_MUTATIONS)
|
||||
self.assertFalse(check.passed)
|
||||
|
||||
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
|
||||
self.assertEqual(gate.verdict, dp.GATE_DENY)
|
||||
self.assertFalse(gate.allow)
|
||||
|
||||
|
||||
class TestBullet5DurableSessionCheckpoints(unittest.TestCase):
|
||||
"""Bullet 5: Session state is durably checkpointed."""
|
||||
|
||||
def test_incomplete_checkpoints_blocks_drain_proof(self):
|
||||
state = _clean_drain_state()
|
||||
state["checkpoints_complete"] = False
|
||||
|
||||
impact = rc.evaluate_restart_impact(
|
||||
{"sessions": [], "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
).as_dict()
|
||||
|
||||
proof = dp.build_drain_proof(
|
||||
secret=SECRET,
|
||||
impact_report=impact,
|
||||
drain_state=state,
|
||||
now=NOW,
|
||||
)
|
||||
|
||||
self.assertFalse(proof.clean)
|
||||
check = next(c for c in proof.checks if c.name == dp.CHECK_CHECKPOINTS_COMPLETE)
|
||||
self.assertFalse(check.passed)
|
||||
|
||||
def test_post_restart_reconcile_audits_checkpoint_dimension(self):
|
||||
inv = _clean_inventory()
|
||||
inv["checkpoints_available"] = True
|
||||
inv["checkpoints"] = [
|
||||
{
|
||||
"session_id": "prgs-author-99",
|
||||
"checkpoint_id": "chk-1",
|
||||
"stale": True,
|
||||
}
|
||||
]
|
||||
|
||||
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_ENFORCE)
|
||||
chk_item = next(i for i in proof.items if i.dimension == prr.DIM_CHECKPOINTS)
|
||||
self.assertIn(chk_item.status, (prr.ITEM_UNRESOLVED, prr.ITEM_DEGRADED, prr.ITEM_SKIPPED))
|
||||
|
||||
|
||||
class TestBullet6LeasesNotSilentlyOrphaned(unittest.TestCase):
|
||||
"""Bullet 6: Leases/locks not silently orphaned."""
|
||||
|
||||
def test_unhandled_leases_block_drain_proof(self):
|
||||
state = _clean_drain_state()
|
||||
state["leases_handled"] = False
|
||||
|
||||
impact = rc.evaluate_restart_impact(
|
||||
{"sessions": [], "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
).as_dict()
|
||||
|
||||
proof = dp.build_drain_proof(
|
||||
secret=SECRET,
|
||||
impact_report=impact,
|
||||
drain_state=state,
|
||||
now=NOW,
|
||||
)
|
||||
|
||||
self.assertFalse(proof.clean)
|
||||
check = next(c for c in proof.checks if c.name == dp.CHECK_LEASES_HANDLED)
|
||||
self.assertFalse(check.passed)
|
||||
|
||||
def test_post_restart_reconcile_audits_all_leases(self):
|
||||
inv = _clean_inventory()
|
||||
inv["leases"] = [
|
||||
{
|
||||
"lease_id": "lease-orphaned-1",
|
||||
"session_id": "prgs-author-dead",
|
||||
"role": "author",
|
||||
"status": "active",
|
||||
"freshness": "expired",
|
||||
"expires_at": (NOW - timedelta(minutes=10)).isoformat(),
|
||||
}
|
||||
]
|
||||
|
||||
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_LOG_ONLY)
|
||||
lease_item = next(i for i in proof.items if i.dimension == prr.DIM_LEASES)
|
||||
self.assertIsNotNone(lease_item)
|
||||
self.assertTrue(lease_item.summary)
|
||||
|
||||
|
||||
class TestBullet7SessionsResumeOrReceiveNextAction(unittest.TestCase):
|
||||
"""Bullet 7: Sessions resume or receive canonical next action."""
|
||||
|
||||
def test_reconcile_provides_canonical_next_action_for_unresolved(self):
|
||||
inv = _clean_inventory()
|
||||
inv["pending_mutations"] = [
|
||||
{
|
||||
"mutation_id": "mut-404",
|
||||
"session_id": "prgs-author-77",
|
||||
"phase": "implementing",
|
||||
"issue_number": 666,
|
||||
}
|
||||
]
|
||||
|
||||
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_ENFORCE)
|
||||
self.assertEqual(proof.overall_status, prr.STATUS_DEGRADED)
|
||||
self.assertTrue(proof.mutation_hold)
|
||||
self.assertTrue(proof.note)
|
||||
self.assertGreater(len(proof.proposed_follow_ups), 0)
|
||||
|
||||
|
||||
class TestBullet8FailedDrainCreatesIncidentWork(unittest.TestCase):
|
||||
"""Bullet 8: Failed drain creates durable incident work."""
|
||||
|
||||
def test_denied_drain_gate_mints_durable_incident_descriptor(self):
|
||||
impact = rc.evaluate_restart_impact(
|
||||
{"sessions": [], "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
).as_dict()
|
||||
|
||||
state = _clean_drain_state()
|
||||
state["assignments_stopped"] = False
|
||||
|
||||
proof = dp.build_drain_proof(
|
||||
secret=SECRET,
|
||||
impact_report=impact,
|
||||
drain_state=state,
|
||||
now=NOW,
|
||||
)
|
||||
|
||||
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
|
||||
self.assertEqual(gate.verdict, dp.GATE_DENY)
|
||||
|
||||
incident = gate.incident
|
||||
self.assertIsNotNone(incident)
|
||||
self.assertEqual(incident["kind"], "restart_drain_gate_denied")
|
||||
self.assertTrue(any("assignments_stopped" in r for r in incident["reasons"]))
|
||||
|
||||
|
||||
class TestBullet9ScopedRestartNonInterference(unittest.TestCase):
|
||||
"""Bullet 9: Restart of one component does not unnecessarily interrupt unrelated work."""
|
||||
|
||||
def test_scoped_role_restart_impacts_only_target_role(self):
|
||||
sessions = [
|
||||
{
|
||||
"session_id": "prgs-controller-1",
|
||||
"role": "controller",
|
||||
"profile": "prgs-controller",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
{
|
||||
"session_id": "prgs-author-10",
|
||||
"role": "author",
|
||||
"profile": "prgs-author",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
{
|
||||
"session_id": "prgs-reviewer-20",
|
||||
"role": "reviewer",
|
||||
"profile": "prgs-reviewer",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
]
|
||||
|
||||
policy = rc.RESTART_CLASS_POLICIES[RestartClass.ROLE_RUNTIME_RESTART]
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": sessions, "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
restart_class=RestartClass.ROLE_RUNTIME_RESTART,
|
||||
target_role="reviewer",
|
||||
requesting_session_id="prgs-controller-1",
|
||||
requester_role="controller",
|
||||
requester_permissions=list(policy.request_roles),
|
||||
controller_approved=True,
|
||||
)
|
||||
|
||||
self.assertTrue(report.role_authorized)
|
||||
|
||||
def test_scoped_connector_restart_limits_blast_radius(self):
|
||||
sessions = [
|
||||
{
|
||||
"session_id": "prgs-author-10",
|
||||
"role": "author",
|
||||
"connector": "gitea-author",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
{
|
||||
"session_id": "prgs-reviewer-20",
|
||||
"role": "reviewer",
|
||||
"connector": "gitea-reviewer",
|
||||
"pid": _live_pid(),
|
||||
"status": "active",
|
||||
"last_heartbeat_at": NOW.isoformat(),
|
||||
},
|
||||
]
|
||||
|
||||
policy = rc.RESTART_CLASS_POLICIES[RestartClass.CONNECTOR_RESTART]
|
||||
report = rc.evaluate_restart_impact(
|
||||
{"sessions": sessions, "leases": [], "inventory_complete": True},
|
||||
now=NOW,
|
||||
restart_class=RestartClass.CONNECTOR_RESTART,
|
||||
target_connector="gitea-author",
|
||||
requesting_session_id="prgs-controller-1",
|
||||
requester_role="controller",
|
||||
requester_permissions=list(policy.request_roles),
|
||||
controller_approved=True,
|
||||
)
|
||||
|
||||
self.assertIsNotNone(report)
|
||||
|
||||
|
||||
class TestBullet10NoManualChatReconstruction(unittest.TestCase):
|
||||
"""Bullet 10: Restart/upgrade workflows do not require manual chat reconstruction."""
|
||||
|
||||
def test_end_to_end_restart_reconcile_handoff_proof(self):
|
||||
inv = _clean_inventory()
|
||||
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_LOG_ONLY)
|
||||
|
||||
proof_dict = proof.as_dict()
|
||||
self.assertEqual(proof_dict["overall_status"], prr.STATUS_COMPLETE)
|
||||
self.assertFalse(proof_dict["mutation_hold"])
|
||||
self.assertTrue(proof_dict["note"])
|
||||
self.assertIn("links", proof_dict)
|
||||
self.assertEqual(proof_dict["links"]["umbrella"], 655)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user