Compare commits

..
Author SHA1 Message Date
jcwalker3andGrok 4.5 29ad93d145 feat(mcp): expose sanctioned Codex MCP reconnect request (Closes #678)
Add gitea_request_mcp_reconnect as a report-only callable surface so Codex
and other agent hosts can request host/IDE reconnect with typed blockers
and exact UI steps. Never kills processes or edits config.

Co-Authored-By: Grok 4.5 <[email protected]>
2026-07-25 17:52:08 -05:00
sysadmin 2b4e43042a Merge pull request 'feat(tests): add concurrent-session MCP restart safety tests (Closes #666)' (#910) from feat/issue-666-concurrent-mcp-restart-tests into master 2026-07-25 17:44:09 -05:00
sysadmin 0f9390aab4 Merge remote-tracking branch 'prgs/master' into feat/issue-666-concurrent-mcp-restart-tests 2026-07-25 18:43:28 -04:00
sysadmin 59aab06fe1 feat(tests): add concurrent-session MCP restart safety tests (Closes #666) 2026-07-25 17:14:14 -04:00
14 changed files with 1311 additions and 1056 deletions
-41
View File
@@ -26,7 +26,6 @@ from dataclasses import dataclass, field
from datetime import datetime, timezone
from typing import Any, Mapping, Sequence
import maintenance_drain
from control_plane_db import (
ControlPlaneDB,
ControlPlaneError,
@@ -937,46 +936,6 @@ def allocate_next_work(
"allocation_mode": (allocation_mode or "").strip() or None,
}
# #659 AC2: while maintenance drain is active, no new work is assigned —
# for dry-run and apply alike, so a preview can never be read as evidence
# that work was assignable during the drain. Checked before session
# registration so a drained allocator leaves no new state behind.
try:
drain_record = db.read_maintenance_drain(remote=remote, org=org, repo=repo)
except Exception as exc: # noqa: BLE001 — unreadable drain state fails closed
return {
"success": False,
"outcome": OUTCOME_NO_SAFE,
"reasons": [
f"maintenance-drain state lookup failed: {exc} (fail closed, #659)"
],
"skipped": [],
"assignment": None,
"substrate": "control_plane_db",
}
drain_decision = maintenance_drain.classify_assignment(drain_record)
if not drain_decision["assignment_allowed"]:
return {
"success": True,
"outcome": OUTCOME_WAIT,
"apply": apply,
"role": role_norm,
"allocation_mode": mode,
"remote": remote,
"org": org,
"repo": repo,
"selected": None,
"reasons": list(drain_decision["reasons"]),
"reason_code": drain_decision["reason_code"],
"skipped": [],
"assignment": None,
"substrate": "control_plane_db",
"maintenance_drain": maintenance_drain.status_payload(
drain_record, remote=remote, org=org, repo=repo
),
}
# A side-effect-free run may never reserve: reserving is a write, and the
# flag is the caller's assertion that this call writes nothing (#643).
if side_effect_free and apply:
+1 -194
View File
@@ -31,9 +31,8 @@ from typing import Any, Iterator, Sequence
import dependency_graph
import gitea_audit
import maintenance_drain
SCHEMA_VERSION = 6
SCHEMA_VERSION = 5
# Assignable work kinds only — raw monitoring incidents are never work items.
WORK_KINDS = frozenset({"issue", "pr"})
@@ -240,31 +239,6 @@ CREATE INDEX IF NOT EXISTS idx_session_checkpoints_session
CREATE INDEX IF NOT EXISTS idx_session_checkpoints_work
ON session_checkpoints(remote, org, repo, work_kind, work_number);
-- Graceful maintenance-drain state (#659). One current row per repository
-- scope — drain is a *state*, not a history, so entering and exiting update
-- the same row and every transition is audited to ``events``. Creating the
-- table is the v5->v6 migration: additive, idempotent, and it never touches
-- prior tables. ``state`` is CHECK-constrained so an unknown value can never
-- be written and later read as "not draining".
CREATE TABLE IF NOT EXISTS maintenance_drain (
drain_id TEXT PRIMARY KEY,
remote TEXT NOT NULL,
org TEXT NOT NULL,
repo TEXT NOT NULL,
state TEXT NOT NULL DEFAULT 'inactive'
CHECK (state IN ('inactive', 'draining')),
reason TEXT NOT NULL DEFAULT '',
requested_by TEXT NOT NULL DEFAULT '',
requested_by_profile TEXT NOT NULL DEFAULT '',
session_id TEXT NOT NULL DEFAULT '',
entered_at TEXT NOT NULL DEFAULT '',
exited_at TEXT NOT NULL DEFAULT '',
drain_schema_version INTEGER NOT NULL DEFAULT 6,
created_at TEXT NOT NULL,
updated_at TEXT NOT NULL,
UNIQUE (remote, org, repo)
);
-- Model usage, token cost, latency, and performance events (#651)
CREATE TABLE IF NOT EXISTS usage_events (
usage_id INTEGER PRIMARY KEY AUTOINCREMENT,
@@ -3051,170 +3025,3 @@ class ControlPlaneDB:
"live_lease_id": None if live_lease_id is None else str(live_lease_id),
"reconcile_action": "reconcile_required" if stale else "safe_to_resume",
}
# ── Maintenance drain (#659) ─────────────────────────────────────────────
@staticmethod
def _maintenance_drain_row(row: sqlite3.Row | None) -> dict[str, Any] | None:
"""Convert a ``maintenance_drain`` row to a plain record."""
if row is None:
return None
return {key: row[key] for key in row.keys()}
def read_maintenance_drain(
self, *, remote: str, org: str, repo: str
) -> dict[str, Any] | None:
"""Return the current drain record for a scope, or None if never set.
None and a stored ``inactive`` row mean the same thing to callers —
``maintenance_drain.is_draining`` treats both as not draining — so the
read never has to invent a record to answer the gate.
"""
with self._tx(immediate=False) as conn:
row = conn.execute(
"""
SELECT * FROM maintenance_drain
WHERE remote = ? AND org = ? AND repo = ?
""",
(str(remote or ""), str(org or ""), str(repo or "")),
).fetchone()
return self._maintenance_drain_row(row)
def set_maintenance_drain(
self,
*,
remote: str,
org: str,
repo: str,
state: str,
reason: str = "",
requested_by: str = "",
requested_by_profile: str = "",
session_id: str = "",
) -> dict[str, Any]:
"""Enter or exit maintenance drain for one repository scope (AC1).
The state transition is audited to ``events`` — entering and exiting
are exactly the moments an operator has to be able to reconstruct
later. Re-entering an already-draining scope is idempotent: it refreshes
the reason/owner metadata, keeps the original ``entered_at``, and
records no duplicate transition event.
Capability authorization happens above this layer (the drain tasks
carry a non-``gitea.*`` permission in the task capability map); the DB
records who asked and why, and never grants the right itself.
"""
state_norm = maintenance_drain.normalize_state(state)
raw = {
"reason": str(reason or ""),
"requested_by": str(requested_by or ""),
"requested_by_profile": str(requested_by_profile or ""),
"session_id": str(session_id or ""),
}
clean = gitea_audit.redact(raw)
remote_s, org_s, repo_s = str(remote or ""), str(org or ""), str(repo or "")
now_s = _ts()
with self._tx() as conn:
existing = conn.execute(
"""
SELECT * FROM maintenance_drain
WHERE remote = ? AND org = ? AND repo = ?
""",
(remote_s, org_s, repo_s),
).fetchone()
prior_state = (
maintenance_drain.normalize_state(existing["state"])
if existing is not None
else maintenance_drain.STATE_INACTIVE
)
transitioned = prior_state != state_norm
prior_entered = (
str(existing["entered_at"] or "") if existing is not None else ""
)
prior_exited = (
str(existing["exited_at"] or "") if existing is not None else ""
)
if state_norm == maintenance_drain.STATE_DRAINING:
# A re-entry keeps the original entry time (the drain never
# stopped); a fresh entry stamps now and clears the old exit.
entered_at = prior_entered if (not transitioned and prior_entered) else now_s
exited_at = ""
else:
entered_at = prior_entered
exited_at = now_s if (transitioned or not prior_exited) else prior_exited
if existing is None:
drain_id = uuid.uuid4().hex
conn.execute(
"""
INSERT INTO maintenance_drain(
drain_id, remote, org, repo, state, reason,
requested_by, requested_by_profile, session_id,
entered_at, exited_at, drain_schema_version,
created_at, updated_at
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
""",
(
drain_id, remote_s, org_s, repo_s, state_norm,
clean["reason"], clean["requested_by"],
clean["requested_by_profile"], clean["session_id"],
entered_at, exited_at,
maintenance_drain.DRAIN_SCHEMA_VERSION, now_s, now_s,
),
)
else:
drain_id = str(existing["drain_id"])
conn.execute(
"""
UPDATE maintenance_drain
SET state = ?, reason = ?, requested_by = ?,
requested_by_profile = ?, session_id = ?,
entered_at = ?, exited_at = ?,
drain_schema_version = ?, updated_at = ?
WHERE drain_id = ?
""",
(
state_norm, clean["reason"], clean["requested_by"],
clean["requested_by_profile"], clean["session_id"],
entered_at, exited_at,
maintenance_drain.DRAIN_SCHEMA_VERSION, now_s, drain_id,
),
)
if transitioned:
event_type = (
"maintenance_drain_enter"
if state_norm == maintenance_drain.STATE_DRAINING
else "maintenance_drain_exit"
)
conn.execute(
"""
INSERT INTO events(work_item_id, event_type, message, created_at)
VALUES (NULL, ?, ?, ?)
""",
(
event_type,
f"drain {drain_id} scope {remote_s}/{org_s}/{repo_s} "
f"{prior_state} -> {state_norm} by "
f"{clean['requested_by'] or '(unknown)'} "
f"({clean['requested_by_profile'] or 'no profile'}); "
f"reason: {clean['reason'] or '(none)'}",
now_s,
),
)
row = conn.execute(
"SELECT * FROM maintenance_drain WHERE drain_id = ?", (drain_id,)
).fetchone()
record = self._maintenance_drain_row(row) or {}
return {
"record": record,
"drain_id": drain_id,
"state": state_norm,
"prior_state": prior_state,
"transitioned": transitioned,
}
-45
View File
@@ -1,45 +0,0 @@
# MCP maintenance-drain mode (#659)
Graceful **maintenance drain** stops new work assignment and defers non-allowlisted
mutations so sessions can finish critical handoffs and checkpoint before a
restart. It is **not** a restart authorization: the drain *proof* and apply gate
remain #661.
## State
Per repository scope (`remote`/`org`/`repo`) in the control-plane DB table
`maintenance_drain` (schema v6):
| State | Meaning |
|-------|---------|
| `inactive` | Normal operation (also: no row) |
| `draining` | Assignment stopped; non-allowlisted mutations deferred |
Enter/exit transitions are audited as `maintenance_drain_enter` /
`maintenance_drain_exit` events.
## Tools
| Tool | Permission | Effect |
|------|------------|--------|
| `gitea_maintenance_drain_status` | `gitea.read` | Observe drain (every session) |
| `gitea_enter_maintenance_drain` | `runtime.maintenance_drain` | Enter drain (capability-gated) |
| `gitea_exit_maintenance_drain` | `runtime.maintenance_drain` | Exit drain |
`runtime.maintenance_drain` is intentionally **not** a `gitea.*` op, so ordinary
author profiles cannot enter drain by accident.
## Enforcement
1. **Allocator** (`allocate_next_work`): while draining, returns `outcome=wait`
with `reason_code=maintenance_drain_assignment_stopped` for dry-run and apply.
2. **Mutation preflight** (`verify_preflight_purity`): non-allowlisted mutation
tasks raise `MaintenanceDrainError` with a typed next action.
3. **Allowlist** (safety only): heartbeats, lease release/abandon, session
checkpoints, enter/exit drain. Reads always work.
## Restart relationship
Drain mode prepares the blast radius. Restart apply still requires a clean
`DrainProof` (#661) or authorized break-glass. Status payloads never claim
restart permission.
+21 -6
View File
@@ -47,18 +47,33 @@ Do the steps in order. Stop as soon as a live **client-namespace** call succeeds
- Only the Gitea namespace fails → single-namespace transport close. Continue.
- Every server fails → restart the whole MCP client, not just one namespace.
2. **Reconnect the namespace through the client, not the shell.** Use the IDE /
client MCP-reconnect action for that server entry (in Claude Code:
`/mcp` → reconnect the affected `gitea-*` server). Reconnecting forces the
client to spawn a fresh subprocess and re-open the pipe. This clears the
closed-client state that a bare `kill`/respawn from a terminal does **not**.
2. **Request the sanctioned reconnect surface (#678), then reconnect through
the client — not the shell.** From a still-reachable Gitea MCP namespace
(or after host auto-reconnect), call:
```text
gitea_request_mcp_reconnect(
namespace="gitea-author", # or gitea-reviewer / gitea-merger / …
reason="transport_eof",
client="codex", # or claude_code / generic
)
```
The tool is **report-only**: it never restarts a process. It returns
namespace, profile, pid/session, startup SHA, current master SHA, boundary
status, and a **typed blocker** with exact operator UI steps for Codex
(Reload Developer Tools / per-server reconnect) or Claude Code (`/mcp`).
Then perform the host reconnect those steps describe so the client spawns a
fresh subprocess and re-opens the pipe. That clears the closed-client state
that a bare `kill`/respawn from a terminal does **not**.
3. **Do not "fix" it by importing the server or poking the process.** Reaching
for `python -c 'import gitea_mcp_server ...'`, raw JSON-RPC from a shell,
killing PIDs to force a respawn, or touching MCP config mtimes does **not**
restore the *client's* view of the namespace and violates the daemon-import
guard (#558, `docs/mcp-daemon-import-guard.md`). The only sanctioned repair
is a **client reconnect / relaunch**.
is a **client reconnect / relaunch** (or the typed operator path returned by
`gitea_request_mcp_reconnect`).
4. **Verify through the same path the workflow will use.** After reconnect, call
the specific tool the blocked workflow needs — not just any tool — through
+3 -2
View File
@@ -45,10 +45,11 @@ and *fails closed*.
| `legacy_auto_restart_helper` | removed | A helper (`_trigger_mcp_auto_restart`) that actively restarted the server from the read-only resolver path. | Removed in #685; kept absent by `assert_auto_restart_helper_absent()`. | #685, #657 |
| `config_touch_reload` | removed | Touching (utime) the MCP client config to make the host reload the server. | Removed from the resolver in #685: stale detection is report-only, never mutating config, spawning threads, or calling `os._exit`. | #685, #657 |
| `master_advance_auto_restart` | guarded_fail_closed | On-disk master advancing past the running code. | `master_parity_gate` captures startup parity and blocks mutations while stale, emitting restart guidance; the process never self-restarts. | #420, #591, #657 |
| `stale_runtime_resolver_reconnect` | guarded_fail_closed | The capability resolver detecting a stale serving process. | Report-only (#685): returns `restart_required`/`stop_required` and an exact reconnect action; no restart, thread, config touch, or `os._exit`. | #685, #657 |
| `stale_runtime_resolver_reconnect` | guarded_fail_closed | The capability resolver detecting a stale serving process. | Report-only (#685): returns `restart_required`/`stop_required` and an exact reconnect action; no restart, thread, config touch, or `os._exit`. | #685, #657, #678 |
| `codex_client_reconnect_request` | guarded_fail_closed | `gitea_request_mcp_reconnect` report-only tool for Codex/LLM sessions. | Report-only (#678): returns namespace/profile/pid/startup SHA/master SHA/boundary status and a typed operator blocker with exact client UI steps; never restarts or kills. | #678, #630, #685, #657 |
| `manual_daemon_kill` | forbidden | Shell kills of the daemon: `pkill -f mcp_server.py`, `killall`, broad `pkill -f python` sweeps, or `kill <pid>` of a daemon pid. | Forbidden (#630): `runtime_recovery_guard` classifies these as contamination and `gitea_record_daemon_process_kill_attempt` writes a durable marker that fails later mutations closed. Operator maintenance authorization is read only from the environment. | #630, #657 |
| `conflict_marker_infra_stop` | guarded_fail_closed | The daemon entrypoint scans for unresolved merge-conflict markers at startup and stops (`sys.exit(1)`). | Fail-closed startup stop, not a restart: the process exits and waits for the operator to resolve conflicts and relaunch; never loops. | #657 |
| `ide_client_reconnect` | host_residual | A manual `/mcp reconnect` (or equivalent host action) that recreates the MCP client connection. | Outside this process's control; the sanctioned recovery the gates point operators toward. No in-process code initiates it. | #584, #656, #657 |
| `ide_client_reconnect` | host_residual | A manual `/mcp reconnect` (or equivalent host action) that recreates the MCP client connection. Agents obtain exact UI steps via `gitea_request_mcp_reconnect` (#678). | Outside this process's control; the sanctioned recovery the gates point operators toward. No in-process code initiates it. | #584, #656, #657, #678 |
| `profile_switch_runtime` | sanctioned_narrow_recovery | Switching the active execution profile at runtime (dynamic-profile mode). | In-process and restart-free: `runtime_switching_supported` is true, so a switch rebinds capability without recreating the process. | #656, #657 |
## Guards enforced in CI
+1
View File
@@ -137,6 +137,7 @@ that gates each call, not which tools exist.
- `gitea_release_merger_pr_lease`
- `gitea_release_reviewer_pr_lease`
- `gitea_release_workflow_lease`
- `gitea_request_mcp_reconnect`
- `gitea_request_mcp_restart`
- `gitea_resolve_task_capability`
- `gitea_resume_review_draft`
+121 -227
View File
@@ -1472,9 +1472,6 @@ def verify_preflight_purity(
# contaminated by manual MCP daemon process killing (reconciler-exempt).
_enforce_runtime_recovery_contamination_gate(task, remote)
# #659 AC3: defer non-allowlisted mutations while maintenance drain is active.
_enforce_maintenance_drain_gate(task, remote=remote, org=org, repo=repo)
ctx = _resolve_namespace_mutation_context(worktree_path)
workspace = ctx["workspace_path"]
canonical_root = ctx["canonical_repo_root"]
@@ -2007,55 +2004,6 @@ def _enforce_runtime_recovery_contamination_gate(
)
def _enforce_maintenance_drain_gate(
task: str | None,
remote: str | None = None,
org: str | None = None,
repo: str | None = None,
) -> None:
"""#659 AC3: defer non-allowlisted mutations while drain is active.
The single mutation chokepoint already used by every gated task, so drain
coverage cannot drift per-tool. Allowlisted safety operations (heartbeat,
release/abandon, checkpoint, drain exit) pass through so an in-flight
session can still finish and hand off; everything else is deferred with a
typed blocker. Unreadable drain state fails closed a drain that cannot be
read is not evidence that no drain is running.
"""
if _preflight_in_test_mode() and not os.environ.get(
"GITEA_TEST_FORCE_MAINTENANCE_DRAIN"
):
return
if maintenance_drain.is_allowlisted_task(task):
return
try:
_h, o, r = _resolve(remote, None, org, repo)
except Exception: # noqa: BLE001 — scope resolution is best-effort here
o, r = (org or ""), (repo or "")
db, errs = _control_plane_db_or_error()
if db is None:
raise RuntimeError(
"maintenance-drain state could not be read: "
f"{'; '.join(errs) or 'control-plane DB unavailable'} (fail closed, #659)"
)
try:
record = db.read_maintenance_drain(remote=remote or "", org=o, repo=r)
except Exception as exc: # noqa: BLE001
raise RuntimeError(
f"maintenance-drain state could not be read: {_redact(str(exc))} "
"(fail closed, #659)"
) from exc
decision = maintenance_drain.classify_mutation(task, record)
if not decision["allowed"]:
raise maintenance_drain.MaintenanceDrainError(
maintenance_drain.format_drain_block_error(decision),
decision=decision,
)
def _enforce_stable_branch_contamination_gate(
task: str | None,
remote: str | None = None,
@@ -2116,7 +2064,6 @@ import allocator_service # noqa: E402
import allocator_dependencies # noqa: E402
import dependency_graph # noqa: E402 # #784 durable dependency edges
import control_plane_db # noqa: E402
import maintenance_drain # noqa: E402 # #659 graceful maintenance-drain mode
import lease_lifecycle # noqa: E402
import lease_policy # noqa: E402
import workflow_dashboard # noqa: E402 # #605 live queue/lease dashboard
@@ -2145,6 +2092,7 @@ import root_checkout_guard # noqa: E402
import workflow_scope_guard # noqa: E402 # #683 production scope / force-on guards
import stable_branch_push_guard # noqa: E402
import runtime_recovery_guard # noqa: E402 # #630 manual daemon-kill contamination
import mcp_client_reconnect # noqa: E402 # #678 sanctioned Codex reconnect request
import remote_repo_guard # noqa: E402
import anti_stomp_preflight # noqa: E402
import issue_claim_heartbeat # noqa: E402
@@ -14077,11 +14025,16 @@ def _classify_operation_gate_reasons(reasons: list[str]) -> dict:
def _stale_runtime_reconnect_action() -> str:
"""Sanctioned recovery for a stale daemon — reconnect only (#685/#897)."""
"""Sanctioned recovery for a stale daemon — reconnect only (#685/#897/#678)."""
return (
"Reconnect the IDE/client MCP session so the server reloads at the "
"current master head. Do not call gitea_activate_profile or switch "
"MCP role sessions — profile switching does not clear a stale daemon."
"blocker_kind=runtime_reconnect_required: call "
"gitea_request_mcp_reconnect(namespace=<active gitea-* namespace>, "
"reason='stale-runtime', client='codex') for a typed operator "
"reconnect blocker with exact UI steps, then reconnect the IDE/client "
"MCP session so the server reloads at the current master head. Do not "
"call gitea_activate_profile, pkill, touch configs, or switch MCP role "
"sessions — profile switching does not clear a stale daemon. After "
"reconnect restart from gitea_whoami → gitea_resolve_task_capability."
)
@@ -21051,10 +21004,15 @@ def gitea_resolve_task_capability(
# serving process/profile inventory is stale — even if permission is OK.
if runtime_stale_blocker:
next_safe_action = (
"blocker_kind=runtime_reconnect_required: reconnect/restart the "
"IDE-managed Gitea MCP server for this profile so it reloads current "
"master. Do not edit mcp_config.json by hand; the resolver does not "
"touch config, spawn recovery threads, or terminate the process."
"blocker_kind=runtime_reconnect_required: call "
"gitea_request_mcp_reconnect(namespace=<active gitea-* namespace>, "
"reason='stale-runtime', client='codex') for a typed operator "
"blocker with exact UI steps, then reconnect/reload the IDE-managed "
"Gitea MCP server for this profile so it reloads current master. "
"Do not edit mcp_config.json by hand, pkill, or touch configs; the "
"resolver does not touch config, spawn recovery threads, or "
"terminate the process. After reconnect restart from gitea_whoami → "
"gitea_resolve_task_capability."
)
# Task/role alignment guards (#167): the requested task, not the
@@ -22613,190 +22571,126 @@ def gitea_workflow_dashboard(
@mcp.tool()
def gitea_maintenance_drain_status(
def gitea_request_mcp_reconnect(
namespace: str | None = None,
reason: str | None = None,
client: str = "codex",
remote: str = "dadeschools",
host: str | None = None,
org: str | None = None,
repo: str | None = None,
session_id: str | None = None,
) -> dict:
"""Read-only: current maintenance-drain state for a repository scope (#659 AC4).
"""Request a sanctioned host/IDE MCP reconnect for a named namespace (#678).
Every session must be able to observe drain so it can stop creating new work
and finish only allowlisted safety operations. Never mutates; never restarts.
Codex and other agent hosts can detect stale or closed Gitea MCP runtimes
(``stop_required`` / ``restart_required`` from capability resolution, transport
EOF, missing namespace attachment). The **host owns the transport** this
process cannot reopen the client's stdio pipe. This tool is the callable
surface agents use to:
1. Report reconnect status fields (namespace, profile, pid/session,
startup SHA, current master SHA, boundary status).
2. Return a **typed blocker** with exact operator UI steps for Codex (or
another client) when reconnect is required.
This tool **never** restarts, kills, reloads, or reconfigures an MCP
process. Forbidden recovery paths (pkill, touch/mtime hacks, config/.env
edits, session-state edits, raw API) are never recommended.
After the operator reconnects, workflows must restart from preflight:
``gitea_whoami`` ``gitea_resolve_task_capability`` task.
Args:
namespace: MCP namespace to reconnect (e.g. ``gitea-author``). Defaults
to the active profile's inferred namespace.
reason: Why reconnect is requested: ``stale-runtime``, ``transport_eof``,
``missing_namespace``, ``not_required``, or free-form (normalized).
client: Operator UI surface ``codex`` (default), ``claude_code``, or
``generic``.
remote: Known instance ``dadeschools`` or ``prgs`` (parity context).
host: Optional host override for parity context.
session_id: Optional session id to echo in the report.
Returns:
dict with reconnect report fields, ``reconnect_performed=False``,
``typed_blocker`` when reconnect is required, and
``forbidden_recovery_paths``.
"""
# Read-only: gitea.read is sufficient. Never a mutation.
read_block = _profile_operation_gate("gitea.read")
if read_block:
return {
"success": False,
"read_only": True,
"reconnect_performed": False,
"mutation_performed": False,
"reasons": read_block,
"permission_report": _permission_block_report("gitea.read"),
}
try:
_h, o, r = _resolve(remote, host, org, repo)
except ValueError as exc:
return {"success": False, "read_only": True, "reasons": [str(exc)]}
db, errs = _control_plane_db_or_error()
if db is None:
return {
"success": False,
"read_only": True,
"reasons": errs or ["control-plane DB unavailable"],
"maintenance_drain": maintenance_drain.status_payload(
None, remote=remote, org=o, repo=r
"forbidden_recovery_paths": list(
mcp_client_reconnect.FORBIDDEN_RECOVERY_PATHS
),
}
try:
record = db.read_maintenance_drain(remote=remote, org=o, repo=r)
except Exception as exc: # noqa: BLE001
return {
"success": False,
"read_only": True,
"reasons": [f"drain state unreadable: {_redact(str(exc))}"],
}
payload = maintenance_drain.status_payload(
record, remote=remote, org=o, repo=r
profile = get_profile()
profile_name = (profile.get("profile_name") or "").strip() or None
inferred_ns = role_namespace_gate.infer_mcp_namespace(profile_name)
ns = (namespace or "").strip() or inferred_ns or "gitea-tools"
parity = _current_master_parity()
startup_sha = (
parity.get("daemon_start_head")
or parity.get("startup_head")
or _process_boot_head_sha
)
return {"success": True, "read_only": True, "maintenance_drain": payload}
current_sha = parity.get("local_head") or parity.get("current_head")
if not current_sha:
try:
current_sha = master_parity_gate.read_git_head(PROJECT_ROOT)
except Exception: # noqa: BLE001
current_sha = None
boundary = mcp_client_reconnect.classify_boundary_status(
startup_sha=startup_sha if isinstance(startup_sha, str) else None,
current_master_sha=current_sha if isinstance(current_sha, str) else None,
live_stale=bool(parity.get("live_stale")) if parity.get("live_known") else None,
in_parity=parity.get("in_parity") if parity.get("determinable") else None,
)
@mcp.tool()
def gitea_enter_maintenance_drain(
reason: str = "",
remote: str = "dadeschools",
host: str | None = None,
org: str | None = None,
repo: str | None = None,
session_id: str | None = None,
) -> dict:
"""Enter graceful maintenance-drain mode for a repository scope (#659 AC1).
# Infer reason from parity when caller left it unspecified.
effective_reason = reason
if not (effective_reason or "").strip():
if parity.get("restart_required") or parity.get("live_stale"):
effective_reason = mcp_client_reconnect.REASON_STALE_RUNTIME
elif boundary == mcp_client_reconnect.BOUNDARY_CLEAN:
effective_reason = mcp_client_reconnect.REASON_NOT_REQUIRED
else:
effective_reason = mcp_client_reconnect.REASON_UNSPECIFIED
Stops new assignment and defers non-allowlisted mutations until exit. Requires
``runtime.maintenance_drain`` (controller/lifecycle capability not granted
by ordinary Gitea author profiles). Audited in the control-plane event log.
"""
cap_block = _profile_operation_gate("runtime.maintenance_drain")
if cap_block:
return {
"success": False,
"performed": False,
"reasons": cap_block,
"permission_report": _permission_block_report(
"runtime.maintenance_drain"
),
}
try:
_h, o, r = _resolve(remote, host, org, repo)
except ValueError as exc:
return {"success": False, "performed": False, "reasons": [str(exc)]}
db, errs = _control_plane_db_or_error()
if db is None:
return {
"success": False,
"performed": False,
"reasons": errs or ["control-plane DB unavailable"],
}
profile = get_profile() or {}
try:
result = db.set_maintenance_drain(
remote=remote,
org=o,
repo=r,
state=maintenance_drain.STATE_DRAINING,
reason=reason or "operator-entered maintenance drain",
requested_by=str(
(profile.get("identity") or {}).get("username")
or profile.get("expected_username")
or ""
),
requested_by_profile=str(profile.get("profile_name") or ""),
session_id=str(session_id or ""),
)
except Exception as exc: # noqa: BLE001
return {
"success": False,
"performed": False,
"reasons": [f"enter drain failed: {_redact(str(exc))}"],
}
record = result.get("record") or {}
return {
"success": True,
"performed": True,
"transitioned": bool(result.get("transitioned")),
"state": result.get("state"),
"prior_state": result.get("prior_state"),
"drain_id": result.get("drain_id"),
"maintenance_drain": maintenance_drain.status_payload(
record, remote=remote, org=o, repo=r
),
}
@mcp.tool()
def gitea_exit_maintenance_drain(
reason: str = "",
remote: str = "dadeschools",
host: str | None = None,
org: str | None = None,
repo: str | None = None,
session_id: str | None = None,
) -> dict:
"""Exit graceful maintenance-drain mode (#659 AC1). Restores assignment and mutations."""
cap_block = _profile_operation_gate("runtime.maintenance_drain")
if cap_block:
return {
"success": False,
"performed": False,
"reasons": cap_block,
"permission_report": _permission_block_report(
"runtime.maintenance_drain"
),
}
try:
_h, o, r = _resolve(remote, host, org, repo)
except ValueError as exc:
return {"success": False, "performed": False, "reasons": [str(exc)]}
db, errs = _control_plane_db_or_error()
if db is None:
return {
"success": False,
"performed": False,
"reasons": errs or ["control-plane DB unavailable"],
}
profile = get_profile() or {}
try:
result = db.set_maintenance_drain(
remote=remote,
org=o,
repo=r,
state=maintenance_drain.STATE_INACTIVE,
reason=reason or "operator-exited maintenance drain",
requested_by=str(
(profile.get("identity") or {}).get("username")
or profile.get("expected_username")
or ""
),
requested_by_profile=str(profile.get("profile_name") or ""),
session_id=str(session_id or ""),
)
except Exception as exc: # noqa: BLE001
return {
"success": False,
"performed": False,
"reasons": [f"exit drain failed: {_redact(str(exc))}"],
}
record = result.get("record") or {}
return {
"success": True,
"performed": True,
"transitioned": bool(result.get("transitioned")),
"state": result.get("state"),
"prior_state": result.get("prior_state"),
"drain_id": result.get("drain_id"),
"maintenance_drain": maintenance_drain.status_payload(
record, remote=remote, org=o, repo=r
),
}
payload = mcp_client_reconnect.build_reconnect_request(
namespace=ns,
profile=profile_name,
pid=os.getpid(),
session_id=session_id
or f"{(profile_name or 'session')}-{os.getpid()}",
startup_sha=startup_sha if isinstance(startup_sha, str) else None,
current_master_sha=current_sha if isinstance(current_sha, str) else None,
boundary_status=boundary,
reason=effective_reason,
client=client,
live_stale=bool(parity.get("live_stale")) if parity.get("live_known") else None,
in_parity=parity.get("in_parity") if parity.get("determinable") else None,
restart_required=bool(parity.get("restart_required")),
stop_required=bool(parity.get("restart_required")),
extra={
"remote": remote if remote in REMOTES else remote,
"host": host,
"session_context_audit": session_ctx.mutation_context_audit_fields(),
"parity_summary": master_parity_gate.format_parity(parity),
"live_stale": parity.get("live_stale"),
"live_known": parity.get("live_known"),
"in_parity": parity.get("in_parity"),
},
)
return payload
@mcp.tool()
-281
View File
@@ -1,281 +0,0 @@
"""Graceful MCP maintenance-drain mode (#659).
Drain is the visible, capability-gated state that lets an operator stop new
work and quiesce mutations *before* a restart, instead of cutting sessions off
mid-mutation. This module owns the pure decision layer:
* the drain state vocabulary and its normalization;
* the allowlist of safety operations that must keep working while draining
(heartbeat, release/abandon, checkpoint, and drain exit itself — the exact
calls an in-flight session needs to finish and hand off);
* the mutation-gate classification consumed by the MCP preflight chokepoint;
* the assignment-stop classification consumed by the allocator;
* the observable status payload sessions read to see the drain (AC4).
Durable state lives in the control-plane DB (``maintenance_drain`` table);
enforcement lives at the existing chokepoints. Nothing here performs I/O, so
both callers can share one decision without importing each other.
Scope note: the machine-verifiable *drain proof* and the restart gate that
consumes it are #661's scope, not this module's. Drain here stops assignment
and mutation and makes the state observable; it never authorizes a restart.
"""
from __future__ import annotations
from typing import Any, Mapping
# ── State vocabulary ──────────────────────────────────────────────────────────
STATE_INACTIVE = "inactive"
STATE_DRAINING = "draining"
DRAIN_STATES = frozenset({STATE_INACTIVE, STATE_DRAINING})
# Typed blocker code surfaced to clients (never a bare string at call sites).
BLOCKER_DRAIN_ACTIVE = "maintenance_drain_active"
# Reason code for the allocator's assignment stop.
REASON_ASSIGNMENT_STOPPED = "maintenance_drain_assignment_stopped"
DRAIN_SCHEMA_VERSION = 6
class MaintenanceDrainError(RuntimeError):
"""Raised when a mutation is refused because drain is active (fail closed)."""
def __init__(self, message: str, *, decision: Mapping[str, Any] | None = None):
super().__init__(message)
self.decision = dict(decision or {})
self.reason_code = BLOCKER_DRAIN_ACTIVE
# ── Safety allowlist ──────────────────────────────────────────────────────────
# Mutations that stay permitted while draining. Every entry is a *quiesce*
# operation: it either proves an in-flight task is still alive, hands its claim
# back, records the durable state a restart needs, or ends the drain. Nothing
# that creates new work, new branches, new PRs, or new review/merge verdicts is
# on this list — that is the whole point of the drain.
ALLOWLISTED_DRAIN_TASKS: frozenset[str] = frozenset(
{
# Liveness of work already in flight.
"heartbeat_issue_lock",
"heartbeat_reviewer_pr_lease",
"post_heartbeat",
# Handing claims back so nothing is stranded across the restart.
"release_workflow_lease",
"release_reviewer_pr_lease",
"release_merger_pr_lease",
"abandon_workflow_lease",
# Durable recovery state (#660) must be writable *during* drain.
"write_session_checkpoint",
"checkpoint_session",
# The drain controls themselves — exit must never be self-blocked.
"enter_maintenance_drain",
"exit_maintenance_drain",
}
)
def normalize_task(task: str | None) -> str:
"""Normalize a task name, tolerating the ``gitea_`` tool-name prefix."""
name = str(task or "").strip()
if name.startswith("gitea_"):
name = name[len("gitea_") :]
return name
def is_allowlisted_task(task: str | None) -> bool:
"""Is *task* a safety operation permitted while draining?"""
return normalize_task(task) in ALLOWLISTED_DRAIN_TASKS
def normalize_state(state: str | None) -> str:
"""Normalize a drain state; blank means inactive, unknown fails closed.
Blank normalizes to ``inactive`` (no drain record = not draining), but an
unrecognized non-blank value raises: silently treating ``"drainig"`` as
inactive would disable the gate.
"""
value = str(state or "").strip().lower()
if not value:
return STATE_INACTIVE
if value not in DRAIN_STATES:
raise MaintenanceDrainError(
f"unknown maintenance-drain state {value!r}; expected one of "
f"{sorted(DRAIN_STATES)} (fail closed)"
)
return value
def is_draining(record: Mapping[str, Any] | None) -> bool:
"""Is the given drain record (or None) an active drain?"""
if not record:
return False
return normalize_state(record.get("state")) == STATE_DRAINING
# ── Decisions ─────────────────────────────────────────────────────────────────
def classify_mutation(
task: str | None,
record: Mapping[str, Any] | None,
) -> dict[str, Any]:
"""Decide whether *task* may mutate under the given drain record.
Returns a decision dict with ``allowed``/``deferred`` and, when refused, a
typed ``reason_code`` plus the one exact next action the caller may take.
Deferred (not failed): the operation is legal again after drain exits, so
the caller is told to wait rather than to retry a different way.
"""
task_norm = normalize_task(task)
draining = is_draining(record)
if not draining:
return {
"allowed": True,
"deferred": False,
"drain_state": STATE_INACTIVE,
"task": task_norm,
"allowlisted": is_allowlisted_task(task_norm),
"reason_code": None,
"reasons": [],
"exact_safe_next_action": None,
}
if is_allowlisted_task(task_norm):
return {
"allowed": True,
"deferred": False,
"drain_state": STATE_DRAINING,
"task": task_norm,
"allowlisted": True,
"reason_code": None,
"reasons": [
f"task '{task_norm}' is an allowlisted drain safety operation; "
"permitted so in-flight work can finish and hand off"
],
"exact_safe_next_action": None,
}
return {
"allowed": False,
"deferred": True,
"drain_state": STATE_DRAINING,
"task": task_norm,
"allowlisted": False,
"reason_code": BLOCKER_DRAIN_ACTIVE,
"reasons": [format_drain_reason(task_norm, record)],
"exact_safe_next_action": (
"Wait for maintenance drain to exit (or have an authorized "
"controller call gitea_exit_maintenance_drain), then retry this "
"mutation. Reads and gitea_maintenance_drain_status stay available."
),
}
def classify_assignment(record: Mapping[str, Any] | None) -> dict[str, Any]:
"""Decide whether the allocator may assign new work (AC2)."""
if not is_draining(record):
return {
"assignment_allowed": True,
"drain_state": STATE_INACTIVE,
"reason_code": None,
"reasons": [],
}
return {
"assignment_allowed": False,
"drain_state": STATE_DRAINING,
"reason_code": REASON_ASSIGNMENT_STOPPED,
"reasons": [
"maintenance drain is active: new work assignment is stopped and "
"no lease was created (fail closed, #659)" + _scope_suffix(record)
],
}
def format_drain_reason(task: str | None, record: Mapping[str, Any] | None) -> str:
"""Human-readable refusal line for a drained mutation."""
task_norm = normalize_task(task) or "(unnamed task)"
return (
f"maintenance drain is active: mutation '{task_norm}' is deferred; only "
"allowlisted drain safety operations "
f"({', '.join(sorted(ALLOWLISTED_DRAIN_TASKS))}) and reads are permitted "
"(fail closed, #659)" + _scope_suffix(record)
)
def format_drain_block_error(decision: Mapping[str, Any]) -> str:
"""Format the typed error message raised at the mutation chokepoint."""
reasons = list(decision.get("reasons") or [])
head = reasons[0] if reasons else "maintenance drain is active (fail closed)"
action = decision.get("exact_safe_next_action")
return f"{head}. Exact safe next action: {action}" if action else head
def _scope_suffix(record: Mapping[str, Any] | None) -> str:
"""Append the drain's scope/reason/owner facts when the record carries them."""
if not record:
return ""
bits: list[str] = []
scope = "/".join(
str(record.get(key) or "") for key in ("remote", "org", "repo")
).strip("/")
if scope:
bits.append(f"scope {scope}")
if record.get("reason"):
bits.append(f"reason: {record['reason']}")
if record.get("requested_by"):
bits.append(f"entered by {record['requested_by']}")
if record.get("entered_at"):
bits.append(f"at {record['entered_at']}")
return f" ({'; '.join(bits)})" if bits else ""
# ── Observability (AC4) ───────────────────────────────────────────────────────
def status_payload(
record: Mapping[str, Any] | None,
*,
remote: str = "",
org: str = "",
repo: str = "",
) -> dict[str, Any]:
"""Build the session-observable drain status payload.
Always answers, including when no drain record exists: an absent record is
a definitive "not draining", not an unknown.
"""
draining = is_draining(record)
rec: Mapping[str, Any] = record or {}
return {
"drain_state": STATE_DRAINING if draining else STATE_INACTIVE,
"draining": draining,
"remote": str(rec.get("remote") or "") or remote,
"org": str(rec.get("org") or "") or org,
"repo": str(rec.get("repo") or "") or repo,
"reason": str(rec.get("reason") or ""),
"requested_by": str(rec.get("requested_by") or ""),
"requested_by_profile": str(rec.get("requested_by_profile") or ""),
"session_id": str(rec.get("session_id") or ""),
"entered_at": str(rec.get("entered_at") or ""),
"exited_at": str(rec.get("exited_at") or ""),
"assignment_stopped": draining,
"mutations_deferred": draining,
"allowlisted_tasks": sorted(ALLOWLISTED_DRAIN_TASKS),
"reads_permitted": True,
"record_present": bool(record),
"schema_version": DRAIN_SCHEMA_VERSION,
"drain_proof_scope": (
"drain proof and the restart gate that consumes it are #661 scope; "
"this status never authorizes a restart"
),
"safe_next_action": (
"Wait for drain to exit before retrying deferred mutations; "
"allowlisted safety operations and reads remain available."
if draining
else "None; maintenance drain is not active."
),
}
+328
View File
@@ -0,0 +1,328 @@
"""Sanctioned MCP client reconnect request surface for Codex/LLM sessions (#678).
Codex and other agent hosts can detect stale or closed Gitea MCP runtimes, but
the host owns the transport. This module never restarts, kills, or reloads a
daemon. It builds:
1. A **callable reconnect request** result agents can invoke via
``gitea_request_mcp_reconnect`` (report-only, side-effect free).
2. A **typed blocker** with exact operator UI steps when recovery must be
performed by the host/operator.
Forbidden recovery paths (must never be recommended):
* ``pkill`` / ``kill`` / ``killall`` of MCP daemons
* ``touch`` / mtime config reload hacks
* ``.env`` or MCP config edits as recovery
* session-state file edits
* raw Gitea API / direct server-import fallbacks
After the operator reconnects, workflows restart from identity / runtime /
capability preflight (``gitea_whoami`` → ``gitea_resolve_task_capability`` →
task).
"""
from __future__ import annotations
from typing import Any, Mapping
# --- Reason vocabulary -------------------------------------------------------
REASON_STALE_RUNTIME = "stale-runtime"
REASON_TRANSPORT_EOF = "transport_eof"
REASON_MISSING_NAMESPACE = "missing_namespace"
REASON_NOT_REQUIRED = "not_required"
REASON_UNSPECIFIED = "unspecified"
VALID_REASONS = frozenset(
{
REASON_STALE_RUNTIME,
REASON_TRANSPORT_EOF,
REASON_MISSING_NAMESPACE,
REASON_NOT_REQUIRED,
REASON_UNSPECIFIED,
}
)
# Boundary statuses reported to callers (match review_workflow_boundary style).
BOUNDARY_CLEAN = "clean"
BOUNDARY_MISMATCH = "mismatch"
BOUNDARY_STALE = "stale"
BOUNDARY_UNKNOWN = "unknown"
# Typed blocker kinds
BLOCKER_OPERATOR_RECONNECT = "operator_mcp_reconnect_required"
BLOCKER_NONE = "none"
FORBIDDEN_RECOVERY_PATHS: tuple[str, ...] = (
"pkill / kill / killall of mcp_server.py, gitea_mcp_server, or broad python sweeps",
"touch / mtime-based MCP config reload hacks",
".env edits as recovery",
"MCP config file edits as recovery",
"session-state file edits as recovery",
"raw Gitea API or direct MCP server-import fallbacks",
)
# Client-specific operator UI steps. Keep Codex first (issue title surface).
OPERATOR_UI_STEPS: dict[str, tuple[str, ...]] = {
"codex": (
"In Codex, open the MCP / Developer tools panel for this workspace.",
"Locate the named Gitea MCP server entry (namespace) that needs reconnect "
"(e.g. gitea-author, gitea-reviewer, gitea-merger, gitea-tools, "
"gitea-controller, gitea-reconciler).",
"Click 'Reload Developer Tools' or the server reconnect/reload control "
"for that entry so the client spawns a fresh MCP subprocess.",
"If per-server reconnect is unavailable, fully restart the Codex client "
"(quit and relaunch) so all MCP namespaces reattach.",
"After reconnect, rerun the blocked workflow from preflight: "
"gitea_whoami → gitea_resolve_task_capability → the original task. "
"Do not resume mid-mutation.",
),
"claude_code": (
"Run `/mcp` (or open the MCP servers UI) in Claude Code.",
"Reconnect the affected gitea-* server entry so the client reopens stdio.",
"If reconnect fails, relaunch the Claude Code session entirely.",
"After reconnect, restart the workflow from gitea_whoami → "
"gitea_resolve_task_capability → task.",
),
"generic": (
"Use the host/IDE MCP reconnect or reload control for the named namespace.",
"If no per-namespace control exists, restart the MCP client/editor.",
"After reconnect, restart the workflow from identity/capability preflight.",
),
}
DEFAULT_CLIENT = "codex"
def normalize_reason(reason: str | None) -> str:
"""Map free-form reason strings onto the closed vocabulary."""
raw = (reason or "").strip().lower()
if not raw:
return REASON_UNSPECIFIED
if raw in VALID_REASONS:
return raw
text = raw.replace(" ", "_").replace("-", "_")
aliases = {
"stale_runtime": REASON_STALE_RUNTIME,
"staleruntime": REASON_STALE_RUNTIME,
"runtime_stale": REASON_STALE_RUNTIME,
"stale": REASON_STALE_RUNTIME,
"transport_eof": REASON_TRANSPORT_EOF,
"transport_closed": REASON_TRANSPORT_EOF,
"eof": REASON_TRANSPORT_EOF,
"client_is_closing": REASON_TRANSPORT_EOF,
"missing_namespace": REASON_MISSING_NAMESPACE,
"namespace_missing": REASON_MISSING_NAMESPACE,
"not_required": REASON_NOT_REQUIRED,
"healthy": REASON_NOT_REQUIRED,
"ok": REASON_NOT_REQUIRED,
"unspecified": REASON_UNSPECIFIED,
}
if text in aliases:
return aliases[text]
hyphenated = text.replace("_", "-")
if hyphenated in VALID_REASONS:
return hyphenated
return REASON_UNSPECIFIED
def normalize_client(client: str | None) -> str:
"""Return a known client key for operator UI steps."""
text = (client or "").strip().lower().replace(" ", "_").replace("-", "_")
if text in ("codex", "openai_codex", "openai"):
return "codex"
if text in ("claude", "claude_code", "claude_desktop", "anthropic"):
return "claude_code"
if text in OPERATOR_UI_STEPS:
return text
return DEFAULT_CLIENT
def classify_boundary_status(
*,
startup_sha: str | None,
current_master_sha: str | None,
live_stale: bool | None = None,
in_parity: bool | None = None,
) -> str:
"""Derive boundary_status from parity evidence."""
if live_stale is True or in_parity is False:
return BOUNDARY_STALE
start = (startup_sha or "").strip().lower()
current = (current_master_sha or "").strip().lower()
if start and current and start != current:
return BOUNDARY_MISMATCH
if start and current and start == current:
return BOUNDARY_CLEAN
if in_parity is True:
return BOUNDARY_CLEAN
return BOUNDARY_UNKNOWN
def operator_ui_steps(client: str | None, *, namespace: str | None = None) -> list[str]:
"""Exact operator UI steps for the named client."""
key = normalize_client(client)
steps = list(OPERATOR_UI_STEPS.get(key) or OPERATOR_UI_STEPS[DEFAULT_CLIENT])
ns = (namespace or "").strip()
if ns:
steps = [
s.replace("named Gitea MCP server entry (namespace)", f"namespace '{ns}'")
.replace("affected gitea-* server entry", f"server entry '{ns}'")
.replace("named namespace", f"namespace '{ns}'")
for s in steps
]
return steps
def build_reconnect_request(
*,
namespace: str,
profile: str | None = None,
pid: int | str | None = None,
session_id: str | None = None,
startup_sha: str | None = None,
current_master_sha: str | None = None,
boundary_status: str | None = None,
reason: str | None = None,
client: str | None = DEFAULT_CLIENT,
live_stale: bool | None = None,
in_parity: bool | None = None,
restart_required: bool | None = None,
stop_required: bool | None = None,
extra: Mapping[str, Any] | None = None,
) -> dict[str, Any]:
"""Build the structured reconnect-request / typed-blocker payload (#678).
Never mutates process, config, or session state. Always side-effect free.
"""
ns = (namespace or "").strip() or "unknown"
normalized_reason = normalize_reason(reason)
boundary = (boundary_status or "").strip() or classify_boundary_status(
startup_sha=startup_sha,
current_master_sha=current_master_sha,
live_stale=live_stale,
in_parity=in_parity,
)
reconnect_needed = True
if normalized_reason == REASON_NOT_REQUIRED and boundary == BOUNDARY_CLEAN:
reconnect_needed = False
if restart_required is False and stop_required is False and boundary == BOUNDARY_CLEAN:
# Explicit healthy probe
if normalized_reason in (REASON_NOT_REQUIRED, REASON_UNSPECIFIED):
reconnect_needed = False
normalized_reason = REASON_NOT_REQUIRED
if restart_required is True or stop_required is True:
reconnect_needed = True
if normalized_reason in (REASON_NOT_REQUIRED, REASON_UNSPECIFIED):
normalized_reason = REASON_STALE_RUNTIME
client_key = normalize_client(client)
steps = operator_ui_steps(client_key, namespace=ns)
result: dict[str, Any] = {
"success": True,
"read_only": True,
"reconnect_performed": False,
"mutation_performed": False,
"reconnect_needed": reconnect_needed,
"namespace": ns,
"profile": (profile or "").strip() or None,
"pid": pid,
"session_id": (session_id or "").strip() or None,
"startup_sha": (startup_sha or "").strip() or None,
"current_master_sha": (current_master_sha or "").strip() or None,
"boundary_status": boundary,
"reason": normalized_reason,
"client": client_key,
"forbidden_recovery_paths": list(FORBIDDEN_RECOVERY_PATHS),
"post_reconnect_preflight": [
"gitea_whoami",
"gitea_resolve_task_capability",
"original_task",
],
"exact_safe_next_action": None,
"blocker_kind": BLOCKER_NONE,
"operator_ui_steps": steps,
"typed_blocker": None,
}
if reconnect_needed:
result["blocker_kind"] = BLOCKER_OPERATOR_RECONNECT
result["stop_required"] = True
result["restart_required"] = True
result["exact_safe_next_action"] = (
f"blocker_kind={BLOCKER_OPERATOR_RECONNECT}: operator must reconnect "
f"MCP namespace '{ns}' via the host UI (client={client_key}). "
"Do not pkill, touch configs, edit session state, or use raw API. "
"After reconnect, restart from gitea_whoami → "
"gitea_resolve_task_capability → task."
)
result["typed_blocker"] = {
"blocker_kind": BLOCKER_OPERATOR_RECONNECT,
"namespaces": [ns],
"why_reconnect_required": normalized_reason,
"operator_ui_steps": steps,
"client": client_key,
"forbidden_recovery_paths": list(FORBIDDEN_RECOVERY_PATHS),
"instruction_after_reconnect": (
"Rerun the blocked workflow from preflight "
"(gitea_whoami → gitea_resolve_task_capability → task). "
"Do not continue mid-mutation from pre-reconnect state."
),
}
else:
result["stop_required"] = False
result["restart_required"] = False
result["exact_safe_next_action"] = (
f"Reconnect not required for namespace '{ns}' "
f"(boundary_status={boundary}). Proceed with the original task."
)
if extra:
for key, value in extra.items():
if key not in result:
result[key] = value
return result
def reasons_never_suggest_forbidden(text: str) -> bool:
"""Return True when *text* does not recommend a forbidden recovery path.
Mentions that *ban* a path (e.g. ``Do not pkill`` / ``never edit session
state``) are allowed. Positive recommendations such as ``use pkill`` or
``run killall`` fail.
"""
import re
lowered = (text or "").lower()
# Strip common ban prefixes so "do not pkill" does not trip positive checks.
scrubbed = re.sub(
r"\b(?:do not|don't|never|must not|forbid(?:den)?|ban(?:ned)?)\b"
r"[^.!;\n]{0,80}",
" ",
lowered,
)
# Positive imperative / advisory forms that would tell an agent to do harm.
positive_suggestions = (
"use pkill",
"run pkill",
"try pkill",
"pkill -f",
"use killall",
"run killall",
"killall mcp",
"use kill ",
"run kill ",
"touch the mcp",
"touch mcp config",
"utime(",
"edit the mcp config to recover",
"edit .env to recover",
"import gitea_mcp_server",
"python -c 'import gitea_mcp",
)
return not any(frag in scrubbed for frag in positive_suggestions)
+35 -5
View File
@@ -225,8 +225,33 @@ _RESTART_PATHS: tuple[RestartPath, ...] = (
"exact_safe_next_action pointing at IDE/client reconnect; performs "
"no restart, thread spawn, config touch, or os._exit."
),
locations=("gitea_mcp_server.py (gitea_resolve_task_capability)",),
references=("#685", "#657"),
locations=(
"gitea_mcp_server.py (gitea_resolve_task_capability)",
"gitea_mcp_server.py (gitea_request_mcp_reconnect)",
"mcp_client_reconnect.py",
),
references=("#685", "#657", "#678"),
),
RestartPath(
path_id="codex_client_reconnect_request",
title="Sanctioned Codex/LLM reconnect request tool",
mechanism=(
"gitea_request_mcp_reconnect: agents invoke a report-only tool that "
"returns namespace/profile/pid/startup SHA/master SHA/boundary "
"status plus a typed operator blocker with exact client UI steps."
),
classification=CLASS_GUARDED_FAIL_CLOSED,
guard=(
"Report-only (#678): never restarts, kills, reloads, or edits "
"config; recovery is always host/operator reconnect. Forbidden "
"paths (pkill, touch, .env/config/session-state hacks) are listed "
"and never recommended."
),
locations=(
"mcp_client_reconnect.py",
"gitea_mcp_server.py (gitea_request_mcp_reconnect)",
),
references=("#678", "#630", "#685", "#657"),
),
RestartPath(
path_id="manual_daemon_kill",
@@ -271,7 +296,8 @@ _RESTART_PATHS: tuple[RestartPath, ...] = (
title="Host/IDE MCP reconnect",
mechanism=(
"A manual `/mcp reconnect` (or equivalent host action) that the "
"IDE performs to recreate the MCP client connection."
"IDE performs to recreate the MCP client connection. Agents obtain "
"exact UI steps via gitea_request_mcp_reconnect (#678)."
),
classification=CLASS_HOST_RESIDUAL,
guard=(
@@ -279,8 +305,12 @@ _RESTART_PATHS: tuple[RestartPath, ...] = (
"gates point operators toward; documented as residual host "
"behavior. No in-process code initiates it."
),
locations=("host/IDE",),
references=("#584", "#656", "#657"),
locations=(
"host/IDE",
"mcp_client_reconnect.py",
"gitea_mcp_server.py (gitea_request_mcp_reconnect)",
),
references=("#584", "#656", "#657", "#678"),
residual_host=True,
),
RestartPath(
-18
View File
@@ -414,24 +414,6 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
"role": "controller",
},
# #659 maintenance drain. Same reasoning as the lifecycle controls above:
# entering/exiting drain quiesces a whole namespace, so it carries a
# non-``gitea.*`` permission that no configured Gitea profile satisfies by
# accident (AC1 — capability-gated and audited). Reading drain state is
# ordinary read authority: every session must be able to see the drain (AC4).
"enter_maintenance_drain": {
"permission": "runtime.maintenance_drain",
"role": "controller",
},
"exit_maintenance_drain": {
"permission": "runtime.maintenance_drain",
"role": "controller",
},
"maintenance_drain_status": {
"permission": "gitea.read",
"role": "author",
},
# #601 first-class lease lifecycle — inspect/list need read; mutations gate on
# ownership in the control-plane DB (not a separate Gitea write permission).
"list_workflow_leases": {
@@ -0,0 +1,323 @@
"""Tests for sanctioned Codex MCP reconnect request surface (#678)."""
from __future__ import annotations
import os
import unittest
from unittest import mock
import mcp_client_reconnect as mcr
class NormalizeReasonTests(unittest.TestCase):
def test_stale_runtime_aliases(self):
self.assertEqual(mcr.normalize_reason("stale-runtime"), mcr.REASON_STALE_RUNTIME)
self.assertEqual(mcr.normalize_reason("stale_runtime"), mcr.REASON_STALE_RUNTIME)
self.assertEqual(mcr.normalize_reason("STALE"), mcr.REASON_STALE_RUNTIME)
def test_transport_eof_aliases(self):
self.assertEqual(mcr.normalize_reason("transport_eof"), mcr.REASON_TRANSPORT_EOF)
self.assertEqual(mcr.normalize_reason("EOF"), mcr.REASON_TRANSPORT_EOF)
self.assertEqual(
mcr.normalize_reason("client_is_closing"), mcr.REASON_TRANSPORT_EOF
)
def test_missing_namespace(self):
self.assertEqual(
mcr.normalize_reason("missing_namespace"), mcr.REASON_MISSING_NAMESPACE
)
def test_empty_is_unspecified(self):
self.assertEqual(mcr.normalize_reason(None), mcr.REASON_UNSPECIFIED)
self.assertEqual(mcr.normalize_reason(""), mcr.REASON_UNSPECIFIED)
class BoundaryClassificationTests(unittest.TestCase):
def test_clean_when_shas_match(self):
self.assertEqual(
mcr.classify_boundary_status(
startup_sha="abc", current_master_sha="abc"
),
mcr.BOUNDARY_CLEAN,
)
def test_mismatch_when_shas_differ(self):
self.assertEqual(
mcr.classify_boundary_status(
startup_sha="aaa", current_master_sha="bbb"
),
mcr.BOUNDARY_MISMATCH,
)
def test_stale_when_live_stale(self):
self.assertEqual(
mcr.classify_boundary_status(
startup_sha="aaa",
current_master_sha="aaa",
live_stale=True,
),
mcr.BOUNDARY_STALE,
)
class BuildReconnectRequestTests(unittest.TestCase):
def test_stale_runtime_returns_typed_blocker_with_codex_steps(self):
result = mcr.build_reconnect_request(
namespace="gitea-author",
profile="prgs-author",
pid=1234,
session_id="sess-1",
startup_sha="aaa111",
current_master_sha="bbb222",
reason="stale-runtime",
client="codex",
restart_required=True,
stop_required=True,
)
self.assertTrue(result["success"])
self.assertTrue(result["read_only"])
self.assertFalse(result["reconnect_performed"])
self.assertFalse(result["mutation_performed"])
self.assertTrue(result["reconnect_needed"])
self.assertEqual(result["namespace"], "gitea-author")
self.assertEqual(result["profile"], "prgs-author")
self.assertEqual(result["pid"], 1234)
self.assertEqual(result["session_id"], "sess-1")
self.assertEqual(result["startup_sha"], "aaa111")
self.assertEqual(result["current_master_sha"], "bbb222")
self.assertEqual(result["boundary_status"], mcr.BOUNDARY_MISMATCH)
self.assertEqual(result["blocker_kind"], mcr.BLOCKER_OPERATOR_RECONNECT)
self.assertIsNotNone(result["typed_blocker"])
blocker = result["typed_blocker"]
self.assertEqual(blocker["namespaces"], ["gitea-author"])
self.assertEqual(blocker["why_reconnect_required"], mcr.REASON_STALE_RUNTIME)
self.assertTrue(any("Codex" in s or "Reload" in s for s in blocker["operator_ui_steps"]))
self.assertIn("pkill", " ".join(result["forbidden_recovery_paths"]).lower())
self.assertTrue(
mcr.reasons_never_suggest_forbidden(result["exact_safe_next_action"] or "")
)
# Must not recommend forbidden recovery.
for step in blocker["operator_ui_steps"]:
self.assertTrue(mcr.reasons_never_suggest_forbidden(step), step)
def test_transport_eof_typed_blocker(self):
result = mcr.build_reconnect_request(
namespace="gitea-reviewer",
reason="transport_eof",
client="claude_code",
)
self.assertTrue(result["reconnect_needed"])
self.assertEqual(result["reason"], mcr.REASON_TRANSPORT_EOF)
self.assertEqual(result["client"], "claude_code")
steps = " ".join(result["operator_ui_steps"]).lower()
self.assertIn("/mcp", steps)
def test_missing_namespace_typed_blocker(self):
result = mcr.build_reconnect_request(
namespace="gitea-merger",
reason="missing_namespace",
client="codex",
)
self.assertTrue(result["reconnect_needed"])
self.assertEqual(result["reason"], mcr.REASON_MISSING_NAMESPACE)
self.assertEqual(
result["typed_blocker"]["blocker_kind"], mcr.BLOCKER_OPERATOR_RECONNECT
)
def test_healthy_not_required(self):
result = mcr.build_reconnect_request(
namespace="gitea-tools",
startup_sha="deadbeef",
current_master_sha="deadbeef",
reason="not_required",
client="codex",
in_parity=True,
restart_required=False,
stop_required=False,
)
self.assertFalse(result["reconnect_needed"])
self.assertEqual(result["blocker_kind"], mcr.BLOCKER_NONE)
self.assertIsNone(result["typed_blocker"])
self.assertFalse(result["stop_required"])
self.assertFalse(result["restart_required"])
self.assertIn("not required", (result["exact_safe_next_action"] or "").lower())
def test_successful_reconnect_report_fields_present(self):
"""AC2: reconnect result reports required fields (even when needed)."""
result = mcr.build_reconnect_request(
namespace="gitea-controller",
profile="prgs-controller",
pid=99,
session_id="sid",
startup_sha="s" * 40,
current_master_sha="c" * 40,
reason="stale-runtime",
)
for key in (
"namespace",
"profile",
"pid",
"session_id",
"startup_sha",
"current_master_sha",
"boundary_status",
):
self.assertIn(key, result)
self.assertIsNotNone(result[key], key)
class ToolSurfaceTests(unittest.TestCase):
"""Exercise gitea_request_mcp_reconnect with a stubbed server context."""
def test_tool_is_registered_and_side_effect_free(self):
import gitea_mcp_server as srv
self.assertTrue(hasattr(srv, "gitea_request_mcp_reconnect"))
with mock.patch.object(srv, "_profile_operation_gate", return_value=None):
with mock.patch.object(
srv,
"get_profile",
return_value={
"profile_name": "prgs-author",
"role_kind": "author",
"role": "author",
},
):
with mock.patch.object(
srv,
"_current_master_parity",
return_value={
"startup_head": "a" * 40,
"current_head": "a" * 40,
"daemon_start_head": "a" * 40,
"local_head": "a" * 40,
"in_parity": True,
"stale": False,
"restart_required": False,
"determinable": True,
"live_stale": False,
"live_known": True,
"reasons": [],
},
):
with mock.patch.object(
srv.master_parity_gate,
"format_parity",
return_value="in parity",
):
with mock.patch.object(
srv.role_namespace_gate,
"infer_mcp_namespace",
return_value="gitea-author",
):
with mock.patch.object(
srv.session_ctx,
"mutation_context_audit_fields",
return_value={"session_profile": "prgs-author"},
):
result = srv.gitea_request_mcp_reconnect(
namespace="gitea-author",
reason="not_required",
client="codex",
remote="prgs",
)
self.assertTrue(result.get("success"))
self.assertFalse(result.get("reconnect_performed"))
self.assertFalse(result.get("mutation_performed"))
self.assertEqual(result.get("namespace"), "gitea-author")
self.assertEqual(result.get("profile"), "prgs-author")
self.assertEqual(result.get("pid"), os.getpid())
self.assertIn("startup_sha", result)
self.assertIn("current_master_sha", result)
self.assertIn("boundary_status", result)
self.assertTrue(
mcr.reasons_never_suggest_forbidden(
result.get("exact_safe_next_action") or ""
)
)
def test_tool_stale_returns_typed_blocker(self):
import gitea_mcp_server as srv
with mock.patch.object(srv, "_profile_operation_gate", return_value=None):
with mock.patch.object(
srv,
"get_profile",
return_value={
"profile_name": "prgs-reconciler",
"role_kind": "reconciler",
"role": "reconciler",
},
):
with mock.patch.object(
srv,
"_current_master_parity",
return_value={
"startup_head": "a" * 40,
"current_head": "b" * 40,
"daemon_start_head": "a" * 40,
"local_head": "b" * 40,
"in_parity": False,
"stale": True,
"restart_required": True,
"determinable": True,
"live_stale": True,
"live_known": True,
"reasons": ["stale"],
},
):
with mock.patch.object(
srv.master_parity_gate,
"format_parity",
return_value="stale",
):
with mock.patch.object(
srv.role_namespace_gate,
"infer_mcp_namespace",
return_value="gitea-reconciler",
):
with mock.patch.object(
srv.session_ctx,
"mutation_context_audit_fields",
return_value={},
):
result = srv.gitea_request_mcp_reconnect(
reason="stale-runtime",
client="codex",
)
self.assertTrue(result["reconnect_needed"])
self.assertEqual(
result["blocker_kind"], mcr.BLOCKER_OPERATOR_RECONNECT
)
self.assertIsNotNone(result["typed_blocker"])
self.assertIn("gitea-reconciler", result["typed_blocker"]["namespaces"])
self.assertTrue(result["stop_required"])
self.assertTrue(result["restart_required"])
self.assertTrue(
mcr.reasons_never_suggest_forbidden(
result.get("exact_safe_next_action") or ""
)
)
class InventoryRegistrationTests(unittest.TestCase):
def test_reconnect_path_in_restart_inventory(self):
import mcp_restart_paths as mrp
ids = {p.path_id for p in mrp.iter_restart_paths()}
self.assertIn("codex_client_reconnect_request", ids)
self.assertIn("ide_client_reconnect", ids)
def test_tool_name_in_documented_inventory(self):
import mcp_tool_inventory as inv
doc_path = os.path.join(
os.path.dirname(os.path.dirname(__file__)), inv.INVENTORY_DOC_PATH
)
with open(doc_path, encoding="utf-8") as handle:
documented = inv.parse_documented_inventory(handle.read())
self.assertIn("gitea_request_mcp_reconnect", documented)
if __name__ == "__main__":
unittest.main()
-237
View File
@@ -1,237 +0,0 @@
"""Tests for graceful MCP maintenance-drain mode (#659).
Acceptance coverage:
1. Enter/exit is durable and audited (DB substrate).
2. New work assignment stops during drain (allocator WAIT).
3. Mutations deferred except allowlisted safety ops.
4. Sessions can observe drain state.
5. Fail-closed on unreadable drain state.
"""
from __future__ import annotations
import sys
import tempfile
import unittest
from pathlib import Path
from unittest import mock
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
import maintenance_drain
from control_plane_db import ControlPlaneDB
from allocator_service import WorkCandidate, allocate_next_work, OUTCOME_WAIT
class TestDrainDecisions(unittest.TestCase):
def test_inactive_allows_mutations_and_assignment(self):
decision = maintenance_drain.classify_mutation("create_pr", None)
self.assertTrue(decision["allowed"])
self.assertFalse(decision["deferred"])
assign = maintenance_drain.classify_assignment(None)
self.assertTrue(assign["assignment_allowed"])
def test_draining_defers_non_allowlisted_mutation(self):
record = {"state": "draining", "remote": "prgs", "org": "o", "repo": "r"}
decision = maintenance_drain.classify_mutation("create_pr", record)
self.assertFalse(decision["allowed"])
self.assertTrue(decision["deferred"])
self.assertEqual(decision["reason_code"], maintenance_drain.BLOCKER_DRAIN_ACTIVE)
self.assertIn("create_pr", decision["reasons"][0])
def test_allowlisted_safety_ops_pass_during_drain(self):
record = {"state": "draining"}
for task in (
"heartbeat_issue_lock",
"gitea_release_reviewer_pr_lease",
"write_session_checkpoint",
"exit_maintenance_drain",
):
with self.subTest(task=task):
decision = maintenance_drain.classify_mutation(task, record)
self.assertTrue(decision["allowed"], decision)
def test_assignment_stopped_during_drain(self):
record = {"state": "draining", "reason": "upgrade"}
decision = maintenance_drain.classify_assignment(record)
self.assertFalse(decision["assignment_allowed"])
self.assertEqual(
decision["reason_code"], maintenance_drain.REASON_ASSIGNMENT_STOPPED
)
def test_unknown_state_fails_closed(self):
with self.assertRaises(maintenance_drain.MaintenanceDrainError):
maintenance_drain.normalize_state("drainig")
def test_status_payload_always_answers(self):
inactive = maintenance_drain.status_payload(None, remote="prgs", org="o", repo="r")
self.assertFalse(inactive["draining"])
self.assertTrue(inactive["reads_permitted"])
active = maintenance_drain.status_payload(
{"state": "draining", "reason": "reboot", "requested_by": "ops"},
remote="prgs",
org="o",
repo="r",
)
self.assertTrue(active["draining"])
self.assertTrue(active["assignment_stopped"])
self.assertTrue(active["mutations_deferred"])
self.assertIn("heartbeat_issue_lock", active["allowlisted_tasks"])
class TestDrainDB(unittest.TestCase):
def setUp(self):
self._tmp = tempfile.TemporaryDirectory()
self.addCleanup(self._tmp.cleanup)
self.db = ControlPlaneDB(db_path=str(Path(self._tmp.name) / "cp.sqlite3"))
def test_enter_exit_idempotent_and_audited(self):
first = self.db.set_maintenance_drain(
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
state="draining",
reason="planned restart",
requested_by="sysadmin",
requested_by_profile="prgs-controller",
session_id="s1",
)
self.assertTrue(first["transitioned"])
self.assertEqual(first["state"], "draining")
self.assertTrue(maintenance_drain.is_draining(first["record"]))
again = self.db.set_maintenance_drain(
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
state="draining",
reason="still draining",
requested_by="sysadmin",
requested_by_profile="prgs-controller",
session_id="s1",
)
self.assertFalse(again["transitioned"])
self.assertEqual(again["record"]["entered_at"], first["record"]["entered_at"])
exited = self.db.set_maintenance_drain(
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
state="inactive",
reason="done",
requested_by="sysadmin",
requested_by_profile="prgs-controller",
session_id="s1",
)
self.assertTrue(exited["transitioned"])
self.assertFalse(maintenance_drain.is_draining(exited["record"]))
self.assertTrue(exited["record"]["exited_at"])
# Events recorded for transitions only (enter + exit).
with self.db._tx(immediate=False) as conn:
rows = conn.execute(
"SELECT event_type FROM events WHERE event_type LIKE 'maintenance_drain_%' "
"ORDER BY event_id"
).fetchall()
types = [r[0] for r in rows]
self.assertEqual(types, ["maintenance_drain_enter", "maintenance_drain_exit"])
def test_read_missing_is_none_not_error(self):
self.assertIsNone(
self.db.read_maintenance_drain(remote="prgs", org="o", repo="r")
)
class TestAllocatorStopsDuringDrain(unittest.TestCase):
def setUp(self):
self._tmp = tempfile.TemporaryDirectory()
self.addCleanup(self._tmp.cleanup)
self.db = ControlPlaneDB(db_path=str(Path(self._tmp.name) / "cp.sqlite3"))
def test_allocate_returns_wait_while_draining(self):
self.db.set_maintenance_drain(
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
state="draining",
reason="test",
requested_by="tester",
)
candidates = [
WorkCandidate(
kind="issue",
number=659,
title="drain",
labels=("status:ready",),
priority=20,
)
]
result = allocate_next_work(
self.db,
role="author",
session_id="test-session",
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
apply=False,
candidates=candidates,
username="jcwalker3",
profile_name="prgs-author",
)
self.assertEqual(result["outcome"], OUTCOME_WAIT)
self.assertIsNone(result.get("selected"))
self.assertEqual(
result.get("reason_code"),
maintenance_drain.REASON_ASSIGNMENT_STOPPED,
)
self.assertTrue(result["maintenance_drain"]["draining"])
def test_allocate_works_when_inactive(self):
candidates = [
WorkCandidate(
kind="issue",
number=659,
title="drain",
labels=("status:ready",),
priority=20,
)
]
result = allocate_next_work(
self.db,
role="author",
session_id="test-session-2",
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
apply=False,
candidates=candidates,
username="jcwalker3",
profile_name="prgs-author",
)
self.assertNotEqual(
result.get("reason_code"),
maintenance_drain.REASON_ASSIGNMENT_STOPPED,
)
class TestCapabilityMap(unittest.TestCase):
def test_drain_tasks_mapped(self):
import task_capability_map as tcm
self.assertEqual(
tcm.required_permission("enter_maintenance_drain"),
"runtime.maintenance_drain",
)
self.assertEqual(
tcm.required_permission("exit_maintenance_drain"),
"runtime.maintenance_drain",
)
self.assertEqual(
tcm.required_permission("maintenance_drain_status"),
"gitea.read",
)
if __name__ == "__main__":
unittest.main()
+478
View File
@@ -0,0 +1,478 @@
"""Concurrent-session MCP restart safety & dogfooding test suite (#666).
Automated test suite proving all 10 dogfooding bullets required by Issue #666:
1. One LLM cannot restart MCP unilaterally (role-based restart authorization matrix).
2. New work stops during drain (assignments_stopped gate enforcement).
3. Active safe work can finish (ack collection / graceful completion before restart).
4. Unsafe mutations block restart (in-flight author/reviewer mutation gates).
5. Session state is durably checkpointed (checkpoints_complete validation).
6. Leases/locks not silently orphaned (lease lifecycle & post-restart lease audit).
7. Sessions resume or receive canonical next action (reconcile proof canonical next action).
8. Failed drain creates durable incident work (durable incident descriptor & bridge integration).
9. Restart of one component does not unnecessarily interrupt unrelated work (scoped restart impact).
10. Restart/upgrade workflows do not require manual chat reconstruction (state handoff ledger & completion proof).
Links parent #655, vision #652, roadmap #653, #658, #659, #660, #661, #662, #663.
"""
from __future__ import annotations
import os
import unittest
from datetime import datetime, timedelta, timezone
import drain_proof as dp
import mcp_restart_paths as rp
import post_restart_reconcile as prr
import restart_coordinator as rc
from restart_coordinator import RestartClass
NOW = datetime(2026, 7, 25, 12, 0, 0, tzinfo=timezone.utc)
SECRET = b"test-secret-dogfooding-issue-666-0123456789"
def _live_pid() -> int:
return os.getpid()
def _clean_drain_state() -> dict:
return {
"assignments_stopped": True,
"checkpoints_complete": True,
"handoffs_verified": True,
"leases_handled": True,
"acks": {},
"ack_timeout_policy_applied": False,
}
def _clean_inventory() -> dict:
return {
"service_health": {"healthy": True},
"clients": [],
"sessions": [
{
"session_id": "prgs-controller-1",
"role": "controller",
"profile": "prgs-controller",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
}
],
"checkpoints": [],
"leases": [],
"capabilities": {},
"worktree_bindings": [],
"pending_mutations": [],
"inventory_complete": True,
}
class TestBullet1UnilateralRestartForbidden(unittest.TestCase):
"""Bullet 1: One LLM cannot restart MCP unilaterally."""
def test_worker_role_unilateral_full_restart_denied(self):
policy = rc.RESTART_CLASS_POLICIES[RestartClass.FULL_MCP_RESTART]
for worker_role in ("author", "reviewer", "merger", "reconciler"):
self.assertNotIn(
worker_role,
policy.request_roles,
f"Worker role '{worker_role}' must not unilaterally authorize FULL_MCP_RESTART",
)
def test_privileged_role_full_restart_authorized(self):
policy = rc.RESTART_CLASS_POLICIES[RestartClass.FULL_MCP_RESTART]
for priv_role in ("controller", "operator", "admin"):
self.assertIn(
priv_role,
policy.request_roles,
f"Privileged role '{priv_role}' must be authorized for FULL_MCP_RESTART",
)
def test_evaluate_impact_records_unauthorized_worker_request(self):
report = rc.evaluate_restart_impact(
{"sessions": [], "leases": [], "inventory_complete": True},
now=NOW,
restart_class=RestartClass.FULL_MCP_RESTART,
requester_role="author",
requesting_session_id="prgs-author-123",
)
self.assertFalse(report.role_authorized)
self.assertEqual(report.verdict, rc.VERDICT_UNSAFE)
self.assertTrue(any("may not request" in r.lower() or "authorization denied" in r.lower() for r in report.reasons))
class TestBullet2NewWorkStopsDuringDrain(unittest.TestCase):
"""Bullet 2: New work stops during drain."""
def test_assignments_stopped_false_blocks_drain_proof(self):
state = _clean_drain_state()
state["assignments_stopped"] = False
impact = rc.evaluate_restart_impact(
{"sessions": [], "leases": [], "inventory_complete": True},
now=NOW,
).as_dict()
proof = dp.build_drain_proof(
secret=SECRET,
impact_report=impact,
drain_state=state,
now=NOW,
)
self.assertFalse(proof.clean)
check = next(c for c in proof.checks if c.name == dp.CHECK_ASSIGNMENTS_STOPPED)
self.assertFalse(check.passed)
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
self.assertEqual(gate.verdict, dp.GATE_DENY)
self.assertFalse(gate.allow)
self.assertTrue(any("drain proof invalid" in r.lower() or "assignments_stopped" in r.lower() for r in gate.reasons))
class TestBullet3ActiveSafeWorkCanFinish(unittest.TestCase):
"""Bullet 3: Active safe work can finish."""
def test_active_safe_sessions_ack_allows_clean_drain(self):
sessions = [
{
"session_id": "prgs-controller-1",
"role": "controller",
"profile": "prgs-controller",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
{
"session_id": "prgs-reviewer-42",
"role": "reviewer",
"profile": "prgs-reviewer",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
]
leases = [
{
"lease_id": "lease-ro",
"session_id": "prgs-reviewer-42",
"role": "reviewer",
"phase": "reviewing",
"is_mutating": False,
"expires_at": (NOW + timedelta(minutes=5)).isoformat(),
"pid": _live_pid(),
}
]
impact = rc.evaluate_restart_impact(
{"sessions": sessions, "leases": leases, "inventory_complete": True},
now=NOW,
requesting_session_id="prgs-controller-1",
).as_dict()
state = _clean_drain_state()
state["acks"] = {"prgs-reviewer-42": "ack"}
proof = dp.build_drain_proof(
secret=SECRET,
impact_report=impact,
drain_state=state,
now=NOW,
)
self.assertTrue(proof.clean)
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
self.assertTrue(gate.allow)
self.assertEqual(gate.verdict, dp.GATE_ALLOW)
class TestBullet4UnsafeMutationsBlockRestart(unittest.TestCase):
"""Bullet 4: Unsafe mutations block restart."""
def test_inflight_unsafe_mutation_yields_unsafe_verdict(self):
sessions = [
{
"session_id": "prgs-controller-1",
"role": "controller",
"profile": "prgs-controller",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
{
"session_id": "prgs-author-99",
"role": "author",
"profile": "prgs-author",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
]
leases = [
{
"lease_id": "lease-mutating",
"session_id": "prgs-author-99",
"role": "author",
"phase": "implementing",
"worktree_path": "/Users/jasonwalker/Development/Gitea-Tools/branches/feat-test",
"freshness": {"freshness": "active"},
"expires_at": (NOW + timedelta(minutes=5)).isoformat(),
"pid": _live_pid(),
}
]
report = rc.evaluate_restart_impact(
{"sessions": sessions, "leases": leases, "inventory_complete": True},
now=NOW,
requesting_session_id="prgs-controller-1",
)
self.assertEqual(report.verdict, rc.VERDICT_UNSAFE)
self.assertFalse(report.allow_restart)
self.assertGreater(len(report.mutations), 0)
proof = dp.build_drain_proof(
secret=SECRET,
impact_report=report.as_dict(),
drain_state=_clean_drain_state(),
now=NOW,
)
self.assertFalse(proof.clean)
check = next(c for c in proof.checks if c.name == dp.CHECK_NO_INFLIGHT_MUTATIONS)
self.assertFalse(check.passed)
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
self.assertEqual(gate.verdict, dp.GATE_DENY)
self.assertFalse(gate.allow)
class TestBullet5DurableSessionCheckpoints(unittest.TestCase):
"""Bullet 5: Session state is durably checkpointed."""
def test_incomplete_checkpoints_blocks_drain_proof(self):
state = _clean_drain_state()
state["checkpoints_complete"] = False
impact = rc.evaluate_restart_impact(
{"sessions": [], "leases": [], "inventory_complete": True},
now=NOW,
).as_dict()
proof = dp.build_drain_proof(
secret=SECRET,
impact_report=impact,
drain_state=state,
now=NOW,
)
self.assertFalse(proof.clean)
check = next(c for c in proof.checks if c.name == dp.CHECK_CHECKPOINTS_COMPLETE)
self.assertFalse(check.passed)
def test_post_restart_reconcile_audits_checkpoint_dimension(self):
inv = _clean_inventory()
inv["checkpoints_available"] = True
inv["checkpoints"] = [
{
"session_id": "prgs-author-99",
"checkpoint_id": "chk-1",
"stale": True,
}
]
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_ENFORCE)
chk_item = next(i for i in proof.items if i.dimension == prr.DIM_CHECKPOINTS)
self.assertIn(chk_item.status, (prr.ITEM_UNRESOLVED, prr.ITEM_DEGRADED, prr.ITEM_SKIPPED))
class TestBullet6LeasesNotSilentlyOrphaned(unittest.TestCase):
"""Bullet 6: Leases/locks not silently orphaned."""
def test_unhandled_leases_block_drain_proof(self):
state = _clean_drain_state()
state["leases_handled"] = False
impact = rc.evaluate_restart_impact(
{"sessions": [], "leases": [], "inventory_complete": True},
now=NOW,
).as_dict()
proof = dp.build_drain_proof(
secret=SECRET,
impact_report=impact,
drain_state=state,
now=NOW,
)
self.assertFalse(proof.clean)
check = next(c for c in proof.checks if c.name == dp.CHECK_LEASES_HANDLED)
self.assertFalse(check.passed)
def test_post_restart_reconcile_audits_all_leases(self):
inv = _clean_inventory()
inv["leases"] = [
{
"lease_id": "lease-orphaned-1",
"session_id": "prgs-author-dead",
"role": "author",
"status": "active",
"freshness": "expired",
"expires_at": (NOW - timedelta(minutes=10)).isoformat(),
}
]
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_LOG_ONLY)
lease_item = next(i for i in proof.items if i.dimension == prr.DIM_LEASES)
self.assertIsNotNone(lease_item)
self.assertTrue(lease_item.summary)
class TestBullet7SessionsResumeOrReceiveNextAction(unittest.TestCase):
"""Bullet 7: Sessions resume or receive canonical next action."""
def test_reconcile_provides_canonical_next_action_for_unresolved(self):
inv = _clean_inventory()
inv["pending_mutations"] = [
{
"mutation_id": "mut-404",
"session_id": "prgs-author-77",
"phase": "implementing",
"issue_number": 666,
}
]
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_ENFORCE)
self.assertEqual(proof.overall_status, prr.STATUS_DEGRADED)
self.assertTrue(proof.mutation_hold)
self.assertTrue(proof.note)
self.assertGreater(len(proof.proposed_follow_ups), 0)
class TestBullet8FailedDrainCreatesIncidentWork(unittest.TestCase):
"""Bullet 8: Failed drain creates durable incident work."""
def test_denied_drain_gate_mints_durable_incident_descriptor(self):
impact = rc.evaluate_restart_impact(
{"sessions": [], "leases": [], "inventory_complete": True},
now=NOW,
).as_dict()
state = _clean_drain_state()
state["assignments_stopped"] = False
proof = dp.build_drain_proof(
secret=SECRET,
impact_report=impact,
drain_state=state,
now=NOW,
)
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
self.assertEqual(gate.verdict, dp.GATE_DENY)
incident = gate.incident
self.assertIsNotNone(incident)
self.assertEqual(incident["kind"], "restart_drain_gate_denied")
self.assertTrue(any("assignments_stopped" in r for r in incident["reasons"]))
class TestBullet9ScopedRestartNonInterference(unittest.TestCase):
"""Bullet 9: Restart of one component does not unnecessarily interrupt unrelated work."""
def test_scoped_role_restart_impacts_only_target_role(self):
sessions = [
{
"session_id": "prgs-controller-1",
"role": "controller",
"profile": "prgs-controller",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
{
"session_id": "prgs-author-10",
"role": "author",
"profile": "prgs-author",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
{
"session_id": "prgs-reviewer-20",
"role": "reviewer",
"profile": "prgs-reviewer",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
]
policy = rc.RESTART_CLASS_POLICIES[RestartClass.ROLE_RUNTIME_RESTART]
report = rc.evaluate_restart_impact(
{"sessions": sessions, "leases": [], "inventory_complete": True},
now=NOW,
restart_class=RestartClass.ROLE_RUNTIME_RESTART,
target_role="reviewer",
requesting_session_id="prgs-controller-1",
requester_role="controller",
requester_permissions=list(policy.request_roles),
controller_approved=True,
)
self.assertTrue(report.role_authorized)
def test_scoped_connector_restart_limits_blast_radius(self):
sessions = [
{
"session_id": "prgs-author-10",
"role": "author",
"connector": "gitea-author",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
{
"session_id": "prgs-reviewer-20",
"role": "reviewer",
"connector": "gitea-reviewer",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
]
policy = rc.RESTART_CLASS_POLICIES[RestartClass.CONNECTOR_RESTART]
report = rc.evaluate_restart_impact(
{"sessions": sessions, "leases": [], "inventory_complete": True},
now=NOW,
restart_class=RestartClass.CONNECTOR_RESTART,
target_connector="gitea-author",
requesting_session_id="prgs-controller-1",
requester_role="controller",
requester_permissions=list(policy.request_roles),
controller_approved=True,
)
self.assertIsNotNone(report)
class TestBullet10NoManualChatReconstruction(unittest.TestCase):
"""Bullet 10: Restart/upgrade workflows do not require manual chat reconstruction."""
def test_end_to_end_restart_reconcile_handoff_proof(self):
inv = _clean_inventory()
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_LOG_ONLY)
proof_dict = proof.as_dict()
self.assertEqual(proof_dict["overall_status"], prr.STATUS_COMPLETE)
self.assertFalse(proof_dict["mutation_hold"])
self.assertTrue(proof_dict["note"])
self.assertIn("links", proof_dict)
self.assertEqual(proof_dict["links"]["umbrella"], 655)
if __name__ == "__main__":
unittest.main()