Compare commits

..
Author SHA1 Message Date
sysadmin f0c6255d7d Merge remote-tracking branch 'prgs/master' into feat/issue-659-maintenance-drain-mode 2026-07-25 18:40:57 -04:00
sysadminandClaude Opus 4.8 77d808e7d4 merge(master): resolve PR #907 allocator drain vs side_effect_free
Keep #659 maintenance-drain assignment stop and #643 side_effect_free
apply guard; session registration remains gated by side_effect_free.

Co-Authored-By: Claude Opus 4.8 (1M context) <[email protected]>
2026-07-25 18:12:03 -04:00
sysadminandClaude Opus 4.8 e91b94db56 feat(mcp): implement graceful maintenance-drain mode (Closes #659)
Add durable per-repo drain state, allocator assignment stop, mutation
deferral with a safety allowlist, observable status, and capability-gated
enter/exit tools. Drain proof/restart gate remain #661.

Co-Authored-By: Claude Opus 4.8 (1M context) <[email protected]>
2026-07-25 17:02:12 -04:00
18 changed files with 1071 additions and 1056 deletions
+41
View File
@@ -26,6 +26,7 @@ from dataclasses import dataclass, field
from datetime import datetime, timezone from datetime import datetime, timezone
from typing import Any, Mapping, Sequence from typing import Any, Mapping, Sequence
import maintenance_drain
from control_plane_db import ( from control_plane_db import (
ControlPlaneDB, ControlPlaneDB,
ControlPlaneError, ControlPlaneError,
@@ -936,6 +937,46 @@ def allocate_next_work(
"allocation_mode": (allocation_mode or "").strip() or None, "allocation_mode": (allocation_mode or "").strip() or None,
} }
# #659 AC2: while maintenance drain is active, no new work is assigned —
# for dry-run and apply alike, so a preview can never be read as evidence
# that work was assignable during the drain. Checked before session
# registration so a drained allocator leaves no new state behind.
try:
drain_record = db.read_maintenance_drain(remote=remote, org=org, repo=repo)
except Exception as exc: # noqa: BLE001 — unreadable drain state fails closed
return {
"success": False,
"outcome": OUTCOME_NO_SAFE,
"reasons": [
f"maintenance-drain state lookup failed: {exc} (fail closed, #659)"
],
"skipped": [],
"assignment": None,
"substrate": "control_plane_db",
}
drain_decision = maintenance_drain.classify_assignment(drain_record)
if not drain_decision["assignment_allowed"]:
return {
"success": True,
"outcome": OUTCOME_WAIT,
"apply": apply,
"role": role_norm,
"allocation_mode": mode,
"remote": remote,
"org": org,
"repo": repo,
"selected": None,
"reasons": list(drain_decision["reasons"]),
"reason_code": drain_decision["reason_code"],
"skipped": [],
"assignment": None,
"substrate": "control_plane_db",
"maintenance_drain": maintenance_drain.status_payload(
drain_record, remote=remote, org=org, repo=repo
),
}
# A side-effect-free run may never reserve: reserving is a write, and the # A side-effect-free run may never reserve: reserving is a write, and the
# flag is the caller's assertion that this call writes nothing (#643). # flag is the caller's assertion that this call writes nothing (#643).
if side_effect_free and apply: if side_effect_free and apply:
+194 -1
View File
@@ -31,8 +31,9 @@ from typing import Any, Iterator, Sequence
import dependency_graph import dependency_graph
import gitea_audit import gitea_audit
import maintenance_drain
SCHEMA_VERSION = 5 SCHEMA_VERSION = 6
# Assignable work kinds only — raw monitoring incidents are never work items. # Assignable work kinds only — raw monitoring incidents are never work items.
WORK_KINDS = frozenset({"issue", "pr"}) WORK_KINDS = frozenset({"issue", "pr"})
@@ -239,6 +240,31 @@ CREATE INDEX IF NOT EXISTS idx_session_checkpoints_session
CREATE INDEX IF NOT EXISTS idx_session_checkpoints_work CREATE INDEX IF NOT EXISTS idx_session_checkpoints_work
ON session_checkpoints(remote, org, repo, work_kind, work_number); ON session_checkpoints(remote, org, repo, work_kind, work_number);
-- Graceful maintenance-drain state (#659). One current row per repository
-- scope — drain is a *state*, not a history, so entering and exiting update
-- the same row and every transition is audited to ``events``. Creating the
-- table is the v5->v6 migration: additive, idempotent, and it never touches
-- prior tables. ``state`` is CHECK-constrained so an unknown value can never
-- be written and later read as "not draining".
CREATE TABLE IF NOT EXISTS maintenance_drain (
drain_id TEXT PRIMARY KEY,
remote TEXT NOT NULL,
org TEXT NOT NULL,
repo TEXT NOT NULL,
state TEXT NOT NULL DEFAULT 'inactive'
CHECK (state IN ('inactive', 'draining')),
reason TEXT NOT NULL DEFAULT '',
requested_by TEXT NOT NULL DEFAULT '',
requested_by_profile TEXT NOT NULL DEFAULT '',
session_id TEXT NOT NULL DEFAULT '',
entered_at TEXT NOT NULL DEFAULT '',
exited_at TEXT NOT NULL DEFAULT '',
drain_schema_version INTEGER NOT NULL DEFAULT 6,
created_at TEXT NOT NULL,
updated_at TEXT NOT NULL,
UNIQUE (remote, org, repo)
);
-- Model usage, token cost, latency, and performance events (#651) -- Model usage, token cost, latency, and performance events (#651)
CREATE TABLE IF NOT EXISTS usage_events ( CREATE TABLE IF NOT EXISTS usage_events (
usage_id INTEGER PRIMARY KEY AUTOINCREMENT, usage_id INTEGER PRIMARY KEY AUTOINCREMENT,
@@ -3025,3 +3051,170 @@ class ControlPlaneDB:
"live_lease_id": None if live_lease_id is None else str(live_lease_id), "live_lease_id": None if live_lease_id is None else str(live_lease_id),
"reconcile_action": "reconcile_required" if stale else "safe_to_resume", "reconcile_action": "reconcile_required" if stale else "safe_to_resume",
} }
# ── Maintenance drain (#659) ─────────────────────────────────────────────
@staticmethod
def _maintenance_drain_row(row: sqlite3.Row | None) -> dict[str, Any] | None:
"""Convert a ``maintenance_drain`` row to a plain record."""
if row is None:
return None
return {key: row[key] for key in row.keys()}
def read_maintenance_drain(
self, *, remote: str, org: str, repo: str
) -> dict[str, Any] | None:
"""Return the current drain record for a scope, or None if never set.
None and a stored ``inactive`` row mean the same thing to callers —
``maintenance_drain.is_draining`` treats both as not draining — so the
read never has to invent a record to answer the gate.
"""
with self._tx(immediate=False) as conn:
row = conn.execute(
"""
SELECT * FROM maintenance_drain
WHERE remote = ? AND org = ? AND repo = ?
""",
(str(remote or ""), str(org or ""), str(repo or "")),
).fetchone()
return self._maintenance_drain_row(row)
def set_maintenance_drain(
self,
*,
remote: str,
org: str,
repo: str,
state: str,
reason: str = "",
requested_by: str = "",
requested_by_profile: str = "",
session_id: str = "",
) -> dict[str, Any]:
"""Enter or exit maintenance drain for one repository scope (AC1).
The state transition is audited to ``events`` — entering and exiting
are exactly the moments an operator has to be able to reconstruct
later. Re-entering an already-draining scope is idempotent: it refreshes
the reason/owner metadata, keeps the original ``entered_at``, and
records no duplicate transition event.
Capability authorization happens above this layer (the drain tasks
carry a non-``gitea.*`` permission in the task capability map); the DB
records who asked and why, and never grants the right itself.
"""
state_norm = maintenance_drain.normalize_state(state)
raw = {
"reason": str(reason or ""),
"requested_by": str(requested_by or ""),
"requested_by_profile": str(requested_by_profile or ""),
"session_id": str(session_id or ""),
}
clean = gitea_audit.redact(raw)
remote_s, org_s, repo_s = str(remote or ""), str(org or ""), str(repo or "")
now_s = _ts()
with self._tx() as conn:
existing = conn.execute(
"""
SELECT * FROM maintenance_drain
WHERE remote = ? AND org = ? AND repo = ?
""",
(remote_s, org_s, repo_s),
).fetchone()
prior_state = (
maintenance_drain.normalize_state(existing["state"])
if existing is not None
else maintenance_drain.STATE_INACTIVE
)
transitioned = prior_state != state_norm
prior_entered = (
str(existing["entered_at"] or "") if existing is not None else ""
)
prior_exited = (
str(existing["exited_at"] or "") if existing is not None else ""
)
if state_norm == maintenance_drain.STATE_DRAINING:
# A re-entry keeps the original entry time (the drain never
# stopped); a fresh entry stamps now and clears the old exit.
entered_at = prior_entered if (not transitioned and prior_entered) else now_s
exited_at = ""
else:
entered_at = prior_entered
exited_at = now_s if (transitioned or not prior_exited) else prior_exited
if existing is None:
drain_id = uuid.uuid4().hex
conn.execute(
"""
INSERT INTO maintenance_drain(
drain_id, remote, org, repo, state, reason,
requested_by, requested_by_profile, session_id,
entered_at, exited_at, drain_schema_version,
created_at, updated_at
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
""",
(
drain_id, remote_s, org_s, repo_s, state_norm,
clean["reason"], clean["requested_by"],
clean["requested_by_profile"], clean["session_id"],
entered_at, exited_at,
maintenance_drain.DRAIN_SCHEMA_VERSION, now_s, now_s,
),
)
else:
drain_id = str(existing["drain_id"])
conn.execute(
"""
UPDATE maintenance_drain
SET state = ?, reason = ?, requested_by = ?,
requested_by_profile = ?, session_id = ?,
entered_at = ?, exited_at = ?,
drain_schema_version = ?, updated_at = ?
WHERE drain_id = ?
""",
(
state_norm, clean["reason"], clean["requested_by"],
clean["requested_by_profile"], clean["session_id"],
entered_at, exited_at,
maintenance_drain.DRAIN_SCHEMA_VERSION, now_s, drain_id,
),
)
if transitioned:
event_type = (
"maintenance_drain_enter"
if state_norm == maintenance_drain.STATE_DRAINING
else "maintenance_drain_exit"
)
conn.execute(
"""
INSERT INTO events(work_item_id, event_type, message, created_at)
VALUES (NULL, ?, ?, ?)
""",
(
event_type,
f"drain {drain_id} scope {remote_s}/{org_s}/{repo_s} "
f"{prior_state} -> {state_norm} by "
f"{clean['requested_by'] or '(unknown)'} "
f"({clean['requested_by_profile'] or 'no profile'}); "
f"reason: {clean['reason'] or '(none)'}",
now_s,
),
)
row = conn.execute(
"SELECT * FROM maintenance_drain WHERE drain_id = ?", (drain_id,)
).fetchone()
record = self._maintenance_drain_row(row) or {}
return {
"record": record,
"drain_id": drain_id,
"state": state_norm,
"prior_state": prior_state,
"transitioned": transitioned,
}
+45
View File
@@ -0,0 +1,45 @@
# MCP maintenance-drain mode (#659)
Graceful **maintenance drain** stops new work assignment and defers non-allowlisted
mutations so sessions can finish critical handoffs and checkpoint before a
restart. It is **not** a restart authorization: the drain *proof* and apply gate
remain #661.
## State
Per repository scope (`remote`/`org`/`repo`) in the control-plane DB table
`maintenance_drain` (schema v6):
| State | Meaning |
|-------|---------|
| `inactive` | Normal operation (also: no row) |
| `draining` | Assignment stopped; non-allowlisted mutations deferred |
Enter/exit transitions are audited as `maintenance_drain_enter` /
`maintenance_drain_exit` events.
## Tools
| Tool | Permission | Effect |
|------|------------|--------|
| `gitea_maintenance_drain_status` | `gitea.read` | Observe drain (every session) |
| `gitea_enter_maintenance_drain` | `runtime.maintenance_drain` | Enter drain (capability-gated) |
| `gitea_exit_maintenance_drain` | `runtime.maintenance_drain` | Exit drain |
`runtime.maintenance_drain` is intentionally **not** a `gitea.*` op, so ordinary
author profiles cannot enter drain by accident.
## Enforcement
1. **Allocator** (`allocate_next_work`): while draining, returns `outcome=wait`
with `reason_code=maintenance_drain_assignment_stopped` for dry-run and apply.
2. **Mutation preflight** (`verify_preflight_purity`): non-allowlisted mutation
tasks raise `MaintenanceDrainError` with a typed next action.
3. **Allowlist** (safety only): heartbeats, lease release/abandon, session
checkpoints, enter/exit drain. Reads always work.
## Restart relationship
Drain mode prepares the blast radius. Restart apply still requires a clean
`DrainProof` (#661) or authorized break-glass. Status payloads never claim
restart permission.
-12
View File
@@ -153,19 +153,7 @@ not a tool argument: a session must never be able to authorize itself.
## Related ## Related
- #630 — manual daemon killing as contaminated recovery (this contrast, enforced). - #630 — manual daemon killing as contaminated recovery (this contrast, enforced).
- #657 — restart-path inventory and daemon classification.
- #686 — manual server launch detection & fail-closed provenance gate.
- #531 / #544 — stale-runtime detection (`ps`-based); sibling failure mode. - #531 / #544 — stale-runtime detection (`ps`-based); sibling failure mode.
- #558 / `docs/mcp-daemon-import-guard.md` — why shell imports are not a repair. - #558 / `docs/mcp-daemon-import-guard.md` — why shell imports are not a repair.
- `docs/mcp-client-registration.md` — per-server registration contract. - `docs/mcp-client-registration.md` — per-server registration contract.
- `docs/mcp-namespace-health.md` — probe sources and mutation enforcement. - `docs/mcp-namespace-health.md` — probe sources and mutation enforcement.
## Sanctioned reconnect vs forbidden manual launch (#686)
In addition to manual process killing (#630), manually launching a duplicate role server from an ad hoc shell (`python3 mcp_server.py`) is forbidden and fail-closed:
- **Why manual launches are unsupported:** A terminal-launched `mcp_server.py` holds its own stdio transport; it can never bind to the IDE client's stdio pipes. It cannot restore a dropped IDE namespace, and a manual duplicate process masks stale client-managed runtimes for that profile, defeating stale-runtime gates.
- **Sanctioned path:** Supported recovery is IDE/client-managed reconnect only (`/mcp reconnect`, IDE restart, or sanctioned reconnect exposure).
- **Fail-closed enforcement (#686):** Mutating tools on a server lacking client-managed launch provenance (`GITEA_CLIENT_MANAGED=1`) refuse execution fail-closed with typed blocker `unsupported_manual_launch` and an exact next action. Unsupported `GITEA_*` env overrides (e.g. `GITEA_DUMMY`) are surfaced in diagnostics rather than silently ignored.
- **Inventory & staleness:** Staleness diagnostics ignore non-client-managed duplicates when evaluating runtime freshness and inventory duplicate processes per profile (#657, #686).
-230
View File
@@ -1,230 +0,0 @@
# Remote-MCP coupling inventory
Every place the Gitea MCP server depends on being a local, client-spawned, stdio-attached
process on the operator's machine.
- **Issue:** #930 (Remote-MCP 01), child 1 of epic #929.
- **Generated against commit:** `7bf4f1258451823a55b36d2157e74f8457165088` (`master`).
- **Anchors:** every `file:line` below resolves at the commit above and at the commit that
adds this document. This change adds one new file and edits no existing file, so no
existing line number shifts between the two.
- **Scope:** documentation only. No server behavior changes in this child.
## How to read an entry
| Field | Meaning |
| ----- | ------- |
| **Anchor** | `file:line` at the commit under review. |
| **Assumes today** | What the code takes for granted while running as a local stdio process. |
| **Observes remotely** | What the same code would actually see on a shared remote host. |
| **Class** | One of: *portable as written*, *needs a seam*, *needs a replacement*, *cannot be remote*. |
| **Owner** | Exactly one epic child (#931#939) responsible for the fix. |
Classification meanings:
- **portable as written** — the code is already transport-, host-, and principal-neutral; it
moves unchanged once its inputs are supplied by a remote-aware caller.
- **needs a seam** — the logic is correct but is wired to a hard-coded local source. It needs
an injection point, not new semantics.
- **needs a replacement** — the semantics themselves are local-only. A remote deployment
needs a differently-defined mechanism, not the same mechanism relocated.
- **cannot be remote** — the operation is inherently about the operator's own machine
(its process table, its keychain, its checkout). It must either stay local behind an
explicit boundary or be deleted from the remote surface.
---
## 1. Transport bind
The transport is bound literally, once, at process start, and the bound value is the root of
the mutation-authorization chain.
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
| -- | ------ | ------------- | ----------------- | ----- | ----- |
| T1 | `gitea_mcp_server.py:23750` | The single production bind call passes the literal `transport="stdio"` immediately before the server loop. | The literal is wrong for any non-stdio deployment; there is no parameter to change it. | needs a seam | #931 |
| T2 | `mcp_daemon_guard.py:45` | `_PRODUCTION_TRANSPORTS = frozenset({"stdio"})` is the closed allowlist of production transports. | A remote transport name is rejected by the allowlist before any other check runs. | needs a seam | #931 |
| T3 | `mcp_daemon_guard.py:174` | `bind_native_mcp_transport` raises `UnsanctionedRuntimeError` for any transport outside `_PRODUCTION_TRANSPORTS` (raise at `mcp_daemon_guard.py:187`). | The remote server fails to start rather than degrading; the failure is correct, but the allowlist is the only thing that must change. | needs a seam | #931 |
| T4 | `mcp_daemon_guard.py:328` | `is_native_mcp_transport()` asserts a process-local runtime record whose `pid` matches `os.getpid()` and whose phase is `transport_bound`. The predicate itself names no transport. | Unchanged semantics: one server process that bound one transport. It stays true on a remote host. | portable as written | #931 |
| T5 | `mcp_daemon_guard.py:349` | `is_production_native_mcp_transport()` adds only a `mode == production` check on top of T4. | Unchanged. | portable as written | #931 |
| T6 | `irrecoverable_provenance.py:497` | `assess_transport_for_auth_mint()` requires production native transport before minting non-forgeable recovery authorization (#709 F1). | The gate is transport-agnostic in form, but its guarantee — "an ordinary Python process cannot reach this" — is currently underwritten by the stdio bind. Under a remote transport the guarantee must be re-derived from the authenticated session, not from the bind. | needs a seam | #931 |
| T7 | `gitea_mcp_server.py:8375` | Consumer: refuses to proceed unless `assess_transport_for_auth_mint()` allows. | Unchanged given a corrected T6. | portable as written | #931 |
| T8 | `gitea_mcp_server.py:8624` | Second consumer of the same gate on the confirmation path. | Unchanged given a corrected T6. | portable as written | #931 |
| T9 | `mcp_server.py:4` | Module docstring asserts "Runs over stdio." as a property of the server. | The stated contract becomes false on the remote deployment and is load-bearing documentation for operators. | needs a replacement | #931 |
## 2. Launch provenance
Mutations fail closed unless the process can prove a client launched it with real stdio pipes
and `GITEA_CLIENT_MANAGED` provenance. Every proof in this section is a statement about the
local operating system.
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
| -- | ------ | ------------- | ----------------- | ----- | ----- |
| P1 | `gitea_mcp_server.py:14588` | `_is_client_managed_process()` derives provenance from `GITEA_CLIENT_MANAGED` / `GITEA_MCP_CLIENT_MANAGED` / `GITEA_SERVER_PROVENANCE` / `GITEA_FORCE_CLIENT_MANAGED` on this process's own environment. | A long-lived remote process has one environment for all callers, so a per-process env var can no longer say anything about the caller that issued a request. | needs a replacement | #934 |
| P2 | `gitea_mcp_server.py:14606` | Falls back to `sys.stdin.isatty()`: an active TTY on stdin means a human launched it from a terminal, so refuse. | A remote server has no meaningful stdin. The signal is absent, not merely different. | cannot be remote | #934 |
| P3 | `gitea_mcp_server.py:14618` | `_provenance_mutation_block()` emits `blocker_kind: "unsupported_manual_launch"` and a "reconnect the IDE/client-managed MCP namespace" remediation. | The block shape is reusable; its predicate and its remediation text are both stdio-specific. | needs a seam | #934 |
| P4 | `gitea_mcp_server.py:20599` | `_check_mcp_runtimes_diagnostics()` shells `ps -o pid,lstart,command -ax` and greps for `mcp_server.py` to find peer role servers. | On a shared host the process table lists unrelated tenants' processes, or none at all under a container. Peer discovery by `ps` has no remote meaning. | cannot be remote | #934 |
| P5 | `gitea_mcp_server.py:20702` | More than one process per `GITEA_MCP_PROFILE` in the local process table is reported as a duplicate-launch fault. | A remote endpoint is expected to serve many concurrent sessions per role. "Two processes for one role" becomes the normal case, so the check inverts from a safety net into a false wall. | cannot be remote | #934 |
| P6 | `gitea_mcp_server.py:20715` | Processes lacking client-managed provenance are ignored for runtime freshness and reported as manual launches. | Same defect as P5: correctness depends on enumerating local peers. | cannot be remote | #934 |
| P7 | `gitea_config.py:1172` | `RECOGNIZED_GITEA_ENV_KEYS` is the allowlist of `GITEA_*` env vars a legitimately launched server may carry; anything else is contamination. | Configuration on a remote host arrives from deployment tooling, not from a client-authored env block. The allowlist keeps working mechanically but stops proving anything about provenance. | needs a replacement | #934 |
| P8 | `gitea_mcp_server.py:20683` | The unsupported-env scan applies `RECOGNIZED_GITEA_ENV_KEYS` to *other* processes' environments harvested via `ps eww <pid>`. | Reading another process's environment is unavailable or prohibited across tenants, and is not exposed in this form outside macOS/BSD `ps`. | cannot be remote | #934 |
| P9 | `mcp_daemon_guard.py:126` | `mark_sanctioned_daemon()` requires the claiming stack frame's resolved absolute path to be the canonical `mcp_server.py` / `gitea_mcp_server.py` next to the guard module; basename spoofing is rejected. | Entrypoint-path identity still exists on a remote host, but it authenticates the *deployment*, not the *caller*. It must be kept and demoted from "authorizes mutations" to "authorizes the process". | needs a seam | #934 |
| P10 | `gitea_config.py:1233` | The client-config generator emits `"GITEA_CLIENT_MANAGED": "1"` into each generated MCP client entry, alongside `GITEA_MCP_CONFIG` / `GITEA_MCP_PROFILE`. | A remote endpoint is addressed by URL and credential, not by a spawn command with an env block. This generator produces the wrong artifact entirely. | needs a replacement | #938 |
| P11 | `mcp_namespace_health.py:232` | Namespace health classifies a namespace as `client_managed` or `manual_launch` from the reported env summary. | During dual-run, local and remote namespaces coexist and must both be classifiable; a two-valued local/manual axis cannot express "remote endpoint, authenticated session". | needs a replacement | #939 |
| P12 | `gitea_mcp_server.py:18161` | The diagnostics payload reports `server_provenance` as exactly `"client_managed"` or `"manual_launch"`. | This is the field a cutover operator reads to confirm which deployment served a call. It must gain a remote value before dual-run parity can be validated. | needs a replacement | #939 |
## 3. Role binding
Role separation is currently enforced by *which process a call reaches*. The process is pinned
to one role for its lifetime by an environment variable.
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
| -- | ------ | ------------- | ----------------- | ----- | ----- |
| R1 | `gitea_config.py:54` | `ENV_PROFILE = "GITEA_MCP_PROFILE"` is the single source of the active profile, read from the process environment. | One shared process serves several principals; a process-wide profile cannot answer "who is calling now". This is the root of the coupling. | needs a replacement | #932 |
| R2 | `review_workflow_load.py:95` | Reads `GITEA_MCP_PROFILE` directly to decide the reviewer workflow binding. | Reads the deployment's profile, not the caller's, silently granting or denying the wrong role. | needs a replacement | #932 |
| R3 | `mcp_discoverability.py:152` | Reads `GITEA_MCP_PROFILE` to describe the namespace to the client. | Correct logic, wrong input source; it needs the request principal injected. | needs a seam | #932 |
| R4 | `webui/deployment_boundary.py:115` | Reads `GITEA_MCP_PROFILE` to classify the deployment boundary for the console. | Same as R3. | needs a seam | #932 |
| R5 | `gitea_mcp_server.py:21106` | Remediation text instructs the operator to "Relaunch the server with `GITEA_MCP_PROFILE` set to a profile that has the required permission". | Relaunching a shared remote endpoint to change one caller's role is not a valid instruction; it would re-role every other session. | needs a replacement | #932 |
| R6 | `native_mcp_preference.py:93` | Detects shell commands that override `GITEA_MCP_PROFILE` away from the session (`native_mcp_preference.py:223`) and flags them as CLI auth divergence. | The divergence check is genuinely useful and survives, but its notion of "the session's profile" must come from the request principal. | needs a seam | #932 |
| R7 | `gitea_mcp_server.py:20671` | Recovers a peer server's role by regexing `GITEA_MCP_PROFILE=` out of that process's environment. | Depends on P4/P8 process-table access; role discovery by peer-env scraping has no remote analogue. | cannot be remote | #932 |
## 4. Credentials
Every token resolves, directly or indirectly, from one human's macOS keychain.
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
| -- | ------ | ------------- | ----------------- | ----- | ----- |
| C1 | `gitea_config.py:956` | `_keychain_token()` shells `security find-generic-password -s <item> -w`. | `security(1)` is a macOS binary reading the calling user's login keychain. It does not exist on a Linux host and would be the wrong identity even on a shared Mac. | cannot be remote | #933 |
| C2 | `gitea_config.py:974` | `resolve_token(profile, keychain_lookup=_keychain_token)` dispatches on `auth.type` of `env` or `keychain`, defaulting the lookup to C1. | The injectable `keychain_lookup` parameter is the existing seam; a remote credential provider plugs in here without changing the dispatch. | needs a seam | #933 |
| C3 | `gitea_config.py:1015` | `keychain_auth(item_id)` constructs the `{"type": "keychain", "id": ...}` reference stored in profiles. | The reference type itself encodes "macOS keychain" into persisted config. A remote provider needs a new auth reference type, not a new value of this one. | needs a replacement | #933 |
| C4 | `mcp_daemon_guard.py:440` | `assert_keychain_access_allowed()` fails closed for git-credential keychain fill outside a sanctioned daemon, with an operator opt-out env var. | The gate protects a mechanism that will not exist remotely. Its replacement must gate the *credential provider* call, not the keychain call, or the protection silently lapses. | needs a replacement | #933 |
| C5 | `sentry_incident_bridge.py:190` | `resolve_token(env)` resolves the Sentry token from an injected env mapping with no keychain path. | Already host-neutral; it is the shape the Gitea credential path should converge on. | portable as written | #933 |
| C6 | `gitea_mcp_server.py:18469` | The profile-audit tool calls `gitea_config.resolve_token(p)` for every configured profile to report "credentials present" without networking. | On a remote host this would materialize every principal's credential inside one process — an audit surface that becomes a credential-aggregation risk. | needs a seam | #933 |
## 5. Runtime freshness
The mutation gate is defined as "the commit this process started at matches the checkout on
this disk, and both match live master". Two of those three terms are local-disk facts.
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
| -- | ------ | ------------- | ----------------- | ----- | ----- |
| F1 | `master_parity_gate.py:168` | `capture_startup_parity(root)` reads git `HEAD` from the server's own root once at startup and returns it as the baseline. | A remote host carries a deployed artifact, not the operator's checkout. Its `HEAD` says nothing about the operator's working tree, which is the thing the gate exists to protect. | cannot be remote | #935 |
| F2 | `master_parity_gate.py:255` | `mutation_safe = determinable and in_parity and live_known and not live_stale` — a conjunction of two local-HEAD comparisons and one live-remote comparison. | Two of the three conjuncts lose meaning, so the whole verdict does. A remote deployment needs a redefined, testable freshness predicate rather than this one relocated. | needs a replacement | #935 |
| F3 | `master_parity_gate.py:164` | The live-remote head is probed and cached per `(root, remote, branch)`, keyed on the local root. | The live-remote probe is the one conjunct that survives; it needs a key that is not the operator's filesystem path. | needs a seam | #935 |
| F4 | `gitea_mcp_server.py:18262` | `gitea_assess_master_parity` publishes `startup_head` / `local_head` / `live_remote_head` / `mutation_safe` as the authoritative mutation-safety verdict. | The tool's contract is consumed by every mutation caller and by the operator; it must keep its shape while its semantics are redefined, or every consumer breaks at once. | needs a replacement | #935 |
| F5 | `gitea_mcp_server.py:23054` | Falls back to `_process_boot_head_sha` — the commit this process booted at — when the parity payload has no `startup_head`. | Same defect as F1, in a fallback path that is easy to miss when F1 is fixed. | needs a seam | #935 |
| F6 | `gitea_mcp_server.py:20615` | Staleness is also inferred from `os.path.getmtime()` of `gitea_mcp_server.py` under `PROJECT_ROOT` (`gitea_mcp_server.py:20611`), compared against peer process start times. | File mtime on a deployed artifact tracks the deploy, not the operator's edits, and the peer start times it is compared against come from the unavailable process table (P4). | cannot be remote | #935 |
## 6. Local filesystem
Author and reviewer tools act directly on the operator's checkout.
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
| -- | ------ | ------------- | ----------------- | ----- | ----- |
| L1 | `gitea_mcp_server.py:10122` | `gitea_bootstrap_author_issue_worktree` creates and binds a git worktree on the server's own disk. | The remote host has no operator checkout to add a worktree to. Executing this remotely would act on the wrong disk while reporting success. | cannot be remote | #936 |
| L2 | `gitea_mcp_server.py:190` | `ACTIVE_WORKTREE_ENV = "GITEA_ACTIVE_WORKTREE"` and `AUTHOR_WORKTREE_ENV` (`gitea_mcp_server.py:191`) carry the active workspace as process-wide environment. | Process-wide workspace state cannot represent per-session workspaces on a shared endpoint. | needs a replacement | #936 |
| L3 | `gitea_mcp_server.py:9801` | Binding a worktree writes `os.environ["GITEA_AUTHOR_WORKTREE"]` and `os.environ["GITEA_ACTIVE_WORKTREE"]` (`gitea_mcp_server.py:9802`), mutating global process state. | One session's bind would silently retarget every other concurrent session in the same process. This is a correctness bug the moment concurrency is real. | needs a replacement | #936 |
| L4 | `reviewer_inventory_worktree.py:48` | `_BRANCHES_WORKTREE_RE = re.compile(r"\bbranches/", re.I)` requires review worktree paths to sit under `branches/`. | A path convention on the operator's machine, asserted as a validation rule. It needs to become a property of a declared workspace, not a substring test. | needs a seam | #936 |
| L5 | `stable_control_runtime.py:54` | `DEV_WORKTREE_SEGMENT = "branches"` classifies a process root as a development worktree by path segment. | Same class of assumption as L4, on the runtime-classification side. | needs a seam | #936 |
| L6 | `mcp_server.py:42` | `check_conflict_markers()` runs at import and `os.walk`s the install directory for unresolved conflict markers, `sys.exit(1)` on a hit. | On a remote host it scans a deployed artifact, which by construction never has conflict markers — so the guard passes trivially and stops protecting the thing it was written to protect. | needs a replacement | #936 |
| L7 | `role_session_router.py:487` | `check_mid_merge()` reports infra-stop from `.git/MERGE_HEAD`, `rebase-merge`, `rebase-apply` and a source conflict scan under the server's project root. | Same inversion as L6: it would report the deployment's git state, not the operator's. | needs a replacement | #936 |
| L8 | `author_issue_bootstrap.py:996` | Enumerates worktrees with `git -C <root> worktree list --porcelain`. | Requires a real local clone with real worktrees; there is nothing equivalent to enumerate remotely. | cannot be remote | #936 |
| L9 | `mcp_server.py:10` | Redirects `sys.stderr` to the fixed path `/tmp/mcp_server_stderr.log` outside pytest. | A single fixed `/tmp` path is shared by every concurrent server on a host and is not a deployment's logging surface. | needs a replacement | #938 |
| L10 | `gitea_mcp_server.py:2314` | `ISSUE_LOCK_FILE = "/tmp/gitea_issue_lock.json"` — the legacy single global lock slot. | One global `/tmp` slot per host cannot represent concurrent remote sessions and is world-visible on a shared machine. | needs a replacement | #937 |
| L11 | `issue_lock_provenance.py:14` | `ISSUE_LOCK_FILE = os.environ.get("GITEA_ISSUE_LOCK_FILE", "/tmp/gitea_issue_lock.json")` keeps the same `/tmp` default in the provenance path. | Same as L10; the env override is a local escape hatch, not a remote design. | needs a replacement | #937 |
## 7. Durable state
Locks, leases, session state, and the control-plane database live in the operator's home
directory and are keyed on local PIDs.
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
| -- | ------ | ------------- | ----------------- | ----- | ----- |
| S1 | `issue_lock_store.py:26` | `DEFAULT_LOCK_DIR = ~/.cache/gitea-tools/issue-locks` — per-issue lock files under one user's home. | A shared endpoint has no single operator home; per-user paths make locks invisible across sessions and hosts. | needs a replacement | #937 |
| S2 | `issue_lock_store.py:83` | `session_pointer_path()` names the session pointer file `session-<os.getpid()>.json`. | Many sessions share one PID on a remote server, so the pointer collapses to a single slot and sessions overwrite each other. | cannot be remote | #937 |
| S3 | `issue_lock_store.py:98` | `is_process_alive(pid)` decides lock liveness by probing the local process table. | A PID recorded by one host is meaningless on another, and may coincidentally match a live unrelated process. | cannot be remote | #937 |
| S4 | `issue_lock_store.py:213` | Lock records stamp `session_pid` and `pid` from `os.getpid()`. | The recorded identity no longer distinguishes sessions; ownership checks silently pass for the wrong caller. | needs a replacement | #937 |
| S5 | `mcp_session_state.py:27` | `DEFAULT_STATE_DIR = ~/.cache/gitea-tools/session-state`, mode `0o700`. | Same home-directory coupling as S1, for review decision locks and workflow proofs. | needs a replacement | #937 |
| S6 | `mcp_session_state.py:559` | Session bodies stamp `session_pid` and `writer_pid` from `os.getpid()` (`mcp_session_state.py:560`). | Writer attribution collapses across concurrent sessions in one process. | needs a replacement | #937 |
| S7 | `control_plane_db.py:47` | `DEFAULT_DB_PATH = ~/.cache/gitea-tools/control-plane/control_plane.sqlite3`. | A per-user SQLite file is not reachable by, or safe for, multiple remote sessions or multiple hosts. | needs a replacement | #937 |
| S8 | `control_plane_db.py:386` | `sqlite3.connect(self.db_path, timeout=30)` — single-writer file locking tuned for one local process. | SQLite's write lock does not extend across hosts and degrades sharply under real concurrency; the store needs a concurrency-safe backend. | needs a replacement | #937 |
| S9 | `control_plane_db.py:1145` | Lease rows record `owner_pid` defaulting to `os.getpid()` (also `control_plane_db.py:2039`). | PID-keyed lease ownership is unusable across hosts and ambiguous within one shared process. | cannot be remote | #937 |
| S10 | `mcp_daemon_guard.py:53` | `_DEFAULT_SESSION_STATE_DIR` is pinned once at transport bind so a later `GITEA_MCP_SESSION_STATE_DIR` change cannot manufacture a second authority domain (#695 AC2). | The single-authority-domain invariant is exactly right and must be preserved; only its backing location needs to move. | needs a seam | #937 |
| S11 | `gitea_mcp_server.py:11875` | Reviewer-lease reclaim reads `owner_pid_alive` from the lease freshness record to decide whether an owner is dead. | Consumes S3/S9; a false "owner alive" or "owner dead" here reclaims or refuses a live lease. This is the highest-consequence consumer of PID liveness. | cannot be remote | #937 |
---
## Summary
### Entries per category
| Category | Entries |
| -------- | ------: |
| 1. Transport bind | 9 |
| 2. Launch provenance | 12 |
| 3. Role binding | 7 |
| 4. Credentials | 6 |
| 5. Runtime freshness | 6 |
| 6. Local filesystem | 11 |
| 7. Durable state | 11 |
| **Total** | **62** |
No category is empty, so no "this category has no coupling" justification is required.
### Entries per classification
| Classification | Entries |
| -------------- | ------: |
| portable as written | 5 |
| needs a seam | 16 |
| needs a replacement | 26 |
| cannot be remote | 15 |
| **Total** | **62** |
### Category × classification
| Category | portable | seam | replacement | cannot | Total |
| -------- | -------: | ---: | ----------: | -----: | ----: |
| 1. Transport bind | 4 | 4 | 1 | 0 | 9 |
| 2. Launch provenance | 0 | 2 | 5 | 5 | 12 |
| 3. Role binding | 0 | 3 | 3 | 1 | 7 |
| 4. Credentials | 1 | 2 | 2 | 1 | 6 |
| 5. Runtime freshness | 0 | 2 | 2 | 2 | 6 |
| 6. Local filesystem | 0 | 2 | 7 | 2 | 11 |
| 7. Durable state | 0 | 1 | 6 | 4 | 11 |
| **Total** | **5** | **16** | **26** | **15** | **62** |
### Entries per epic child
Every child from 2 through 10 is named by at least one entry, and every entry names exactly
one child.
| Child | Issue | Title | Entries | IDs |
| ----: | ----- | ----- | ------: | --- |
| 2 | #931 | Transport-neutral bind seam | 9 | T1T9 |
| 3 | #932 | Per-request principal resolution | 7 | R1R7 |
| 4 | #933 | Server-side credential provider | 6 | C1C6 |
| 5 | #934 | Remote-session provenance | 9 | P1P9 |
| 6 | #935 | Redefined master-parity gate | 6 | F1F6 |
| 7 | #936 | Local-filesystem vs remotable tool split | 8 | L1L8 |
| 8 | #937 | Concurrency-safe session, lock, and lease state | 13 | L10, L11, S1S11 |
| 9 | #938 | Authenticated remote MCP endpoint | 2 | P10, L9 |
| 10 | #939 | Dual-run cutover and rollback | 2 | P11, P12 |
| | | **Total** | **62** | |
## Notes for downstream children
- **The three highest-risk entries are P5, F2, and S11.** Each is a guard that does not
merely stop working remotely — it inverts. P5 turns concurrency into a reported fault,
F2 returns a verdict computed from terms that no longer mean anything, and S11 reclaims
or refuses leases on a PID-liveness answer that is wrong rather than unknown. A gate that
fails open while still reporting green is worse than one that fails to start.
- **T4, T5, T7, T8, and C5 are the portable core.** They show the target shape: predicates
over injected inputs, with no reference to the host, the process table, or the operator's
disk.
- **The keychain seam already exists** at C2 (`resolve_token`'s injectable `keychain_lookup`).
#933 should widen that seam rather than introduce a parallel path, and must remember C4 —
the guard protecting the old mechanism has to be re-pointed, or the protection lapses
silently when the mechanism is replaced.
- **`branches/` appears as a validation rule in at least two independent places** (L4, L5).
Path-substring conventions tend to have more copies than expected; #936 should re-grep
rather than trust this list to be exhaustive for that specific pattern.
+1 -50
View File
@@ -1169,57 +1169,10 @@ def server_command():
return python, [os.path.join(root, "mcp_server.py")] return python, [os.path.join(root, "mcp_server.py")]
RECOGNIZED_GITEA_ENV_KEYS = frozenset({
"GITEA_MCP_CONFIG",
"GITEA_MCP_PROFILE",
"GITEA_PROFILE_NAME",
"GITEA_SERVICE",
"GITEA_EXECUTION_ROLE",
"GITEA_CLIENT_MANAGED",
"GITEA_MCP_CLIENT_MANAGED",
"GITEA_SERVER_PROVENANCE",
"GITEA_AUTHOR_WORKTREE",
"GITEA_ACTIVE_WORKTREE",
"GITEA_DISABLE_KEYCHAIN",
"GITEA_CONTROL_PLANE_DB",
"GITEA_DB_PATH",
"GITEA_LOG_LEVEL",
"GITEA_DEBUG",
"GITEA_HMAC_SECRET",
"GITEA_IRRECOVERABLE_HMAC_SECRET",
"GITEA_FORCE_MCP_RUNTIME_CHECK",
"GITEA_FORCE_CLIENT_MANAGED",
})
RECOGNIZED_GITEA_ENV_PREFIXES = (
"GITEA_TOKEN_",
"GITEA_PASS_",
"GITEA_USER_",
"GITEA_URL_",
"GITEA_HOST_",
"GITEA_REMOTE_",
"GITEA_HTTP_HEADER_",
)
def get_unconsumed_gitea_env_overrides(env=None) -> dict[str, str]:
"""Find unsupported GITEA_* env vars present in *env* (defaults to os.environ)."""
target = os.environ if env is None else env
unconsumed = {}
for key, value in target.items():
if key.startswith("GITEA_"):
if key in RECOGNIZED_GITEA_ENV_KEYS:
continue
if any(key.startswith(p) for p in RECOGNIZED_GITEA_ENV_PREFIXES):
continue
unconsumed[key] = str(value)
return unconsumed
def launcher_entry(profile_name, config_path=None): def launcher_entry(profile_name, config_path=None):
"""Return a thin MCP launcher entry for *profile_name*. """Return a thin MCP launcher entry for *profile_name*.
Contains command/args and the GITEA_MCP_* / GITEA_CLIENT_MANAGED env vars — never a token Contains only command/args and the two GITEA_MCP_* env vars — never a token
or password. Suitable for Claude / Gemini / Codex ``mcpServers`` blocks. or password. Suitable for Claude / Gemini / Codex ``mcpServers`` blocks.
""" """
command, args = server_command() command, args = server_command()
@@ -1230,13 +1183,11 @@ def launcher_entry(profile_name, config_path=None):
"env": { "env": {
"GITEA_MCP_CONFIG": config_path or DEFAULT_CONFIG_PATH, "GITEA_MCP_CONFIG": config_path or DEFAULT_CONFIG_PATH,
"GITEA_MCP_PROFILE": profile_name, "GITEA_MCP_PROFILE": profile_name,
"GITEA_CLIENT_MANAGED": "1",
}, },
} }
} }
def keychain_set(item_id, token, account=None, runner=subprocess.run): def keychain_set(item_id, token, account=None, runner=subprocess.run):
"""Store *token* in the macOS keychain under service *item_id*. """Store *token* in the macOS keychain under service *item_id*.
+244 -117
View File
@@ -1472,6 +1472,9 @@ def verify_preflight_purity(
# contaminated by manual MCP daemon process killing (reconciler-exempt). # contaminated by manual MCP daemon process killing (reconciler-exempt).
_enforce_runtime_recovery_contamination_gate(task, remote) _enforce_runtime_recovery_contamination_gate(task, remote)
# #659 AC3: defer non-allowlisted mutations while maintenance drain is active.
_enforce_maintenance_drain_gate(task, remote=remote, org=org, repo=repo)
ctx = _resolve_namespace_mutation_context(worktree_path) ctx = _resolve_namespace_mutation_context(worktree_path)
workspace = ctx["workspace_path"] workspace = ctx["workspace_path"]
canonical_root = ctx["canonical_repo_root"] canonical_root = ctx["canonical_repo_root"]
@@ -2004,6 +2007,55 @@ def _enforce_runtime_recovery_contamination_gate(
) )
def _enforce_maintenance_drain_gate(
task: str | None,
remote: str | None = None,
org: str | None = None,
repo: str | None = None,
) -> None:
"""#659 AC3: defer non-allowlisted mutations while drain is active.
The single mutation chokepoint already used by every gated task, so drain
coverage cannot drift per-tool. Allowlisted safety operations (heartbeat,
release/abandon, checkpoint, drain exit) pass through so an in-flight
session can still finish and hand off; everything else is deferred with a
typed blocker. Unreadable drain state fails closed a drain that cannot be
read is not evidence that no drain is running.
"""
if _preflight_in_test_mode() and not os.environ.get(
"GITEA_TEST_FORCE_MAINTENANCE_DRAIN"
):
return
if maintenance_drain.is_allowlisted_task(task):
return
try:
_h, o, r = _resolve(remote, None, org, repo)
except Exception: # noqa: BLE001 — scope resolution is best-effort here
o, r = (org or ""), (repo or "")
db, errs = _control_plane_db_or_error()
if db is None:
raise RuntimeError(
"maintenance-drain state could not be read: "
f"{'; '.join(errs) or 'control-plane DB unavailable'} (fail closed, #659)"
)
try:
record = db.read_maintenance_drain(remote=remote or "", org=o, repo=r)
except Exception as exc: # noqa: BLE001
raise RuntimeError(
f"maintenance-drain state could not be read: {_redact(str(exc))} "
"(fail closed, #659)"
) from exc
decision = maintenance_drain.classify_mutation(task, record)
if not decision["allowed"]:
raise maintenance_drain.MaintenanceDrainError(
maintenance_drain.format_drain_block_error(decision),
decision=decision,
)
def _enforce_stable_branch_contamination_gate( def _enforce_stable_branch_contamination_gate(
task: str | None, task: str | None,
remote: str | None = None, remote: str | None = None,
@@ -2064,6 +2116,7 @@ import allocator_service # noqa: E402
import allocator_dependencies # noqa: E402 import allocator_dependencies # noqa: E402
import dependency_graph # noqa: E402 # #784 durable dependency edges import dependency_graph # noqa: E402 # #784 durable dependency edges
import control_plane_db # noqa: E402 import control_plane_db # noqa: E402
import maintenance_drain # noqa: E402 # #659 graceful maintenance-drain mode
import lease_lifecycle # noqa: E402 import lease_lifecycle # noqa: E402
import lease_policy # noqa: E402 import lease_policy # noqa: E402
import workflow_dashboard # noqa: E402 # #605 live queue/lease dashboard import workflow_dashboard # noqa: E402 # #605 live queue/lease dashboard
@@ -14585,56 +14638,6 @@ def _session_context_mutation_block(
return blocked return blocked
def _is_client_managed_process() -> bool:
"""Check whether the current MCP server process has client-managed launch provenance (#686)."""
val = (
os.environ.get("GITEA_CLIENT_MANAGED")
or os.environ.get("GITEA_MCP_CLIENT_MANAGED")
or os.environ.get("GITEA_SERVER_PROVENANCE")
or os.environ.get("GITEA_FORCE_CLIENT_MANAGED")
or ""
).strip().lower()
if val in ("0", "false", "no", "manual", "manual_launch"):
return False
if val in ("1", "true", "yes", "client_managed"):
return True
# A terminal launch has an active TTY on stdin
try:
if sys.stdin and sys.stdin.isatty():
return False
except Exception:
pass
# Standard client launch or test runner with stdio pipe and profile env
if "GITEA_MCP_CONFIG" in os.environ or "GITEA_MCP_PROFILE" in os.environ or "GITEA_PROFILE_NAME" in os.environ:
return True
return False
def _provenance_mutation_block(**extra_fields) -> dict | None:
"""Refuse mutating tool calls on processes lacking client-managed launch provenance (#686)."""
if _is_client_managed_process():
return None
unconsumed = gitea_config.get_unconsumed_gitea_env_overrides()
blocked = {
"success": False,
"performed": False,
"blocker_kind": "unsupported_manual_launch",
"reasons": [
"mutation denied: server process was launched manually from a terminal without client-managed provenance (fail closed). Manually launched mcp_server.py processes cannot receive IDE stdio or serve workflow mutations."
],
"exact_next_action": "BLOCKED + RECONNECT: Reconnect the IDE/client-managed MCP server namespace instead of an ad hoc terminal launch. Hand-launched processes and mcp_config.json hand-edits are classified as workflow contamination.",
"provenance": "manual_launch",
"unconsumed_gitea_env": unconsumed,
}
blocked.update(extra_fields)
return blocked
def _profile_permission_block(required_operation: str, **extra_fields) -> dict | None: def _profile_permission_block(required_operation: str, **extra_fields) -> dict | None:
"""Structured operation-gate denial for gated tools (#69, #142, #897). """Structured operation-gate denial for gated tools (#69, #142, #897).
@@ -14651,10 +14654,6 @@ def _profile_permission_block(required_operation: str, **extra_fields) -> dict |
# #714: evaluate active profile only — never auto-switch. # #714: evaluate active profile only — never auto-switch.
_ensure_matching_profile(required_operation, req_role, extra_fields.get("remote")) _ensure_matching_profile(required_operation, req_role, extra_fields.get("remote"))
prov_block = _provenance_mutation_block(**extra_fields)
if prov_block is not None:
return prov_block
reasons = _profile_operation_gate(required_operation) reasons = _profile_operation_gate(required_operation)
if reasons: if reasons:
return _build_operation_gate_refusal( return _build_operation_gate_refusal(
@@ -14688,10 +14687,6 @@ def _namespace_mutation_block(mutation_task: str, **extra_fields) -> dict | None
# #714: evaluate active profile only — never auto-switch. # #714: evaluate active profile only — never auto-switch.
_ensure_matching_profile(required_permission, required_role, extra_fields.get("remote")) _ensure_matching_profile(required_permission, required_role, extra_fields.get("remote"))
prov_block = _provenance_mutation_block(**extra_fields)
if prov_block is not None:
return prov_block
try: try:
profile = get_profile() profile = get_profile()
except Exception as exc: except Exception as exc:
@@ -18139,9 +18134,6 @@ def gitea_get_runtime_context(
source="gitea_get_runtime_context", source="gitea_get_runtime_context",
) )
is_client_managed = _is_client_managed_process()
unconsumed_env = gitea_config.get_unconsumed_gitea_env_overrides()
result = { result = {
"active_profile": profile["profile_name"], "active_profile": profile["profile_name"],
"authenticated_username": username, "authenticated_username": username,
@@ -18158,9 +18150,6 @@ def gitea_get_runtime_context(
"review_merge_blocked_reasons": blocked_reasons, "review_merge_blocked_reasons": blocked_reasons,
"suggested_fix": suggested_fix, "suggested_fix": suggested_fix,
"safe_next_action": safe_next_action, "safe_next_action": safe_next_action,
"server_provenance": "client_managed" if is_client_managed else "manual_launch",
"is_client_managed": is_client_managed,
"unconsumed_gitea_env": unconsumed_env,
"preflight_ready": preflight["preflight_ready"], "preflight_ready": preflight["preflight_ready"],
"preflight_block_reasons": preflight["preflight_block_reasons"], "preflight_block_reasons": preflight["preflight_block_reasons"],
"preflight_workspace": preflight.get("preflight_workspace"), "preflight_workspace": preflight.get("preflight_workspace"),
@@ -18174,13 +18163,6 @@ def gitea_get_runtime_context(
PROJECT_ROOT), PROJECT_ROOT),
} }
if not is_client_managed:
result["safe_next_action"] = (
"BLOCKED + RECONNECT: Serving process lacks client-managed launch provenance (manual launch). "
"Reconnect the IDE/client-managed MCP server namespace instead of an ad hoc terminal launch."
)
# #702: read-only visibility into the inherited GITEA_ACTIVE_WORKTREE # #702: read-only visibility into the inherited GITEA_ACTIVE_WORKTREE
# binding; recovery itself runs during capability resolution. # binding; recovery itself runs during capability resolution.
try: try:
@@ -20638,9 +20620,7 @@ def _check_mcp_runtimes_diagnostics(task: str, matching_profiles: list[str]) ->
self_pid = os.getpid() self_pid = os.getpid()
self_stale = False self_stale = False
all_profile_procs: dict[str, list[dict]] = {} running_profiles = {}
unsupported_env_found = set()
for line in proc.stdout.splitlines()[1:]: for line in proc.stdout.splitlines()[1:]:
line = line.strip() line = line.strip()
if not line or "mcp_server.py" not in line: if not line or "mcp_server.py" not in line:
@@ -20672,55 +20652,16 @@ def _check_mcp_runtimes_diagnostics(task: str, matching_profiles: list[str]) ->
if match: if match:
profile = match.group(1) profile = match.group(1)
is_client_managed = bool(
re.search(r'\bGITEA_CLIENT_MANAGED=(1|true|yes|client_managed)\b', env_out, re.IGNORECASE)
or re.search(r'\bGITEA_MCP_CLIENT_MANAGED=(1|true|yes|client_managed)\b', env_out, re.IGNORECASE)
or re.search(r'\bGITEA_SERVER_PROVENANCE=client_managed\b', env_out, re.IGNORECASE)
)
for env_match in re.finditer(r'\b(GITEA_[A-Z0-9_]+)=([^\s]+)', env_out):
k, v = env_match.group(1), env_match.group(2)
if k not in gitea_config.RECOGNIZED_GITEA_ENV_KEYS and not any(k.startswith(p) for p in gitea_config.RECOGNIZED_GITEA_ENV_PREFIXES):
unsupported_env_found.add(f"{k}={v}")
is_stale = (start_time < code_mtime) or git_stale is_stale = (start_time < code_mtime) or git_stale
if pid == self_pid and is_stale: if pid == self_pid and is_stale:
self_stale = True self_stale = True
proc_info = { if profile not in running_profiles or start_time > running_profiles[profile]["start_time"]:
running_profiles[profile] = {
"pid": pid, "pid": pid,
"start_time": start_time, "start_time": start_time,
"is_stale": is_stale, "is_stale": is_stale
"is_client_managed": is_client_managed,
} }
if profile not in all_profile_procs:
all_profile_procs[profile] = []
all_profile_procs[profile].append(proc_info)
running_profiles = {}
for profile, procs in all_profile_procs.items():
if len(procs) > 1:
pids_str = ", ".join(str(p["pid"]) for p in procs)
reasons.append(
f"stale-runtime: Duplicate MCP server process(es) detected for profile '{profile}' (PIDs: {pids_str}). "
"Manual or duplicate launches defeat staleness detection and cannot receive client stdio."
)
client_procs = [p for p in procs if p["is_client_managed"]]
if client_procs:
client_procs.sort(key=lambda p: p["start_time"], reverse=True)
running_profiles[profile] = client_procs[0]
else:
pids_str = ", ".join(str(p["pid"]) for p in procs)
reasons.append(
f"stale-runtime: Manually launched MCP process(es) detected without client-managed provenance for profile '{profile}' (PIDs: {pids_str}). "
"Manual launches cannot serve client stdio and are ignored for runtime freshness."
)
if unsupported_env_found:
reasons.append(
f"unsupported-env: Unsupported GITEA_* environment variable override(s) detected: {', '.join(sorted(unsupported_env_found))}. "
"Unknown env overrides are unsupported."
)
if self_stale: if self_stale:
# #685: report-only — no config utime, no thread, no os._exit. # #685: report-only — no config utime, no thread, no os._exit.
@@ -20755,7 +20696,6 @@ def _check_mcp_runtimes_diagnostics(task: str, matching_profiles: list[str]) ->
return reasons return reasons
@mcp.tool() @mcp.tool()
def gitea_resolve_task_capability( def gitea_resolve_task_capability(
task: str, task: str,
@@ -22672,6 +22612,193 @@ def gitea_workflow_dashboard(
return payload return payload
@mcp.tool()
def gitea_maintenance_drain_status(
remote: str = "dadeschools",
host: str | None = None,
org: str | None = None,
repo: str | None = None,
) -> dict:
"""Read-only: current maintenance-drain state for a repository scope (#659 AC4).
Every session must be able to observe drain so it can stop creating new work
and finish only allowlisted safety operations. Never mutates; never restarts.
"""
read_block = _profile_operation_gate("gitea.read")
if read_block:
return {
"success": False,
"read_only": True,
"reasons": read_block,
"permission_report": _permission_block_report("gitea.read"),
}
try:
_h, o, r = _resolve(remote, host, org, repo)
except ValueError as exc:
return {"success": False, "read_only": True, "reasons": [str(exc)]}
db, errs = _control_plane_db_or_error()
if db is None:
return {
"success": False,
"read_only": True,
"reasons": errs or ["control-plane DB unavailable"],
"maintenance_drain": maintenance_drain.status_payload(
None, remote=remote, org=o, repo=r
),
}
try:
record = db.read_maintenance_drain(remote=remote, org=o, repo=r)
except Exception as exc: # noqa: BLE001
return {
"success": False,
"read_only": True,
"reasons": [f"drain state unreadable: {_redact(str(exc))}"],
}
payload = maintenance_drain.status_payload(
record, remote=remote, org=o, repo=r
)
return {"success": True, "read_only": True, "maintenance_drain": payload}
@mcp.tool()
def gitea_enter_maintenance_drain(
reason: str = "",
remote: str = "dadeschools",
host: str | None = None,
org: str | None = None,
repo: str | None = None,
session_id: str | None = None,
) -> dict:
"""Enter graceful maintenance-drain mode for a repository scope (#659 AC1).
Stops new assignment and defers non-allowlisted mutations until exit. Requires
``runtime.maintenance_drain`` (controller/lifecycle capability not granted
by ordinary Gitea author profiles). Audited in the control-plane event log.
"""
cap_block = _profile_operation_gate("runtime.maintenance_drain")
if cap_block:
return {
"success": False,
"performed": False,
"reasons": cap_block,
"permission_report": _permission_block_report(
"runtime.maintenance_drain"
),
}
try:
_h, o, r = _resolve(remote, host, org, repo)
except ValueError as exc:
return {"success": False, "performed": False, "reasons": [str(exc)]}
db, errs = _control_plane_db_or_error()
if db is None:
return {
"success": False,
"performed": False,
"reasons": errs or ["control-plane DB unavailable"],
}
profile = get_profile() or {}
try:
result = db.set_maintenance_drain(
remote=remote,
org=o,
repo=r,
state=maintenance_drain.STATE_DRAINING,
reason=reason or "operator-entered maintenance drain",
requested_by=str(
(profile.get("identity") or {}).get("username")
or profile.get("expected_username")
or ""
),
requested_by_profile=str(profile.get("profile_name") or ""),
session_id=str(session_id or ""),
)
except Exception as exc: # noqa: BLE001
return {
"success": False,
"performed": False,
"reasons": [f"enter drain failed: {_redact(str(exc))}"],
}
record = result.get("record") or {}
return {
"success": True,
"performed": True,
"transitioned": bool(result.get("transitioned")),
"state": result.get("state"),
"prior_state": result.get("prior_state"),
"drain_id": result.get("drain_id"),
"maintenance_drain": maintenance_drain.status_payload(
record, remote=remote, org=o, repo=r
),
}
@mcp.tool()
def gitea_exit_maintenance_drain(
reason: str = "",
remote: str = "dadeschools",
host: str | None = None,
org: str | None = None,
repo: str | None = None,
session_id: str | None = None,
) -> dict:
"""Exit graceful maintenance-drain mode (#659 AC1). Restores assignment and mutations."""
cap_block = _profile_operation_gate("runtime.maintenance_drain")
if cap_block:
return {
"success": False,
"performed": False,
"reasons": cap_block,
"permission_report": _permission_block_report(
"runtime.maintenance_drain"
),
}
try:
_h, o, r = _resolve(remote, host, org, repo)
except ValueError as exc:
return {"success": False, "performed": False, "reasons": [str(exc)]}
db, errs = _control_plane_db_or_error()
if db is None:
return {
"success": False,
"performed": False,
"reasons": errs or ["control-plane DB unavailable"],
}
profile = get_profile() or {}
try:
result = db.set_maintenance_drain(
remote=remote,
org=o,
repo=r,
state=maintenance_drain.STATE_INACTIVE,
reason=reason or "operator-exited maintenance drain",
requested_by=str(
(profile.get("identity") or {}).get("username")
or profile.get("expected_username")
or ""
),
requested_by_profile=str(profile.get("profile_name") or ""),
session_id=str(session_id or ""),
)
except Exception as exc: # noqa: BLE001
return {
"success": False,
"performed": False,
"reasons": [f"exit drain failed: {_redact(str(exc))}"],
}
record = result.get("record") or {}
return {
"success": True,
"performed": True,
"transitioned": bool(result.get("transitioned")),
"state": result.get("state"),
"prior_state": result.get("prior_state"),
"drain_id": result.get("drain_id"),
"maintenance_drain": maintenance_drain.status_payload(
record, remote=remote, org=o, repo=r
),
}
@mcp.tool() @mcp.tool()
def gitea_request_mcp_restart( def gitea_request_mcp_restart(
remote: str = "dadeschools", remote: str = "dadeschools",
+281
View File
@@ -0,0 +1,281 @@
"""Graceful MCP maintenance-drain mode (#659).
Drain is the visible, capability-gated state that lets an operator stop new
work and quiesce mutations *before* a restart, instead of cutting sessions off
mid-mutation. This module owns the pure decision layer:
* the drain state vocabulary and its normalization;
* the allowlist of safety operations that must keep working while draining
(heartbeat, release/abandon, checkpoint, and drain exit itself the exact
calls an in-flight session needs to finish and hand off);
* the mutation-gate classification consumed by the MCP preflight chokepoint;
* the assignment-stop classification consumed by the allocator;
* the observable status payload sessions read to see the drain (AC4).
Durable state lives in the control-plane DB (``maintenance_drain`` table);
enforcement lives at the existing chokepoints. Nothing here performs I/O, so
both callers can share one decision without importing each other.
Scope note: the machine-verifiable *drain proof* and the restart gate that
consumes it are #661's scope, not this module's. Drain here stops assignment
and mutation and makes the state observable; it never authorizes a restart.
"""
from __future__ import annotations
from typing import Any, Mapping
# ── State vocabulary ──────────────────────────────────────────────────────────
STATE_INACTIVE = "inactive"
STATE_DRAINING = "draining"
DRAIN_STATES = frozenset({STATE_INACTIVE, STATE_DRAINING})
# Typed blocker code surfaced to clients (never a bare string at call sites).
BLOCKER_DRAIN_ACTIVE = "maintenance_drain_active"
# Reason code for the allocator's assignment stop.
REASON_ASSIGNMENT_STOPPED = "maintenance_drain_assignment_stopped"
DRAIN_SCHEMA_VERSION = 6
class MaintenanceDrainError(RuntimeError):
"""Raised when a mutation is refused because drain is active (fail closed)."""
def __init__(self, message: str, *, decision: Mapping[str, Any] | None = None):
super().__init__(message)
self.decision = dict(decision or {})
self.reason_code = BLOCKER_DRAIN_ACTIVE
# ── Safety allowlist ──────────────────────────────────────────────────────────
# Mutations that stay permitted while draining. Every entry is a *quiesce*
# operation: it either proves an in-flight task is still alive, hands its claim
# back, records the durable state a restart needs, or ends the drain. Nothing
# that creates new work, new branches, new PRs, or new review/merge verdicts is
# on this list — that is the whole point of the drain.
ALLOWLISTED_DRAIN_TASKS: frozenset[str] = frozenset(
{
# Liveness of work already in flight.
"heartbeat_issue_lock",
"heartbeat_reviewer_pr_lease",
"post_heartbeat",
# Handing claims back so nothing is stranded across the restart.
"release_workflow_lease",
"release_reviewer_pr_lease",
"release_merger_pr_lease",
"abandon_workflow_lease",
# Durable recovery state (#660) must be writable *during* drain.
"write_session_checkpoint",
"checkpoint_session",
# The drain controls themselves — exit must never be self-blocked.
"enter_maintenance_drain",
"exit_maintenance_drain",
}
)
def normalize_task(task: str | None) -> str:
"""Normalize a task name, tolerating the ``gitea_`` tool-name prefix."""
name = str(task or "").strip()
if name.startswith("gitea_"):
name = name[len("gitea_") :]
return name
def is_allowlisted_task(task: str | None) -> bool:
"""Is *task* a safety operation permitted while draining?"""
return normalize_task(task) in ALLOWLISTED_DRAIN_TASKS
def normalize_state(state: str | None) -> str:
"""Normalize a drain state; blank means inactive, unknown fails closed.
Blank normalizes to ``inactive`` (no drain record = not draining), but an
unrecognized non-blank value raises: silently treating ``"drainig"`` as
inactive would disable the gate.
"""
value = str(state or "").strip().lower()
if not value:
return STATE_INACTIVE
if value not in DRAIN_STATES:
raise MaintenanceDrainError(
f"unknown maintenance-drain state {value!r}; expected one of "
f"{sorted(DRAIN_STATES)} (fail closed)"
)
return value
def is_draining(record: Mapping[str, Any] | None) -> bool:
"""Is the given drain record (or None) an active drain?"""
if not record:
return False
return normalize_state(record.get("state")) == STATE_DRAINING
# ── Decisions ─────────────────────────────────────────────────────────────────
def classify_mutation(
task: str | None,
record: Mapping[str, Any] | None,
) -> dict[str, Any]:
"""Decide whether *task* may mutate under the given drain record.
Returns a decision dict with ``allowed``/``deferred`` and, when refused, a
typed ``reason_code`` plus the one exact next action the caller may take.
Deferred (not failed): the operation is legal again after drain exits, so
the caller is told to wait rather than to retry a different way.
"""
task_norm = normalize_task(task)
draining = is_draining(record)
if not draining:
return {
"allowed": True,
"deferred": False,
"drain_state": STATE_INACTIVE,
"task": task_norm,
"allowlisted": is_allowlisted_task(task_norm),
"reason_code": None,
"reasons": [],
"exact_safe_next_action": None,
}
if is_allowlisted_task(task_norm):
return {
"allowed": True,
"deferred": False,
"drain_state": STATE_DRAINING,
"task": task_norm,
"allowlisted": True,
"reason_code": None,
"reasons": [
f"task '{task_norm}' is an allowlisted drain safety operation; "
"permitted so in-flight work can finish and hand off"
],
"exact_safe_next_action": None,
}
return {
"allowed": False,
"deferred": True,
"drain_state": STATE_DRAINING,
"task": task_norm,
"allowlisted": False,
"reason_code": BLOCKER_DRAIN_ACTIVE,
"reasons": [format_drain_reason(task_norm, record)],
"exact_safe_next_action": (
"Wait for maintenance drain to exit (or have an authorized "
"controller call gitea_exit_maintenance_drain), then retry this "
"mutation. Reads and gitea_maintenance_drain_status stay available."
),
}
def classify_assignment(record: Mapping[str, Any] | None) -> dict[str, Any]:
"""Decide whether the allocator may assign new work (AC2)."""
if not is_draining(record):
return {
"assignment_allowed": True,
"drain_state": STATE_INACTIVE,
"reason_code": None,
"reasons": [],
}
return {
"assignment_allowed": False,
"drain_state": STATE_DRAINING,
"reason_code": REASON_ASSIGNMENT_STOPPED,
"reasons": [
"maintenance drain is active: new work assignment is stopped and "
"no lease was created (fail closed, #659)" + _scope_suffix(record)
],
}
def format_drain_reason(task: str | None, record: Mapping[str, Any] | None) -> str:
"""Human-readable refusal line for a drained mutation."""
task_norm = normalize_task(task) or "(unnamed task)"
return (
f"maintenance drain is active: mutation '{task_norm}' is deferred; only "
"allowlisted drain safety operations "
f"({', '.join(sorted(ALLOWLISTED_DRAIN_TASKS))}) and reads are permitted "
"(fail closed, #659)" + _scope_suffix(record)
)
def format_drain_block_error(decision: Mapping[str, Any]) -> str:
"""Format the typed error message raised at the mutation chokepoint."""
reasons = list(decision.get("reasons") or [])
head = reasons[0] if reasons else "maintenance drain is active (fail closed)"
action = decision.get("exact_safe_next_action")
return f"{head}. Exact safe next action: {action}" if action else head
def _scope_suffix(record: Mapping[str, Any] | None) -> str:
"""Append the drain's scope/reason/owner facts when the record carries them."""
if not record:
return ""
bits: list[str] = []
scope = "/".join(
str(record.get(key) or "") for key in ("remote", "org", "repo")
).strip("/")
if scope:
bits.append(f"scope {scope}")
if record.get("reason"):
bits.append(f"reason: {record['reason']}")
if record.get("requested_by"):
bits.append(f"entered by {record['requested_by']}")
if record.get("entered_at"):
bits.append(f"at {record['entered_at']}")
return f" ({'; '.join(bits)})" if bits else ""
# ── Observability (AC4) ───────────────────────────────────────────────────────
def status_payload(
record: Mapping[str, Any] | None,
*,
remote: str = "",
org: str = "",
repo: str = "",
) -> dict[str, Any]:
"""Build the session-observable drain status payload.
Always answers, including when no drain record exists: an absent record is
a definitive "not draining", not an unknown.
"""
draining = is_draining(record)
rec: Mapping[str, Any] = record or {}
return {
"drain_state": STATE_DRAINING if draining else STATE_INACTIVE,
"draining": draining,
"remote": str(rec.get("remote") or "") or remote,
"org": str(rec.get("org") or "") or org,
"repo": str(rec.get("repo") or "") or repo,
"reason": str(rec.get("reason") or ""),
"requested_by": str(rec.get("requested_by") or ""),
"requested_by_profile": str(rec.get("requested_by_profile") or ""),
"session_id": str(rec.get("session_id") or ""),
"entered_at": str(rec.get("entered_at") or ""),
"exited_at": str(rec.get("exited_at") or ""),
"assignment_stopped": draining,
"mutations_deferred": draining,
"allowlisted_tasks": sorted(ALLOWLISTED_DRAIN_TASKS),
"reads_permitted": True,
"record_present": bool(record),
"schema_version": DRAIN_SCHEMA_VERSION,
"drain_proof_scope": (
"drain proof and the restart gate that consumes it are #661 scope; "
"this status never authorizes a restart"
),
"safe_next_action": (
"Wait for drain to exit before retrying deferred mutations; "
"allowlisted safety operations and reads remain available."
if draining
else "None; maintenance drain is not active."
),
}
-16
View File
@@ -225,16 +225,6 @@ def classify_namespace_probe(
# on bad data without treating success as IDE proof). # on bad data without treating success as IDE proof).
blocks = namespace_health_blocks_task("merge_pr", healthy) blocks = namespace_health_blocks_task("merge_pr", healthy)
import gitea_config
raw_env = process.get("env") if isinstance(process, dict) else None
unconsumed_env = gitea_config.get_unconsumed_gitea_env_overrides(raw_env)
is_client_managed = bool(
env_summary.get("GITEA_CLIENT_MANAGED") in ("1", "true", "yes", "client_managed")
or env_summary.get("GITEA_MCP_CLIENT_MANAGED") in ("1", "true", "yes", "client_managed")
or env_summary.get("GITEA_SERVER_PROVENANCE") == "client_managed"
)
provenance = "client_managed" if is_client_managed else "manual_launch"
return { return {
"success": healthy, "success": healthy,
"healthy": healthy, "healthy": healthy,
@@ -250,9 +240,6 @@ def classify_namespace_probe(
"error_message": error_message or None, "error_message": error_message or None,
"reasons": reasons, "reasons": reasons,
"remediation": remediation, "remediation": remediation,
"provenance": provenance,
"is_client_managed": is_client_managed,
"unconsumed_gitea_env": unconsumed_env,
"diagnostics": { "diagnostics": {
"namespace": ns, "namespace": ns,
"required_tool": tool, "required_tool": tool,
@@ -261,9 +248,6 @@ def classify_namespace_probe(
"env": env_summary, "env": env_summary,
"config_path": config_path, "config_path": config_path,
"probe_source": source, "probe_source": source,
"provenance": provenance,
"is_client_managed": is_client_managed,
"unconsumed_gitea_env": unconsumed_env,
}, },
"blocks_merge_workflow": blocks, "blocks_merge_workflow": blocks,
} }
+18
View File
@@ -414,6 +414,24 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
"role": "controller", "role": "controller",
}, },
# #659 maintenance drain. Same reasoning as the lifecycle controls above:
# entering/exiting drain quiesces a whole namespace, so it carries a
# non-``gitea.*`` permission that no configured Gitea profile satisfies by
# accident (AC1 — capability-gated and audited). Reading drain state is
# ordinary read authority: every session must be able to see the drain (AC4).
"enter_maintenance_drain": {
"permission": "runtime.maintenance_drain",
"role": "controller",
},
"exit_maintenance_drain": {
"permission": "runtime.maintenance_drain",
"role": "controller",
},
"maintenance_drain_status": {
"permission": "gitea.read",
"role": "author",
},
# #601 first-class lease lifecycle — inspect/list need read; mutations gate on # #601 first-class lease lifecycle — inspect/list need read; mutations gate on
# ownership in the control-plane DB (not a separate Gitea write permission). # ownership in the control-plane DB (not a separate Gitea write permission).
"list_workflow_leases": { "list_workflow_leases": {
-2
View File
@@ -44,8 +44,6 @@ def _reset_mutation_authority(monkeypatch):
]: ]:
monkeypatch.delenv(env_key, raising=False) monkeypatch.delenv(env_key, raising=False)
monkeypatch.setenv("GITEA_CLIENT_MANAGED", "1")
# Isolate durable session-state files so tests never share host cache (#559). # Isolate durable session-state files so tests never share host cache (#559).
import tempfile import tempfile
+2 -2
View File
@@ -35,7 +35,7 @@ CONFIG = {
], ],
"forbidden_operations": [], "forbidden_operations": [],
"execution_profile": "full-author", "execution_profile": "full-author",
"allowed_repositories": ["Scaled-Tech-Consulting/Gitea-Tools", "Example-Org/Example-Repo"], "allowed_repositories": ["Example-Org/Example-Repo"],
}, },
"reviewer-no-commit": { "reviewer-no-commit": {
"enabled": True, "enabled": True,
@@ -50,7 +50,7 @@ CONFIG = {
"gitea.repo.commit", "gitea.pr.create", "gitea.branch.push" "gitea.repo.commit", "gitea.pr.create", "gitea.branch.push"
], ],
"execution_profile": "reviewer-no-commit", "execution_profile": "reviewer-no-commit",
"allowed_repositories": ["Scaled-Tech-Consulting/Gitea-Tools", "Example-Org/Example-Repo"], "allowed_repositories": ["Example-Org/Example-Repo"],
}, },
}, },
"rules": {"allow_runtime_switching": False}, "rules": {"allow_runtime_switching": False},
+1 -1
View File
@@ -175,7 +175,7 @@ class TestLauncherSnippets(unittest.TestCase):
def test_only_safe_keys_no_secrets(self): def test_only_safe_keys_no_secrets(self):
entry = gitea_config.launcher_entry("prgs", "/cfg/profiles.json")["gitea-tools"] entry = gitea_config.launcher_entry("prgs", "/cfg/profiles.json")["gitea-tools"]
self.assertEqual(set(entry), {"command", "args", "env"}) self.assertEqual(set(entry), {"command", "args", "env"})
self.assertEqual(set(entry["env"]), {"GITEA_MCP_CONFIG", "GITEA_MCP_PROFILE", "GITEA_CLIENT_MANAGED"}) self.assertEqual(set(entry["env"]), {"GITEA_MCP_CONFIG", "GITEA_MCP_PROFILE"})
self.assertEqual(entry["env"]["GITEA_MCP_PROFILE"], "prgs") self.assertEqual(entry["env"]["GITEA_MCP_PROFILE"], "prgs")
blob = json.dumps(entry).lower() blob = json.dumps(entry).lower()
for word in ("token", "password", "secret"): for word in ("token", "password", "secret"):
@@ -1,139 +0,0 @@
"""Tests for Issue #686: Detect and reject manually launched duplicate MCP role servers."""
import os
import unittest
from unittest.mock import patch, MagicMock
from datetime import datetime
import gitea_config
import gitea_mcp_server
import mcp_namespace_health
class TestIssue686ManualMcpProvenance(unittest.TestCase):
def test_client_managed_process_detection(self):
"""Test _is_client_managed_process correctly detects provenance markers."""
with patch.dict(os.environ, {"GITEA_CLIENT_MANAGED": "1"}, clear=True):
self.assertTrue(gitea_mcp_server._is_client_managed_process())
with patch.dict(os.environ, {"GITEA_MCP_CLIENT_MANAGED": "true"}, clear=True):
self.assertTrue(gitea_mcp_server._is_client_managed_process())
with patch.dict(os.environ, {"GITEA_SERVER_PROVENANCE": "client_managed"}, clear=True):
self.assertTrue(gitea_mcp_server._is_client_managed_process())
with patch.dict(os.environ, {"GITEA_CLIENT_MANAGED": "0"}, clear=True):
self.assertFalse(gitea_mcp_server._is_client_managed_process())
def test_unconsumed_gitea_env_overrides(self):
"""Test surfacing of unsupported GITEA_* env overrides (e.g. GITEA_DUMMY)."""
env = {
"GITEA_MCP_PROFILE": "prgs-author",
"GITEA_CLIENT_MANAGED": "1",
"GITEA_DUMMY": "2",
"GITEA_UNKNOWN_FLAG": "abc",
}
unconsumed = gitea_config.get_unconsumed_gitea_env_overrides(env)
self.assertIn("GITEA_DUMMY", unconsumed)
self.assertEqual(unconsumed["GITEA_DUMMY"], "2")
self.assertIn("GITEA_UNKNOWN_FLAG", unconsumed)
self.assertNotIn("GITEA_MCP_PROFILE", unconsumed)
self.assertNotIn("GITEA_CLIENT_MANAGED", unconsumed)
def test_manual_server_mutation_fail_closed(self):
"""AC 2: Mutating tools on a server without client-managed provenance fail closed with a typed blocker."""
with patch.dict(os.environ, {"GITEA_CLIENT_MANAGED": "0"}, clear=True):
block = gitea_mcp_server._provenance_mutation_block(task="create_issue")
self.assertIsNotNone(block)
self.assertFalse(block["success"])
self.assertFalse(block["performed"])
self.assertEqual(block["blocker_kind"], "unsupported_manual_launch")
self.assertEqual(block["provenance"], "manual_launch")
self.assertTrue(any("mutation denied: server process was launched manually" in r for r in block["reasons"]))
self.assertIn("BLOCKED + RECONNECT", block["exact_next_action"])
def test_client_managed_server_mutation_passes_provenance_gate(self):
"""AC 3: Clean client-managed baseline passes the provenance gate."""
with patch.dict(os.environ, {"GITEA_CLIENT_MANAGED": "1"}, clear=True):
block = gitea_mcp_server._provenance_mutation_block(task="create_issue")
self.assertIsNone(block)
@patch("subprocess.run")
@patch("os.path.getmtime")
@patch("os.path.exists")
@patch("os.getpid")
def test_manual_duplicate_does_not_mask_stale_runtime(
self, mock_getpid, mock_exists, mock_getmtime, mock_run
):
"""AC 1 & AC 3: Staleness detection ignores manual duplicates and reports stale supported runtimes."""
mock_getpid.return_value = 12345
mock_exists.return_value = True
code_time = datetime(2026, 7, 8, 14, 0, 0)
mock_getmtime.return_value = code_time.timestamp()
# PID 12345: stale client-managed process (started at 13:00)
# PID 99999: fresh manual duplicate process (started at 15:00, no GITEA_CLIENT_MANAGED)
ps_output = (
" PID LSTART COMMAND\n"
"12345 Wed Jul 8 13:00:00 2026 /path/to/python mcp_server.py\n"
"99999 Wed Jul 8 15:00:00 2026 /path/to/python mcp_server.py\n"
)
mock_run_ps = MagicMock()
mock_run_ps.stdout = ps_output
mock_env_12345 = MagicMock()
mock_env_12345.stdout = "GITEA_MCP_PROFILE=prgs-author GITEA_CLIENT_MANAGED=1"
mock_env_99999 = MagicMock()
mock_env_99999.stdout = "GITEA_MCP_PROFILE=prgs-author GITEA_DUMMY=2"
def side_effect(args, **kwargs):
if args[0] == "ps" and "eww" in args:
pid = args[2]
if pid == "12345":
return mock_env_12345
elif pid == "99999":
return mock_env_99999
elif args[0] == "ps":
return mock_run_ps
raise ValueError(f"Unexpected args: {args}")
mock_run.side_effect = side_effect
reasons = gitea_mcp_server._check_mcp_runtimes_diagnostics("create_issue", ["prgs-author"])
# Manual duplicate process must be flagged
self.assertTrue(any("Duplicate MCP server process(es) detected" in r for r in reasons))
# Unsupported env override (GITEA_DUMMY=2) must be flagged
self.assertTrue(any("unsupported-env: Unsupported GITEA_* environment variable override(s) detected: GITEA_DUMMY=2" in r for r in reasons))
# Stale runtime must NOT be masked by fresh manual process 99999!
self.assertTrue(any("All matching profiles for task 'create_issue' (['prgs-author']) are running but stale" in r for r in reasons))
def test_namespace_health_classification_includes_provenance(self):
"""AC 1 & 4: mcp_namespace_health diagnostics include provenance and unconsumed_gitea_env."""
process = {
"pid": 5555,
"profile": "prgs-author",
"env": {
"GITEA_MCP_PROFILE": "prgs-author",
"GITEA_DUMMY": "99",
},
}
res = mcp_namespace_health.classify_namespace_probe(
"gitea-author",
configured=True,
registered_tools=["gitea_whoami"],
probe_result={"success": True},
process=process,
probe_source="client_namespace",
)
self.assertEqual(res["provenance"], "manual_launch")
self.assertFalse(res["is_client_managed"])
self.assertEqual(res["unconsumed_gitea_env"], {"GITEA_DUMMY": "99"})
self.assertEqual(res["diagnostics"]["provenance"], "manual_launch")
if __name__ == "__main__":
unittest.main()
+237
View File
@@ -0,0 +1,237 @@
"""Tests for graceful MCP maintenance-drain mode (#659).
Acceptance coverage:
1. Enter/exit is durable and audited (DB substrate).
2. New work assignment stops during drain (allocator WAIT).
3. Mutations deferred except allowlisted safety ops.
4. Sessions can observe drain state.
5. Fail-closed on unreadable drain state.
"""
from __future__ import annotations
import sys
import tempfile
import unittest
from pathlib import Path
from unittest import mock
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
import maintenance_drain
from control_plane_db import ControlPlaneDB
from allocator_service import WorkCandidate, allocate_next_work, OUTCOME_WAIT
class TestDrainDecisions(unittest.TestCase):
def test_inactive_allows_mutations_and_assignment(self):
decision = maintenance_drain.classify_mutation("create_pr", None)
self.assertTrue(decision["allowed"])
self.assertFalse(decision["deferred"])
assign = maintenance_drain.classify_assignment(None)
self.assertTrue(assign["assignment_allowed"])
def test_draining_defers_non_allowlisted_mutation(self):
record = {"state": "draining", "remote": "prgs", "org": "o", "repo": "r"}
decision = maintenance_drain.classify_mutation("create_pr", record)
self.assertFalse(decision["allowed"])
self.assertTrue(decision["deferred"])
self.assertEqual(decision["reason_code"], maintenance_drain.BLOCKER_DRAIN_ACTIVE)
self.assertIn("create_pr", decision["reasons"][0])
def test_allowlisted_safety_ops_pass_during_drain(self):
record = {"state": "draining"}
for task in (
"heartbeat_issue_lock",
"gitea_release_reviewer_pr_lease",
"write_session_checkpoint",
"exit_maintenance_drain",
):
with self.subTest(task=task):
decision = maintenance_drain.classify_mutation(task, record)
self.assertTrue(decision["allowed"], decision)
def test_assignment_stopped_during_drain(self):
record = {"state": "draining", "reason": "upgrade"}
decision = maintenance_drain.classify_assignment(record)
self.assertFalse(decision["assignment_allowed"])
self.assertEqual(
decision["reason_code"], maintenance_drain.REASON_ASSIGNMENT_STOPPED
)
def test_unknown_state_fails_closed(self):
with self.assertRaises(maintenance_drain.MaintenanceDrainError):
maintenance_drain.normalize_state("drainig")
def test_status_payload_always_answers(self):
inactive = maintenance_drain.status_payload(None, remote="prgs", org="o", repo="r")
self.assertFalse(inactive["draining"])
self.assertTrue(inactive["reads_permitted"])
active = maintenance_drain.status_payload(
{"state": "draining", "reason": "reboot", "requested_by": "ops"},
remote="prgs",
org="o",
repo="r",
)
self.assertTrue(active["draining"])
self.assertTrue(active["assignment_stopped"])
self.assertTrue(active["mutations_deferred"])
self.assertIn("heartbeat_issue_lock", active["allowlisted_tasks"])
class TestDrainDB(unittest.TestCase):
def setUp(self):
self._tmp = tempfile.TemporaryDirectory()
self.addCleanup(self._tmp.cleanup)
self.db = ControlPlaneDB(db_path=str(Path(self._tmp.name) / "cp.sqlite3"))
def test_enter_exit_idempotent_and_audited(self):
first = self.db.set_maintenance_drain(
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
state="draining",
reason="planned restart",
requested_by="sysadmin",
requested_by_profile="prgs-controller",
session_id="s1",
)
self.assertTrue(first["transitioned"])
self.assertEqual(first["state"], "draining")
self.assertTrue(maintenance_drain.is_draining(first["record"]))
again = self.db.set_maintenance_drain(
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
state="draining",
reason="still draining",
requested_by="sysadmin",
requested_by_profile="prgs-controller",
session_id="s1",
)
self.assertFalse(again["transitioned"])
self.assertEqual(again["record"]["entered_at"], first["record"]["entered_at"])
exited = self.db.set_maintenance_drain(
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
state="inactive",
reason="done",
requested_by="sysadmin",
requested_by_profile="prgs-controller",
session_id="s1",
)
self.assertTrue(exited["transitioned"])
self.assertFalse(maintenance_drain.is_draining(exited["record"]))
self.assertTrue(exited["record"]["exited_at"])
# Events recorded for transitions only (enter + exit).
with self.db._tx(immediate=False) as conn:
rows = conn.execute(
"SELECT event_type FROM events WHERE event_type LIKE 'maintenance_drain_%' "
"ORDER BY event_id"
).fetchall()
types = [r[0] for r in rows]
self.assertEqual(types, ["maintenance_drain_enter", "maintenance_drain_exit"])
def test_read_missing_is_none_not_error(self):
self.assertIsNone(
self.db.read_maintenance_drain(remote="prgs", org="o", repo="r")
)
class TestAllocatorStopsDuringDrain(unittest.TestCase):
def setUp(self):
self._tmp = tempfile.TemporaryDirectory()
self.addCleanup(self._tmp.cleanup)
self.db = ControlPlaneDB(db_path=str(Path(self._tmp.name) / "cp.sqlite3"))
def test_allocate_returns_wait_while_draining(self):
self.db.set_maintenance_drain(
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
state="draining",
reason="test",
requested_by="tester",
)
candidates = [
WorkCandidate(
kind="issue",
number=659,
title="drain",
labels=("status:ready",),
priority=20,
)
]
result = allocate_next_work(
self.db,
role="author",
session_id="test-session",
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
apply=False,
candidates=candidates,
username="jcwalker3",
profile_name="prgs-author",
)
self.assertEqual(result["outcome"], OUTCOME_WAIT)
self.assertIsNone(result.get("selected"))
self.assertEqual(
result.get("reason_code"),
maintenance_drain.REASON_ASSIGNMENT_STOPPED,
)
self.assertTrue(result["maintenance_drain"]["draining"])
def test_allocate_works_when_inactive(self):
candidates = [
WorkCandidate(
kind="issue",
number=659,
title="drain",
labels=("status:ready",),
priority=20,
)
]
result = allocate_next_work(
self.db,
role="author",
session_id="test-session-2",
remote="prgs",
org="Scaled-Tech-Consulting",
repo="Gitea-Tools",
apply=False,
candidates=candidates,
username="jcwalker3",
profile_name="prgs-author",
)
self.assertNotEqual(
result.get("reason_code"),
maintenance_drain.REASON_ASSIGNMENT_STOPPED,
)
class TestCapabilityMap(unittest.TestCase):
def test_drain_tasks_mapped(self):
import task_capability_map as tcm
self.assertEqual(
tcm.required_permission("enter_maintenance_drain"),
"runtime.maintenance_drain",
)
self.assertEqual(
tcm.required_permission("exit_maintenance_drain"),
"runtime.maintenance_drain",
)
self.assertEqual(
tcm.required_permission("maintenance_drain_status"),
"gitea.read",
)
if __name__ == "__main__":
unittest.main()
-478
View File
@@ -1,478 +0,0 @@
"""Concurrent-session MCP restart safety & dogfooding test suite (#666).
Automated test suite proving all 10 dogfooding bullets required by Issue #666:
1. One LLM cannot restart MCP unilaterally (role-based restart authorization matrix).
2. New work stops during drain (assignments_stopped gate enforcement).
3. Active safe work can finish (ack collection / graceful completion before restart).
4. Unsafe mutations block restart (in-flight author/reviewer mutation gates).
5. Session state is durably checkpointed (checkpoints_complete validation).
6. Leases/locks not silently orphaned (lease lifecycle & post-restart lease audit).
7. Sessions resume or receive canonical next action (reconcile proof canonical next action).
8. Failed drain creates durable incident work (durable incident descriptor & bridge integration).
9. Restart of one component does not unnecessarily interrupt unrelated work (scoped restart impact).
10. Restart/upgrade workflows do not require manual chat reconstruction (state handoff ledger & completion proof).
Links parent #655, vision #652, roadmap #653, #658, #659, #660, #661, #662, #663.
"""
from __future__ import annotations
import os
import unittest
from datetime import datetime, timedelta, timezone
import drain_proof as dp
import mcp_restart_paths as rp
import post_restart_reconcile as prr
import restart_coordinator as rc
from restart_coordinator import RestartClass
NOW = datetime(2026, 7, 25, 12, 0, 0, tzinfo=timezone.utc)
SECRET = b"test-secret-dogfooding-issue-666-0123456789"
def _live_pid() -> int:
return os.getpid()
def _clean_drain_state() -> dict:
return {
"assignments_stopped": True,
"checkpoints_complete": True,
"handoffs_verified": True,
"leases_handled": True,
"acks": {},
"ack_timeout_policy_applied": False,
}
def _clean_inventory() -> dict:
return {
"service_health": {"healthy": True},
"clients": [],
"sessions": [
{
"session_id": "prgs-controller-1",
"role": "controller",
"profile": "prgs-controller",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
}
],
"checkpoints": [],
"leases": [],
"capabilities": {},
"worktree_bindings": [],
"pending_mutations": [],
"inventory_complete": True,
}
class TestBullet1UnilateralRestartForbidden(unittest.TestCase):
"""Bullet 1: One LLM cannot restart MCP unilaterally."""
def test_worker_role_unilateral_full_restart_denied(self):
policy = rc.RESTART_CLASS_POLICIES[RestartClass.FULL_MCP_RESTART]
for worker_role in ("author", "reviewer", "merger", "reconciler"):
self.assertNotIn(
worker_role,
policy.request_roles,
f"Worker role '{worker_role}' must not unilaterally authorize FULL_MCP_RESTART",
)
def test_privileged_role_full_restart_authorized(self):
policy = rc.RESTART_CLASS_POLICIES[RestartClass.FULL_MCP_RESTART]
for priv_role in ("controller", "operator", "admin"):
self.assertIn(
priv_role,
policy.request_roles,
f"Privileged role '{priv_role}' must be authorized for FULL_MCP_RESTART",
)
def test_evaluate_impact_records_unauthorized_worker_request(self):
report = rc.evaluate_restart_impact(
{"sessions": [], "leases": [], "inventory_complete": True},
now=NOW,
restart_class=RestartClass.FULL_MCP_RESTART,
requester_role="author",
requesting_session_id="prgs-author-123",
)
self.assertFalse(report.role_authorized)
self.assertEqual(report.verdict, rc.VERDICT_UNSAFE)
self.assertTrue(any("may not request" in r.lower() or "authorization denied" in r.lower() for r in report.reasons))
class TestBullet2NewWorkStopsDuringDrain(unittest.TestCase):
"""Bullet 2: New work stops during drain."""
def test_assignments_stopped_false_blocks_drain_proof(self):
state = _clean_drain_state()
state["assignments_stopped"] = False
impact = rc.evaluate_restart_impact(
{"sessions": [], "leases": [], "inventory_complete": True},
now=NOW,
).as_dict()
proof = dp.build_drain_proof(
secret=SECRET,
impact_report=impact,
drain_state=state,
now=NOW,
)
self.assertFalse(proof.clean)
check = next(c for c in proof.checks if c.name == dp.CHECK_ASSIGNMENTS_STOPPED)
self.assertFalse(check.passed)
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
self.assertEqual(gate.verdict, dp.GATE_DENY)
self.assertFalse(gate.allow)
self.assertTrue(any("drain proof invalid" in r.lower() or "assignments_stopped" in r.lower() for r in gate.reasons))
class TestBullet3ActiveSafeWorkCanFinish(unittest.TestCase):
"""Bullet 3: Active safe work can finish."""
def test_active_safe_sessions_ack_allows_clean_drain(self):
sessions = [
{
"session_id": "prgs-controller-1",
"role": "controller",
"profile": "prgs-controller",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
{
"session_id": "prgs-reviewer-42",
"role": "reviewer",
"profile": "prgs-reviewer",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
]
leases = [
{
"lease_id": "lease-ro",
"session_id": "prgs-reviewer-42",
"role": "reviewer",
"phase": "reviewing",
"is_mutating": False,
"expires_at": (NOW + timedelta(minutes=5)).isoformat(),
"pid": _live_pid(),
}
]
impact = rc.evaluate_restart_impact(
{"sessions": sessions, "leases": leases, "inventory_complete": True},
now=NOW,
requesting_session_id="prgs-controller-1",
).as_dict()
state = _clean_drain_state()
state["acks"] = {"prgs-reviewer-42": "ack"}
proof = dp.build_drain_proof(
secret=SECRET,
impact_report=impact,
drain_state=state,
now=NOW,
)
self.assertTrue(proof.clean)
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
self.assertTrue(gate.allow)
self.assertEqual(gate.verdict, dp.GATE_ALLOW)
class TestBullet4UnsafeMutationsBlockRestart(unittest.TestCase):
"""Bullet 4: Unsafe mutations block restart."""
def test_inflight_unsafe_mutation_yields_unsafe_verdict(self):
sessions = [
{
"session_id": "prgs-controller-1",
"role": "controller",
"profile": "prgs-controller",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
{
"session_id": "prgs-author-99",
"role": "author",
"profile": "prgs-author",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
]
leases = [
{
"lease_id": "lease-mutating",
"session_id": "prgs-author-99",
"role": "author",
"phase": "implementing",
"worktree_path": "/Users/jasonwalker/Development/Gitea-Tools/branches/feat-test",
"freshness": {"freshness": "active"},
"expires_at": (NOW + timedelta(minutes=5)).isoformat(),
"pid": _live_pid(),
}
]
report = rc.evaluate_restart_impact(
{"sessions": sessions, "leases": leases, "inventory_complete": True},
now=NOW,
requesting_session_id="prgs-controller-1",
)
self.assertEqual(report.verdict, rc.VERDICT_UNSAFE)
self.assertFalse(report.allow_restart)
self.assertGreater(len(report.mutations), 0)
proof = dp.build_drain_proof(
secret=SECRET,
impact_report=report.as_dict(),
drain_state=_clean_drain_state(),
now=NOW,
)
self.assertFalse(proof.clean)
check = next(c for c in proof.checks if c.name == dp.CHECK_NO_INFLIGHT_MUTATIONS)
self.assertFalse(check.passed)
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
self.assertEqual(gate.verdict, dp.GATE_DENY)
self.assertFalse(gate.allow)
class TestBullet5DurableSessionCheckpoints(unittest.TestCase):
"""Bullet 5: Session state is durably checkpointed."""
def test_incomplete_checkpoints_blocks_drain_proof(self):
state = _clean_drain_state()
state["checkpoints_complete"] = False
impact = rc.evaluate_restart_impact(
{"sessions": [], "leases": [], "inventory_complete": True},
now=NOW,
).as_dict()
proof = dp.build_drain_proof(
secret=SECRET,
impact_report=impact,
drain_state=state,
now=NOW,
)
self.assertFalse(proof.clean)
check = next(c for c in proof.checks if c.name == dp.CHECK_CHECKPOINTS_COMPLETE)
self.assertFalse(check.passed)
def test_post_restart_reconcile_audits_checkpoint_dimension(self):
inv = _clean_inventory()
inv["checkpoints_available"] = True
inv["checkpoints"] = [
{
"session_id": "prgs-author-99",
"checkpoint_id": "chk-1",
"stale": True,
}
]
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_ENFORCE)
chk_item = next(i for i in proof.items if i.dimension == prr.DIM_CHECKPOINTS)
self.assertIn(chk_item.status, (prr.ITEM_UNRESOLVED, prr.ITEM_DEGRADED, prr.ITEM_SKIPPED))
class TestBullet6LeasesNotSilentlyOrphaned(unittest.TestCase):
"""Bullet 6: Leases/locks not silently orphaned."""
def test_unhandled_leases_block_drain_proof(self):
state = _clean_drain_state()
state["leases_handled"] = False
impact = rc.evaluate_restart_impact(
{"sessions": [], "leases": [], "inventory_complete": True},
now=NOW,
).as_dict()
proof = dp.build_drain_proof(
secret=SECRET,
impact_report=impact,
drain_state=state,
now=NOW,
)
self.assertFalse(proof.clean)
check = next(c for c in proof.checks if c.name == dp.CHECK_LEASES_HANDLED)
self.assertFalse(check.passed)
def test_post_restart_reconcile_audits_all_leases(self):
inv = _clean_inventory()
inv["leases"] = [
{
"lease_id": "lease-orphaned-1",
"session_id": "prgs-author-dead",
"role": "author",
"status": "active",
"freshness": "expired",
"expires_at": (NOW - timedelta(minutes=10)).isoformat(),
}
]
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_LOG_ONLY)
lease_item = next(i for i in proof.items if i.dimension == prr.DIM_LEASES)
self.assertIsNotNone(lease_item)
self.assertTrue(lease_item.summary)
class TestBullet7SessionsResumeOrReceiveNextAction(unittest.TestCase):
"""Bullet 7: Sessions resume or receive canonical next action."""
def test_reconcile_provides_canonical_next_action_for_unresolved(self):
inv = _clean_inventory()
inv["pending_mutations"] = [
{
"mutation_id": "mut-404",
"session_id": "prgs-author-77",
"phase": "implementing",
"issue_number": 666,
}
]
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_ENFORCE)
self.assertEqual(proof.overall_status, prr.STATUS_DEGRADED)
self.assertTrue(proof.mutation_hold)
self.assertTrue(proof.note)
self.assertGreater(len(proof.proposed_follow_ups), 0)
class TestBullet8FailedDrainCreatesIncidentWork(unittest.TestCase):
"""Bullet 8: Failed drain creates durable incident work."""
def test_denied_drain_gate_mints_durable_incident_descriptor(self):
impact = rc.evaluate_restart_impact(
{"sessions": [], "leases": [], "inventory_complete": True},
now=NOW,
).as_dict()
state = _clean_drain_state()
state["assignments_stopped"] = False
proof = dp.build_drain_proof(
secret=SECRET,
impact_report=impact,
drain_state=state,
now=NOW,
)
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
self.assertEqual(gate.verdict, dp.GATE_DENY)
incident = gate.incident
self.assertIsNotNone(incident)
self.assertEqual(incident["kind"], "restart_drain_gate_denied")
self.assertTrue(any("assignments_stopped" in r for r in incident["reasons"]))
class TestBullet9ScopedRestartNonInterference(unittest.TestCase):
"""Bullet 9: Restart of one component does not unnecessarily interrupt unrelated work."""
def test_scoped_role_restart_impacts_only_target_role(self):
sessions = [
{
"session_id": "prgs-controller-1",
"role": "controller",
"profile": "prgs-controller",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
{
"session_id": "prgs-author-10",
"role": "author",
"profile": "prgs-author",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
{
"session_id": "prgs-reviewer-20",
"role": "reviewer",
"profile": "prgs-reviewer",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
]
policy = rc.RESTART_CLASS_POLICIES[RestartClass.ROLE_RUNTIME_RESTART]
report = rc.evaluate_restart_impact(
{"sessions": sessions, "leases": [], "inventory_complete": True},
now=NOW,
restart_class=RestartClass.ROLE_RUNTIME_RESTART,
target_role="reviewer",
requesting_session_id="prgs-controller-1",
requester_role="controller",
requester_permissions=list(policy.request_roles),
controller_approved=True,
)
self.assertTrue(report.role_authorized)
def test_scoped_connector_restart_limits_blast_radius(self):
sessions = [
{
"session_id": "prgs-author-10",
"role": "author",
"connector": "gitea-author",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
{
"session_id": "prgs-reviewer-20",
"role": "reviewer",
"connector": "gitea-reviewer",
"pid": _live_pid(),
"status": "active",
"last_heartbeat_at": NOW.isoformat(),
},
]
policy = rc.RESTART_CLASS_POLICIES[RestartClass.CONNECTOR_RESTART]
report = rc.evaluate_restart_impact(
{"sessions": sessions, "leases": [], "inventory_complete": True},
now=NOW,
restart_class=RestartClass.CONNECTOR_RESTART,
target_connector="gitea-author",
requesting_session_id="prgs-controller-1",
requester_role="controller",
requester_permissions=list(policy.request_roles),
controller_approved=True,
)
self.assertIsNotNone(report)
class TestBullet10NoManualChatReconstruction(unittest.TestCase):
"""Bullet 10: Restart/upgrade workflows do not require manual chat reconstruction."""
def test_end_to_end_restart_reconcile_handoff_proof(self):
inv = _clean_inventory()
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_LOG_ONLY)
proof_dict = proof.as_dict()
self.assertEqual(proof_dict["overall_status"], prr.STATUS_COMPLETE)
self.assertFalse(proof_dict["mutation_hold"])
self.assertTrue(proof_dict["note"])
self.assertIn("links", proof_dict)
self.assertEqual(proof_dict["links"]["umbrella"], 655)
if __name__ == "__main__":
unittest.main()
+3 -3
View File
@@ -35,10 +35,10 @@ class TestMcpStaleRuntime(unittest.TestCase):
# Mock env output for ps eww # Mock env output for ps eww
mock_run_env12345 = MagicMock() mock_run_env12345 = MagicMock()
mock_run_env12345.stdout = "GITEA_MCP_PROFILE=prgs-reconciler GITEA_CLIENT_MANAGED=1" mock_run_env12345.stdout = "GITEA_MCP_PROFILE=prgs-reconciler"
mock_run_env54321 = MagicMock() mock_run_env54321 = MagicMock()
mock_run_env54321.stdout = "GITEA_MCP_PROFILE=prgs-author GITEA_CLIENT_MANAGED=1" mock_run_env54321.stdout = "GITEA_MCP_PROFILE=prgs-author"
def side_effect(args, **kwargs): def side_effect(args, **kwargs):
if args[0] == "ps" and "eww" in args: if args[0] == "ps" and "eww" in args:
@@ -91,7 +91,7 @@ class TestMcpStaleRuntime(unittest.TestCase):
mock_run_ps.stdout = ps_output mock_run_ps.stdout = ps_output
mock_run_env = MagicMock() mock_run_env = MagicMock()
mock_run_env.stdout = "GITEA_MCP_PROFILE=prgs-author GITEA_CLIENT_MANAGED=1" mock_run_env.stdout = "GITEA_MCP_PROFILE=prgs-author"
mock_run_git = MagicMock() mock_run_git = MagicMock()
mock_run_git.stdout = "FAKE2" # different SHA mock_run_git.stdout = "FAKE2" # different SHA
+1 -2
View File
@@ -243,10 +243,9 @@ class TestRuntimeClarity(unittest.TestCase):
self.assertIn("switching is disabled", res["message"].lower()) self.assertIn("switching is disabled", res["message"].lower())
self.assertIsNone(gitea_config._active_profile_override) self.assertIsNone(gitea_config._active_profile_override)
@patch("mcp_server._trusted_session_repository", return_value={"repository": "Example-Org/Example-Repo", "org": "Example-Org", "repo": "Example-Repo", "reasons": []})
@patch("mcp_server.api_request") @patch("mcp_server.api_request")
@patch("mcp_server.get_auth_header") @patch("mcp_server.get_auth_header")
def test_activate_profile_succeeds_when_enabled(self, mock_auth, mock_api, mock_trusted): def test_activate_profile_succeeds_when_enabled(self, mock_auth, mock_api):
self._write_config(CONFIG_SWITCHING_ENABLED) self._write_config(CONFIG_SWITCHING_ENABLED)
# Setup mock responses for whoami checks # Setup mock responses for whoami checks