Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e344e68a13 | ||
|
|
c0c6d14b73 | ||
|
|
108cbfa173 | ||
|
|
0dbff6dcd5 | ||
|
|
4a1cc63e94 | ||
|
|
6596b259fc | ||
|
|
b0868be6b3 | ||
|
|
1a38ef95e3 | ||
|
|
324a0c8a8d | ||
|
|
34968475c7 | ||
|
|
a6c7d1491e | ||
|
|
9d96cf4cfa | ||
|
|
7978008709 | ||
|
|
6a53308473 | ||
|
|
626be8b178 | ||
|
|
c763161702 | ||
|
|
3f584352df |
+136
-21
@@ -37,6 +37,48 @@ CANONICAL_ROOT_ENV = "GITEA_CANONICAL_REPOSITORY_ROOT"
|
||||
# Candidate git remote names probed when deriving repository identity.
|
||||
_IDENTITY_REMOTE_CANDIDATES = ("prgs", "origin", "dadeschools", "mdcps")
|
||||
|
||||
# Repository-authority modes (#973 B10). Exactly two values are supported.
|
||||
# ``mode`` selects how repository authority is established, so an unrecognised
|
||||
# value must never be normalised onto one of these: aliasing a trusted mode is
|
||||
# precisely the defect. Omitting the argument keeps the documented safe default,
|
||||
# ``validation``.
|
||||
MODE_VALIDATION = "validation"
|
||||
MODE_DERIVATION = "derivation"
|
||||
SUPPORTED_MODES: tuple[str, ...] = (MODE_VALIDATION, MODE_DERIVATION)
|
||||
|
||||
# Reason code emitted when an unsupported mode is refused. Mirrors the existing
|
||||
# ``webui.sanctioned_restart.DENY_UNKNOWN_MODE`` convention so callers and tests
|
||||
# can assert the refusal cause rather than string-matching prose.
|
||||
DENY_UNKNOWN_MODE = "unknown_mode"
|
||||
|
||||
|
||||
def unsupported_mode_reason(mode: object) -> str | None:
|
||||
"""Precise rejection reason for *mode*, or None when *mode* is supported.
|
||||
|
||||
Only the two documented string values are accepted, compared exactly — no
|
||||
stripping, no case folding — so misspellings and whitespace variants are
|
||||
refused rather than coerced. An empty string, ``None``, and any non-string
|
||||
are all *explicitly supplied* unsupported values and are refused on the same
|
||||
footing; none of them is normalised to a supported mode. Omitting the
|
||||
argument entirely never reaches here with an unsupported value because the
|
||||
parameter default is ``"validation"``.
|
||||
"""
|
||||
if isinstance(mode, str) and mode in SUPPORTED_MODES:
|
||||
return None
|
||||
supported = ", ".join(repr(m) for m in SUPPORTED_MODES)
|
||||
if not isinstance(mode, str):
|
||||
return (
|
||||
f"unsupported repository-authority mode {mode!r} of type "
|
||||
f"{type(mode).__name__}: only {supported} are supported; the mode "
|
||||
"is refused before any repository assessment and no repository "
|
||||
"identity was resolved through it (fail closed)"
|
||||
)
|
||||
return (
|
||||
f"unsupported repository-authority mode {mode!r}: only {supported} are "
|
||||
"supported; the mode is refused before any repository assessment and no "
|
||||
"repository identity was resolved through it (fail closed)"
|
||||
)
|
||||
|
||||
|
||||
def configured_canonical_root(
|
||||
profile: Mapping | None,
|
||||
@@ -132,6 +174,7 @@ def assess_canonical_repository_root(
|
||||
process_project_root: str,
|
||||
remote: str | None = None,
|
||||
require_binding: bool = False,
|
||||
mode: str = "validation",
|
||||
) -> dict:
|
||||
"""Validate the canonical repository root binding, failing closed on forgery.
|
||||
|
||||
@@ -140,15 +183,41 @@ def assess_canonical_repository_root(
|
||||
``configured`` (whether a cross-repo binding was declared),
|
||||
``resolved_slug`` and ``source``.
|
||||
|
||||
Without a configured binding the single-repo default is preserved: the
|
||||
canonical root is derived from *process_project_root* and never blocks
|
||||
(unless *require_binding* explicitly demands one).
|
||||
*mode* accepts exactly the two values in :data:`SUPPORTED_MODES`:
|
||||
- ``"validation"`` (default): an independently trusted expected repository
|
||||
slug is known (or required). Candidate root's observed identity must match.
|
||||
Unprovable or missing expected identity fails closed when *require_binding* is True.
|
||||
- ``"derivation"``: caller is deriving the canonical repository identity.
|
||||
No expected slug exists yet by design. Derivation succeeds if the configured
|
||||
root exists, is a git repository, and carries a resolvable git remote identity.
|
||||
|
||||
With a configured binding the path must exist, be a git repository, and —
|
||||
when *expected_slug* is known — carry a matching repository identity. A
|
||||
mismatched or (when *require_binding*) unprovable identity is a forged or
|
||||
conflicting binding and fails closed.
|
||||
Every other explicitly supplied value — unknown strings, misspellings, the
|
||||
empty string, ``None``, and non-strings — is refused with ``proven`` False,
|
||||
``block`` True, and ``reason_code`` :data:`DENY_UNKNOWN_MODE` (#973 B10).
|
||||
Omitting *mode* entirely keeps the documented ``"validation"`` default.
|
||||
"""
|
||||
# #973 B10: refuse an unsupported repository-authority mode as the very first
|
||||
# act, before any candidate-root existence check, path or symlink resolution,
|
||||
# git top-level discovery, remote-URL or repository-identity discovery, and
|
||||
# before any expected-versus-observed comparison or validation/derivation
|
||||
# behaviour. An unsupported mode previously fell into the catch-all ``else``
|
||||
# on the configured-root path (silently receiving validation semantics) and
|
||||
# skipped the identity comparison entirely on the single-repository default
|
||||
# path (strictly weaker than validation), so matching identities could return
|
||||
# ``proven`` True. No repository identity may be resolved through a mode the
|
||||
# contract does not define.
|
||||
mode_reason = unsupported_mode_reason(mode)
|
||||
if mode_reason is not None:
|
||||
return _assessment(
|
||||
proven=False,
|
||||
reasons=[mode_reason],
|
||||
configured=bool((configured_value or "").strip()),
|
||||
canonical_repo_root=None,
|
||||
resolved_slug=None,
|
||||
source=source,
|
||||
reason_code=DENY_UNKNOWN_MODE,
|
||||
)
|
||||
|
||||
process_root = os.path.realpath(process_project_root)
|
||||
declared = (configured_value or "").strip()
|
||||
|
||||
@@ -168,12 +237,31 @@ def assess_canonical_repository_root(
|
||||
)
|
||||
# Single-repo default: canonical root follows the install checkout.
|
||||
derived = resolve_repo_toplevel(process_root) or process_root
|
||||
resolved_slug = repository_identity_slug(derived, remote=remote)
|
||||
reasons: list[str] = []
|
||||
# #973 B10: the allowlist above guarantees *mode* is one of the two
|
||||
# supported values here, so an unsupported value can no longer skip this
|
||||
# identity comparison and end up strictly weaker than validation.
|
||||
if mode == MODE_VALIDATION and expected_slug:
|
||||
expected = expected_slug.strip()
|
||||
if resolved_slug and resolved_slug.lower() != expected.lower():
|
||||
reasons.append(
|
||||
f"canonical repository root identity mismatch: '{derived}' resolves "
|
||||
f"to repository '{resolved_slug}' but expected repository identity "
|
||||
f"is '{expected}' (forged or conflicting binding, fail closed)"
|
||||
)
|
||||
elif not resolved_slug and require_binding:
|
||||
reasons.append(
|
||||
f"canonical repository root '{derived}' has no resolvable git "
|
||||
f"remote identity to confirm authorization for '{expected}' "
|
||||
"(fail closed)"
|
||||
)
|
||||
return _assessment(
|
||||
proven=True,
|
||||
reasons=[],
|
||||
proven=not reasons,
|
||||
reasons=reasons,
|
||||
configured=False,
|
||||
canonical_repo_root=derived,
|
||||
resolved_slug=None,
|
||||
resolved_slug=resolved_slug,
|
||||
source=None,
|
||||
)
|
||||
|
||||
@@ -207,19 +295,36 @@ def assess_canonical_repository_root(
|
||||
|
||||
resolved_slug = repository_identity_slug(toplevel, remote=remote)
|
||||
reasons: list[str] = []
|
||||
expected = (expected_slug or "").strip() or None
|
||||
if expected:
|
||||
if resolved_slug and resolved_slug.lower() != expected.lower():
|
||||
|
||||
if mode == MODE_DERIVATION:
|
||||
if not resolved_slug:
|
||||
reasons.append(
|
||||
f"canonical repository root identity mismatch: '{toplevel}' resolves "
|
||||
f"to repository '{resolved_slug}' but the session is authorized for "
|
||||
f"'{expected}' (forged or conflicting binding, fail closed)"
|
||||
f"configured canonical repository root '{toplevel}' has no resolvable "
|
||||
"git remote identity (fail closed)"
|
||||
)
|
||||
elif not resolved_slug and require_binding:
|
||||
else:
|
||||
# #973 B10: MODE_VALIDATION only. This arm is no longer a catch-all — the
|
||||
# allowlist above admits no third value, so an unsupported mode can no
|
||||
# longer silently receive validation semantics here.
|
||||
expected = (expected_slug or "").strip() or None
|
||||
if expected:
|
||||
if resolved_slug and resolved_slug.lower() != expected.lower():
|
||||
reasons.append(
|
||||
f"canonical repository root identity mismatch: '{toplevel}' resolves "
|
||||
f"to repository '{resolved_slug}' but the session is authorized for "
|
||||
f"'{expected}' (forged or conflicting binding, fail closed)"
|
||||
)
|
||||
elif not resolved_slug and require_binding:
|
||||
reasons.append(
|
||||
f"canonical repository root '{toplevel}' has no resolvable git "
|
||||
f"remote identity to confirm authorization for '{expected}' "
|
||||
"(fail closed)"
|
||||
)
|
||||
elif require_binding:
|
||||
reasons.append(
|
||||
f"canonical repository root '{toplevel}' has no resolvable git "
|
||||
f"remote identity to confirm authorization for '{expected}' "
|
||||
"(fail closed)"
|
||||
f"canonical repository root '{toplevel}' has configured value "
|
||||
f"'{configured_value}' but authoritative expected repository identity "
|
||||
"is unprovable or missing (fail closed)"
|
||||
)
|
||||
|
||||
return _assessment(
|
||||
@@ -252,10 +357,19 @@ def _assessment(
|
||||
proven: bool,
|
||||
reasons: list[str],
|
||||
configured: bool,
|
||||
canonical_repo_root: str,
|
||||
canonical_repo_root: str | None,
|
||||
resolved_slug: str | None,
|
||||
source: str | None,
|
||||
reason_code: str | None = None,
|
||||
) -> dict:
|
||||
"""Build the assessment payload.
|
||||
|
||||
``canonical_repo_root`` is None only when the assessment refused to resolve
|
||||
one at all — today exactly the unsupported-mode refusal (#973 B10), which
|
||||
must not derive a trusted repository identity through an undefined mode.
|
||||
``reason_code`` is a machine-checkable refusal cause; None for every
|
||||
ordinary (non-coded) outcome.
|
||||
"""
|
||||
return {
|
||||
"proven": proven,
|
||||
"block": not proven,
|
||||
@@ -264,4 +378,5 @@ def _assessment(
|
||||
"canonical_repo_root": canonical_repo_root,
|
||||
"resolved_slug": resolved_slug,
|
||||
"source": source,
|
||||
"reason_code": reason_code,
|
||||
}
|
||||
|
||||
+305
-226
@@ -27,12 +27,12 @@ import uuid
|
||||
from contextlib import contextmanager
|
||||
from dataclasses import dataclass
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Any, Iterator, Mapping, Sequence
|
||||
from typing import Any, Iterator, Sequence
|
||||
|
||||
import dependency_graph
|
||||
import gitea_audit
|
||||
|
||||
SCHEMA_VERSION = 6
|
||||
SCHEMA_VERSION = 5
|
||||
|
||||
# Assignable work kinds only — raw monitoring incidents are never work items.
|
||||
WORK_KINDS = frozenset({"issue", "pr"})
|
||||
@@ -280,6 +280,22 @@ def _ts(dt: datetime | None = None) -> str:
|
||||
return value.astimezone(timezone.utc).replace(microsecond=0).isoformat().replace("+00:00", "Z")
|
||||
|
||||
|
||||
def _realpath_or_raw(value: str | None) -> str:
|
||||
"""Normalize a filesystem path for compare-and-swap equality (#970).
|
||||
|
||||
Symlinks and ``..`` segments must not make two spellings of the same path
|
||||
look different, but an unresolvable path must still compare as itself
|
||||
rather than collapsing to empty — an empty result means "no path given".
|
||||
"""
|
||||
text = (value or "").strip()
|
||||
if not text:
|
||||
return ""
|
||||
try:
|
||||
return os.path.realpath(os.path.abspath(text))
|
||||
except Exception:
|
||||
return text
|
||||
|
||||
|
||||
def _parse_ts(value: str | None) -> datetime | None:
|
||||
if not value:
|
||||
return None
|
||||
@@ -419,7 +435,6 @@ class ControlPlaneDB:
|
||||
self._migrate_incident_links_null_scope(conn)
|
||||
self._migrate_lease_lifecycle_columns(conn)
|
||||
self._migrate_session_ownership_columns(conn)
|
||||
self._migrate_session_lifecycle_columns(conn)
|
||||
self._migrate_usage_events_table(conn)
|
||||
conn.execute(
|
||||
"INSERT OR REPLACE INTO schema_meta(key, value) VALUES (?, ?)",
|
||||
@@ -813,7 +828,6 @@ class ControlPlaneDB:
|
||||
pid: int | None = None,
|
||||
status: str = "active",
|
||||
controller_instance_id: str | None = None,
|
||||
owner_process_started_at: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Register/refresh a session row.
|
||||
|
||||
@@ -823,22 +837,16 @@ class ControlPlaneDB:
|
||||
controller instance can. It is never overwritten with ``None``, so a
|
||||
heartbeat from a caller that does not supply one cannot erase
|
||||
ownership.
|
||||
|
||||
*owner_process_started_at* (#969) is the OS start time of the owner
|
||||
process at registration time. When present it is retained across
|
||||
heartbeats (never cleared by ``None``) so later PID-reuse checks do
|
||||
not depend on a live ``ps`` probe of a long-dead process.
|
||||
"""
|
||||
now = _ts()
|
||||
instance = (controller_instance_id or "").strip() or None
|
||||
proc_start = (owner_process_started_at or "").strip() or None
|
||||
with self._tx() as conn:
|
||||
existing = conn.execute(
|
||||
"SELECT session_id FROM sessions WHERE session_id = ?",
|
||||
(session_id,),
|
||||
).fetchone()
|
||||
if existing:
|
||||
if instance is None and proc_start is None:
|
||||
if instance is None:
|
||||
conn.execute(
|
||||
"""
|
||||
UPDATE sessions
|
||||
@@ -848,21 +856,7 @@ class ControlPlaneDB:
|
||||
""",
|
||||
(role, profile, namespace, pid, now, status, session_id),
|
||||
)
|
||||
elif instance is None:
|
||||
conn.execute(
|
||||
"""
|
||||
UPDATE sessions
|
||||
SET role = ?, profile = ?, namespace = ?, pid = ?,
|
||||
last_heartbeat_at = ?, status = ?,
|
||||
owner_process_started_at = COALESCE(?, owner_process_started_at)
|
||||
WHERE session_id = ?
|
||||
""",
|
||||
(
|
||||
role, profile, namespace, pid, now, status,
|
||||
proc_start, session_id,
|
||||
),
|
||||
)
|
||||
elif proc_start is None:
|
||||
else:
|
||||
conn.execute(
|
||||
"""
|
||||
UPDATE sessions
|
||||
@@ -876,33 +870,18 @@ class ControlPlaneDB:
|
||||
instance, session_id,
|
||||
),
|
||||
)
|
||||
else:
|
||||
conn.execute(
|
||||
"""
|
||||
UPDATE sessions
|
||||
SET role = ?, profile = ?, namespace = ?, pid = ?,
|
||||
last_heartbeat_at = ?, status = ?,
|
||||
controller_instance_id = ?,
|
||||
owner_process_started_at = COALESCE(?, owner_process_started_at)
|
||||
WHERE session_id = ?
|
||||
""",
|
||||
(
|
||||
role, profile, namespace, pid, now, status,
|
||||
instance, proc_start, session_id,
|
||||
),
|
||||
)
|
||||
else:
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO sessions(
|
||||
session_id, role, profile, namespace, pid,
|
||||
started_at, last_heartbeat_at, status,
|
||||
controller_instance_id, owner_process_started_at
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
controller_instance_id
|
||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||
""",
|
||||
(
|
||||
session_id, role, profile, namespace, pid, now, now,
|
||||
status, instance, proc_start,
|
||||
status, instance,
|
||||
),
|
||||
)
|
||||
row = conn.execute(
|
||||
@@ -918,165 +897,6 @@ class ControlPlaneDB:
|
||||
(_ts(), session_id),
|
||||
)
|
||||
|
||||
def retire_session(
|
||||
self,
|
||||
*,
|
||||
session_id: str,
|
||||
reason: str,
|
||||
actor_session_id: str | None = None,
|
||||
details: Mapping[str, Any] | None = None,
|
||||
now: datetime | None = None,
|
||||
terminal_status: str = "retired",
|
||||
) -> dict[str, Any]:
|
||||
"""Terminalize one session row if it is still non-terminal (#969).
|
||||
|
||||
CAS on non-terminal status: concurrent retirements of the same row
|
||||
yield exactly one ``retired`` outcome and subsequent
|
||||
``already_terminal`` outcomes. Never deletes historical rows. Writes a
|
||||
durable ``session_retired`` event for audit.
|
||||
"""
|
||||
moment = _ts(now)
|
||||
reason_s = (reason or "").strip() or "unspecified"
|
||||
term = (terminal_status or "retired").strip().lower() or "retired"
|
||||
terminal_set = {
|
||||
"retired",
|
||||
"ended",
|
||||
"terminal",
|
||||
"dead",
|
||||
"stale",
|
||||
"orphaned",
|
||||
}
|
||||
with self._tx() as conn:
|
||||
row = conn.execute(
|
||||
"SELECT * FROM sessions WHERE session_id = ?",
|
||||
(session_id,),
|
||||
).fetchone()
|
||||
if row is None:
|
||||
return {
|
||||
"outcome": "missing",
|
||||
"reason": reason_s,
|
||||
"prior_status": None,
|
||||
"new_status": None,
|
||||
"details": {"session_id": session_id},
|
||||
}
|
||||
prior = dict(row)
|
||||
prior_status = str(prior.get("status") or "").strip().lower()
|
||||
if prior_status in terminal_set:
|
||||
return {
|
||||
"outcome": "already_terminal",
|
||||
"reason": reason_s,
|
||||
"prior_status": prior_status,
|
||||
"new_status": prior_status,
|
||||
"details": {"session_id": session_id, "idempotent": True},
|
||||
}
|
||||
|
||||
# Optional: refuse when an active lease still names this session.
|
||||
active_lease = conn.execute(
|
||||
"""
|
||||
SELECT lease_id, status, expires_at FROM leases
|
||||
WHERE session_id = ? AND status = 'active'
|
||||
LIMIT 1
|
||||
""",
|
||||
(session_id,),
|
||||
).fetchone()
|
||||
if active_lease is not None:
|
||||
return {
|
||||
"outcome": "blocked",
|
||||
"reason": "live_lease",
|
||||
"prior_status": prior_status,
|
||||
"new_status": prior_status,
|
||||
"details": {
|
||||
"session_id": session_id,
|
||||
"lease_id": active_lease["lease_id"],
|
||||
"blocker": "active_lease_row",
|
||||
},
|
||||
}
|
||||
|
||||
cols = {
|
||||
r[1] for r in conn.execute("PRAGMA table_info(sessions)").fetchall()
|
||||
}
|
||||
if "retired_at" in cols and "retire_reason" in cols:
|
||||
conn.execute(
|
||||
"""
|
||||
UPDATE sessions
|
||||
SET status = ?, retired_at = ?, retire_reason = ?
|
||||
WHERE session_id = ? AND status = ?
|
||||
""",
|
||||
(term, moment, reason_s, session_id, prior.get("status")),
|
||||
)
|
||||
else:
|
||||
conn.execute(
|
||||
"""
|
||||
UPDATE sessions
|
||||
SET status = ?
|
||||
WHERE session_id = ? AND status = ?
|
||||
""",
|
||||
(term, session_id, prior.get("status")),
|
||||
)
|
||||
changed = conn.execute(
|
||||
"SELECT changes()"
|
||||
).fetchone()[0]
|
||||
if not changed:
|
||||
# Lost CAS race — re-read.
|
||||
refreshed = conn.execute(
|
||||
"SELECT status FROM sessions WHERE session_id = ?",
|
||||
(session_id,),
|
||||
).fetchone()
|
||||
cur = (
|
||||
str(refreshed["status"]).strip().lower()
|
||||
if refreshed is not None
|
||||
else None
|
||||
)
|
||||
return {
|
||||
"outcome": "already_terminal"
|
||||
if cur in terminal_set
|
||||
else "blocked",
|
||||
"reason": reason_s,
|
||||
"prior_status": prior_status,
|
||||
"new_status": cur,
|
||||
"details": {
|
||||
"session_id": session_id,
|
||||
"cas_lost": True,
|
||||
},
|
||||
}
|
||||
|
||||
detail_payload: dict[str, Any] = {
|
||||
"session_id": session_id,
|
||||
"prior_status": prior_status,
|
||||
"new_status": term,
|
||||
"reason": reason_s,
|
||||
"actor_session_id": actor_session_id,
|
||||
"pid": prior.get("pid"),
|
||||
"role": prior.get("role"),
|
||||
"profile": prior.get("profile"),
|
||||
}
|
||||
if isinstance(details, Mapping):
|
||||
for key, value in details.items():
|
||||
if key not in detail_payload:
|
||||
detail_payload[key] = value
|
||||
message = (
|
||||
f"session {session_id} retired reason={reason_s} "
|
||||
f"prior_status={prior_status}"
|
||||
)
|
||||
# events.work_item_id is nullable; session retirement is not work-scoped.
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO events(work_item_id, event_type, message, created_at)
|
||||
VALUES (NULL, 'session_retired', ?, ?)
|
||||
""",
|
||||
(message[:2000], moment),
|
||||
)
|
||||
# Also persist a structured JSON line in the message when short enough
|
||||
# by appending a compact summary (full detail stays in return value /
|
||||
# audit log; events.message is human-readable).
|
||||
return {
|
||||
"outcome": "retired",
|
||||
"reason": reason_s,
|
||||
"prior_status": prior_status,
|
||||
"new_status": term,
|
||||
"details": detail_payload,
|
||||
}
|
||||
|
||||
def list_sessions(
|
||||
self,
|
||||
*,
|
||||
@@ -1835,29 +1655,6 @@ class ControlPlaneDB:
|
||||
if name not in cols:
|
||||
conn.execute(f"ALTER TABLE sessions ADD COLUMN {name} {decl}")
|
||||
|
||||
_SESSION_LIFECYCLE_COLUMNS: tuple[tuple[str, str], ...] = (
|
||||
("owner_process_started_at", "TEXT"),
|
||||
("retired_at", "TEXT"),
|
||||
("retire_reason", "TEXT"),
|
||||
)
|
||||
|
||||
def _migrate_session_lifecycle_columns(self, conn: sqlite3.Connection) -> None:
|
||||
"""Add session retirement / PID-reuse provenance columns (#969).
|
||||
|
||||
Additive and idempotent. Pre-existing rows migrate with NULL; retirement
|
||||
fills ``retired_at`` / ``retire_reason``, and new upserts may record
|
||||
``owner_process_started_at`` for stronger identity checks.
|
||||
"""
|
||||
cols = {
|
||||
row[1]
|
||||
for row in conn.execute("PRAGMA table_info(sessions)").fetchall()
|
||||
}
|
||||
if not cols:
|
||||
return
|
||||
for name, decl in self._SESSION_LIFECYCLE_COLUMNS:
|
||||
if name not in cols:
|
||||
conn.execute(f"ALTER TABLE sessions ADD COLUMN {name} {decl}")
|
||||
|
||||
def list_active_claims(
|
||||
self,
|
||||
*,
|
||||
@@ -2132,6 +1929,168 @@ class ControlPlaneDB:
|
||||
),
|
||||
)
|
||||
|
||||
def retire_lease_worktree_path(
|
||||
self,
|
||||
lease_id: str,
|
||||
*,
|
||||
expected_path: str | None = None,
|
||||
expected_status: str | None = None,
|
||||
expected_session_id: str | None = None,
|
||||
expected_owner_pid: int | None = None,
|
||||
reason: str = "missing_worktree_path_retired",
|
||||
) -> dict[str, Any]:
|
||||
"""Retire a missing worktree_path binding from a control-plane lease (#970).
|
||||
|
||||
Clears worktree_path on the lease row, updates provenance_json with
|
||||
durable retirement audit proof, and writes a worktree_binding_retired
|
||||
event.
|
||||
|
||||
The update is a compare-and-swap (#970 review 644 B1/B3): the caller
|
||||
states the exact path it audited and, when known, the lease status,
|
||||
owning session, and owner pid it classified against. Every stated value
|
||||
must still match the stored row, and the ``UPDATE`` itself is keyed on
|
||||
the stored ``worktree_path``, so a concurrent writer that moved or
|
||||
replaced the binding between audit and apply loses the race instead of
|
||||
having its value silently overwritten. A mismatch raises and mutates
|
||||
nothing.
|
||||
|
||||
``expected_path`` is mandatory: a retirement that does not name the path
|
||||
it intends to clear cannot be safe against concurrent recreation.
|
||||
"""
|
||||
now_s = _ts()
|
||||
expected_norm = _realpath_or_raw(expected_path)
|
||||
if not expected_norm:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire lease {lease_id} worktree_path: expected_path is "
|
||||
"required for compare-and-swap retirement (fail closed)"
|
||||
)
|
||||
with self._tx() as conn:
|
||||
cols = self._lease_columns(conn)
|
||||
row = conn.execute(
|
||||
"SELECT * FROM leases WHERE lease_id = ?",
|
||||
(lease_id,),
|
||||
).fetchone()
|
||||
if not row:
|
||||
raise ControlPlaneError(f"unknown lease_id {lease_id}")
|
||||
|
||||
record = dict(row)
|
||||
current_wt = (record.get("worktree_path") or "").strip()
|
||||
|
||||
if not current_wt:
|
||||
# Idempotent: the binding this caller audited is already gone.
|
||||
return {
|
||||
"lease_id": lease_id,
|
||||
"retired": False,
|
||||
"already_retired": True,
|
||||
"prior_worktree_path": "",
|
||||
"expected_worktree_path": expected_path,
|
||||
"reason": reason,
|
||||
"compare_and_swap": {
|
||||
"matched": True,
|
||||
"outcome": "already_retired",
|
||||
},
|
||||
}
|
||||
|
||||
if _realpath_or_raw(current_wt) != expected_norm:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire lease {lease_id} worktree_path: expected "
|
||||
f"'{expected_path}' does not match current '{current_wt}' "
|
||||
"(fail closed)"
|
||||
)
|
||||
|
||||
for field, expected_value in (
|
||||
("status", expected_status),
|
||||
("session_id", expected_session_id),
|
||||
):
|
||||
if expected_value is None:
|
||||
continue
|
||||
current_value = record.get(field)
|
||||
if str(current_value or "").strip() != str(expected_value).strip():
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire lease {lease_id} worktree_path: lease "
|
||||
f"{field} changed since audit (expected "
|
||||
f"'{expected_value}', found '{current_value}'); fail closed"
|
||||
)
|
||||
|
||||
if expected_owner_pid is not None:
|
||||
current_pid = record.get("owner_pid")
|
||||
if current_pid is not None and int(current_pid) != int(expected_owner_pid):
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire lease {lease_id} worktree_path: lease "
|
||||
f"owner_pid changed since audit (expected "
|
||||
f"{expected_owner_pid}, found {current_pid}); fail closed"
|
||||
)
|
||||
|
||||
# Parse and update provenance_json
|
||||
raw_prov = record.get("provenance_json") or "{}"
|
||||
try:
|
||||
prov = json.loads(raw_prov) if isinstance(raw_prov, str) else dict(raw_prov)
|
||||
except Exception:
|
||||
prov = {}
|
||||
if not isinstance(prov, dict):
|
||||
prov = {}
|
||||
|
||||
prior_path = current_wt
|
||||
prov.update({
|
||||
"worktree_path_retired": True,
|
||||
"retired_worktree_path": prior_path,
|
||||
"retired_at": now_s,
|
||||
"retirement_reason": reason,
|
||||
"retired_from_status": record.get("status"),
|
||||
"retired_from_session_id": record.get("session_id"),
|
||||
"worktree_path": "",
|
||||
})
|
||||
prov_json = json.dumps(prov)
|
||||
|
||||
if "worktree_path" in cols:
|
||||
# CAS: keyed on the exact stored path this caller audited.
|
||||
cur = conn.execute(
|
||||
"UPDATE leases SET worktree_path = '', provenance_json = ? "
|
||||
"WHERE lease_id = ? AND worktree_path = ?",
|
||||
(prov_json, lease_id, record.get("worktree_path")),
|
||||
)
|
||||
if cur.rowcount != 1:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire lease {lease_id} worktree_path: "
|
||||
"compare-and-swap matched no row (concurrent change); "
|
||||
"fail closed"
|
||||
)
|
||||
else:
|
||||
conn.execute(
|
||||
"UPDATE leases SET provenance_json = ? WHERE lease_id = ?",
|
||||
(prov_json, lease_id),
|
||||
)
|
||||
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO events(work_item_id, event_type, message, created_at)
|
||||
VALUES (?, 'worktree_binding_retired', ?, ?)
|
||||
""",
|
||||
(
|
||||
record["work_item_id"],
|
||||
f"lease {lease_id} worktree_path '{prior_path}' retired: {reason}",
|
||||
now_s,
|
||||
),
|
||||
)
|
||||
|
||||
return {
|
||||
"lease_id": lease_id,
|
||||
"retired": True,
|
||||
"already_retired": False,
|
||||
"prior_worktree_path": prior_path,
|
||||
"expected_worktree_path": expected_path,
|
||||
"retired_at": now_s,
|
||||
"reason": reason,
|
||||
"compare_and_swap": {
|
||||
"matched": True,
|
||||
"outcome": "retired",
|
||||
"expected_status": expected_status,
|
||||
"expected_session_id": expected_session_id,
|
||||
"expected_owner_pid": expected_owner_pid,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def abandon_lease(
|
||||
self,
|
||||
*,
|
||||
@@ -3270,3 +3229,123 @@ class ControlPlaneDB:
|
||||
"live_lease_id": None if live_lease_id is None else str(live_lease_id),
|
||||
"reconcile_action": "reconcile_required" if stale else "safe_to_resume",
|
||||
}
|
||||
|
||||
def retire_session_checkpoint_worktree_path(
|
||||
self,
|
||||
session_id: str,
|
||||
*,
|
||||
checkpoint_id: str | None = None,
|
||||
expected_path: str | None = None,
|
||||
expected_status: str | None = None,
|
||||
reason: str = "missing_worktree_path_retired",
|
||||
) -> dict[str, Any]:
|
||||
"""Retire a missing worktree_path from session_checkpoints (#970).
|
||||
|
||||
Compare-and-swap, mirroring :meth:`retire_lease_worktree_path` (#970
|
||||
review 644 B3). ``expected_path`` names the exact stored path the caller
|
||||
audited; the guarded ``UPDATE`` is keyed on that stored value, so a
|
||||
checkpoint whose path was moved, replaced, or concurrently rewritten
|
||||
after the audit is refused without mutation rather than blindly cleared.
|
||||
|
||||
Exactly one checkpoint row is targeted: by ``checkpoint_id`` when given,
|
||||
otherwise by ``session_id``, which must identify a single row.
|
||||
"""
|
||||
now_s = _ts()
|
||||
expected_norm = _realpath_or_raw(expected_path)
|
||||
if not expected_norm:
|
||||
raise ControlPlaneError(
|
||||
"cannot retire session checkpoint worktree_path: expected_path "
|
||||
"is required for compare-and-swap retirement (fail closed)"
|
||||
)
|
||||
if not checkpoint_id and not (session_id or "").strip():
|
||||
raise ControlPlaneError(
|
||||
"cannot retire session checkpoint worktree_path: checkpoint_id "
|
||||
"or session_id is required (fail closed)"
|
||||
)
|
||||
|
||||
with self._tx() as conn:
|
||||
if checkpoint_id:
|
||||
selector_sql = "SELECT * FROM session_checkpoints WHERE checkpoint_id = ?"
|
||||
selector_params: tuple[Any, ...] = (checkpoint_id,)
|
||||
selector_desc = f"checkpoint_id '{checkpoint_id}'"
|
||||
else:
|
||||
selector_sql = "SELECT * FROM session_checkpoints WHERE session_id = ?"
|
||||
selector_params = (session_id,)
|
||||
selector_desc = f"session_id '{session_id}'"
|
||||
|
||||
rows = [dict(r) for r in conn.execute(selector_sql, selector_params).fetchall()]
|
||||
if not rows:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire session checkpoint worktree_path: no "
|
||||
f"checkpoint matches {selector_desc} (fail closed)"
|
||||
)
|
||||
if len(rows) > 1:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire session checkpoint worktree_path: "
|
||||
f"{selector_desc} matches {len(rows)} checkpoints; supply an "
|
||||
"exact checkpoint_id (fail closed)"
|
||||
)
|
||||
|
||||
record = rows[0]
|
||||
target_checkpoint_id = record.get("checkpoint_id")
|
||||
current_wt = (record.get("worktree_path") or "").strip()
|
||||
|
||||
if not current_wt:
|
||||
# Idempotent: the binding this caller audited is already gone.
|
||||
return {
|
||||
"session_id": session_id,
|
||||
"checkpoint_id": target_checkpoint_id,
|
||||
"retired": False,
|
||||
"already_retired": True,
|
||||
"prior_worktree_path": "",
|
||||
"expected_worktree_path": expected_path,
|
||||
"reason": reason,
|
||||
"compare_and_swap": {
|
||||
"matched": True,
|
||||
"outcome": "already_retired",
|
||||
},
|
||||
}
|
||||
|
||||
if _realpath_or_raw(current_wt) != expected_norm:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire session checkpoint worktree_path for "
|
||||
f"{selector_desc}: expected '{expected_path}' does not match "
|
||||
f"current '{current_wt}' (fail closed)"
|
||||
)
|
||||
|
||||
if expected_status is not None:
|
||||
current_status = record.get("status")
|
||||
if str(current_status or "").strip() != str(expected_status).strip():
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire session checkpoint worktree_path for "
|
||||
f"{selector_desc}: status changed since audit (expected "
|
||||
f"'{expected_status}', found '{current_status}'); fail closed"
|
||||
)
|
||||
|
||||
cur = conn.execute(
|
||||
"UPDATE session_checkpoints SET worktree_path = '', updated_at = ? "
|
||||
"WHERE checkpoint_id = ? AND worktree_path = ?",
|
||||
(now_s, target_checkpoint_id, record.get("worktree_path")),
|
||||
)
|
||||
if cur.rowcount != 1:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire session checkpoint worktree_path for "
|
||||
f"{selector_desc}: compare-and-swap matched no row "
|
||||
"(concurrent change); fail closed"
|
||||
)
|
||||
|
||||
return {
|
||||
"session_id": session_id,
|
||||
"checkpoint_id": target_checkpoint_id,
|
||||
"retired": True,
|
||||
"already_retired": False,
|
||||
"prior_worktree_path": current_wt,
|
||||
"expected_worktree_path": expected_path,
|
||||
"retired_at": now_s,
|
||||
"reason": reason,
|
||||
"compare_and_swap": {
|
||||
"matched": True,
|
||||
"outcome": "retired",
|
||||
"expected_status": expected_status,
|
||||
},
|
||||
}
|
||||
|
||||
@@ -0,0 +1,203 @@
|
||||
# Instance-level fleet identity and health snapshots
|
||||
|
||||
Issue **#978**. Companion primitives: **#948** (worker ownership), **#975**
|
||||
(heartbeat lifecycle), **#951** (restart receipts).
|
||||
|
||||
## Why this exists
|
||||
|
||||
The multi-instance fleet gate (#963) needs a *native* answer to:
|
||||
|
||||
* which application launches are live;
|
||||
* which of the five namespace workers belong to each launch;
|
||||
* whether two Codex (or Claude, Grok, …) launches are distinct;
|
||||
* whether any live collision is real rather than a shared client type.
|
||||
|
||||
Process tables, PID proximity, configuration files, and direct SQLite access
|
||||
are **not** production evidence. The sanctioned surface is the read-only MCP
|
||||
tool `gitea_snapshot_instance_fleet` on **controller** and **reconciler**
|
||||
namespaces.
|
||||
|
||||
## Identity hierarchy
|
||||
|
||||
| Identity | Scope | Who generates it | Lifetime |
|
||||
| --- | --- | --- | --- |
|
||||
| `client_type` | Application family (`codex`, `claude_code`, `gemini`, `grok`, …) | Launcher sets `GITEA_MCP_CLIENT` | Stable for the product |
|
||||
| `client_instance_id` | One running application launch | **Trusted launcher**, once per launch, as `GITEA_MCP_CLIENT_INSTANCE` | Fresh launch → new ID; reconnect of same launch → same ID; full app restart → new ID |
|
||||
| `fleet_run_id` | Operator-approved enrollment / canary cohort | Operator / controller sets `GITEA_MCP_FLEET_RUN_ID` | Duration of the approved rollout |
|
||||
| `namespace` | `author` \| `reviewer` \| `merger` \| `controller` \| `reconciler` | Profile / MCP server binding | Process lifetime |
|
||||
| `worker_id` / `worker_identity` | One namespace worker process | Runtime registry at first registration | New on worker restart; not reused |
|
||||
| `session_id` | Client session ownership | Launcher `GITEA_MCP_CLIENT_SESSION` or runtime | Session lifetime |
|
||||
| `generation_id` | One daemon launch | Runtime at process boot | New on process restart |
|
||||
| `process_identity` / PID | OS process | Runtime | Process lifetime |
|
||||
|
||||
### Rules
|
||||
|
||||
1. Multiple active instances **may** share the same `client_type`.
|
||||
2. Every application launch receives a **distinct** `client_instance_id`.
|
||||
3. All five namespace workers of one launch report the **same**
|
||||
`client_instance_id`.
|
||||
4. Each namespace worker has a **distinct** `worker_identity`, process
|
||||
identity, generation, and PID.
|
||||
5. Instance identity is **never** inferred from PID proximity, timestamps, or
|
||||
client type alone.
|
||||
6. Live reuse of one `client_instance_id` with conflicting workers fails closed.
|
||||
7. Sharing only a profile or `client_type` is **not** a duplicate.
|
||||
|
||||
This deliberately **replaces** any permanent `exactly_one_per_profile` fleet
|
||||
model (#949 assumption) as the operating rule for multi-instance fleets.
|
||||
|
||||
## How five workers join one instance
|
||||
|
||||
1. The host starts one application instance (for example one Codex session).
|
||||
2. The **production application launcher**
|
||||
(`mcp_application_launcher.build_application_mcp_servers` /
|
||||
`gitea_config.multi_namespace_launcher_entries`) mints exactly one
|
||||
`client_instance_id` via `mcp_fleet_snapshot.generate_client_instance_id`
|
||||
and injects the same env into every namespace worker:
|
||||
|
||||
```text
|
||||
GITEA_MCP_CLIENT=<client_type>
|
||||
GITEA_MCP_CLIENT_INSTANCE=<client_instance_id>
|
||||
GITEA_MCP_INSTANCE_PROVENANCE=trusted_launcher
|
||||
GITEA_MCP_CLIENT_SESSION=<session_id> # optional but recommended
|
||||
GITEA_MCP_FLEET_RUN_ID=<enrollment id> # when on an approved canary
|
||||
GITEA_CLIENT_MANAGED=1
|
||||
GITEA_MCP_PROFILE=<role-profile>
|
||||
```
|
||||
|
||||
3. The MCP client starts the five namespace processes (`gitea-author`,
|
||||
`gitea-reviewer`, `gitea-merger`, `gitea-controller`, `gitea-reconciler`)
|
||||
from that generated `mcpServers` map — each entry carries the **same**
|
||||
instance ID.
|
||||
4. Each worker registers once into the worker registry with its own
|
||||
`worker_identity`, `namespace`, `generation_id`, and PID, but the shared
|
||||
`client_instance_id`. Workers never mint a trusted instance ID themselves.
|
||||
|
||||
Only IDs matching the launcher format `inst-<client>-<timestamp>-<digest>`
|
||||
are trusted. Missing, `legacy-pid-*` / `pid-*` placeholders, and ordinary
|
||||
user-supplied strings fail closed as untrusted and cannot authorize
|
||||
multi-instance fleet mutation safety.
|
||||
|
||||
Worker reconnect (same process, same registration) keeps the instance ID.
|
||||
Worker restart (new process) mints a new worker identity and generation but
|
||||
must still receive the same `GITEA_MCP_CLIENT_INSTANCE` from the parent
|
||||
application if it is the same launch. Full application restart mints a new
|
||||
`client_instance_id` (call the launcher without a prior ID).
|
||||
|
||||
### Mutation gate (instance-aware)
|
||||
|
||||
The capability / runtime diagnostic gate
|
||||
(`_check_mcp_runtimes_diagnostics`) is **instance-aware**:
|
||||
|
||||
* Two legitimate application instances that share a profile (same
|
||||
`GITEA_MCP_PROFILE`) are allowed when each has a distinct trusted
|
||||
`client_instance_id`.
|
||||
* More than one live worker for the same
|
||||
`(client_instance_id, profile/namespace)` fails closed as a duplicate
|
||||
namespace worker.
|
||||
* Multiple processes sharing a profile **without** trusted instance
|
||||
evidence still fail closed (indistinguishable from a duplicate).
|
||||
|
||||
## Approved fleet manifest
|
||||
|
||||
An operator (or controller enrollment step) obtains an approved manifest as a
|
||||
list of expected instances, for example:
|
||||
|
||||
```json
|
||||
{
|
||||
"instances": [
|
||||
{
|
||||
"client_type": "codex",
|
||||
"client_instance_id": "inst-codex-20260730T120000Z-abc123def456",
|
||||
"fleet_run_id": "canary-963-2026-07-30",
|
||||
"namespaces": ["author", "reviewer", "merger", "controller", "reconciler"]
|
||||
},
|
||||
{
|
||||
"client_type": "codex",
|
||||
"client_instance_id": "inst-codex-20260730T120100Z-fed654cba321",
|
||||
"fleet_run_id": "canary-963-2026-07-30",
|
||||
"namespaces": ["author", "reviewer", "merger", "controller", "reconciler"]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
Pass that list as `expected_manifest` to `gitea_snapshot_instance_fleet`.
|
||||
When a manifest is supplied:
|
||||
|
||||
* instances on the manifest but not live → `missing_expected`;
|
||||
* live instances not on the manifest → `unmanifested`;
|
||||
* both are **active blockers** for fleet-gate safety.
|
||||
|
||||
Without a manifest, the snapshot still enumerates the live fleet and classifies
|
||||
identity collisions; it does not invent enrollment policy.
|
||||
|
||||
## Snapshot consistency
|
||||
|
||||
Each snapshot includes:
|
||||
|
||||
* `snapshot_at` — UTC timestamp;
|
||||
* `consistency_token` / `registry_revision` — content digest over worker
|
||||
identity, instance id, status, heartbeat, fencing, and generation.
|
||||
|
||||
Two successive snapshots can prove heartbeat continuity via
|
||||
`mcp_fleet_snapshot.compare_snapshot_heartbeats`.
|
||||
|
||||
## Classification (active vs historical)
|
||||
|
||||
**Active blockers** (make `live_fleet_safe=false`):
|
||||
|
||||
* missing expected instance;
|
||||
* unmanifested extra instance;
|
||||
* duplicate namespace worker within one instance;
|
||||
* live `client_instance_id` collision;
|
||||
* reused worker / session / generation / process / PID / fencing identity;
|
||||
* orphaned or unowned workers;
|
||||
* unknown client;
|
||||
* foreign-repository workers;
|
||||
* old-revision workers;
|
||||
* stale workers still marked active;
|
||||
* legacy incomplete instance identity.
|
||||
|
||||
**Historical dead rows** (`status` released/superseded) are reported as
|
||||
`historical_dead` findings with `active_blocker=false`. They never
|
||||
automatically make the live fleet unsafe.
|
||||
|
||||
## Fail-closed behaviour and recovery
|
||||
|
||||
| Condition | Mutation safety | Diagnostic reads | Recovery |
|
||||
| --- | --- | --- | --- |
|
||||
| Two instances, same `client_type`, distinct IDs | Safe (if otherwise healthy) | Available | None needed |
|
||||
| Live reuse of one `client_instance_id` | Unsafe | Available | Stop the colliding launch or re-issue a distinct ID |
|
||||
| Two workers, same namespace, one instance | Unsafe | Available | Stop the extra worker |
|
||||
| Legacy registration without trusted instance ID | Unsafe for fleet mutation | Available | Relaunch with launcher-issued `GITEA_MCP_CLIENT_INSTANCE` |
|
||||
| Historical dead row only | Does not block alone | Available | No action required for fleet safety |
|
||||
| Registry unavailable | Snapshot fails closed | N/A | Repair registry path / reconnect namespaces |
|
||||
|
||||
Incomplete or untrusted instance identity **cannot** authorize unsafe
|
||||
mutation. Diagnostic reads remain available where `gitea.read` allows.
|
||||
|
||||
## Permissions
|
||||
|
||||
* **Allowed:** `controller`, `reconciler` with `gitea.read`.
|
||||
* **Denied:** author, reviewer, merger (even with `gitea.read` for other tools).
|
||||
* **No new mutation permissions** are granted to any role.
|
||||
|
||||
## Related surfaces
|
||||
|
||||
* `mcp_worker_identity` — registry, worker identity, heartbeats (#948, #975).
|
||||
* `mcp_fleet_snapshot` — pure snapshot + classification (#978).
|
||||
* `gitea_snapshot_instance_fleet` — sanctioned MCP tool (#978).
|
||||
* `gitea_get_runtime_context` — single-process view (not fleet-wide).
|
||||
* [stale-worker-retirement.md](stale-worker-retirement.md) — CAS-protected
|
||||
retirement of conclusively stale registrations (#980). Note that the
|
||||
`registry_revision` this snapshot returns is **time-seeded** and unsuitable
|
||||
for compare-and-swap; retirement derives its own stable
|
||||
`registry_fingerprint` from registry content alone.
|
||||
|
||||
## Non-goals
|
||||
|
||||
* Starting the #963 canary.
|
||||
* Purging historical registry rows.
|
||||
* Preserving “exactly one process per profile” as the permanent model.
|
||||
* Using shell / process-table / SQLite inspection as production fleet evidence.
|
||||
@@ -51,6 +51,7 @@ that gates each call, not which tools exist.
|
||||
- `gitea_adopt_merger_pr_lease`
|
||||
- `gitea_adopt_workflow_lease`
|
||||
- `gitea_allocate_next_work`
|
||||
- `gitea_apply_stale_worker_retirement`
|
||||
- `gitea_assess_already_landed_reconciliation`
|
||||
- `gitea_assess_conflict_fix_classification`
|
||||
- `gitea_assess_conflict_fix_push`
|
||||
@@ -65,6 +66,7 @@ that gates each call, not which tools exist.
|
||||
- `gitea_assess_work_issue_duplicate`
|
||||
- `gitea_assess_worktree_cleanup_integrity`
|
||||
- `gitea_audit_config`
|
||||
- `gitea_audit_missing_worktree_bindings`
|
||||
- `gitea_audit_runtime_recovery_contamination`
|
||||
- `gitea_audit_stable_branch_contamination`
|
||||
- `gitea_audit_worktree_cleanup`
|
||||
@@ -123,19 +125,24 @@ that gates each call, not which tools exist.
|
||||
- `gitea_observability_link_issue`
|
||||
- `gitea_observability_list_projects`
|
||||
- `gitea_observability_reconcile_incident`
|
||||
- `gitea_plan_stale_worker_retirement`
|
||||
- `gitea_post_heartbeat`
|
||||
- `gitea_publish_unpublished_issue_branch`
|
||||
- `gitea_quarantine_contaminated_review`
|
||||
- `gitea_rebind_dirty_same_claimant_author_session`
|
||||
- `gitea_reclaim_expired_workflow_lease`
|
||||
- `gitea_reconcile_after_restart`
|
||||
- `gitea_reconcile_already_landed_pr`
|
||||
- `gitea_reconcile_issue_claims`
|
||||
- `gitea_reconcile_merged_cleanups`
|
||||
- `gitea_reconcile_missing_worktree_bindings`
|
||||
- `gitea_reconcile_superseded_by_merged_pr`
|
||||
- `gitea_record_daemon_process_kill_attempt`
|
||||
- `gitea_record_irrecoverable_decision_lock_provenance`
|
||||
- `gitea_record_pre_review_command`
|
||||
- `gitea_record_shell_spawn_outcome`
|
||||
- `gitea_record_stable_branch_push_attempt`
|
||||
- `gitea_recover_dirty_orphaned_issue_worktree`
|
||||
- `gitea_recover_incomplete_bootstrap_lock`
|
||||
- `gitea_release_merger_pr_lease`
|
||||
- `gitea_release_reviewer_pr_lease`
|
||||
@@ -154,6 +161,7 @@ that gates each call, not which tools exist.
|
||||
- `gitea_sentry_reconcile_issue`
|
||||
- `gitea_sentry_watchdog`
|
||||
- `gitea_set_issue_labels`
|
||||
- `gitea_snapshot_instance_fleet`
|
||||
- `gitea_submit_pr_review`
|
||||
- `gitea_update_pr_branch_by_merge`
|
||||
- `gitea_validate_review_final_report`
|
||||
|
||||
@@ -19,11 +19,7 @@ The assessor classifies:
|
||||
|
||||
- **service_health** — process healthy / parity mutation-safe
|
||||
- **clients** — connected client descriptors (optional inventory)
|
||||
- **sessions** — active session rows with dead or reused owner pids are
|
||||
unresolved until retired via `#969` (`session_lifecycle` /
|
||||
`apply_session_cleanup=true` on `gitea_reconcile_after_restart`, or
|
||||
`gitea_retire_stale_workflow_sessions`). Live owners, live leases, and live
|
||||
client-managed sessions are never retired.
|
||||
- **sessions** — active session rows with dead owner pids are unresolved
|
||||
- **checkpoints** — soft-depends on #660; skipped with reason when schema absent
|
||||
- **leases** — live control-plane leases after restart
|
||||
- **capabilities** — master-parity / stale-runtime (#610)
|
||||
|
||||
@@ -0,0 +1,282 @@
|
||||
# Stale worker retirement (#980)
|
||||
|
||||
The #978 instance-fleet snapshot made registry accuracy observable but
|
||||
deliberately read-only: a registry full of rows whose owning processes are long
|
||||
gone stays full. This document describes the sanctioned way to retire those
|
||||
rows — a dry-run-first, compare-and-swap-protected workflow available only to
|
||||
controller and reconciler namespaces.
|
||||
|
||||
Related: [instance-fleet-identity.md](instance-fleet-identity.md) (#978),
|
||||
[post-restart-reconcile.md](post-restart-reconcile.md) (#662).
|
||||
|
||||
## Why a dedicated registry token
|
||||
|
||||
`mcp_fleet_snapshot._consistency_token` seeds its digest with `snapshot_at`,
|
||||
formatted at second precision. Its output — surfaced as `registry_revision` and
|
||||
`consistency_token` on the snapshot — therefore changes on **every call**, even
|
||||
when no registry row changed. Any compare-and-swap gated on it can never pass:
|
||||
a dry-run/apply cycle spanning more than one second aborts unconditionally.
|
||||
|
||||
That token remains useful as an observation stamp and is unchanged. #980 adds a
|
||||
separate, *stable* token instead:
|
||||
|
||||
| Token | Module | Derived from | Stable across time? |
|
||||
| --- | --- | --- | --- |
|
||||
| `registry_revision` / `consistency_token` | `mcp_fleet_snapshot` | `snapshot_at` + a subset of row fields | **No** — moves every second |
|
||||
| `registry_fingerprint` | `mcp_fleet_retirement` | canonical retirement-relevant row content only | **Yes** |
|
||||
| `candidate_fingerprint` | `mcp_fleet_retirement` | canonical content of the selected candidate rows | **Yes** |
|
||||
|
||||
`registry_fingerprint` guarantees:
|
||||
|
||||
* identical canonical registry contents always produce the same token, whenever
|
||||
they are observed;
|
||||
* row iteration order never affects the token (serialized rows are sorted);
|
||||
* any retirement-relevant change moves it — row creation or deletion, identity
|
||||
change, heartbeat or TTL change, ownership change, registration-state change,
|
||||
PID change, or repository-binding change.
|
||||
|
||||
The exact field set is `mcp_fleet_retirement.FINGERPRINT_FIELDS`. Deliberately
|
||||
excluded: `token_fingerprint` (credential-adjacent, never a retirement input),
|
||||
the four `*_revision` columns (revision drift is an independent restart concern
|
||||
and is not part of the eligibility conjunction), and the `retired_*` bookkeeping
|
||||
columns this feature adds. Numeric values are canonicalised, so a TTL that
|
||||
round-trips through SQLite as `900.0` hashes identically to `900`.
|
||||
|
||||
## Eligibility — the conjunction
|
||||
|
||||
A registration is retired only when **every** one of these holds. Any missing
|
||||
or contradictory evidence preserves the row.
|
||||
|
||||
| Requirement | Preserve reason code when it fails |
|
||||
| --- | --- |
|
||||
| `status` is `active` | `already_terminal_registration` |
|
||||
| Every field the conjunction reads is present (`REQUIRED_IDENTITY_FIELDS`) | `incomplete_registry_identity` |
|
||||
| `last_heartbeat_at` parses as a UTC stamp | `unparsable_heartbeat` |
|
||||
| Worker is not live | `worker_live` |
|
||||
| PID probe returns a definite answer | `pid_liveness_unknown` |
|
||||
| PID probe says the process is gone | `pid_alive` |
|
||||
| Heartbeat has expired under the canonical TTL | `heartbeat_not_expired` |
|
||||
| `ownership_state` is exactly `stale` | `ambiguous_ownership_state` |
|
||||
| Repository binding present and canonical | `repository_binding_ambiguous` |
|
||||
| No identity evidence shared with a live or unprobeable worker | `conflicting_identity_evidence` |
|
||||
| No other active row claims the same (instance, namespace) while one may be live | `client_instance_conflict` |
|
||||
| Not an active workflow-lease owner | `protected_active_workflow_owner` |
|
||||
| Instance identity is launcher-minted (`inst-…`) | `untrusted_identity_provenance` |
|
||||
| Row's `host_id` matches the host running retirement | `host_binding_unproven` |
|
||||
| Boot identity is known on both sides | `boot_identity_unknown` |
|
||||
| The pid number is not occupied by a different incarnation | `pid_reuse_detected` |
|
||||
|
||||
Eligible rows carry `eligible_stale_orphan`.
|
||||
|
||||
Three properties are worth stating explicitly:
|
||||
|
||||
* **`pid_alive` can only withdraw liveness, never grant it**
|
||||
(`WorkerRegistry.is_live`, #948 AC7). A heartbeat-lapsed but still-running
|
||||
process therefore classifies as `stale` in the snapshot, yet #980's added
|
||||
`pid_alive is False` requirement preserves it. An unprobeable PID (`None`)
|
||||
also fails closed.
|
||||
* **Affirmative identity proof is required (review 657 B2).** An earlier
|
||||
revision required only that the pre-existing registry columns were non-null —
|
||||
which a legacy `legacy-pid-…` row satisfies trivially, so a row that proved
|
||||
nothing about *which* process it described was retireable. Retirement now
|
||||
needs both halves of a positive proof:
|
||||
* **Attribution** — a launcher-minted `inst-…` `client_instance_id`, so the
|
||||
row is known to belong to one specific application launch rather than
|
||||
having been inferred from pid proximity.
|
||||
* **Fencing** — `host_id`, `boot_id`, and `process_start_time`, which turn a
|
||||
bare pid into a statement about one process: which machine it ran on, which
|
||||
boot of that machine, and which incarnation of that pid number.
|
||||
|
||||
**Consequence, stated plainly:** registrations written before these columns
|
||||
existed, and any row on a legacy instance identity, are preserved
|
||||
*permanently*. They are retired only after their worker re-registers under a
|
||||
trusted identity — never on weaker evidence. That the alternative would leave
|
||||
legacy rows outstanding indefinitely is not a reason to relax the proof.
|
||||
* **A live pid is an absolute block.** Even across a boot boundary, where the
|
||||
number provably cannot belong to the registered process, an occupied pid
|
||||
preserves the row rather than being argued away by the fencing proof.
|
||||
|
||||
Multiple processes belonging to one legitimate worker cohort are not treated as
|
||||
multiple independent workers: the fleet model from #948/#978 is preserved
|
||||
unchanged, and sharing a role or profile is never a duplicate.
|
||||
|
||||
## Tools
|
||||
|
||||
### `gitea_plan_stale_worker_retirement`
|
||||
|
||||
Read-only. Controller and reconciler only.
|
||||
|
||||
| Parameter | Meaning |
|
||||
| --- | --- |
|
||||
| `remote` | `dadeschools` or `prgs` |
|
||||
| `host`, `org`, `repo` | Optional overrides (audit context) |
|
||||
| `canonical_repository` | Expected repository binding; defaults to the process root |
|
||||
|
||||
Returns `registry_fingerprint`, `candidate_fingerprint`,
|
||||
`candidate_worker_identities`, per-worker `candidates` and `preserved` entries
|
||||
(each with `reason_code`, `detail`, and structured `evidence`),
|
||||
`preserved_reason_counts`, `assessed_count`, `candidate_count`,
|
||||
`preserved_count`, and `protected_active_workflow_owners`.
|
||||
`mutation_performed` is always `false` and `read_only` is always `true`.
|
||||
|
||||
Planning is deterministic: the same authoritative registry contents produce the
|
||||
same plan and the same tokens regardless of when they are observed.
|
||||
|
||||
### `gitea_apply_stale_worker_retirement`
|
||||
|
||||
Mutating. Controller and reconciler only.
|
||||
|
||||
| Parameter | Meaning |
|
||||
| --- | --- |
|
||||
| `registry_fingerprint` | The exact stable token the plan returned |
|
||||
| `candidate_fingerprint` | The exact candidate-set token the plan returned |
|
||||
| `worker_identities` | The exact candidate identities (list, or JSON / comma-separated string) |
|
||||
| `remote`, `host`, `org`, `repo` | As above |
|
||||
| `canonical_repository` | Must match the value the plan used |
|
||||
|
||||
Before touching the registry, apply fails closed on: profile permission, role
|
||||
kind, master parity (`mutation_safe`), stable-runtime mode, capability
|
||||
resolution refreshed immediately before mutation, worker-registry availability,
|
||||
workflow-lease enumeration failure, and daemon-cohort uniqueness
|
||||
(`classify_cohort`).
|
||||
|
||||
A matching token is necessary but never sufficient. Inside one
|
||||
`BEGIN IMMEDIATE` transaction (`WorkerRegistry.retire_stale_workers`) the
|
||||
server:
|
||||
|
||||
1. re-reads the authoritative rows;
|
||||
2. recomputes `registry_fingerprint` from *those* rows and compares — a mismatch
|
||||
returns `registry_revision_moved` with `retired_count: 0` and no write;
|
||||
3. recomputes the eligibility plan from *those* rows — re-reading the active
|
||||
workflow leases rather than reusing the set captured before the transaction
|
||||
opened, so a lease acquired after planning still preserves its worker — and
|
||||
compares `candidate_fingerprint`. A mismatch returns `candidate_set_moved`
|
||||
with `retired_count: 0` and no write; a lease-enumeration failure raises and
|
||||
rolls the transaction back;
|
||||
4. revalidates every requested identity against that fresh plan;
|
||||
5. retires each survivor with a guarded `UPDATE` that additionally asserts
|
||||
`status`, `last_heartbeat_at`, `generation_id`, `session_id`,
|
||||
`fencing_epoch`, and `pid` are unchanged. A guard that matches no row
|
||||
preserves the worker with `row_changed_since_plan`.
|
||||
|
||||
There is no window between a safety check and its matching write, so a worker
|
||||
that comes back to life, changes ownership, or is retired concurrently cannot be
|
||||
removed on the strength of a stale observation. Any exception — including a
|
||||
commit failure — rolls the whole transaction back and returns
|
||||
`transaction_failed` with `success: false`, `retired_count: 0`, and
|
||||
`mutation_performed: false`; a partial write can never be reported as success.
|
||||
|
||||
### Outcomes
|
||||
|
||||
| `outcome` | Meaning | `mutation_performed` |
|
||||
| --- | --- | --- |
|
||||
| `planned` | Dry-run result | `false` |
|
||||
| `applied` | Transaction ran; see `retired` / `preserved` | `true` only if something was retired |
|
||||
| `registry_revision_moved` | Registry changed between plan and apply | `false` |
|
||||
| `candidate_set_moved` | Eligibility verdict changed between plan and apply | `false` |
|
||||
| `already_retired` | Every requested row is already retired (idempotent replay) | `false` |
|
||||
| `nothing_requested` | Empty target list | `false` |
|
||||
| `transaction_failed` | Rolled back; nothing retired | `false` |
|
||||
|
||||
## What retirement does to the fleet snapshot
|
||||
|
||||
A retired row keeps its history: `status` moves to `retired` and `retired_at`,
|
||||
`retired_by`, `retirement_reason` are recorded. Nothing is deleted. Because
|
||||
`retired` is not `active`, the #978 snapshot counts the row as **historical**,
|
||||
not stale, so `stale_worker_count` falls and historical rows never make the live
|
||||
fleet unsafe by themselves.
|
||||
|
||||
**Retirement does not repair untrusted live identity.** Live workers registered
|
||||
under legacy `pid-`/`proc-` instance identities are preserved untouched and
|
||||
keep their `legacy_incomplete_identity` blockers. Retiring every stale row can
|
||||
therefore legitimately produce:
|
||||
|
||||
* `stale_worker_count: 0`
|
||||
* residual live `legacy_incomplete_identity` blockers
|
||||
* `live_fleet_safe: false`
|
||||
|
||||
That is a truthful result, and the `post_apply` block reports the remaining
|
||||
blockers rather than claiming the fleet became safe. Trusted
|
||||
`client_instance_id` propagation through launchers is a separate enrollment
|
||||
problem.
|
||||
|
||||
## Permissions
|
||||
|
||||
Plan and apply are authorized differently, and deliberately so (review 657 B1).
|
||||
|
||||
| | Plan | Apply |
|
||||
| --- | --- | --- |
|
||||
| Capability | `gitea.read` | `gitea.worker_registry.retire` |
|
||||
| Nature | observational; opens no transaction, writes nothing | mutation |
|
||||
| Roles | `controller`, `reconciler` | `controller`, `reconciler` |
|
||||
|
||||
An earlier revision authorized apply with `gitea.read` alone, reasoning that
|
||||
the mutation lands in the local control-plane registry rather than in Gitea.
|
||||
That the write is local makes it **no less a mutation**: sharing an
|
||||
observational permission class with plan meant any profile that could *look*
|
||||
could also *destroy*. Apply now requires its own capability.
|
||||
|
||||
* **Denied:** author, reviewer, merger, and every ordinary read-only profile —
|
||||
they lack the capability, so they fail closed on the permission itself rather
|
||||
than on the role check alone. The role restriction remains as defence in
|
||||
depth: a profile mistakenly granted the capability still cannot reach apply
|
||||
from an author, reviewer, or merger role.
|
||||
* The capability is checked at entry **and** re-resolved immediately before the
|
||||
registry mutation, so a profile change mid-call cannot be outrun.
|
||||
* **No new Gitea write permission is introduced.**
|
||||
`gitea.worker_registry.retire` authorizes exactly one local control-plane
|
||||
transition (`worker_registrations.status -> retired`) and grants no branch,
|
||||
issue, PR, review, merge, or restart authority. No author permission is
|
||||
broadened.
|
||||
* The fleet snapshot remains observational: nothing here turns it into a gate on
|
||||
ordinary author work.
|
||||
|
||||
### Operator step
|
||||
|
||||
No profile holds `gitea.worker_registry.retire` by default, so apply is inert
|
||||
until an operator adds it to the `allowed_operations` of the controller or
|
||||
reconciler profile in `profiles.json`. Removing it again immediately and
|
||||
completely revokes apply, while leaving plan and every other capability
|
||||
untouched. That grant is a configuration change and is outside the scope of the
|
||||
code that implements this feature.
|
||||
|
||||
## External-state fencing
|
||||
|
||||
`BEGIN IMMEDIATE` locks the worker registry and nothing else, so two inputs the
|
||||
decision depends on sit outside the transaction's isolation domain: the
|
||||
workflow-lease table in a separate control-plane database, and OS process
|
||||
liveness. Re-reading them once during revalidation is not sufficient — the
|
||||
per-target loop runs afterwards, so a lease acquired (or a pid revived) after
|
||||
revalidation but before a given row's `UPDATE` would go unnoticed, and the
|
||||
registry-column guard cannot catch it because no registry column changed.
|
||||
|
||||
Two mechanisms close that window, both applied per target immediately before
|
||||
its own write:
|
||||
|
||||
* **`external_fence_fn`** — a version token over active leases
|
||||
(`external_state_fingerprint`), captured inside the transaction *before* the
|
||||
authoritative read and re-compared before every guarded `UPDATE`. Any movement
|
||||
raises, rolling back the whole transaction: once the world has changed, every
|
||||
remaining per-row decision was computed against a world that no longer exists.
|
||||
An unreadable lease store raises rather than returning a token, because
|
||||
"unreadable" must not silently compare equal to "unchanged".
|
||||
* **`liveness_fn`** — a re-probe of process liveness and fencing identity that
|
||||
must affirmatively re-establish that this exact process is gone. It compares
|
||||
`process_start_time`, so a pid number reused since the plan is refused rather
|
||||
than accepted.
|
||||
|
||||
A caller that supplies no `liveness_fn` retires nothing
|
||||
(`liveness_reprobe_unavailable`) rather than proceeding unfenced. The guarded
|
||||
`UPDATE` additionally asserts `host_id`, `boot_id`, and `process_start_time` are
|
||||
unchanged, and all three participate in the CAS token, so fencing movement
|
||||
alone is enough to abort.
|
||||
|
||||
## Non-goals
|
||||
|
||||
* Killing or restarting processes.
|
||||
* Editing session files or configuration.
|
||||
* Direct database cleanup outside the sanctioned transaction.
|
||||
* Retiring live workers.
|
||||
* Backfilling trusted identity for legacy workers.
|
||||
* Rewriting worker ownership.
|
||||
* Cleaning unrelated workflow-lease or issue-claim registries.
|
||||
+90
-15
@@ -1180,6 +1180,10 @@ RECOGNIZED_GITEA_ENV_KEYS = frozenset({
|
||||
"GITEA_SERVER_PROVENANCE",
|
||||
"GITEA_AUTHOR_WORKTREE",
|
||||
"GITEA_ACTIVE_WORKTREE",
|
||||
"GITEA_REVIEWER_WORKTREE",
|
||||
"GITEA_MERGER_WORKTREE",
|
||||
"GITEA_CANONICAL_REPOSITORY_ROOT",
|
||||
"GITEA_MCP_SESSION_STATE_TTL_HOURS",
|
||||
"GITEA_DISABLE_KEYCHAIN",
|
||||
"GITEA_CONTROL_PLANE_DB",
|
||||
"GITEA_DB_PATH",
|
||||
@@ -1189,6 +1193,31 @@ RECOGNIZED_GITEA_ENV_KEYS = frozenset({
|
||||
"GITEA_IRRECOVERABLE_HMAC_SECRET",
|
||||
"GITEA_FORCE_MCP_RUNTIME_CHECK",
|
||||
"GITEA_FORCE_CLIENT_MANAGED",
|
||||
# #975: the client-identity inputs the server actually consumes at startup
|
||||
# (CLIENT_NAME_ENV / CLIENT_INSTANCE_ENV / CLIENT_SESSION_ENV in
|
||||
# gitea_mcp_server). Production read them while this allowlist omitted them,
|
||||
# so the peer-env scan classified them as unsupported overrides and the
|
||||
# capability resolver refused every mutation fleet-wide. Named individually
|
||||
# on purpose: no prefix is added, so an unrecognised GITEA_* override is
|
||||
# still refused exactly as it was before.
|
||||
"GITEA_MCP_CLIENT",
|
||||
"GITEA_MCP_CLIENT_INSTANCE",
|
||||
"GITEA_MCP_CLIENT_SESSION",
|
||||
# #978: operator-approved fleet enrollment identity shared by one launch.
|
||||
"GITEA_MCP_FLEET_RUN_ID",
|
||||
"GITEA_MCP_PROCESS_IDENTITY",
|
||||
# #978 B1/B2: launcher-sealed provenance + peer-visible worker identity so
|
||||
# the instance-aware mutation gate can distinguish two legitimate
|
||||
# application instances that share a profile from a true duplicate worker.
|
||||
"GITEA_MCP_INSTANCE_PROVENANCE",
|
||||
"GITEA_MCP_WORKER_IDENTITY",
|
||||
"GITEA_MCP_GENERATION_ID",
|
||||
# #975 review 652 B1: production also consumes HEARTBEAT_INTERVAL_ENV from
|
||||
# mcp_worker_identity via gitea_mcp_server._start_worker_heartbeat. Omitting
|
||||
# it reproduced the same unsupported-env → runtime_reconnect_required
|
||||
# failure mode for the documented operator override. Named individually;
|
||||
# no GITEA_* / GITEA_WORKER_* prefix is added.
|
||||
"GITEA_WORKER_HEARTBEAT_INTERVAL_SECONDS",
|
||||
})
|
||||
|
||||
RECOGNIZED_GITEA_ENV_PREFIXES = (
|
||||
@@ -1216,24 +1245,70 @@ def get_unconsumed_gitea_env_overrides(env=None) -> dict[str, str]:
|
||||
return unconsumed
|
||||
|
||||
|
||||
def launcher_entry(profile_name, config_path=None):
|
||||
def launcher_entry(
|
||||
profile_name,
|
||||
config_path=None,
|
||||
*,
|
||||
client_type=None,
|
||||
client_instance_id=None,
|
||||
fleet_run_id=None,
|
||||
session_id=None,
|
||||
launch_nonce=None,
|
||||
):
|
||||
"""Return a thin MCP launcher entry for *profile_name*.
|
||||
|
||||
Contains command/args and the GITEA_MCP_* / GITEA_CLIENT_MANAGED env vars — never a token
|
||||
or password. Suitable for Claude / Gemini / Codex ``mcpServers`` blocks.
|
||||
Contains command/args and the GITEA_MCP_* / GITEA_CLIENT_MANAGED env vars —
|
||||
never a token or password. Suitable for Claude / Gemini / Codex
|
||||
``mcpServers`` blocks.
|
||||
|
||||
#978 B1: production launches mint (or reuse) a trusted
|
||||
``GITEA_MCP_CLIENT_INSTANCE`` so workers never fall back to the legacy
|
||||
placeholder identity on the real serve path. Pass *client_instance_id* to
|
||||
resume the same launch; omit it for a fresh launch (new ID).
|
||||
"""
|
||||
command, args = server_command()
|
||||
return {
|
||||
"gitea-tools": {
|
||||
"command": command,
|
||||
"args": args,
|
||||
"env": {
|
||||
"GITEA_MCP_CONFIG": config_path or DEFAULT_CONFIG_PATH,
|
||||
"GITEA_MCP_PROFILE": profile_name,
|
||||
"GITEA_CLIENT_MANAGED": "1",
|
||||
},
|
||||
}
|
||||
}
|
||||
import mcp_application_launcher as app_launcher
|
||||
|
||||
built = app_launcher.launcher_entry_for_profile(
|
||||
profile_name,
|
||||
client_type=client_type or "unknown",
|
||||
config_path=config_path or DEFAULT_CONFIG_PATH,
|
||||
client_instance_id=client_instance_id,
|
||||
fleet_run_id=fleet_run_id,
|
||||
session_id=session_id,
|
||||
launch_nonce=launch_nonce,
|
||||
server_key="gitea-tools",
|
||||
)
|
||||
# Public shape stays {server_key: {command, args, env}} for existing callers.
|
||||
return {"gitea-tools": built["gitea-tools"]}
|
||||
|
||||
|
||||
def multi_namespace_launcher_entries(
|
||||
profile_by_namespace,
|
||||
*,
|
||||
client_type,
|
||||
config_path=None,
|
||||
client_instance_id=None,
|
||||
fleet_run_id=None,
|
||||
session_id=None,
|
||||
launch_nonce=None,
|
||||
):
|
||||
"""Build production mcpServers for all five namespaces of one application launch.
|
||||
|
||||
One shared trusted ``client_instance_id`` is minted (or reused) and
|
||||
propagated to every namespace worker env. See
|
||||
:func:`mcp_application_launcher.build_application_mcp_servers`.
|
||||
"""
|
||||
import mcp_application_launcher as app_launcher
|
||||
|
||||
return app_launcher.build_application_mcp_servers(
|
||||
profile_by_namespace,
|
||||
client_type=client_type,
|
||||
config_path=config_path or DEFAULT_CONFIG_PATH,
|
||||
client_instance_id=client_instance_id,
|
||||
fleet_run_id=fleet_run_id,
|
||||
session_id=session_id,
|
||||
launch_nonce=launch_nonce,
|
||||
)
|
||||
|
||||
|
||||
|
||||
|
||||
+1400
-203
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,393 @@
|
||||
"""Production application launcher for multi-namespace MCP fleets (#978 B1).
|
||||
|
||||
One real LLM application launch mints exactly one trusted
|
||||
``client_instance_id`` and propagates it to every Gitea MCP namespace worker
|
||||
started for that launch. Workers never invent a trusted instance identity from
|
||||
PID proximity, timestamps, or ordinary untrusted environment values.
|
||||
|
||||
This module is the production serve-path authority for instance identity.
|
||||
Tests and fixtures may call the same functions, but production registration
|
||||
receives the identity from the env this launcher builds — not from a hand-set
|
||||
test-only helper that bypasses it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import secrets
|
||||
from typing import Any, Mapping
|
||||
|
||||
import mcp_fleet_snapshot as fleet
|
||||
import mcp_worker_identity as mwi
|
||||
|
||||
#: Canonical MCP server names for the five role namespaces.
|
||||
NAMESPACE_SERVER_NAMES: tuple[str, ...] = (
|
||||
"gitea-author",
|
||||
"gitea-reviewer",
|
||||
"gitea-merger",
|
||||
"gitea-controller",
|
||||
"gitea-reconciler",
|
||||
)
|
||||
|
||||
#: Map MCP server name → role/namespace kind.
|
||||
SERVER_TO_NAMESPACE: dict[str, str] = {
|
||||
"gitea-author": "author",
|
||||
"gitea-reviewer": "reviewer",
|
||||
"gitea-merger": "merger",
|
||||
"gitea-controller": "controller",
|
||||
"gitea-reconciler": "reconciler",
|
||||
}
|
||||
|
||||
SANCTIONED_NAMESPACES: tuple[str, ...] = (
|
||||
"author",
|
||||
"reviewer",
|
||||
"merger",
|
||||
"controller",
|
||||
"reconciler",
|
||||
)
|
||||
|
||||
# Env the launcher may set on every worker of one application launch.
|
||||
CLIENT_NAME_ENV = "GITEA_MCP_CLIENT"
|
||||
CLIENT_INSTANCE_ENV = fleet.CLIENT_INSTANCE_ENV
|
||||
FLEET_RUN_ENV = fleet.FLEET_RUN_ENV
|
||||
CLIENT_SESSION_ENV = "GITEA_MCP_CLIENT_SESSION"
|
||||
CLIENT_MANAGED_ENV = "GITEA_CLIENT_MANAGED"
|
||||
PROFILE_ENV = "GITEA_MCP_PROFILE"
|
||||
CONFIG_ENV = "GITEA_MCP_CONFIG"
|
||||
INSTANCE_PROVENANCE_ENV = "GITEA_MCP_INSTANCE_PROVENANCE"
|
||||
WORKER_IDENTITY_ENV = "GITEA_MCP_WORKER_IDENTITY"
|
||||
GENERATION_ID_ENV = "GITEA_MCP_GENERATION_ID"
|
||||
|
||||
# Marker the trusted launcher alone writes; ordinary user env without this
|
||||
# marker is never classified as launcher-trusted provenance.
|
||||
LAUNCHER_PROVENANCE_VALUE = fleet.INSTANCE_ID_PROVENANCE_TRUSTED
|
||||
|
||||
|
||||
def mint_application_launch(
|
||||
client_type: str | None,
|
||||
*,
|
||||
launch_nonce: str | None = None,
|
||||
fleet_run_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
now=None,
|
||||
) -> dict[str, Any]:
|
||||
"""Mint one trusted application-instance identity for a production launch.
|
||||
|
||||
Called exactly once per real application launch. The returned
|
||||
``client_instance_id`` is injected into every namespace worker environment
|
||||
for that launch. A second call (separate launch) yields a different ID.
|
||||
"""
|
||||
client = mwi.normalize_client_name(client_type)
|
||||
instance_id = fleet.generate_client_instance_id(
|
||||
client, launch_nonce=launch_nonce, now=now
|
||||
)
|
||||
assessment = fleet.assess_instance_identity(instance_id)
|
||||
if not assessment["trusted"]:
|
||||
# generate_client_instance_id always produces a trusted format; fail
|
||||
# closed if that invariant ever breaks rather than shipping untrusted.
|
||||
raise RuntimeError(
|
||||
f"launcher produced untrusted client_instance_id {instance_id!r}: "
|
||||
f"{assessment.get('reasons')}"
|
||||
)
|
||||
session = (session_id or "").strip() or f"launch-{secrets.token_hex(12)}"
|
||||
return {
|
||||
"client_type": client,
|
||||
"client_instance_id": instance_id,
|
||||
"instance_id_provenance": LAUNCHER_PROVENANCE_VALUE,
|
||||
"instance_identity_trusted": True,
|
||||
"fleet_run_id": (fleet_run_id or "").strip() or None,
|
||||
"session_id": session,
|
||||
"namespaces": list(SANCTIONED_NAMESPACES),
|
||||
"namespace_server_names": list(NAMESPACE_SERVER_NAMES),
|
||||
}
|
||||
|
||||
|
||||
def namespace_worker_env(
|
||||
*,
|
||||
profile_name: str,
|
||||
client_type: str | None,
|
||||
client_instance_id: str,
|
||||
config_path: str | None = None,
|
||||
fleet_run_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
extra_env: Mapping[str, str] | None = None,
|
||||
) -> dict[str, str]:
|
||||
"""Build the environment for one namespace worker of a trusted launch.
|
||||
|
||||
Never invents a client_instance_id. The caller must supply the launch-minted
|
||||
identity so all five workers receive the same value.
|
||||
"""
|
||||
assessment = fleet.assess_instance_identity(client_instance_id)
|
||||
if not assessment["trusted"]:
|
||||
raise ValueError(
|
||||
"namespace_worker_env refuses untrusted client_instance_id "
|
||||
f"{client_instance_id!r}: {assessment.get('reasons')}"
|
||||
)
|
||||
client = mwi.normalize_client_name(client_type)
|
||||
env: dict[str, str] = {
|
||||
PROFILE_ENV: str(profile_name),
|
||||
CLIENT_MANAGED_ENV: "1",
|
||||
"GITEA_MCP_CLIENT": client,
|
||||
CLIENT_INSTANCE_ENV: assessment["client_instance_id"],
|
||||
INSTANCE_PROVENANCE_ENV: LAUNCHER_PROVENANCE_VALUE,
|
||||
}
|
||||
if config_path:
|
||||
env[CONFIG_ENV] = str(config_path)
|
||||
if fleet_run_id:
|
||||
env[FLEET_RUN_ENV] = str(fleet_run_id)
|
||||
if session_id:
|
||||
env[CLIENT_SESSION_ENV] = str(session_id)
|
||||
if extra_env:
|
||||
# Never let untrusted callers override the trusted instance keys.
|
||||
protected = {
|
||||
CLIENT_INSTANCE_ENV,
|
||||
INSTANCE_PROVENANCE_ENV,
|
||||
"GITEA_MCP_CLIENT",
|
||||
CLIENT_MANAGED_ENV,
|
||||
}
|
||||
for key, value in extra_env.items():
|
||||
if key in protected:
|
||||
continue
|
||||
env[str(key)] = str(value)
|
||||
return env
|
||||
|
||||
|
||||
def build_application_mcp_servers(
|
||||
profile_by_namespace: Mapping[str, str],
|
||||
*,
|
||||
client_type: str | None,
|
||||
config_path: str | None = None,
|
||||
client_instance_id: str | None = None,
|
||||
fleet_run_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
launch_nonce: str | None = None,
|
||||
command: str | None = None,
|
||||
args: list[str] | None = None,
|
||||
now=None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build a production ``mcpServers`` map for one application launch.
|
||||
|
||||
Mints one ``client_instance_id`` when *client_instance_id* is omitted (fresh
|
||||
launch). When the caller supplies a previously minted trusted ID (resume of
|
||||
the same launch / config rewrite), that ID is reused so reconnect keeps
|
||||
attribution. A full new application restart omits the ID and receives a
|
||||
fresh mint.
|
||||
|
||||
Every namespace server entry receives the **same** instance ID. Separate
|
||||
calls with no supplied ID receive distinct IDs.
|
||||
"""
|
||||
missing = [
|
||||
ns for ns in SANCTIONED_NAMESPACES if not profile_by_namespace.get(ns)
|
||||
]
|
||||
if missing:
|
||||
raise ValueError(
|
||||
"build_application_mcp_servers requires a profile for every "
|
||||
f"sanctioned namespace; missing: {missing}"
|
||||
)
|
||||
|
||||
if client_instance_id is None:
|
||||
launch = mint_application_launch(
|
||||
client_type,
|
||||
launch_nonce=launch_nonce,
|
||||
fleet_run_id=fleet_run_id,
|
||||
session_id=session_id,
|
||||
now=now,
|
||||
)
|
||||
else:
|
||||
assessment = fleet.assess_instance_identity(client_instance_id)
|
||||
if not assessment["trusted"]:
|
||||
raise ValueError(
|
||||
"refusing to propagate untrusted client_instance_id "
|
||||
f"{client_instance_id!r} into production launch envs: "
|
||||
f"{assessment.get('reasons')}"
|
||||
)
|
||||
launch = {
|
||||
"client_type": mwi.normalize_client_name(client_type),
|
||||
"client_instance_id": assessment["client_instance_id"],
|
||||
"instance_id_provenance": LAUNCHER_PROVENANCE_VALUE,
|
||||
"instance_identity_trusted": True,
|
||||
"fleet_run_id": (fleet_run_id or "").strip() or None,
|
||||
"session_id": (session_id or "").strip()
|
||||
or f"launch-{secrets.token_hex(12)}",
|
||||
"namespaces": list(SANCTIONED_NAMESPACES),
|
||||
"namespace_server_names": list(NAMESPACE_SERVER_NAMES),
|
||||
}
|
||||
|
||||
# Resolve command/args from the production server entry when not provided.
|
||||
if command is None or args is None:
|
||||
import gitea_config
|
||||
|
||||
cmd, cmd_args = gitea_config.server_command()
|
||||
command = command or cmd
|
||||
args = args if args is not None else list(cmd_args)
|
||||
|
||||
servers: dict[str, Any] = {}
|
||||
shared_id = launch["client_instance_id"]
|
||||
for namespace in SANCTIONED_NAMESPACES:
|
||||
server_name = f"gitea-{namespace}"
|
||||
profile = profile_by_namespace[namespace]
|
||||
env = namespace_worker_env(
|
||||
profile_name=profile,
|
||||
client_type=launch["client_type"],
|
||||
client_instance_id=shared_id,
|
||||
config_path=config_path,
|
||||
fleet_run_id=launch.get("fleet_run_id"),
|
||||
session_id=launch.get("session_id"),
|
||||
)
|
||||
servers[server_name] = {
|
||||
"command": command,
|
||||
"args": list(args),
|
||||
"env": env,
|
||||
}
|
||||
|
||||
return {
|
||||
"mcpServers": servers,
|
||||
"launch": launch,
|
||||
"client_instance_id": shared_id,
|
||||
"client_type": launch["client_type"],
|
||||
"namespaces": list(SANCTIONED_NAMESPACES),
|
||||
"shared_instance_id_across_namespaces": True,
|
||||
"namespace_count": len(SANCTIONED_NAMESPACES),
|
||||
}
|
||||
|
||||
|
||||
def launcher_entry_for_profile(
|
||||
profile_name: str,
|
||||
*,
|
||||
client_type: str | None = None,
|
||||
config_path: str | None = None,
|
||||
client_instance_id: str | None = None,
|
||||
fleet_run_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
server_key: str = "gitea-tools",
|
||||
launch_nonce: str | None = None,
|
||||
now=None,
|
||||
) -> dict[str, Any]:
|
||||
"""Thin single-server production launcher entry with trusted instance ID.
|
||||
|
||||
Used when only one namespace is being configured. Still mints (or reuses)
|
||||
a trusted ``client_instance_id`` so production never relies on the legacy
|
||||
placeholder identity for normal launches.
|
||||
"""
|
||||
import gitea_config
|
||||
|
||||
if client_instance_id is None:
|
||||
launch = mint_application_launch(
|
||||
client_type or "unknown",
|
||||
launch_nonce=launch_nonce,
|
||||
fleet_run_id=fleet_run_id,
|
||||
session_id=session_id,
|
||||
now=now,
|
||||
)
|
||||
client_instance_id = launch["client_instance_id"]
|
||||
client = launch["client_type"]
|
||||
fleet_run = launch.get("fleet_run_id")
|
||||
session = launch.get("session_id")
|
||||
else:
|
||||
assessment = fleet.assess_instance_identity(client_instance_id)
|
||||
if not assessment["trusted"]:
|
||||
raise ValueError(
|
||||
f"untrusted client_instance_id {client_instance_id!r}"
|
||||
)
|
||||
client = mwi.normalize_client_name(client_type)
|
||||
fleet_run = (fleet_run_id or "").strip() or None
|
||||
session = (session_id or "").strip() or None
|
||||
client_instance_id = assessment["client_instance_id"]
|
||||
|
||||
command, args = gitea_config.server_command()
|
||||
env = namespace_worker_env(
|
||||
profile_name=profile_name,
|
||||
client_type=client,
|
||||
client_instance_id=client_instance_id,
|
||||
config_path=config_path or gitea_config.DEFAULT_CONFIG_PATH,
|
||||
fleet_run_id=fleet_run,
|
||||
session_id=session,
|
||||
)
|
||||
return {
|
||||
server_key: {
|
||||
"command": command,
|
||||
"args": args,
|
||||
"env": env,
|
||||
},
|
||||
"client_instance_id": client_instance_id,
|
||||
"client_type": client,
|
||||
}
|
||||
|
||||
|
||||
def collect_instance_ids_from_mcp_servers(
|
||||
mcp_servers: Mapping[str, Any],
|
||||
) -> dict[str, Any]:
|
||||
"""Inspect a production mcpServers map for shared instance attribution.
|
||||
|
||||
Returns the unique set of client_instance_id values across gitea-* servers
|
||||
and whether all five namespaces share exactly one trusted ID.
|
||||
"""
|
||||
ids: list[str] = []
|
||||
by_server: dict[str, str | None] = {}
|
||||
for name in NAMESPACE_SERVER_NAMES:
|
||||
entry = mcp_servers.get(name) or {}
|
||||
env = entry.get("env") or {}
|
||||
raw = (env.get(CLIENT_INSTANCE_ENV) or "").strip() or None
|
||||
by_server[name] = raw
|
||||
if raw:
|
||||
ids.append(raw)
|
||||
unique = sorted(set(ids))
|
||||
trusted = [
|
||||
i
|
||||
for i in unique
|
||||
if fleet.assess_instance_identity(i)["trusted"]
|
||||
]
|
||||
return {
|
||||
"server_instance_ids": by_server,
|
||||
"unique_instance_ids": unique,
|
||||
"trusted_instance_ids": trusted,
|
||||
"shared_single_trusted_id": (
|
||||
len(unique) == 1
|
||||
and len(trusted) == 1
|
||||
and all(by_server.get(n) == unique[0] for n in NAMESPACE_SERVER_NAMES)
|
||||
),
|
||||
"namespace_server_count": sum(
|
||||
1 for n in NAMESPACE_SERVER_NAMES if n in mcp_servers
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def inherit_or_refuse_client_instance(
|
||||
env: Mapping[str, str] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Resolve instance identity for a worker process at serve time.
|
||||
|
||||
Production workers inherit the launcher-issued ID. They never mint a
|
||||
trusted ID themselves. Missing / legacy / malformed values fail soft into
|
||||
an untrusted assessment so registration can still record a diagnostic row
|
||||
without authorizing multi-instance fleet mutation.
|
||||
"""
|
||||
source = dict(env if env is not None else os.environ)
|
||||
raw = (source.get(CLIENT_INSTANCE_ENV) or "").strip() or None
|
||||
provenance_marker = (source.get(INSTANCE_PROVENANCE_ENV) or "").strip()
|
||||
assessment = fleet.assess_instance_identity(raw)
|
||||
# Ordinary user-supplied values without launcher provenance marker are
|
||||
# still format-checked by assess_instance_identity. When the format is
|
||||
# trusted but the launcher marker is absent, keep the ID but note that
|
||||
# provenance is not launcher-sealed (operator hand-set or legacy config).
|
||||
if assessment["trusted"] and provenance_marker != LAUNCHER_PROVENANCE_VALUE:
|
||||
assessment = dict(assessment)
|
||||
assessment["launcher_sealed"] = False
|
||||
assessment["reasons"] = list(assessment.get("reasons") or []) + [
|
||||
f"{INSTANCE_PROVENANCE_ENV} is not {LAUNCHER_PROVENANCE_VALUE!r}; "
|
||||
"identity format is valid but not sealed by the production launcher"
|
||||
]
|
||||
else:
|
||||
assessment = dict(assessment)
|
||||
assessment["launcher_sealed"] = bool(
|
||||
assessment["trusted"]
|
||||
and provenance_marker == LAUNCHER_PROVENANCE_VALUE
|
||||
)
|
||||
assessment["fleet_run_id"] = (source.get(FLEET_RUN_ENV) or "").strip() or None
|
||||
assessment["session_id"] = (
|
||||
(source.get(CLIENT_SESSION_ENV) or "").strip() or None
|
||||
)
|
||||
assessment["client_type"] = mwi.normalize_client_name(
|
||||
(source.get("GITEA_MCP_CLIENT") or "").strip() or None
|
||||
)
|
||||
return assessment
|
||||
@@ -0,0 +1,776 @@
|
||||
"""CAS-protected retirement planning for stale worker registrations (#980).
|
||||
|
||||
#978 (merged PR #979) made the fleet observable: every registered namespace
|
||||
worker, its instance attribution, its heartbeat freshness, and a structured
|
||||
classification. It deliberately stopped there — the snapshot is read-only and
|
||||
the control plane still had no sanctioned way to retire registry rows whose
|
||||
owning process is conclusively gone.
|
||||
|
||||
This module is the *decision layer* for that retirement. It is pure: callers
|
||||
supply registry rows, a clock, and a PID probe; nothing here opens SQLite,
|
||||
scans process tables, or mutates state. The transactional apply lives in
|
||||
:meth:`mcp_worker_identity.WorkerRegistry.retire_stale_workers`, which calls
|
||||
back into these same pure functions so plan and apply can never disagree about
|
||||
what "the registry looks like" or "which rows are eligible".
|
||||
|
||||
Why a separate token
|
||||
--------------------
|
||||
|
||||
``mcp_fleet_snapshot._consistency_token`` seeds its digest with ``snapshot_at``
|
||||
at second precision, so ``registry_revision`` changes on every call even when
|
||||
no registry row changed. A compare-and-swap gated on it can never pass — a
|
||||
dry-run/apply cycle spanning more than one second aborts unconditionally. That
|
||||
token is still useful as an observation stamp, so it is left exactly as it is;
|
||||
#980 gets its own :func:`registry_fingerprint`, derived *only* from canonical
|
||||
retirement-relevant row content:
|
||||
|
||||
* identical registry contents observed at any two times produce the same token,
|
||||
* row order never affects the token (serialized rows are sorted),
|
||||
* any create/delete/identity/liveness/ownership/registration-state change to a
|
||||
retirement-relevant field changes the token.
|
||||
|
||||
Fail-closed posture
|
||||
-------------------
|
||||
|
||||
A worker is retired only when the control plane *conclusively* establishes it
|
||||
is a stale orphan. Missing evidence is never read as permission: an unprobeable
|
||||
PID, an unparsable heartbeat, a row that shares identity evidence with a live
|
||||
or unprobeable worker, a foreign or absent repository binding, or a worker that
|
||||
still owns an active workflow lease all preserve the row.
|
||||
|
||||
Trusted launcher identity (``inst-…`` provenance) is deliberately *not* part of
|
||||
the conjunction. #980 places trusted ``client_instance_id`` propagation out of
|
||||
scope and lists "backfilling trusted identity for legacy workers" as a non-goal;
|
||||
requiring it here would preserve every legacy row forever and make the feature
|
||||
inert. What *is* required is that the registry fields the conjunction reads are
|
||||
actually present — see :data:`REQUIRED_IDENTITY_FIELDS`.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
from datetime import datetime
|
||||
from typing import Any, Callable, Iterable, Mapping, Sequence
|
||||
|
||||
import mcp_fleet_snapshot as fleet
|
||||
import mcp_worker_identity as mwi
|
||||
|
||||
# --- Outcomes -------------------------------------------------------------
|
||||
|
||||
OUTCOME_PLANNED = "planned"
|
||||
OUTCOME_APPLIED = "applied"
|
||||
OUTCOME_REGISTRY_MOVED = "registry_revision_moved"
|
||||
OUTCOME_CANDIDATES_MOVED = "candidate_set_moved"
|
||||
OUTCOME_ALREADY_RETIRED = "already_retired"
|
||||
OUTCOME_NOTHING_REQUESTED = "nothing_requested"
|
||||
|
||||
# --- Reason codes ---------------------------------------------------------
|
||||
|
||||
#: The only reason code that authorizes retirement.
|
||||
REASON_ELIGIBLE = "eligible_stale_orphan"
|
||||
|
||||
REASON_ALREADY_TERMINAL = "already_terminal_registration"
|
||||
REASON_AMBIGUOUS_OWNERSHIP = "ambiguous_ownership_state"
|
||||
REASON_CONFLICTING_IDENTITY = "conflicting_identity_evidence"
|
||||
REASON_FOREIGN_REPOSITORY = "repository_binding_ambiguous"
|
||||
REASON_HEARTBEAT_FRESH = "heartbeat_not_expired"
|
||||
REASON_INCOMPLETE_IDENTITY = "incomplete_registry_identity"
|
||||
REASON_NOT_IN_PLAN = "not_in_current_plan"
|
||||
REASON_PID_ALIVE = "pid_alive"
|
||||
REASON_PID_UNKNOWN = "pid_liveness_unknown"
|
||||
REASON_PROTECTED_OWNER = "protected_active_workflow_owner"
|
||||
REASON_ROW_CHANGED = "row_changed_since_plan"
|
||||
REASON_ROW_MISSING = "registration_missing"
|
||||
REASON_UNPARSABLE_HEARTBEAT = "unparsable_heartbeat"
|
||||
REASON_WORKER_LIVE = "worker_live"
|
||||
|
||||
# --- #980 review 657 B2: affirmative identity/liveness proof ---------------
|
||||
|
||||
#: The registration's instance identity is not launcher-minted (``inst-…``),
|
||||
#: so nothing proves which application launch this row belongs to.
|
||||
REASON_UNTRUSTED_PROVENANCE = "untrusted_identity_provenance"
|
||||
#: The row does not record which host its pid belongs to, or records a
|
||||
#: different host than the one probing. A local pid probe cannot speak for a
|
||||
#: process on another machine.
|
||||
REASON_HOST_UNPROVEN = "host_binding_unproven"
|
||||
#: Boot identity is missing on the row or unobtainable here, so a recorded pid
|
||||
#: cannot be compared against a live pid at all.
|
||||
REASON_BOOT_UNKNOWN = "boot_identity_unknown"
|
||||
#: The pid is alive but belongs to a different process incarnation than the one
|
||||
#: registered — reported distinctly from a plain live worker.
|
||||
REASON_PID_REUSED = "pid_reuse_detected"
|
||||
#: Two active registrations claim one client instance within one namespace.
|
||||
REASON_INSTANCE_CONFLICT = "client_instance_conflict"
|
||||
#: The immediate pre-write re-probe could not re-establish death (#980 B3).
|
||||
REASON_LIVENESS_REPROBE = "liveness_reprobe_refused"
|
||||
|
||||
#: Instance-identity prefix minted by the trusted launcher. Kept in sync with
|
||||
#: ``mcp_fleet_snapshot._TRUSTED_INSTANCE_PREFIX`` through
|
||||
#: :func:`mcp_fleet_snapshot.assess_instance_identity`, which stays the single
|
||||
#: authority on what "trusted" means — this module never re-implements it.
|
||||
TRUSTED_INSTANCE_PREFIX = "inst-"
|
||||
|
||||
#: Registry columns that must carry a usable value before the eligibility
|
||||
#: conjunction can even be evaluated. Absence is ambiguity, not permission.
|
||||
#:
|
||||
#: #980 review 657 B2 added the fencing triple. Before it, "complete identity"
|
||||
#: meant only that the pre-existing columns were non-null, which a legacy
|
||||
#: ``legacy-pid-…`` row satisfies trivially — so a row that proved nothing about
|
||||
#: *which* process it described was retireable. The triple is what makes a
|
||||
#: recorded pid interpretable: which machine it ran on, which boot of that
|
||||
#: machine, and which incarnation of that pid number. A registration written
|
||||
#: before these columns existed carries NULL and is therefore preserved
|
||||
#: permanently, which is the intended fail-closed outcome.
|
||||
REQUIRED_IDENTITY_FIELDS: tuple[str, ...] = (
|
||||
"worker_identity",
|
||||
"client_instance_id",
|
||||
"session_id",
|
||||
"generation_id",
|
||||
"status",
|
||||
"started_at",
|
||||
"last_heartbeat_at",
|
||||
"heartbeat_ttl_seconds",
|
||||
"pid",
|
||||
"host_id",
|
||||
"boot_id",
|
||||
"process_start_time",
|
||||
)
|
||||
|
||||
#: Canonical retirement-relevant content. Ordering here is fixed and part of
|
||||
#: the token contract; adding a field changes every fingerprint, so a change
|
||||
#: here is a deliberate contract revision.
|
||||
#:
|
||||
#: Deliberately excluded: ``token_fingerprint`` (credential-adjacent, never a
|
||||
#: retirement input), the four ``*_revision`` columns (revision drift is an
|
||||
#: independent restart concern and is not part of the eligibility conjunction),
|
||||
#: and the ``retired_*`` bookkeeping columns this feature adds.
|
||||
FINGERPRINT_FIELDS: tuple[str, ...] = (
|
||||
"worker_identity",
|
||||
"client_name",
|
||||
"client_instance_id",
|
||||
"session_id",
|
||||
"generation_id",
|
||||
"role",
|
||||
"profile",
|
||||
"namespace",
|
||||
"remote",
|
||||
"repository_binding",
|
||||
"pid",
|
||||
"process_identity",
|
||||
"transport",
|
||||
"started_at",
|
||||
"last_heartbeat_at",
|
||||
"heartbeat_ttl_seconds",
|
||||
"fencing_epoch",
|
||||
"status",
|
||||
"fleet_run_id",
|
||||
"authenticated_account",
|
||||
"instance_id_provenance",
|
||||
# #980 review 657 B3: fencing evidence is a retirement input, so moving it
|
||||
# must move the CAS token. Without these, a row whose host, boot, or
|
||||
# process incarnation changed would hash identically to the row the plan
|
||||
# approved.
|
||||
"host_id",
|
||||
"boot_id",
|
||||
"process_start_time",
|
||||
)
|
||||
|
||||
_FINGERPRINT_VERSION = "registryfp-v1"
|
||||
_CANDIDATE_VERSION = "candidatefp-v1"
|
||||
_UNIT = "\x1f"
|
||||
_RECORD = "\x1e"
|
||||
|
||||
|
||||
def _canon(value: Any) -> str:
|
||||
"""Stable text for one field value, independent of Python/SQLite typing.
|
||||
|
||||
``900`` and ``900.0`` are the same TTL and must hash the same; a value that
|
||||
round-trips through SQLite as REAL must not produce a different token than
|
||||
the same value supplied by a caller as ``int``.
|
||||
"""
|
||||
if value is None:
|
||||
return ""
|
||||
if isinstance(value, bool):
|
||||
return "true" if value else "false"
|
||||
if isinstance(value, float):
|
||||
if value != value or value in (float("inf"), float("-inf")):
|
||||
return repr(value)
|
||||
if value.is_integer():
|
||||
return str(int(value))
|
||||
return repr(value)
|
||||
if isinstance(value, int):
|
||||
return str(value)
|
||||
return str(value)
|
||||
|
||||
|
||||
def _serialize_row(row: Mapping[str, Any]) -> str:
|
||||
return _UNIT.join(f"{name}={_canon(row.get(name))}" for name in FINGERPRINT_FIELDS)
|
||||
|
||||
|
||||
def _digest(version: str, serialized: Sequence[str], prefix: str) -> str:
|
||||
ordered = sorted(serialized)
|
||||
material = _RECORD.join([version, str(len(ordered)), *ordered])
|
||||
return f"{prefix}-{hashlib.sha256(material.encode('utf-8')).hexdigest()[:32]}"
|
||||
|
||||
|
||||
def registry_fingerprint(rows: Iterable[Mapping[str, Any]]) -> str:
|
||||
"""Content-derived compare-and-swap token for the worker registry (#980).
|
||||
|
||||
Derived exclusively from :data:`FINGERPRINT_FIELDS` across every row. It
|
||||
contains no ``snapshot_at``, wall-clock, request, or report-generation
|
||||
time, so two observations of an unchanged registry always agree, and the
|
||||
serialized rows are sorted so iteration order cannot perturb the digest.
|
||||
"""
|
||||
return _digest(
|
||||
_FINGERPRINT_VERSION,
|
||||
[_serialize_row(row) for row in rows],
|
||||
"registryfp",
|
||||
)
|
||||
|
||||
|
||||
def candidate_fingerprint(candidate_rows: Iterable[Mapping[str, Any]]) -> str:
|
||||
"""Exact-candidate-set token over the selected rows' canonical content.
|
||||
|
||||
A matching :func:`registry_fingerprint` already implies these rows are
|
||||
unchanged; this second token additionally pins *which* rows the operator
|
||||
approved, so an apply can never widen or narrow the approved set.
|
||||
"""
|
||||
return _digest(
|
||||
_CANDIDATE_VERSION,
|
||||
[_serialize_row(row) for row in candidate_rows],
|
||||
"candidatefp",
|
||||
)
|
||||
|
||||
|
||||
def _probe_pid(
|
||||
pid: Any, pid_alive_probe: Callable[[int | None], bool | None] | None
|
||||
) -> bool | None:
|
||||
if pid_alive_probe is None or pid is None:
|
||||
return None
|
||||
try:
|
||||
result = pid_alive_probe(pid)
|
||||
except Exception:
|
||||
return None
|
||||
return None if result is None else bool(result)
|
||||
|
||||
|
||||
def _reuse_detected(
|
||||
row: Mapping[str, Any],
|
||||
live_start_time: str | None,
|
||||
current_host_id: str | None,
|
||||
current_boot_id: str | None,
|
||||
) -> bool:
|
||||
"""Is the pid occupied by a *different* incarnation than the one recorded?
|
||||
|
||||
Only meaningful when the recorded pid is comparable to the live one — same
|
||||
machine, same boot. Across hosts or boots the number is unrelated by
|
||||
construction and reuse is not the interesting question.
|
||||
"""
|
||||
recorded_host = (row.get("host_id") or "").strip()
|
||||
recorded_boot = (row.get("boot_id") or "").strip()
|
||||
if not current_host_id or recorded_host != current_host_id:
|
||||
return False
|
||||
if not current_boot_id or recorded_boot != current_boot_id:
|
||||
return False
|
||||
recorded_start = (row.get("process_start_time") or "").strip()
|
||||
return bool(live_start_time) and live_start_time != recorded_start
|
||||
|
||||
|
||||
def _instance_key(row: Mapping[str, Any]) -> tuple[str, str] | None:
|
||||
"""The (instance, namespace) pair #978 requires to be unique among live rows."""
|
||||
instance = _canon(row.get("client_instance_id"))
|
||||
namespace = _canon(row.get("namespace"))
|
||||
if not instance or not namespace:
|
||||
return None
|
||||
return (instance, namespace)
|
||||
|
||||
|
||||
def _probe_start_time(
|
||||
pid: Any, start_time_probe: Callable[[Any], str | None] | None
|
||||
) -> str | None:
|
||||
if start_time_probe is None or pid is None:
|
||||
return None
|
||||
try:
|
||||
return start_time_probe(pid)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _evidence(
|
||||
row: Mapping[str, Any],
|
||||
snapshot_row: Mapping[str, Any],
|
||||
pid_alive: bool | None,
|
||||
) -> dict[str, Any]:
|
||||
liveness = snapshot_row.get("liveness") or {}
|
||||
return {
|
||||
"worker_identity": row.get("worker_identity"),
|
||||
"client_type": snapshot_row.get("client_type"),
|
||||
"client_instance_id": row.get("client_instance_id"),
|
||||
"fleet_run_id": row.get("fleet_run_id"),
|
||||
"namespace": row.get("namespace"),
|
||||
"profile": row.get("profile"),
|
||||
"declared_role": row.get("role"),
|
||||
"session_id": row.get("session_id"),
|
||||
"generation_id": row.get("generation_id"),
|
||||
"fencing_epoch": row.get("fencing_epoch"),
|
||||
"process_identity": snapshot_row.get("process_identity"),
|
||||
"pid": row.get("pid"),
|
||||
"pid_alive": pid_alive,
|
||||
"host_id": row.get("host_id"),
|
||||
"boot_id": row.get("boot_id"),
|
||||
"process_start_time": row.get("process_start_time"),
|
||||
"repository_binding": row.get("repository_binding"),
|
||||
"foreign_repository": bool(snapshot_row.get("foreign_repository")),
|
||||
"status": row.get("status"),
|
||||
"started_at": row.get("started_at"),
|
||||
"last_heartbeat_at": row.get("last_heartbeat_at"),
|
||||
"heartbeat_ttl_seconds": row.get("heartbeat_ttl_seconds"),
|
||||
"heartbeat_age_seconds": liveness.get("heartbeat_age_seconds"),
|
||||
"heartbeat_fresh": liveness.get("heartbeat_fresh"),
|
||||
"live": bool(snapshot_row.get("live")),
|
||||
"ownership_state": snapshot_row.get("ownership_state"),
|
||||
"instance_id_provenance": snapshot_row.get("instance_id_provenance"),
|
||||
"instance_identity_trusted": bool(
|
||||
snapshot_row.get("instance_identity_trusted")
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def _conflict_keys(
|
||||
row: Mapping[str, Any], snapshot_row: Mapping[str, Any]
|
||||
) -> list[tuple[str, str]]:
|
||||
keys: list[tuple[str, str]] = []
|
||||
for name, value in (
|
||||
("session_id", row.get("session_id")),
|
||||
("generation_id", row.get("generation_id")),
|
||||
("process_identity", snapshot_row.get("process_identity")),
|
||||
("pid", row.get("pid")),
|
||||
):
|
||||
text = _canon(value)
|
||||
if text:
|
||||
keys.append((name, text))
|
||||
return keys
|
||||
|
||||
|
||||
def external_state_fingerprint(
|
||||
leases: Iterable[Mapping[str, Any]],
|
||||
*,
|
||||
liveness: Iterable[tuple[Any, Any]] = (),
|
||||
) -> str:
|
||||
"""Version token over the external state a retirement decision consumed.
|
||||
|
||||
#980 review 657 B3: ``BEGIN IMMEDIATE`` on the worker registry does not
|
||||
cover the control-plane lease table or the OS process table, so those
|
||||
inputs can move while the transaction is open. This token lets the
|
||||
transaction detect that movement: it is captured before the authoritative
|
||||
read and re-compared immediately before every guarded write, and any
|
||||
difference aborts rather than retiring against evidence that has changed.
|
||||
|
||||
Only ownership-relevant lease fields participate, so unrelated churn (a
|
||||
heartbeat timestamp advancing on an unrelated lease) does not cause
|
||||
spurious aborts, while an acquire, release, or owner change always does.
|
||||
"""
|
||||
lease_units: list[str] = []
|
||||
for lease in leases:
|
||||
lease_units.append(
|
||||
_UNIT.join(
|
||||
f"{name}={_canon(lease.get(name))}"
|
||||
for name in (
|
||||
"lease_id",
|
||||
"role",
|
||||
"target",
|
||||
"status",
|
||||
"session_id",
|
||||
"owner_session_id",
|
||||
"owner_pid",
|
||||
"session_pid",
|
||||
"generation",
|
||||
)
|
||||
)
|
||||
)
|
||||
for pid, alive in liveness:
|
||||
lease_units.append(f"liveness{_UNIT}pid={_canon(pid)}{_UNIT}alive={_canon(alive)}")
|
||||
return _digest("externalfp-v1", lease_units, "externalfp")
|
||||
|
||||
|
||||
def assess_retirement_identity_proof(
|
||||
row: Mapping[str, Any],
|
||||
snapshot_row: Mapping[str, Any],
|
||||
*,
|
||||
current_host_id: str | None,
|
||||
current_boot_id: str | None,
|
||||
live_start_time: str | None,
|
||||
pid_alive: bool | None,
|
||||
) -> tuple[str, str] | None:
|
||||
"""Affirmative proof that this row names one specific, now-dead process.
|
||||
|
||||
Returns ``None`` when the proof holds, or ``(reason_code, detail)`` naming
|
||||
the first thing that could not be established. #980 review 657 B2: absence
|
||||
of evidence is never read as staleness, so every branch here refuses on
|
||||
*missing* information exactly as firmly as on contradictory information.
|
||||
|
||||
The proof has two independent halves and needs both:
|
||||
|
||||
* **Attribution** — a launcher-minted ``inst-…`` instance identity, so the
|
||||
row is known to belong to one specific application launch rather than
|
||||
having been inferred from pid proximity.
|
||||
* **Fencing** — the row's host matches the host doing the probing, boot
|
||||
identity is known on both sides, and the recorded process incarnation
|
||||
agrees with whatever currently occupies that pid number.
|
||||
|
||||
Requiring trusted attribution means pre-#978 ``legacy-pid-…`` rows are
|
||||
preserved permanently. That is deliberate. The reviewer specifically
|
||||
rejected the argument that legacy rows "would remain forever" as grounds
|
||||
for a weaker proof, and #980 lists backfilling trusted identity for legacy
|
||||
workers as a non-goal — so those rows are retired only after their worker
|
||||
re-registers under a trusted identity, never on weaker evidence.
|
||||
"""
|
||||
if not snapshot_row.get("instance_identity_trusted"):
|
||||
return (
|
||||
REASON_UNTRUSTED_PROVENANCE,
|
||||
"client_instance_id "
|
||||
f"{row.get('client_instance_id')!r} is not launcher-minted "
|
||||
f"({snapshot_row.get('instance_id_provenance')!r}); nothing proves "
|
||||
"which application launch this registration belongs to",
|
||||
)
|
||||
|
||||
recorded_host = (row.get("host_id") or "").strip()
|
||||
if not current_host_id:
|
||||
return (
|
||||
REASON_HOST_UNPROVEN,
|
||||
"this process cannot establish its own host identity, so a local "
|
||||
"pid probe cannot be attributed to any machine",
|
||||
)
|
||||
if recorded_host != current_host_id:
|
||||
return (
|
||||
REASON_HOST_UNPROVEN,
|
||||
f"registration is bound to host {recorded_host!r} but retirement is "
|
||||
f"running on {current_host_id!r}; a local pid probe says nothing "
|
||||
"about a process on another machine",
|
||||
)
|
||||
|
||||
recorded_boot = (row.get("boot_id") or "").strip()
|
||||
if not current_boot_id:
|
||||
return (
|
||||
REASON_BOOT_UNKNOWN,
|
||||
"the current boot identity could not be determined, so a recorded "
|
||||
"pid cannot be compared against a live pid",
|
||||
)
|
||||
|
||||
recorded_start = (row.get("process_start_time") or "").strip()
|
||||
|
||||
if recorded_boot != current_boot_id:
|
||||
# A different boot is the strongest possible death evidence: every pid
|
||||
# from a previous boot is gone, and pid numbers restart, so whatever
|
||||
# occupies this number now is unrelated by construction.
|
||||
return None
|
||||
|
||||
# Same boot: the pid number is directly comparable, so the recorded
|
||||
# incarnation must still agree with whatever holds that number.
|
||||
if pid_alive and live_start_time and live_start_time != recorded_start:
|
||||
return (
|
||||
REASON_PID_REUSED,
|
||||
f"pid {row.get('pid')!r} is alive but started at "
|
||||
f"{live_start_time!r}, not the registered {recorded_start!r}; the "
|
||||
"number was reused by an unrelated process and this registration's "
|
||||
"own liveness is therefore unproven",
|
||||
)
|
||||
if pid_alive:
|
||||
return (
|
||||
REASON_PID_ALIVE,
|
||||
f"recorded pid {row.get('pid')!r} is still running on this host and "
|
||||
"boot",
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _missing_identity_fields(row: Mapping[str, Any]) -> list[str]:
|
||||
missing: list[str] = []
|
||||
for name in REQUIRED_IDENTITY_FIELDS:
|
||||
value = row.get(name)
|
||||
if value is None or (isinstance(value, str) and not value.strip()):
|
||||
missing.append(name)
|
||||
return missing
|
||||
|
||||
|
||||
def plan_stale_worker_retirement(
|
||||
rows: Iterable[Mapping[str, Any]],
|
||||
*,
|
||||
now: datetime | None = None,
|
||||
pid_alive_probe: Callable[[int | None], bool | None] | None = None,
|
||||
canonical_repository: str | None = None,
|
||||
protected_worker_identities: Iterable[str] | None = None,
|
||||
protected_session_ids: Iterable[str] | None = None,
|
||||
protected_pids: Iterable[Any] | None = None,
|
||||
current_host_id: str | None = None,
|
||||
current_boot_id: str | None = None,
|
||||
start_time_probe: Callable[[Any], str | None] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Decide, without mutating anything, which registrations may be retired.
|
||||
|
||||
Every row lands in exactly one of ``candidates`` (eligible) or
|
||||
``preserved`` (with the reason code that stopped it), so the output
|
||||
explains the whole registry rather than only the interesting part.
|
||||
"""
|
||||
all_rows = [dict(row) for row in rows]
|
||||
protected_ids = {str(w) for w in (protected_worker_identities or []) if w}
|
||||
protected_sessions = {str(s) for s in (protected_session_ids or []) if s}
|
||||
protected_pid_set = {_canon(p) for p in (protected_pids or []) if p is not None}
|
||||
|
||||
snapshots: dict[int, dict[str, Any]] = {}
|
||||
pid_alive_by_index: dict[int, bool | None] = {}
|
||||
start_time_by_index: dict[int, str | None] = {}
|
||||
for index, row in enumerate(all_rows):
|
||||
pid_alive = _probe_pid(row.get("pid"), pid_alive_probe)
|
||||
pid_alive_by_index[index] = pid_alive
|
||||
start_time_by_index[index] = _probe_start_time(
|
||||
row.get("pid"), start_time_probe
|
||||
)
|
||||
snapshots[index] = fleet.build_worker_snapshot_row(
|
||||
row,
|
||||
now=now,
|
||||
pid_alive_probe=(lambda _pid, _value=pid_alive: _value),
|
||||
canonical_repository=canonical_repository,
|
||||
)
|
||||
|
||||
# Identity evidence owned by a worker that is live, or whose liveness could
|
||||
# not be established, is ambiguous: anything sharing it is preserved.
|
||||
ambiguous_keys: set[tuple[str, str]] = set()
|
||||
identity_counts: dict[str, int] = {}
|
||||
for index, row in enumerate(all_rows):
|
||||
identity = _canon(row.get("worker_identity"))
|
||||
if identity:
|
||||
identity_counts[identity] = identity_counts.get(identity, 0) + 1
|
||||
snapshot_row = snapshots[index]
|
||||
liveness = snapshot_row.get("liveness") or {}
|
||||
unresolved = (
|
||||
pid_alive_by_index[index] is None
|
||||
or liveness.get("heartbeat_fresh") is None
|
||||
)
|
||||
if snapshot_row.get("live") or (
|
||||
str(row.get("status") or "") == mwi.STATUS_ACTIVE and unresolved
|
||||
):
|
||||
ambiguous_keys.update(_conflict_keys(row, snapshot_row))
|
||||
|
||||
# Two active registrations claiming one (client_instance_id, namespace)
|
||||
# violate the #978 uniqueness invariant — but only when one of them might
|
||||
# still be running. Several *dead* rows accumulating on one slot across
|
||||
# restarts is ordinary history and every one of them is safely retirable;
|
||||
# a slot shared with a live or unprobeable worker is genuinely ambiguous,
|
||||
# because which row that process belongs to cannot be settled from the
|
||||
# registry alone.
|
||||
#
|
||||
# The key is deliberately the (instance, namespace) pair, not the instance
|
||||
# alone: one legitimate cohort is exactly one instance spread across
|
||||
# distinct namespaces, so keying on the instance would make every cohort
|
||||
# look self-conflicting and preserve the whole fleet forever.
|
||||
instance_members: dict[tuple[str, str], list[int]] = {}
|
||||
for index, row in enumerate(all_rows):
|
||||
if str(row.get("status") or "") != mwi.STATUS_ACTIVE:
|
||||
continue
|
||||
key = _instance_key(row)
|
||||
if key is None:
|
||||
continue
|
||||
instance_members.setdefault(key, []).append(index)
|
||||
instance_conflicts: dict[tuple[str, str], bool] = {}
|
||||
for key, members in instance_members.items():
|
||||
if len(members) < 2:
|
||||
continue
|
||||
contested = any(
|
||||
snapshots[i].get("live") or pid_alive_by_index[i] is None
|
||||
for i in members
|
||||
)
|
||||
if contested:
|
||||
instance_conflicts[key] = True
|
||||
|
||||
candidates: list[dict[str, Any]] = []
|
||||
candidate_rows: list[Mapping[str, Any]] = []
|
||||
preserved: list[dict[str, Any]] = []
|
||||
|
||||
for index, row in enumerate(all_rows):
|
||||
snapshot_row = snapshots[index]
|
||||
pid_alive = pid_alive_by_index[index]
|
||||
liveness = snapshot_row.get("liveness") or {}
|
||||
evidence = _evidence(row, snapshot_row, pid_alive)
|
||||
blocked: tuple[str, str] | None = None
|
||||
|
||||
identity = _canon(row.get("worker_identity"))
|
||||
missing = _missing_identity_fields(row)
|
||||
shared = sorted(
|
||||
f"{name}={value}"
|
||||
for name, value in _conflict_keys(row, snapshot_row)
|
||||
if (name, value) in ambiguous_keys
|
||||
)
|
||||
binding = (row.get("repository_binding") or "").strip()
|
||||
protected_hits: list[str] = []
|
||||
if identity and identity in protected_ids:
|
||||
protected_hits.append(f"worker_identity={identity}")
|
||||
if _canon(row.get("session_id")) in protected_sessions:
|
||||
protected_hits.append(f"session_id={_canon(row.get('session_id'))}")
|
||||
if _canon(row.get("pid")) in protected_pid_set:
|
||||
protected_hits.append(f"pid={_canon(row.get('pid'))}")
|
||||
|
||||
if identity and identity_counts.get(identity, 0) > 1:
|
||||
blocked = (
|
||||
REASON_CONFLICTING_IDENTITY,
|
||||
f"worker identity {identity!r} appears on more than one registry row",
|
||||
)
|
||||
elif str(row.get("status") or "") != mwi.STATUS_ACTIVE:
|
||||
blocked = (
|
||||
REASON_ALREADY_TERMINAL,
|
||||
f"registration status is {row.get('status')!r}; nothing to retire",
|
||||
)
|
||||
elif missing:
|
||||
blocked = (
|
||||
REASON_INCOMPLETE_IDENTITY,
|
||||
"registry row is missing field(s) the retirement conjunction "
|
||||
f"reads: {missing}",
|
||||
)
|
||||
elif mwi._parse_ts(row.get("last_heartbeat_at")) is None:
|
||||
blocked = (
|
||||
REASON_UNPARSABLE_HEARTBEAT,
|
||||
"last_heartbeat_at is not a parsable UTC stamp; liveness is unknown",
|
||||
)
|
||||
elif snapshot_row.get("live"):
|
||||
blocked = (REASON_WORKER_LIVE, "worker is live and must not be retired")
|
||||
elif pid_alive is None:
|
||||
blocked = (
|
||||
REASON_PID_UNKNOWN,
|
||||
f"pid {row.get('pid')!r} could not be probed; liveness is unproven",
|
||||
)
|
||||
elif pid_alive and _reuse_detected(
|
||||
row, start_time_by_index[index], current_host_id, current_boot_id
|
||||
):
|
||||
# Reported before the generic live-pid branch so the operator sees
|
||||
# *why* the number is occupied: an unrelated process inherited it,
|
||||
# which means this registration's own liveness is unproven rather
|
||||
# than positively established.
|
||||
blocked = (
|
||||
REASON_PID_REUSED,
|
||||
f"pid {row.get('pid')!r} is alive but started at "
|
||||
f"{start_time_by_index[index]!r}, not the registered "
|
||||
f"{row.get('process_start_time')!r}; the number was reused",
|
||||
)
|
||||
elif pid_alive:
|
||||
blocked = (
|
||||
REASON_PID_ALIVE,
|
||||
f"recorded pid {row.get('pid')!r} is still running",
|
||||
)
|
||||
elif liveness.get("heartbeat_fresh") is not False:
|
||||
blocked = (
|
||||
REASON_HEARTBEAT_FRESH,
|
||||
"heartbeat has not expired under the canonical TTL policy",
|
||||
)
|
||||
elif snapshot_row.get("ownership_state") != "stale":
|
||||
blocked = (
|
||||
REASON_AMBIGUOUS_OWNERSHIP,
|
||||
"ownership_state is "
|
||||
f"{snapshot_row.get('ownership_state')!r}, not 'stale'",
|
||||
)
|
||||
elif not binding or snapshot_row.get("foreign_repository"):
|
||||
blocked = (
|
||||
REASON_FOREIGN_REPOSITORY,
|
||||
"repository binding is absent or does not match the canonical "
|
||||
"repository; retirement scope is ambiguous",
|
||||
)
|
||||
elif shared:
|
||||
blocked = (
|
||||
REASON_CONFLICTING_IDENTITY,
|
||||
"identity evidence is shared with a live or unprobeable worker: "
|
||||
f"{shared}",
|
||||
)
|
||||
elif protected_hits:
|
||||
blocked = (
|
||||
REASON_PROTECTED_OWNER,
|
||||
"worker still owns active workflow state requiring separate "
|
||||
f"reconciliation: {protected_hits}",
|
||||
)
|
||||
elif instance_conflicts.get(_instance_key(row)):
|
||||
blocked = (
|
||||
REASON_INSTANCE_CONFLICT,
|
||||
"another active registration claims client_instance_id "
|
||||
f"{row.get('client_instance_id')!r} in namespace "
|
||||
f"{row.get('namespace')!r}; instance ownership is ambiguous",
|
||||
)
|
||||
else:
|
||||
# Affirmative identity + fencing proof runs last: everything above
|
||||
# establishes the row is *inert*, and this establishes it is
|
||||
# unambiguously *this* worker (#980 review 657 B2).
|
||||
blocked = assess_retirement_identity_proof(
|
||||
row,
|
||||
snapshot_row,
|
||||
current_host_id=current_host_id,
|
||||
current_boot_id=current_boot_id,
|
||||
live_start_time=start_time_by_index[index],
|
||||
pid_alive=pid_alive,
|
||||
)
|
||||
|
||||
if blocked is not None:
|
||||
preserved.append(
|
||||
{
|
||||
"worker_identity": row.get("worker_identity"),
|
||||
"reason_code": blocked[0],
|
||||
"detail": blocked[1],
|
||||
"evidence": evidence,
|
||||
}
|
||||
)
|
||||
continue
|
||||
|
||||
candidates.append(
|
||||
{
|
||||
"worker_identity": row.get("worker_identity"),
|
||||
"reason_code": REASON_ELIGIBLE,
|
||||
"detail": (
|
||||
"dead pid, expired heartbeat, stale ownership, unambiguous "
|
||||
"identity, canonical repository binding, no active workflow "
|
||||
"ownership"
|
||||
),
|
||||
"evidence": evidence,
|
||||
}
|
||||
)
|
||||
candidate_rows.append(row)
|
||||
|
||||
counts: dict[str, int] = {}
|
||||
for entry in preserved:
|
||||
counts[entry["reason_code"]] = counts.get(entry["reason_code"], 0) + 1
|
||||
|
||||
return {
|
||||
"success": True,
|
||||
"read_only": True,
|
||||
"mutation_performed": False,
|
||||
"outcome": OUTCOME_PLANNED,
|
||||
"registry_fingerprint": registry_fingerprint(all_rows),
|
||||
"candidate_fingerprint": candidate_fingerprint(candidate_rows),
|
||||
"assessed_count": len(all_rows),
|
||||
"candidate_count": len(candidates),
|
||||
"preserved_count": len(preserved),
|
||||
"candidates": candidates,
|
||||
"candidate_worker_identities": [c["worker_identity"] for c in candidates],
|
||||
"preserved": preserved,
|
||||
"preserved_reason_counts": counts,
|
||||
"protected_inputs": {
|
||||
"worker_identities": sorted(protected_ids),
|
||||
"session_ids": sorted(protected_sessions),
|
||||
"pids": sorted(protected_pid_set),
|
||||
},
|
||||
"canonical_repository": canonical_repository,
|
||||
"fencing_context": {
|
||||
"current_host_id": current_host_id,
|
||||
"current_boot_id": current_boot_id,
|
||||
"start_time_probe_available": start_time_probe is not None,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def summarize_plan(plan: Mapping[str, Any]) -> dict[str, Any]:
|
||||
"""Compact, log-safe view of a plan or apply result."""
|
||||
return {
|
||||
"outcome": plan.get("outcome"),
|
||||
"registry_fingerprint": plan.get("registry_fingerprint"),
|
||||
"candidate_fingerprint": plan.get("candidate_fingerprint"),
|
||||
"assessed_count": plan.get("assessed_count"),
|
||||
"candidate_count": plan.get("candidate_count"),
|
||||
"retired_count": plan.get("retired_count"),
|
||||
"preserved_count": plan.get("preserved_count"),
|
||||
"mutation_performed": plan.get("mutation_performed"),
|
||||
}
|
||||
@@ -0,0 +1,869 @@
|
||||
"""Instance-level fleet identity and health snapshots (#978).
|
||||
|
||||
#948 established per-worker ownership; #975 made heartbeats keep those rows
|
||||
live. Neither surface could enumerate the fleet at *instance* granularity:
|
||||
which application launch owns which five namespace workers, whether two
|
||||
Codex launches are distinct, or whether a live collision is real rather than
|
||||
a shared client type.
|
||||
|
||||
This module is pure. Callers supply registry rows (and optional enrichments);
|
||||
nothing here opens SQLite, scans process tables, or mutates state. Production
|
||||
evidence for the fleet gate is the snapshot returned by the sanctioned
|
||||
controller/reconciler tool that wraps this assessor.
|
||||
|
||||
Identity hierarchy (highest → lowest):
|
||||
|
||||
* ``client_type`` — application family (``codex``, ``claude_code``, …)
|
||||
* ``client_instance_id`` — one running application launch (trusted launcher)
|
||||
* ``fleet_run_id`` — operator-approved enrollment / canary cohort
|
||||
* ``worker_id`` / ``worker_identity`` — one namespace worker process
|
||||
* ``namespace`` — author | reviewer | merger | controller | reconciler
|
||||
|
||||
Multiple simultaneous instances of the same ``client_type`` are first-class.
|
||||
Sharing only a profile or client type is never a duplicate.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import secrets
|
||||
from collections import defaultdict
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Callable, Iterable, Mapping
|
||||
|
||||
import mcp_worker_identity as mwi
|
||||
|
||||
# --- Classification labels ------------------------------------------------
|
||||
|
||||
CLASS_EXPECTED = "expected_enrolled"
|
||||
CLASS_MISSING = "missing_expected"
|
||||
CLASS_UNMANIFESTED = "unmanifested"
|
||||
CLASS_DUPLICATE_NAMESPACE = "duplicate_namespace_worker"
|
||||
CLASS_INSTANCE_ID_COLLISION = "instance_id_collision"
|
||||
CLASS_WORKER_ID_COLLISION = "worker_identity_collision"
|
||||
CLASS_SESSION_COLLISION = "session_identity_collision"
|
||||
CLASS_GENERATION_COLLISION = "generation_identity_collision"
|
||||
CLASS_PROCESS_COLLISION = "process_identity_collision"
|
||||
CLASS_PID_COLLISION = "pid_collision"
|
||||
CLASS_OWNERSHIP_COLLISION = "ownership_fencing_collision"
|
||||
CLASS_ORPHANED = "orphaned_unowned"
|
||||
CLASS_UNKNOWN_CLIENT = "unknown_client"
|
||||
CLASS_FOREIGN_REPOSITORY = "foreign_repository"
|
||||
CLASS_OLD_REVISION = "old_revision"
|
||||
CLASS_STALE_WORKER = "stale_orphaned_worker"
|
||||
CLASS_LEGACY_INCOMPLETE = "legacy_incomplete_identity"
|
||||
CLASS_HISTORICAL = "historical_dead"
|
||||
CLASS_HEALTHY = "healthy"
|
||||
|
||||
#: Active blockers that make the live fleet unsafe for mutation-gated work.
|
||||
ACTIVE_BLOCKER_CLASSES = frozenset(
|
||||
{
|
||||
CLASS_MISSING,
|
||||
CLASS_UNMANIFESTED,
|
||||
CLASS_DUPLICATE_NAMESPACE,
|
||||
CLASS_INSTANCE_ID_COLLISION,
|
||||
CLASS_WORKER_ID_COLLISION,
|
||||
CLASS_SESSION_COLLISION,
|
||||
CLASS_GENERATION_COLLISION,
|
||||
CLASS_PROCESS_COLLISION,
|
||||
CLASS_PID_COLLISION,
|
||||
CLASS_OWNERSHIP_COLLISION,
|
||||
CLASS_ORPHANED,
|
||||
CLASS_UNKNOWN_CLIENT,
|
||||
CLASS_FOREIGN_REPOSITORY,
|
||||
CLASS_OLD_REVISION,
|
||||
CLASS_STALE_WORKER,
|
||||
CLASS_LEGACY_INCOMPLETE,
|
||||
}
|
||||
)
|
||||
|
||||
SANCTIONED_NAMESPACES = frozenset(
|
||||
{"author", "reviewer", "merger", "controller", "reconciler"}
|
||||
)
|
||||
|
||||
INSTANCE_ID_PROVENANCE_TRUSTED = "trusted_launcher"
|
||||
INSTANCE_ID_PROVENANCE_LEGACY = "legacy_incomplete"
|
||||
INSTANCE_ID_PROVENANCE_MISSING = "missing"
|
||||
|
||||
CLIENT_INSTANCE_ENV = "GITEA_MCP_CLIENT_INSTANCE"
|
||||
FLEET_RUN_ENV = "GITEA_MCP_FLEET_RUN_ID"
|
||||
PROCESS_IDENTITY_ENV = "GITEA_MCP_PROCESS_IDENTITY"
|
||||
INSTANCE_PROVENANCE_ENV = "GITEA_MCP_INSTANCE_PROVENANCE"
|
||||
|
||||
_LEGACY_INSTANCE_PREFIXES = ("pid-", "proc-", "legacy-")
|
||||
#: Trusted launcher-issued IDs use the reserved ``inst-`` prefix
|
||||
#: (see :func:`generate_client_instance_id`). Ordinary user-supplied strings
|
||||
#: without that prefix never count as trusted attribution.
|
||||
_TRUSTED_INSTANCE_PREFIX = "inst-"
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def _ts(value: datetime) -> str:
|
||||
return value.astimezone(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||
|
||||
|
||||
def generate_client_instance_id(
|
||||
client_type: str | None,
|
||||
*,
|
||||
launch_nonce: str | None = None,
|
||||
now: datetime | None = None,
|
||||
) -> str:
|
||||
"""Mint a distinct instance ID for one application launch (#978).
|
||||
|
||||
The trusted launcher (or host that starts all five namespace workers)
|
||||
generates this once per launch and injects it as ``GITEA_MCP_CLIENT_INSTANCE``
|
||||
into every worker environment. Workers never invent their own instance ID
|
||||
from PID proximity or timestamps.
|
||||
"""
|
||||
client = mwi.normalize_client_name(client_type)
|
||||
stamp = (now or _utc_now()).astimezone(timezone.utc).strftime(
|
||||
mwi.IDENTITY_TIMESTAMP_FORMAT
|
||||
)
|
||||
nonce = launch_nonce if launch_nonce is not None else secrets.token_hex(16)
|
||||
digest = hashlib.sha256(
|
||||
f"{client}\x1f{stamp}\x1f{nonce}".encode("utf-8")
|
||||
).hexdigest()[:12]
|
||||
return f"inst-{client}-{stamp}-{digest}"
|
||||
|
||||
|
||||
def assess_instance_identity(
|
||||
raw_instance_id: str | None,
|
||||
*,
|
||||
source: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Classify whether a client_instance_id is trusted enough for mutation.
|
||||
|
||||
Trusted instance IDs are non-empty, match the launcher-minted ``inst-…``
|
||||
format from :func:`generate_client_instance_id`, and are not pre-#978
|
||||
PID/proc/legacy placeholders. Ordinary user-supplied or malformed values
|
||||
fail closed as untrusted so they cannot spoof multi-instance attribution.
|
||||
Incomplete identities remain visible for diagnosis but cannot authorize
|
||||
unsafe mutation.
|
||||
"""
|
||||
text = (raw_instance_id or "").strip()
|
||||
if not text:
|
||||
return {
|
||||
"client_instance_id": None,
|
||||
"complete": False,
|
||||
"trusted": False,
|
||||
"provenance": INSTANCE_ID_PROVENANCE_MISSING,
|
||||
"reasons": [
|
||||
"client_instance_id is missing; the trusted launcher must set "
|
||||
f"{CLIENT_INSTANCE_ENV} once per application launch"
|
||||
],
|
||||
}
|
||||
lowered = text.lower()
|
||||
if lowered.startswith(_LEGACY_INSTANCE_PREFIXES) or source == "pid_fallback":
|
||||
return {
|
||||
"client_instance_id": text,
|
||||
"complete": False,
|
||||
"trusted": False,
|
||||
"provenance": INSTANCE_ID_PROVENANCE_LEGACY,
|
||||
"reasons": [
|
||||
f"client_instance_id {text!r} is a legacy PID/process fallback, "
|
||||
"not a trusted launcher-issued instance identity"
|
||||
],
|
||||
}
|
||||
if not text.startswith(_TRUSTED_INSTANCE_PREFIX) or len(text) <= len(
|
||||
_TRUSTED_INSTANCE_PREFIX
|
||||
):
|
||||
return {
|
||||
"client_instance_id": text,
|
||||
"complete": False,
|
||||
"trusted": False,
|
||||
"provenance": INSTANCE_ID_PROVENANCE_LEGACY,
|
||||
"reasons": [
|
||||
f"client_instance_id {text!r} is malformed or user-supplied and "
|
||||
f"does not use the trusted launcher prefix "
|
||||
f"{_TRUSTED_INSTANCE_PREFIX!r}; refusing trusted attribution"
|
||||
],
|
||||
}
|
||||
return {
|
||||
"client_instance_id": text,
|
||||
"complete": True,
|
||||
"trusted": True,
|
||||
"provenance": INSTANCE_ID_PROVENANCE_TRUSTED,
|
||||
"reasons": [],
|
||||
}
|
||||
|
||||
|
||||
def resolve_client_instance_from_env(
|
||||
env: Mapping[str, str] | None = None,
|
||||
*,
|
||||
pid: int | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Resolve instance identity from launcher env without inventing one.
|
||||
|
||||
When the trusted key is absent, return incomplete evidence rather than a
|
||||
silent ``pid-<n>`` identity. Callers that still need a non-empty registry
|
||||
key may choose a legacy placeholder deliberately; they must not treat it as
|
||||
trusted.
|
||||
"""
|
||||
source = dict(env or {})
|
||||
raw = (source.get(CLIENT_INSTANCE_ENV) or "").strip()
|
||||
assessment = assess_instance_identity(raw or None)
|
||||
assessment["fleet_run_id"] = (source.get(FLEET_RUN_ENV) or "").strip() or None
|
||||
assessment["process_identity"] = (
|
||||
(source.get(PROCESS_IDENTITY_ENV) or "").strip()
|
||||
or (f"pid-{pid}" if pid is not None else None)
|
||||
)
|
||||
return assessment
|
||||
|
||||
|
||||
def _public_worker(record: Mapping[str, Any]) -> dict[str, Any]:
|
||||
return {
|
||||
"worker_id": record.get("worker_identity") or record.get("worker_id"),
|
||||
"worker_identity": record.get("worker_identity") or record.get("worker_id"),
|
||||
"client_type": mwi.normalize_client_name(
|
||||
record.get("client_name") or record.get("client_type")
|
||||
),
|
||||
"client_instance_id": record.get("client_instance_id"),
|
||||
"fleet_run_id": record.get("fleet_run_id"),
|
||||
"namespace": record.get("namespace"),
|
||||
"profile": record.get("profile"),
|
||||
"declared_role": record.get("role") or record.get("declared_role"),
|
||||
"authenticated_account": record.get("authenticated_account"),
|
||||
"session_id": record.get("session_id"),
|
||||
"generation_id": record.get("generation_id"),
|
||||
"process_identity": record.get("process_identity")
|
||||
or (
|
||||
f"pid-{record['pid']}"
|
||||
if record.get("pid") is not None
|
||||
else None
|
||||
),
|
||||
"pid": record.get("pid"),
|
||||
"repository_binding": record.get("repository_binding"),
|
||||
"remote": record.get("remote"),
|
||||
"startup_revision": record.get("startup_revision"),
|
||||
"loaded_revision": record.get("loaded_revision"),
|
||||
"parity_revision": record.get("parity_revision"),
|
||||
"live_revision": record.get("live_revision"),
|
||||
"runtime_provenance": record.get("runtime_provenance")
|
||||
or record.get("transport"),
|
||||
"transport": record.get("transport"),
|
||||
"started_at": record.get("started_at"),
|
||||
"last_heartbeat_at": record.get("last_heartbeat_at"),
|
||||
"heartbeat_ttl_seconds": record.get("heartbeat_ttl_seconds"),
|
||||
"fencing_epoch": record.get("fencing_epoch"),
|
||||
"status": record.get("status"),
|
||||
"instance_id_provenance": record.get("instance_id_provenance"),
|
||||
}
|
||||
|
||||
|
||||
def _liveness(
|
||||
record: Mapping[str, Any],
|
||||
*,
|
||||
now: datetime | None,
|
||||
pid_alive_probe: Callable[[int | None], bool | None] | None,
|
||||
) -> dict[str, Any]:
|
||||
pid_alive = None
|
||||
if pid_alive_probe is not None and record.get("pid") is not None:
|
||||
try:
|
||||
pid_alive = pid_alive_probe(record.get("pid"))
|
||||
except Exception:
|
||||
pid_alive = None
|
||||
return mwi.WorkerRegistry.is_live(record, now=now, pid_alive=pid_alive)
|
||||
|
||||
|
||||
def _consistency_token(rows: Iterable[Mapping[str, Any]], snapshot_at: str) -> str:
|
||||
material = [snapshot_at]
|
||||
for row in sorted(
|
||||
rows,
|
||||
key=lambda r: (
|
||||
str(r.get("worker_identity") or ""),
|
||||
str(r.get("last_heartbeat_at") or ""),
|
||||
str(r.get("fencing_epoch") or ""),
|
||||
),
|
||||
):
|
||||
material.append(
|
||||
"|".join(
|
||||
[
|
||||
str(row.get("worker_identity") or ""),
|
||||
str(row.get("client_instance_id") or ""),
|
||||
str(row.get("status") or ""),
|
||||
str(row.get("last_heartbeat_at") or ""),
|
||||
str(row.get("fencing_epoch") or ""),
|
||||
str(row.get("generation_id") or ""),
|
||||
]
|
||||
)
|
||||
)
|
||||
digest = hashlib.sha256("\n".join(material).encode("utf-8")).hexdigest()[:16]
|
||||
return f"fleetrev-{digest}"
|
||||
|
||||
|
||||
def build_worker_snapshot_row(
|
||||
record: Mapping[str, Any],
|
||||
*,
|
||||
now: datetime | None = None,
|
||||
pid_alive_probe: Callable[[int | None], bool | None] | None = None,
|
||||
canonical_repository: str | None = None,
|
||||
expected_live_revision: str | None = None,
|
||||
heartbeat_supervised: bool | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""One point-in-time worker row for the fleet snapshot."""
|
||||
stamp = now or _utc_now()
|
||||
base = _public_worker(record)
|
||||
identity = assess_instance_identity(
|
||||
base.get("client_instance_id"),
|
||||
source=record.get("instance_id_source"),
|
||||
)
|
||||
liveness = _liveness(record, now=stamp, pid_alive_probe=pid_alive_probe)
|
||||
is_historical = str(record.get("status") or "") != mwi.STATUS_ACTIVE
|
||||
live = bool(liveness.get("live")) and not is_historical
|
||||
|
||||
repo = (base.get("repository_binding") or "").strip() or None
|
||||
foreign_repo = bool(
|
||||
canonical_repository
|
||||
and repo
|
||||
and repo.rstrip("/") != str(canonical_repository).rstrip("/")
|
||||
)
|
||||
old_revision = False
|
||||
if expected_live_revision:
|
||||
for key in ("startup_revision", "loaded_revision", "parity_revision", "live_revision"):
|
||||
rev = (base.get(key) or "").strip()
|
||||
if rev and rev != expected_live_revision:
|
||||
old_revision = True
|
||||
break
|
||||
|
||||
ownership_state = "historical" if is_historical else (
|
||||
"live" if live else "stale"
|
||||
)
|
||||
if live and not identity["trusted"]:
|
||||
ownership_state = "live_untrusted_identity"
|
||||
if live and not base.get("session_id"):
|
||||
ownership_state = "orphaned"
|
||||
|
||||
mutation_safe = bool(
|
||||
live
|
||||
and identity["trusted"]
|
||||
and not foreign_repo
|
||||
and not old_revision
|
||||
and ownership_state == "live"
|
||||
and base.get("client_type") != mwi.UNKNOWN_CLIENT
|
||||
)
|
||||
|
||||
restart_required = bool(
|
||||
old_revision
|
||||
or (live and not liveness.get("heartbeat_fresh", True))
|
||||
)
|
||||
|
||||
return {
|
||||
**base,
|
||||
"instance_identity": identity,
|
||||
"client_instance_id": identity["client_instance_id"] or base.get("client_instance_id"),
|
||||
"instance_id_provenance": identity["provenance"],
|
||||
"instance_identity_trusted": identity["trusted"],
|
||||
"live": live,
|
||||
"historical": is_historical,
|
||||
"liveness": liveness,
|
||||
"heartbeat": {
|
||||
"registered": bool(base.get("last_heartbeat_at")),
|
||||
"supervised": heartbeat_supervised,
|
||||
"age_seconds": liveness.get("heartbeat_age_seconds"),
|
||||
"ttl_seconds": liveness.get("heartbeat_ttl_seconds"),
|
||||
"fresh": liveness.get("heartbeat_fresh"),
|
||||
"last_heartbeat_at": base.get("last_heartbeat_at"),
|
||||
},
|
||||
"fencing": {
|
||||
"fencing_epoch": base.get("fencing_epoch"),
|
||||
"generation_id": base.get("generation_id"),
|
||||
},
|
||||
"ownership_state": ownership_state,
|
||||
"foreign_repository": foreign_repo,
|
||||
"old_revision": old_revision,
|
||||
"stale": not live and not is_historical,
|
||||
"restart_required": restart_required,
|
||||
"mutation_safe": mutation_safe,
|
||||
"conflicting_live_sessions": [],
|
||||
}
|
||||
|
||||
|
||||
def _collision_groups(
|
||||
live_rows: list[dict[str, Any]],
|
||||
key_fn,
|
||||
) -> dict[str, list[dict[str, Any]]]:
|
||||
groups: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
||||
for row in live_rows:
|
||||
key = key_fn(row)
|
||||
if key is None or key == "" or key == "None":
|
||||
continue
|
||||
groups[str(key)].append(row)
|
||||
return {k: v for k, v in groups.items() if len(v) > 1}
|
||||
|
||||
|
||||
def snapshot_instance_fleet(
|
||||
records: Iterable[Mapping[str, Any]],
|
||||
*,
|
||||
expected_manifest: list[Mapping[str, Any]] | None = None,
|
||||
now: datetime | None = None,
|
||||
pid_alive_probe: Callable[[int | None], bool | None] | None = None,
|
||||
canonical_repository: str | None = None,
|
||||
expected_live_revision: str | None = None,
|
||||
registry_revision: str | None = None,
|
||||
known_client_types: Iterable[str] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Authoritative point-in-time fleet snapshot with classification (#978).
|
||||
|
||||
Historical dead rows are reported separately and never automatically make
|
||||
the live fleet unsafe.
|
||||
"""
|
||||
stamp = now or _utc_now()
|
||||
snapshot_at = _ts(stamp)
|
||||
known = {
|
||||
mwi.normalize_client_name(c)
|
||||
for c in (known_client_types or mwi.CLIENT_ALIASES.values())
|
||||
}
|
||||
known.discard(mwi.UNKNOWN_CLIENT)
|
||||
|
||||
all_rows: list[dict[str, Any]] = []
|
||||
for record in records:
|
||||
all_rows.append(
|
||||
build_worker_snapshot_row(
|
||||
record,
|
||||
now=stamp,
|
||||
pid_alive_probe=pid_alive_probe,
|
||||
canonical_repository=canonical_repository,
|
||||
expected_live_revision=expected_live_revision,
|
||||
)
|
||||
)
|
||||
|
||||
live_rows = [r for r in all_rows if r["live"]]
|
||||
historical_rows = [r for r in all_rows if r["historical"]]
|
||||
stale_rows = [r for r in all_rows if r["stale"]]
|
||||
|
||||
# --- identity collisions among live workers ---
|
||||
findings: list[dict[str, Any]] = []
|
||||
|
||||
def _finding(
|
||||
classification: str,
|
||||
*,
|
||||
severity: str,
|
||||
workers: list[dict[str, Any]] | None = None,
|
||||
instance_ids: list[str] | None = None,
|
||||
detail: str,
|
||||
active_blocker: bool,
|
||||
) -> None:
|
||||
findings.append(
|
||||
{
|
||||
"classification": classification,
|
||||
"severity": severity,
|
||||
"active_blocker": active_blocker,
|
||||
"detail": detail,
|
||||
"client_instance_ids": instance_ids or sorted(
|
||||
{
|
||||
str(w.get("client_instance_id"))
|
||||
for w in (workers or [])
|
||||
if w.get("client_instance_id")
|
||||
}
|
||||
),
|
||||
"worker_identities": [
|
||||
w.get("worker_identity") for w in (workers or [])
|
||||
],
|
||||
}
|
||||
)
|
||||
|
||||
# Duplicate worker identity (should not happen with PK, still detect)
|
||||
for wid, group in _collision_groups(
|
||||
live_rows, lambda r: r.get("worker_identity")
|
||||
).items():
|
||||
_finding(
|
||||
CLASS_WORKER_ID_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=f"worker identity {wid!r} is claimed by {len(group)} live workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Reused session identity across live workers
|
||||
for sid, group in _collision_groups(live_rows, lambda r: r.get("session_id")).items():
|
||||
# Same session may appear once; collision only when multiple workers share it
|
||||
# across different worker identities (always true for group size > 1).
|
||||
_finding(
|
||||
CLASS_SESSION_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=f"session identity {sid!r} is reused by {len(group)} live workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Generation claimed by multiple live sessions/workers is a conflict when
|
||||
# the workers are not the five sanctioned namespaces of one instance.
|
||||
for gen, group in _collision_groups(
|
||||
live_rows, lambda r: r.get("generation_id")
|
||||
).items():
|
||||
namespaces = {g.get("namespace") for g in group if g.get("namespace")}
|
||||
instance_ids = {g.get("client_instance_id") for g in group}
|
||||
# Multiple workers under one generation is only valid if they share one
|
||||
# instance and distinct namespaces. Same generation + same namespace = bad.
|
||||
by_ns: dict[str, list] = defaultdict(list)
|
||||
for g in group:
|
||||
by_ns[str(g.get("namespace") or "")].append(g)
|
||||
ns_dups = {ns: rows for ns, rows in by_ns.items() if ns and len(rows) > 1}
|
||||
if ns_dups or len(instance_ids) > 1:
|
||||
_finding(
|
||||
CLASS_GENERATION_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=(
|
||||
f"generation {gen!r} is contested across namespaces/instances "
|
||||
f"(namespaces={sorted(namespaces)}, "
|
||||
f"instances={sorted(str(i) for i in instance_ids if i)})"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Process identity / PID collisions across distinct workers
|
||||
for proc, group in _collision_groups(
|
||||
live_rows, lambda r: r.get("process_identity")
|
||||
).items():
|
||||
if len({r.get("worker_identity") for r in group}) > 1:
|
||||
_finding(
|
||||
CLASS_PROCESS_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=f"process identity {proc!r} is shared by distinct live workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
for pid, group in _collision_groups(live_rows, lambda r: r.get("pid")).items():
|
||||
if len({r.get("worker_identity") for r in group}) > 1:
|
||||
_finding(
|
||||
CLASS_PID_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=f"PID {pid} is shared by distinct live workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Fencing/ownership: same fencing epoch on different workers of different instances
|
||||
for epoch, group in _collision_groups(
|
||||
live_rows,
|
||||
lambda r: (
|
||||
f"{r.get('generation_id')}:{r.get('fencing_epoch')}"
|
||||
if r.get("generation_id") is not None and r.get("fencing_epoch") is not None
|
||||
else None
|
||||
),
|
||||
).items():
|
||||
if len({r.get("client_instance_id") for r in group}) > 1:
|
||||
_finding(
|
||||
CLASS_OWNERSHIP_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=(
|
||||
f"fencing token {epoch!r} spans more than one client_instance_id"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Per-instance grouping
|
||||
by_instance: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
||||
unkeyed_live: list[dict[str, Any]] = []
|
||||
for row in live_rows:
|
||||
iid = row.get("client_instance_id")
|
||||
if not iid:
|
||||
unkeyed_live.append(row)
|
||||
continue
|
||||
by_instance[str(iid)].append(row)
|
||||
|
||||
instances: list[dict[str, Any]] = []
|
||||
for iid, workers in sorted(by_instance.items()):
|
||||
client_types = sorted({w.get("client_type") for w in workers if w.get("client_type")})
|
||||
trusted = all(w.get("instance_identity_trusted") for w in workers)
|
||||
namespaces = [w.get("namespace") for w in workers]
|
||||
ns_counts: dict[str, int] = defaultdict(int)
|
||||
for ns in namespaces:
|
||||
if ns:
|
||||
ns_counts[str(ns)] += 1
|
||||
dup_ns = sorted(ns for ns, n in ns_counts.items() if n > 1)
|
||||
if dup_ns:
|
||||
_finding(
|
||||
CLASS_DUPLICATE_NAMESPACE,
|
||||
severity="blocker",
|
||||
workers=[w for w in workers if w.get("namespace") in dup_ns],
|
||||
instance_ids=[iid],
|
||||
detail=(
|
||||
f"instance {iid!r} has more than one live worker for "
|
||||
f"namespace(s) {dup_ns}"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Live reuse of one instance ID with incompatible client types
|
||||
if len(client_types) > 1:
|
||||
_finding(
|
||||
CLASS_INSTANCE_ID_COLLISION,
|
||||
severity="blocker",
|
||||
workers=workers,
|
||||
instance_ids=[iid],
|
||||
detail=(
|
||||
f"client_instance_id {iid!r} is live under multiple client "
|
||||
f"types {client_types}"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
if not trusted:
|
||||
_finding(
|
||||
CLASS_LEGACY_INCOMPLETE,
|
||||
severity="blocker",
|
||||
workers=workers,
|
||||
instance_ids=[iid],
|
||||
detail=(
|
||||
f"instance {iid!r} lacks trusted launcher-issued instance "
|
||||
"identity; diagnostic reads remain available"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
unknown = [w for w in workers if w.get("client_type") == mwi.UNKNOWN_CLIENT]
|
||||
if unknown:
|
||||
_finding(
|
||||
CLASS_UNKNOWN_CLIENT,
|
||||
severity="blocker",
|
||||
workers=unknown,
|
||||
instance_ids=[iid],
|
||||
detail=f"instance {iid!r} has worker(s) with unknown client_type",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
foreign = [w for w in workers if w.get("foreign_repository")]
|
||||
if foreign:
|
||||
_finding(
|
||||
CLASS_FOREIGN_REPOSITORY,
|
||||
severity="blocker",
|
||||
workers=foreign,
|
||||
instance_ids=[iid],
|
||||
detail=f"instance {iid!r} has foreign-repository workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
old = [w for w in workers if w.get("old_revision")]
|
||||
if old:
|
||||
_finding(
|
||||
CLASS_OLD_REVISION,
|
||||
severity="blocker",
|
||||
workers=old,
|
||||
instance_ids=[iid],
|
||||
detail=f"instance {iid!r} has old-revision workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
orphans = [w for w in workers if w.get("ownership_state") == "orphaned"]
|
||||
if orphans:
|
||||
_finding(
|
||||
CLASS_ORPHANED,
|
||||
severity="blocker",
|
||||
workers=orphans,
|
||||
instance_ids=[iid],
|
||||
detail=f"instance {iid!r} has orphaned/unowned workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
instances.append(
|
||||
{
|
||||
"client_instance_id": iid,
|
||||
"client_types": client_types,
|
||||
"client_type": client_types[0] if len(client_types) == 1 else None,
|
||||
"fleet_run_ids": sorted(
|
||||
{w.get("fleet_run_id") for w in workers if w.get("fleet_run_id")}
|
||||
),
|
||||
"worker_count": len(workers),
|
||||
"namespaces": sorted({n for n in namespaces if n}),
|
||||
"namespace_counts": dict(ns_counts),
|
||||
"duplicate_namespaces": dup_ns,
|
||||
"trusted_instance_identity": trusted,
|
||||
"workers": workers,
|
||||
"mutation_safe": all(w.get("mutation_safe") for w in workers)
|
||||
and not dup_ns
|
||||
and trusted,
|
||||
}
|
||||
)
|
||||
|
||||
for row in unkeyed_live:
|
||||
_finding(
|
||||
CLASS_LEGACY_INCOMPLETE,
|
||||
severity="blocker",
|
||||
workers=[row],
|
||||
detail="live worker has no client_instance_id",
|
||||
active_blocker=True,
|
||||
)
|
||||
if row.get("client_type") == mwi.UNKNOWN_CLIENT:
|
||||
_finding(
|
||||
CLASS_UNKNOWN_CLIENT,
|
||||
severity="blocker",
|
||||
workers=[row],
|
||||
detail="live worker has unknown client_type and no instance id",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
for row in stale_rows:
|
||||
_finding(
|
||||
CLASS_STALE_WORKER,
|
||||
severity="warning",
|
||||
workers=[row],
|
||||
detail=(
|
||||
f"worker {row.get('worker_identity')!r} is active in the registry "
|
||||
"but not live (stale heartbeat or dead pid)"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
for row in historical_rows:
|
||||
_finding(
|
||||
CLASS_HISTORICAL,
|
||||
severity="info",
|
||||
workers=[row],
|
||||
detail=(
|
||||
f"historical registration {row.get('worker_identity')!r} "
|
||||
f"(status={row.get('status')!r}) is not an active blocker"
|
||||
),
|
||||
active_blocker=False,
|
||||
)
|
||||
|
||||
# Manifest comparison
|
||||
expected = list(expected_manifest or [])
|
||||
expected_ids = {
|
||||
str(item.get("client_instance_id")).strip()
|
||||
for item in expected
|
||||
if (item.get("client_instance_id") or "").strip()
|
||||
}
|
||||
live_ids = set(by_instance.keys())
|
||||
missing_ids = sorted(expected_ids - live_ids)
|
||||
unmanifested_ids = sorted(live_ids - expected_ids) if expected_ids else []
|
||||
|
||||
for iid in missing_ids:
|
||||
_finding(
|
||||
CLASS_MISSING,
|
||||
severity="blocker",
|
||||
instance_ids=[iid],
|
||||
detail=f"expected enrolled instance {iid!r} is missing from the live fleet",
|
||||
active_blocker=True,
|
||||
)
|
||||
for iid in unmanifested_ids:
|
||||
_finding(
|
||||
CLASS_UNMANIFESTED,
|
||||
severity="blocker",
|
||||
instance_ids=[iid],
|
||||
workers=by_instance.get(iid, []),
|
||||
detail=(
|
||||
f"live instance {iid!r} is not on the approved fleet manifest "
|
||||
"(unmanifested)"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Same client_type multi-instance is healthy when each has distinct instance IDs
|
||||
by_type: dict[str, list[str]] = defaultdict(list)
|
||||
for inst in instances:
|
||||
for ct in inst.get("client_types") or []:
|
||||
by_type[str(ct)].append(inst["client_instance_id"])
|
||||
multi_instance_same_type = {
|
||||
ct: ids for ct, ids in by_type.items() if len(ids) > 1
|
||||
}
|
||||
|
||||
active_blockers = [f for f in findings if f.get("active_blocker")]
|
||||
historical_only = [f for f in findings if f.get("classification") == CLASS_HISTORICAL]
|
||||
live_safe = not active_blockers
|
||||
|
||||
consistency = registry_revision or _consistency_token(all_rows, snapshot_at)
|
||||
|
||||
return {
|
||||
"success": True,
|
||||
"read_only": True,
|
||||
"snapshot_at": snapshot_at,
|
||||
"consistency_token": consistency,
|
||||
"registry_revision": consistency,
|
||||
"live_worker_count": len(live_rows),
|
||||
"historical_worker_count": len(historical_rows),
|
||||
"stale_worker_count": len(stale_rows),
|
||||
"instance_count": len(instances),
|
||||
"workers": all_rows,
|
||||
"live_workers": live_rows,
|
||||
"historical_workers": historical_rows,
|
||||
"stale_workers": stale_rows,
|
||||
"instances": instances,
|
||||
"multi_instance_same_client_type": multi_instance_same_type,
|
||||
"same_client_type_not_duplicate": True,
|
||||
"expected_manifest": [
|
||||
{
|
||||
"client_instance_id": item.get("client_instance_id"),
|
||||
"client_type": item.get("client_type"),
|
||||
"fleet_run_id": item.get("fleet_run_id"),
|
||||
"namespaces": item.get("namespaces"),
|
||||
}
|
||||
for item in expected
|
||||
],
|
||||
"missing_expected_instance_ids": missing_ids,
|
||||
"unmanifested_instance_ids": unmanifested_ids,
|
||||
"findings": findings,
|
||||
"active_blockers": active_blockers,
|
||||
"historical_findings": historical_only,
|
||||
"live_fleet_safe": live_safe,
|
||||
"mutation_safe": live_safe and all(
|
||||
inst.get("mutation_safe") for inst in instances
|
||||
)
|
||||
if instances
|
||||
else live_safe,
|
||||
"classification_model": {
|
||||
"exactly_one_process_per_profile": False,
|
||||
"exactly_one_instance_per_client_type": False,
|
||||
"multiple_instances_per_client_type": True,
|
||||
"duplicate_requires": [
|
||||
"live client_instance_id collision",
|
||||
"duplicate namespace worker within one instance",
|
||||
"reused worker/session/generation/process/pid/fencing identity",
|
||||
"cross-instance ownership collision",
|
||||
"unmanifested instance when a manifest is required",
|
||||
],
|
||||
},
|
||||
"reasons": [f["detail"] for f in active_blockers],
|
||||
}
|
||||
|
||||
|
||||
def compare_snapshot_heartbeats(
|
||||
earlier: Mapping[str, Any],
|
||||
later: Mapping[str, Any],
|
||||
) -> dict[str, Any]:
|
||||
"""Prove heartbeat continuity and stable ownership across two snapshots."""
|
||||
earlier_live = {
|
||||
w.get("worker_identity"): w for w in earlier.get("live_workers") or []
|
||||
}
|
||||
later_live = {
|
||||
w.get("worker_identity"): w for w in later.get("live_workers") or []
|
||||
}
|
||||
shared = sorted(set(earlier_live) & set(later_live))
|
||||
continuity: list[dict[str, Any]] = []
|
||||
stable_ownership = True
|
||||
for wid in shared:
|
||||
a = earlier_live[wid]
|
||||
b = later_live[wid]
|
||||
same_instance = a.get("client_instance_id") == b.get("client_instance_id")
|
||||
same_session = a.get("session_id") == b.get("session_id")
|
||||
same_generation = a.get("generation_id") == b.get("generation_id")
|
||||
hb_advanced_or_equal = True
|
||||
if a.get("last_heartbeat_at") and b.get("last_heartbeat_at"):
|
||||
hb_advanced_or_equal = b["last_heartbeat_at"] >= a["last_heartbeat_at"]
|
||||
if not (same_instance and same_session and same_generation):
|
||||
stable_ownership = False
|
||||
continuity.append(
|
||||
{
|
||||
"worker_identity": wid,
|
||||
"same_client_instance_id": same_instance,
|
||||
"same_session_id": same_session,
|
||||
"same_generation_id": same_generation,
|
||||
"heartbeat_non_decreasing": hb_advanced_or_equal,
|
||||
"earlier_heartbeat": a.get("last_heartbeat_at"),
|
||||
"later_heartbeat": b.get("last_heartbeat_at"),
|
||||
}
|
||||
)
|
||||
return {
|
||||
"shared_live_workers": shared,
|
||||
"continuity": continuity,
|
||||
"stable_ownership": stable_ownership
|
||||
and all(c["heartbeat_non_decreasing"] for c in continuity),
|
||||
"dropped_workers": sorted(set(earlier_live) - set(later_live)),
|
||||
"new_workers": sorted(set(later_live) - set(earlier_live)),
|
||||
}
|
||||
@@ -0,0 +1,205 @@
|
||||
"""Host, boot, and process-start fencing evidence for retirement safety (#980).
|
||||
|
||||
Review 657 B2/B3 established that a local ``os.kill(pid, 0)`` probe is not, on
|
||||
its own, evidence that a *particular registered worker* is gone:
|
||||
|
||||
* **Host ambiguity.** A registry row written on host A records pid 1234. Probing
|
||||
pid 1234 on host B answers a question nobody asked. "Not running here" is not
|
||||
"not running".
|
||||
* **PID reuse.** Pid 1234 may be alive and belong to an unrelated process that
|
||||
the kernel handed the number to after the original exited. The naive probe
|
||||
reads that as "the worker is live" (safe, over-preserving) — but the converse
|
||||
matters too: evidence captured about pid 1234 at time T must not be honoured
|
||||
at time T+n if the process behind that number changed in between.
|
||||
* **Boot boundaries.** Every pid from a previous boot is conclusively gone, and
|
||||
pid numbers restart, so a recorded pid is only comparable to a live pid when
|
||||
both belong to the same boot.
|
||||
|
||||
This module supplies the three pieces of evidence that turn a bare pid into a
|
||||
statement about one specific process:
|
||||
|
||||
``host_id``
|
||||
Which machine the pid belongs to.
|
||||
``boot_id``
|
||||
Which boot of that machine the pid belongs to. Pids are only comparable
|
||||
within a single boot.
|
||||
``process_start_time``
|
||||
Which *incarnation* of that pid number. Two processes on the same host and
|
||||
boot sharing a pid number cannot share a start time, so comparing start
|
||||
times defeats reuse.
|
||||
|
||||
Every probe here is read-only, never raises, and returns ``None`` when the
|
||||
evidence cannot be established. ``None`` means *unknown*, and the retirement
|
||||
conjunction is required to treat unknown as "preserve", never as "safe".
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import platform
|
||||
import subprocess
|
||||
|
||||
# --- Host identity --------------------------------------------------------
|
||||
|
||||
|
||||
def current_host_id() -> str | None:
|
||||
"""Stable identifier for the machine this process runs on.
|
||||
|
||||
Deliberately the kernel node name rather than anything network-derived: it
|
||||
does not change when an interface goes down or a VPN reassigns an address,
|
||||
and retirement must not become unsafe because DNS moved.
|
||||
"""
|
||||
try:
|
||||
node = (platform.node() or "").strip()
|
||||
except Exception:
|
||||
return None
|
||||
return node or None
|
||||
|
||||
|
||||
# --- Boot identity --------------------------------------------------------
|
||||
|
||||
|
||||
def _linux_boot_id() -> str | None:
|
||||
try:
|
||||
with open("/proc/sys/kernel/random/boot_id", encoding="utf-8") as handle:
|
||||
value = handle.read().strip()
|
||||
except Exception:
|
||||
return None
|
||||
return value or None
|
||||
|
||||
|
||||
def _darwin_boot_id() -> str | None:
|
||||
"""macOS boot identity, derived from ``kern.boottime``.
|
||||
|
||||
``sysctl`` prints e.g. ``{ sec = 1785400000, usec = 123456 } Wed Jul 30 ...``.
|
||||
Only the integer seconds are kept: the trailing human-readable date is
|
||||
locale-dependent and would make the identifier unstable across environments
|
||||
for the very same boot.
|
||||
"""
|
||||
try:
|
||||
completed = subprocess.run(
|
||||
["/usr/sbin/sysctl", "-n", "kern.boottime"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=5,
|
||||
check=False,
|
||||
)
|
||||
except Exception:
|
||||
return None
|
||||
if completed.returncode != 0:
|
||||
return None
|
||||
text = (completed.stdout or "").strip()
|
||||
marker = "sec = "
|
||||
start = text.find(marker)
|
||||
if start < 0:
|
||||
return None
|
||||
tail = text[start + len(marker) :]
|
||||
digits = ""
|
||||
for char in tail:
|
||||
if char.isdigit():
|
||||
digits += char
|
||||
else:
|
||||
break
|
||||
return f"boot-{digits}" if digits else None
|
||||
|
||||
|
||||
def current_boot_id() -> str | None:
|
||||
"""Identifier for the current boot, or ``None`` when it cannot be proven."""
|
||||
system = ""
|
||||
try:
|
||||
system = (platform.system() or "").strip().lower()
|
||||
except Exception:
|
||||
system = ""
|
||||
if system == "linux":
|
||||
return _linux_boot_id()
|
||||
if system == "darwin":
|
||||
return _darwin_boot_id()
|
||||
# An unrecognised platform yields no boot evidence rather than a guess.
|
||||
return None
|
||||
|
||||
|
||||
# --- Process start time ---------------------------------------------------
|
||||
|
||||
|
||||
def _linux_process_start_time(pid: int) -> str | None:
|
||||
"""Field 22 of ``/proc/<pid>/stat`` — start time in clock ticks since boot.
|
||||
|
||||
The executable name in field 2 is parenthesised and may itself contain
|
||||
spaces and parentheses, so the fields are located from the *last* ``)``
|
||||
rather than by splitting the whole line.
|
||||
"""
|
||||
try:
|
||||
with open(f"/proc/{pid}/stat", encoding="utf-8") as handle:
|
||||
raw = handle.read()
|
||||
except Exception:
|
||||
return None
|
||||
close = raw.rfind(")")
|
||||
if close < 0:
|
||||
return None
|
||||
fields = raw[close + 1 :].split()
|
||||
# After the ')' the next field is state (field 3), so field 22 is index 19.
|
||||
if len(fields) < 20:
|
||||
return None
|
||||
value = fields[19].strip()
|
||||
return f"ticks-{value}" if value else None
|
||||
|
||||
|
||||
def _darwin_process_start_time(pid: int) -> str | None:
|
||||
"""macOS process start, from ``ps -o lstart=``.
|
||||
|
||||
``lstart`` is the absolute wall-clock start of that pid's current
|
||||
incarnation. Two processes reusing one pid number report different values,
|
||||
which is exactly the discrimination reuse detection needs.
|
||||
"""
|
||||
try:
|
||||
completed = subprocess.run(
|
||||
["/bin/ps", "-o", "lstart=", "-p", str(int(pid))],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=5,
|
||||
check=False,
|
||||
)
|
||||
except Exception:
|
||||
return None
|
||||
if completed.returncode != 0:
|
||||
return None
|
||||
value = " ".join((completed.stdout or "").split())
|
||||
return f"lstart-{value}" if value else None
|
||||
|
||||
|
||||
def process_start_time(pid: int | None) -> str | None:
|
||||
"""Start-time token for *pid*'s current incarnation, or ``None``.
|
||||
|
||||
``None`` is returned both when the pid does not exist and when the platform
|
||||
cannot answer. Callers must not read either case as evidence of death — a
|
||||
dead pid is established by the liveness probe, and this value only ever
|
||||
*withdraws* a retirement that pid-level evidence would otherwise allow.
|
||||
"""
|
||||
if pid is None:
|
||||
return None
|
||||
try:
|
||||
numeric = int(pid)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
if numeric <= 0:
|
||||
return None
|
||||
system = ""
|
||||
try:
|
||||
system = (platform.system() or "").strip().lower()
|
||||
except Exception:
|
||||
system = ""
|
||||
if system == "linux":
|
||||
return _linux_process_start_time(numeric)
|
||||
if system == "darwin":
|
||||
return _darwin_process_start_time(numeric)
|
||||
return None
|
||||
|
||||
|
||||
def current_process_fencing(pid: int | None = None) -> dict[str, str | None]:
|
||||
"""The full fencing triple for *pid* (defaults to this process)."""
|
||||
target = os.getpid() if pid is None else pid
|
||||
return {
|
||||
"host_id": current_host_id(),
|
||||
"boot_id": current_boot_id(),
|
||||
"process_start_time": process_start_time(target),
|
||||
}
|
||||
+909
-3
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+153
-27
@@ -8,8 +8,10 @@ poison workspace purity checks in another namespace.
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
import author_mutation_worktree as amw
|
||||
import canonical_repository_root as crr
|
||||
|
||||
ACTIVE_WORKTREE_ENV = amw.ACTIVE_WORKTREE_ENV
|
||||
AUTHOR_WORKTREE_ENV = amw.AUTHOR_WORKTREE_ENV
|
||||
@@ -152,6 +154,60 @@ def resolve_namespace_workspace(
|
||||
return os.path.realpath(process_project_root), "MCP server process root (default)"
|
||||
|
||||
|
||||
def verify_git_common_directory_membership(
|
||||
workspace_path: str,
|
||||
canonical_repo_root: str,
|
||||
) -> tuple[bool, str | None]:
|
||||
"""Verify that workspace_path belongs to canonical_repo_root via git common-dir or branches containment."""
|
||||
ws = (workspace_path or "").strip()
|
||||
root = (canonical_repo_root or "").strip()
|
||||
if not ws or not root:
|
||||
return False, "empty workspace or canonical root path"
|
||||
|
||||
try:
|
||||
real_ws = os.path.realpath(os.path.abspath(ws))
|
||||
real_root = os.path.realpath(os.path.abspath(root))
|
||||
except Exception as exc:
|
||||
return False, f"invalid workspace or root path: {exc}"
|
||||
|
||||
if real_ws == real_root:
|
||||
return True, None
|
||||
|
||||
if not os.path.isdir(real_ws):
|
||||
return False, f"workspace directory '{real_ws}' does not exist"
|
||||
|
||||
try:
|
||||
res = subprocess.run(
|
||||
["git", "-C", real_ws, "rev-parse", "--git-common-dir"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
if res.returncode == 0:
|
||||
common_raw = (res.stdout or "").strip()
|
||||
common_dir = amw._realpath_git_common_dir(real_ws, common_raw)
|
||||
real_common = os.path.realpath(common_dir)
|
||||
canonical_git = os.path.realpath(os.path.join(real_root, ".git"))
|
||||
|
||||
if real_common in (canonical_git, real_root):
|
||||
return True, None
|
||||
return (
|
||||
False,
|
||||
f"workspace '{real_ws}' git common directory '{real_common}' does not match "
|
||||
f"canonical repository root '{real_root}' (.git at '{canonical_git}')"
|
||||
)
|
||||
else:
|
||||
return (
|
||||
False,
|
||||
f"workspace '{real_ws}' is not a valid git repository or git rev-parse failed"
|
||||
)
|
||||
except Exception as exc:
|
||||
return (
|
||||
False,
|
||||
f"failed to inspect git common directory for workspace '{real_ws}': {exc}"
|
||||
)
|
||||
|
||||
|
||||
def resolve_namespace_mutation_context(
|
||||
*,
|
||||
role_kind: str,
|
||||
@@ -163,6 +219,8 @@ def resolve_namespace_mutation_context(
|
||||
worktree: str | None = None,
|
||||
profile_name: str | None = None,
|
||||
configured_canonical_root: str | None = None,
|
||||
expected_slug: str | None = None,
|
||||
remote: str | None = None,
|
||||
) -> dict:
|
||||
"""Shared workspace resolution for runtime_context and mutation guards.
|
||||
|
||||
@@ -180,11 +238,41 @@ def resolve_namespace_mutation_context(
|
||||
env_map = env if env is not None else os.environ
|
||||
process_root = os.path.realpath(process_project_root)
|
||||
role = normalize_role_kind(role_kind, profile_name=profile_name)
|
||||
configured = (configured_canonical_root or "").strip()
|
||||
if configured:
|
||||
canonical_root = os.path.realpath(configured)
|
||||
|
||||
configured_val = (configured_canonical_root or "").strip()
|
||||
if configured_val:
|
||||
crr_assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=configured_val,
|
||||
source="configured_canonical_root",
|
||||
expected_slug=expected_slug,
|
||||
process_project_root=process_root,
|
||||
remote=remote,
|
||||
require_binding=True,
|
||||
)
|
||||
# #973 B10: a refused repository-authority mode deliberately resolves no
|
||||
# canonical root, so downstream guards keep evaluating the install
|
||||
# checkout rather than an identity derived through an undefined mode.
|
||||
# ``roots_aligned`` still follows ``proven`` and stays False, and the
|
||||
# refusal (with its reason_code) rides along in
|
||||
# ``canonical_root_assessment`` — this is a fail-closed fallback, never a
|
||||
# normalisation of the mode.
|
||||
canonical_root = crr_assessment["canonical_repo_root"] or amw.resolve_canonical_repo_root(
|
||||
process_root, process_root
|
||||
)
|
||||
roots_aligned = crr_assessment["proven"]
|
||||
else:
|
||||
canonical_root = amw.resolve_canonical_repo_root(process_root, process_root)
|
||||
crr_assessment = {
|
||||
"proven": True,
|
||||
"block": False,
|
||||
"reasons": [],
|
||||
"configured": False,
|
||||
"canonical_repo_root": amw.resolve_canonical_repo_root(process_root, process_root),
|
||||
"resolved_slug": None,
|
||||
"source": None,
|
||||
"reason_code": None,
|
||||
}
|
||||
canonical_root = crr_assessment["canonical_repo_root"]
|
||||
roots_aligned = (canonical_root == process_root)
|
||||
|
||||
durable: dict | None = None
|
||||
if role == "author":
|
||||
@@ -234,7 +322,10 @@ def resolve_namespace_mutation_context(
|
||||
"ignored_bindings": demotions + (pollution.get("ignored_bindings") or []),
|
||||
"process_project_root": process_root,
|
||||
"canonical_repo_root": canonical_root,
|
||||
"roots_aligned": canonical_root == process_root,
|
||||
"roots_aligned": roots_aligned,
|
||||
"canonical_root_assessment": crr_assessment,
|
||||
"expected_slug": expected_slug,
|
||||
"remote": remote,
|
||||
}
|
||||
if durable is not None:
|
||||
result["author_worktree_resolution"] = durable
|
||||
@@ -246,6 +337,15 @@ def resolve_namespace_mutation_context(
|
||||
result["author_worktree_reasons"] = list(durable.get("reasons") or [])
|
||||
result["author_worktree_blocker_kind"] = durable.get("blocker_kind")
|
||||
result["operator_recovery"] = durable.get("operator_recovery")
|
||||
else:
|
||||
path_exists = os.path.exists(workspace)
|
||||
result["path_exists"] = path_exists
|
||||
result["in_git_worktree_list"] = (
|
||||
amw.path_in_git_worktree_list(workspace, canonical_root)
|
||||
if path_exists
|
||||
else False
|
||||
)
|
||||
result["bound_worktree_missing"] = not path_exists
|
||||
return result
|
||||
|
||||
|
||||
@@ -378,8 +478,8 @@ def format_namespace_workspace_binding_error(
|
||||
def assess_namespace_mutation_workspace(
|
||||
*,
|
||||
role_kind: str,
|
||||
worktree_path: str | None,
|
||||
worktree: str | None,
|
||||
worktree_path: str | None = None,
|
||||
worktree: str | None = None,
|
||||
process_project_root: str,
|
||||
env: dict[str, str] | os._Environ | None = None,
|
||||
session_lease_worktree: str | None = None,
|
||||
@@ -387,6 +487,8 @@ def assess_namespace_mutation_workspace(
|
||||
profile_name: str | None = None,
|
||||
current_branch: str | None = None,
|
||||
configured_canonical_root: str | None = None,
|
||||
expected_slug: str | None = None,
|
||||
remote: str | None = None,
|
||||
) -> dict:
|
||||
"""Evaluate namespace workspace binding before preflight/mutation."""
|
||||
ctx = resolve_namespace_mutation_context(
|
||||
@@ -399,6 +501,8 @@ def assess_namespace_mutation_workspace(
|
||||
session_lock_worktree=session_lock_worktree,
|
||||
profile_name=profile_name,
|
||||
configured_canonical_root=configured_canonical_root,
|
||||
expected_slug=expected_slug,
|
||||
remote=remote,
|
||||
)
|
||||
mutation_workspace = ctx["workspace_path"]
|
||||
binding_source = ctx["workspace_binding_source"]
|
||||
@@ -422,6 +526,27 @@ def assess_namespace_mutation_workspace(
|
||||
|
||||
reasons = list(metadata.get("reasons") or [])
|
||||
operator_recovery = ctx.get("operator_recovery")
|
||||
|
||||
crr_reasons = list(ctx.get("canonical_root_assessment", {}).get("reasons") or [])
|
||||
if crr_reasons:
|
||||
reasons.extend(crr_reasons)
|
||||
|
||||
path_exists = ctx.get("path_exists")
|
||||
if path_exists is None:
|
||||
path_exists = os.path.exists(mutation_workspace)
|
||||
|
||||
if not path_exists:
|
||||
if role != "author":
|
||||
reasons.append(
|
||||
f"{role} mutation blocked: configured workspace directory '{mutation_workspace}' does not exist (nonexistent worktree)"
|
||||
)
|
||||
else:
|
||||
valid_common, common_err = verify_git_common_directory_membership(
|
||||
mutation_workspace, ctx["canonical_repo_root"]
|
||||
)
|
||||
if not valid_common and common_err:
|
||||
reasons.append(common_err)
|
||||
|
||||
if role == "author":
|
||||
# #618 durable resolution already validated existence, membership,
|
||||
# branches/, lock ownership, and traversal safety when present.
|
||||
@@ -438,26 +563,27 @@ def assess_namespace_mutation_workspace(
|
||||
)
|
||||
if branches["block"]:
|
||||
reasons.extend(branches["reasons"])
|
||||
elif (
|
||||
role == "reviewer"
|
||||
and mutation_workspace == process_root
|
||||
and not amw.is_path_under_branches(mutation_workspace, ctx["canonical_repo_root"])
|
||||
):
|
||||
reasons.append(
|
||||
f"{role} mutation blocked: workspace is the stable control checkout; "
|
||||
f"create or reconnect to a session-owned worktree under branches/ "
|
||||
f"or set {ROLE_WORKTREE_ENVS.get(role, ACTIVE_WORKTREE_ENV)} / "
|
||||
f"{ACTIVE_WORKTREE_ENV}"
|
||||
)
|
||||
elif (
|
||||
role in {"reviewer", "merger"}
|
||||
and mutation_workspace != process_root
|
||||
and not amw.is_path_under_branches(mutation_workspace, ctx["canonical_repo_root"])
|
||||
):
|
||||
reasons.append(
|
||||
f"{role} mutation blocked: workspace '{mutation_workspace}' is not under "
|
||||
f"'{ctx['canonical_repo_root']}/branches/'"
|
||||
)
|
||||
elif role in {"reviewer", "merger"}:
|
||||
if mutation_workspace == process_root and not amw.is_path_under_branches(mutation_workspace, ctx["canonical_repo_root"]):
|
||||
reasons.append(
|
||||
f"{role} mutation blocked: workspace is the stable control checkout; "
|
||||
f"create or reconnect to a session-owned worktree under branches/ "
|
||||
f"or set {ROLE_WORKTREE_ENVS.get(role, ACTIVE_WORKTREE_ENV)} / "
|
||||
f"{ACTIVE_WORKTREE_ENV}"
|
||||
)
|
||||
elif mutation_workspace != process_root and not amw.is_path_under_branches(mutation_workspace, ctx["canonical_repo_root"]):
|
||||
reasons.append(
|
||||
f"{role} mutation blocked: workspace '{mutation_workspace}' is not under "
|
||||
f"'{ctx['canonical_repo_root']}/branches/'"
|
||||
)
|
||||
|
||||
if path_exists and amw.is_path_under_branches(mutation_workspace, ctx["canonical_repo_root"]):
|
||||
in_list = ctx.get("in_git_worktree_list")
|
||||
if in_list is False:
|
||||
reasons.append(
|
||||
f"{role} mutation blocked: workspace '{mutation_workspace}' is under branches/ "
|
||||
f"but is not registered in git worktree list for '{ctx['canonical_repo_root']}'"
|
||||
)
|
||||
|
||||
block = bool(reasons)
|
||||
return {
|
||||
|
||||
+12
-57
@@ -475,65 +475,24 @@ def reconcile_after_restart(
|
||||
)
|
||||
)
|
||||
|
||||
# --- sessions (#969: dead-owner / PID-reuse lifecycle) ---------------
|
||||
# Prefer a precomputed fleet report from the gather/apply path when present;
|
||||
# otherwise classify pure from inventory (injectable checkers stay default).
|
||||
# --- sessions -------------------------------------------------------
|
||||
sessions = [s for s in (inventory.get("sessions") or []) if isinstance(s, Mapping)]
|
||||
leases_for_sessions = [
|
||||
L for L in (inventory.get("leases") or []) if isinstance(L, Mapping)
|
||||
orphan_sessions = [
|
||||
s
|
||||
for s in sessions
|
||||
if str(s.get("status") or "").lower() == "active"
|
||||
and s.get("pid") is not None
|
||||
and not lease_lifecycle.is_process_alive(s.get("pid"))
|
||||
]
|
||||
fleet_report = inventory.get("session_fleet")
|
||||
if isinstance(fleet_report, Mapping) and "retireable_session_ids" in fleet_report:
|
||||
fleet_details = dict(fleet_report)
|
||||
retireable_ids = list(fleet_details.get("retireable_session_ids") or [])
|
||||
resolved = bool(fleet_details.get("sessions_dimension_resolved", not retireable_ids))
|
||||
else:
|
||||
try:
|
||||
import session_lifecycle as _sl
|
||||
|
||||
client_managed = inventory.get("client_managed_session_ids")
|
||||
cm_set = None
|
||||
if isinstance(client_managed, (list, tuple, set, frozenset)):
|
||||
cm_set = {str(x) for x in client_managed}
|
||||
fleet = _sl.classify_sessions(
|
||||
sessions,
|
||||
leases=leases_for_sessions,
|
||||
now=started,
|
||||
client_managed_sessions=cm_set,
|
||||
)
|
||||
fleet_details = fleet.as_dict()
|
||||
retireable_ids = list(fleet_details.get("retireable_session_ids") or [])
|
||||
resolved = bool(fleet_details.get("sessions_dimension_resolved"))
|
||||
except Exception as exc: # noqa: BLE001 — fail closed to legacy signal
|
||||
# Legacy fallback: dead-pid active rows only (pre-#969 behaviour).
|
||||
orphan_sessions = [
|
||||
s
|
||||
for s in sessions
|
||||
if str(s.get("status") or "").lower() == "active"
|
||||
and s.get("pid") is not None
|
||||
and not lease_lifecycle.is_process_alive(s.get("pid"))
|
||||
]
|
||||
retireable_ids = [s.get("session_id") for s in orphan_sessions]
|
||||
resolved = not orphan_sessions
|
||||
fleet_details = {
|
||||
"total_sessions": len(sessions),
|
||||
"retireable_session_ids": retireable_ids,
|
||||
"legacy_fallback": True,
|
||||
"fallback_error": str(exc),
|
||||
}
|
||||
|
||||
if not resolved and retireable_ids:
|
||||
if orphan_sessions:
|
||||
items.append(
|
||||
_item(
|
||||
DIM_SESSIONS,
|
||||
ITEM_UNRESOLVED,
|
||||
f"{len(retireable_ids)} session row(s) with dead/reused owner "
|
||||
f"await retirement",
|
||||
f"{len(orphan_sessions)} active session row(s) with dead owner pid",
|
||||
details={
|
||||
"orphan_session_ids": retireable_ids,
|
||||
"retireable_session_ids": retireable_ids,
|
||||
"orphan_session_ids": [s.get("session_id") for s in orphan_sessions],
|
||||
"total_sessions": len(sessions),
|
||||
"fleet": fleet_details,
|
||||
},
|
||||
follow_up=True,
|
||||
)
|
||||
@@ -543,12 +502,8 @@ def reconcile_after_restart(
|
||||
_item(
|
||||
DIM_SESSIONS,
|
||||
ITEM_RESOLVED,
|
||||
f"{len(sessions)} session row(s) reconciled "
|
||||
f"(no retireable dead/reused owners)",
|
||||
details={
|
||||
"total_sessions": len(sessions),
|
||||
"fleet": fleet_details,
|
||||
},
|
||||
f"{len(sessions)} session row(s) reconciled (no dead-pid orphans)",
|
||||
details={"total_sessions": len(sessions)},
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
@@ -1,703 +0,0 @@
|
||||
"""Safe lifecycle for workflow session rows with dead or reused owners (#969).
|
||||
|
||||
Post-restart reconciliation previously left hundreds of ``active`` session rows
|
||||
with dead owner PIDs permanently unresolved. A PID existence check alone is not
|
||||
enough: operating systems reuse PIDs, so an unrelated live process can appear to
|
||||
own a historical session.
|
||||
|
||||
This module is the pure classification + apply core for session retirement:
|
||||
|
||||
* distinguish live, disconnected, stale, protected, and terminal records
|
||||
* refuse retirement when a live lease or live verified owner remains
|
||||
* protect live client-managed sessions
|
||||
* detect PID reuse via process start time vs session start / heartbeat
|
||||
* terminalize confirmed-stale rows idempotently with durable audit events
|
||||
* stay safe under concurrent reconciles (CAS on status)
|
||||
|
||||
Design mirrors ``lease_lifecycle`` / ``post_restart_reconcile``:
|
||||
|
||||
* Pure classification accepts injectable checkers so unit tests never touch
|
||||
real processes.
|
||||
* Apply mutations go only through :meth:`ControlPlaneDB.retire_session`.
|
||||
* Historical rows are never deleted; status moves to a terminal value and an
|
||||
events-row records the action + reason.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
from dataclasses import dataclass, field
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Any, Callable, Mapping, Sequence
|
||||
|
||||
import control_plane_db as cpd
|
||||
import lease_lifecycle
|
||||
|
||||
# Session status vocabulary.
|
||||
SESSION_STATUS_ACTIVE = "active"
|
||||
SESSION_STATUS_RETIRED = "retired"
|
||||
SESSION_STATUS_ENDED = "ended"
|
||||
SESSION_STATUS_TERMINAL = "terminal"
|
||||
|
||||
TERMINAL_SESSION_STATUSES = frozenset(
|
||||
{
|
||||
SESSION_STATUS_RETIRED,
|
||||
SESSION_STATUS_ENDED,
|
||||
SESSION_STATUS_TERMINAL,
|
||||
"dead",
|
||||
"stale",
|
||||
"orphaned",
|
||||
}
|
||||
)
|
||||
|
||||
# Classification outcomes for one session row.
|
||||
CLASS_LIVE = "live"
|
||||
CLASS_DISCONNECTED = "disconnected"
|
||||
CLASS_STALE = "stale"
|
||||
CLASS_PROTECTED = "protected"
|
||||
CLASS_TERMINAL = "terminal"
|
||||
|
||||
# Stable reason codes (audit + tests).
|
||||
REASON_ALREADY_TERMINAL = "already_terminal"
|
||||
REASON_LIVE_OWNER = "live_owner"
|
||||
REASON_LIVE_LEASE = "live_lease"
|
||||
REASON_CLIENT_MANAGED_LIVE = "client_managed_live"
|
||||
REASON_DEAD_OWNER = "dead_owner"
|
||||
REASON_PID_REUSE = "pid_reuse"
|
||||
REASON_HEARTBEAT_STALE_DEAD = "heartbeat_stale_dead_owner"
|
||||
REASON_MISSING_PID = "missing_pid_no_lease"
|
||||
|
||||
# Event type written to control-plane events table.
|
||||
EVENT_SESSION_RETIRED = "session_retired"
|
||||
|
||||
# Default heartbeat window before a still-alive PID is treated as disconnected
|
||||
# rather than live (does not alone authorize retirement).
|
||||
DEFAULT_HEARTBEAT_STALE_SECONDS = 900
|
||||
|
||||
# PID reuse: process start must be strictly later than session started_at by
|
||||
# more than this skew (ps lstart is second-resolution; clocks can lag).
|
||||
PID_REUSE_SKEW = timedelta(seconds=2)
|
||||
|
||||
# Live lease freshness values that block retirement.
|
||||
_LIVE_LEASE_FRESHNESS = frozenset({"active", "live"})
|
||||
|
||||
|
||||
class SessionLifecycleError(RuntimeError):
|
||||
"""Fail-closed session lifecycle policy error."""
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def _parse_ts(value: str | datetime | None) -> datetime | None:
|
||||
if value is None:
|
||||
return None
|
||||
if isinstance(value, datetime):
|
||||
if value.tzinfo is None:
|
||||
return value.replace(tzinfo=timezone.utc)
|
||||
return value.astimezone(timezone.utc)
|
||||
return cpd._parse_ts(str(value))
|
||||
|
||||
|
||||
def _ts(dt: datetime | None = None) -> str:
|
||||
return cpd._ts(dt)
|
||||
|
||||
|
||||
def process_start_time(pid: int | None) -> datetime | None:
|
||||
"""Return the OS start time of *pid*, or None when it cannot be resolved.
|
||||
|
||||
Uses ``ps -o lstart=`` (POSIX). Failures return None rather than inventing
|
||||
evidence — missing start time never authorizes retirement of a live PID.
|
||||
"""
|
||||
if pid is None:
|
||||
return None
|
||||
try:
|
||||
pid_i = int(pid)
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
if pid_i <= 0:
|
||||
return None
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["ps", "-o", "lstart=", "-p", str(pid_i)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
timeout=2,
|
||||
)
|
||||
except (OSError, subprocess.SubprocessError):
|
||||
return None
|
||||
if proc.returncode != 0:
|
||||
return None
|
||||
text = (proc.stdout or "").strip()
|
||||
if not text:
|
||||
return None
|
||||
try:
|
||||
# Example: "Wed Jul 29 09:14:36 2026"
|
||||
naive = datetime.strptime(text, "%a %b %d %H:%M:%S %Y")
|
||||
return naive.replace(tzinfo=timezone.utc)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def _lease_freshness_label(lease: Mapping[str, Any]) -> str:
|
||||
fr = lease.get("freshness")
|
||||
if isinstance(fr, Mapping):
|
||||
return str(fr.get("freshness") or "").strip().lower()
|
||||
if fr:
|
||||
return str(fr).strip().lower()
|
||||
# Fall back to classifying a raw lease row.
|
||||
try:
|
||||
return str(
|
||||
lease_lifecycle.classify_lease_freshness(lease).get("freshness") or ""
|
||||
).strip().lower()
|
||||
except Exception: # noqa: BLE001 — pure classifier must not raise on bad rows
|
||||
status = str(lease.get("status") or "").strip().lower()
|
||||
return status or "unknown"
|
||||
|
||||
|
||||
def live_lease_session_ids(
|
||||
leases: Sequence[Mapping[str, Any]] | None,
|
||||
*,
|
||||
now: datetime | None = None,
|
||||
pid_checker: Callable[[int | None], bool] = lease_lifecycle.is_process_alive,
|
||||
) -> set[str]:
|
||||
"""Session ids that still hold a live (or ambiguous-active) workflow lease."""
|
||||
live: set[str] = set()
|
||||
moment = now or _utc_now()
|
||||
for lease in leases or ():
|
||||
if not isinstance(lease, Mapping):
|
||||
continue
|
||||
status = str(lease.get("status") or "").strip().lower()
|
||||
if status and status not in {
|
||||
lease_lifecycle.LEASE_STATUS_ACTIVE,
|
||||
"",
|
||||
}:
|
||||
# Explicit terminal lease statuses never protect a session.
|
||||
if status in {
|
||||
lease_lifecycle.LEASE_STATUS_RELEASED,
|
||||
lease_lifecycle.LEASE_STATUS_EXPIRED,
|
||||
lease_lifecycle.LEASE_STATUS_ABANDONED,
|
||||
}:
|
||||
continue
|
||||
freshness = _lease_freshness_label(lease)
|
||||
if freshness in _LIVE_LEASE_FRESHNESS or freshness in {"", "unknown"}:
|
||||
# Ambiguous active rows: re-check with authoritative classifier.
|
||||
try:
|
||||
fr = lease_lifecycle.classify_lease_freshness(
|
||||
lease, now=moment, pid_checker=pid_checker
|
||||
)
|
||||
freshness = str(fr.get("freshness") or "").strip().lower()
|
||||
except Exception: # noqa: BLE001
|
||||
freshness = "unknown"
|
||||
if freshness in _LIVE_LEASE_FRESHNESS:
|
||||
sid = str(lease.get("session_id") or "").strip()
|
||||
if sid:
|
||||
live.add(sid)
|
||||
elif freshness == "unknown" and status in {
|
||||
lease_lifecycle.LEASE_STATUS_ACTIVE,
|
||||
"",
|
||||
}:
|
||||
# Fail closed: active lease with unknown freshness blocks retirement.
|
||||
sid = str(lease.get("session_id") or "").strip()
|
||||
if sid:
|
||||
live.add(sid)
|
||||
return live
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SessionClassification:
|
||||
"""Classification of one workflow session row."""
|
||||
|
||||
session_id: str
|
||||
classification: str
|
||||
reason: str
|
||||
retireable: bool
|
||||
status: str | None
|
||||
pid: int | None
|
||||
pid_alive: bool | None
|
||||
pid_reused: bool
|
||||
heartbeat_stale: bool
|
||||
has_live_lease: bool
|
||||
client_managed: bool
|
||||
details: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"session_id": self.session_id,
|
||||
"classification": self.classification,
|
||||
"reason": self.reason,
|
||||
"retireable": self.retireable,
|
||||
"status": self.status,
|
||||
"pid": self.pid,
|
||||
"pid_alive": self.pid_alive,
|
||||
"pid_reused": self.pid_reused,
|
||||
"heartbeat_stale": self.heartbeat_stale,
|
||||
"has_live_lease": self.has_live_lease,
|
||||
"client_managed": self.client_managed,
|
||||
"details": dict(self.details),
|
||||
}
|
||||
|
||||
|
||||
def classify_session(
|
||||
row: Mapping[str, Any],
|
||||
*,
|
||||
now: datetime | None = None,
|
||||
pid_checker: Callable[[int | None], bool] = lease_lifecycle.is_process_alive,
|
||||
process_start_probe: Callable[[int | None], datetime | None] = process_start_time,
|
||||
live_lease_sessions: set[str] | frozenset[str] | None = None,
|
||||
client_managed_sessions: set[str] | frozenset[str] | None = None,
|
||||
heartbeat_stale_seconds: int = DEFAULT_HEARTBEAT_STALE_SECONDS,
|
||||
) -> SessionClassification:
|
||||
"""Classify one session row for retirement decisions (#969).
|
||||
|
||||
Rules (first match wins where noted):
|
||||
|
||||
1. Non-active / already-terminal status → ``terminal`` (not retireable).
|
||||
2. Session holds a live lease → ``protected`` (never retire).
|
||||
3. PID missing and no live lease → ``stale`` (retireable: missing owner).
|
||||
4. PID alive + process start after session start → ``stale`` (PID reuse).
|
||||
5. PID alive + client-managed → ``live`` protected (never retire).
|
||||
6. PID alive + fresh heartbeat → ``live``.
|
||||
7. PID alive + stale heartbeat → ``disconnected`` (not retireable alone).
|
||||
8. PID dead → ``stale`` (retireable).
|
||||
"""
|
||||
moment = now or _utc_now()
|
||||
session_id = str(row.get("session_id") or "").strip()
|
||||
status = str(row.get("status") or "").strip().lower() or None
|
||||
raw_pid = row.get("pid")
|
||||
try:
|
||||
pid = int(raw_pid) if raw_pid is not None else None
|
||||
except (TypeError, ValueError):
|
||||
pid = None
|
||||
|
||||
client_managed = bool(
|
||||
row.get("client_managed")
|
||||
or row.get("is_client_managed")
|
||||
or (
|
||||
client_managed_sessions is not None
|
||||
and session_id in client_managed_sessions
|
||||
)
|
||||
)
|
||||
has_live_lease = bool(
|
||||
live_lease_sessions is not None and session_id in live_lease_sessions
|
||||
)
|
||||
|
||||
hb = _parse_ts(row.get("last_heartbeat_at"))
|
||||
heartbeat_stale = bool(
|
||||
hb is not None
|
||||
and (moment - hb).total_seconds() > max(0, int(heartbeat_stale_seconds))
|
||||
)
|
||||
if hb is None and status == SESSION_STATUS_ACTIVE:
|
||||
# No heartbeat evidence: treat as stale for liveness bookkeeping only.
|
||||
heartbeat_stale = True
|
||||
|
||||
started = _parse_ts(row.get("started_at"))
|
||||
recorded_proc_start = _parse_ts(row.get("owner_process_started_at"))
|
||||
|
||||
if status in TERMINAL_SESSION_STATUSES:
|
||||
return SessionClassification(
|
||||
session_id=session_id,
|
||||
classification=CLASS_TERMINAL,
|
||||
reason=REASON_ALREADY_TERMINAL,
|
||||
retireable=False,
|
||||
status=status,
|
||||
pid=pid,
|
||||
pid_alive=None,
|
||||
pid_reused=False,
|
||||
heartbeat_stale=heartbeat_stale,
|
||||
has_live_lease=has_live_lease,
|
||||
client_managed=client_managed,
|
||||
)
|
||||
|
||||
if has_live_lease:
|
||||
return SessionClassification(
|
||||
session_id=session_id,
|
||||
classification=CLASS_PROTECTED,
|
||||
reason=REASON_LIVE_LEASE,
|
||||
retireable=False,
|
||||
status=status,
|
||||
pid=pid,
|
||||
pid_alive=pid_checker(pid) if pid is not None else None,
|
||||
pid_reused=False,
|
||||
heartbeat_stale=heartbeat_stale,
|
||||
has_live_lease=True,
|
||||
client_managed=client_managed,
|
||||
details={"blocker": "live_workflow_lease"},
|
||||
)
|
||||
|
||||
if pid is None:
|
||||
return SessionClassification(
|
||||
session_id=session_id,
|
||||
classification=CLASS_STALE,
|
||||
reason=REASON_MISSING_PID,
|
||||
retireable=True,
|
||||
status=status,
|
||||
pid=None,
|
||||
pid_alive=False,
|
||||
pid_reused=False,
|
||||
heartbeat_stale=heartbeat_stale,
|
||||
has_live_lease=False,
|
||||
client_managed=client_managed,
|
||||
)
|
||||
|
||||
pid_alive = bool(pid_checker(pid))
|
||||
pid_reused = False
|
||||
proc_start: datetime | None = None
|
||||
|
||||
if pid_alive:
|
||||
proc_start = recorded_proc_start or process_start_probe(pid)
|
||||
# PID reuse: live process started after the session row itself was
|
||||
# created. Anchor on started_at only — last_heartbeat alone is not a
|
||||
# safe bound (synthetic inventories and long-lived processes would
|
||||
# false-positive against a live ``ps`` probe).
|
||||
if proc_start is not None and started is not None:
|
||||
if proc_start > (started + PID_REUSE_SKEW):
|
||||
pid_reused = True
|
||||
|
||||
if pid_reused:
|
||||
return SessionClassification(
|
||||
session_id=session_id,
|
||||
classification=CLASS_STALE,
|
||||
reason=REASON_PID_REUSE,
|
||||
retireable=True,
|
||||
status=status,
|
||||
pid=pid,
|
||||
pid_alive=True,
|
||||
pid_reused=True,
|
||||
heartbeat_stale=heartbeat_stale,
|
||||
has_live_lease=False,
|
||||
client_managed=client_managed,
|
||||
details={
|
||||
"process_started_at": _ts(proc_start) if proc_start else None,
|
||||
"session_started_at": row.get("started_at"),
|
||||
"last_heartbeat_at": row.get("last_heartbeat_at"),
|
||||
},
|
||||
)
|
||||
|
||||
if pid_alive and client_managed:
|
||||
return SessionClassification(
|
||||
session_id=session_id,
|
||||
classification=CLASS_LIVE,
|
||||
reason=REASON_CLIENT_MANAGED_LIVE,
|
||||
retireable=False,
|
||||
status=status,
|
||||
pid=pid,
|
||||
pid_alive=True,
|
||||
pid_reused=False,
|
||||
heartbeat_stale=heartbeat_stale,
|
||||
has_live_lease=False,
|
||||
client_managed=True,
|
||||
)
|
||||
|
||||
if pid_alive and not heartbeat_stale:
|
||||
return SessionClassification(
|
||||
session_id=session_id,
|
||||
classification=CLASS_LIVE,
|
||||
reason=REASON_LIVE_OWNER,
|
||||
retireable=False,
|
||||
status=status,
|
||||
pid=pid,
|
||||
pid_alive=True,
|
||||
pid_reused=False,
|
||||
heartbeat_stale=False,
|
||||
has_live_lease=False,
|
||||
client_managed=client_managed,
|
||||
)
|
||||
|
||||
if pid_alive and heartbeat_stale:
|
||||
# Process still exists but has not heartbeated — disconnected, not
|
||||
# confirmed stale. Do not retire; operator/reconnect owns next step.
|
||||
return SessionClassification(
|
||||
session_id=session_id,
|
||||
classification=CLASS_DISCONNECTED,
|
||||
reason=REASON_LIVE_OWNER,
|
||||
retireable=False,
|
||||
status=status,
|
||||
pid=pid,
|
||||
pid_alive=True,
|
||||
pid_reused=False,
|
||||
heartbeat_stale=True,
|
||||
has_live_lease=False,
|
||||
client_managed=client_managed,
|
||||
details={"note": "alive_pid_stale_heartbeat_not_retired"},
|
||||
)
|
||||
|
||||
# PID dead (or checker said not alive).
|
||||
reason = REASON_DEAD_OWNER
|
||||
if heartbeat_stale:
|
||||
reason = REASON_HEARTBEAT_STALE_DEAD
|
||||
return SessionClassification(
|
||||
session_id=session_id,
|
||||
classification=CLASS_STALE,
|
||||
reason=reason,
|
||||
retireable=True,
|
||||
status=status,
|
||||
pid=pid,
|
||||
pid_alive=False,
|
||||
pid_reused=False,
|
||||
heartbeat_stale=heartbeat_stale,
|
||||
has_live_lease=False,
|
||||
client_managed=client_managed,
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SessionFleetReport:
|
||||
"""Fleet-wide classification summary for reconcile + apply."""
|
||||
|
||||
classifications: tuple[SessionClassification, ...]
|
||||
live_count: int
|
||||
disconnected_count: int
|
||||
stale_count: int
|
||||
protected_count: int
|
||||
terminal_count: int
|
||||
retireable: tuple[SessionClassification, ...]
|
||||
counts_by_reason: dict[str, int]
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"live_count": self.live_count,
|
||||
"disconnected_count": self.disconnected_count,
|
||||
"stale_count": self.stale_count,
|
||||
"protected_count": self.protected_count,
|
||||
"terminal_count": self.terminal_count,
|
||||
"retireable_count": len(self.retireable),
|
||||
"retireable_session_ids": [c.session_id for c in self.retireable],
|
||||
"counts_by_reason": dict(self.counts_by_reason),
|
||||
"classifications": [c.as_dict() for c in self.classifications],
|
||||
"sessions_dimension_resolved": len(self.retireable) == 0,
|
||||
}
|
||||
|
||||
|
||||
def classify_sessions(
|
||||
sessions: Sequence[Mapping[str, Any]] | None,
|
||||
*,
|
||||
leases: Sequence[Mapping[str, Any]] | None = None,
|
||||
now: datetime | None = None,
|
||||
pid_checker: Callable[[int | None], bool] = lease_lifecycle.is_process_alive,
|
||||
process_start_probe: Callable[[int | None], datetime | None] = process_start_time,
|
||||
client_managed_sessions: set[str] | frozenset[str] | None = None,
|
||||
heartbeat_stale_seconds: int = DEFAULT_HEARTBEAT_STALE_SECONDS,
|
||||
) -> SessionFleetReport:
|
||||
"""Classify a fleet of session rows against live leases (#969)."""
|
||||
moment = now or _utc_now()
|
||||
live_leases = live_lease_session_ids(
|
||||
leases, now=moment, pid_checker=pid_checker
|
||||
)
|
||||
results: list[SessionClassification] = []
|
||||
for row in sessions or ():
|
||||
if not isinstance(row, Mapping):
|
||||
continue
|
||||
results.append(
|
||||
classify_session(
|
||||
row,
|
||||
now=moment,
|
||||
pid_checker=pid_checker,
|
||||
process_start_probe=process_start_probe,
|
||||
live_lease_sessions=live_leases,
|
||||
client_managed_sessions=client_managed_sessions,
|
||||
heartbeat_stale_seconds=heartbeat_stale_seconds,
|
||||
)
|
||||
)
|
||||
|
||||
live_count = sum(1 for c in results if c.classification == CLASS_LIVE)
|
||||
disconnected_count = sum(
|
||||
1 for c in results if c.classification == CLASS_DISCONNECTED
|
||||
)
|
||||
stale_count = sum(1 for c in results if c.classification == CLASS_STALE)
|
||||
protected_count = sum(1 for c in results if c.classification == CLASS_PROTECTED)
|
||||
terminal_count = sum(1 for c in results if c.classification == CLASS_TERMINAL)
|
||||
retireable = tuple(c for c in results if c.retireable)
|
||||
by_reason: dict[str, int] = {}
|
||||
for c in results:
|
||||
by_reason[c.reason] = by_reason.get(c.reason, 0) + 1
|
||||
|
||||
return SessionFleetReport(
|
||||
classifications=tuple(results),
|
||||
live_count=live_count,
|
||||
disconnected_count=disconnected_count,
|
||||
stale_count=stale_count,
|
||||
protected_count=protected_count,
|
||||
terminal_count=terminal_count,
|
||||
retireable=retireable,
|
||||
counts_by_reason=by_reason,
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RetirementResult:
|
||||
"""Outcome of one session retirement attempt."""
|
||||
|
||||
session_id: str
|
||||
outcome: str # retired | already_terminal | skipped | blocked | missing
|
||||
reason: str
|
||||
prior_status: str | None = None
|
||||
new_status: str | None = None
|
||||
details: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
def as_dict(self) -> dict[str, Any]:
|
||||
return {
|
||||
"session_id": self.session_id,
|
||||
"outcome": self.outcome,
|
||||
"reason": self.reason,
|
||||
"prior_status": self.prior_status,
|
||||
"new_status": self.new_status,
|
||||
"details": dict(self.details),
|
||||
}
|
||||
|
||||
|
||||
def apply_session_retirements(
|
||||
db: cpd.ControlPlaneDB,
|
||||
report: SessionFleetReport,
|
||||
*,
|
||||
dry_run: bool = False,
|
||||
actor_session_id: str | None = None,
|
||||
now: datetime | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Terminalize every retireable session in *report* (idempotent).
|
||||
|
||||
Concurrent reconciles are safe: each retirement is a CAS on
|
||||
``status='active'`` (or other non-terminal). A second pass that sees the
|
||||
same session already retired records ``already_terminal`` rather than
|
||||
duplicating audit noise beyond a single no-op outcome.
|
||||
"""
|
||||
moment = now or _utc_now()
|
||||
results: list[RetirementResult] = []
|
||||
retired = 0
|
||||
already = 0
|
||||
blocked = 0
|
||||
missing = 0
|
||||
|
||||
for classification in report.retireable:
|
||||
if not classification.retireable:
|
||||
continue
|
||||
if dry_run:
|
||||
results.append(
|
||||
RetirementResult(
|
||||
session_id=classification.session_id,
|
||||
outcome="skipped",
|
||||
reason=classification.reason,
|
||||
prior_status=classification.status,
|
||||
new_status=SESSION_STATUS_RETIRED,
|
||||
details={"dry_run": True, **classification.details},
|
||||
)
|
||||
)
|
||||
continue
|
||||
|
||||
applied = db.retire_session(
|
||||
session_id=classification.session_id,
|
||||
reason=classification.reason,
|
||||
actor_session_id=actor_session_id,
|
||||
details={
|
||||
"classification": classification.classification,
|
||||
"pid": classification.pid,
|
||||
"pid_alive": classification.pid_alive,
|
||||
"pid_reused": classification.pid_reused,
|
||||
"client_managed": classification.client_managed,
|
||||
**classification.details,
|
||||
},
|
||||
now=moment,
|
||||
)
|
||||
outcome = str(applied.get("outcome") or "missing")
|
||||
results.append(
|
||||
RetirementResult(
|
||||
session_id=classification.session_id,
|
||||
outcome=outcome,
|
||||
reason=str(applied.get("reason") or classification.reason),
|
||||
prior_status=applied.get("prior_status"),
|
||||
new_status=applied.get("new_status"),
|
||||
details=dict(applied.get("details") or {}),
|
||||
)
|
||||
)
|
||||
if outcome == "retired":
|
||||
retired += 1
|
||||
elif outcome == "already_terminal":
|
||||
already += 1
|
||||
elif outcome == "blocked":
|
||||
blocked += 1
|
||||
else:
|
||||
missing += 1
|
||||
|
||||
# Non-retireable classifications are recorded for audit completeness when
|
||||
# dry-run lists the fleet, but apply only mutates retireable rows.
|
||||
return {
|
||||
"success": True,
|
||||
"dry_run": dry_run,
|
||||
"retired_count": retired,
|
||||
"already_terminal_count": already,
|
||||
"blocked_count": blocked,
|
||||
"missing_count": missing,
|
||||
"planned_count": len(report.retireable),
|
||||
"results": [r.as_dict() for r in results],
|
||||
"audit_action": EVENT_SESSION_RETIRED,
|
||||
"actor_session_id": actor_session_id,
|
||||
"recorded_at": _ts(moment),
|
||||
}
|
||||
|
||||
|
||||
def retire_stale_sessions(
|
||||
db: cpd.ControlPlaneDB,
|
||||
*,
|
||||
sessions: Sequence[Mapping[str, Any]] | None = None,
|
||||
leases: Sequence[Mapping[str, Any]] | None = None,
|
||||
dry_run: bool = False,
|
||||
actor_session_id: str | None = None,
|
||||
now: datetime | None = None,
|
||||
pid_checker: Callable[[int | None], bool] = lease_lifecycle.is_process_alive,
|
||||
process_start_probe: Callable[[int | None], datetime | None] = process_start_time,
|
||||
client_managed_sessions: set[str] | frozenset[str] | None = None,
|
||||
heartbeat_stale_seconds: int = DEFAULT_HEARTBEAT_STALE_SECONDS,
|
||||
session_limit: int = 500,
|
||||
) -> dict[str, Any]:
|
||||
"""End-to-end plan + apply for stale session retirement (#969).
|
||||
|
||||
When *sessions* is omitted the control-plane DB is inventoried (active
|
||||
rows only). Callers that already gathered inventory should pass it.
|
||||
"""
|
||||
moment = now or _utc_now()
|
||||
if sessions is None:
|
||||
sessions = db.list_sessions(
|
||||
statuses=(SESSION_STATUS_ACTIVE,),
|
||||
limit=max(1, int(session_limit)),
|
||||
)
|
||||
if leases is None:
|
||||
try:
|
||||
leases = db.list_leases(statuses=("active",), limit=max(1, int(session_limit)))
|
||||
except Exception: # noqa: BLE001
|
||||
leases = []
|
||||
|
||||
report = classify_sessions(
|
||||
sessions,
|
||||
leases=leases,
|
||||
now=moment,
|
||||
pid_checker=pid_checker,
|
||||
process_start_probe=process_start_probe,
|
||||
client_managed_sessions=client_managed_sessions,
|
||||
heartbeat_stale_seconds=heartbeat_stale_seconds,
|
||||
)
|
||||
apply_result = apply_session_retirements(
|
||||
db,
|
||||
report,
|
||||
dry_run=dry_run,
|
||||
actor_session_id=actor_session_id,
|
||||
now=moment,
|
||||
)
|
||||
return {
|
||||
"success": True,
|
||||
"fleet": report.as_dict(),
|
||||
"apply": apply_result,
|
||||
"sessions_dimension_resolved": (
|
||||
report.as_dict()["sessions_dimension_resolved"]
|
||||
if dry_run
|
||||
else apply_result["planned_count"]
|
||||
== (
|
||||
apply_result["retired_count"]
|
||||
+ apply_result["already_terminal_count"]
|
||||
)
|
||||
and apply_result["blocked_count"] == 0
|
||||
),
|
||||
}
|
||||
+69
-8
@@ -153,9 +153,8 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.branch.push",
|
||||
"role": "author",
|
||||
},
|
||||
# #662: post-restart reconcile is inventory + pure classification.
|
||||
# #969: optional apply_session_cleanup retires confirmed-stale session rows
|
||||
# through the same tool; durable follow-up Gitea issues remain separate.
|
||||
# #662: post-restart reconcile is read-only inventory + pure classification.
|
||||
# Durable follow-up issue creation is a separate apply path (not this task).
|
||||
"reconcile_after_restart": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
@@ -164,14 +163,55 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
# #969: explicit plan/apply path for dead-owner session retirement.
|
||||
"retire_stale_workflow_sessions": {
|
||||
# #978: instance-level fleet identity/health snapshot. gitea.read is the
|
||||
# operation gate; the tool additionally restricts role_kind to
|
||||
# controller|reconciler so author/reviewer/merger cannot use it as a
|
||||
# mutation surface and no unrelated write permission is introduced.
|
||||
"snapshot_instance_fleet": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
"role": "controller",
|
||||
},
|
||||
"gitea_retire_stale_workflow_sessions": {
|
||||
"gitea_snapshot_instance_fleet": {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
"role": "controller",
|
||||
},
|
||||
# #980: CAS-protected retirement of conclusively stale worker
|
||||
# registrations.
|
||||
#
|
||||
# Planning is observational and stays on ``gitea.read``: it opens no
|
||||
# transaction, writes nothing, and returns only what a fleet snapshot
|
||||
# already exposes to the same roles.
|
||||
#
|
||||
# Applying is a mutation and review 657 B1 established that ``gitea.read``
|
||||
# cannot authorize it. The mutation landing in the local control-plane
|
||||
# registry rather than in Gitea makes it *no less* a mutation, and sharing
|
||||
# an observational permission class with plan meant any profile that could
|
||||
# look could also destroy. It now requires its own permission,
|
||||
# ``gitea.worker_registry.retire``, which no profile holds by default — so
|
||||
# author, reviewer, merger, and ordinary read-only profiles fail closed on
|
||||
# the permission itself rather than relying on the role check alone. The
|
||||
# role restriction (controller/reconciler), runtime parity, cohort
|
||||
# uniqueness, and the exact registry + candidate fingerprints all remain,
|
||||
# and are now defence in depth behind the capability rather than a
|
||||
# substitute for it.
|
||||
#
|
||||
# Granting the permission is a deliberate operator act in profiles.json;
|
||||
# removing it from a profile immediately and completely revokes apply.
|
||||
"plan_stale_worker_retirement": {
|
||||
"permission": "gitea.read",
|
||||
"role": "controller",
|
||||
},
|
||||
"gitea_plan_stale_worker_retirement": {
|
||||
"permission": "gitea.read",
|
||||
"role": "controller",
|
||||
},
|
||||
"apply_stale_worker_retirement": {
|
||||
"permission": "gitea.worker_registry.retire",
|
||||
"role": "controller",
|
||||
},
|
||||
"gitea_apply_stale_worker_retirement": {
|
||||
"permission": "gitea.worker_registry.retire",
|
||||
"role": "controller",
|
||||
},
|
||||
# #644: Phase 2 Web Console recovery tasks.
|
||||
"clear_stale_binding": {
|
||||
@@ -396,6 +436,25 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.branch.delete",
|
||||
"role": "reconciler",
|
||||
},
|
||||
# #970: auditing missing worktree bindings is read-only; retiring one is a
|
||||
# control-plane cleanup mutation and carries the same reconciler-only
|
||||
# authority as any other reconciliation cleanup (review 644 B2).
|
||||
"audit_missing_worktree_bindings": {
|
||||
"permission": "gitea.read",
|
||||
"role": "reconciler",
|
||||
},
|
||||
"gitea_audit_missing_worktree_bindings": {
|
||||
"permission": "gitea.read",
|
||||
"role": "reconciler",
|
||||
},
|
||||
"reconcile_missing_worktree_bindings": {
|
||||
"permission": "gitea.branch.delete",
|
||||
"role": "reconciler",
|
||||
},
|
||||
"gitea_reconcile_missing_worktree_bindings": {
|
||||
"permission": "gitea.branch.delete",
|
||||
"role": "reconciler",
|
||||
},
|
||||
"work_issue": {
|
||||
"permission": "gitea.pr.create",
|
||||
"role": "author",
|
||||
@@ -663,6 +722,8 @@ ROLE_EXCLUSIVE_TASKS: frozenset[str] = frozenset(
|
||||
"delete_branch",
|
||||
"cleanup_merged_pr_branch",
|
||||
"reconciliation_cleanup",
|
||||
"reconcile_missing_worktree_bindings",
|
||||
"gitea_reconcile_missing_worktree_bindings",
|
||||
"work_issue",
|
||||
"work-issue",
|
||||
}
|
||||
|
||||
@@ -175,8 +175,21 @@ class TestLauncherSnippets(unittest.TestCase):
|
||||
def test_only_safe_keys_no_secrets(self):
|
||||
entry = gitea_config.launcher_entry("prgs", "/cfg/profiles.json")["gitea-tools"]
|
||||
self.assertEqual(set(entry), {"command", "args", "env"})
|
||||
self.assertEqual(set(entry["env"]), {"GITEA_MCP_CONFIG", "GITEA_MCP_PROFILE", "GITEA_CLIENT_MANAGED"})
|
||||
# #978 B1: production launcher also injects trusted client identity.
|
||||
required = {
|
||||
"GITEA_MCP_CONFIG",
|
||||
"GITEA_MCP_PROFILE",
|
||||
"GITEA_CLIENT_MANAGED",
|
||||
"GITEA_MCP_CLIENT",
|
||||
"GITEA_MCP_CLIENT_INSTANCE",
|
||||
"GITEA_MCP_INSTANCE_PROVENANCE",
|
||||
}
|
||||
self.assertTrue(required.issubset(set(entry["env"])), entry["env"])
|
||||
self.assertEqual(entry["env"]["GITEA_MCP_PROFILE"], "prgs")
|
||||
self.assertTrue(
|
||||
entry["env"]["GITEA_MCP_CLIENT_INSTANCE"].startswith("inst-"),
|
||||
entry["env"]["GITEA_MCP_CLIENT_INSTANCE"],
|
||||
)
|
||||
blob = json.dumps(entry).lower()
|
||||
for word in ("token", "password", "secret"):
|
||||
self.assertNotIn(word, blob)
|
||||
|
||||
@@ -37,7 +37,7 @@ class ControlPlaneDBTest(unittest.TestCase):
|
||||
rows = dict(conn.execute("SELECT key, value FROM schema_meta").fetchall())
|
||||
finally:
|
||||
conn.close()
|
||||
self.assertEqual(rows["schema_version"], "6")
|
||||
self.assertEqual(rows["schema_version"], "5")
|
||||
self.assertIn("DB coordinates", rows["architecture"])
|
||||
self.assertIn("bridge", rows["architecture"].lower())
|
||||
|
||||
@@ -868,7 +868,7 @@ class SessionCheckpointTest(unittest.TestCase):
|
||||
conn.close()
|
||||
self.assertIn("session_checkpoints", names)
|
||||
record = self._write()
|
||||
self.assertEqual(record["checkpoint_schema_version"], 6)
|
||||
self.assertEqual(record["checkpoint_schema_version"], 5)
|
||||
|
||||
# AC2 — checkpoints written for multi-role session fixtures.
|
||||
def test_multi_role_fixtures_each_get_a_row(self) -> None:
|
||||
|
||||
@@ -313,6 +313,7 @@ class TestCanonicalRootGuardBinding(_ServerHarness):
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="derivation",
|
||||
)
|
||||
self.assertFalse(got.get("block"), got.get("reasons"))
|
||||
self.assertEqual(got["resolved_slug"], TARGET_SLUG)
|
||||
|
||||
@@ -241,9 +241,10 @@ class TestNamespaceContextUsesConfiguredRoot(unittest.TestCase):
|
||||
process_project_root=self.install,
|
||||
env={},
|
||||
configured_canonical_root=self.target,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertEqual(ctx["canonical_repo_root"], self.target)
|
||||
self.assertFalse(ctx["roots_aligned"])
|
||||
self.assertTrue(ctx["roots_aligned"])
|
||||
|
||||
def test_target_worktree_is_member_of_target_root(self):
|
||||
got = nwb.amw.assess_workspace_repo_membership(
|
||||
@@ -268,6 +269,7 @@ class TestNamespaceContextUsesConfiguredRoot(unittest.TestCase):
|
||||
env={},
|
||||
current_branch="feat/issue-1",
|
||||
configured_canonical_root=self.target,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertFalse(assessment["block"], assessment.get("reasons"))
|
||||
self.assertEqual(assessment["canonical_repo_root"], self.target)
|
||||
|
||||
@@ -1,480 +0,0 @@
|
||||
"""Tests for dead-owner / PID-reuse session retirement (#969)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
import threading
|
||||
import unittest
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
import control_plane_db as cpd
|
||||
import post_restart_reconcile as prr
|
||||
import session_lifecycle as sl
|
||||
|
||||
NOW = datetime(2026, 7, 29, 12, 0, 0, tzinfo=timezone.utc)
|
||||
EARLIER = NOW - timedelta(hours=2)
|
||||
LATER = NOW + timedelta(minutes=5)
|
||||
|
||||
|
||||
def _session(
|
||||
session_id: str,
|
||||
*,
|
||||
pid: int | None = 4242,
|
||||
status: str = "active",
|
||||
started_at: datetime = EARLIER,
|
||||
last_heartbeat_at: datetime | None = None,
|
||||
client_managed: bool = False,
|
||||
owner_process_started_at: datetime | None = None,
|
||||
role: str = "author",
|
||||
) -> dict:
|
||||
hb = last_heartbeat_at if last_heartbeat_at is not None else started_at
|
||||
row = {
|
||||
"session_id": session_id,
|
||||
"role": role,
|
||||
"profile": "prgs-author",
|
||||
"pid": pid,
|
||||
"status": status,
|
||||
"started_at": cpd._ts(started_at),
|
||||
"last_heartbeat_at": cpd._ts(hb),
|
||||
"client_managed": client_managed,
|
||||
}
|
||||
if owner_process_started_at is not None:
|
||||
row["owner_process_started_at"] = cpd._ts(owner_process_started_at)
|
||||
return row
|
||||
|
||||
|
||||
def _alive(pids: set[int]):
|
||||
def _check(pid):
|
||||
try:
|
||||
return int(pid) in pids
|
||||
except (TypeError, ValueError):
|
||||
return False
|
||||
|
||||
return _check
|
||||
|
||||
|
||||
def _starts(mapping: dict[int, datetime]):
|
||||
def _probe(pid):
|
||||
try:
|
||||
return mapping.get(int(pid))
|
||||
except (TypeError, ValueError):
|
||||
return None
|
||||
|
||||
return _probe
|
||||
|
||||
|
||||
class ClassifyDeadOwnerTests(unittest.TestCase):
|
||||
def test_dead_owner_is_stale_and_retireable(self) -> None:
|
||||
c = sl.classify_session(
|
||||
_session("ghost", pid=2_000_000_000),
|
||||
now=NOW,
|
||||
pid_checker=_alive(set()),
|
||||
process_start_probe=_starts({}),
|
||||
)
|
||||
self.assertEqual(c.classification, sl.CLASS_STALE)
|
||||
self.assertTrue(c.retireable)
|
||||
self.assertIn(c.reason, {sl.REASON_DEAD_OWNER, sl.REASON_HEARTBEAT_STALE_DEAD})
|
||||
|
||||
def test_missing_pid_is_stale(self) -> None:
|
||||
c = sl.classify_session(
|
||||
_session("no-pid", pid=None),
|
||||
now=NOW,
|
||||
pid_checker=_alive(set()),
|
||||
)
|
||||
self.assertEqual(c.classification, sl.CLASS_STALE)
|
||||
self.assertEqual(c.reason, sl.REASON_MISSING_PID)
|
||||
self.assertTrue(c.retireable)
|
||||
|
||||
|
||||
class PidReuseTests(unittest.TestCase):
|
||||
def test_pid_reuse_marks_stale_not_live(self) -> None:
|
||||
# Process with same PID started AFTER the session was recorded.
|
||||
c = sl.classify_session(
|
||||
_session("reused", pid=77, started_at=EARLIER, last_heartbeat_at=EARLIER),
|
||||
now=NOW,
|
||||
pid_checker=_alive({77}),
|
||||
process_start_probe=_starts({77: LATER}),
|
||||
)
|
||||
self.assertEqual(c.classification, sl.CLASS_STALE)
|
||||
self.assertEqual(c.reason, sl.REASON_PID_REUSE)
|
||||
self.assertTrue(c.pid_reused)
|
||||
self.assertTrue(c.retireable)
|
||||
|
||||
def test_matching_process_start_is_live(self) -> None:
|
||||
c = sl.classify_session(
|
||||
_session(
|
||||
"same-proc",
|
||||
pid=88,
|
||||
started_at=EARLIER,
|
||||
last_heartbeat_at=NOW - timedelta(seconds=30),
|
||||
owner_process_started_at=EARLIER - timedelta(seconds=5),
|
||||
),
|
||||
now=NOW,
|
||||
pid_checker=_alive({88}),
|
||||
process_start_probe=_starts({88: EARLIER - timedelta(seconds=5)}),
|
||||
)
|
||||
self.assertEqual(c.classification, sl.CLASS_LIVE)
|
||||
self.assertFalse(c.retireable)
|
||||
|
||||
|
||||
class LiveOwnerAndLeaseTests(unittest.TestCase):
|
||||
def test_live_owner_not_retired(self) -> None:
|
||||
c = sl.classify_session(
|
||||
_session(
|
||||
"live",
|
||||
pid=os.getpid(),
|
||||
last_heartbeat_at=NOW - timedelta(seconds=10),
|
||||
),
|
||||
now=NOW,
|
||||
pid_checker=_alive({os.getpid()}),
|
||||
process_start_probe=_starts({os.getpid(): EARLIER}),
|
||||
)
|
||||
self.assertEqual(c.classification, sl.CLASS_LIVE)
|
||||
self.assertFalse(c.retireable)
|
||||
|
||||
def test_live_lease_blocks_retirement_even_if_pid_dead(self) -> None:
|
||||
c = sl.classify_session(
|
||||
_session("leased", pid=99999),
|
||||
now=NOW,
|
||||
pid_checker=_alive(set()),
|
||||
live_lease_sessions={"leased"},
|
||||
)
|
||||
self.assertEqual(c.classification, sl.CLASS_PROTECTED)
|
||||
self.assertEqual(c.reason, sl.REASON_LIVE_LEASE)
|
||||
self.assertFalse(c.retireable)
|
||||
|
||||
def test_client_managed_live_never_retired(self) -> None:
|
||||
c = sl.classify_session(
|
||||
_session(
|
||||
"client",
|
||||
pid=55,
|
||||
client_managed=True,
|
||||
last_heartbeat_at=NOW - timedelta(seconds=5),
|
||||
),
|
||||
now=NOW,
|
||||
pid_checker=_alive({55}),
|
||||
process_start_probe=_starts({55: EARLIER}),
|
||||
)
|
||||
self.assertEqual(c.classification, sl.CLASS_LIVE)
|
||||
self.assertEqual(c.reason, sl.REASON_CLIENT_MANAGED_LIVE)
|
||||
self.assertFalse(c.retireable)
|
||||
|
||||
def test_client_managed_dead_pid_is_retireable(self) -> None:
|
||||
# Dead client process is not a live client-managed session.
|
||||
c = sl.classify_session(
|
||||
_session("client-dead", pid=56, client_managed=True),
|
||||
now=NOW,
|
||||
pid_checker=_alive(set()),
|
||||
)
|
||||
self.assertEqual(c.classification, sl.CLASS_STALE)
|
||||
self.assertTrue(c.retireable)
|
||||
|
||||
|
||||
class TerminalAndDisconnectedTests(unittest.TestCase):
|
||||
def test_already_terminal_not_retireable(self) -> None:
|
||||
c = sl.classify_session(
|
||||
_session("done", status="retired", pid=1),
|
||||
now=NOW,
|
||||
pid_checker=_alive(set()),
|
||||
)
|
||||
self.assertEqual(c.classification, sl.CLASS_TERMINAL)
|
||||
self.assertFalse(c.retireable)
|
||||
|
||||
def test_alive_stale_heartbeat_is_disconnected_not_retired(self) -> None:
|
||||
c = sl.classify_session(
|
||||
_session(
|
||||
"quiet",
|
||||
pid=66,
|
||||
last_heartbeat_at=NOW - timedelta(hours=5),
|
||||
),
|
||||
now=NOW,
|
||||
pid_checker=_alive({66}),
|
||||
process_start_probe=_starts({66: EARLIER}),
|
||||
)
|
||||
self.assertEqual(c.classification, sl.CLASS_DISCONNECTED)
|
||||
self.assertFalse(c.retireable)
|
||||
|
||||
|
||||
class FleetAndApplyTests(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.db_path = os.path.join(self._tmp.name, "cp.sqlite3")
|
||||
self.db = cpd.ControlPlaneDB(self.db_path)
|
||||
|
||||
def tearDown(self) -> None:
|
||||
self._tmp.cleanup()
|
||||
|
||||
def test_mixed_fleet_and_apply_retires_only_stale(self) -> None:
|
||||
live_pid = os.getpid()
|
||||
# Use wall-clock "now" so upsert timestamps align with classification.
|
||||
moment = datetime.now(timezone.utc)
|
||||
proc_start = moment - timedelta(hours=1)
|
||||
self.db.upsert_session(
|
||||
session_id="s-live",
|
||||
role="author",
|
||||
pid=live_pid,
|
||||
status="active",
|
||||
owner_process_started_at=cpd._ts(proc_start),
|
||||
)
|
||||
self.db.upsert_session(
|
||||
session_id="s-dead", role="reviewer", pid=2_000_000_001, status="active"
|
||||
)
|
||||
self.db.upsert_session(
|
||||
session_id="s-ended", role="merger", pid=3, status="ended"
|
||||
)
|
||||
|
||||
sessions = self.db.list_sessions(limit=50)
|
||||
report = sl.classify_sessions(
|
||||
sessions,
|
||||
leases=[],
|
||||
now=moment,
|
||||
pid_checker=_alive({live_pid}),
|
||||
process_start_probe=_starts({live_pid: proc_start}),
|
||||
)
|
||||
self.assertGreaterEqual(report.stale_count, 1)
|
||||
self.assertTrue(
|
||||
any(c.session_id == "s-dead" and c.retireable for c in report.classifications)
|
||||
)
|
||||
self.assertTrue(
|
||||
any(
|
||||
c.session_id == "s-live" and not c.retireable
|
||||
for c in report.classifications
|
||||
)
|
||||
)
|
||||
|
||||
first = sl.apply_session_retirements(
|
||||
self.db, report, dry_run=False, actor_session_id="actor-1", now=moment
|
||||
)
|
||||
self.assertGreaterEqual(first["retired_count"], 1)
|
||||
# After retirement, active list should exclude s-dead.
|
||||
active = {
|
||||
s["session_id"]
|
||||
for s in self.db.list_sessions(statuses=("active",), limit=50)
|
||||
}
|
||||
self.assertNotIn("s-dead", active)
|
||||
self.assertIn("s-live", active)
|
||||
|
||||
# Repeated cleanup is idempotent.
|
||||
report2 = sl.classify_sessions(
|
||||
self.db.list_sessions(limit=50),
|
||||
leases=[],
|
||||
now=moment,
|
||||
pid_checker=_alive({live_pid}),
|
||||
process_start_probe=_starts({live_pid: proc_start}),
|
||||
)
|
||||
second = sl.apply_session_retirements(
|
||||
self.db, report2, dry_run=False, actor_session_id="actor-1", now=moment
|
||||
)
|
||||
# No double-retirement of the same row as a new mutation.
|
||||
self.assertEqual(second["retired_count"], 0)
|
||||
|
||||
# Durable audit event present.
|
||||
import sqlite3
|
||||
|
||||
conn = sqlite3.connect(self.db_path)
|
||||
try:
|
||||
events = conn.execute(
|
||||
"SELECT event_type, message FROM events WHERE event_type = ?",
|
||||
("session_retired",),
|
||||
).fetchall()
|
||||
finally:
|
||||
conn.close()
|
||||
self.assertTrue(events)
|
||||
self.assertTrue(any("s-dead" in (m or "") for _, m in events))
|
||||
|
||||
def test_live_lease_blocks_db_retirement(self) -> None:
|
||||
self.db.upsert_session(
|
||||
session_id="s-leased", role="author", pid=2_000_000_002, status="active"
|
||||
)
|
||||
self.db.upsert_work_item(
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
kind="issue",
|
||||
number=969,
|
||||
)
|
||||
result = self.db.assign_and_lease(
|
||||
session_id="s-leased",
|
||||
role="author",
|
||||
remote="prgs",
|
||||
org="org",
|
||||
repo="repo",
|
||||
kind="issue",
|
||||
number=969,
|
||||
)
|
||||
self.assertEqual(result.outcome, "assigned")
|
||||
|
||||
# Inventory-style lease with explicit live freshness (authoritative for
|
||||
# the pure classifier). DB apply also blocks on the active lease row.
|
||||
leases = [
|
||||
{
|
||||
"lease_id": result.lease_id,
|
||||
"session_id": "s-leased",
|
||||
"status": "active",
|
||||
"freshness": {"freshness": "active"},
|
||||
}
|
||||
]
|
||||
report = sl.classify_sessions(
|
||||
self.db.list_sessions(statuses=("active",), limit=20),
|
||||
leases=leases,
|
||||
now=NOW,
|
||||
pid_checker=_alive(set()),
|
||||
)
|
||||
# Classifier protects via live lease set.
|
||||
self.assertTrue(
|
||||
any(
|
||||
c.session_id == "s-leased" and c.classification == sl.CLASS_PROTECTED
|
||||
for c in report.classifications
|
||||
)
|
||||
)
|
||||
apply = sl.apply_session_retirements(
|
||||
self.db, report, dry_run=False, actor_session_id="actor", now=NOW
|
||||
)
|
||||
self.assertEqual(apply["retired_count"], 0)
|
||||
active = {
|
||||
s["session_id"]
|
||||
for s in self.db.list_sessions(statuses=("active",), limit=20)
|
||||
}
|
||||
self.assertIn("s-leased", active)
|
||||
|
||||
# Direct DB CAS also refuses while an active lease row remains.
|
||||
blocked = self.db.retire_session(
|
||||
session_id="s-leased",
|
||||
reason=sl.REASON_DEAD_OWNER,
|
||||
actor_session_id="actor",
|
||||
now=NOW,
|
||||
)
|
||||
self.assertEqual(blocked["outcome"], "blocked")
|
||||
self.assertEqual(blocked["reason"], "live_lease")
|
||||
|
||||
def test_concurrent_retirement_is_idempotent(self) -> None:
|
||||
for i in range(20):
|
||||
self.db.upsert_session(
|
||||
session_id=f"ghost-{i}",
|
||||
role="author",
|
||||
pid=3_000_000 + i,
|
||||
status="active",
|
||||
)
|
||||
|
||||
def _worker() -> dict:
|
||||
return sl.retire_stale_sessions(
|
||||
self.db,
|
||||
dry_run=False,
|
||||
actor_session_id=f"actor-{threading.get_ident()}",
|
||||
now=NOW,
|
||||
pid_checker=_alive(set()),
|
||||
process_start_probe=_starts({}),
|
||||
session_limit=100,
|
||||
)
|
||||
|
||||
outcomes = []
|
||||
with ThreadPoolExecutor(max_workers=4) as pool:
|
||||
futs = [pool.submit(_worker) for _ in range(4)]
|
||||
for fut in as_completed(futs):
|
||||
outcomes.append(fut.result())
|
||||
|
||||
total_retired = sum(o["apply"]["retired_count"] for o in outcomes)
|
||||
# Exactly one successful retirement per ghost row across all workers.
|
||||
self.assertEqual(total_retired, 20)
|
||||
active = {
|
||||
s["session_id"]
|
||||
for s in self.db.list_sessions(statuses=("active",), limit=100)
|
||||
}
|
||||
for i in range(20):
|
||||
self.assertNotIn(f"ghost-{i}", active)
|
||||
|
||||
|
||||
class ReconcileIntegrationTests(unittest.TestCase):
|
||||
def test_unresolved_until_retired_then_resolved(self) -> None:
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"incomplete_reasons": [],
|
||||
"service_health": {"healthy": True},
|
||||
"clients": [{"session_id": "c1", "connected": True}],
|
||||
"sessions": [
|
||||
_session("ghost", pid=2_000_000_099, last_heartbeat_at=EARLIER),
|
||||
],
|
||||
"leases": [],
|
||||
"checkpoints_available": False,
|
||||
"worktree_bindings": [],
|
||||
"pending_mutations": [],
|
||||
"capabilities": {"stale": False},
|
||||
"boot_head_sha": "a" * 40,
|
||||
"current_head_sha": "a" * 40,
|
||||
"queue_state": {"safe_to_resume": True},
|
||||
}
|
||||
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_LOG_ONLY)
|
||||
sess = next(i for i in proof.items if i.dimension == prr.DIM_SESSIONS)
|
||||
self.assertEqual(sess.status, prr.ITEM_UNRESOLVED)
|
||||
self.assertIn("ghost", sess.details.get("orphan_session_ids") or [])
|
||||
|
||||
# After retirement inventory (no active orphans) resolves.
|
||||
inv2 = dict(inv)
|
||||
inv2["sessions"] = []
|
||||
inv2["session_fleet"] = {
|
||||
"retireable_session_ids": [],
|
||||
"sessions_dimension_resolved": True,
|
||||
"live_count": 0,
|
||||
"stale_count": 0,
|
||||
}
|
||||
proof2 = prr.reconcile_after_restart(inv2, now=NOW, mode=prr.MODE_LOG_ONLY)
|
||||
sess2 = next(i for i in proof2.items if i.dimension == prr.DIM_SESSIONS)
|
||||
self.assertEqual(sess2.status, prr.ITEM_RESOLVED)
|
||||
|
||||
def test_legacy_orphan_key_still_populated(self) -> None:
|
||||
inv = {
|
||||
"inventory_complete": True,
|
||||
"service_health": {"healthy": True},
|
||||
"clients": [],
|
||||
"sessions": [_session("ghost", pid=2_000_000_100)],
|
||||
"leases": [],
|
||||
"checkpoints_available": False,
|
||||
"worktree_bindings": [],
|
||||
"pending_mutations": [],
|
||||
"capabilities": {"stale": False},
|
||||
"boot_head_sha": "a" * 40,
|
||||
"current_head_sha": "a" * 40,
|
||||
"queue_state": {"safe_to_resume": True},
|
||||
}
|
||||
proof = prr.reconcile_after_restart(inv, now=NOW)
|
||||
sess = next(i for i in proof.items if i.dimension == prr.DIM_SESSIONS)
|
||||
self.assertIn("orphan_session_ids", sess.details)
|
||||
|
||||
|
||||
class SchemaMigrationTests(unittest.TestCase):
|
||||
def test_lifecycle_columns_present(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as tmp:
|
||||
path = os.path.join(tmp, "cp.sqlite3")
|
||||
db = cpd.ControlPlaneDB(path)
|
||||
db.upsert_session(session_id="s1", role="author", pid=1)
|
||||
out = db.retire_session(
|
||||
session_id="s1",
|
||||
reason=sl.REASON_DEAD_OWNER,
|
||||
actor_session_id="tester",
|
||||
now=NOW,
|
||||
)
|
||||
self.assertEqual(out["outcome"], "retired")
|
||||
rows = db.list_sessions(limit=5)
|
||||
# May not appear under active filter
|
||||
all_rows = db.list_sessions(limit=5)
|
||||
# Re-open raw to check columns
|
||||
import sqlite3
|
||||
|
||||
conn = sqlite3.connect(path)
|
||||
try:
|
||||
cols = {r[1] for r in conn.execute("PRAGMA table_info(sessions)")}
|
||||
version = conn.execute(
|
||||
"SELECT value FROM schema_meta WHERE key='schema_version'"
|
||||
).fetchone()[0]
|
||||
finally:
|
||||
conn.close()
|
||||
self.assertIn("retired_at", cols)
|
||||
self.assertIn("retire_reason", cols)
|
||||
self.assertIn("owner_process_started_at", cols)
|
||||
self.assertEqual(version, "6")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,739 @@
|
||||
"""Regression tests for Issue #973 blocker B10: repository-authority mode contract.
|
||||
|
||||
Before this repair ``assess_canonical_repository_root`` never checked its
|
||||
``mode`` argument against an allowlist. Both dispatch points were permissive:
|
||||
|
||||
* the configured-root path tested ``mode == "derivation"`` and sent every other
|
||||
value into a catch-all ``else``, so an unsupported mode silently received
|
||||
*validation* semantics, and
|
||||
* the single-repository default path tested ``mode == "validation"``, so an
|
||||
unsupported mode skipped the identity comparison entirely and was strictly
|
||||
*weaker* than validation.
|
||||
|
||||
The measured consequence was that ``mode="invalid_mode"`` with matching expected
|
||||
and observed identities returned ``proven: True`` / ``block: False`` with no
|
||||
reasons, and that an unsupported mode passed on the default path where
|
||||
``"validation"`` correctly blocked.
|
||||
|
||||
These tests exercise the production module directly with real git repositories —
|
||||
no patched stand-in for the function under test — and drive the production
|
||||
enforcement and mutation-context consumers rather than only the intermediate
|
||||
assessment.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import inspect
|
||||
import os
|
||||
import subprocess
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
import canonical_repository_root as crr
|
||||
import gitea_config
|
||||
import gitea_mcp_server as mcp_server
|
||||
import namespace_workspace_binding as nwb
|
||||
import stable_control_runtime
|
||||
|
||||
|
||||
INSTALL_SLUG = "Scaled-Tech-Consulting/Gitea-Tools"
|
||||
TARGET_SLUG = "Scaled-Tech-Consulting/mcp-control-plane"
|
||||
FOREIGN_SLUG = "Someone-Else/Evil-Repo"
|
||||
|
||||
# Explicitly supplied values that must all be refused. Omission is *not* in this
|
||||
# list: omitting the argument keeps the documented ``"validation"`` default.
|
||||
UNSUPPORTED_STRING_MODES = (
|
||||
"invalid_mode",
|
||||
"",
|
||||
"validaton", # misspelling
|
||||
"derivaton", # misspelling
|
||||
"Validation", # case variant
|
||||
"DERIVATION", # case variant
|
||||
" validation", # leading whitespace
|
||||
"validation ", # trailing whitespace
|
||||
"derivation\n", # trailing newline
|
||||
"validation,derivation",
|
||||
)
|
||||
|
||||
UNSUPPORTED_NON_STRING_MODES = (
|
||||
None,
|
||||
0,
|
||||
1,
|
||||
True,
|
||||
False,
|
||||
3.14,
|
||||
[],
|
||||
["validation"],
|
||||
{},
|
||||
{"mode": "validation"},
|
||||
("validation",),
|
||||
object(),
|
||||
)
|
||||
|
||||
|
||||
def _init_repo(path: str, remote_url: str, *, user: str = "Test User") -> None:
|
||||
os.makedirs(path, exist_ok=True)
|
||||
subprocess.run(["git", "init", "-b", "master"], cwd=path, check=True,
|
||||
capture_output=True)
|
||||
subprocess.run(["git", "config", "user.email", "[email protected]"], cwd=path,
|
||||
check=True, capture_output=True)
|
||||
subprocess.run(["git", "config", "user.name", user], cwd=path, check=True,
|
||||
capture_output=True)
|
||||
with open(os.path.join(path, "README.md"), "w") as handle:
|
||||
handle.write(f"{os.path.basename(path)}\n")
|
||||
subprocess.run(["git", "add", "README.md"], cwd=path, check=True,
|
||||
capture_output=True)
|
||||
subprocess.run(["git", "commit", "-m", "initial"], cwd=path, check=True,
|
||||
capture_output=True)
|
||||
subprocess.run(["git", "remote", "add", "prgs", remote_url], cwd=path,
|
||||
check=True, capture_output=True)
|
||||
|
||||
|
||||
class _CanonicalRootFixture(unittest.TestCase):
|
||||
"""Real install / target / foreign git repositories, as in the #973 suite."""
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.tmp_dir = os.path.realpath(self._tmp.name)
|
||||
|
||||
self.install_root = os.path.join(self.tmp_dir, "Gitea-Tools")
|
||||
_init_repo(
|
||||
self.install_root,
|
||||
"https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools.git",
|
||||
)
|
||||
|
||||
self.target_root = os.path.join(self.tmp_dir, "mcp-control-plane")
|
||||
_init_repo(
|
||||
self.target_root,
|
||||
"https://gitea.prgs.cc/Scaled-Tech-Consulting/mcp-control-plane.git",
|
||||
)
|
||||
|
||||
self.evil_root = os.path.join(self.tmp_dir, "Evil-Repo")
|
||||
_init_repo(
|
||||
self.evil_root,
|
||||
"https://gitea.prgs.cc/Someone-Else/Evil-Repo.git",
|
||||
user="Evil User",
|
||||
)
|
||||
|
||||
self.target_branches = os.path.join(self.target_root, "branches")
|
||||
self.target_worktree = os.path.join(self.target_branches, "rev-pr-99")
|
||||
subprocess.run(
|
||||
["git", "worktree", "add", "-b", "rev-pr-99", self.target_worktree],
|
||||
cwd=self.target_root, check=True, capture_output=True,
|
||||
)
|
||||
|
||||
def tearDown(self):
|
||||
self._tmp.cleanup()
|
||||
|
||||
def assertRefusedForMode(self, assessment: dict, mode) -> None:
|
||||
"""Assert a fail-closed refusal attributable to *mode* and nothing else."""
|
||||
self.assertFalse(assessment["proven"], assessment)
|
||||
self.assertTrue(assessment["block"], assessment)
|
||||
self.assertEqual(assessment["reason_code"], crr.DENY_UNKNOWN_MODE, assessment)
|
||||
self.assertEqual(len(assessment["reasons"]), 1, assessment)
|
||||
reason = assessment["reasons"][0]
|
||||
self.assertIn("unsupported repository-authority mode", reason)
|
||||
self.assertIn(repr(mode), reason)
|
||||
# No trusted repository identity may be derived through an invalid mode.
|
||||
self.assertIsNone(assessment["resolved_slug"], assessment)
|
||||
self.assertIsNone(assessment["canonical_repo_root"], assessment)
|
||||
|
||||
|
||||
class TestB10UnsupportedModeIsRejected(_CanonicalRootFixture):
|
||||
"""Direct assessment tests for invalid-mode parsing."""
|
||||
|
||||
def test_invalid_string_with_missing_expected_identity(self):
|
||||
"""G1: blocks for the mode, not incidentally for a missing identity."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=None,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="invalid_mode",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "invalid_mode")
|
||||
self.assertFalse(
|
||||
any("unprovable or missing" in r for r in assessment["reasons"]),
|
||||
"must block because the mode is unsupported, not because the expected "
|
||||
"identity happened to be missing",
|
||||
)
|
||||
|
||||
def test_invalid_string_with_matching_identities(self):
|
||||
"""G2: the contract violation — matching identities used to return proven."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="invalid_mode",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "invalid_mode")
|
||||
|
||||
def test_invalid_string_with_conflicting_identities(self):
|
||||
"""G3: refused for the mode, not for the incidental identity mismatch."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.evil_root,
|
||||
source="env",
|
||||
expected_slug=INSTALL_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="invalid_mode",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "invalid_mode")
|
||||
self.assertFalse(
|
||||
any("identity mismatch" in r for r in assessment["reasons"]),
|
||||
"the identity comparison must not have run at all",
|
||||
)
|
||||
|
||||
def test_empty_string_mode(self):
|
||||
"""G4a: an explicitly supplied empty string is an unsupported value."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "")
|
||||
|
||||
def test_explicit_none_mode(self):
|
||||
"""G4b: explicit None is refused; it is not treated as omission."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode=None,
|
||||
)
|
||||
self.assertRefusedForMode(assessment, None)
|
||||
self.assertIn("of type NoneType", assessment["reasons"][0])
|
||||
|
||||
def test_representative_non_string_modes(self):
|
||||
for mode in UNSUPPORTED_NON_STRING_MODES:
|
||||
with self.subTest(mode=repr(mode)):
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode=mode,
|
||||
)
|
||||
self.assertRefusedForMode(assessment, mode)
|
||||
self.assertIn(
|
||||
f"of type {type(mode).__name__}", assessment["reasons"][0]
|
||||
)
|
||||
|
||||
def test_unknown_strings_misspellings_and_whitespace_variants(self):
|
||||
for mode in UNSUPPORTED_STRING_MODES:
|
||||
with self.subTest(mode=repr(mode)):
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode=mode,
|
||||
)
|
||||
self.assertRefusedForMode(assessment, mode)
|
||||
|
||||
def test_supported_modes_are_exactly_two(self):
|
||||
self.assertEqual(
|
||||
crr.SUPPORTED_MODES, ("validation", "derivation")
|
||||
)
|
||||
self.assertIsNone(crr.unsupported_mode_reason("validation"))
|
||||
self.assertIsNone(crr.unsupported_mode_reason("derivation"))
|
||||
self.assertIsNotNone(crr.unsupported_mode_reason("invalid_mode"))
|
||||
|
||||
|
||||
class TestB10SupportedModesUnchanged(_CanonicalRootFixture):
|
||||
"""The repair must not disturb the two documented modes."""
|
||||
|
||||
def test_omitted_mode_defaults_to_validation(self):
|
||||
"""G4c/G4d: omission still selects validation, proven by both outcomes."""
|
||||
matching = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
)
|
||||
self.assertTrue(matching["proven"], matching)
|
||||
self.assertFalse(matching["block"], matching)
|
||||
self.assertIsNone(matching["reason_code"], matching)
|
||||
self.assertEqual(matching["resolved_slug"], TARGET_SLUG)
|
||||
|
||||
conflicting = crr.assess_canonical_repository_root(
|
||||
configured_value=self.evil_root,
|
||||
source="env",
|
||||
expected_slug=INSTALL_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
)
|
||||
self.assertFalse(conflicting["proven"], conflicting)
|
||||
self.assertTrue(conflicting["block"], conflicting)
|
||||
self.assertIsNone(conflicting["reason_code"], conflicting)
|
||||
self.assertTrue(
|
||||
any("identity mismatch" in r for r in conflicting["reasons"]),
|
||||
"omission must behave exactly like explicit validation",
|
||||
)
|
||||
|
||||
def test_explicit_validation_retains_strict_behavior(self):
|
||||
"""G5a/G5b/G5c."""
|
||||
ok = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root, source="env",
|
||||
expected_slug=TARGET_SLUG, process_project_root=self.install_root,
|
||||
remote="prgs", require_binding=True, mode="validation",
|
||||
)
|
||||
self.assertTrue(ok["proven"], ok)
|
||||
|
||||
mismatch = crr.assess_canonical_repository_root(
|
||||
configured_value=self.evil_root, source="env",
|
||||
expected_slug=INSTALL_SLUG, process_project_root=self.install_root,
|
||||
remote="prgs", require_binding=True, mode="validation",
|
||||
)
|
||||
self.assertTrue(mismatch["block"], mismatch)
|
||||
self.assertTrue(any("identity mismatch" in r for r in mismatch["reasons"]))
|
||||
|
||||
unprovable = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root, source="env",
|
||||
expected_slug=None, process_project_root=self.install_root,
|
||||
remote="prgs", require_binding=True, mode="validation",
|
||||
)
|
||||
self.assertTrue(unprovable["block"], unprovable)
|
||||
self.assertTrue(
|
||||
any("unprovable or missing" in r for r in unprovable["reasons"])
|
||||
)
|
||||
|
||||
def test_explicit_derivation_retains_trusted_derivation(self):
|
||||
"""G5d: derivation still resolves identity with no expected slug."""
|
||||
derived = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root, source="env",
|
||||
expected_slug=None, process_project_root=self.install_root,
|
||||
remote="prgs", require_binding=True, mode="derivation",
|
||||
)
|
||||
self.assertTrue(derived["proven"], derived)
|
||||
self.assertFalse(derived["block"], derived)
|
||||
self.assertIsNone(derived["reason_code"], derived)
|
||||
self.assertEqual(derived["resolved_slug"], TARGET_SLUG)
|
||||
|
||||
def test_derivation_without_resolvable_remote_still_fails_closed(self):
|
||||
no_remote = os.path.join(self.tmp_dir, "no-remote-target")
|
||||
os.makedirs(no_remote)
|
||||
subprocess.run(["git", "init", "-b", "master"], cwd=no_remote, check=True,
|
||||
capture_output=True)
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=no_remote, source="env", expected_slug=None,
|
||||
process_project_root=self.install_root, remote="prgs",
|
||||
require_binding=True, mode="derivation",
|
||||
)
|
||||
self.assertTrue(assessment["block"], assessment)
|
||||
self.assertTrue(
|
||||
any("no resolvable" in r for r in assessment["reasons"]), assessment
|
||||
)
|
||||
|
||||
|
||||
class TestB10SingleRepositoryDefaultPath(_CanonicalRootFixture):
|
||||
"""G6: on the unconfigured path an invalid mode used to be weaker than validation."""
|
||||
|
||||
def _assess(self, mode_kwargs: dict) -> dict:
|
||||
return crr.assess_canonical_repository_root(
|
||||
configured_value=None,
|
||||
source=None,
|
||||
expected_slug=FOREIGN_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=False,
|
||||
**mode_kwargs,
|
||||
)
|
||||
|
||||
def test_validation_blocks_a_foreign_expected_identity(self):
|
||||
"""G6b: the reference behaviour the invalid mode must not undercut."""
|
||||
got = self._assess({"mode": "validation"})
|
||||
self.assertFalse(got["proven"], got)
|
||||
self.assertTrue(got["block"], got)
|
||||
self.assertTrue(any("identity mismatch" in r for r in got["reasons"]))
|
||||
|
||||
def test_invalid_mode_no_longer_passes_where_validation_blocks(self):
|
||||
"""G6a: identical inputs, only the mode differs — must not fail open."""
|
||||
invalid = self._assess({"mode": "invalid_mode"})
|
||||
self.assertRefusedForMode(invalid, "invalid_mode")
|
||||
|
||||
validation = self._assess({"mode": "validation"})
|
||||
self.assertEqual(
|
||||
invalid["proven"], validation["proven"],
|
||||
"an unsupported mode must never be more permissive than validation",
|
||||
)
|
||||
self.assertTrue(invalid["block"] and validation["block"])
|
||||
|
||||
def test_empty_string_mode_on_default_path(self):
|
||||
"""G6c."""
|
||||
self.assertRefusedForMode(self._assess({"mode": ""}), "")
|
||||
|
||||
def test_omitted_mode_on_default_path_still_validates(self):
|
||||
got = self._assess({})
|
||||
self.assertFalse(got["proven"], got)
|
||||
self.assertTrue(any("identity mismatch" in r for r in got["reasons"]), got)
|
||||
|
||||
|
||||
class TestB10RejectionOrdering(_CanonicalRootFixture):
|
||||
"""Refusal must precede every form of candidate-root or Git inspection."""
|
||||
|
||||
def test_no_git_or_identity_discovery_runs_for_an_unsupported_mode(self):
|
||||
with patch.object(crr, "resolve_repo_toplevel") as toplevel, \
|
||||
patch.object(crr, "repository_identity_slug") as identity:
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="invalid_mode",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "invalid_mode")
|
||||
toplevel.assert_not_called()
|
||||
identity.assert_not_called()
|
||||
|
||||
def test_the_same_spies_do_fire_for_a_supported_mode(self):
|
||||
"""Control: proves the previous test's assertions are not vacuous."""
|
||||
with patch.object(crr, "resolve_repo_toplevel",
|
||||
wraps=crr.resolve_repo_toplevel) as toplevel, \
|
||||
patch.object(crr, "repository_identity_slug",
|
||||
wraps=crr.repository_identity_slug) as identity:
|
||||
crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="validation",
|
||||
)
|
||||
toplevel.assert_called()
|
||||
identity.assert_called()
|
||||
|
||||
def test_nonexistent_candidate_root_still_reports_the_mode_refusal(self):
|
||||
"""No mocks: the existence check cannot have run before the refusal."""
|
||||
nonexistent = os.path.join(self.tmp_dir, "no-such-repository")
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=nonexistent,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="invalid_mode",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "invalid_mode")
|
||||
self.assertFalse(
|
||||
any("does not exist" in r for r in assessment["reasons"]), assessment
|
||||
)
|
||||
|
||||
def test_symlinked_candidate_root_is_not_resolved_for_an_unsupported_mode(self):
|
||||
link = os.path.join(self.tmp_dir, "target-alias")
|
||||
os.symlink(self.target_root, link)
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=link,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="invalid_mode",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "invalid_mode")
|
||||
# The refusal payload must not leak a resolved path for the candidate.
|
||||
self.assertIsNone(assessment["canonical_repo_root"], assessment)
|
||||
|
||||
|
||||
class TestB10NoModeInjectionSurface(unittest.TestCase):
|
||||
"""``mode`` must not be reachable from requests, environment, or config."""
|
||||
|
||||
PRODUCTION_MODULES = ("gitea_mcp_server.py", "namespace_workspace_binding.py")
|
||||
|
||||
def _repo_root(self) -> str:
|
||||
return os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
def test_production_call_sites_only_pass_allowlisted_literal_modes(self):
|
||||
"""Valid hardcoded modes reach the correct path; nothing else is passed."""
|
||||
seen: list[tuple[str, str | None]] = []
|
||||
for name in self.PRODUCTION_MODULES:
|
||||
path = os.path.join(self._repo_root(), name)
|
||||
with open(path) as handle:
|
||||
tree = ast.parse(handle.read())
|
||||
for node in ast.walk(tree):
|
||||
if not isinstance(node, ast.Call):
|
||||
continue
|
||||
func = node.func
|
||||
called = (
|
||||
func.attr if isinstance(func, ast.Attribute)
|
||||
else getattr(func, "id", None)
|
||||
)
|
||||
if called != "assess_canonical_repository_root":
|
||||
continue
|
||||
supplied = [k for k in node.keywords if k.arg == "mode"]
|
||||
if not supplied:
|
||||
seen.append((name, None))
|
||||
continue
|
||||
value = supplied[0].value
|
||||
self.assertIsInstance(
|
||||
value, ast.Constant,
|
||||
f"{name}: mode must be a literal, never a variable or expression",
|
||||
)
|
||||
self.assertIn(
|
||||
value.value, crr.SUPPORTED_MODES,
|
||||
f"{name}: unsupported mode literal {value.value!r}",
|
||||
)
|
||||
seen.append((name, value.value))
|
||||
self.assertTrue(seen, "expected production call sites to be found")
|
||||
# Both documented modes are exercised by production, and omission is used.
|
||||
self.assertIn(None, [mode for _, mode in seen])
|
||||
self.assertIn("derivation", [mode for _, mode in seen])
|
||||
|
||||
def test_no_public_entry_point_exposes_a_mode_parameter(self):
|
||||
for func in (
|
||||
nwb.resolve_namespace_mutation_context,
|
||||
nwb.assess_namespace_mutation_workspace,
|
||||
mcp_server._resolve_namespace_mutation_context,
|
||||
mcp_server._enforce_canonical_repository_root,
|
||||
mcp_server._canonical_repository_slug,
|
||||
mcp_server._trusted_session_repository,
|
||||
mcp_server._resolve_expected_repository_slug,
|
||||
):
|
||||
with self.subTest(func=func.__name__):
|
||||
self.assertNotIn("mode", inspect.signature(func).parameters)
|
||||
|
||||
def test_no_environment_key_selects_a_repository_authority_mode(self):
|
||||
for key in gitea_config.RECOGNIZED_GITEA_ENV_KEYS:
|
||||
self.assertNotIn(
|
||||
"CANONICAL_REPOSITORY_MODE", key.upper(),
|
||||
f"{key} would expose a repository-authority mode selector",
|
||||
)
|
||||
# An invented mode-ish variable is simply not consumed by crr.
|
||||
env = {
|
||||
"GITEA_CANONICAL_REPOSITORY_ROOT": "/some/path",
|
||||
"GITEA_CANONICAL_REPOSITORY_MODE": "invalid_mode",
|
||||
}
|
||||
value, source = crr.configured_canonical_root(None, env)
|
||||
self.assertEqual(value, "/some/path")
|
||||
self.assertNotIn("mode", (source or "").lower())
|
||||
self.assertIn(
|
||||
"GITEA_CANONICAL_REPOSITORY_MODE",
|
||||
gitea_config.get_unconsumed_gitea_env_overrides(env),
|
||||
"an unknown GITEA_* key must still be rejected as unrecognised",
|
||||
)
|
||||
|
||||
def test_repository_configuration_carries_no_mode_field(self):
|
||||
profile = {
|
||||
"canonical_repository_root": "/some/path",
|
||||
"mode": "invalid_mode",
|
||||
}
|
||||
value, source = crr.configured_canonical_root(profile, {})
|
||||
self.assertEqual(value, "/some/path")
|
||||
self.assertEqual(source, "profile canonical_repository_root")
|
||||
|
||||
|
||||
class TestB10ProductionEnforcementPaths(_CanonicalRootFixture):
|
||||
"""Production enforcement, mutation-context, reviewer and merger consumers."""
|
||||
|
||||
def _force_invalid_mode(self):
|
||||
"""Simulate a future call site threading an unsupported mode.
|
||||
|
||||
The real ``assess_canonical_repository_root`` still executes — only the
|
||||
caller-side argument is substituted — so the refusal under test is
|
||||
produced by production code, not by a stand-in. No caller-controlled
|
||||
``mode`` parameter is added to any production signature to achieve this.
|
||||
"""
|
||||
real = crr.assess_canonical_repository_root
|
||||
|
||||
def _wrapper(**kwargs):
|
||||
kwargs["mode"] = "invalid_mode"
|
||||
return real(**kwargs)
|
||||
|
||||
return patch.object(crr, "assess_canonical_repository_root", _wrapper)
|
||||
|
||||
def test_mutation_context_fails_closed_under_a_refused_mode(self):
|
||||
with self._force_invalid_mode():
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug=TARGET_SLUG,
|
||||
remote="prgs",
|
||||
)
|
||||
assessment = ctx["canonical_root_assessment"]
|
||||
self.assertFalse(ctx["roots_aligned"], ctx)
|
||||
self.assertTrue(assessment["block"], assessment)
|
||||
self.assertEqual(assessment["reason_code"], crr.DENY_UNKNOWN_MODE)
|
||||
self.assertIsNone(assessment["resolved_slug"])
|
||||
# The refused mode must not yield the candidate root as canonical.
|
||||
self.assertNotEqual(ctx["canonical_repo_root"], self.target_root)
|
||||
self.assertEqual(ctx["canonical_repo_root"], self.install_root)
|
||||
|
||||
def test_mutation_context_unchanged_for_the_supported_default(self):
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug=TARGET_SLUG,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertTrue(ctx["roots_aligned"], ctx)
|
||||
self.assertEqual(ctx["canonical_repo_root"], self.target_root)
|
||||
self.assertIsNone(ctx["canonical_root_assessment"]["reason_code"])
|
||||
|
||||
def test_reviewer_and_merger_authorization_fails_closed_under_a_refused_mode(self):
|
||||
for role in ("reviewer", "merger"):
|
||||
with self.subTest(role=role), self._force_invalid_mode():
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind=role,
|
||||
worktree_path=self.target_worktree,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug=TARGET_SLUG,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertTrue(assessment["block"], assessment)
|
||||
self.assertTrue(
|
||||
any("unsupported repository-authority mode" in r
|
||||
for r in assessment["reasons"]),
|
||||
assessment,
|
||||
)
|
||||
|
||||
def test_reviewer_and_merger_authorization_unchanged_for_supported_modes(self):
|
||||
for role in ("reviewer", "merger"):
|
||||
with self.subTest(role=role):
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind=role,
|
||||
worktree_path=self.target_worktree,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug=TARGET_SLUG,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertFalse(assessment["block"], assessment)
|
||||
|
||||
def test_final_mutation_gate_blocks_when_the_mode_refusal_unaligns_roots(self):
|
||||
with self._force_invalid_mode():
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug=TARGET_SLUG,
|
||||
remote="prgs",
|
||||
)
|
||||
report = stable_control_runtime.build_runtime_report(
|
||||
process_root=self.install_root,
|
||||
checkout_branch="master",
|
||||
runtime_head="abcdef123456",
|
||||
active_task_workspace=ctx["workspace_path"],
|
||||
canonical_repository_root=ctx["canonical_repo_root"],
|
||||
workspace_roots_aligned=ctx["roots_aligned"],
|
||||
)
|
||||
gate = stable_control_runtime.assess_runtime_mutation_gate(report)
|
||||
self.assertTrue(gate["block"], gate)
|
||||
|
||||
def test_enforce_canonical_repository_root_uses_the_validation_default(self):
|
||||
"""B8 boundary intact: a foreign configured root raises through production."""
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server, "_configured_canonical_root",
|
||||
return_value=(self.evil_root, "env")), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context",
|
||||
return_value=None):
|
||||
with self.assertRaises(RuntimeError) as raised:
|
||||
mcp_server._enforce_canonical_repository_root(remote="prgs")
|
||||
self.assertIn("identity mismatch", str(raised.exception))
|
||||
|
||||
def test_enforce_canonical_repository_root_refuses_a_threaded_invalid_mode(self):
|
||||
bound = {"org": "Scaled-Tech-Consulting",
|
||||
"repository": "mcp-control-plane", "remote": "prgs"}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server, "_configured_canonical_root",
|
||||
return_value=(self.target_root, "env")), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context",
|
||||
return_value=bound), \
|
||||
self._force_invalid_mode():
|
||||
with self.assertRaises(RuntimeError) as raised:
|
||||
mcp_server._enforce_canonical_repository_root(remote="prgs")
|
||||
self.assertIn("unsupported repository-authority mode", str(raised.exception))
|
||||
|
||||
def test_enforce_canonical_repository_root_passes_for_a_valid_binding(self):
|
||||
bound = {"org": "Scaled-Tech-Consulting",
|
||||
"repository": "mcp-control-plane", "remote": "prgs"}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server, "_configured_canonical_root",
|
||||
return_value=(self.target_root, "env")), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context",
|
||||
return_value=bound), \
|
||||
patch.object(mcp_server.session_ctx, "assess_session_context",
|
||||
return_value={"block": False, "reasons": []}), \
|
||||
patch.object(mcp_server, "get_profile",
|
||||
return_value={"profile_name": "prgs-reviewer"}):
|
||||
mcp_server._enforce_canonical_repository_root(remote="prgs")
|
||||
|
||||
def test_canonical_repository_slug_derivation_unbroken(self):
|
||||
"""B9 boundary intact: legitimate cross-repository derivation still works."""
|
||||
profile = {
|
||||
"profile_name": "prgs-author",
|
||||
"allowed_repositories": [TARGET_SLUG],
|
||||
"canonical_repository_root": self.target_root,
|
||||
}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context",
|
||||
return_value=None):
|
||||
slug, reasons = mcp_server._canonical_repository_slug(profile, "prgs")
|
||||
self.assertEqual(slug, TARGET_SLUG, reasons)
|
||||
self.assertEqual(reasons, [])
|
||||
|
||||
def test_canonical_repository_slug_fails_closed_under_a_refused_mode(self):
|
||||
profile = {
|
||||
"profile_name": "prgs-author",
|
||||
"allowed_repositories": [TARGET_SLUG],
|
||||
"canonical_repository_root": self.target_root,
|
||||
}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context",
|
||||
return_value=None), \
|
||||
self._force_invalid_mode():
|
||||
slug, reasons = mcp_server._canonical_repository_slug(profile, "prgs")
|
||||
self.assertIsNone(slug)
|
||||
self.assertTrue(
|
||||
any("unsupported repository-authority mode" in r for r in reasons),
|
||||
reasons,
|
||||
)
|
||||
result = mcp_server._trusted_session_repository(
|
||||
profile, "prgs", for_mutation=True
|
||||
)
|
||||
self.assertIsNone(result["org"])
|
||||
self.assertIsNone(result["repository"])
|
||||
self.assertTrue(result["reasons"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,546 @@
|
||||
"""Regression tests for Issue #973: validated cross-repository canonical roots."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest.mock import patch, MagicMock
|
||||
from pathlib import Path
|
||||
import subprocess
|
||||
|
||||
import gitea_config
|
||||
import namespace_workspace_binding as nwb
|
||||
import canonical_repository_root as crr
|
||||
import stable_control_runtime
|
||||
import gitea_mcp_server as mcp_server
|
||||
|
||||
|
||||
class TestIssue973RecognizedEnvKeys(unittest.TestCase):
|
||||
"""Test recognized environment variable keys under #973."""
|
||||
|
||||
def test_recognized_gitea_env_keys(self):
|
||||
for key in (
|
||||
"GITEA_CANONICAL_REPOSITORY_ROOT",
|
||||
"GITEA_REVIEWER_WORKTREE",
|
||||
"GITEA_MERGER_WORKTREE",
|
||||
"GITEA_MCP_SESSION_STATE_TTL_HOURS",
|
||||
):
|
||||
self.assertIn(key, gitea_config.RECOGNIZED_GITEA_ENV_KEYS)
|
||||
|
||||
def test_get_unconsumed_gitea_env_overrides_ignores_recognized(self):
|
||||
env = {
|
||||
"GITEA_CANONICAL_REPOSITORY_ROOT": "/some/path",
|
||||
"GITEA_REVIEWER_WORKTREE": "/some/reviewer/path",
|
||||
"GITEA_MERGER_WORKTREE": "/some/merger/path",
|
||||
"GITEA_MCP_SESSION_STATE_TTL_HOURS": "24",
|
||||
"GITEA_UNRECOGNIZED_FOO_VAR": "bar",
|
||||
}
|
||||
unconsumed = gitea_config.get_unconsumed_gitea_env_overrides(env)
|
||||
self.assertNotIn("GITEA_CANONICAL_REPOSITORY_ROOT", unconsumed)
|
||||
self.assertNotIn("GITEA_REVIEWER_WORKTREE", unconsumed)
|
||||
self.assertNotIn("GITEA_MERGER_WORKTREE", unconsumed)
|
||||
self.assertNotIn("GITEA_MCP_SESSION_STATE_TTL_HOURS", unconsumed)
|
||||
self.assertIn("GITEA_UNRECOGNIZED_FOO_VAR", unconsumed)
|
||||
|
||||
|
||||
class TestIssue973CrossRepoCanonicalRoots(unittest.TestCase):
|
||||
"""Test workspace binding and canonical root validation for cross-repo namespaces."""
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.tmp_dir = os.path.realpath(self._tmp.name)
|
||||
|
||||
# Create simulated installation root
|
||||
self.install_root = os.path.join(self.tmp_dir, "Gitea-Tools")
|
||||
os.makedirs(self.install_root)
|
||||
subprocess.run(["git", "init", "-b", "master"], cwd=self.install_root, check=True)
|
||||
subprocess.run(["git", "config", "user.email", "[email protected]"], cwd=self.install_root, check=True)
|
||||
subprocess.run(["git", "config", "user.name", "Test User"], cwd=self.install_root, check=True)
|
||||
with open(os.path.join(self.install_root, "README.md"), "w") as f:
|
||||
f.write("install\n")
|
||||
subprocess.run(["git", "add", "README.md"], cwd=self.install_root, check=True)
|
||||
subprocess.run(["git", "commit", "-m", "initial"], cwd=self.install_root, check=True)
|
||||
|
||||
# Create simulated target repository root
|
||||
self.target_root = os.path.join(self.tmp_dir, "mcp-control-plane")
|
||||
os.makedirs(self.target_root)
|
||||
subprocess.run(["git", "init", "-b", "master"], cwd=self.target_root, check=True)
|
||||
subprocess.run(["git", "config", "user.email", "[email protected]"], cwd=self.target_root, check=True)
|
||||
subprocess.run(["git", "config", "user.name", "Test User"], cwd=self.target_root, check=True)
|
||||
with open(os.path.join(self.target_root, "README.md"), "w") as f:
|
||||
f.write("target\n")
|
||||
subprocess.run(["git", "add", "README.md"], cwd=self.target_root, check=True)
|
||||
subprocess.run(["git", "commit", "-m", "initial"], cwd=self.target_root, check=True)
|
||||
|
||||
# Add remotes to simulate real git repositories with identities
|
||||
subprocess.run(["git", "remote", "add", "prgs", "https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools.git"], cwd=self.install_root, check=True)
|
||||
subprocess.run(["git", "remote", "add", "prgs", "https://gitea.prgs.cc/Scaled-Tech-Consulting/mcp-control-plane.git"], cwd=self.target_root, check=True)
|
||||
|
||||
# Create simulated foreign repository root
|
||||
self.evil_root = os.path.join(self.tmp_dir, "Evil-Repo")
|
||||
os.makedirs(self.evil_root)
|
||||
subprocess.run(["git", "init", "-b", "master"], cwd=self.evil_root, check=True)
|
||||
subprocess.run(["git", "config", "user.email", "[email protected]"], cwd=self.evil_root, check=True)
|
||||
subprocess.run(["git", "config", "user.name", "Evil User"], cwd=self.evil_root, check=True)
|
||||
with open(os.path.join(self.evil_root, "README.md"), "w") as f:
|
||||
f.write("evil\n")
|
||||
subprocess.run(["git", "add", "README.md"], cwd=self.evil_root, check=True)
|
||||
subprocess.run(["git", "commit", "-m", "initial"], cwd=self.evil_root, check=True)
|
||||
subprocess.run(["git", "remote", "add", "prgs", "https://gitea.prgs.cc/Someone-Else/Evil-Repo.git"], cwd=self.evil_root, check=True)
|
||||
|
||||
# Create branches/ directory and a valid registered worktree in target repository
|
||||
self.target_branches = os.path.join(self.target_root, "branches")
|
||||
self.target_worktree = os.path.join(self.target_branches, "rev-pr-99")
|
||||
subprocess.run(["git", "worktree", "add", "-b", "rev-pr-99", self.target_worktree], cwd=self.target_root, check=True)
|
||||
|
||||
def tearDown(self):
|
||||
self._tmp.cleanup()
|
||||
|
||||
def test_valid_same_repository_configuration(self):
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=None,
|
||||
)
|
||||
self.assertEqual(ctx["canonical_repo_root"], self.install_root)
|
||||
self.assertTrue(ctx["roots_aligned"])
|
||||
self.assertTrue(ctx["canonical_root_assessment"]["proven"])
|
||||
|
||||
def test_valid_cross_repo_canonical_root(self):
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertEqual(ctx["canonical_repo_root"], self.target_root)
|
||||
self.assertTrue(ctx["roots_aligned"])
|
||||
self.assertTrue(ctx["canonical_root_assessment"]["proven"])
|
||||
|
||||
def test_expected_repository_identity_match(self):
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="test",
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertTrue(assessment["proven"])
|
||||
self.assertFalse(assessment["block"])
|
||||
|
||||
def test_foreign_repository_identity_mismatch(self):
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.evil_root,
|
||||
source="test",
|
||||
expected_slug="Scaled-Tech-Consulting/Gitea-Tools",
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertFalse(assessment["proven"])
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("identity mismatch" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_native_repository_binding_mismatch(self):
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.evil_root,
|
||||
expected_slug="Scaled-Tech-Consulting/Gitea-Tools",
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertFalse(ctx["roots_aligned"])
|
||||
self.assertFalse(ctx["canonical_root_assessment"]["proven"])
|
||||
self.assertTrue(any("identity mismatch" in r for r in ctx["canonical_root_assessment"]["reasons"]))
|
||||
|
||||
def test_unpatched_foreign_configured_root_derives_expected_from_process_root_and_blocks(self):
|
||||
"""B8: Production path test where foreign configured root cannot self-authorize."""
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server, "_configured_canonical_root", return_value=(self.evil_root, "env")):
|
||||
expected_slug = mcp_server._resolve_expected_repository_slug("prgs")
|
||||
self.assertEqual(expected_slug, "Scaled-Tech-Consulting/Gitea-Tools")
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.evil_root,
|
||||
source="env",
|
||||
expected_slug=expected_slug,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
)
|
||||
self.assertFalse(assessment["proven"])
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertEqual(assessment["resolved_slug"], "Someone-Else/Evil-Repo")
|
||||
self.assertTrue(any("identity mismatch" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_unpatched_valid_cross_repo_matching_session_context(self):
|
||||
"""B8: Valid cross-repo namespace matches when session context is bound to target repo."""
|
||||
bound_ctx = {"org": "Scaled-Tech-Consulting", "repository": "mcp-control-plane", "remote": "prgs"}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context", return_value=bound_ctx):
|
||||
expected_slug = mcp_server._resolve_expected_repository_slug("prgs")
|
||||
self.assertEqual(expected_slug, "Scaled-Tech-Consulting/mcp-control-plane")
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=expected_slug,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
)
|
||||
self.assertTrue(assessment["proven"])
|
||||
self.assertFalse(assessment["block"])
|
||||
self.assertEqual(assessment["resolved_slug"], "Scaled-Tech-Consulting/mcp-control-plane")
|
||||
|
||||
def test_unprovable_expected_identity_fails_closed(self):
|
||||
"""B8: If expected repository identity is unprovable for a configured root, fail closed."""
|
||||
no_remote_root = os.path.join(self.tmp_dir, "no-remote-process-root")
|
||||
os.makedirs(no_remote_root)
|
||||
subprocess.run(["git", "init", "-b", "master"], cwd=no_remote_root, check=True)
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", no_remote_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context", return_value=None):
|
||||
expected_slug = mcp_server._resolve_expected_repository_slug("prgs")
|
||||
self.assertIsNone(expected_slug)
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=expected_slug,
|
||||
process_project_root=no_remote_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
)
|
||||
self.assertFalse(assessment["proven"])
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("unprovable or missing" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_missing_canonical_root(self):
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="author",
|
||||
worktree_path=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root="",
|
||||
)
|
||||
self.assertEqual(ctx["canonical_repo_root"], self.install_root)
|
||||
self.assertTrue(ctx["roots_aligned"])
|
||||
|
||||
def test_nonexistent_configured_canonical_root(self):
|
||||
nonexistent = os.path.join(self.tmp_dir, "nonexistent-repo")
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=nonexistent,
|
||||
)
|
||||
self.assertFalse(ctx["roots_aligned"])
|
||||
self.assertFalse(ctx["canonical_root_assessment"]["proven"])
|
||||
self.assertTrue(any("does not exist" in r for r in ctx["canonical_root_assessment"]["reasons"]))
|
||||
|
||||
def test_non_git_configured_canonical_root(self):
|
||||
non_git = os.path.join(self.tmp_dir, "non-git-dir")
|
||||
os.makedirs(non_git)
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=non_git,
|
||||
)
|
||||
self.assertFalse(ctx["roots_aligned"])
|
||||
self.assertFalse(ctx["canonical_root_assessment"]["proven"])
|
||||
self.assertTrue(any("not a git repository" in r for r in ctx["canonical_root_assessment"]["reasons"]))
|
||||
|
||||
def test_git_common_directory_membership_matching(self):
|
||||
valid, err = nwb.verify_git_common_directory_membership(
|
||||
self.target_worktree, self.target_root
|
||||
)
|
||||
self.assertTrue(valid, err)
|
||||
self.assertIsNone(err)
|
||||
|
||||
def test_foreign_git_common_directory(self):
|
||||
# Foreign worktree created under install_root
|
||||
install_branches = os.path.join(self.install_root, "branches")
|
||||
foreign_wt = os.path.join(install_branches, "foreign-wt")
|
||||
subprocess.run(["git", "worktree", "add", "-b", "foreign-wt", foreign_wt], cwd=self.install_root, check=True)
|
||||
|
||||
valid, err = nwb.verify_git_common_directory_membership(
|
||||
foreign_wt, self.target_root
|
||||
)
|
||||
self.assertFalse(valid)
|
||||
self.assertIn("does not match", err)
|
||||
|
||||
def test_normalized_path_aliases(self):
|
||||
alias_path = self.target_worktree + "/../rev-pr-99/./"
|
||||
valid, err = nwb.verify_git_common_directory_membership(
|
||||
alias_path, self.target_root
|
||||
)
|
||||
self.assertTrue(valid, err)
|
||||
|
||||
def test_safe_symlink_identity(self):
|
||||
link_path = os.path.join(self.target_branches, "symlink-rev-99")
|
||||
try:
|
||||
os.symlink(self.target_worktree, link_path)
|
||||
valid, err = nwb.verify_git_common_directory_membership(
|
||||
link_path, self.target_root
|
||||
)
|
||||
self.assertTrue(valid, err)
|
||||
finally:
|
||||
if os.path.exists(link_path):
|
||||
os.unlink(link_path)
|
||||
|
||||
def test_symlink_escape_or_foreign_alias(self):
|
||||
outside_dir = os.path.join(self.tmp_dir, "outside-target")
|
||||
os.makedirs(outside_dir)
|
||||
link_escape = os.path.join(self.target_branches, "escape-link")
|
||||
try:
|
||||
os.symlink(outside_dir, link_escape)
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind="reviewer",
|
||||
worktree_path=link_escape,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(assessment["block"])
|
||||
finally:
|
||||
if os.path.exists(link_escape):
|
||||
os.unlink(link_escape)
|
||||
|
||||
def test_reviewer_worktree_registered_and_valid(self):
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertFalse(assessment["block"])
|
||||
|
||||
def test_reviewer_worktree_unregistered_blocks(self):
|
||||
unreg_wt = os.path.join(self.target_branches, "unregistered-reviewer")
|
||||
os.makedirs(unreg_wt)
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind="reviewer",
|
||||
worktree_path=unreg_wt,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("is not registered in git worktree list" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_merger_worktree_registered_and_valid(self):
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind="merger",
|
||||
worktree_path=self.target_worktree,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertFalse(assessment["block"])
|
||||
|
||||
def test_merger_worktree_unregistered_blocks(self):
|
||||
unreg_wt = os.path.join(self.target_branches, "unregistered-merger")
|
||||
os.makedirs(unreg_wt)
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind="merger",
|
||||
worktree_path=unreg_wt,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("is not registered in git worktree list" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_reviewer_or_merger_worktree_outside_branches_blocks(self):
|
||||
outside_wt = os.path.join(self.target_root, "outside_branches_wt")
|
||||
os.makedirs(outside_wt)
|
||||
for r_kind in ("reviewer", "merger"):
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind=r_kind,
|
||||
worktree_path=outside_wt,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("is not under" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_nonexistent_reviewer_or_merger_worktree_blocks(self):
|
||||
nonexistent_wt = os.path.join(self.target_branches, "nonexistent-wt")
|
||||
for r_kind in ("reviewer", "merger"):
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind=r_kind,
|
||||
worktree_path=nonexistent_wt,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("does not exist" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_safe_and_unsafe_mutation_alignment_outcomes(self):
|
||||
# Safe alignment (same repo)
|
||||
report_safe = stable_control_runtime.build_runtime_report(
|
||||
process_root=self.install_root,
|
||||
checkout_branch="master",
|
||||
runtime_head="abcdef123456",
|
||||
active_task_workspace=self.install_root,
|
||||
canonical_repository_root=self.install_root,
|
||||
workspace_roots_aligned=True,
|
||||
)
|
||||
gate_safe = stable_control_runtime.assess_runtime_mutation_gate(report_safe)
|
||||
self.assertFalse(gate_safe["block"])
|
||||
|
||||
# Unsafe alignment
|
||||
report_unsafe = stable_control_runtime.build_runtime_report(
|
||||
process_root=self.install_root,
|
||||
checkout_branch="master",
|
||||
runtime_head="abcdef123456",
|
||||
active_task_workspace=self.target_worktree,
|
||||
canonical_repository_root=self.target_root,
|
||||
workspace_roots_aligned=False,
|
||||
)
|
||||
gate_unsafe = stable_control_runtime.assess_runtime_mutation_gate(report_unsafe)
|
||||
self.assertTrue(gate_unsafe["block"])
|
||||
|
||||
def test_reviewer_lease_lifecycle_production_path(self):
|
||||
"""B5: Automated regression for reviewer lease acquire and release through production path."""
|
||||
mock_whoami = {
|
||||
"authenticated": True,
|
||||
"username": "sysadmin",
|
||||
"remote": "prgs",
|
||||
"profile": {
|
||||
"profile_name": "prgs-reviewer",
|
||||
"role": "reviewer",
|
||||
"role_kind": "reviewer",
|
||||
"allowed_operations": ["gitea.read", "gitea.pr.comment", "gitea.pr.approve", "gitea.pr.request_changes"],
|
||||
"forbidden_operations": [],
|
||||
},
|
||||
}
|
||||
|
||||
def mock_api_request(method, url, auth=None, json_data=None):
|
||||
if method == "GET":
|
||||
return {"number": 99, "head": {"sha": "abc1234"}, "state": "open", "merged": False, "merged_at": None}
|
||||
elif method == "POST":
|
||||
return {"id": 9999, "body": (json_data or {}).get("body", "")}
|
||||
return {}
|
||||
|
||||
with patch.object(mcp_server, "gitea_whoami", return_value=mock_whoami), \
|
||||
patch.object(mcp_server, "get_profile", return_value=mock_whoami["profile"]), \
|
||||
patch.object(mcp_server, "_effective_workspace_role", return_value="reviewer"), \
|
||||
patch.object(mcp_server, "_configured_canonical_root", return_value=(self.target_root, "env")), \
|
||||
patch.object(mcp_server, "_reviewer_session_worktree", return_value=self.target_worktree), \
|
||||
patch.object(mcp_server, "_auth", return_value="token mock-token"), \
|
||||
patch.object(mcp_server, "_fetch_pr_comments", return_value=[]), \
|
||||
patch.object(mcp_server, "api_request", side_effect=mock_api_request), \
|
||||
patch("reviewer_pr_lease.assess_acquire_lease", return_value={"acquire_allowed": True, "reasons": [], "lease_body": "<!-- LEASE -->"}), \
|
||||
patch("reviewer_pr_lease.find_active_reviewer_lease", return_value={"session_id": "sid-123", "reviewer": "sysadmin"}), \
|
||||
patch("reviewer_pr_lease.get_session_lease", return_value={"session_id": "sid-123", "reviewer": "sysadmin"}), \
|
||||
patch("reviewer_pr_lease.clear_session_lease") as mock_clear:
|
||||
|
||||
acq_res = mcp_server.gitea_acquire_reviewer_pr_lease(
|
||||
pr_number=99,
|
||||
remote="prgs",
|
||||
worktree=self.target_worktree,
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(acq_res.get("success"), acq_res)
|
||||
|
||||
rel_res = mcp_server.gitea_release_reviewer_pr_lease(
|
||||
pr_number=99,
|
||||
worktree=self.target_worktree,
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(rel_res.get("success"), rel_res)
|
||||
mock_clear.assert_called_once()
|
||||
|
||||
def test_unbound_session_derives_and_retains_configured_target_repository(self):
|
||||
"""B9: Unbound session with a legitimate configured target root derives and retains target repository."""
|
||||
prof = {
|
||||
"profile_name": "prgs-author",
|
||||
"allowed_repositories": ["Scaled-Tech-Consulting/mcp-control-plane"],
|
||||
"canonical_repository_root": self.target_root,
|
||||
}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context", return_value=None):
|
||||
slug, reasons = mcp_server._canonical_repository_slug(prof, "prgs")
|
||||
self.assertEqual(slug, "Scaled-Tech-Consulting/mcp-control-plane", f"reasons: {reasons}")
|
||||
self.assertEqual(reasons, [])
|
||||
res = mcp_server._trusted_session_repository(prof, "prgs")
|
||||
self.assertEqual(res["org"], "Scaled-Tech-Consulting")
|
||||
self.assertEqual(res["repository"], "mcp-control-plane")
|
||||
|
||||
def test_process_root_a_configured_target_b_retains_b(self):
|
||||
"""B9: Process root A (Gitea-Tools) and configured target B (mcp-control-plane) keep B as expected identity when bound."""
|
||||
bound_ctx = {"org": "Scaled-Tech-Consulting", "repository": "mcp-control-plane", "remote": "prgs"}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context", return_value=bound_ctx):
|
||||
expected = mcp_server._resolve_expected_repository_slug("prgs")
|
||||
self.assertEqual(expected, "Scaled-Tech-Consulting/mcp-control-plane")
|
||||
|
||||
def test_request_supplied_repository_c_cannot_replace_derived_or_bound_repository_b(self):
|
||||
"""B9: Request-supplied repository C (Timesheet) cannot replace bound repository B (mcp-control-plane)."""
|
||||
bound_ctx = {"org": "Scaled-Tech-Consulting", "repository": "mcp-control-plane", "remote": "prgs"}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context", return_value=bound_ctx):
|
||||
expected = mcp_server._resolve_expected_repository_slug("prgs", org="Scaled-Tech-Consulting", repo="Timesheet")
|
||||
self.assertEqual(expected, "Scaled-Tech-Consulting/mcp-control-plane")
|
||||
|
||||
def test_known_expected_repo_a_plus_candidate_root_b_fails_closed(self):
|
||||
"""B9: Known expected repository A plus candidate root B fails closed in validation mode."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug="Scaled-Tech-Consulting/Gitea-Tools",
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="validation",
|
||||
)
|
||||
self.assertFalse(assessment["proven"])
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("identity mismatch" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_validation_mode_with_unprovable_expected_identity_fails_closed(self):
|
||||
"""B9: Validation mode with expected_slug=None and require_binding=True fails closed."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=None,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="validation",
|
||||
)
|
||||
self.assertFalse(assessment["proven"])
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("unprovable or missing" in r for r in assessment["reasons"]))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -245,10 +245,15 @@ class TestNamespaceWorkspaceIntegration(unittest.TestCase):
|
||||
def test_pr487_style_merge_binds_clean_merger_workspace(
|
||||
self, _exists, _isdir, mock_run
|
||||
):
|
||||
mock_run.return_value = MagicMock(returncode=0, stdout=f"{CONTROL_ROOT}/.git\n")
|
||||
def mock_git(cmd, *args, **kwargs):
|
||||
if "rev-parse" in cmd:
|
||||
return MagicMock(returncode=0, stdout=f"{CONTROL_ROOT}/.git\n")
|
||||
return MagicMock(returncode=0, stdout=f"worktree {CONTROL_ROOT}\nworktree {MERGER_CLEAN}\n")
|
||||
mock_run.side_effect = mock_git
|
||||
os.environ[nwb.AUTHOR_WORKTREE_ENV] = AUTHOR_DIRTY
|
||||
os.environ[nwb.MERGER_WORKTREE_ENV] = MERGER_CLEAN
|
||||
srv._preflight_resolved_role = "reviewer"
|
||||
with mock.patch.object(srv, "PROJECT_ROOT", MCP_PROCESS_ROOT):
|
||||
with mock.patch("gitea_mcp_server.get_profile", return_value=self._merger_profile()):
|
||||
resolved = srv._verify_role_mutation_workspace("prgs")
|
||||
self.assertEqual(resolved, os.path.realpath(MCP_PROCESS_ROOT))
|
||||
self.assertEqual(resolved, os.path.realpath(MERGER_CLEAN))
|
||||
@@ -153,6 +153,12 @@ EXPECTED_ROLE_EXCLUSIVE_TASKS = frozenset(
|
||||
"delete_branch",
|
||||
"cleanup_merged_pr_branch",
|
||||
"reconciliation_cleanup",
|
||||
# #970 review 644 B2: retiring a missing worktree binding is a
|
||||
# control-plane cleanup mutation, so it carries the same reconciler-only
|
||||
# authority as every other reconciliation cleanup. Permission alone must
|
||||
# not authorize it.
|
||||
"reconcile_missing_worktree_bindings",
|
||||
"gitea_reconcile_missing_worktree_bindings",
|
||||
"work_issue",
|
||||
"work-issue",
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user