Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c0c6d14b73 | ||
|
|
108cbfa173 | ||
|
|
0dbff6dcd5 | ||
|
|
4a1cc63e94 | ||
|
|
6596b259fc | ||
|
|
b0868be6b3 | ||
|
|
1a38ef95e3 | ||
|
|
324a0c8a8d | ||
|
|
34968475c7 | ||
|
|
a6c7d1491e | ||
|
|
9d96cf4cfa | ||
|
|
7978008709 | ||
|
|
6a53308473 | ||
|
|
626be8b178 | ||
|
|
c763161702 | ||
|
|
3f584352df | ||
|
|
956fa15fe3 | ||
|
|
fa510dd28d | ||
|
|
1dd30ecb15 | ||
|
|
8eada1fbe4 |
+136
-21
@@ -37,6 +37,48 @@ CANONICAL_ROOT_ENV = "GITEA_CANONICAL_REPOSITORY_ROOT"
|
||||
# Candidate git remote names probed when deriving repository identity.
|
||||
_IDENTITY_REMOTE_CANDIDATES = ("prgs", "origin", "dadeschools", "mdcps")
|
||||
|
||||
# Repository-authority modes (#973 B10). Exactly two values are supported.
|
||||
# ``mode`` selects how repository authority is established, so an unrecognised
|
||||
# value must never be normalised onto one of these: aliasing a trusted mode is
|
||||
# precisely the defect. Omitting the argument keeps the documented safe default,
|
||||
# ``validation``.
|
||||
MODE_VALIDATION = "validation"
|
||||
MODE_DERIVATION = "derivation"
|
||||
SUPPORTED_MODES: tuple[str, ...] = (MODE_VALIDATION, MODE_DERIVATION)
|
||||
|
||||
# Reason code emitted when an unsupported mode is refused. Mirrors the existing
|
||||
# ``webui.sanctioned_restart.DENY_UNKNOWN_MODE`` convention so callers and tests
|
||||
# can assert the refusal cause rather than string-matching prose.
|
||||
DENY_UNKNOWN_MODE = "unknown_mode"
|
||||
|
||||
|
||||
def unsupported_mode_reason(mode: object) -> str | None:
|
||||
"""Precise rejection reason for *mode*, or None when *mode* is supported.
|
||||
|
||||
Only the two documented string values are accepted, compared exactly — no
|
||||
stripping, no case folding — so misspellings and whitespace variants are
|
||||
refused rather than coerced. An empty string, ``None``, and any non-string
|
||||
are all *explicitly supplied* unsupported values and are refused on the same
|
||||
footing; none of them is normalised to a supported mode. Omitting the
|
||||
argument entirely never reaches here with an unsupported value because the
|
||||
parameter default is ``"validation"``.
|
||||
"""
|
||||
if isinstance(mode, str) and mode in SUPPORTED_MODES:
|
||||
return None
|
||||
supported = ", ".join(repr(m) for m in SUPPORTED_MODES)
|
||||
if not isinstance(mode, str):
|
||||
return (
|
||||
f"unsupported repository-authority mode {mode!r} of type "
|
||||
f"{type(mode).__name__}: only {supported} are supported; the mode "
|
||||
"is refused before any repository assessment and no repository "
|
||||
"identity was resolved through it (fail closed)"
|
||||
)
|
||||
return (
|
||||
f"unsupported repository-authority mode {mode!r}: only {supported} are "
|
||||
"supported; the mode is refused before any repository assessment and no "
|
||||
"repository identity was resolved through it (fail closed)"
|
||||
)
|
||||
|
||||
|
||||
def configured_canonical_root(
|
||||
profile: Mapping | None,
|
||||
@@ -132,6 +174,7 @@ def assess_canonical_repository_root(
|
||||
process_project_root: str,
|
||||
remote: str | None = None,
|
||||
require_binding: bool = False,
|
||||
mode: str = "validation",
|
||||
) -> dict:
|
||||
"""Validate the canonical repository root binding, failing closed on forgery.
|
||||
|
||||
@@ -140,15 +183,41 @@ def assess_canonical_repository_root(
|
||||
``configured`` (whether a cross-repo binding was declared),
|
||||
``resolved_slug`` and ``source``.
|
||||
|
||||
Without a configured binding the single-repo default is preserved: the
|
||||
canonical root is derived from *process_project_root* and never blocks
|
||||
(unless *require_binding* explicitly demands one).
|
||||
*mode* accepts exactly the two values in :data:`SUPPORTED_MODES`:
|
||||
- ``"validation"`` (default): an independently trusted expected repository
|
||||
slug is known (or required). Candidate root's observed identity must match.
|
||||
Unprovable or missing expected identity fails closed when *require_binding* is True.
|
||||
- ``"derivation"``: caller is deriving the canonical repository identity.
|
||||
No expected slug exists yet by design. Derivation succeeds if the configured
|
||||
root exists, is a git repository, and carries a resolvable git remote identity.
|
||||
|
||||
With a configured binding the path must exist, be a git repository, and —
|
||||
when *expected_slug* is known — carry a matching repository identity. A
|
||||
mismatched or (when *require_binding*) unprovable identity is a forged or
|
||||
conflicting binding and fails closed.
|
||||
Every other explicitly supplied value — unknown strings, misspellings, the
|
||||
empty string, ``None``, and non-strings — is refused with ``proven`` False,
|
||||
``block`` True, and ``reason_code`` :data:`DENY_UNKNOWN_MODE` (#973 B10).
|
||||
Omitting *mode* entirely keeps the documented ``"validation"`` default.
|
||||
"""
|
||||
# #973 B10: refuse an unsupported repository-authority mode as the very first
|
||||
# act, before any candidate-root existence check, path or symlink resolution,
|
||||
# git top-level discovery, remote-URL or repository-identity discovery, and
|
||||
# before any expected-versus-observed comparison or validation/derivation
|
||||
# behaviour. An unsupported mode previously fell into the catch-all ``else``
|
||||
# on the configured-root path (silently receiving validation semantics) and
|
||||
# skipped the identity comparison entirely on the single-repository default
|
||||
# path (strictly weaker than validation), so matching identities could return
|
||||
# ``proven`` True. No repository identity may be resolved through a mode the
|
||||
# contract does not define.
|
||||
mode_reason = unsupported_mode_reason(mode)
|
||||
if mode_reason is not None:
|
||||
return _assessment(
|
||||
proven=False,
|
||||
reasons=[mode_reason],
|
||||
configured=bool((configured_value or "").strip()),
|
||||
canonical_repo_root=None,
|
||||
resolved_slug=None,
|
||||
source=source,
|
||||
reason_code=DENY_UNKNOWN_MODE,
|
||||
)
|
||||
|
||||
process_root = os.path.realpath(process_project_root)
|
||||
declared = (configured_value or "").strip()
|
||||
|
||||
@@ -168,12 +237,31 @@ def assess_canonical_repository_root(
|
||||
)
|
||||
# Single-repo default: canonical root follows the install checkout.
|
||||
derived = resolve_repo_toplevel(process_root) or process_root
|
||||
resolved_slug = repository_identity_slug(derived, remote=remote)
|
||||
reasons: list[str] = []
|
||||
# #973 B10: the allowlist above guarantees *mode* is one of the two
|
||||
# supported values here, so an unsupported value can no longer skip this
|
||||
# identity comparison and end up strictly weaker than validation.
|
||||
if mode == MODE_VALIDATION and expected_slug:
|
||||
expected = expected_slug.strip()
|
||||
if resolved_slug and resolved_slug.lower() != expected.lower():
|
||||
reasons.append(
|
||||
f"canonical repository root identity mismatch: '{derived}' resolves "
|
||||
f"to repository '{resolved_slug}' but expected repository identity "
|
||||
f"is '{expected}' (forged or conflicting binding, fail closed)"
|
||||
)
|
||||
elif not resolved_slug and require_binding:
|
||||
reasons.append(
|
||||
f"canonical repository root '{derived}' has no resolvable git "
|
||||
f"remote identity to confirm authorization for '{expected}' "
|
||||
"(fail closed)"
|
||||
)
|
||||
return _assessment(
|
||||
proven=True,
|
||||
reasons=[],
|
||||
proven=not reasons,
|
||||
reasons=reasons,
|
||||
configured=False,
|
||||
canonical_repo_root=derived,
|
||||
resolved_slug=None,
|
||||
resolved_slug=resolved_slug,
|
||||
source=None,
|
||||
)
|
||||
|
||||
@@ -207,19 +295,36 @@ def assess_canonical_repository_root(
|
||||
|
||||
resolved_slug = repository_identity_slug(toplevel, remote=remote)
|
||||
reasons: list[str] = []
|
||||
expected = (expected_slug or "").strip() or None
|
||||
if expected:
|
||||
if resolved_slug and resolved_slug.lower() != expected.lower():
|
||||
|
||||
if mode == MODE_DERIVATION:
|
||||
if not resolved_slug:
|
||||
reasons.append(
|
||||
f"canonical repository root identity mismatch: '{toplevel}' resolves "
|
||||
f"to repository '{resolved_slug}' but the session is authorized for "
|
||||
f"'{expected}' (forged or conflicting binding, fail closed)"
|
||||
f"configured canonical repository root '{toplevel}' has no resolvable "
|
||||
"git remote identity (fail closed)"
|
||||
)
|
||||
elif not resolved_slug and require_binding:
|
||||
else:
|
||||
# #973 B10: MODE_VALIDATION only. This arm is no longer a catch-all — the
|
||||
# allowlist above admits no third value, so an unsupported mode can no
|
||||
# longer silently receive validation semantics here.
|
||||
expected = (expected_slug or "").strip() or None
|
||||
if expected:
|
||||
if resolved_slug and resolved_slug.lower() != expected.lower():
|
||||
reasons.append(
|
||||
f"canonical repository root identity mismatch: '{toplevel}' resolves "
|
||||
f"to repository '{resolved_slug}' but the session is authorized for "
|
||||
f"'{expected}' (forged or conflicting binding, fail closed)"
|
||||
)
|
||||
elif not resolved_slug and require_binding:
|
||||
reasons.append(
|
||||
f"canonical repository root '{toplevel}' has no resolvable git "
|
||||
f"remote identity to confirm authorization for '{expected}' "
|
||||
"(fail closed)"
|
||||
)
|
||||
elif require_binding:
|
||||
reasons.append(
|
||||
f"canonical repository root '{toplevel}' has no resolvable git "
|
||||
f"remote identity to confirm authorization for '{expected}' "
|
||||
"(fail closed)"
|
||||
f"canonical repository root '{toplevel}' has configured value "
|
||||
f"'{configured_value}' but authoritative expected repository identity "
|
||||
"is unprovable or missing (fail closed)"
|
||||
)
|
||||
|
||||
return _assessment(
|
||||
@@ -252,10 +357,19 @@ def _assessment(
|
||||
proven: bool,
|
||||
reasons: list[str],
|
||||
configured: bool,
|
||||
canonical_repo_root: str,
|
||||
canonical_repo_root: str | None,
|
||||
resolved_slug: str | None,
|
||||
source: str | None,
|
||||
reason_code: str | None = None,
|
||||
) -> dict:
|
||||
"""Build the assessment payload.
|
||||
|
||||
``canonical_repo_root`` is None only when the assessment refused to resolve
|
||||
one at all — today exactly the unsupported-mode refusal (#973 B10), which
|
||||
must not derive a trusted repository identity through an undefined mode.
|
||||
``reason_code`` is a machine-checkable refusal cause; None for every
|
||||
ordinary (non-coded) outcome.
|
||||
"""
|
||||
return {
|
||||
"proven": proven,
|
||||
"block": not proven,
|
||||
@@ -264,4 +378,5 @@ def _assessment(
|
||||
"canonical_repo_root": canonical_repo_root,
|
||||
"resolved_slug": resolved_slug,
|
||||
"source": source,
|
||||
"reason_code": reason_code,
|
||||
}
|
||||
|
||||
@@ -280,6 +280,22 @@ def _ts(dt: datetime | None = None) -> str:
|
||||
return value.astimezone(timezone.utc).replace(microsecond=0).isoformat().replace("+00:00", "Z")
|
||||
|
||||
|
||||
def _realpath_or_raw(value: str | None) -> str:
|
||||
"""Normalize a filesystem path for compare-and-swap equality (#970).
|
||||
|
||||
Symlinks and ``..`` segments must not make two spellings of the same path
|
||||
look different, but an unresolvable path must still compare as itself
|
||||
rather than collapsing to empty — an empty result means "no path given".
|
||||
"""
|
||||
text = (value or "").strip()
|
||||
if not text:
|
||||
return ""
|
||||
try:
|
||||
return os.path.realpath(os.path.abspath(text))
|
||||
except Exception:
|
||||
return text
|
||||
|
||||
|
||||
def _parse_ts(value: str | None) -> datetime | None:
|
||||
if not value:
|
||||
return None
|
||||
@@ -1913,6 +1929,168 @@ class ControlPlaneDB:
|
||||
),
|
||||
)
|
||||
|
||||
def retire_lease_worktree_path(
|
||||
self,
|
||||
lease_id: str,
|
||||
*,
|
||||
expected_path: str | None = None,
|
||||
expected_status: str | None = None,
|
||||
expected_session_id: str | None = None,
|
||||
expected_owner_pid: int | None = None,
|
||||
reason: str = "missing_worktree_path_retired",
|
||||
) -> dict[str, Any]:
|
||||
"""Retire a missing worktree_path binding from a control-plane lease (#970).
|
||||
|
||||
Clears worktree_path on the lease row, updates provenance_json with
|
||||
durable retirement audit proof, and writes a worktree_binding_retired
|
||||
event.
|
||||
|
||||
The update is a compare-and-swap (#970 review 644 B1/B3): the caller
|
||||
states the exact path it audited and, when known, the lease status,
|
||||
owning session, and owner pid it classified against. Every stated value
|
||||
must still match the stored row, and the ``UPDATE`` itself is keyed on
|
||||
the stored ``worktree_path``, so a concurrent writer that moved or
|
||||
replaced the binding between audit and apply loses the race instead of
|
||||
having its value silently overwritten. A mismatch raises and mutates
|
||||
nothing.
|
||||
|
||||
``expected_path`` is mandatory: a retirement that does not name the path
|
||||
it intends to clear cannot be safe against concurrent recreation.
|
||||
"""
|
||||
now_s = _ts()
|
||||
expected_norm = _realpath_or_raw(expected_path)
|
||||
if not expected_norm:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire lease {lease_id} worktree_path: expected_path is "
|
||||
"required for compare-and-swap retirement (fail closed)"
|
||||
)
|
||||
with self._tx() as conn:
|
||||
cols = self._lease_columns(conn)
|
||||
row = conn.execute(
|
||||
"SELECT * FROM leases WHERE lease_id = ?",
|
||||
(lease_id,),
|
||||
).fetchone()
|
||||
if not row:
|
||||
raise ControlPlaneError(f"unknown lease_id {lease_id}")
|
||||
|
||||
record = dict(row)
|
||||
current_wt = (record.get("worktree_path") or "").strip()
|
||||
|
||||
if not current_wt:
|
||||
# Idempotent: the binding this caller audited is already gone.
|
||||
return {
|
||||
"lease_id": lease_id,
|
||||
"retired": False,
|
||||
"already_retired": True,
|
||||
"prior_worktree_path": "",
|
||||
"expected_worktree_path": expected_path,
|
||||
"reason": reason,
|
||||
"compare_and_swap": {
|
||||
"matched": True,
|
||||
"outcome": "already_retired",
|
||||
},
|
||||
}
|
||||
|
||||
if _realpath_or_raw(current_wt) != expected_norm:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire lease {lease_id} worktree_path: expected "
|
||||
f"'{expected_path}' does not match current '{current_wt}' "
|
||||
"(fail closed)"
|
||||
)
|
||||
|
||||
for field, expected_value in (
|
||||
("status", expected_status),
|
||||
("session_id", expected_session_id),
|
||||
):
|
||||
if expected_value is None:
|
||||
continue
|
||||
current_value = record.get(field)
|
||||
if str(current_value or "").strip() != str(expected_value).strip():
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire lease {lease_id} worktree_path: lease "
|
||||
f"{field} changed since audit (expected "
|
||||
f"'{expected_value}', found '{current_value}'); fail closed"
|
||||
)
|
||||
|
||||
if expected_owner_pid is not None:
|
||||
current_pid = record.get("owner_pid")
|
||||
if current_pid is not None and int(current_pid) != int(expected_owner_pid):
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire lease {lease_id} worktree_path: lease "
|
||||
f"owner_pid changed since audit (expected "
|
||||
f"{expected_owner_pid}, found {current_pid}); fail closed"
|
||||
)
|
||||
|
||||
# Parse and update provenance_json
|
||||
raw_prov = record.get("provenance_json") or "{}"
|
||||
try:
|
||||
prov = json.loads(raw_prov) if isinstance(raw_prov, str) else dict(raw_prov)
|
||||
except Exception:
|
||||
prov = {}
|
||||
if not isinstance(prov, dict):
|
||||
prov = {}
|
||||
|
||||
prior_path = current_wt
|
||||
prov.update({
|
||||
"worktree_path_retired": True,
|
||||
"retired_worktree_path": prior_path,
|
||||
"retired_at": now_s,
|
||||
"retirement_reason": reason,
|
||||
"retired_from_status": record.get("status"),
|
||||
"retired_from_session_id": record.get("session_id"),
|
||||
"worktree_path": "",
|
||||
})
|
||||
prov_json = json.dumps(prov)
|
||||
|
||||
if "worktree_path" in cols:
|
||||
# CAS: keyed on the exact stored path this caller audited.
|
||||
cur = conn.execute(
|
||||
"UPDATE leases SET worktree_path = '', provenance_json = ? "
|
||||
"WHERE lease_id = ? AND worktree_path = ?",
|
||||
(prov_json, lease_id, record.get("worktree_path")),
|
||||
)
|
||||
if cur.rowcount != 1:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire lease {lease_id} worktree_path: "
|
||||
"compare-and-swap matched no row (concurrent change); "
|
||||
"fail closed"
|
||||
)
|
||||
else:
|
||||
conn.execute(
|
||||
"UPDATE leases SET provenance_json = ? WHERE lease_id = ?",
|
||||
(prov_json, lease_id),
|
||||
)
|
||||
|
||||
conn.execute(
|
||||
"""
|
||||
INSERT INTO events(work_item_id, event_type, message, created_at)
|
||||
VALUES (?, 'worktree_binding_retired', ?, ?)
|
||||
""",
|
||||
(
|
||||
record["work_item_id"],
|
||||
f"lease {lease_id} worktree_path '{prior_path}' retired: {reason}",
|
||||
now_s,
|
||||
),
|
||||
)
|
||||
|
||||
return {
|
||||
"lease_id": lease_id,
|
||||
"retired": True,
|
||||
"already_retired": False,
|
||||
"prior_worktree_path": prior_path,
|
||||
"expected_worktree_path": expected_path,
|
||||
"retired_at": now_s,
|
||||
"reason": reason,
|
||||
"compare_and_swap": {
|
||||
"matched": True,
|
||||
"outcome": "retired",
|
||||
"expected_status": expected_status,
|
||||
"expected_session_id": expected_session_id,
|
||||
"expected_owner_pid": expected_owner_pid,
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def abandon_lease(
|
||||
self,
|
||||
*,
|
||||
@@ -3051,3 +3229,123 @@ class ControlPlaneDB:
|
||||
"live_lease_id": None if live_lease_id is None else str(live_lease_id),
|
||||
"reconcile_action": "reconcile_required" if stale else "safe_to_resume",
|
||||
}
|
||||
|
||||
def retire_session_checkpoint_worktree_path(
|
||||
self,
|
||||
session_id: str,
|
||||
*,
|
||||
checkpoint_id: str | None = None,
|
||||
expected_path: str | None = None,
|
||||
expected_status: str | None = None,
|
||||
reason: str = "missing_worktree_path_retired",
|
||||
) -> dict[str, Any]:
|
||||
"""Retire a missing worktree_path from session_checkpoints (#970).
|
||||
|
||||
Compare-and-swap, mirroring :meth:`retire_lease_worktree_path` (#970
|
||||
review 644 B3). ``expected_path`` names the exact stored path the caller
|
||||
audited; the guarded ``UPDATE`` is keyed on that stored value, so a
|
||||
checkpoint whose path was moved, replaced, or concurrently rewritten
|
||||
after the audit is refused without mutation rather than blindly cleared.
|
||||
|
||||
Exactly one checkpoint row is targeted: by ``checkpoint_id`` when given,
|
||||
otherwise by ``session_id``, which must identify a single row.
|
||||
"""
|
||||
now_s = _ts()
|
||||
expected_norm = _realpath_or_raw(expected_path)
|
||||
if not expected_norm:
|
||||
raise ControlPlaneError(
|
||||
"cannot retire session checkpoint worktree_path: expected_path "
|
||||
"is required for compare-and-swap retirement (fail closed)"
|
||||
)
|
||||
if not checkpoint_id and not (session_id or "").strip():
|
||||
raise ControlPlaneError(
|
||||
"cannot retire session checkpoint worktree_path: checkpoint_id "
|
||||
"or session_id is required (fail closed)"
|
||||
)
|
||||
|
||||
with self._tx() as conn:
|
||||
if checkpoint_id:
|
||||
selector_sql = "SELECT * FROM session_checkpoints WHERE checkpoint_id = ?"
|
||||
selector_params: tuple[Any, ...] = (checkpoint_id,)
|
||||
selector_desc = f"checkpoint_id '{checkpoint_id}'"
|
||||
else:
|
||||
selector_sql = "SELECT * FROM session_checkpoints WHERE session_id = ?"
|
||||
selector_params = (session_id,)
|
||||
selector_desc = f"session_id '{session_id}'"
|
||||
|
||||
rows = [dict(r) for r in conn.execute(selector_sql, selector_params).fetchall()]
|
||||
if not rows:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire session checkpoint worktree_path: no "
|
||||
f"checkpoint matches {selector_desc} (fail closed)"
|
||||
)
|
||||
if len(rows) > 1:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire session checkpoint worktree_path: "
|
||||
f"{selector_desc} matches {len(rows)} checkpoints; supply an "
|
||||
"exact checkpoint_id (fail closed)"
|
||||
)
|
||||
|
||||
record = rows[0]
|
||||
target_checkpoint_id = record.get("checkpoint_id")
|
||||
current_wt = (record.get("worktree_path") or "").strip()
|
||||
|
||||
if not current_wt:
|
||||
# Idempotent: the binding this caller audited is already gone.
|
||||
return {
|
||||
"session_id": session_id,
|
||||
"checkpoint_id": target_checkpoint_id,
|
||||
"retired": False,
|
||||
"already_retired": True,
|
||||
"prior_worktree_path": "",
|
||||
"expected_worktree_path": expected_path,
|
||||
"reason": reason,
|
||||
"compare_and_swap": {
|
||||
"matched": True,
|
||||
"outcome": "already_retired",
|
||||
},
|
||||
}
|
||||
|
||||
if _realpath_or_raw(current_wt) != expected_norm:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire session checkpoint worktree_path for "
|
||||
f"{selector_desc}: expected '{expected_path}' does not match "
|
||||
f"current '{current_wt}' (fail closed)"
|
||||
)
|
||||
|
||||
if expected_status is not None:
|
||||
current_status = record.get("status")
|
||||
if str(current_status or "").strip() != str(expected_status).strip():
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire session checkpoint worktree_path for "
|
||||
f"{selector_desc}: status changed since audit (expected "
|
||||
f"'{expected_status}', found '{current_status}'); fail closed"
|
||||
)
|
||||
|
||||
cur = conn.execute(
|
||||
"UPDATE session_checkpoints SET worktree_path = '', updated_at = ? "
|
||||
"WHERE checkpoint_id = ? AND worktree_path = ?",
|
||||
(now_s, target_checkpoint_id, record.get("worktree_path")),
|
||||
)
|
||||
if cur.rowcount != 1:
|
||||
raise ControlPlaneError(
|
||||
f"cannot retire session checkpoint worktree_path for "
|
||||
f"{selector_desc}: compare-and-swap matched no row "
|
||||
"(concurrent change); fail closed"
|
||||
)
|
||||
|
||||
return {
|
||||
"session_id": session_id,
|
||||
"checkpoint_id": target_checkpoint_id,
|
||||
"retired": True,
|
||||
"already_retired": False,
|
||||
"prior_worktree_path": current_wt,
|
||||
"expected_worktree_path": expected_path,
|
||||
"retired_at": now_s,
|
||||
"reason": reason,
|
||||
"compare_and_swap": {
|
||||
"matched": True,
|
||||
"outcome": "retired",
|
||||
"expected_status": expected_status,
|
||||
},
|
||||
}
|
||||
|
||||
@@ -0,0 +1,203 @@
|
||||
# Instance-level fleet identity and health snapshots
|
||||
|
||||
Issue **#978**. Companion primitives: **#948** (worker ownership), **#975**
|
||||
(heartbeat lifecycle), **#951** (restart receipts).
|
||||
|
||||
## Why this exists
|
||||
|
||||
The multi-instance fleet gate (#963) needs a *native* answer to:
|
||||
|
||||
* which application launches are live;
|
||||
* which of the five namespace workers belong to each launch;
|
||||
* whether two Codex (or Claude, Grok, …) launches are distinct;
|
||||
* whether any live collision is real rather than a shared client type.
|
||||
|
||||
Process tables, PID proximity, configuration files, and direct SQLite access
|
||||
are **not** production evidence. The sanctioned surface is the read-only MCP
|
||||
tool `gitea_snapshot_instance_fleet` on **controller** and **reconciler**
|
||||
namespaces.
|
||||
|
||||
## Identity hierarchy
|
||||
|
||||
| Identity | Scope | Who generates it | Lifetime |
|
||||
| --- | --- | --- | --- |
|
||||
| `client_type` | Application family (`codex`, `claude_code`, `gemini`, `grok`, …) | Launcher sets `GITEA_MCP_CLIENT` | Stable for the product |
|
||||
| `client_instance_id` | One running application launch | **Trusted launcher**, once per launch, as `GITEA_MCP_CLIENT_INSTANCE` | Fresh launch → new ID; reconnect of same launch → same ID; full app restart → new ID |
|
||||
| `fleet_run_id` | Operator-approved enrollment / canary cohort | Operator / controller sets `GITEA_MCP_FLEET_RUN_ID` | Duration of the approved rollout |
|
||||
| `namespace` | `author` \| `reviewer` \| `merger` \| `controller` \| `reconciler` | Profile / MCP server binding | Process lifetime |
|
||||
| `worker_id` / `worker_identity` | One namespace worker process | Runtime registry at first registration | New on worker restart; not reused |
|
||||
| `session_id` | Client session ownership | Launcher `GITEA_MCP_CLIENT_SESSION` or runtime | Session lifetime |
|
||||
| `generation_id` | One daemon launch | Runtime at process boot | New on process restart |
|
||||
| `process_identity` / PID | OS process | Runtime | Process lifetime |
|
||||
|
||||
### Rules
|
||||
|
||||
1. Multiple active instances **may** share the same `client_type`.
|
||||
2. Every application launch receives a **distinct** `client_instance_id`.
|
||||
3. All five namespace workers of one launch report the **same**
|
||||
`client_instance_id`.
|
||||
4. Each namespace worker has a **distinct** `worker_identity`, process
|
||||
identity, generation, and PID.
|
||||
5. Instance identity is **never** inferred from PID proximity, timestamps, or
|
||||
client type alone.
|
||||
6. Live reuse of one `client_instance_id` with conflicting workers fails closed.
|
||||
7. Sharing only a profile or `client_type` is **not** a duplicate.
|
||||
|
||||
This deliberately **replaces** any permanent `exactly_one_per_profile` fleet
|
||||
model (#949 assumption) as the operating rule for multi-instance fleets.
|
||||
|
||||
## How five workers join one instance
|
||||
|
||||
1. The host starts one application instance (for example one Codex session).
|
||||
2. The **production application launcher**
|
||||
(`mcp_application_launcher.build_application_mcp_servers` /
|
||||
`gitea_config.multi_namespace_launcher_entries`) mints exactly one
|
||||
`client_instance_id` via `mcp_fleet_snapshot.generate_client_instance_id`
|
||||
and injects the same env into every namespace worker:
|
||||
|
||||
```text
|
||||
GITEA_MCP_CLIENT=<client_type>
|
||||
GITEA_MCP_CLIENT_INSTANCE=<client_instance_id>
|
||||
GITEA_MCP_INSTANCE_PROVENANCE=trusted_launcher
|
||||
GITEA_MCP_CLIENT_SESSION=<session_id> # optional but recommended
|
||||
GITEA_MCP_FLEET_RUN_ID=<enrollment id> # when on an approved canary
|
||||
GITEA_CLIENT_MANAGED=1
|
||||
GITEA_MCP_PROFILE=<role-profile>
|
||||
```
|
||||
|
||||
3. The MCP client starts the five namespace processes (`gitea-author`,
|
||||
`gitea-reviewer`, `gitea-merger`, `gitea-controller`, `gitea-reconciler`)
|
||||
from that generated `mcpServers` map — each entry carries the **same**
|
||||
instance ID.
|
||||
4. Each worker registers once into the worker registry with its own
|
||||
`worker_identity`, `namespace`, `generation_id`, and PID, but the shared
|
||||
`client_instance_id`. Workers never mint a trusted instance ID themselves.
|
||||
|
||||
Only IDs matching the launcher format `inst-<client>-<timestamp>-<digest>`
|
||||
are trusted. Missing, `legacy-pid-*` / `pid-*` placeholders, and ordinary
|
||||
user-supplied strings fail closed as untrusted and cannot authorize
|
||||
multi-instance fleet mutation safety.
|
||||
|
||||
Worker reconnect (same process, same registration) keeps the instance ID.
|
||||
Worker restart (new process) mints a new worker identity and generation but
|
||||
must still receive the same `GITEA_MCP_CLIENT_INSTANCE` from the parent
|
||||
application if it is the same launch. Full application restart mints a new
|
||||
`client_instance_id` (call the launcher without a prior ID).
|
||||
|
||||
### Mutation gate (instance-aware)
|
||||
|
||||
The capability / runtime diagnostic gate
|
||||
(`_check_mcp_runtimes_diagnostics`) is **instance-aware**:
|
||||
|
||||
* Two legitimate application instances that share a profile (same
|
||||
`GITEA_MCP_PROFILE`) are allowed when each has a distinct trusted
|
||||
`client_instance_id`.
|
||||
* More than one live worker for the same
|
||||
`(client_instance_id, profile/namespace)` fails closed as a duplicate
|
||||
namespace worker.
|
||||
* Multiple processes sharing a profile **without** trusted instance
|
||||
evidence still fail closed (indistinguishable from a duplicate).
|
||||
|
||||
## Approved fleet manifest
|
||||
|
||||
An operator (or controller enrollment step) obtains an approved manifest as a
|
||||
list of expected instances, for example:
|
||||
|
||||
```json
|
||||
{
|
||||
"instances": [
|
||||
{
|
||||
"client_type": "codex",
|
||||
"client_instance_id": "inst-codex-20260730T120000Z-abc123def456",
|
||||
"fleet_run_id": "canary-963-2026-07-30",
|
||||
"namespaces": ["author", "reviewer", "merger", "controller", "reconciler"]
|
||||
},
|
||||
{
|
||||
"client_type": "codex",
|
||||
"client_instance_id": "inst-codex-20260730T120100Z-fed654cba321",
|
||||
"fleet_run_id": "canary-963-2026-07-30",
|
||||
"namespaces": ["author", "reviewer", "merger", "controller", "reconciler"]
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
Pass that list as `expected_manifest` to `gitea_snapshot_instance_fleet`.
|
||||
When a manifest is supplied:
|
||||
|
||||
* instances on the manifest but not live → `missing_expected`;
|
||||
* live instances not on the manifest → `unmanifested`;
|
||||
* both are **active blockers** for fleet-gate safety.
|
||||
|
||||
Without a manifest, the snapshot still enumerates the live fleet and classifies
|
||||
identity collisions; it does not invent enrollment policy.
|
||||
|
||||
## Snapshot consistency
|
||||
|
||||
Each snapshot includes:
|
||||
|
||||
* `snapshot_at` — UTC timestamp;
|
||||
* `consistency_token` / `registry_revision` — content digest over worker
|
||||
identity, instance id, status, heartbeat, fencing, and generation.
|
||||
|
||||
Two successive snapshots can prove heartbeat continuity via
|
||||
`mcp_fleet_snapshot.compare_snapshot_heartbeats`.
|
||||
|
||||
## Classification (active vs historical)
|
||||
|
||||
**Active blockers** (make `live_fleet_safe=false`):
|
||||
|
||||
* missing expected instance;
|
||||
* unmanifested extra instance;
|
||||
* duplicate namespace worker within one instance;
|
||||
* live `client_instance_id` collision;
|
||||
* reused worker / session / generation / process / PID / fencing identity;
|
||||
* orphaned or unowned workers;
|
||||
* unknown client;
|
||||
* foreign-repository workers;
|
||||
* old-revision workers;
|
||||
* stale workers still marked active;
|
||||
* legacy incomplete instance identity.
|
||||
|
||||
**Historical dead rows** (`status` released/superseded) are reported as
|
||||
`historical_dead` findings with `active_blocker=false`. They never
|
||||
automatically make the live fleet unsafe.
|
||||
|
||||
## Fail-closed behaviour and recovery
|
||||
|
||||
| Condition | Mutation safety | Diagnostic reads | Recovery |
|
||||
| --- | --- | --- | --- |
|
||||
| Two instances, same `client_type`, distinct IDs | Safe (if otherwise healthy) | Available | None needed |
|
||||
| Live reuse of one `client_instance_id` | Unsafe | Available | Stop the colliding launch or re-issue a distinct ID |
|
||||
| Two workers, same namespace, one instance | Unsafe | Available | Stop the extra worker |
|
||||
| Legacy registration without trusted instance ID | Unsafe for fleet mutation | Available | Relaunch with launcher-issued `GITEA_MCP_CLIENT_INSTANCE` |
|
||||
| Historical dead row only | Does not block alone | Available | No action required for fleet safety |
|
||||
| Registry unavailable | Snapshot fails closed | N/A | Repair registry path / reconnect namespaces |
|
||||
|
||||
Incomplete or untrusted instance identity **cannot** authorize unsafe
|
||||
mutation. Diagnostic reads remain available where `gitea.read` allows.
|
||||
|
||||
## Permissions
|
||||
|
||||
* **Allowed:** `controller`, `reconciler` with `gitea.read`.
|
||||
* **Denied:** author, reviewer, merger (even with `gitea.read` for other tools).
|
||||
* **No new mutation permissions** are granted to any role.
|
||||
|
||||
## Related surfaces
|
||||
|
||||
* `mcp_worker_identity` — registry, worker identity, heartbeats (#948, #975).
|
||||
* `mcp_fleet_snapshot` — pure snapshot + classification (#978).
|
||||
* `gitea_snapshot_instance_fleet` — sanctioned MCP tool (#978).
|
||||
* `gitea_get_runtime_context` — single-process view (not fleet-wide).
|
||||
* [stale-worker-retirement.md](stale-worker-retirement.md) — CAS-protected
|
||||
retirement of conclusively stale registrations (#980). Note that the
|
||||
`registry_revision` this snapshot returns is **time-seeded** and unsuitable
|
||||
for compare-and-swap; retirement derives its own stable
|
||||
`registry_fingerprint` from registry content alone.
|
||||
|
||||
## Non-goals
|
||||
|
||||
* Starting the #963 canary.
|
||||
* Purging historical registry rows.
|
||||
* Preserving “exactly one process per profile” as the permanent model.
|
||||
* Using shell / process-table / SQLite inspection as production fleet evidence.
|
||||
@@ -51,6 +51,7 @@ that gates each call, not which tools exist.
|
||||
- `gitea_adopt_merger_pr_lease`
|
||||
- `gitea_adopt_workflow_lease`
|
||||
- `gitea_allocate_next_work`
|
||||
- `gitea_apply_stale_worker_retirement`
|
||||
- `gitea_assess_already_landed_reconciliation`
|
||||
- `gitea_assess_conflict_fix_classification`
|
||||
- `gitea_assess_conflict_fix_push`
|
||||
@@ -65,6 +66,7 @@ that gates each call, not which tools exist.
|
||||
- `gitea_assess_work_issue_duplicate`
|
||||
- `gitea_assess_worktree_cleanup_integrity`
|
||||
- `gitea_audit_config`
|
||||
- `gitea_audit_missing_worktree_bindings`
|
||||
- `gitea_audit_runtime_recovery_contamination`
|
||||
- `gitea_audit_stable_branch_contamination`
|
||||
- `gitea_audit_worktree_cleanup`
|
||||
@@ -123,19 +125,24 @@ that gates each call, not which tools exist.
|
||||
- `gitea_observability_link_issue`
|
||||
- `gitea_observability_list_projects`
|
||||
- `gitea_observability_reconcile_incident`
|
||||
- `gitea_plan_stale_worker_retirement`
|
||||
- `gitea_post_heartbeat`
|
||||
- `gitea_publish_unpublished_issue_branch`
|
||||
- `gitea_quarantine_contaminated_review`
|
||||
- `gitea_rebind_dirty_same_claimant_author_session`
|
||||
- `gitea_reclaim_expired_workflow_lease`
|
||||
- `gitea_reconcile_after_restart`
|
||||
- `gitea_reconcile_already_landed_pr`
|
||||
- `gitea_reconcile_issue_claims`
|
||||
- `gitea_reconcile_merged_cleanups`
|
||||
- `gitea_reconcile_missing_worktree_bindings`
|
||||
- `gitea_reconcile_superseded_by_merged_pr`
|
||||
- `gitea_record_daemon_process_kill_attempt`
|
||||
- `gitea_record_irrecoverable_decision_lock_provenance`
|
||||
- `gitea_record_pre_review_command`
|
||||
- `gitea_record_shell_spawn_outcome`
|
||||
- `gitea_record_stable_branch_push_attempt`
|
||||
- `gitea_recover_dirty_orphaned_issue_worktree`
|
||||
- `gitea_recover_incomplete_bootstrap_lock`
|
||||
- `gitea_release_merger_pr_lease`
|
||||
- `gitea_release_reviewer_pr_lease`
|
||||
@@ -154,6 +161,7 @@ that gates each call, not which tools exist.
|
||||
- `gitea_sentry_reconcile_issue`
|
||||
- `gitea_sentry_watchdog`
|
||||
- `gitea_set_issue_labels`
|
||||
- `gitea_snapshot_instance_fleet`
|
||||
- `gitea_submit_pr_review`
|
||||
- `gitea_update_pr_branch_by_merge`
|
||||
- `gitea_validate_review_final_report`
|
||||
|
||||
@@ -8,10 +8,10 @@
|
||||
"document. #930's inventory had no such guard and its gitea_mcp_server.py",
|
||||
"anchors drifted between 7bf4f125 and aad5c8b4."
|
||||
],
|
||||
"generated_against_commit": "ca5f078d8a575ea3e2991771f8b4ea85e3dcaaa0",
|
||||
"generated_against_commit": "1dd30ecb1508b559868c2d5d94367bc055d5138e",
|
||||
"anchors": [
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:24864",
|
||||
"anchor": "gitea_mcp_server.py:25087",
|
||||
"expect": "mcp_daemon_guard.bind_native_mcp_transport()"
|
||||
},
|
||||
{
|
||||
@@ -27,11 +27,11 @@
|
||||
"expect": "def assess_transport_for_auth_mint"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:9191",
|
||||
"anchor": "gitea_mcp_server.py:9192",
|
||||
"expect": "assess_transport_for_auth_mint()"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:9440",
|
||||
"anchor": "gitea_mcp_server.py:9441",
|
||||
"expect": "assess_transport_for_auth_mint()"
|
||||
},
|
||||
{
|
||||
@@ -39,31 +39,31 @@
|
||||
"expect": "The transport is selected by deployment configuration"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:15481",
|
||||
"anchor": "gitea_mcp_server.py:15630",
|
||||
"expect": "def _is_client_managed_process"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:15511",
|
||||
"anchor": "gitea_mcp_server.py:15644",
|
||||
"expect": "def _provenance_mutation_block"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:15519",
|
||||
"anchor": "gitea_mcp_server.py:15652",
|
||||
"expect": "unsupported_manual_launch"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:19070",
|
||||
"anchor": "gitea_mcp_server.py:19217",
|
||||
"expect": "server_provenance"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:21581",
|
||||
"anchor": "gitea_mcp_server.py:21741",
|
||||
"expect": "def _check_mcp_runtimes_diagnostics"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:21601",
|
||||
"anchor": "gitea_mcp_server.py:21761",
|
||||
"expect": "\"ps\", \"-o\", \"pid,lstart,command\""
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:21645",
|
||||
"anchor": "gitea_mcp_server.py:21805",
|
||||
"expect": "\"ps\", \"eww\""
|
||||
},
|
||||
{
|
||||
@@ -107,19 +107,19 @@
|
||||
"expect": "def assert_keychain_access_allowed"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:19327",
|
||||
"anchor": "gitea_mcp_server.py:19487",
|
||||
"expect": "def gitea_list_profiles"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:19378",
|
||||
"anchor": "gitea_mcp_server.py:19538",
|
||||
"expect": "gitea_config.resolve_token(p)"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:19691",
|
||||
"anchor": "gitea_mcp_server.py:19851",
|
||||
"expect": "def gitea_audit_config"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:19713",
|
||||
"anchor": "gitea_mcp_server.py:19873",
|
||||
"expect": "service_summaries(config)"
|
||||
},
|
||||
{
|
||||
@@ -135,19 +135,19 @@
|
||||
"expect": "_keychain_token(auth.get(\"id\"))"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:17776",
|
||||
"anchor": "gitea_mcp_server.py:17909",
|
||||
"expect": "\"jenkins-mcp\""
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:17782",
|
||||
"anchor": "gitea_mcp_server.py:17915",
|
||||
"expect": "external-mcp"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:17803",
|
||||
"anchor": "gitea_mcp_server.py:17936",
|
||||
"expect": "\"glitchtip-mcp\""
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:17808",
|
||||
"anchor": "gitea_mcp_server.py:17941",
|
||||
"expect": "external-mcp"
|
||||
},
|
||||
{
|
||||
@@ -183,7 +183,7 @@
|
||||
"expect": "mutation_safe"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:19171",
|
||||
"anchor": "gitea_mcp_server.py:19331",
|
||||
"expect": "def gitea_assess_master_parity"
|
||||
},
|
||||
{
|
||||
@@ -195,11 +195,11 @@
|
||||
"expect": "AUTHOR_WORKTREE_ENV"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:2351",
|
||||
"anchor": "gitea_mcp_server.py:2352",
|
||||
"expect": "/tmp/gitea_issue_lock.json"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:10956",
|
||||
"anchor": "gitea_mcp_server.py:10957",
|
||||
"expect": "def gitea_bootstrap_author_issue_worktree"
|
||||
},
|
||||
{
|
||||
@@ -239,7 +239,7 @@
|
||||
"expect": "os.getpid()"
|
||||
},
|
||||
{
|
||||
"anchor": "gitea_mcp_server.py:12870",
|
||||
"anchor": "gitea_mcp_server.py:12871",
|
||||
"expect": "owner_pid_alive"
|
||||
}
|
||||
]
|
||||
|
||||
@@ -5,7 +5,7 @@ What the adversary is, what each boundary protects, and which services may share
|
||||
- **Issue:** #956 (Remote-MCP threat model), child of epic #929, cross-linked to #955.
|
||||
- **Depends on:** #930 (closed) — `docs/remote-mcp/coupling-inventory.md`.
|
||||
- **Blocks:** #932, #933, #934, #938.
|
||||
- **Generated against commit:** `ca5f078d8a575ea3e2991771f8b4ea85e3dcaaa0` (#708's
|
||||
- **Generated against commit:** `1dd30ecb1508b559868c2d5d94367bc055d5138e` (#708's
|
||||
namespace-attachment gate). Originally generated against
|
||||
`aad5c8b42361d380a8eeb07b94b90815e594c2c5` (`master`), re-anchored at
|
||||
`a143cd065ba06e1a2bdc5143a19ec156e53650ef` when #931's transport bind seam shifted the
|
||||
@@ -31,7 +31,7 @@ document cites an anchor the fixture does not cover.
|
||||
|
||||
This guard exists because #930 did not have one. Its inventory was generated at
|
||||
`7bf4f125`; by `aad5c8b4` its `gitea_mcp_server.py` anchors had drifted — the transport
|
||||
bind it cited at line 23750 now lives at `gitea_mcp_server.py:24864`, and its
|
||||
bind it cited at line 23750 now lives at `gitea_mcp_server.py:25087`, and its
|
||||
client-managed provenance anchor at 14588 now lands in an unrelated function. Nothing
|
||||
failed, because nothing checked. Anchors into a ~24,700-line module rot silently, and a
|
||||
security document that cannot prove its own citations is worse than none, because it is
|
||||
@@ -75,19 +75,19 @@ authenticate the *caller*, not the *intent*.
|
||||
## 3. Trust boundaries
|
||||
|
||||
"Crossing requires today" is what the code actually enforces at
|
||||
`ca5f078d8a575ea3e2991771f8b4ea85e3dcaaa0`, not what the design intends.
|
||||
`1dd30ecb1508b559868c2d5d94367bc055d5138e`, not what the design intends.
|
||||
|
||||
| ID | Boundary | Protects | Crossing requires today | Crossing must require remotely |
|
||||
| -- | -------- | -------- | ----------------------- | ------------------------------ |
|
||||
| B1 | LLM client ↔ MCP server session | A1, A3, A10 — that a mutating session was established through the sanctioned client path | A single configured bind (`gitea_mcp_server.py:24864`) validated against one closed allowlist (`mcp_daemon_guard.py:49`, `mcp_daemon_guard.py:195`) — since #931 the identifier comes from deployment configuration and defaults to the local transport, so the boundary no longer rests on a literal, but it still rests on the *bind* rather than on an authenticated caller; client-managed provenance (`gitea_mcp_server.py:15481`) or a refusal (`gitea_mcp_server.py:15519`); production transport before recovery-authorization mint (`irrecoverable_provenance.py:497`, consumed at `gitea_mcp_server.py:9191` and `gitea_mcp_server.py:9440`) | An authenticated handshake issuing a server-side session identity bound to a principal, with the transport recorded in provenance. The physical proof (a pipe) must become a cryptographic one. |
|
||||
| B1 | LLM client ↔ MCP server session | A1, A3, A10 — that a mutating session was established through the sanctioned client path | A single configured bind (`gitea_mcp_server.py:25087`) validated against one closed allowlist (`mcp_daemon_guard.py:49`, `mcp_daemon_guard.py:195`) — since #931 the identifier comes from deployment configuration and defaults to the local transport, so the boundary no longer rests on a literal, but it still rests on the *bind* rather than on an authenticated caller; client-managed provenance (`gitea_mcp_server.py:15630`) or a refusal (`gitea_mcp_server.py:15652`); production transport before recovery-authorization mint (`irrecoverable_provenance.py:497`, consumed at `gitea_mcp_server.py:9192` and `gitea_mcp_server.py:9441`) | An authenticated handshake issuing a server-side session identity bound to a principal, with the transport recorded in provenance. The physical proof (a pipe) must become a cryptographic one. |
|
||||
| B2 | Role ↔ role | A9 — that author, reviewer, merger, and reconciler are distinct authorities | **The process boundary only.** The role is a property of the process, read once from `GITEA_MCP_PROFILE` (`gitea_config.py:54`). A caller gets author permissions by connecting to the author process. Review and merge are the operations singled out for extra care (`gitea_config.py:97`) | A per-request principal, so the role follows from the credential presented and cannot be selected by reaching a different endpoint. |
|
||||
| B3 | MCP server ↔ credential store | A3, A8 — that only sanctioned code turns a profile into a token | `_keychain_token` shelling out to the login keychain (`gitea_config.py:956`), dispatched by `resolve_token` (`gitea_config.py:974`) with the reference type built at `gitea_config.py:1015`, gated by `assert_keychain_access_allowed` (`mcp_daemon_guard.py:583`). Inline secrets are rejected at config load (`gitea_config.py:294`) | A credential provider keyed by the *request* principal, returning only that principal's credential, with the source recorded and the value never returned. |
|
||||
| B4 | MCP server ↔ Gitea | A1, A2 — that only authorized calls reach the forge | A bearer token over TLS. Server-side, nothing distinguishes one role's token from another beyond the account it belongs to | Unchanged at the forge; the endpoint in front of it must refuse unauthenticated and plaintext connections before tool dispatch. |
|
||||
| B5 | MCP server ↔ caller's filesystem | A7 — that a tool acts on the *caller's* disk or refuses | Nothing. The server's disk *is* the caller's disk. Worktree bootstrap writes directly (`gitea_mcp_server.py:10956`); the active workspace is process-global (`gitea_mcp_server.py:193`, `gitea_mcp_server.py:194`) | An explicit per-tool classification, enforced at dispatch, refusing filesystem tools over a transport that cannot reach the caller's disk. A green verdict about the wrong disk is the failure to prevent. |
|
||||
| B6 | MCP server ↔ coordination state | A6, A9 — mutual exclusion | Local files and a local SQLite database, with liveness judged from the local process table (`issue_lock_store.py:98`), keyed on paths under one user's home (`issue_lock_store.py:26`, `mcp_session_state.py:27`, `control_plane_db.py:47`) and on `os.getpid()` (`control_plane_db.py:1145`, `gitea_mcp_server.py:12870`). A legacy global slot still exists at `gitea_mcp_server.py:2351`, and the session-pointer file is named per PID (`issue_lock_store.py:83`) | One authority per ownership question, with liveness from session identity and expiry, and atomic acquire, renew, and release across hosts. |
|
||||
| B5 | MCP server ↔ caller's filesystem | A7 — that a tool acts on the *caller's* disk or refuses | Nothing. The server's disk *is* the caller's disk. Worktree bootstrap writes directly (`gitea_mcp_server.py:10957`); the active workspace is process-global (`gitea_mcp_server.py:193`, `gitea_mcp_server.py:194`) | An explicit per-tool classification, enforced at dispatch, refusing filesystem tools over a transport that cannot reach the caller's disk. A green verdict about the wrong disk is the failure to prevent. |
|
||||
| B6 | MCP server ↔ coordination state | A6, A9 — mutual exclusion | Local files and a local SQLite database, with liveness judged from the local process table (`issue_lock_store.py:98`), keyed on paths under one user's home (`issue_lock_store.py:26`, `mcp_session_state.py:27`, `control_plane_db.py:47`) and on `os.getpid()` (`control_plane_db.py:1145`, `gitea_mcp_server.py:12871`). A legacy global slot still exists at `gitea_mcp_server.py:2352`, and the session-pointer file is named per PID (`issue_lock_store.py:83`) | One authority per ownership question, with liveness from session identity and expiry, and atomic acquire, renew, and release across hosts. |
|
||||
| B7 | Gitea integration ↔ unrelated integrations | A4, A5 — that a Gitea compromise is not a CI and observability compromise | **Nothing.** See §5. The Gitea server reads Jenkins and GlitchTip secrets (`gitea_config.py:851`, reached from `gitea_config.py:837`) and holds the Sentry token (`sentry_incident_bridge.py:190`) | A hard process boundary. This is the boundary #956 exists to create. |
|
||||
| B8 | Tenant ↔ tenant (`prgs` / `mdcps` / `local-lab`) | A2 — that one organization's compromise is not another's | Convention. One configuration declares all three contexts; `resolve_service` fails closed on a *disabled* context (`gitea_config.py:704`) but the credentials of enabled ones remain reachable in-process. A per-profile repository scope exists (`gitea_config.py:499`) | Separate deployments, or at minimum per-tenant credential scopes with no process able to resolve both. |
|
||||
| B9 | Deployed code ↔ merged policy | A1, A10 — that the running server enforces the rules that were actually merged | Comparing this process's startup commit against this disk (`master_parity_gate.py:168`), conjoined into a single verdict (`master_parity_gate.py:255`) published by `gitea_mcp_server.py:19171` | Freshness defined against the deployed build identity, with an explicit fail-closed verdict when undeterminable. |
|
||||
| B9 | Deployed code ↔ merged policy | A1, A10 — that the running server enforces the rules that were actually merged | Comparing this process's startup commit against this disk (`master_parity_gate.py:168`), conjoined into a single verdict (`master_parity_gate.py:255`) published by `gitea_mcp_server.py:19331` | Freshness defined against the deployed build identity, with an explicit fail-closed verdict when undeterminable. |
|
||||
|
||||
### What no boundary constrains
|
||||
|
||||
@@ -132,10 +132,10 @@ Two flows deserve attention because neither is obvious from the code:
|
||||
|
||||
1. **The keychain flow fans out.** B3 is drawn once but resolves credentials for *every*
|
||||
configured profile and service, not only the active one. `gitea_list_profiles`
|
||||
(`gitea_mcp_server.py:19327`) reports each profile's credential status by calling
|
||||
`resolve_token` on it (`gitea_mcp_server.py:19378`), and `gitea_audit_config`
|
||||
(`gitea_mcp_server.py:19691`) reports service credential status through
|
||||
`service_summaries` (`gitea_mcp_server.py:19713`).
|
||||
(`gitea_mcp_server.py:19487`) reports each profile's credential status by calling
|
||||
`resolve_token` on it (`gitea_mcp_server.py:19538`), and `gitea_audit_config`
|
||||
(`gitea_mcp_server.py:19851`) reports service credential status through
|
||||
`service_summaries` (`gitea_mcp_server.py:19873`).
|
||||
2. **The return path is a flow too.** Content read from Gitea travels back into the model
|
||||
and is treated as instruction. This is the ADV2 edge, and it is the only edge in the
|
||||
diagram with no authentication on it, because it is not a request.
|
||||
@@ -180,16 +180,16 @@ the credential, and an attacker holding the token does not call our tools.
|
||||
|
||||
**Finding 3 — Any one role process can resolve every other role's credential.** This is not
|
||||
inferred; it is demonstrated by tool output. `gitea_list_profiles`
|
||||
(`gitea_mcp_server.py:19327`) called from the **author** session reports
|
||||
(`gitea_mcp_server.py:19487`) called from the **author** session reports
|
||||
`identity_status: "credentials present"` for `prgs-merger`, `prgs-reviewer`,
|
||||
`prgs-reconciler`, and every `mdcps` profile, because it calls `resolve_token` on each one
|
||||
(`gitea_mcp_server.py:19378`). The author process does not merely *have access to* the
|
||||
(`gitea_mcp_server.py:19538`). The author process does not merely *have access to* the
|
||||
merger's credential — it reads it to answer a status query. B2 is not a credential boundary
|
||||
in either direction.
|
||||
|
||||
**Finding 4 — The Gitea server reads CI and observability secrets.** `gitea_audit_config`
|
||||
(`gitea_mcp_server.py:19691`) reports `MDCPS Jenkins: enabled, read-only, authenticated`.
|
||||
That word `authenticated` is produced by `service_summaries` (`gitea_mcp_server.py:19713`,
|
||||
(`gitea_mcp_server.py:19851`) reports `MDCPS Jenkins: enabled, read-only, authenticated`.
|
||||
That word `authenticated` is produced by `service_summaries` (`gitea_mcp_server.py:19873`,
|
||||
defined at `gitea_config.py:837`), whose default check calls `_keychain_token` on the
|
||||
service's own keychain reference (`gitea_config.py:851`). Producing that one line requires
|
||||
the Gitea MCP server to read the Jenkins secret and the GlitchTip secret out of the
|
||||
@@ -197,8 +197,8 @@ keychain. B7 does not exist.
|
||||
|
||||
**Finding 5 — Jenkins and GlitchTip are already decomposed; the reach is residual.** Their
|
||||
tools live in separately registered servers, marked `external-mcp`
|
||||
(`gitea_mcp_server.py:17776`, `gitea_mcp_server.py:17782`, `gitea_mcp_server.py:17803`,
|
||||
`gitea_mcp_server.py:17808`) with their own expected tool sets (`mcp_discoverability.py:9`,
|
||||
(`gitea_mcp_server.py:17909`, `gitea_mcp_server.py:17915`, `gitea_mcp_server.py:17936`,
|
||||
`gitea_mcp_server.py:17941`) with their own expected tool sets (`mcp_discoverability.py:9`,
|
||||
`mcp_discoverability.py:17`). The correct decomposition was already chosen. What remains is
|
||||
a leak across it: the credential *references* still live in the Gitea configuration and are
|
||||
still resolved by the Gitea process. #75 bundled these services into one control-plane
|
||||
@@ -209,8 +209,8 @@ GlitchTip, the Sentry bridge runs *inside* the Gitea server, resolving its token
|
||||
process environment (`sentry_incident_bridge.py:190`) and sending it as a bearer header
|
||||
(`sentry_incident_bridge.py:289`). Being an environment variable rather than a keychain item
|
||||
makes it strictly worse: it needs no keychain prompt and is inherited by every subprocess the
|
||||
server spawns — including the `ps` invocations at `gitea_mcp_server.py:21601` and
|
||||
`gitea_mcp_server.py:21645`, reached from `gitea_mcp_server.py:21581`.
|
||||
server spawns — including the `ps` invocations at `gitea_mcp_server.py:21761` and
|
||||
`gitea_mcp_server.py:21805`, reached from `gitea_mcp_server.py:21741`.
|
||||
|
||||
**Finding 7 — The highest-value coordination asset has the weakest gate.** A6 is protected
|
||||
by filesystem permissions alone (CR14). Corrupting a lease requires no Gitea credential,
|
||||
@@ -219,8 +219,8 @@ assumes. Every other asset costs an attacker a credential; this one costs nothin
|
||||
local access, which is exactly ADV5's position.
|
||||
|
||||
**Finding 8 — Provenance authenticates the launch, not the caller.** `server_provenance` is
|
||||
reported as exactly `client_managed` or `manual_launch` (`gitea_mcp_server.py:19070`),
|
||||
derived from environment inspection (`gitea_mcp_server.py:15481`) with the recognized-key
|
||||
reported as exactly `client_managed` or `manual_launch` (`gitea_mcp_server.py:19217`),
|
||||
derived from environment inspection (`gitea_mcp_server.py:15630`) with the recognized-key
|
||||
allowlist at `gitea_config.py:1172` and the generator that emits the marker at
|
||||
`gitea_config.py:1233`. Every one of those facts is fixed at process start. A client that is
|
||||
trustworthy at launch and compromised a minute later remains `client_managed` for the life
|
||||
@@ -259,7 +259,7 @@ holds the token and calls the API instead of the tool.
|
||||
|
||||
**D3 — Credential resolution is scoped to the request principal.** A session must resolve its
|
||||
own credential and must have no path to any other principal's. The resolve-every-profile
|
||||
behavior behind `gitea_mcp_server.py:19378` and `gitea_mcp_server.py:19713` must report
|
||||
behavior behind `gitea_mcp_server.py:19538` and `gitea_mcp_server.py:19873` must report
|
||||
configured-or-not from configuration alone, without resolving the secret.
|
||||
|
||||
*Rationale.* Finding 3. An audit surface that proves a credential exists by fetching it is a
|
||||
@@ -336,11 +336,11 @@ The client is attached to the local fleet over stdio.
|
||||
|
||||
| Boundary | What ADV1 reaches | Stopped by |
|
||||
| -------- | ----------------- | ---------- |
|
||||
| B1 | Everything the fleet serves. The client *is* the sanctioned launcher: it satisfies the client-managed check (`gitea_mcp_server.py:15481`) by construction, and provenance is never re-verified after launch (Finding 8). | Nothing. The guard authenticates the launch, not the caller. |
|
||||
| B1 | Everything the fleet serves. The client *is* the sanctioned launcher: it satisfies the client-managed check (`gitea_mcp_server.py:15630`) by construction, and provenance is never re-verified after launch (Finding 8). | Nothing. The guard authenticates the launch, not the caller. |
|
||||
| B2 | All five roles — it is attached to all five namespaces. It can author a PR, approve it from the reviewer namespace, and merge it from the merger namespace. | Only the in-process self-review check, which compares `jcwalker3` (author) against `sysadmin` (reviewer) and **passes**, because Finding 1 made them different accounts while leaving reviewer and merger identical. A9 falls in one sequence of legitimate calls. |
|
||||
| B3 | Every credential in CR1–CR10 via CR13, with no additional prompt — the daemon is already sanctioned, so `assert_keychain_access_allowed` (`mcp_daemon_guard.py:583`) returns immediately. | Nothing. |
|
||||
| B4 | A1 and A2 in full. | Branch protection at the forge, to the extent configured. |
|
||||
| B5 | The operator's checkout and every worktree, through the author tools (`gitea_mcp_server.py:10956`), plus the shared stderr path at `mcp_server.py:13`. | Nothing; the server's disk is the target disk. |
|
||||
| B5 | The operator's checkout and every worktree, through the author tools (`gitea_mcp_server.py:10957`), plus the shared stderr path at `mcp_server.py:13`. | Nothing; the server's disk is the target disk. |
|
||||
| B6 | All coordination state — no credential required (CR14). It can forge lease ownership and clear decision locks. | Filesystem permissions, which it already satisfies. |
|
||||
| B7 | Jenkins (A4) and GlitchTip (A5) secrets via Finding 4, and CR11/CR12 from its own environment. | Nothing. |
|
||||
| B8 | Both tenants. | Nothing in-process; only the disabled-context check (`gitea_config.py:704`), which does not apply to enabled contexts. |
|
||||
|
||||
@@ -0,0 +1,203 @@
|
||||
# Stale worker retirement (#980)
|
||||
|
||||
The #978 instance-fleet snapshot made registry accuracy observable but
|
||||
deliberately read-only: a registry full of rows whose owning processes are long
|
||||
gone stays full. This document describes the sanctioned way to retire those
|
||||
rows — a dry-run-first, compare-and-swap-protected workflow available only to
|
||||
controller and reconciler namespaces.
|
||||
|
||||
Related: [instance-fleet-identity.md](instance-fleet-identity.md) (#978),
|
||||
[post-restart-reconcile.md](post-restart-reconcile.md) (#662).
|
||||
|
||||
## Why a dedicated registry token
|
||||
|
||||
`mcp_fleet_snapshot._consistency_token` seeds its digest with `snapshot_at`,
|
||||
formatted at second precision. Its output — surfaced as `registry_revision` and
|
||||
`consistency_token` on the snapshot — therefore changes on **every call**, even
|
||||
when no registry row changed. Any compare-and-swap gated on it can never pass:
|
||||
a dry-run/apply cycle spanning more than one second aborts unconditionally.
|
||||
|
||||
That token remains useful as an observation stamp and is unchanged. #980 adds a
|
||||
separate, *stable* token instead:
|
||||
|
||||
| Token | Module | Derived from | Stable across time? |
|
||||
| --- | --- | --- | --- |
|
||||
| `registry_revision` / `consistency_token` | `mcp_fleet_snapshot` | `snapshot_at` + a subset of row fields | **No** — moves every second |
|
||||
| `registry_fingerprint` | `mcp_fleet_retirement` | canonical retirement-relevant row content only | **Yes** |
|
||||
| `candidate_fingerprint` | `mcp_fleet_retirement` | canonical content of the selected candidate rows | **Yes** |
|
||||
|
||||
`registry_fingerprint` guarantees:
|
||||
|
||||
* identical canonical registry contents always produce the same token, whenever
|
||||
they are observed;
|
||||
* row iteration order never affects the token (serialized rows are sorted);
|
||||
* any retirement-relevant change moves it — row creation or deletion, identity
|
||||
change, heartbeat or TTL change, ownership change, registration-state change,
|
||||
PID change, or repository-binding change.
|
||||
|
||||
The exact field set is `mcp_fleet_retirement.FINGERPRINT_FIELDS`. Deliberately
|
||||
excluded: `token_fingerprint` (credential-adjacent, never a retirement input),
|
||||
the four `*_revision` columns (revision drift is an independent restart concern
|
||||
and is not part of the eligibility conjunction), and the `retired_*` bookkeeping
|
||||
columns this feature adds. Numeric values are canonicalised, so a TTL that
|
||||
round-trips through SQLite as `900.0` hashes identically to `900`.
|
||||
|
||||
## Eligibility — the conjunction
|
||||
|
||||
A registration is retired only when **every** one of these holds. Any missing
|
||||
or contradictory evidence preserves the row.
|
||||
|
||||
| Requirement | Preserve reason code when it fails |
|
||||
| --- | --- |
|
||||
| `status` is `active` | `already_terminal_registration` |
|
||||
| Every field the conjunction reads is present (`REQUIRED_IDENTITY_FIELDS`) | `incomplete_registry_identity` |
|
||||
| `last_heartbeat_at` parses as a UTC stamp | `unparsable_heartbeat` |
|
||||
| Worker is not live | `worker_live` |
|
||||
| PID probe returns a definite answer | `pid_liveness_unknown` |
|
||||
| PID probe says the process is gone | `pid_alive` |
|
||||
| Heartbeat has expired under the canonical TTL | `heartbeat_not_expired` |
|
||||
| `ownership_state` is exactly `stale` | `ambiguous_ownership_state` |
|
||||
| Repository binding present and canonical | `repository_binding_ambiguous` |
|
||||
| No identity evidence shared with a live or unprobeable worker | `conflicting_identity_evidence` |
|
||||
| Not an active workflow-lease owner | `protected_active_workflow_owner` |
|
||||
|
||||
Eligible rows carry `eligible_stale_orphan`.
|
||||
|
||||
Two properties are worth stating explicitly:
|
||||
|
||||
* **`pid_alive` can only withdraw liveness, never grant it**
|
||||
(`WorkerRegistry.is_live`, #948 AC7). A heartbeat-lapsed but still-running
|
||||
process therefore classifies as `stale` in the snapshot, yet #980's added
|
||||
`pid_alive is False` requirement preserves it. An unprobeable PID (`None`)
|
||||
also fails closed.
|
||||
* **Trusted launcher identity is not required.** #980 places trusted
|
||||
`client_instance_id` propagation out of scope and lists backfilling trusted
|
||||
identity for legacy workers as a non-goal. Requiring `inst-…` provenance here
|
||||
would preserve every legacy row forever and make the feature inert. What is
|
||||
required is that the registry *fields* the conjunction reads are present.
|
||||
|
||||
Multiple processes belonging to one legitimate worker cohort are not treated as
|
||||
multiple independent workers: the fleet model from #948/#978 is preserved
|
||||
unchanged, and sharing a role or profile is never a duplicate.
|
||||
|
||||
## Tools
|
||||
|
||||
### `gitea_plan_stale_worker_retirement`
|
||||
|
||||
Read-only. Controller and reconciler only.
|
||||
|
||||
| Parameter | Meaning |
|
||||
| --- | --- |
|
||||
| `remote` | `dadeschools` or `prgs` |
|
||||
| `host`, `org`, `repo` | Optional overrides (audit context) |
|
||||
| `canonical_repository` | Expected repository binding; defaults to the process root |
|
||||
|
||||
Returns `registry_fingerprint`, `candidate_fingerprint`,
|
||||
`candidate_worker_identities`, per-worker `candidates` and `preserved` entries
|
||||
(each with `reason_code`, `detail`, and structured `evidence`),
|
||||
`preserved_reason_counts`, `assessed_count`, `candidate_count`,
|
||||
`preserved_count`, and `protected_active_workflow_owners`.
|
||||
`mutation_performed` is always `false` and `read_only` is always `true`.
|
||||
|
||||
Planning is deterministic: the same authoritative registry contents produce the
|
||||
same plan and the same tokens regardless of when they are observed.
|
||||
|
||||
### `gitea_apply_stale_worker_retirement`
|
||||
|
||||
Mutating. Controller and reconciler only.
|
||||
|
||||
| Parameter | Meaning |
|
||||
| --- | --- |
|
||||
| `registry_fingerprint` | The exact stable token the plan returned |
|
||||
| `candidate_fingerprint` | The exact candidate-set token the plan returned |
|
||||
| `worker_identities` | The exact candidate identities (list, or JSON / comma-separated string) |
|
||||
| `remote`, `host`, `org`, `repo` | As above |
|
||||
| `canonical_repository` | Must match the value the plan used |
|
||||
|
||||
Before touching the registry, apply fails closed on: profile permission, role
|
||||
kind, master parity (`mutation_safe`), stable-runtime mode, capability
|
||||
resolution refreshed immediately before mutation, worker-registry availability,
|
||||
workflow-lease enumeration failure, and daemon-cohort uniqueness
|
||||
(`classify_cohort`).
|
||||
|
||||
A matching token is necessary but never sufficient. Inside one
|
||||
`BEGIN IMMEDIATE` transaction (`WorkerRegistry.retire_stale_workers`) the
|
||||
server:
|
||||
|
||||
1. re-reads the authoritative rows;
|
||||
2. recomputes `registry_fingerprint` from *those* rows and compares — a mismatch
|
||||
returns `registry_revision_moved` with `retired_count: 0` and no write;
|
||||
3. recomputes the eligibility plan from *those* rows — re-reading the active
|
||||
workflow leases rather than reusing the set captured before the transaction
|
||||
opened, so a lease acquired after planning still preserves its worker — and
|
||||
compares `candidate_fingerprint`. A mismatch returns `candidate_set_moved`
|
||||
with `retired_count: 0` and no write; a lease-enumeration failure raises and
|
||||
rolls the transaction back;
|
||||
4. revalidates every requested identity against that fresh plan;
|
||||
5. retires each survivor with a guarded `UPDATE` that additionally asserts
|
||||
`status`, `last_heartbeat_at`, `generation_id`, `session_id`,
|
||||
`fencing_epoch`, and `pid` are unchanged. A guard that matches no row
|
||||
preserves the worker with `row_changed_since_plan`.
|
||||
|
||||
There is no window between a safety check and its matching write, so a worker
|
||||
that comes back to life, changes ownership, or is retired concurrently cannot be
|
||||
removed on the strength of a stale observation. Any exception — including a
|
||||
commit failure — rolls the whole transaction back and returns
|
||||
`transaction_failed` with `success: false`, `retired_count: 0`, and
|
||||
`mutation_performed: false`; a partial write can never be reported as success.
|
||||
|
||||
### Outcomes
|
||||
|
||||
| `outcome` | Meaning | `mutation_performed` |
|
||||
| --- | --- | --- |
|
||||
| `planned` | Dry-run result | `false` |
|
||||
| `applied` | Transaction ran; see `retired` / `preserved` | `true` only if something was retired |
|
||||
| `registry_revision_moved` | Registry changed between plan and apply | `false` |
|
||||
| `candidate_set_moved` | Eligibility verdict changed between plan and apply | `false` |
|
||||
| `already_retired` | Every requested row is already retired (idempotent replay) | `false` |
|
||||
| `nothing_requested` | Empty target list | `false` |
|
||||
| `transaction_failed` | Rolled back; nothing retired | `false` |
|
||||
|
||||
## What retirement does to the fleet snapshot
|
||||
|
||||
A retired row keeps its history: `status` moves to `retired` and `retired_at`,
|
||||
`retired_by`, `retirement_reason` are recorded. Nothing is deleted. Because
|
||||
`retired` is not `active`, the #978 snapshot counts the row as **historical**,
|
||||
not stale, so `stale_worker_count` falls and historical rows never make the live
|
||||
fleet unsafe by themselves.
|
||||
|
||||
**Retirement does not repair untrusted live identity.** Live workers registered
|
||||
under legacy `pid-`/`proc-` instance identities are preserved untouched and
|
||||
keep their `legacy_incomplete_identity` blockers. Retiring every stale row can
|
||||
therefore legitimately produce:
|
||||
|
||||
* `stale_worker_count: 0`
|
||||
* residual live `legacy_incomplete_identity` blockers
|
||||
* `live_fleet_safe: false`
|
||||
|
||||
That is a truthful result, and the `post_apply` block reports the remaining
|
||||
blockers rather than claiming the fleet became safe. Trusted
|
||||
`client_instance_id` propagation through launchers is a separate enrollment
|
||||
problem.
|
||||
|
||||
## Permissions
|
||||
|
||||
* **Allowed:** `controller`, `reconciler`.
|
||||
* **Denied:** author, reviewer, merger — they keep `gitea.read` for diagnosis
|
||||
elsewhere and are refused this surface by role.
|
||||
* The Gitea operation gate stays `gitea.read` because the mutation lands in the
|
||||
**local control-plane worker registry**, not in Gitea — the same model the
|
||||
#601 lease lifecycle uses. **No new Gitea write permission is introduced for
|
||||
any profile**, and no author permission is broadened.
|
||||
* The fleet snapshot remains observational: nothing here turns it into a gate on
|
||||
ordinary author work.
|
||||
|
||||
## Non-goals
|
||||
|
||||
* Killing or restarting processes.
|
||||
* Editing session files or configuration.
|
||||
* Direct database cleanup outside the sanctioned transaction.
|
||||
* Retiring live workers.
|
||||
* Backfilling trusted identity for legacy workers.
|
||||
* Rewriting worker ownership.
|
||||
* Cleaning unrelated workflow-lease or issue-claim registries.
|
||||
+90
-15
@@ -1180,6 +1180,10 @@ RECOGNIZED_GITEA_ENV_KEYS = frozenset({
|
||||
"GITEA_SERVER_PROVENANCE",
|
||||
"GITEA_AUTHOR_WORKTREE",
|
||||
"GITEA_ACTIVE_WORKTREE",
|
||||
"GITEA_REVIEWER_WORKTREE",
|
||||
"GITEA_MERGER_WORKTREE",
|
||||
"GITEA_CANONICAL_REPOSITORY_ROOT",
|
||||
"GITEA_MCP_SESSION_STATE_TTL_HOURS",
|
||||
"GITEA_DISABLE_KEYCHAIN",
|
||||
"GITEA_CONTROL_PLANE_DB",
|
||||
"GITEA_DB_PATH",
|
||||
@@ -1189,6 +1193,31 @@ RECOGNIZED_GITEA_ENV_KEYS = frozenset({
|
||||
"GITEA_IRRECOVERABLE_HMAC_SECRET",
|
||||
"GITEA_FORCE_MCP_RUNTIME_CHECK",
|
||||
"GITEA_FORCE_CLIENT_MANAGED",
|
||||
# #975: the client-identity inputs the server actually consumes at startup
|
||||
# (CLIENT_NAME_ENV / CLIENT_INSTANCE_ENV / CLIENT_SESSION_ENV in
|
||||
# gitea_mcp_server). Production read them while this allowlist omitted them,
|
||||
# so the peer-env scan classified them as unsupported overrides and the
|
||||
# capability resolver refused every mutation fleet-wide. Named individually
|
||||
# on purpose: no prefix is added, so an unrecognised GITEA_* override is
|
||||
# still refused exactly as it was before.
|
||||
"GITEA_MCP_CLIENT",
|
||||
"GITEA_MCP_CLIENT_INSTANCE",
|
||||
"GITEA_MCP_CLIENT_SESSION",
|
||||
# #978: operator-approved fleet enrollment identity shared by one launch.
|
||||
"GITEA_MCP_FLEET_RUN_ID",
|
||||
"GITEA_MCP_PROCESS_IDENTITY",
|
||||
# #978 B1/B2: launcher-sealed provenance + peer-visible worker identity so
|
||||
# the instance-aware mutation gate can distinguish two legitimate
|
||||
# application instances that share a profile from a true duplicate worker.
|
||||
"GITEA_MCP_INSTANCE_PROVENANCE",
|
||||
"GITEA_MCP_WORKER_IDENTITY",
|
||||
"GITEA_MCP_GENERATION_ID",
|
||||
# #975 review 652 B1: production also consumes HEARTBEAT_INTERVAL_ENV from
|
||||
# mcp_worker_identity via gitea_mcp_server._start_worker_heartbeat. Omitting
|
||||
# it reproduced the same unsupported-env → runtime_reconnect_required
|
||||
# failure mode for the documented operator override. Named individually;
|
||||
# no GITEA_* / GITEA_WORKER_* prefix is added.
|
||||
"GITEA_WORKER_HEARTBEAT_INTERVAL_SECONDS",
|
||||
})
|
||||
|
||||
RECOGNIZED_GITEA_ENV_PREFIXES = (
|
||||
@@ -1216,24 +1245,70 @@ def get_unconsumed_gitea_env_overrides(env=None) -> dict[str, str]:
|
||||
return unconsumed
|
||||
|
||||
|
||||
def launcher_entry(profile_name, config_path=None):
|
||||
def launcher_entry(
|
||||
profile_name,
|
||||
config_path=None,
|
||||
*,
|
||||
client_type=None,
|
||||
client_instance_id=None,
|
||||
fleet_run_id=None,
|
||||
session_id=None,
|
||||
launch_nonce=None,
|
||||
):
|
||||
"""Return a thin MCP launcher entry for *profile_name*.
|
||||
|
||||
Contains command/args and the GITEA_MCP_* / GITEA_CLIENT_MANAGED env vars — never a token
|
||||
or password. Suitable for Claude / Gemini / Codex ``mcpServers`` blocks.
|
||||
Contains command/args and the GITEA_MCP_* / GITEA_CLIENT_MANAGED env vars —
|
||||
never a token or password. Suitable for Claude / Gemini / Codex
|
||||
``mcpServers`` blocks.
|
||||
|
||||
#978 B1: production launches mint (or reuse) a trusted
|
||||
``GITEA_MCP_CLIENT_INSTANCE`` so workers never fall back to the legacy
|
||||
placeholder identity on the real serve path. Pass *client_instance_id* to
|
||||
resume the same launch; omit it for a fresh launch (new ID).
|
||||
"""
|
||||
command, args = server_command()
|
||||
return {
|
||||
"gitea-tools": {
|
||||
"command": command,
|
||||
"args": args,
|
||||
"env": {
|
||||
"GITEA_MCP_CONFIG": config_path or DEFAULT_CONFIG_PATH,
|
||||
"GITEA_MCP_PROFILE": profile_name,
|
||||
"GITEA_CLIENT_MANAGED": "1",
|
||||
},
|
||||
}
|
||||
}
|
||||
import mcp_application_launcher as app_launcher
|
||||
|
||||
built = app_launcher.launcher_entry_for_profile(
|
||||
profile_name,
|
||||
client_type=client_type or "unknown",
|
||||
config_path=config_path or DEFAULT_CONFIG_PATH,
|
||||
client_instance_id=client_instance_id,
|
||||
fleet_run_id=fleet_run_id,
|
||||
session_id=session_id,
|
||||
launch_nonce=launch_nonce,
|
||||
server_key="gitea-tools",
|
||||
)
|
||||
# Public shape stays {server_key: {command, args, env}} for existing callers.
|
||||
return {"gitea-tools": built["gitea-tools"]}
|
||||
|
||||
|
||||
def multi_namespace_launcher_entries(
|
||||
profile_by_namespace,
|
||||
*,
|
||||
client_type,
|
||||
config_path=None,
|
||||
client_instance_id=None,
|
||||
fleet_run_id=None,
|
||||
session_id=None,
|
||||
launch_nonce=None,
|
||||
):
|
||||
"""Build production mcpServers for all five namespaces of one application launch.
|
||||
|
||||
One shared trusted ``client_instance_id`` is minted (or reused) and
|
||||
propagated to every namespace worker env. See
|
||||
:func:`mcp_application_launcher.build_application_mcp_servers`.
|
||||
"""
|
||||
import mcp_application_launcher as app_launcher
|
||||
|
||||
return app_launcher.build_application_mcp_servers(
|
||||
profile_by_namespace,
|
||||
client_type=client_type,
|
||||
config_path=config_path or DEFAULT_CONFIG_PATH,
|
||||
client_instance_id=client_instance_id,
|
||||
fleet_run_id=fleet_run_id,
|
||||
session_id=session_id,
|
||||
launch_nonce=launch_nonce,
|
||||
)
|
||||
|
||||
|
||||
|
||||
|
||||
+1480
-54
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,393 @@
|
||||
"""Production application launcher for multi-namespace MCP fleets (#978 B1).
|
||||
|
||||
One real LLM application launch mints exactly one trusted
|
||||
``client_instance_id`` and propagates it to every Gitea MCP namespace worker
|
||||
started for that launch. Workers never invent a trusted instance identity from
|
||||
PID proximity, timestamps, or ordinary untrusted environment values.
|
||||
|
||||
This module is the production serve-path authority for instance identity.
|
||||
Tests and fixtures may call the same functions, but production registration
|
||||
receives the identity from the env this launcher builds — not from a hand-set
|
||||
test-only helper that bypasses it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import secrets
|
||||
from typing import Any, Mapping
|
||||
|
||||
import mcp_fleet_snapshot as fleet
|
||||
import mcp_worker_identity as mwi
|
||||
|
||||
#: Canonical MCP server names for the five role namespaces.
|
||||
NAMESPACE_SERVER_NAMES: tuple[str, ...] = (
|
||||
"gitea-author",
|
||||
"gitea-reviewer",
|
||||
"gitea-merger",
|
||||
"gitea-controller",
|
||||
"gitea-reconciler",
|
||||
)
|
||||
|
||||
#: Map MCP server name → role/namespace kind.
|
||||
SERVER_TO_NAMESPACE: dict[str, str] = {
|
||||
"gitea-author": "author",
|
||||
"gitea-reviewer": "reviewer",
|
||||
"gitea-merger": "merger",
|
||||
"gitea-controller": "controller",
|
||||
"gitea-reconciler": "reconciler",
|
||||
}
|
||||
|
||||
SANCTIONED_NAMESPACES: tuple[str, ...] = (
|
||||
"author",
|
||||
"reviewer",
|
||||
"merger",
|
||||
"controller",
|
||||
"reconciler",
|
||||
)
|
||||
|
||||
# Env the launcher may set on every worker of one application launch.
|
||||
CLIENT_NAME_ENV = "GITEA_MCP_CLIENT"
|
||||
CLIENT_INSTANCE_ENV = fleet.CLIENT_INSTANCE_ENV
|
||||
FLEET_RUN_ENV = fleet.FLEET_RUN_ENV
|
||||
CLIENT_SESSION_ENV = "GITEA_MCP_CLIENT_SESSION"
|
||||
CLIENT_MANAGED_ENV = "GITEA_CLIENT_MANAGED"
|
||||
PROFILE_ENV = "GITEA_MCP_PROFILE"
|
||||
CONFIG_ENV = "GITEA_MCP_CONFIG"
|
||||
INSTANCE_PROVENANCE_ENV = "GITEA_MCP_INSTANCE_PROVENANCE"
|
||||
WORKER_IDENTITY_ENV = "GITEA_MCP_WORKER_IDENTITY"
|
||||
GENERATION_ID_ENV = "GITEA_MCP_GENERATION_ID"
|
||||
|
||||
# Marker the trusted launcher alone writes; ordinary user env without this
|
||||
# marker is never classified as launcher-trusted provenance.
|
||||
LAUNCHER_PROVENANCE_VALUE = fleet.INSTANCE_ID_PROVENANCE_TRUSTED
|
||||
|
||||
|
||||
def mint_application_launch(
|
||||
client_type: str | None,
|
||||
*,
|
||||
launch_nonce: str | None = None,
|
||||
fleet_run_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
now=None,
|
||||
) -> dict[str, Any]:
|
||||
"""Mint one trusted application-instance identity for a production launch.
|
||||
|
||||
Called exactly once per real application launch. The returned
|
||||
``client_instance_id`` is injected into every namespace worker environment
|
||||
for that launch. A second call (separate launch) yields a different ID.
|
||||
"""
|
||||
client = mwi.normalize_client_name(client_type)
|
||||
instance_id = fleet.generate_client_instance_id(
|
||||
client, launch_nonce=launch_nonce, now=now
|
||||
)
|
||||
assessment = fleet.assess_instance_identity(instance_id)
|
||||
if not assessment["trusted"]:
|
||||
# generate_client_instance_id always produces a trusted format; fail
|
||||
# closed if that invariant ever breaks rather than shipping untrusted.
|
||||
raise RuntimeError(
|
||||
f"launcher produced untrusted client_instance_id {instance_id!r}: "
|
||||
f"{assessment.get('reasons')}"
|
||||
)
|
||||
session = (session_id or "").strip() or f"launch-{secrets.token_hex(12)}"
|
||||
return {
|
||||
"client_type": client,
|
||||
"client_instance_id": instance_id,
|
||||
"instance_id_provenance": LAUNCHER_PROVENANCE_VALUE,
|
||||
"instance_identity_trusted": True,
|
||||
"fleet_run_id": (fleet_run_id or "").strip() or None,
|
||||
"session_id": session,
|
||||
"namespaces": list(SANCTIONED_NAMESPACES),
|
||||
"namespace_server_names": list(NAMESPACE_SERVER_NAMES),
|
||||
}
|
||||
|
||||
|
||||
def namespace_worker_env(
|
||||
*,
|
||||
profile_name: str,
|
||||
client_type: str | None,
|
||||
client_instance_id: str,
|
||||
config_path: str | None = None,
|
||||
fleet_run_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
extra_env: Mapping[str, str] | None = None,
|
||||
) -> dict[str, str]:
|
||||
"""Build the environment for one namespace worker of a trusted launch.
|
||||
|
||||
Never invents a client_instance_id. The caller must supply the launch-minted
|
||||
identity so all five workers receive the same value.
|
||||
"""
|
||||
assessment = fleet.assess_instance_identity(client_instance_id)
|
||||
if not assessment["trusted"]:
|
||||
raise ValueError(
|
||||
"namespace_worker_env refuses untrusted client_instance_id "
|
||||
f"{client_instance_id!r}: {assessment.get('reasons')}"
|
||||
)
|
||||
client = mwi.normalize_client_name(client_type)
|
||||
env: dict[str, str] = {
|
||||
PROFILE_ENV: str(profile_name),
|
||||
CLIENT_MANAGED_ENV: "1",
|
||||
"GITEA_MCP_CLIENT": client,
|
||||
CLIENT_INSTANCE_ENV: assessment["client_instance_id"],
|
||||
INSTANCE_PROVENANCE_ENV: LAUNCHER_PROVENANCE_VALUE,
|
||||
}
|
||||
if config_path:
|
||||
env[CONFIG_ENV] = str(config_path)
|
||||
if fleet_run_id:
|
||||
env[FLEET_RUN_ENV] = str(fleet_run_id)
|
||||
if session_id:
|
||||
env[CLIENT_SESSION_ENV] = str(session_id)
|
||||
if extra_env:
|
||||
# Never let untrusted callers override the trusted instance keys.
|
||||
protected = {
|
||||
CLIENT_INSTANCE_ENV,
|
||||
INSTANCE_PROVENANCE_ENV,
|
||||
"GITEA_MCP_CLIENT",
|
||||
CLIENT_MANAGED_ENV,
|
||||
}
|
||||
for key, value in extra_env.items():
|
||||
if key in protected:
|
||||
continue
|
||||
env[str(key)] = str(value)
|
||||
return env
|
||||
|
||||
|
||||
def build_application_mcp_servers(
|
||||
profile_by_namespace: Mapping[str, str],
|
||||
*,
|
||||
client_type: str | None,
|
||||
config_path: str | None = None,
|
||||
client_instance_id: str | None = None,
|
||||
fleet_run_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
launch_nonce: str | None = None,
|
||||
command: str | None = None,
|
||||
args: list[str] | None = None,
|
||||
now=None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build a production ``mcpServers`` map for one application launch.
|
||||
|
||||
Mints one ``client_instance_id`` when *client_instance_id* is omitted (fresh
|
||||
launch). When the caller supplies a previously minted trusted ID (resume of
|
||||
the same launch / config rewrite), that ID is reused so reconnect keeps
|
||||
attribution. A full new application restart omits the ID and receives a
|
||||
fresh mint.
|
||||
|
||||
Every namespace server entry receives the **same** instance ID. Separate
|
||||
calls with no supplied ID receive distinct IDs.
|
||||
"""
|
||||
missing = [
|
||||
ns for ns in SANCTIONED_NAMESPACES if not profile_by_namespace.get(ns)
|
||||
]
|
||||
if missing:
|
||||
raise ValueError(
|
||||
"build_application_mcp_servers requires a profile for every "
|
||||
f"sanctioned namespace; missing: {missing}"
|
||||
)
|
||||
|
||||
if client_instance_id is None:
|
||||
launch = mint_application_launch(
|
||||
client_type,
|
||||
launch_nonce=launch_nonce,
|
||||
fleet_run_id=fleet_run_id,
|
||||
session_id=session_id,
|
||||
now=now,
|
||||
)
|
||||
else:
|
||||
assessment = fleet.assess_instance_identity(client_instance_id)
|
||||
if not assessment["trusted"]:
|
||||
raise ValueError(
|
||||
"refusing to propagate untrusted client_instance_id "
|
||||
f"{client_instance_id!r} into production launch envs: "
|
||||
f"{assessment.get('reasons')}"
|
||||
)
|
||||
launch = {
|
||||
"client_type": mwi.normalize_client_name(client_type),
|
||||
"client_instance_id": assessment["client_instance_id"],
|
||||
"instance_id_provenance": LAUNCHER_PROVENANCE_VALUE,
|
||||
"instance_identity_trusted": True,
|
||||
"fleet_run_id": (fleet_run_id or "").strip() or None,
|
||||
"session_id": (session_id or "").strip()
|
||||
or f"launch-{secrets.token_hex(12)}",
|
||||
"namespaces": list(SANCTIONED_NAMESPACES),
|
||||
"namespace_server_names": list(NAMESPACE_SERVER_NAMES),
|
||||
}
|
||||
|
||||
# Resolve command/args from the production server entry when not provided.
|
||||
if command is None or args is None:
|
||||
import gitea_config
|
||||
|
||||
cmd, cmd_args = gitea_config.server_command()
|
||||
command = command or cmd
|
||||
args = args if args is not None else list(cmd_args)
|
||||
|
||||
servers: dict[str, Any] = {}
|
||||
shared_id = launch["client_instance_id"]
|
||||
for namespace in SANCTIONED_NAMESPACES:
|
||||
server_name = f"gitea-{namespace}"
|
||||
profile = profile_by_namespace[namespace]
|
||||
env = namespace_worker_env(
|
||||
profile_name=profile,
|
||||
client_type=launch["client_type"],
|
||||
client_instance_id=shared_id,
|
||||
config_path=config_path,
|
||||
fleet_run_id=launch.get("fleet_run_id"),
|
||||
session_id=launch.get("session_id"),
|
||||
)
|
||||
servers[server_name] = {
|
||||
"command": command,
|
||||
"args": list(args),
|
||||
"env": env,
|
||||
}
|
||||
|
||||
return {
|
||||
"mcpServers": servers,
|
||||
"launch": launch,
|
||||
"client_instance_id": shared_id,
|
||||
"client_type": launch["client_type"],
|
||||
"namespaces": list(SANCTIONED_NAMESPACES),
|
||||
"shared_instance_id_across_namespaces": True,
|
||||
"namespace_count": len(SANCTIONED_NAMESPACES),
|
||||
}
|
||||
|
||||
|
||||
def launcher_entry_for_profile(
|
||||
profile_name: str,
|
||||
*,
|
||||
client_type: str | None = None,
|
||||
config_path: str | None = None,
|
||||
client_instance_id: str | None = None,
|
||||
fleet_run_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
server_key: str = "gitea-tools",
|
||||
launch_nonce: str | None = None,
|
||||
now=None,
|
||||
) -> dict[str, Any]:
|
||||
"""Thin single-server production launcher entry with trusted instance ID.
|
||||
|
||||
Used when only one namespace is being configured. Still mints (or reuses)
|
||||
a trusted ``client_instance_id`` so production never relies on the legacy
|
||||
placeholder identity for normal launches.
|
||||
"""
|
||||
import gitea_config
|
||||
|
||||
if client_instance_id is None:
|
||||
launch = mint_application_launch(
|
||||
client_type or "unknown",
|
||||
launch_nonce=launch_nonce,
|
||||
fleet_run_id=fleet_run_id,
|
||||
session_id=session_id,
|
||||
now=now,
|
||||
)
|
||||
client_instance_id = launch["client_instance_id"]
|
||||
client = launch["client_type"]
|
||||
fleet_run = launch.get("fleet_run_id")
|
||||
session = launch.get("session_id")
|
||||
else:
|
||||
assessment = fleet.assess_instance_identity(client_instance_id)
|
||||
if not assessment["trusted"]:
|
||||
raise ValueError(
|
||||
f"untrusted client_instance_id {client_instance_id!r}"
|
||||
)
|
||||
client = mwi.normalize_client_name(client_type)
|
||||
fleet_run = (fleet_run_id or "").strip() or None
|
||||
session = (session_id or "").strip() or None
|
||||
client_instance_id = assessment["client_instance_id"]
|
||||
|
||||
command, args = gitea_config.server_command()
|
||||
env = namespace_worker_env(
|
||||
profile_name=profile_name,
|
||||
client_type=client,
|
||||
client_instance_id=client_instance_id,
|
||||
config_path=config_path or gitea_config.DEFAULT_CONFIG_PATH,
|
||||
fleet_run_id=fleet_run,
|
||||
session_id=session,
|
||||
)
|
||||
return {
|
||||
server_key: {
|
||||
"command": command,
|
||||
"args": args,
|
||||
"env": env,
|
||||
},
|
||||
"client_instance_id": client_instance_id,
|
||||
"client_type": client,
|
||||
}
|
||||
|
||||
|
||||
def collect_instance_ids_from_mcp_servers(
|
||||
mcp_servers: Mapping[str, Any],
|
||||
) -> dict[str, Any]:
|
||||
"""Inspect a production mcpServers map for shared instance attribution.
|
||||
|
||||
Returns the unique set of client_instance_id values across gitea-* servers
|
||||
and whether all five namespaces share exactly one trusted ID.
|
||||
"""
|
||||
ids: list[str] = []
|
||||
by_server: dict[str, str | None] = {}
|
||||
for name in NAMESPACE_SERVER_NAMES:
|
||||
entry = mcp_servers.get(name) or {}
|
||||
env = entry.get("env") or {}
|
||||
raw = (env.get(CLIENT_INSTANCE_ENV) or "").strip() or None
|
||||
by_server[name] = raw
|
||||
if raw:
|
||||
ids.append(raw)
|
||||
unique = sorted(set(ids))
|
||||
trusted = [
|
||||
i
|
||||
for i in unique
|
||||
if fleet.assess_instance_identity(i)["trusted"]
|
||||
]
|
||||
return {
|
||||
"server_instance_ids": by_server,
|
||||
"unique_instance_ids": unique,
|
||||
"trusted_instance_ids": trusted,
|
||||
"shared_single_trusted_id": (
|
||||
len(unique) == 1
|
||||
and len(trusted) == 1
|
||||
and all(by_server.get(n) == unique[0] for n in NAMESPACE_SERVER_NAMES)
|
||||
),
|
||||
"namespace_server_count": sum(
|
||||
1 for n in NAMESPACE_SERVER_NAMES if n in mcp_servers
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def inherit_or_refuse_client_instance(
|
||||
env: Mapping[str, str] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Resolve instance identity for a worker process at serve time.
|
||||
|
||||
Production workers inherit the launcher-issued ID. They never mint a
|
||||
trusted ID themselves. Missing / legacy / malformed values fail soft into
|
||||
an untrusted assessment so registration can still record a diagnostic row
|
||||
without authorizing multi-instance fleet mutation.
|
||||
"""
|
||||
source = dict(env if env is not None else os.environ)
|
||||
raw = (source.get(CLIENT_INSTANCE_ENV) or "").strip() or None
|
||||
provenance_marker = (source.get(INSTANCE_PROVENANCE_ENV) or "").strip()
|
||||
assessment = fleet.assess_instance_identity(raw)
|
||||
# Ordinary user-supplied values without launcher provenance marker are
|
||||
# still format-checked by assess_instance_identity. When the format is
|
||||
# trusted but the launcher marker is absent, keep the ID but note that
|
||||
# provenance is not launcher-sealed (operator hand-set or legacy config).
|
||||
if assessment["trusted"] and provenance_marker != LAUNCHER_PROVENANCE_VALUE:
|
||||
assessment = dict(assessment)
|
||||
assessment["launcher_sealed"] = False
|
||||
assessment["reasons"] = list(assessment.get("reasons") or []) + [
|
||||
f"{INSTANCE_PROVENANCE_ENV} is not {LAUNCHER_PROVENANCE_VALUE!r}; "
|
||||
"identity format is valid but not sealed by the production launcher"
|
||||
]
|
||||
else:
|
||||
assessment = dict(assessment)
|
||||
assessment["launcher_sealed"] = bool(
|
||||
assessment["trusted"]
|
||||
and provenance_marker == LAUNCHER_PROVENANCE_VALUE
|
||||
)
|
||||
assessment["fleet_run_id"] = (source.get(FLEET_RUN_ENV) or "").strip() or None
|
||||
assessment["session_id"] = (
|
||||
(source.get(CLIENT_SESSION_ENV) or "").strip() or None
|
||||
)
|
||||
assessment["client_type"] = mwi.normalize_client_name(
|
||||
(source.get("GITEA_MCP_CLIENT") or "").strip() or None
|
||||
)
|
||||
return assessment
|
||||
+21
-9
@@ -92,7 +92,15 @@ OPERATOR_UI_STEPS: dict[str, tuple[str, ...]] = {
|
||||
),
|
||||
}
|
||||
|
||||
DEFAULT_CLIENT = "codex"
|
||||
#: What an *unidentified* client gets. #948: this is deliberately the
|
||||
#: host-agnostic step set rather than a specific product. Defaulting to one
|
||||
#: vendor emitted Codex UI steps to a Gemini/Antigravity operator, who then had
|
||||
#: no reachable recovery path — the guidance named a panel they do not have.
|
||||
DEFAULT_CLIENT = "generic"
|
||||
|
||||
#: The historical default, kept addressable by name so Codex callers still get
|
||||
#: Codex steps, without it silently becoming the fallback for unknown clients.
|
||||
LEGACY_DEFAULT_CLIENT = "codex"
|
||||
|
||||
|
||||
def normalize_reason(reason: str | None) -> str:
|
||||
@@ -128,14 +136,18 @@ def normalize_reason(reason: str | None) -> str:
|
||||
|
||||
|
||||
def normalize_client(client: str | None) -> str:
|
||||
"""Return a known client key for operator UI steps."""
|
||||
text = (client or "").strip().lower().replace(" ", "_").replace("-", "_")
|
||||
if text in ("codex", "openai_codex", "openai"):
|
||||
return "codex"
|
||||
if text in ("claude", "claude_code", "claude_desktop", "anthropic"):
|
||||
return "claude_code"
|
||||
if text in OPERATOR_UI_STEPS:
|
||||
return text
|
||||
"""Return the UI-step key for a client.
|
||||
|
||||
#948: alias resolution is shared with ``mcp_worker_identity`` so a client
|
||||
name means the same thing wherever it is read. A name we recognise but have
|
||||
no bespoke steps for — Gemini, Antigravity, Grok — resolves to the generic
|
||||
host-agnostic steps rather than to another vendor's panel.
|
||||
"""
|
||||
import mcp_worker_identity
|
||||
|
||||
canonical = mcp_worker_identity.normalize_client_name(client)
|
||||
if canonical in OPERATOR_UI_STEPS:
|
||||
return canonical
|
||||
return DEFAULT_CLIENT
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,479 @@
|
||||
"""CAS-protected retirement planning for stale worker registrations (#980).
|
||||
|
||||
#978 (merged PR #979) made the fleet observable: every registered namespace
|
||||
worker, its instance attribution, its heartbeat freshness, and a structured
|
||||
classification. It deliberately stopped there — the snapshot is read-only and
|
||||
the control plane still had no sanctioned way to retire registry rows whose
|
||||
owning process is conclusively gone.
|
||||
|
||||
This module is the *decision layer* for that retirement. It is pure: callers
|
||||
supply registry rows, a clock, and a PID probe; nothing here opens SQLite,
|
||||
scans process tables, or mutates state. The transactional apply lives in
|
||||
:meth:`mcp_worker_identity.WorkerRegistry.retire_stale_workers`, which calls
|
||||
back into these same pure functions so plan and apply can never disagree about
|
||||
what "the registry looks like" or "which rows are eligible".
|
||||
|
||||
Why a separate token
|
||||
--------------------
|
||||
|
||||
``mcp_fleet_snapshot._consistency_token`` seeds its digest with ``snapshot_at``
|
||||
at second precision, so ``registry_revision`` changes on every call even when
|
||||
no registry row changed. A compare-and-swap gated on it can never pass — a
|
||||
dry-run/apply cycle spanning more than one second aborts unconditionally. That
|
||||
token is still useful as an observation stamp, so it is left exactly as it is;
|
||||
#980 gets its own :func:`registry_fingerprint`, derived *only* from canonical
|
||||
retirement-relevant row content:
|
||||
|
||||
* identical registry contents observed at any two times produce the same token,
|
||||
* row order never affects the token (serialized rows are sorted),
|
||||
* any create/delete/identity/liveness/ownership/registration-state change to a
|
||||
retirement-relevant field changes the token.
|
||||
|
||||
Fail-closed posture
|
||||
-------------------
|
||||
|
||||
A worker is retired only when the control plane *conclusively* establishes it
|
||||
is a stale orphan. Missing evidence is never read as permission: an unprobeable
|
||||
PID, an unparsable heartbeat, a row that shares identity evidence with a live
|
||||
or unprobeable worker, a foreign or absent repository binding, or a worker that
|
||||
still owns an active workflow lease all preserve the row.
|
||||
|
||||
Trusted launcher identity (``inst-…`` provenance) is deliberately *not* part of
|
||||
the conjunction. #980 places trusted ``client_instance_id`` propagation out of
|
||||
scope and lists "backfilling trusted identity for legacy workers" as a non-goal;
|
||||
requiring it here would preserve every legacy row forever and make the feature
|
||||
inert. What *is* required is that the registry fields the conjunction reads are
|
||||
actually present — see :data:`REQUIRED_IDENTITY_FIELDS`.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
from datetime import datetime
|
||||
from typing import Any, Callable, Iterable, Mapping, Sequence
|
||||
|
||||
import mcp_fleet_snapshot as fleet
|
||||
import mcp_worker_identity as mwi
|
||||
|
||||
# --- Outcomes -------------------------------------------------------------
|
||||
|
||||
OUTCOME_PLANNED = "planned"
|
||||
OUTCOME_APPLIED = "applied"
|
||||
OUTCOME_REGISTRY_MOVED = "registry_revision_moved"
|
||||
OUTCOME_CANDIDATES_MOVED = "candidate_set_moved"
|
||||
OUTCOME_ALREADY_RETIRED = "already_retired"
|
||||
OUTCOME_NOTHING_REQUESTED = "nothing_requested"
|
||||
|
||||
# --- Reason codes ---------------------------------------------------------
|
||||
|
||||
#: The only reason code that authorizes retirement.
|
||||
REASON_ELIGIBLE = "eligible_stale_orphan"
|
||||
|
||||
REASON_ALREADY_TERMINAL = "already_terminal_registration"
|
||||
REASON_AMBIGUOUS_OWNERSHIP = "ambiguous_ownership_state"
|
||||
REASON_CONFLICTING_IDENTITY = "conflicting_identity_evidence"
|
||||
REASON_FOREIGN_REPOSITORY = "repository_binding_ambiguous"
|
||||
REASON_HEARTBEAT_FRESH = "heartbeat_not_expired"
|
||||
REASON_INCOMPLETE_IDENTITY = "incomplete_registry_identity"
|
||||
REASON_NOT_IN_PLAN = "not_in_current_plan"
|
||||
REASON_PID_ALIVE = "pid_alive"
|
||||
REASON_PID_UNKNOWN = "pid_liveness_unknown"
|
||||
REASON_PROTECTED_OWNER = "protected_active_workflow_owner"
|
||||
REASON_ROW_CHANGED = "row_changed_since_plan"
|
||||
REASON_ROW_MISSING = "registration_missing"
|
||||
REASON_UNPARSABLE_HEARTBEAT = "unparsable_heartbeat"
|
||||
REASON_WORKER_LIVE = "worker_live"
|
||||
|
||||
#: Registry columns that must carry a usable value before the eligibility
|
||||
#: conjunction can even be evaluated. Absence is ambiguity, not permission.
|
||||
REQUIRED_IDENTITY_FIELDS: tuple[str, ...] = (
|
||||
"worker_identity",
|
||||
"client_instance_id",
|
||||
"session_id",
|
||||
"generation_id",
|
||||
"status",
|
||||
"started_at",
|
||||
"last_heartbeat_at",
|
||||
"heartbeat_ttl_seconds",
|
||||
"pid",
|
||||
)
|
||||
|
||||
#: Canonical retirement-relevant content. Ordering here is fixed and part of
|
||||
#: the token contract; adding a field changes every fingerprint, so a change
|
||||
#: here is a deliberate contract revision.
|
||||
#:
|
||||
#: Deliberately excluded: ``token_fingerprint`` (credential-adjacent, never a
|
||||
#: retirement input), the four ``*_revision`` columns (revision drift is an
|
||||
#: independent restart concern and is not part of the eligibility conjunction),
|
||||
#: and the ``retired_*`` bookkeeping columns this feature adds.
|
||||
FINGERPRINT_FIELDS: tuple[str, ...] = (
|
||||
"worker_identity",
|
||||
"client_name",
|
||||
"client_instance_id",
|
||||
"session_id",
|
||||
"generation_id",
|
||||
"role",
|
||||
"profile",
|
||||
"namespace",
|
||||
"remote",
|
||||
"repository_binding",
|
||||
"pid",
|
||||
"process_identity",
|
||||
"transport",
|
||||
"started_at",
|
||||
"last_heartbeat_at",
|
||||
"heartbeat_ttl_seconds",
|
||||
"fencing_epoch",
|
||||
"status",
|
||||
"fleet_run_id",
|
||||
"authenticated_account",
|
||||
"instance_id_provenance",
|
||||
)
|
||||
|
||||
_FINGERPRINT_VERSION = "registryfp-v1"
|
||||
_CANDIDATE_VERSION = "candidatefp-v1"
|
||||
_UNIT = "\x1f"
|
||||
_RECORD = "\x1e"
|
||||
|
||||
|
||||
def _canon(value: Any) -> str:
|
||||
"""Stable text for one field value, independent of Python/SQLite typing.
|
||||
|
||||
``900`` and ``900.0`` are the same TTL and must hash the same; a value that
|
||||
round-trips through SQLite as REAL must not produce a different token than
|
||||
the same value supplied by a caller as ``int``.
|
||||
"""
|
||||
if value is None:
|
||||
return ""
|
||||
if isinstance(value, bool):
|
||||
return "true" if value else "false"
|
||||
if isinstance(value, float):
|
||||
if value != value or value in (float("inf"), float("-inf")):
|
||||
return repr(value)
|
||||
if value.is_integer():
|
||||
return str(int(value))
|
||||
return repr(value)
|
||||
if isinstance(value, int):
|
||||
return str(value)
|
||||
return str(value)
|
||||
|
||||
|
||||
def _serialize_row(row: Mapping[str, Any]) -> str:
|
||||
return _UNIT.join(f"{name}={_canon(row.get(name))}" for name in FINGERPRINT_FIELDS)
|
||||
|
||||
|
||||
def _digest(version: str, serialized: Sequence[str], prefix: str) -> str:
|
||||
ordered = sorted(serialized)
|
||||
material = _RECORD.join([version, str(len(ordered)), *ordered])
|
||||
return f"{prefix}-{hashlib.sha256(material.encode('utf-8')).hexdigest()[:32]}"
|
||||
|
||||
|
||||
def registry_fingerprint(rows: Iterable[Mapping[str, Any]]) -> str:
|
||||
"""Content-derived compare-and-swap token for the worker registry (#980).
|
||||
|
||||
Derived exclusively from :data:`FINGERPRINT_FIELDS` across every row. It
|
||||
contains no ``snapshot_at``, wall-clock, request, or report-generation
|
||||
time, so two observations of an unchanged registry always agree, and the
|
||||
serialized rows are sorted so iteration order cannot perturb the digest.
|
||||
"""
|
||||
return _digest(
|
||||
_FINGERPRINT_VERSION,
|
||||
[_serialize_row(row) for row in rows],
|
||||
"registryfp",
|
||||
)
|
||||
|
||||
|
||||
def candidate_fingerprint(candidate_rows: Iterable[Mapping[str, Any]]) -> str:
|
||||
"""Exact-candidate-set token over the selected rows' canonical content.
|
||||
|
||||
A matching :func:`registry_fingerprint` already implies these rows are
|
||||
unchanged; this second token additionally pins *which* rows the operator
|
||||
approved, so an apply can never widen or narrow the approved set.
|
||||
"""
|
||||
return _digest(
|
||||
_CANDIDATE_VERSION,
|
||||
[_serialize_row(row) for row in candidate_rows],
|
||||
"candidatefp",
|
||||
)
|
||||
|
||||
|
||||
def _probe_pid(
|
||||
pid: Any, pid_alive_probe: Callable[[int | None], bool | None] | None
|
||||
) -> bool | None:
|
||||
if pid_alive_probe is None or pid is None:
|
||||
return None
|
||||
try:
|
||||
result = pid_alive_probe(pid)
|
||||
except Exception:
|
||||
return None
|
||||
return None if result is None else bool(result)
|
||||
|
||||
|
||||
def _evidence(
|
||||
row: Mapping[str, Any],
|
||||
snapshot_row: Mapping[str, Any],
|
||||
pid_alive: bool | None,
|
||||
) -> dict[str, Any]:
|
||||
liveness = snapshot_row.get("liveness") or {}
|
||||
return {
|
||||
"worker_identity": row.get("worker_identity"),
|
||||
"client_type": snapshot_row.get("client_type"),
|
||||
"client_instance_id": row.get("client_instance_id"),
|
||||
"fleet_run_id": row.get("fleet_run_id"),
|
||||
"namespace": row.get("namespace"),
|
||||
"profile": row.get("profile"),
|
||||
"declared_role": row.get("role"),
|
||||
"session_id": row.get("session_id"),
|
||||
"generation_id": row.get("generation_id"),
|
||||
"fencing_epoch": row.get("fencing_epoch"),
|
||||
"process_identity": snapshot_row.get("process_identity"),
|
||||
"pid": row.get("pid"),
|
||||
"pid_alive": pid_alive,
|
||||
"repository_binding": row.get("repository_binding"),
|
||||
"foreign_repository": bool(snapshot_row.get("foreign_repository")),
|
||||
"status": row.get("status"),
|
||||
"started_at": row.get("started_at"),
|
||||
"last_heartbeat_at": row.get("last_heartbeat_at"),
|
||||
"heartbeat_ttl_seconds": row.get("heartbeat_ttl_seconds"),
|
||||
"heartbeat_age_seconds": liveness.get("heartbeat_age_seconds"),
|
||||
"heartbeat_fresh": liveness.get("heartbeat_fresh"),
|
||||
"live": bool(snapshot_row.get("live")),
|
||||
"ownership_state": snapshot_row.get("ownership_state"),
|
||||
"instance_id_provenance": snapshot_row.get("instance_id_provenance"),
|
||||
"instance_identity_trusted": bool(
|
||||
snapshot_row.get("instance_identity_trusted")
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def _conflict_keys(
|
||||
row: Mapping[str, Any], snapshot_row: Mapping[str, Any]
|
||||
) -> list[tuple[str, str]]:
|
||||
keys: list[tuple[str, str]] = []
|
||||
for name, value in (
|
||||
("session_id", row.get("session_id")),
|
||||
("generation_id", row.get("generation_id")),
|
||||
("process_identity", snapshot_row.get("process_identity")),
|
||||
("pid", row.get("pid")),
|
||||
):
|
||||
text = _canon(value)
|
||||
if text:
|
||||
keys.append((name, text))
|
||||
return keys
|
||||
|
||||
|
||||
def _missing_identity_fields(row: Mapping[str, Any]) -> list[str]:
|
||||
missing: list[str] = []
|
||||
for name in REQUIRED_IDENTITY_FIELDS:
|
||||
value = row.get(name)
|
||||
if value is None or (isinstance(value, str) and not value.strip()):
|
||||
missing.append(name)
|
||||
return missing
|
||||
|
||||
|
||||
def plan_stale_worker_retirement(
|
||||
rows: Iterable[Mapping[str, Any]],
|
||||
*,
|
||||
now: datetime | None = None,
|
||||
pid_alive_probe: Callable[[int | None], bool | None] | None = None,
|
||||
canonical_repository: str | None = None,
|
||||
protected_worker_identities: Iterable[str] | None = None,
|
||||
protected_session_ids: Iterable[str] | None = None,
|
||||
protected_pids: Iterable[Any] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Decide, without mutating anything, which registrations may be retired.
|
||||
|
||||
Every row lands in exactly one of ``candidates`` (eligible) or
|
||||
``preserved`` (with the reason code that stopped it), so the output
|
||||
explains the whole registry rather than only the interesting part.
|
||||
"""
|
||||
all_rows = [dict(row) for row in rows]
|
||||
protected_ids = {str(w) for w in (protected_worker_identities or []) if w}
|
||||
protected_sessions = {str(s) for s in (protected_session_ids or []) if s}
|
||||
protected_pid_set = {_canon(p) for p in (protected_pids or []) if p is not None}
|
||||
|
||||
snapshots: dict[int, dict[str, Any]] = {}
|
||||
pid_alive_by_index: dict[int, bool | None] = {}
|
||||
for index, row in enumerate(all_rows):
|
||||
pid_alive = _probe_pid(row.get("pid"), pid_alive_probe)
|
||||
pid_alive_by_index[index] = pid_alive
|
||||
snapshots[index] = fleet.build_worker_snapshot_row(
|
||||
row,
|
||||
now=now,
|
||||
pid_alive_probe=(lambda _pid, _value=pid_alive: _value),
|
||||
canonical_repository=canonical_repository,
|
||||
)
|
||||
|
||||
# Identity evidence owned by a worker that is live, or whose liveness could
|
||||
# not be established, is ambiguous: anything sharing it is preserved.
|
||||
ambiguous_keys: set[tuple[str, str]] = set()
|
||||
identity_counts: dict[str, int] = {}
|
||||
for index, row in enumerate(all_rows):
|
||||
identity = _canon(row.get("worker_identity"))
|
||||
if identity:
|
||||
identity_counts[identity] = identity_counts.get(identity, 0) + 1
|
||||
snapshot_row = snapshots[index]
|
||||
liveness = snapshot_row.get("liveness") or {}
|
||||
unresolved = (
|
||||
pid_alive_by_index[index] is None
|
||||
or liveness.get("heartbeat_fresh") is None
|
||||
)
|
||||
if snapshot_row.get("live") or (
|
||||
str(row.get("status") or "") == mwi.STATUS_ACTIVE and unresolved
|
||||
):
|
||||
ambiguous_keys.update(_conflict_keys(row, snapshot_row))
|
||||
|
||||
candidates: list[dict[str, Any]] = []
|
||||
candidate_rows: list[Mapping[str, Any]] = []
|
||||
preserved: list[dict[str, Any]] = []
|
||||
|
||||
for index, row in enumerate(all_rows):
|
||||
snapshot_row = snapshots[index]
|
||||
pid_alive = pid_alive_by_index[index]
|
||||
liveness = snapshot_row.get("liveness") or {}
|
||||
evidence = _evidence(row, snapshot_row, pid_alive)
|
||||
blocked: tuple[str, str] | None = None
|
||||
|
||||
identity = _canon(row.get("worker_identity"))
|
||||
missing = _missing_identity_fields(row)
|
||||
shared = sorted(
|
||||
f"{name}={value}"
|
||||
for name, value in _conflict_keys(row, snapshot_row)
|
||||
if (name, value) in ambiguous_keys
|
||||
)
|
||||
binding = (row.get("repository_binding") or "").strip()
|
||||
protected_hits: list[str] = []
|
||||
if identity and identity in protected_ids:
|
||||
protected_hits.append(f"worker_identity={identity}")
|
||||
if _canon(row.get("session_id")) in protected_sessions:
|
||||
protected_hits.append(f"session_id={_canon(row.get('session_id'))}")
|
||||
if _canon(row.get("pid")) in protected_pid_set:
|
||||
protected_hits.append(f"pid={_canon(row.get('pid'))}")
|
||||
|
||||
if identity and identity_counts.get(identity, 0) > 1:
|
||||
blocked = (
|
||||
REASON_CONFLICTING_IDENTITY,
|
||||
f"worker identity {identity!r} appears on more than one registry row",
|
||||
)
|
||||
elif str(row.get("status") or "") != mwi.STATUS_ACTIVE:
|
||||
blocked = (
|
||||
REASON_ALREADY_TERMINAL,
|
||||
f"registration status is {row.get('status')!r}; nothing to retire",
|
||||
)
|
||||
elif missing:
|
||||
blocked = (
|
||||
REASON_INCOMPLETE_IDENTITY,
|
||||
"registry row is missing field(s) the retirement conjunction "
|
||||
f"reads: {missing}",
|
||||
)
|
||||
elif mwi._parse_ts(row.get("last_heartbeat_at")) is None:
|
||||
blocked = (
|
||||
REASON_UNPARSABLE_HEARTBEAT,
|
||||
"last_heartbeat_at is not a parsable UTC stamp; liveness is unknown",
|
||||
)
|
||||
elif snapshot_row.get("live"):
|
||||
blocked = (REASON_WORKER_LIVE, "worker is live and must not be retired")
|
||||
elif pid_alive is None:
|
||||
blocked = (
|
||||
REASON_PID_UNKNOWN,
|
||||
f"pid {row.get('pid')!r} could not be probed; liveness is unproven",
|
||||
)
|
||||
elif pid_alive:
|
||||
blocked = (
|
||||
REASON_PID_ALIVE,
|
||||
f"recorded pid {row.get('pid')!r} is still running",
|
||||
)
|
||||
elif liveness.get("heartbeat_fresh") is not False:
|
||||
blocked = (
|
||||
REASON_HEARTBEAT_FRESH,
|
||||
"heartbeat has not expired under the canonical TTL policy",
|
||||
)
|
||||
elif snapshot_row.get("ownership_state") != "stale":
|
||||
blocked = (
|
||||
REASON_AMBIGUOUS_OWNERSHIP,
|
||||
"ownership_state is "
|
||||
f"{snapshot_row.get('ownership_state')!r}, not 'stale'",
|
||||
)
|
||||
elif not binding or snapshot_row.get("foreign_repository"):
|
||||
blocked = (
|
||||
REASON_FOREIGN_REPOSITORY,
|
||||
"repository binding is absent or does not match the canonical "
|
||||
"repository; retirement scope is ambiguous",
|
||||
)
|
||||
elif shared:
|
||||
blocked = (
|
||||
REASON_CONFLICTING_IDENTITY,
|
||||
"identity evidence is shared with a live or unprobeable worker: "
|
||||
f"{shared}",
|
||||
)
|
||||
elif protected_hits:
|
||||
blocked = (
|
||||
REASON_PROTECTED_OWNER,
|
||||
"worker still owns active workflow state requiring separate "
|
||||
f"reconciliation: {protected_hits}",
|
||||
)
|
||||
|
||||
if blocked is not None:
|
||||
preserved.append(
|
||||
{
|
||||
"worker_identity": row.get("worker_identity"),
|
||||
"reason_code": blocked[0],
|
||||
"detail": blocked[1],
|
||||
"evidence": evidence,
|
||||
}
|
||||
)
|
||||
continue
|
||||
|
||||
candidates.append(
|
||||
{
|
||||
"worker_identity": row.get("worker_identity"),
|
||||
"reason_code": REASON_ELIGIBLE,
|
||||
"detail": (
|
||||
"dead pid, expired heartbeat, stale ownership, unambiguous "
|
||||
"identity, canonical repository binding, no active workflow "
|
||||
"ownership"
|
||||
),
|
||||
"evidence": evidence,
|
||||
}
|
||||
)
|
||||
candidate_rows.append(row)
|
||||
|
||||
counts: dict[str, int] = {}
|
||||
for entry in preserved:
|
||||
counts[entry["reason_code"]] = counts.get(entry["reason_code"], 0) + 1
|
||||
|
||||
return {
|
||||
"success": True,
|
||||
"read_only": True,
|
||||
"mutation_performed": False,
|
||||
"outcome": OUTCOME_PLANNED,
|
||||
"registry_fingerprint": registry_fingerprint(all_rows),
|
||||
"candidate_fingerprint": candidate_fingerprint(candidate_rows),
|
||||
"assessed_count": len(all_rows),
|
||||
"candidate_count": len(candidates),
|
||||
"preserved_count": len(preserved),
|
||||
"candidates": candidates,
|
||||
"candidate_worker_identities": [c["worker_identity"] for c in candidates],
|
||||
"preserved": preserved,
|
||||
"preserved_reason_counts": counts,
|
||||
"protected_inputs": {
|
||||
"worker_identities": sorted(protected_ids),
|
||||
"session_ids": sorted(protected_sessions),
|
||||
"pids": sorted(protected_pid_set),
|
||||
},
|
||||
"canonical_repository": canonical_repository,
|
||||
}
|
||||
|
||||
|
||||
def summarize_plan(plan: Mapping[str, Any]) -> dict[str, Any]:
|
||||
"""Compact, log-safe view of a plan or apply result."""
|
||||
return {
|
||||
"outcome": plan.get("outcome"),
|
||||
"registry_fingerprint": plan.get("registry_fingerprint"),
|
||||
"candidate_fingerprint": plan.get("candidate_fingerprint"),
|
||||
"assessed_count": plan.get("assessed_count"),
|
||||
"candidate_count": plan.get("candidate_count"),
|
||||
"retired_count": plan.get("retired_count"),
|
||||
"preserved_count": plan.get("preserved_count"),
|
||||
"mutation_performed": plan.get("mutation_performed"),
|
||||
}
|
||||
@@ -0,0 +1,869 @@
|
||||
"""Instance-level fleet identity and health snapshots (#978).
|
||||
|
||||
#948 established per-worker ownership; #975 made heartbeats keep those rows
|
||||
live. Neither surface could enumerate the fleet at *instance* granularity:
|
||||
which application launch owns which five namespace workers, whether two
|
||||
Codex launches are distinct, or whether a live collision is real rather than
|
||||
a shared client type.
|
||||
|
||||
This module is pure. Callers supply registry rows (and optional enrichments);
|
||||
nothing here opens SQLite, scans process tables, or mutates state. Production
|
||||
evidence for the fleet gate is the snapshot returned by the sanctioned
|
||||
controller/reconciler tool that wraps this assessor.
|
||||
|
||||
Identity hierarchy (highest → lowest):
|
||||
|
||||
* ``client_type`` — application family (``codex``, ``claude_code``, …)
|
||||
* ``client_instance_id`` — one running application launch (trusted launcher)
|
||||
* ``fleet_run_id`` — operator-approved enrollment / canary cohort
|
||||
* ``worker_id`` / ``worker_identity`` — one namespace worker process
|
||||
* ``namespace`` — author | reviewer | merger | controller | reconciler
|
||||
|
||||
Multiple simultaneous instances of the same ``client_type`` are first-class.
|
||||
Sharing only a profile or client type is never a duplicate.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import secrets
|
||||
from collections import defaultdict
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Callable, Iterable, Mapping
|
||||
|
||||
import mcp_worker_identity as mwi
|
||||
|
||||
# --- Classification labels ------------------------------------------------
|
||||
|
||||
CLASS_EXPECTED = "expected_enrolled"
|
||||
CLASS_MISSING = "missing_expected"
|
||||
CLASS_UNMANIFESTED = "unmanifested"
|
||||
CLASS_DUPLICATE_NAMESPACE = "duplicate_namespace_worker"
|
||||
CLASS_INSTANCE_ID_COLLISION = "instance_id_collision"
|
||||
CLASS_WORKER_ID_COLLISION = "worker_identity_collision"
|
||||
CLASS_SESSION_COLLISION = "session_identity_collision"
|
||||
CLASS_GENERATION_COLLISION = "generation_identity_collision"
|
||||
CLASS_PROCESS_COLLISION = "process_identity_collision"
|
||||
CLASS_PID_COLLISION = "pid_collision"
|
||||
CLASS_OWNERSHIP_COLLISION = "ownership_fencing_collision"
|
||||
CLASS_ORPHANED = "orphaned_unowned"
|
||||
CLASS_UNKNOWN_CLIENT = "unknown_client"
|
||||
CLASS_FOREIGN_REPOSITORY = "foreign_repository"
|
||||
CLASS_OLD_REVISION = "old_revision"
|
||||
CLASS_STALE_WORKER = "stale_orphaned_worker"
|
||||
CLASS_LEGACY_INCOMPLETE = "legacy_incomplete_identity"
|
||||
CLASS_HISTORICAL = "historical_dead"
|
||||
CLASS_HEALTHY = "healthy"
|
||||
|
||||
#: Active blockers that make the live fleet unsafe for mutation-gated work.
|
||||
ACTIVE_BLOCKER_CLASSES = frozenset(
|
||||
{
|
||||
CLASS_MISSING,
|
||||
CLASS_UNMANIFESTED,
|
||||
CLASS_DUPLICATE_NAMESPACE,
|
||||
CLASS_INSTANCE_ID_COLLISION,
|
||||
CLASS_WORKER_ID_COLLISION,
|
||||
CLASS_SESSION_COLLISION,
|
||||
CLASS_GENERATION_COLLISION,
|
||||
CLASS_PROCESS_COLLISION,
|
||||
CLASS_PID_COLLISION,
|
||||
CLASS_OWNERSHIP_COLLISION,
|
||||
CLASS_ORPHANED,
|
||||
CLASS_UNKNOWN_CLIENT,
|
||||
CLASS_FOREIGN_REPOSITORY,
|
||||
CLASS_OLD_REVISION,
|
||||
CLASS_STALE_WORKER,
|
||||
CLASS_LEGACY_INCOMPLETE,
|
||||
}
|
||||
)
|
||||
|
||||
SANCTIONED_NAMESPACES = frozenset(
|
||||
{"author", "reviewer", "merger", "controller", "reconciler"}
|
||||
)
|
||||
|
||||
INSTANCE_ID_PROVENANCE_TRUSTED = "trusted_launcher"
|
||||
INSTANCE_ID_PROVENANCE_LEGACY = "legacy_incomplete"
|
||||
INSTANCE_ID_PROVENANCE_MISSING = "missing"
|
||||
|
||||
CLIENT_INSTANCE_ENV = "GITEA_MCP_CLIENT_INSTANCE"
|
||||
FLEET_RUN_ENV = "GITEA_MCP_FLEET_RUN_ID"
|
||||
PROCESS_IDENTITY_ENV = "GITEA_MCP_PROCESS_IDENTITY"
|
||||
INSTANCE_PROVENANCE_ENV = "GITEA_MCP_INSTANCE_PROVENANCE"
|
||||
|
||||
_LEGACY_INSTANCE_PREFIXES = ("pid-", "proc-", "legacy-")
|
||||
#: Trusted launcher-issued IDs use the reserved ``inst-`` prefix
|
||||
#: (see :func:`generate_client_instance_id`). Ordinary user-supplied strings
|
||||
#: without that prefix never count as trusted attribution.
|
||||
_TRUSTED_INSTANCE_PREFIX = "inst-"
|
||||
|
||||
|
||||
def _utc_now() -> datetime:
|
||||
return datetime.now(timezone.utc)
|
||||
|
||||
|
||||
def _ts(value: datetime) -> str:
|
||||
return value.astimezone(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||
|
||||
|
||||
def generate_client_instance_id(
|
||||
client_type: str | None,
|
||||
*,
|
||||
launch_nonce: str | None = None,
|
||||
now: datetime | None = None,
|
||||
) -> str:
|
||||
"""Mint a distinct instance ID for one application launch (#978).
|
||||
|
||||
The trusted launcher (or host that starts all five namespace workers)
|
||||
generates this once per launch and injects it as ``GITEA_MCP_CLIENT_INSTANCE``
|
||||
into every worker environment. Workers never invent their own instance ID
|
||||
from PID proximity or timestamps.
|
||||
"""
|
||||
client = mwi.normalize_client_name(client_type)
|
||||
stamp = (now or _utc_now()).astimezone(timezone.utc).strftime(
|
||||
mwi.IDENTITY_TIMESTAMP_FORMAT
|
||||
)
|
||||
nonce = launch_nonce if launch_nonce is not None else secrets.token_hex(16)
|
||||
digest = hashlib.sha256(
|
||||
f"{client}\x1f{stamp}\x1f{nonce}".encode("utf-8")
|
||||
).hexdigest()[:12]
|
||||
return f"inst-{client}-{stamp}-{digest}"
|
||||
|
||||
|
||||
def assess_instance_identity(
|
||||
raw_instance_id: str | None,
|
||||
*,
|
||||
source: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Classify whether a client_instance_id is trusted enough for mutation.
|
||||
|
||||
Trusted instance IDs are non-empty, match the launcher-minted ``inst-…``
|
||||
format from :func:`generate_client_instance_id`, and are not pre-#978
|
||||
PID/proc/legacy placeholders. Ordinary user-supplied or malformed values
|
||||
fail closed as untrusted so they cannot spoof multi-instance attribution.
|
||||
Incomplete identities remain visible for diagnosis but cannot authorize
|
||||
unsafe mutation.
|
||||
"""
|
||||
text = (raw_instance_id or "").strip()
|
||||
if not text:
|
||||
return {
|
||||
"client_instance_id": None,
|
||||
"complete": False,
|
||||
"trusted": False,
|
||||
"provenance": INSTANCE_ID_PROVENANCE_MISSING,
|
||||
"reasons": [
|
||||
"client_instance_id is missing; the trusted launcher must set "
|
||||
f"{CLIENT_INSTANCE_ENV} once per application launch"
|
||||
],
|
||||
}
|
||||
lowered = text.lower()
|
||||
if lowered.startswith(_LEGACY_INSTANCE_PREFIXES) or source == "pid_fallback":
|
||||
return {
|
||||
"client_instance_id": text,
|
||||
"complete": False,
|
||||
"trusted": False,
|
||||
"provenance": INSTANCE_ID_PROVENANCE_LEGACY,
|
||||
"reasons": [
|
||||
f"client_instance_id {text!r} is a legacy PID/process fallback, "
|
||||
"not a trusted launcher-issued instance identity"
|
||||
],
|
||||
}
|
||||
if not text.startswith(_TRUSTED_INSTANCE_PREFIX) or len(text) <= len(
|
||||
_TRUSTED_INSTANCE_PREFIX
|
||||
):
|
||||
return {
|
||||
"client_instance_id": text,
|
||||
"complete": False,
|
||||
"trusted": False,
|
||||
"provenance": INSTANCE_ID_PROVENANCE_LEGACY,
|
||||
"reasons": [
|
||||
f"client_instance_id {text!r} is malformed or user-supplied and "
|
||||
f"does not use the trusted launcher prefix "
|
||||
f"{_TRUSTED_INSTANCE_PREFIX!r}; refusing trusted attribution"
|
||||
],
|
||||
}
|
||||
return {
|
||||
"client_instance_id": text,
|
||||
"complete": True,
|
||||
"trusted": True,
|
||||
"provenance": INSTANCE_ID_PROVENANCE_TRUSTED,
|
||||
"reasons": [],
|
||||
}
|
||||
|
||||
|
||||
def resolve_client_instance_from_env(
|
||||
env: Mapping[str, str] | None = None,
|
||||
*,
|
||||
pid: int | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Resolve instance identity from launcher env without inventing one.
|
||||
|
||||
When the trusted key is absent, return incomplete evidence rather than a
|
||||
silent ``pid-<n>`` identity. Callers that still need a non-empty registry
|
||||
key may choose a legacy placeholder deliberately; they must not treat it as
|
||||
trusted.
|
||||
"""
|
||||
source = dict(env or {})
|
||||
raw = (source.get(CLIENT_INSTANCE_ENV) or "").strip()
|
||||
assessment = assess_instance_identity(raw or None)
|
||||
assessment["fleet_run_id"] = (source.get(FLEET_RUN_ENV) or "").strip() or None
|
||||
assessment["process_identity"] = (
|
||||
(source.get(PROCESS_IDENTITY_ENV) or "").strip()
|
||||
or (f"pid-{pid}" if pid is not None else None)
|
||||
)
|
||||
return assessment
|
||||
|
||||
|
||||
def _public_worker(record: Mapping[str, Any]) -> dict[str, Any]:
|
||||
return {
|
||||
"worker_id": record.get("worker_identity") or record.get("worker_id"),
|
||||
"worker_identity": record.get("worker_identity") or record.get("worker_id"),
|
||||
"client_type": mwi.normalize_client_name(
|
||||
record.get("client_name") or record.get("client_type")
|
||||
),
|
||||
"client_instance_id": record.get("client_instance_id"),
|
||||
"fleet_run_id": record.get("fleet_run_id"),
|
||||
"namespace": record.get("namespace"),
|
||||
"profile": record.get("profile"),
|
||||
"declared_role": record.get("role") or record.get("declared_role"),
|
||||
"authenticated_account": record.get("authenticated_account"),
|
||||
"session_id": record.get("session_id"),
|
||||
"generation_id": record.get("generation_id"),
|
||||
"process_identity": record.get("process_identity")
|
||||
or (
|
||||
f"pid-{record['pid']}"
|
||||
if record.get("pid") is not None
|
||||
else None
|
||||
),
|
||||
"pid": record.get("pid"),
|
||||
"repository_binding": record.get("repository_binding"),
|
||||
"remote": record.get("remote"),
|
||||
"startup_revision": record.get("startup_revision"),
|
||||
"loaded_revision": record.get("loaded_revision"),
|
||||
"parity_revision": record.get("parity_revision"),
|
||||
"live_revision": record.get("live_revision"),
|
||||
"runtime_provenance": record.get("runtime_provenance")
|
||||
or record.get("transport"),
|
||||
"transport": record.get("transport"),
|
||||
"started_at": record.get("started_at"),
|
||||
"last_heartbeat_at": record.get("last_heartbeat_at"),
|
||||
"heartbeat_ttl_seconds": record.get("heartbeat_ttl_seconds"),
|
||||
"fencing_epoch": record.get("fencing_epoch"),
|
||||
"status": record.get("status"),
|
||||
"instance_id_provenance": record.get("instance_id_provenance"),
|
||||
}
|
||||
|
||||
|
||||
def _liveness(
|
||||
record: Mapping[str, Any],
|
||||
*,
|
||||
now: datetime | None,
|
||||
pid_alive_probe: Callable[[int | None], bool | None] | None,
|
||||
) -> dict[str, Any]:
|
||||
pid_alive = None
|
||||
if pid_alive_probe is not None and record.get("pid") is not None:
|
||||
try:
|
||||
pid_alive = pid_alive_probe(record.get("pid"))
|
||||
except Exception:
|
||||
pid_alive = None
|
||||
return mwi.WorkerRegistry.is_live(record, now=now, pid_alive=pid_alive)
|
||||
|
||||
|
||||
def _consistency_token(rows: Iterable[Mapping[str, Any]], snapshot_at: str) -> str:
|
||||
material = [snapshot_at]
|
||||
for row in sorted(
|
||||
rows,
|
||||
key=lambda r: (
|
||||
str(r.get("worker_identity") or ""),
|
||||
str(r.get("last_heartbeat_at") or ""),
|
||||
str(r.get("fencing_epoch") or ""),
|
||||
),
|
||||
):
|
||||
material.append(
|
||||
"|".join(
|
||||
[
|
||||
str(row.get("worker_identity") or ""),
|
||||
str(row.get("client_instance_id") or ""),
|
||||
str(row.get("status") or ""),
|
||||
str(row.get("last_heartbeat_at") or ""),
|
||||
str(row.get("fencing_epoch") or ""),
|
||||
str(row.get("generation_id") or ""),
|
||||
]
|
||||
)
|
||||
)
|
||||
digest = hashlib.sha256("\n".join(material).encode("utf-8")).hexdigest()[:16]
|
||||
return f"fleetrev-{digest}"
|
||||
|
||||
|
||||
def build_worker_snapshot_row(
|
||||
record: Mapping[str, Any],
|
||||
*,
|
||||
now: datetime | None = None,
|
||||
pid_alive_probe: Callable[[int | None], bool | None] | None = None,
|
||||
canonical_repository: str | None = None,
|
||||
expected_live_revision: str | None = None,
|
||||
heartbeat_supervised: bool | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""One point-in-time worker row for the fleet snapshot."""
|
||||
stamp = now or _utc_now()
|
||||
base = _public_worker(record)
|
||||
identity = assess_instance_identity(
|
||||
base.get("client_instance_id"),
|
||||
source=record.get("instance_id_source"),
|
||||
)
|
||||
liveness = _liveness(record, now=stamp, pid_alive_probe=pid_alive_probe)
|
||||
is_historical = str(record.get("status") or "") != mwi.STATUS_ACTIVE
|
||||
live = bool(liveness.get("live")) and not is_historical
|
||||
|
||||
repo = (base.get("repository_binding") or "").strip() or None
|
||||
foreign_repo = bool(
|
||||
canonical_repository
|
||||
and repo
|
||||
and repo.rstrip("/") != str(canonical_repository).rstrip("/")
|
||||
)
|
||||
old_revision = False
|
||||
if expected_live_revision:
|
||||
for key in ("startup_revision", "loaded_revision", "parity_revision", "live_revision"):
|
||||
rev = (base.get(key) or "").strip()
|
||||
if rev and rev != expected_live_revision:
|
||||
old_revision = True
|
||||
break
|
||||
|
||||
ownership_state = "historical" if is_historical else (
|
||||
"live" if live else "stale"
|
||||
)
|
||||
if live and not identity["trusted"]:
|
||||
ownership_state = "live_untrusted_identity"
|
||||
if live and not base.get("session_id"):
|
||||
ownership_state = "orphaned"
|
||||
|
||||
mutation_safe = bool(
|
||||
live
|
||||
and identity["trusted"]
|
||||
and not foreign_repo
|
||||
and not old_revision
|
||||
and ownership_state == "live"
|
||||
and base.get("client_type") != mwi.UNKNOWN_CLIENT
|
||||
)
|
||||
|
||||
restart_required = bool(
|
||||
old_revision
|
||||
or (live and not liveness.get("heartbeat_fresh", True))
|
||||
)
|
||||
|
||||
return {
|
||||
**base,
|
||||
"instance_identity": identity,
|
||||
"client_instance_id": identity["client_instance_id"] or base.get("client_instance_id"),
|
||||
"instance_id_provenance": identity["provenance"],
|
||||
"instance_identity_trusted": identity["trusted"],
|
||||
"live": live,
|
||||
"historical": is_historical,
|
||||
"liveness": liveness,
|
||||
"heartbeat": {
|
||||
"registered": bool(base.get("last_heartbeat_at")),
|
||||
"supervised": heartbeat_supervised,
|
||||
"age_seconds": liveness.get("heartbeat_age_seconds"),
|
||||
"ttl_seconds": liveness.get("heartbeat_ttl_seconds"),
|
||||
"fresh": liveness.get("heartbeat_fresh"),
|
||||
"last_heartbeat_at": base.get("last_heartbeat_at"),
|
||||
},
|
||||
"fencing": {
|
||||
"fencing_epoch": base.get("fencing_epoch"),
|
||||
"generation_id": base.get("generation_id"),
|
||||
},
|
||||
"ownership_state": ownership_state,
|
||||
"foreign_repository": foreign_repo,
|
||||
"old_revision": old_revision,
|
||||
"stale": not live and not is_historical,
|
||||
"restart_required": restart_required,
|
||||
"mutation_safe": mutation_safe,
|
||||
"conflicting_live_sessions": [],
|
||||
}
|
||||
|
||||
|
||||
def _collision_groups(
|
||||
live_rows: list[dict[str, Any]],
|
||||
key_fn,
|
||||
) -> dict[str, list[dict[str, Any]]]:
|
||||
groups: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
||||
for row in live_rows:
|
||||
key = key_fn(row)
|
||||
if key is None or key == "" or key == "None":
|
||||
continue
|
||||
groups[str(key)].append(row)
|
||||
return {k: v for k, v in groups.items() if len(v) > 1}
|
||||
|
||||
|
||||
def snapshot_instance_fleet(
|
||||
records: Iterable[Mapping[str, Any]],
|
||||
*,
|
||||
expected_manifest: list[Mapping[str, Any]] | None = None,
|
||||
now: datetime | None = None,
|
||||
pid_alive_probe: Callable[[int | None], bool | None] | None = None,
|
||||
canonical_repository: str | None = None,
|
||||
expected_live_revision: str | None = None,
|
||||
registry_revision: str | None = None,
|
||||
known_client_types: Iterable[str] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Authoritative point-in-time fleet snapshot with classification (#978).
|
||||
|
||||
Historical dead rows are reported separately and never automatically make
|
||||
the live fleet unsafe.
|
||||
"""
|
||||
stamp = now or _utc_now()
|
||||
snapshot_at = _ts(stamp)
|
||||
known = {
|
||||
mwi.normalize_client_name(c)
|
||||
for c in (known_client_types or mwi.CLIENT_ALIASES.values())
|
||||
}
|
||||
known.discard(mwi.UNKNOWN_CLIENT)
|
||||
|
||||
all_rows: list[dict[str, Any]] = []
|
||||
for record in records:
|
||||
all_rows.append(
|
||||
build_worker_snapshot_row(
|
||||
record,
|
||||
now=stamp,
|
||||
pid_alive_probe=pid_alive_probe,
|
||||
canonical_repository=canonical_repository,
|
||||
expected_live_revision=expected_live_revision,
|
||||
)
|
||||
)
|
||||
|
||||
live_rows = [r for r in all_rows if r["live"]]
|
||||
historical_rows = [r for r in all_rows if r["historical"]]
|
||||
stale_rows = [r for r in all_rows if r["stale"]]
|
||||
|
||||
# --- identity collisions among live workers ---
|
||||
findings: list[dict[str, Any]] = []
|
||||
|
||||
def _finding(
|
||||
classification: str,
|
||||
*,
|
||||
severity: str,
|
||||
workers: list[dict[str, Any]] | None = None,
|
||||
instance_ids: list[str] | None = None,
|
||||
detail: str,
|
||||
active_blocker: bool,
|
||||
) -> None:
|
||||
findings.append(
|
||||
{
|
||||
"classification": classification,
|
||||
"severity": severity,
|
||||
"active_blocker": active_blocker,
|
||||
"detail": detail,
|
||||
"client_instance_ids": instance_ids or sorted(
|
||||
{
|
||||
str(w.get("client_instance_id"))
|
||||
for w in (workers or [])
|
||||
if w.get("client_instance_id")
|
||||
}
|
||||
),
|
||||
"worker_identities": [
|
||||
w.get("worker_identity") for w in (workers or [])
|
||||
],
|
||||
}
|
||||
)
|
||||
|
||||
# Duplicate worker identity (should not happen with PK, still detect)
|
||||
for wid, group in _collision_groups(
|
||||
live_rows, lambda r: r.get("worker_identity")
|
||||
).items():
|
||||
_finding(
|
||||
CLASS_WORKER_ID_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=f"worker identity {wid!r} is claimed by {len(group)} live workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Reused session identity across live workers
|
||||
for sid, group in _collision_groups(live_rows, lambda r: r.get("session_id")).items():
|
||||
# Same session may appear once; collision only when multiple workers share it
|
||||
# across different worker identities (always true for group size > 1).
|
||||
_finding(
|
||||
CLASS_SESSION_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=f"session identity {sid!r} is reused by {len(group)} live workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Generation claimed by multiple live sessions/workers is a conflict when
|
||||
# the workers are not the five sanctioned namespaces of one instance.
|
||||
for gen, group in _collision_groups(
|
||||
live_rows, lambda r: r.get("generation_id")
|
||||
).items():
|
||||
namespaces = {g.get("namespace") for g in group if g.get("namespace")}
|
||||
instance_ids = {g.get("client_instance_id") for g in group}
|
||||
# Multiple workers under one generation is only valid if they share one
|
||||
# instance and distinct namespaces. Same generation + same namespace = bad.
|
||||
by_ns: dict[str, list] = defaultdict(list)
|
||||
for g in group:
|
||||
by_ns[str(g.get("namespace") or "")].append(g)
|
||||
ns_dups = {ns: rows for ns, rows in by_ns.items() if ns and len(rows) > 1}
|
||||
if ns_dups or len(instance_ids) > 1:
|
||||
_finding(
|
||||
CLASS_GENERATION_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=(
|
||||
f"generation {gen!r} is contested across namespaces/instances "
|
||||
f"(namespaces={sorted(namespaces)}, "
|
||||
f"instances={sorted(str(i) for i in instance_ids if i)})"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Process identity / PID collisions across distinct workers
|
||||
for proc, group in _collision_groups(
|
||||
live_rows, lambda r: r.get("process_identity")
|
||||
).items():
|
||||
if len({r.get("worker_identity") for r in group}) > 1:
|
||||
_finding(
|
||||
CLASS_PROCESS_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=f"process identity {proc!r} is shared by distinct live workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
for pid, group in _collision_groups(live_rows, lambda r: r.get("pid")).items():
|
||||
if len({r.get("worker_identity") for r in group}) > 1:
|
||||
_finding(
|
||||
CLASS_PID_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=f"PID {pid} is shared by distinct live workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Fencing/ownership: same fencing epoch on different workers of different instances
|
||||
for epoch, group in _collision_groups(
|
||||
live_rows,
|
||||
lambda r: (
|
||||
f"{r.get('generation_id')}:{r.get('fencing_epoch')}"
|
||||
if r.get("generation_id") is not None and r.get("fencing_epoch") is not None
|
||||
else None
|
||||
),
|
||||
).items():
|
||||
if len({r.get("client_instance_id") for r in group}) > 1:
|
||||
_finding(
|
||||
CLASS_OWNERSHIP_COLLISION,
|
||||
severity="blocker",
|
||||
workers=group,
|
||||
detail=(
|
||||
f"fencing token {epoch!r} spans more than one client_instance_id"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Per-instance grouping
|
||||
by_instance: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
||||
unkeyed_live: list[dict[str, Any]] = []
|
||||
for row in live_rows:
|
||||
iid = row.get("client_instance_id")
|
||||
if not iid:
|
||||
unkeyed_live.append(row)
|
||||
continue
|
||||
by_instance[str(iid)].append(row)
|
||||
|
||||
instances: list[dict[str, Any]] = []
|
||||
for iid, workers in sorted(by_instance.items()):
|
||||
client_types = sorted({w.get("client_type") for w in workers if w.get("client_type")})
|
||||
trusted = all(w.get("instance_identity_trusted") for w in workers)
|
||||
namespaces = [w.get("namespace") for w in workers]
|
||||
ns_counts: dict[str, int] = defaultdict(int)
|
||||
for ns in namespaces:
|
||||
if ns:
|
||||
ns_counts[str(ns)] += 1
|
||||
dup_ns = sorted(ns for ns, n in ns_counts.items() if n > 1)
|
||||
if dup_ns:
|
||||
_finding(
|
||||
CLASS_DUPLICATE_NAMESPACE,
|
||||
severity="blocker",
|
||||
workers=[w for w in workers if w.get("namespace") in dup_ns],
|
||||
instance_ids=[iid],
|
||||
detail=(
|
||||
f"instance {iid!r} has more than one live worker for "
|
||||
f"namespace(s) {dup_ns}"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Live reuse of one instance ID with incompatible client types
|
||||
if len(client_types) > 1:
|
||||
_finding(
|
||||
CLASS_INSTANCE_ID_COLLISION,
|
||||
severity="blocker",
|
||||
workers=workers,
|
||||
instance_ids=[iid],
|
||||
detail=(
|
||||
f"client_instance_id {iid!r} is live under multiple client "
|
||||
f"types {client_types}"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
if not trusted:
|
||||
_finding(
|
||||
CLASS_LEGACY_INCOMPLETE,
|
||||
severity="blocker",
|
||||
workers=workers,
|
||||
instance_ids=[iid],
|
||||
detail=(
|
||||
f"instance {iid!r} lacks trusted launcher-issued instance "
|
||||
"identity; diagnostic reads remain available"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
unknown = [w for w in workers if w.get("client_type") == mwi.UNKNOWN_CLIENT]
|
||||
if unknown:
|
||||
_finding(
|
||||
CLASS_UNKNOWN_CLIENT,
|
||||
severity="blocker",
|
||||
workers=unknown,
|
||||
instance_ids=[iid],
|
||||
detail=f"instance {iid!r} has worker(s) with unknown client_type",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
foreign = [w for w in workers if w.get("foreign_repository")]
|
||||
if foreign:
|
||||
_finding(
|
||||
CLASS_FOREIGN_REPOSITORY,
|
||||
severity="blocker",
|
||||
workers=foreign,
|
||||
instance_ids=[iid],
|
||||
detail=f"instance {iid!r} has foreign-repository workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
old = [w for w in workers if w.get("old_revision")]
|
||||
if old:
|
||||
_finding(
|
||||
CLASS_OLD_REVISION,
|
||||
severity="blocker",
|
||||
workers=old,
|
||||
instance_ids=[iid],
|
||||
detail=f"instance {iid!r} has old-revision workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
orphans = [w for w in workers if w.get("ownership_state") == "orphaned"]
|
||||
if orphans:
|
||||
_finding(
|
||||
CLASS_ORPHANED,
|
||||
severity="blocker",
|
||||
workers=orphans,
|
||||
instance_ids=[iid],
|
||||
detail=f"instance {iid!r} has orphaned/unowned workers",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
instances.append(
|
||||
{
|
||||
"client_instance_id": iid,
|
||||
"client_types": client_types,
|
||||
"client_type": client_types[0] if len(client_types) == 1 else None,
|
||||
"fleet_run_ids": sorted(
|
||||
{w.get("fleet_run_id") for w in workers if w.get("fleet_run_id")}
|
||||
),
|
||||
"worker_count": len(workers),
|
||||
"namespaces": sorted({n for n in namespaces if n}),
|
||||
"namespace_counts": dict(ns_counts),
|
||||
"duplicate_namespaces": dup_ns,
|
||||
"trusted_instance_identity": trusted,
|
||||
"workers": workers,
|
||||
"mutation_safe": all(w.get("mutation_safe") for w in workers)
|
||||
and not dup_ns
|
||||
and trusted,
|
||||
}
|
||||
)
|
||||
|
||||
for row in unkeyed_live:
|
||||
_finding(
|
||||
CLASS_LEGACY_INCOMPLETE,
|
||||
severity="blocker",
|
||||
workers=[row],
|
||||
detail="live worker has no client_instance_id",
|
||||
active_blocker=True,
|
||||
)
|
||||
if row.get("client_type") == mwi.UNKNOWN_CLIENT:
|
||||
_finding(
|
||||
CLASS_UNKNOWN_CLIENT,
|
||||
severity="blocker",
|
||||
workers=[row],
|
||||
detail="live worker has unknown client_type and no instance id",
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
for row in stale_rows:
|
||||
_finding(
|
||||
CLASS_STALE_WORKER,
|
||||
severity="warning",
|
||||
workers=[row],
|
||||
detail=(
|
||||
f"worker {row.get('worker_identity')!r} is active in the registry "
|
||||
"but not live (stale heartbeat or dead pid)"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
for row in historical_rows:
|
||||
_finding(
|
||||
CLASS_HISTORICAL,
|
||||
severity="info",
|
||||
workers=[row],
|
||||
detail=(
|
||||
f"historical registration {row.get('worker_identity')!r} "
|
||||
f"(status={row.get('status')!r}) is not an active blocker"
|
||||
),
|
||||
active_blocker=False,
|
||||
)
|
||||
|
||||
# Manifest comparison
|
||||
expected = list(expected_manifest or [])
|
||||
expected_ids = {
|
||||
str(item.get("client_instance_id")).strip()
|
||||
for item in expected
|
||||
if (item.get("client_instance_id") or "").strip()
|
||||
}
|
||||
live_ids = set(by_instance.keys())
|
||||
missing_ids = sorted(expected_ids - live_ids)
|
||||
unmanifested_ids = sorted(live_ids - expected_ids) if expected_ids else []
|
||||
|
||||
for iid in missing_ids:
|
||||
_finding(
|
||||
CLASS_MISSING,
|
||||
severity="blocker",
|
||||
instance_ids=[iid],
|
||||
detail=f"expected enrolled instance {iid!r} is missing from the live fleet",
|
||||
active_blocker=True,
|
||||
)
|
||||
for iid in unmanifested_ids:
|
||||
_finding(
|
||||
CLASS_UNMANIFESTED,
|
||||
severity="blocker",
|
||||
instance_ids=[iid],
|
||||
workers=by_instance.get(iid, []),
|
||||
detail=(
|
||||
f"live instance {iid!r} is not on the approved fleet manifest "
|
||||
"(unmanifested)"
|
||||
),
|
||||
active_blocker=True,
|
||||
)
|
||||
|
||||
# Same client_type multi-instance is healthy when each has distinct instance IDs
|
||||
by_type: dict[str, list[str]] = defaultdict(list)
|
||||
for inst in instances:
|
||||
for ct in inst.get("client_types") or []:
|
||||
by_type[str(ct)].append(inst["client_instance_id"])
|
||||
multi_instance_same_type = {
|
||||
ct: ids for ct, ids in by_type.items() if len(ids) > 1
|
||||
}
|
||||
|
||||
active_blockers = [f for f in findings if f.get("active_blocker")]
|
||||
historical_only = [f for f in findings if f.get("classification") == CLASS_HISTORICAL]
|
||||
live_safe = not active_blockers
|
||||
|
||||
consistency = registry_revision or _consistency_token(all_rows, snapshot_at)
|
||||
|
||||
return {
|
||||
"success": True,
|
||||
"read_only": True,
|
||||
"snapshot_at": snapshot_at,
|
||||
"consistency_token": consistency,
|
||||
"registry_revision": consistency,
|
||||
"live_worker_count": len(live_rows),
|
||||
"historical_worker_count": len(historical_rows),
|
||||
"stale_worker_count": len(stale_rows),
|
||||
"instance_count": len(instances),
|
||||
"workers": all_rows,
|
||||
"live_workers": live_rows,
|
||||
"historical_workers": historical_rows,
|
||||
"stale_workers": stale_rows,
|
||||
"instances": instances,
|
||||
"multi_instance_same_client_type": multi_instance_same_type,
|
||||
"same_client_type_not_duplicate": True,
|
||||
"expected_manifest": [
|
||||
{
|
||||
"client_instance_id": item.get("client_instance_id"),
|
||||
"client_type": item.get("client_type"),
|
||||
"fleet_run_id": item.get("fleet_run_id"),
|
||||
"namespaces": item.get("namespaces"),
|
||||
}
|
||||
for item in expected
|
||||
],
|
||||
"missing_expected_instance_ids": missing_ids,
|
||||
"unmanifested_instance_ids": unmanifested_ids,
|
||||
"findings": findings,
|
||||
"active_blockers": active_blockers,
|
||||
"historical_findings": historical_only,
|
||||
"live_fleet_safe": live_safe,
|
||||
"mutation_safe": live_safe and all(
|
||||
inst.get("mutation_safe") for inst in instances
|
||||
)
|
||||
if instances
|
||||
else live_safe,
|
||||
"classification_model": {
|
||||
"exactly_one_process_per_profile": False,
|
||||
"exactly_one_instance_per_client_type": False,
|
||||
"multiple_instances_per_client_type": True,
|
||||
"duplicate_requires": [
|
||||
"live client_instance_id collision",
|
||||
"duplicate namespace worker within one instance",
|
||||
"reused worker/session/generation/process/pid/fencing identity",
|
||||
"cross-instance ownership collision",
|
||||
"unmanifested instance when a manifest is required",
|
||||
],
|
||||
},
|
||||
"reasons": [f["detail"] for f in active_blockers],
|
||||
}
|
||||
|
||||
|
||||
def compare_snapshot_heartbeats(
|
||||
earlier: Mapping[str, Any],
|
||||
later: Mapping[str, Any],
|
||||
) -> dict[str, Any]:
|
||||
"""Prove heartbeat continuity and stable ownership across two snapshots."""
|
||||
earlier_live = {
|
||||
w.get("worker_identity"): w for w in earlier.get("live_workers") or []
|
||||
}
|
||||
later_live = {
|
||||
w.get("worker_identity"): w for w in later.get("live_workers") or []
|
||||
}
|
||||
shared = sorted(set(earlier_live) & set(later_live))
|
||||
continuity: list[dict[str, Any]] = []
|
||||
stable_ownership = True
|
||||
for wid in shared:
|
||||
a = earlier_live[wid]
|
||||
b = later_live[wid]
|
||||
same_instance = a.get("client_instance_id") == b.get("client_instance_id")
|
||||
same_session = a.get("session_id") == b.get("session_id")
|
||||
same_generation = a.get("generation_id") == b.get("generation_id")
|
||||
hb_advanced_or_equal = True
|
||||
if a.get("last_heartbeat_at") and b.get("last_heartbeat_at"):
|
||||
hb_advanced_or_equal = b["last_heartbeat_at"] >= a["last_heartbeat_at"]
|
||||
if not (same_instance and same_session and same_generation):
|
||||
stable_ownership = False
|
||||
continuity.append(
|
||||
{
|
||||
"worker_identity": wid,
|
||||
"same_client_instance_id": same_instance,
|
||||
"same_session_id": same_session,
|
||||
"same_generation_id": same_generation,
|
||||
"heartbeat_non_decreasing": hb_advanced_or_equal,
|
||||
"earlier_heartbeat": a.get("last_heartbeat_at"),
|
||||
"later_heartbeat": b.get("last_heartbeat_at"),
|
||||
}
|
||||
)
|
||||
return {
|
||||
"shared_live_workers": shared,
|
||||
"continuity": continuity,
|
||||
"stable_ownership": stable_ownership
|
||||
and all(c["heartbeat_non_decreasing"] for c in continuity),
|
||||
"dropped_workers": sorted(set(earlier_live) - set(later_live)),
|
||||
"new_workers": sorted(set(later_live) - set(earlier_live)),
|
||||
}
|
||||
+58
-5
@@ -95,6 +95,15 @@ SAFE_ENV_KEYS = (
|
||||
"GITEA_MCP_CONFIG",
|
||||
)
|
||||
|
||||
# #948: provenance used to be derived from the summary this allowlist produces.
|
||||
# The allowlist never carried a provenance key, so that derivation could only
|
||||
# ever evaluate to ``manual_launch`` — whatever the process actually was — while
|
||||
# ``gitea_get_runtime_context`` read the live environment and reported
|
||||
# ``client_managed`` for the same process. Provenance is no longer derived here.
|
||||
# It comes from ``mcp_worker_identity.assess_provenance``, the single authority
|
||||
# every surface shares. This allowlist keeps its original and only job: deciding
|
||||
# which env values are safe to echo back in diagnostics.
|
||||
|
||||
|
||||
def assess_connected_namespace_attachment(
|
||||
*,
|
||||
@@ -443,12 +452,23 @@ def classify_namespace_probe(
|
||||
profile: str | None = None,
|
||||
configured: bool = True,
|
||||
probe_source: str | None = None,
|
||||
worker_identity: str | None = None,
|
||||
generation_id: str | None = None,
|
||||
registry: Any | None = None,
|
||||
pid_alive_probe: Any | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Classify whether a required tool is callable through a live namespace.
|
||||
|
||||
``registered_tools`` is static/server-side evidence. ``probe_result`` is
|
||||
live invocation evidence. Only ``probe_source=client_namespace`` proves the
|
||||
IDE-managed path; ``offline_spawn`` is an offline subprocess check only.
|
||||
|
||||
#948: ``worker_identity``/``generation_id``/``registry`` carry the
|
||||
client/session ownership evidence. Provenance is resolved by
|
||||
``mcp_worker_identity.assess_provenance`` — the same call
|
||||
``gitea_get_runtime_context`` makes — so the two surfaces cannot report
|
||||
different provenance for one process. Omitting them yields the fail-closed
|
||||
``unproven`` verdict, never a fabricated ``client_managed``.
|
||||
"""
|
||||
ns = (namespace or "").strip()
|
||||
tool = required_tool or REQUIRED_NAMESPACE_TOOLS.get(ns) or "gitea_whoami"
|
||||
@@ -565,14 +585,31 @@ def classify_namespace_probe(
|
||||
blocks = namespace_health_blocks_task("merge_pr", healthy)
|
||||
|
||||
import gitea_config
|
||||
import mcp_worker_identity
|
||||
|
||||
raw_env = process.get("env") if isinstance(process, dict) else None
|
||||
unconsumed_env = gitea_config.get_unconsumed_gitea_env_overrides(raw_env)
|
||||
is_client_managed = bool(
|
||||
env_summary.get("GITEA_CLIENT_MANAGED") in ("1", "true", "yes", "client_managed")
|
||||
or env_summary.get("GITEA_MCP_CLIENT_MANAGED") in ("1", "true", "yes", "client_managed")
|
||||
or env_summary.get("GITEA_SERVER_PROVENANCE") == "client_managed"
|
||||
|
||||
# #948: one authority, shared with gitea_get_runtime_context. The env is
|
||||
# passed whole rather than through SAFE_ENV_KEYS — the allowlist exists to
|
||||
# decide what may be *echoed*, and using it to decide what may be *believed*
|
||||
# is what made this surface structurally unable to report client_managed.
|
||||
# ``declared_only``: ``process`` describes an observed peer, not this
|
||||
# interpreter. Its stdin is unavailable and its launcher-config env is
|
||||
# inherited from whatever shell started it, so only an explicit declaration
|
||||
# is evidence. Absence of one is ``unproven``, not an asserted manual launch.
|
||||
provenance_verdict = mcp_worker_identity.assess_provenance(
|
||||
registry=registry,
|
||||
worker_identity=worker_identity,
|
||||
generation_id=generation_id,
|
||||
env=raw_env if isinstance(raw_env, dict) else {},
|
||||
namespace=ns,
|
||||
profile=profile_name,
|
||||
pid_alive_probe=pid_alive_probe,
|
||||
declared_only=True,
|
||||
)
|
||||
provenance = "client_managed" if is_client_managed else "manual_launch"
|
||||
provenance = provenance_verdict["provenance"]
|
||||
is_client_managed = provenance_verdict["is_client_managed"]
|
||||
|
||||
return {
|
||||
"success": healthy,
|
||||
@@ -591,6 +628,15 @@ def classify_namespace_probe(
|
||||
"remediation": remediation,
|
||||
"provenance": provenance,
|
||||
"is_client_managed": is_client_managed,
|
||||
# Every non-client-session verdict fails closed. Consumers that only
|
||||
# need "may this mutate?" read this and stay correct across the #948
|
||||
# vocabulary split between ``manual_launch`` and ``unproven``.
|
||||
"provenance_fail_closed": provenance_verdict["fail_closed"],
|
||||
"provenance_assessment": provenance_verdict,
|
||||
"worker_identity": provenance_verdict["worker_identity"],
|
||||
"session_id": provenance_verdict["session_id"],
|
||||
"generation_id": provenance_verdict["generation_id"],
|
||||
"client_name": provenance_verdict["client_name"],
|
||||
"unconsumed_gitea_env": unconsumed_env,
|
||||
"diagnostics": {
|
||||
"namespace": ns,
|
||||
@@ -602,6 +648,13 @@ def classify_namespace_probe(
|
||||
"probe_source": source,
|
||||
"provenance": provenance,
|
||||
"is_client_managed": is_client_managed,
|
||||
"provenance_fail_closed": provenance_verdict["fail_closed"],
|
||||
"provenance_blocker_kind": provenance_verdict["blocker_kind"],
|
||||
"provenance_scope": provenance_verdict["scope"],
|
||||
"worker_identity": provenance_verdict["worker_identity"],
|
||||
"session_id": provenance_verdict["session_id"],
|
||||
"generation_id": provenance_verdict["generation_id"],
|
||||
"client_name": provenance_verdict["client_name"],
|
||||
"unconsumed_gitea_env": unconsumed_env,
|
||||
},
|
||||
"blocks_merge_workflow": blocks,
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
+153
-27
@@ -8,8 +8,10 @@ poison workspace purity checks in another namespace.
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
import author_mutation_worktree as amw
|
||||
import canonical_repository_root as crr
|
||||
|
||||
ACTIVE_WORKTREE_ENV = amw.ACTIVE_WORKTREE_ENV
|
||||
AUTHOR_WORKTREE_ENV = amw.AUTHOR_WORKTREE_ENV
|
||||
@@ -152,6 +154,60 @@ def resolve_namespace_workspace(
|
||||
return os.path.realpath(process_project_root), "MCP server process root (default)"
|
||||
|
||||
|
||||
def verify_git_common_directory_membership(
|
||||
workspace_path: str,
|
||||
canonical_repo_root: str,
|
||||
) -> tuple[bool, str | None]:
|
||||
"""Verify that workspace_path belongs to canonical_repo_root via git common-dir or branches containment."""
|
||||
ws = (workspace_path or "").strip()
|
||||
root = (canonical_repo_root or "").strip()
|
||||
if not ws or not root:
|
||||
return False, "empty workspace or canonical root path"
|
||||
|
||||
try:
|
||||
real_ws = os.path.realpath(os.path.abspath(ws))
|
||||
real_root = os.path.realpath(os.path.abspath(root))
|
||||
except Exception as exc:
|
||||
return False, f"invalid workspace or root path: {exc}"
|
||||
|
||||
if real_ws == real_root:
|
||||
return True, None
|
||||
|
||||
if not os.path.isdir(real_ws):
|
||||
return False, f"workspace directory '{real_ws}' does not exist"
|
||||
|
||||
try:
|
||||
res = subprocess.run(
|
||||
["git", "-C", real_ws, "rev-parse", "--git-common-dir"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
if res.returncode == 0:
|
||||
common_raw = (res.stdout or "").strip()
|
||||
common_dir = amw._realpath_git_common_dir(real_ws, common_raw)
|
||||
real_common = os.path.realpath(common_dir)
|
||||
canonical_git = os.path.realpath(os.path.join(real_root, ".git"))
|
||||
|
||||
if real_common in (canonical_git, real_root):
|
||||
return True, None
|
||||
return (
|
||||
False,
|
||||
f"workspace '{real_ws}' git common directory '{real_common}' does not match "
|
||||
f"canonical repository root '{real_root}' (.git at '{canonical_git}')"
|
||||
)
|
||||
else:
|
||||
return (
|
||||
False,
|
||||
f"workspace '{real_ws}' is not a valid git repository or git rev-parse failed"
|
||||
)
|
||||
except Exception as exc:
|
||||
return (
|
||||
False,
|
||||
f"failed to inspect git common directory for workspace '{real_ws}': {exc}"
|
||||
)
|
||||
|
||||
|
||||
def resolve_namespace_mutation_context(
|
||||
*,
|
||||
role_kind: str,
|
||||
@@ -163,6 +219,8 @@ def resolve_namespace_mutation_context(
|
||||
worktree: str | None = None,
|
||||
profile_name: str | None = None,
|
||||
configured_canonical_root: str | None = None,
|
||||
expected_slug: str | None = None,
|
||||
remote: str | None = None,
|
||||
) -> dict:
|
||||
"""Shared workspace resolution for runtime_context and mutation guards.
|
||||
|
||||
@@ -180,11 +238,41 @@ def resolve_namespace_mutation_context(
|
||||
env_map = env if env is not None else os.environ
|
||||
process_root = os.path.realpath(process_project_root)
|
||||
role = normalize_role_kind(role_kind, profile_name=profile_name)
|
||||
configured = (configured_canonical_root or "").strip()
|
||||
if configured:
|
||||
canonical_root = os.path.realpath(configured)
|
||||
|
||||
configured_val = (configured_canonical_root or "").strip()
|
||||
if configured_val:
|
||||
crr_assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=configured_val,
|
||||
source="configured_canonical_root",
|
||||
expected_slug=expected_slug,
|
||||
process_project_root=process_root,
|
||||
remote=remote,
|
||||
require_binding=True,
|
||||
)
|
||||
# #973 B10: a refused repository-authority mode deliberately resolves no
|
||||
# canonical root, so downstream guards keep evaluating the install
|
||||
# checkout rather than an identity derived through an undefined mode.
|
||||
# ``roots_aligned`` still follows ``proven`` and stays False, and the
|
||||
# refusal (with its reason_code) rides along in
|
||||
# ``canonical_root_assessment`` — this is a fail-closed fallback, never a
|
||||
# normalisation of the mode.
|
||||
canonical_root = crr_assessment["canonical_repo_root"] or amw.resolve_canonical_repo_root(
|
||||
process_root, process_root
|
||||
)
|
||||
roots_aligned = crr_assessment["proven"]
|
||||
else:
|
||||
canonical_root = amw.resolve_canonical_repo_root(process_root, process_root)
|
||||
crr_assessment = {
|
||||
"proven": True,
|
||||
"block": False,
|
||||
"reasons": [],
|
||||
"configured": False,
|
||||
"canonical_repo_root": amw.resolve_canonical_repo_root(process_root, process_root),
|
||||
"resolved_slug": None,
|
||||
"source": None,
|
||||
"reason_code": None,
|
||||
}
|
||||
canonical_root = crr_assessment["canonical_repo_root"]
|
||||
roots_aligned = (canonical_root == process_root)
|
||||
|
||||
durable: dict | None = None
|
||||
if role == "author":
|
||||
@@ -234,7 +322,10 @@ def resolve_namespace_mutation_context(
|
||||
"ignored_bindings": demotions + (pollution.get("ignored_bindings") or []),
|
||||
"process_project_root": process_root,
|
||||
"canonical_repo_root": canonical_root,
|
||||
"roots_aligned": canonical_root == process_root,
|
||||
"roots_aligned": roots_aligned,
|
||||
"canonical_root_assessment": crr_assessment,
|
||||
"expected_slug": expected_slug,
|
||||
"remote": remote,
|
||||
}
|
||||
if durable is not None:
|
||||
result["author_worktree_resolution"] = durable
|
||||
@@ -246,6 +337,15 @@ def resolve_namespace_mutation_context(
|
||||
result["author_worktree_reasons"] = list(durable.get("reasons") or [])
|
||||
result["author_worktree_blocker_kind"] = durable.get("blocker_kind")
|
||||
result["operator_recovery"] = durable.get("operator_recovery")
|
||||
else:
|
||||
path_exists = os.path.exists(workspace)
|
||||
result["path_exists"] = path_exists
|
||||
result["in_git_worktree_list"] = (
|
||||
amw.path_in_git_worktree_list(workspace, canonical_root)
|
||||
if path_exists
|
||||
else False
|
||||
)
|
||||
result["bound_worktree_missing"] = not path_exists
|
||||
return result
|
||||
|
||||
|
||||
@@ -378,8 +478,8 @@ def format_namespace_workspace_binding_error(
|
||||
def assess_namespace_mutation_workspace(
|
||||
*,
|
||||
role_kind: str,
|
||||
worktree_path: str | None,
|
||||
worktree: str | None,
|
||||
worktree_path: str | None = None,
|
||||
worktree: str | None = None,
|
||||
process_project_root: str,
|
||||
env: dict[str, str] | os._Environ | None = None,
|
||||
session_lease_worktree: str | None = None,
|
||||
@@ -387,6 +487,8 @@ def assess_namespace_mutation_workspace(
|
||||
profile_name: str | None = None,
|
||||
current_branch: str | None = None,
|
||||
configured_canonical_root: str | None = None,
|
||||
expected_slug: str | None = None,
|
||||
remote: str | None = None,
|
||||
) -> dict:
|
||||
"""Evaluate namespace workspace binding before preflight/mutation."""
|
||||
ctx = resolve_namespace_mutation_context(
|
||||
@@ -399,6 +501,8 @@ def assess_namespace_mutation_workspace(
|
||||
session_lock_worktree=session_lock_worktree,
|
||||
profile_name=profile_name,
|
||||
configured_canonical_root=configured_canonical_root,
|
||||
expected_slug=expected_slug,
|
||||
remote=remote,
|
||||
)
|
||||
mutation_workspace = ctx["workspace_path"]
|
||||
binding_source = ctx["workspace_binding_source"]
|
||||
@@ -422,6 +526,27 @@ def assess_namespace_mutation_workspace(
|
||||
|
||||
reasons = list(metadata.get("reasons") or [])
|
||||
operator_recovery = ctx.get("operator_recovery")
|
||||
|
||||
crr_reasons = list(ctx.get("canonical_root_assessment", {}).get("reasons") or [])
|
||||
if crr_reasons:
|
||||
reasons.extend(crr_reasons)
|
||||
|
||||
path_exists = ctx.get("path_exists")
|
||||
if path_exists is None:
|
||||
path_exists = os.path.exists(mutation_workspace)
|
||||
|
||||
if not path_exists:
|
||||
if role != "author":
|
||||
reasons.append(
|
||||
f"{role} mutation blocked: configured workspace directory '{mutation_workspace}' does not exist (nonexistent worktree)"
|
||||
)
|
||||
else:
|
||||
valid_common, common_err = verify_git_common_directory_membership(
|
||||
mutation_workspace, ctx["canonical_repo_root"]
|
||||
)
|
||||
if not valid_common and common_err:
|
||||
reasons.append(common_err)
|
||||
|
||||
if role == "author":
|
||||
# #618 durable resolution already validated existence, membership,
|
||||
# branches/, lock ownership, and traversal safety when present.
|
||||
@@ -438,26 +563,27 @@ def assess_namespace_mutation_workspace(
|
||||
)
|
||||
if branches["block"]:
|
||||
reasons.extend(branches["reasons"])
|
||||
elif (
|
||||
role == "reviewer"
|
||||
and mutation_workspace == process_root
|
||||
and not amw.is_path_under_branches(mutation_workspace, ctx["canonical_repo_root"])
|
||||
):
|
||||
reasons.append(
|
||||
f"{role} mutation blocked: workspace is the stable control checkout; "
|
||||
f"create or reconnect to a session-owned worktree under branches/ "
|
||||
f"or set {ROLE_WORKTREE_ENVS.get(role, ACTIVE_WORKTREE_ENV)} / "
|
||||
f"{ACTIVE_WORKTREE_ENV}"
|
||||
)
|
||||
elif (
|
||||
role in {"reviewer", "merger"}
|
||||
and mutation_workspace != process_root
|
||||
and not amw.is_path_under_branches(mutation_workspace, ctx["canonical_repo_root"])
|
||||
):
|
||||
reasons.append(
|
||||
f"{role} mutation blocked: workspace '{mutation_workspace}' is not under "
|
||||
f"'{ctx['canonical_repo_root']}/branches/'"
|
||||
)
|
||||
elif role in {"reviewer", "merger"}:
|
||||
if mutation_workspace == process_root and not amw.is_path_under_branches(mutation_workspace, ctx["canonical_repo_root"]):
|
||||
reasons.append(
|
||||
f"{role} mutation blocked: workspace is the stable control checkout; "
|
||||
f"create or reconnect to a session-owned worktree under branches/ "
|
||||
f"or set {ROLE_WORKTREE_ENVS.get(role, ACTIVE_WORKTREE_ENV)} / "
|
||||
f"{ACTIVE_WORKTREE_ENV}"
|
||||
)
|
||||
elif mutation_workspace != process_root and not amw.is_path_under_branches(mutation_workspace, ctx["canonical_repo_root"]):
|
||||
reasons.append(
|
||||
f"{role} mutation blocked: workspace '{mutation_workspace}' is not under "
|
||||
f"'{ctx['canonical_repo_root']}/branches/'"
|
||||
)
|
||||
|
||||
if path_exists and amw.is_path_under_branches(mutation_workspace, ctx["canonical_repo_root"]):
|
||||
in_list = ctx.get("in_git_worktree_list")
|
||||
if in_list is False:
|
||||
reasons.append(
|
||||
f"{role} mutation blocked: workspace '{mutation_workspace}' is under branches/ "
|
||||
f"but is not registered in git worktree list for '{ctx['canonical_repo_root']}'"
|
||||
)
|
||||
|
||||
block = bool(reasons)
|
||||
return {
|
||||
|
||||
@@ -163,6 +163,44 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.read",
|
||||
"role": "author",
|
||||
},
|
||||
# #978: instance-level fleet identity/health snapshot. gitea.read is the
|
||||
# operation gate; the tool additionally restricts role_kind to
|
||||
# controller|reconciler so author/reviewer/merger cannot use it as a
|
||||
# mutation surface and no unrelated write permission is introduced.
|
||||
"snapshot_instance_fleet": {
|
||||
"permission": "gitea.read",
|
||||
"role": "controller",
|
||||
},
|
||||
"gitea_snapshot_instance_fleet": {
|
||||
"permission": "gitea.read",
|
||||
"role": "controller",
|
||||
},
|
||||
# #980: CAS-protected retirement of conclusively stale worker
|
||||
# registrations. The mutation lands in the local control-plane worker
|
||||
# registry, not in Gitea, so — exactly like the #601 lease lifecycle — the
|
||||
# Gitea operation gate stays ``gitea.read`` and no new Gitea write
|
||||
# permission is introduced for any profile. The real authority is enforced
|
||||
# in the tools themselves: role_kind must be controller or reconciler, the
|
||||
# runtime must be parity-clean and cohort-unique, and apply additionally
|
||||
# requires the exact stable registry + candidate fingerprints returned by
|
||||
# the plan. Author, reviewer, and merger profiles keep gitea.read for
|
||||
# diagnosis elsewhere and are refused this surface.
|
||||
"plan_stale_worker_retirement": {
|
||||
"permission": "gitea.read",
|
||||
"role": "controller",
|
||||
},
|
||||
"gitea_plan_stale_worker_retirement": {
|
||||
"permission": "gitea.read",
|
||||
"role": "controller",
|
||||
},
|
||||
"apply_stale_worker_retirement": {
|
||||
"permission": "gitea.read",
|
||||
"role": "controller",
|
||||
},
|
||||
"gitea_apply_stale_worker_retirement": {
|
||||
"permission": "gitea.read",
|
||||
"role": "controller",
|
||||
},
|
||||
# #644: Phase 2 Web Console recovery tasks.
|
||||
"clear_stale_binding": {
|
||||
"permission": "gitea.read",
|
||||
@@ -386,6 +424,25 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
||||
"permission": "gitea.branch.delete",
|
||||
"role": "reconciler",
|
||||
},
|
||||
# #970: auditing missing worktree bindings is read-only; retiring one is a
|
||||
# control-plane cleanup mutation and carries the same reconciler-only
|
||||
# authority as any other reconciliation cleanup (review 644 B2).
|
||||
"audit_missing_worktree_bindings": {
|
||||
"permission": "gitea.read",
|
||||
"role": "reconciler",
|
||||
},
|
||||
"gitea_audit_missing_worktree_bindings": {
|
||||
"permission": "gitea.read",
|
||||
"role": "reconciler",
|
||||
},
|
||||
"reconcile_missing_worktree_bindings": {
|
||||
"permission": "gitea.branch.delete",
|
||||
"role": "reconciler",
|
||||
},
|
||||
"gitea_reconcile_missing_worktree_bindings": {
|
||||
"permission": "gitea.branch.delete",
|
||||
"role": "reconciler",
|
||||
},
|
||||
"work_issue": {
|
||||
"permission": "gitea.pr.create",
|
||||
"role": "author",
|
||||
@@ -653,6 +710,8 @@ ROLE_EXCLUSIVE_TASKS: frozenset[str] = frozenset(
|
||||
"delete_branch",
|
||||
"cleanup_merged_pr_branch",
|
||||
"reconciliation_cleanup",
|
||||
"reconcile_missing_worktree_bindings",
|
||||
"gitea_reconcile_missing_worktree_bindings",
|
||||
"work_issue",
|
||||
"work-issue",
|
||||
}
|
||||
|
||||
@@ -175,8 +175,21 @@ class TestLauncherSnippets(unittest.TestCase):
|
||||
def test_only_safe_keys_no_secrets(self):
|
||||
entry = gitea_config.launcher_entry("prgs", "/cfg/profiles.json")["gitea-tools"]
|
||||
self.assertEqual(set(entry), {"command", "args", "env"})
|
||||
self.assertEqual(set(entry["env"]), {"GITEA_MCP_CONFIG", "GITEA_MCP_PROFILE", "GITEA_CLIENT_MANAGED"})
|
||||
# #978 B1: production launcher also injects trusted client identity.
|
||||
required = {
|
||||
"GITEA_MCP_CONFIG",
|
||||
"GITEA_MCP_PROFILE",
|
||||
"GITEA_CLIENT_MANAGED",
|
||||
"GITEA_MCP_CLIENT",
|
||||
"GITEA_MCP_CLIENT_INSTANCE",
|
||||
"GITEA_MCP_INSTANCE_PROVENANCE",
|
||||
}
|
||||
self.assertTrue(required.issubset(set(entry["env"])), entry["env"])
|
||||
self.assertEqual(entry["env"]["GITEA_MCP_PROFILE"], "prgs")
|
||||
self.assertTrue(
|
||||
entry["env"]["GITEA_MCP_CLIENT_INSTANCE"].startswith("inst-"),
|
||||
entry["env"]["GITEA_MCP_CLIENT_INSTANCE"],
|
||||
)
|
||||
blob = json.dumps(entry).lower()
|
||||
for word in ("token", "password", "secret"):
|
||||
self.assertNotIn(word, blob)
|
||||
|
||||
@@ -313,6 +313,7 @@ class TestCanonicalRootGuardBinding(_ServerHarness):
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="derivation",
|
||||
)
|
||||
self.assertFalse(got.get("block"), got.get("reasons"))
|
||||
self.assertEqual(got["resolved_slug"], TARGET_SLUG)
|
||||
|
||||
@@ -112,7 +112,19 @@ class TestIssue686ManualMcpProvenance(unittest.TestCase):
|
||||
self.assertTrue(any("All matching profiles for task 'create_issue' (['prgs-author']) are running but stale" in r for r in reasons))
|
||||
|
||||
def test_namespace_health_classification_includes_provenance(self):
|
||||
"""AC 1 & 4: mcp_namespace_health diagnostics include provenance and unconsumed_gitea_env."""
|
||||
"""AC 1 & 4: mcp_namespace_health diagnostics include provenance and unconsumed_gitea_env.
|
||||
|
||||
#948 narrowed the vocabulary here. This process carries no client-managed
|
||||
declaration, so the old code labelled it ``manual_launch`` — asserting a
|
||||
hand-launched terminal process it had no evidence for, and contradicting
|
||||
``gitea_get_runtime_context``, which read the same process and reported
|
||||
``client_managed``. Absence of proof is now reported as ``unproven``.
|
||||
|
||||
The #686 wall itself is unchanged and still asserted below:
|
||||
``is_client_managed`` stays False, so nothing previously refused is now
|
||||
permitted. Only the label on the *reason* changed, so remediation names
|
||||
the proof that is actually missing.
|
||||
"""
|
||||
process = {
|
||||
"pid": 5555,
|
||||
"profile": "prgs-author",
|
||||
@@ -129,10 +141,12 @@ class TestIssue686ManualMcpProvenance(unittest.TestCase):
|
||||
process=process,
|
||||
probe_source="client_namespace",
|
||||
)
|
||||
self.assertEqual(res["provenance"], "manual_launch")
|
||||
self.assertEqual(res["provenance"], "unproven")
|
||||
self.assertFalse(res["is_client_managed"])
|
||||
# The wall is intact: no client-managed proof still fails closed.
|
||||
self.assertTrue(res["provenance_fail_closed"])
|
||||
self.assertEqual(res["unconsumed_gitea_env"], {"GITEA_DUMMY": "99"})
|
||||
self.assertEqual(res["diagnostics"]["provenance"], "manual_launch")
|
||||
self.assertEqual(res["diagnostics"]["provenance"], "unproven")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -241,9 +241,10 @@ class TestNamespaceContextUsesConfiguredRoot(unittest.TestCase):
|
||||
process_project_root=self.install,
|
||||
env={},
|
||||
configured_canonical_root=self.target,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertEqual(ctx["canonical_repo_root"], self.target)
|
||||
self.assertFalse(ctx["roots_aligned"])
|
||||
self.assertTrue(ctx["roots_aligned"])
|
||||
|
||||
def test_target_worktree_is_member_of_target_root(self):
|
||||
got = nwb.amw.assess_workspace_repo_membership(
|
||||
@@ -268,6 +269,7 @@ class TestNamespaceContextUsesConfiguredRoot(unittest.TestCase):
|
||||
env={},
|
||||
current_branch="feat/issue-1",
|
||||
configured_canonical_root=self.target,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertFalse(assessment["block"], assessment.get("reasons"))
|
||||
self.assertEqual(assessment["canonical_repo_root"], self.target)
|
||||
|
||||
@@ -0,0 +1,662 @@
|
||||
"""Client/session-aware runtime ownership and provenance (#948).
|
||||
|
||||
Covers the reproduced contradiction that motivated the issue: one surface
|
||||
reporting ``client_managed`` while another reported ``manual_launch`` for the
|
||||
same process, remediation hardcoded to one vendor, and a profile-wide duplicate
|
||||
wall that could not tell two healthy clients apart.
|
||||
|
||||
All client and session identifiers here are synthetic.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
import mcp_client_reconnect
|
||||
import mcp_namespace_health
|
||||
import mcp_worker_identity as mwi
|
||||
|
||||
|
||||
NOW = datetime(2026, 7, 29, 6, 0, 0, tzinfo=timezone.utc)
|
||||
|
||||
|
||||
def _registry() -> mwi.WorkerRegistry:
|
||||
"""A registry on a throwaway path; never the operator's real one."""
|
||||
handle, path = tempfile.mkstemp(suffix=".sqlite3")
|
||||
os.close(handle)
|
||||
os.unlink(path)
|
||||
return mwi.WorkerRegistry(path)
|
||||
|
||||
|
||||
def _attach(
|
||||
registry: mwi.WorkerRegistry,
|
||||
*,
|
||||
client: str,
|
||||
session: str,
|
||||
generation: str,
|
||||
profile: str = "prgs-reviewer",
|
||||
role: str = "reviewer",
|
||||
pid: int = 4242,
|
||||
now: datetime = NOW,
|
||||
ttl: float = 900.0,
|
||||
) -> dict:
|
||||
"""Register one synthetic worker and return the outcome."""
|
||||
identity = mwi.generate_worker_identity(client, session, now=now)
|
||||
outcome = registry.register(
|
||||
worker_identity=identity,
|
||||
client_name=client,
|
||||
client_instance_id=f"inst-{session}",
|
||||
session_id=session,
|
||||
generation_id=generation,
|
||||
role=role,
|
||||
profile=profile,
|
||||
pid=pid,
|
||||
heartbeat_ttl_seconds=ttl,
|
||||
now=now,
|
||||
)
|
||||
outcome["identity"] = identity
|
||||
return outcome
|
||||
|
||||
|
||||
class IdentityFormatTests(unittest.TestCase):
|
||||
"""AC27-29: collision-resistant `<llm-name>-<UTC-timestamp>-<short-sha>`."""
|
||||
|
||||
def test_identity_matches_required_format(self):
|
||||
identity = mwi.generate_worker_identity("Gemini", "sess-0001", now=NOW)
|
||||
parsed = mwi.parse_worker_identity(identity)
|
||||
self.assertTrue(parsed["valid"], parsed["reasons"])
|
||||
self.assertEqual(parsed["client_name"], "gemini")
|
||||
self.assertEqual(parsed["minted_at"], "20260729T060000Z")
|
||||
self.assertEqual(len(parsed["digest"]), 12)
|
||||
|
||||
def test_digest_varies_with_session_and_nonce(self):
|
||||
base = dict(timestamp_ns=1, now=NOW)
|
||||
a = mwi.generate_worker_identity("codex", "sess-A", nonce="n", **base)
|
||||
b = mwi.generate_worker_identity("codex", "sess-B", nonce="n", **base)
|
||||
c = mwi.generate_worker_identity("codex", "sess-A", nonce="m", **base)
|
||||
self.assertNotEqual(a, b, "session must feed the digest")
|
||||
self.assertNotEqual(a, c, "nonce must feed the digest")
|
||||
|
||||
def test_identity_is_not_role_or_profile(self):
|
||||
"""AC26: identity is independent of role and profile."""
|
||||
args = dict(timestamp_ns=7, nonce="fixed", now=NOW)
|
||||
same = mwi.generate_worker_identity("claude", "sess-1", **args)
|
||||
self.assertEqual(same, mwi.generate_worker_identity("claude", "sess-1", **args))
|
||||
# Nothing role- or profile-derived appears in the identity.
|
||||
self.assertNotIn("reviewer", same)
|
||||
self.assertNotIn("prgs", same)
|
||||
|
||||
def test_malformed_identity_rejected(self):
|
||||
self.assertFalse(mwi.parse_worker_identity("prgs-reviewer")["valid"])
|
||||
self.assertFalse(mwi.parse_worker_identity("")["valid"])
|
||||
self.assertFalse(mwi.parse_worker_identity(None)["valid"])
|
||||
|
||||
|
||||
class PerClientAttachmentTests(unittest.TestCase):
|
||||
"""Every supported client attaches and is reported as itself."""
|
||||
|
||||
def _assert_attached_as(self, client: str, expected_name: str):
|
||||
registry = _registry()
|
||||
outcome = _attach(
|
||||
registry, client=client, session=f"sess-{client}", generation="gen-1"
|
||||
)
|
||||
self.assertTrue(outcome["registered"], outcome["reasons"])
|
||||
|
||||
verdict = mwi.assess_provenance(
|
||||
registry=registry, worker_identity=outcome["identity"], env={}, now=NOW
|
||||
)
|
||||
self.assertEqual(verdict["session_ownership"], mwi.OWNERSHIP_OWNED)
|
||||
self.assertEqual(verdict["provenance"], mwi.PROVENANCE_CLIENT_SESSION)
|
||||
self.assertEqual(verdict["client_name"], expected_name)
|
||||
self.assertTrue(verdict["session_owned"])
|
||||
self.assertFalse(verdict["fail_closed"])
|
||||
return verdict
|
||||
|
||||
def test_codex_attachment(self):
|
||||
self._assert_attached_as("codex", "codex")
|
||||
|
||||
def test_gemini_attachment(self):
|
||||
self._assert_attached_as("gemini", "gemini")
|
||||
|
||||
def test_antigravity_attachment(self):
|
||||
self._assert_attached_as("antigravity", "antigravity")
|
||||
|
||||
def test_claude_attachment(self):
|
||||
self._assert_attached_as("claude", "claude_code")
|
||||
|
||||
def test_unknown_client_is_named_not_guessed(self):
|
||||
verdict = self._assert_attached_as("some_new_llm", "some_new_llm")
|
||||
self.assertNotEqual(verdict["client_name"], "codex")
|
||||
|
||||
|
||||
class SessionLifecycleTests(unittest.TestCase):
|
||||
def test_same_client_new_session_gets_distinct_identity(self):
|
||||
registry = _registry()
|
||||
first = _attach(registry, client="codex", session="sess-1", generation="gen-1")
|
||||
second = _attach(registry, client="codex", session="sess-2", generation="gen-2")
|
||||
self.assertTrue(first["registered"])
|
||||
self.assertTrue(second["registered"])
|
||||
self.assertNotEqual(first["identity"], second["identity"])
|
||||
|
||||
# Both are live and neither blocks the other.
|
||||
cohort = mwi.classify_cohort(registry.list_workers(), now=NOW)
|
||||
self.assertEqual(cohort["live_worker_count"], 2)
|
||||
self.assertFalse(cohort["blocked"], cohort["reasons"])
|
||||
|
||||
def test_different_client_attaches_after_previous_session_ends(self):
|
||||
"""AC14: expiry then takeover with a higher fencing epoch."""
|
||||
registry = _registry()
|
||||
gone = _attach(
|
||||
registry, client="codex", session="sess-old", generation="gen-shared", ttl=60
|
||||
)
|
||||
later = NOW + timedelta(hours=1)
|
||||
self.assertFalse(
|
||||
registry.is_live(registry.get(gone["identity"]), now=later)["live"]
|
||||
)
|
||||
|
||||
arriving = _attach(
|
||||
registry,
|
||||
client="gemini",
|
||||
session="sess-new",
|
||||
generation="gen-other",
|
||||
now=later,
|
||||
)
|
||||
claim = registry.claim_generation(
|
||||
worker_identity=arriving["identity"],
|
||||
generation_id="gen-shared",
|
||||
now=later,
|
||||
)
|
||||
self.assertTrue(claim["claimed"], claim["reasons"])
|
||||
self.assertIn(gone["identity"], claim["superseded_workers"])
|
||||
self.assertGreater(claim["fencing_epoch"], gone["fencing_epoch"])
|
||||
|
||||
def test_superseded_session_is_fenced_on_resume(self):
|
||||
"""AC15/AC16: the prior session cannot heartbeat its way back."""
|
||||
registry = _registry()
|
||||
old = _attach(
|
||||
registry, client="codex", session="sess-old", generation="gen-shared", ttl=60
|
||||
)
|
||||
later = NOW + timedelta(hours=1)
|
||||
new = _attach(
|
||||
registry, client="gemini", session="sess-new", generation="gen-x", now=later
|
||||
)
|
||||
registry.claim_generation(
|
||||
worker_identity=new["identity"], generation_id="gen-shared", now=later
|
||||
)
|
||||
|
||||
resumed = registry.heartbeat(
|
||||
worker_identity=old["identity"],
|
||||
fencing_epoch=old["fencing_epoch"],
|
||||
now=later,
|
||||
)
|
||||
self.assertFalse(resumed["renewed"])
|
||||
self.assertFalse(resumed["mutation_performed"])
|
||||
self.assertEqual(resumed["blocker_kind"], mwi.BLOCKER_FENCED)
|
||||
|
||||
def test_heartbeat_renews_only_the_owning_lease(self):
|
||||
"""AC11: a wrong epoch never renews, and never mutates."""
|
||||
registry = _registry()
|
||||
worker = _attach(registry, client="codex", session="s", generation="g")
|
||||
good = registry.heartbeat(
|
||||
worker_identity=worker["identity"],
|
||||
fencing_epoch=worker["fencing_epoch"],
|
||||
now=NOW + timedelta(minutes=5),
|
||||
)
|
||||
self.assertTrue(good["renewed"])
|
||||
|
||||
bad = registry.heartbeat(
|
||||
worker_identity=worker["identity"],
|
||||
fencing_epoch=worker["fencing_epoch"] + 99,
|
||||
now=NOW + timedelta(minutes=6),
|
||||
)
|
||||
self.assertFalse(bad["renewed"])
|
||||
self.assertFalse(bad["mutation_performed"])
|
||||
self.assertEqual(
|
||||
registry.get(worker["identity"])["last_heartbeat_at"],
|
||||
good["last_heartbeat_at"],
|
||||
"a refused heartbeat must not advance the record",
|
||||
)
|
||||
|
||||
|
||||
class ConflictAndCollisionTests(unittest.TestCase):
|
||||
def test_two_live_sessions_cannot_claim_one_generation(self):
|
||||
registry = _registry()
|
||||
first = _attach(registry, client="codex", session="s1", generation="gen-shared")
|
||||
second = _attach(registry, client="gemini", session="s2", generation="gen-other")
|
||||
|
||||
claim = registry.claim_generation(
|
||||
worker_identity=second["identity"],
|
||||
generation_id="gen-shared",
|
||||
now=NOW,
|
||||
)
|
||||
self.assertFalse(claim["claimed"])
|
||||
self.assertFalse(claim["mutation_performed"])
|
||||
self.assertEqual(claim["blocker_kind"], mwi.BLOCKER_CONFLICTING_SESSIONS)
|
||||
self.assertEqual(
|
||||
claim["conflicting_owners"][0]["worker_identity"], first["identity"]
|
||||
)
|
||||
# The sanctioned recovery must never be "kill the other process".
|
||||
self.assertIn("Do not kill", claim["exact_next_action"])
|
||||
|
||||
def test_contested_generation_fails_closed_in_assessment(self):
|
||||
registry = _registry()
|
||||
first = _attach(registry, client="codex", session="s1", generation="gen-shared")
|
||||
_attach(registry, client="gemini", session="s2", generation="gen-shared")
|
||||
|
||||
verdict = mwi.assess_provenance(
|
||||
registry=registry, worker_identity=first["identity"], env={}, now=NOW
|
||||
)
|
||||
self.assertEqual(verdict["session_ownership"], mwi.OWNERSHIP_CONTESTED)
|
||||
self.assertTrue(verdict["fail_closed"])
|
||||
self.assertEqual(verdict["blocker_kind"], mwi.BLOCKER_CONTRADICTORY)
|
||||
self.assertTrue(verdict["conflicting_live_sessions"])
|
||||
|
||||
def test_identity_collision_is_refused_without_corrupting_existing(self):
|
||||
"""AC31: never replace, adopt, merge with, or corrupt the incumbent."""
|
||||
registry = _registry()
|
||||
incumbent = _attach(registry, client="codex", session="s1", generation="gen-1")
|
||||
before = registry.get(incumbent["identity"])
|
||||
|
||||
collided = registry.register(
|
||||
worker_identity=incumbent["identity"],
|
||||
client_name="gemini",
|
||||
client_instance_id="inst-other",
|
||||
session_id="s2",
|
||||
generation_id="gen-2",
|
||||
pid=9999,
|
||||
now=NOW,
|
||||
)
|
||||
self.assertFalse(collided["registered"])
|
||||
self.assertTrue(collided["collision"])
|
||||
self.assertFalse(collided["mutation_performed"])
|
||||
self.assertEqual(collided["blocker_kind"], mwi.BLOCKER_IDENTITY_COLLISION)
|
||||
self.assertEqual(collided["collision_kind"], "active_worker")
|
||||
self.assertEqual(
|
||||
registry.get(incumbent["identity"]), before, "incumbent must be untouched"
|
||||
)
|
||||
|
||||
def test_after_collision_a_regenerated_identity_registers(self):
|
||||
"""AC32/AC35: forced collision, safe regeneration, successful replacement."""
|
||||
registry = _registry()
|
||||
fixed = dict(timestamp_ns=99, nonce="deterministic", now=NOW)
|
||||
forced = mwi.generate_worker_identity("codex", "sess-collide", **fixed)
|
||||
first = registry.register(
|
||||
worker_identity=forced,
|
||||
client_name="codex",
|
||||
client_instance_id="inst-1",
|
||||
session_id="sess-collide",
|
||||
generation_id="gen-1",
|
||||
now=NOW,
|
||||
)
|
||||
self.assertTrue(first["registered"])
|
||||
|
||||
# A second worker deriving the same inputs collides deterministically.
|
||||
again = mwi.generate_worker_identity("codex", "sess-collide", **fixed)
|
||||
self.assertEqual(again, forced)
|
||||
self.assertTrue(
|
||||
registry.register(
|
||||
worker_identity=again,
|
||||
client_name="codex",
|
||||
client_instance_id="inst-2",
|
||||
session_id="sess-collide",
|
||||
generation_id="gen-2",
|
||||
now=NOW,
|
||||
)["collision"]
|
||||
)
|
||||
|
||||
replacement = mwi.generate_worker_identity(
|
||||
"codex", "sess-collide", timestamp_ns=100, nonce="different", now=NOW
|
||||
)
|
||||
self.assertNotEqual(replacement, forced)
|
||||
self.assertTrue(
|
||||
registry.register(
|
||||
worker_identity=replacement,
|
||||
client_name="codex",
|
||||
client_instance_id="inst-2",
|
||||
session_id="sess-collide",
|
||||
generation_id="gen-2",
|
||||
now=NOW,
|
||||
)["registered"]
|
||||
)
|
||||
|
||||
def test_restarted_worker_inherits_nothing(self):
|
||||
"""AC33/AC34: a restart mints a new identity and no prior epoch."""
|
||||
registry = _registry()
|
||||
before = _attach(
|
||||
registry, client="codex", session="sess-before", generation="gen-1", ttl=60
|
||||
)
|
||||
later = NOW + timedelta(hours=2)
|
||||
after = _attach(
|
||||
registry, client="codex", session="sess-after", generation="gen-2", now=later
|
||||
)
|
||||
self.assertNotEqual(before["identity"], after["identity"])
|
||||
self.assertNotEqual(
|
||||
registry.get(after["identity"])["generation_id"],
|
||||
registry.get(before["identity"])["generation_id"],
|
||||
)
|
||||
|
||||
|
||||
class LivenessTests(unittest.TestCase):
|
||||
def test_stale_session_record_is_not_live(self):
|
||||
registry = _registry()
|
||||
worker = _attach(registry, client="codex", session="s", generation="g", ttl=300)
|
||||
stale = registry.is_live(
|
||||
registry.get(worker["identity"]), now=NOW + timedelta(hours=1)
|
||||
)
|
||||
self.assertFalse(stale["live"])
|
||||
self.assertFalse(stale["heartbeat_fresh"])
|
||||
|
||||
def test_liveness_is_not_pid_comparison_alone(self):
|
||||
"""AC7: a live PID does not resurrect an expired registration."""
|
||||
registry = _registry()
|
||||
worker = _attach(registry, client="codex", session="s", generation="g", ttl=60)
|
||||
verdict = registry.is_live(
|
||||
registry.get(worker["identity"]),
|
||||
now=NOW + timedelta(hours=1),
|
||||
pid_alive=True,
|
||||
)
|
||||
self.assertFalse(
|
||||
verdict["live"], "a live PID must not override a dead heartbeat"
|
||||
)
|
||||
|
||||
def test_dead_pid_withdraws_liveness_from_a_fresh_heartbeat(self):
|
||||
registry = _registry()
|
||||
worker = _attach(registry, client="codex", session="s", generation="g")
|
||||
verdict = registry.is_live(
|
||||
registry.get(worker["identity"]), now=NOW, pid_alive=False
|
||||
)
|
||||
self.assertFalse(verdict["live"])
|
||||
|
||||
def test_stale_ownership_does_not_permanently_strand_a_daemon(self):
|
||||
registry = _registry()
|
||||
stranded = _attach(
|
||||
registry, client="codex", session="s-old", generation="gen-daemon", ttl=60
|
||||
)
|
||||
later = NOW + timedelta(hours=3)
|
||||
rescuer = _attach(
|
||||
registry, client="claude", session="s-new", generation="gen-tmp", now=later
|
||||
)
|
||||
claim = registry.claim_generation(
|
||||
worker_identity=rescuer["identity"],
|
||||
generation_id="gen-daemon",
|
||||
now=later,
|
||||
)
|
||||
self.assertTrue(claim["claimed"], claim["reasons"])
|
||||
self.assertIn(stranded["identity"], claim["superseded_workers"])
|
||||
|
||||
|
||||
class EvidenceTests(unittest.TestCase):
|
||||
def test_env_flag_alone_does_not_prove_session_ownership(self):
|
||||
verdict = mwi.assess_provenance(
|
||||
registry=None,
|
||||
worker_identity=None,
|
||||
env={"GITEA_CLIENT_MANAGED": "1", "GITEA_MCP_SANCTIONED_DAEMON": "1"},
|
||||
now=NOW,
|
||||
)
|
||||
self.assertFalse(verdict["session_owned"])
|
||||
self.assertEqual(verdict["session_ownership"], mwi.OWNERSHIP_UNOWNED)
|
||||
self.assertTrue(verdict["env_flag_only"])
|
||||
self.assertTrue(verdict["fail_closed"])
|
||||
self.assertNotIn(mwi.EVIDENCE_ATTACHMENT_RECORD, verdict["evidence"])
|
||||
self.assertFalse(verdict["env_signal"]["proves_session_ownership"])
|
||||
|
||||
def test_env_flag_still_answers_the_launch_question(self):
|
||||
"""The #686 wall is preserved: env decides launch, not ownership."""
|
||||
self.assertTrue(
|
||||
mwi.assess_launch_provenance({"GITEA_CLIENT_MANAGED": "1"})["client_managed"]
|
||||
)
|
||||
self.assertFalse(
|
||||
mwi.assess_launch_provenance({"GITEA_CLIENT_MANAGED": "0"})["client_managed"]
|
||||
)
|
||||
self.assertFalse(
|
||||
mwi.assess_launch_provenance({}, stdin_is_tty=True)["client_managed"]
|
||||
)
|
||||
self.assertTrue(
|
||||
mwi.assess_launch_provenance({"GITEA_MCP_PROFILE": "prgs-author"})[
|
||||
"client_managed"
|
||||
]
|
||||
)
|
||||
|
||||
def test_missing_evidence_is_unproven_not_manual(self):
|
||||
"""A missing proof must not be reported as a hand-launched process."""
|
||||
verdict = mwi.assess_provenance(
|
||||
registry=None, worker_identity=None, env={}, now=NOW
|
||||
)
|
||||
self.assertEqual(verdict["provenance"], mwi.PROVENANCE_UNPROVEN)
|
||||
self.assertNotEqual(verdict["provenance"], mwi.PROVENANCE_MANUAL)
|
||||
self.assertTrue(verdict["fail_closed"])
|
||||
|
||||
def test_declared_manual_launch_is_reported_as_manual(self):
|
||||
verdict = mwi.assess_provenance(
|
||||
registry=None,
|
||||
worker_identity=None,
|
||||
env={"GITEA_CLIENT_MANAGED": "0"},
|
||||
now=NOW,
|
||||
)
|
||||
self.assertEqual(verdict["provenance"], mwi.PROVENANCE_MANUAL)
|
||||
|
||||
def test_fail_closed_refusal_names_its_scope_not_the_profile(self):
|
||||
"""AC17/AC41: no refusal is profile-wide."""
|
||||
verdict = mwi.assess_provenance(
|
||||
registry=None,
|
||||
worker_identity=None,
|
||||
env={},
|
||||
profile="prgs-reviewer",
|
||||
role="reviewer",
|
||||
now=NOW,
|
||||
)
|
||||
self.assertFalse(verdict["scope"]["profile_wide"])
|
||||
self.assertEqual(verdict["blocker_kind"], mwi.BLOCKER_NO_ATTACHMENT)
|
||||
|
||||
|
||||
class CohortScopingTests(unittest.TestCase):
|
||||
def test_shared_profile_with_distinct_identities_does_not_block(self):
|
||||
"""AC40: profile is not a singleton identity."""
|
||||
registry = _registry()
|
||||
_attach(
|
||||
registry,
|
||||
client="codex",
|
||||
session="s1",
|
||||
generation="g1",
|
||||
profile="prgs-reviewer",
|
||||
)
|
||||
_attach(
|
||||
registry,
|
||||
client="gemini",
|
||||
session="s2",
|
||||
generation="g2",
|
||||
profile="prgs-reviewer",
|
||||
)
|
||||
|
||||
cohort = mwi.classify_cohort(registry.list_workers(), now=NOW)
|
||||
self.assertFalse(cohort["blocked"], cohort["reasons"])
|
||||
self.assertEqual(cohort["blocker_kind"], mwi.BLOCKER_NONE)
|
||||
self.assertIn("prgs-reviewer", cohort["shared_profiles"])
|
||||
self.assertTrue(cohort["profile_sharing_permitted"])
|
||||
self.assertEqual(cohort["blocked_worker_identities"], [])
|
||||
|
||||
def test_duplicate_cohort_records_block_only_the_offenders(self):
|
||||
registry = _registry()
|
||||
_attach(registry, client="codex", session="s1", generation="gen-contested")
|
||||
_attach(registry, client="gemini", session="s2", generation="gen-contested")
|
||||
_attach(registry, client="claude", session="s3", generation="gen-fine")
|
||||
|
||||
cohort = mwi.classify_cohort(registry.list_workers(), now=NOW)
|
||||
self.assertTrue(cohort["blocked"])
|
||||
self.assertEqual(cohort["contested_generations"], ["gen-contested"])
|
||||
self.assertEqual(len(cohort["blocked_worker_identities"]), 2)
|
||||
|
||||
def test_mixed_runtime_generations_are_scoped_independently(self):
|
||||
"""AC17: one stale generation does not wall unrelated healthy ones."""
|
||||
registry = _registry()
|
||||
stale = _attach(
|
||||
registry, client="codex", session="s1", generation="gen-stale", ttl=60
|
||||
)
|
||||
healthy_a = _attach(registry, client="gemini", session="s2", generation="gen-a")
|
||||
healthy_b = _attach(registry, client="claude", session="s3", generation="gen-b")
|
||||
|
||||
scoped = mwi.scope_runtime_failure(
|
||||
failure_kind="stale-runtime",
|
||||
worker_identity=stale["identity"],
|
||||
profile="prgs-reviewer",
|
||||
all_live_workers=registry.list_workers(),
|
||||
)
|
||||
self.assertFalse(scoped["profile_wide"])
|
||||
self.assertFalse(scoped["fleet_wide"])
|
||||
self.assertEqual(len(scoped["affected_workers"]), 1)
|
||||
self.assertEqual(scoped["unaffected_worker_count"], 2)
|
||||
unaffected = {w["worker_identity"] for w in scoped["unaffected_workers"]}
|
||||
self.assertEqual(unaffected, {healthy_a["identity"], healthy_b["identity"]})
|
||||
|
||||
|
||||
class HardcodedClientRegressionTests(unittest.TestCase):
|
||||
def test_unknown_client_does_not_resolve_to_codex(self):
|
||||
for name in ("gemini", "antigravity", "grok", "some_new_llm", "", None):
|
||||
with self.subTest(client=name):
|
||||
self.assertNotEqual(
|
||||
mcp_client_reconnect.normalize_client(name),
|
||||
"codex",
|
||||
"an unidentified client must never be handed Codex UI steps",
|
||||
)
|
||||
|
||||
def test_known_clients_still_get_their_own_steps(self):
|
||||
self.assertEqual(mcp_client_reconnect.normalize_client("codex"), "codex")
|
||||
self.assertEqual(
|
||||
mcp_client_reconnect.normalize_client("claude_code"), "claude_code"
|
||||
)
|
||||
|
||||
def test_generic_steps_do_not_name_a_specific_vendor(self):
|
||||
steps = " ".join(mcp_client_reconnect.operator_ui_steps("gemini"))
|
||||
self.assertNotIn("Codex", steps)
|
||||
|
||||
def test_reconnect_client_is_derived_from_the_attachment_record(self):
|
||||
registry = _registry()
|
||||
worker = _attach(registry, client="antigravity", session="s", generation="g")
|
||||
verdict = mwi.assess_provenance(
|
||||
registry=registry, worker_identity=worker["identity"], env={}, now=NOW
|
||||
)
|
||||
self.assertEqual(mwi.reconnect_client_for(verdict), "antigravity")
|
||||
|
||||
def test_reconnect_client_is_unknown_rather_than_guessed(self):
|
||||
self.assertEqual(mwi.reconnect_client_for({}), mwi.UNKNOWN_CLIENT)
|
||||
|
||||
|
||||
class RemoteBindingTests(unittest.TestCase):
|
||||
def test_explicit_prgs_selection_is_honoured(self):
|
||||
resolved = mwi.resolve_bound_remote(
|
||||
requested_remote="prgs", bound_remote="prgs", default_remote="dadeschools"
|
||||
)
|
||||
self.assertEqual(resolved["remote"], "prgs")
|
||||
self.assertFalse(resolved["drifted"])
|
||||
|
||||
def test_omitted_remote_uses_the_binding_not_the_library_default(self):
|
||||
"""The reported dadeschools host drift."""
|
||||
resolved = mwi.resolve_bound_remote(
|
||||
requested_remote=None, bound_remote="prgs", default_remote="dadeschools"
|
||||
)
|
||||
self.assertEqual(resolved["remote"], "prgs")
|
||||
self.assertNotEqual(resolved["remote"], "dadeschools")
|
||||
self.assertEqual(resolved["resolved_from"], "session_binding")
|
||||
|
||||
def test_contradicting_the_binding_is_refused(self):
|
||||
resolved = mwi.resolve_bound_remote(
|
||||
requested_remote="dadeschools",
|
||||
bound_remote="prgs",
|
||||
default_remote="dadeschools",
|
||||
)
|
||||
self.assertEqual(resolved["remote"], "prgs")
|
||||
self.assertTrue(resolved["drifted"])
|
||||
self.assertFalse(resolved["honoured_request"])
|
||||
|
||||
def test_unbound_session_falls_back_and_says_so(self):
|
||||
resolved = mwi.resolve_bound_remote(
|
||||
requested_remote=None, bound_remote=None, default_remote="dadeschools"
|
||||
)
|
||||
self.assertEqual(resolved["remote"], "dadeschools")
|
||||
self.assertEqual(resolved["resolved_from"], "library_default")
|
||||
self.assertTrue(resolved["reasons"])
|
||||
|
||||
|
||||
class SurfaceAgreementTests(unittest.TestCase):
|
||||
"""The reproduced contradiction: two surfaces, one process, two answers."""
|
||||
|
||||
def test_namespace_health_and_direct_assessment_agree(self):
|
||||
registry = _registry()
|
||||
worker = _attach(
|
||||
registry,
|
||||
client="gemini",
|
||||
session="sess-agree",
|
||||
generation="gen-agree",
|
||||
profile="prgs-reviewer",
|
||||
)
|
||||
env = {"GITEA_MCP_PROFILE": "prgs-reviewer", "GITEA_CLIENT_MANAGED": "1"}
|
||||
|
||||
direct = mwi.assess_provenance(
|
||||
registry=registry,
|
||||
worker_identity=worker["identity"],
|
||||
env=env,
|
||||
profile="prgs-reviewer",
|
||||
)
|
||||
health = mcp_namespace_health.classify_namespace_probe(
|
||||
"gitea-reviewer",
|
||||
configured=True,
|
||||
registered_tools=["gitea_whoami"],
|
||||
probe_result={"success": True},
|
||||
probe_source="client_namespace",
|
||||
process={"pid": 4242, "profile": "prgs-reviewer", "env": env},
|
||||
registry=registry,
|
||||
worker_identity=worker["identity"],
|
||||
)
|
||||
|
||||
self.assertEqual(health["provenance"], direct["provenance"])
|
||||
self.assertEqual(health["is_client_managed"], direct["is_client_managed"])
|
||||
self.assertEqual(health["worker_identity"], direct["worker_identity"])
|
||||
self.assertEqual(health["session_id"], "sess-agree")
|
||||
self.assertEqual(health["client_name"], "gemini")
|
||||
|
||||
def test_namespace_health_can_report_client_managed_at_all(self):
|
||||
"""The old derivation was structurally incapable of this."""
|
||||
env = {"GITEA_CLIENT_MANAGED": "1", "GITEA_MCP_PROFILE": "prgs-author"}
|
||||
health = mcp_namespace_health.classify_namespace_probe(
|
||||
"gitea-author",
|
||||
configured=True,
|
||||
registered_tools=["gitea_whoami"],
|
||||
probe_result={"success": True},
|
||||
probe_source="client_namespace",
|
||||
process={"pid": 1234, "profile": "prgs-author", "env": env},
|
||||
)
|
||||
self.assertTrue(
|
||||
health["is_client_managed"],
|
||||
"a client-managed launch must be reportable as client-managed",
|
||||
)
|
||||
|
||||
def test_namespace_health_without_attachment_fails_closed(self):
|
||||
health = mcp_namespace_health.classify_namespace_probe(
|
||||
"gitea-author",
|
||||
configured=True,
|
||||
registered_tools=["gitea_whoami"],
|
||||
probe_result={"success": True},
|
||||
probe_source="client_namespace",
|
||||
process={"pid": 1234, "profile": "prgs-author", "env": {}},
|
||||
)
|
||||
self.assertTrue(health["provenance_fail_closed"])
|
||||
self.assertEqual(health["provenance"], mwi.PROVENANCE_UNPROVEN)
|
||||
self.assertIsNone(health["session_id"])
|
||||
|
||||
def test_no_false_reconnect_loop_for_an_owned_session(self):
|
||||
"""A proven owner must not be told to reconnect."""
|
||||
registry = _registry()
|
||||
worker = _attach(registry, client="claude", session="s", generation="g")
|
||||
verdict = mwi.assess_provenance(
|
||||
registry=registry, worker_identity=worker["identity"], env={}, now=NOW
|
||||
)
|
||||
self.assertFalse(verdict["fail_closed"])
|
||||
self.assertEqual(verdict["blocker_kind"], mwi.BLOCKER_NONE)
|
||||
self.assertEqual(verdict["reasons"], [])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,739 @@
|
||||
"""Regression tests for Issue #973 blocker B10: repository-authority mode contract.
|
||||
|
||||
Before this repair ``assess_canonical_repository_root`` never checked its
|
||||
``mode`` argument against an allowlist. Both dispatch points were permissive:
|
||||
|
||||
* the configured-root path tested ``mode == "derivation"`` and sent every other
|
||||
value into a catch-all ``else``, so an unsupported mode silently received
|
||||
*validation* semantics, and
|
||||
* the single-repository default path tested ``mode == "validation"``, so an
|
||||
unsupported mode skipped the identity comparison entirely and was strictly
|
||||
*weaker* than validation.
|
||||
|
||||
The measured consequence was that ``mode="invalid_mode"`` with matching expected
|
||||
and observed identities returned ``proven: True`` / ``block: False`` with no
|
||||
reasons, and that an unsupported mode passed on the default path where
|
||||
``"validation"`` correctly blocked.
|
||||
|
||||
These tests exercise the production module directly with real git repositories —
|
||||
no patched stand-in for the function under test — and drive the production
|
||||
enforcement and mutation-context consumers rather than only the intermediate
|
||||
assessment.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import inspect
|
||||
import os
|
||||
import subprocess
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
import canonical_repository_root as crr
|
||||
import gitea_config
|
||||
import gitea_mcp_server as mcp_server
|
||||
import namespace_workspace_binding as nwb
|
||||
import stable_control_runtime
|
||||
|
||||
|
||||
INSTALL_SLUG = "Scaled-Tech-Consulting/Gitea-Tools"
|
||||
TARGET_SLUG = "Scaled-Tech-Consulting/mcp-control-plane"
|
||||
FOREIGN_SLUG = "Someone-Else/Evil-Repo"
|
||||
|
||||
# Explicitly supplied values that must all be refused. Omission is *not* in this
|
||||
# list: omitting the argument keeps the documented ``"validation"`` default.
|
||||
UNSUPPORTED_STRING_MODES = (
|
||||
"invalid_mode",
|
||||
"",
|
||||
"validaton", # misspelling
|
||||
"derivaton", # misspelling
|
||||
"Validation", # case variant
|
||||
"DERIVATION", # case variant
|
||||
" validation", # leading whitespace
|
||||
"validation ", # trailing whitespace
|
||||
"derivation\n", # trailing newline
|
||||
"validation,derivation",
|
||||
)
|
||||
|
||||
UNSUPPORTED_NON_STRING_MODES = (
|
||||
None,
|
||||
0,
|
||||
1,
|
||||
True,
|
||||
False,
|
||||
3.14,
|
||||
[],
|
||||
["validation"],
|
||||
{},
|
||||
{"mode": "validation"},
|
||||
("validation",),
|
||||
object(),
|
||||
)
|
||||
|
||||
|
||||
def _init_repo(path: str, remote_url: str, *, user: str = "Test User") -> None:
|
||||
os.makedirs(path, exist_ok=True)
|
||||
subprocess.run(["git", "init", "-b", "master"], cwd=path, check=True,
|
||||
capture_output=True)
|
||||
subprocess.run(["git", "config", "user.email", "[email protected]"], cwd=path,
|
||||
check=True, capture_output=True)
|
||||
subprocess.run(["git", "config", "user.name", user], cwd=path, check=True,
|
||||
capture_output=True)
|
||||
with open(os.path.join(path, "README.md"), "w") as handle:
|
||||
handle.write(f"{os.path.basename(path)}\n")
|
||||
subprocess.run(["git", "add", "README.md"], cwd=path, check=True,
|
||||
capture_output=True)
|
||||
subprocess.run(["git", "commit", "-m", "initial"], cwd=path, check=True,
|
||||
capture_output=True)
|
||||
subprocess.run(["git", "remote", "add", "prgs", remote_url], cwd=path,
|
||||
check=True, capture_output=True)
|
||||
|
||||
|
||||
class _CanonicalRootFixture(unittest.TestCase):
|
||||
"""Real install / target / foreign git repositories, as in the #973 suite."""
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.tmp_dir = os.path.realpath(self._tmp.name)
|
||||
|
||||
self.install_root = os.path.join(self.tmp_dir, "Gitea-Tools")
|
||||
_init_repo(
|
||||
self.install_root,
|
||||
"https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools.git",
|
||||
)
|
||||
|
||||
self.target_root = os.path.join(self.tmp_dir, "mcp-control-plane")
|
||||
_init_repo(
|
||||
self.target_root,
|
||||
"https://gitea.prgs.cc/Scaled-Tech-Consulting/mcp-control-plane.git",
|
||||
)
|
||||
|
||||
self.evil_root = os.path.join(self.tmp_dir, "Evil-Repo")
|
||||
_init_repo(
|
||||
self.evil_root,
|
||||
"https://gitea.prgs.cc/Someone-Else/Evil-Repo.git",
|
||||
user="Evil User",
|
||||
)
|
||||
|
||||
self.target_branches = os.path.join(self.target_root, "branches")
|
||||
self.target_worktree = os.path.join(self.target_branches, "rev-pr-99")
|
||||
subprocess.run(
|
||||
["git", "worktree", "add", "-b", "rev-pr-99", self.target_worktree],
|
||||
cwd=self.target_root, check=True, capture_output=True,
|
||||
)
|
||||
|
||||
def tearDown(self):
|
||||
self._tmp.cleanup()
|
||||
|
||||
def assertRefusedForMode(self, assessment: dict, mode) -> None:
|
||||
"""Assert a fail-closed refusal attributable to *mode* and nothing else."""
|
||||
self.assertFalse(assessment["proven"], assessment)
|
||||
self.assertTrue(assessment["block"], assessment)
|
||||
self.assertEqual(assessment["reason_code"], crr.DENY_UNKNOWN_MODE, assessment)
|
||||
self.assertEqual(len(assessment["reasons"]), 1, assessment)
|
||||
reason = assessment["reasons"][0]
|
||||
self.assertIn("unsupported repository-authority mode", reason)
|
||||
self.assertIn(repr(mode), reason)
|
||||
# No trusted repository identity may be derived through an invalid mode.
|
||||
self.assertIsNone(assessment["resolved_slug"], assessment)
|
||||
self.assertIsNone(assessment["canonical_repo_root"], assessment)
|
||||
|
||||
|
||||
class TestB10UnsupportedModeIsRejected(_CanonicalRootFixture):
|
||||
"""Direct assessment tests for invalid-mode parsing."""
|
||||
|
||||
def test_invalid_string_with_missing_expected_identity(self):
|
||||
"""G1: blocks for the mode, not incidentally for a missing identity."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=None,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="invalid_mode",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "invalid_mode")
|
||||
self.assertFalse(
|
||||
any("unprovable or missing" in r for r in assessment["reasons"]),
|
||||
"must block because the mode is unsupported, not because the expected "
|
||||
"identity happened to be missing",
|
||||
)
|
||||
|
||||
def test_invalid_string_with_matching_identities(self):
|
||||
"""G2: the contract violation — matching identities used to return proven."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="invalid_mode",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "invalid_mode")
|
||||
|
||||
def test_invalid_string_with_conflicting_identities(self):
|
||||
"""G3: refused for the mode, not for the incidental identity mismatch."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.evil_root,
|
||||
source="env",
|
||||
expected_slug=INSTALL_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="invalid_mode",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "invalid_mode")
|
||||
self.assertFalse(
|
||||
any("identity mismatch" in r for r in assessment["reasons"]),
|
||||
"the identity comparison must not have run at all",
|
||||
)
|
||||
|
||||
def test_empty_string_mode(self):
|
||||
"""G4a: an explicitly supplied empty string is an unsupported value."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "")
|
||||
|
||||
def test_explicit_none_mode(self):
|
||||
"""G4b: explicit None is refused; it is not treated as omission."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode=None,
|
||||
)
|
||||
self.assertRefusedForMode(assessment, None)
|
||||
self.assertIn("of type NoneType", assessment["reasons"][0])
|
||||
|
||||
def test_representative_non_string_modes(self):
|
||||
for mode in UNSUPPORTED_NON_STRING_MODES:
|
||||
with self.subTest(mode=repr(mode)):
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode=mode,
|
||||
)
|
||||
self.assertRefusedForMode(assessment, mode)
|
||||
self.assertIn(
|
||||
f"of type {type(mode).__name__}", assessment["reasons"][0]
|
||||
)
|
||||
|
||||
def test_unknown_strings_misspellings_and_whitespace_variants(self):
|
||||
for mode in UNSUPPORTED_STRING_MODES:
|
||||
with self.subTest(mode=repr(mode)):
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode=mode,
|
||||
)
|
||||
self.assertRefusedForMode(assessment, mode)
|
||||
|
||||
def test_supported_modes_are_exactly_two(self):
|
||||
self.assertEqual(
|
||||
crr.SUPPORTED_MODES, ("validation", "derivation")
|
||||
)
|
||||
self.assertIsNone(crr.unsupported_mode_reason("validation"))
|
||||
self.assertIsNone(crr.unsupported_mode_reason("derivation"))
|
||||
self.assertIsNotNone(crr.unsupported_mode_reason("invalid_mode"))
|
||||
|
||||
|
||||
class TestB10SupportedModesUnchanged(_CanonicalRootFixture):
|
||||
"""The repair must not disturb the two documented modes."""
|
||||
|
||||
def test_omitted_mode_defaults_to_validation(self):
|
||||
"""G4c/G4d: omission still selects validation, proven by both outcomes."""
|
||||
matching = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
)
|
||||
self.assertTrue(matching["proven"], matching)
|
||||
self.assertFalse(matching["block"], matching)
|
||||
self.assertIsNone(matching["reason_code"], matching)
|
||||
self.assertEqual(matching["resolved_slug"], TARGET_SLUG)
|
||||
|
||||
conflicting = crr.assess_canonical_repository_root(
|
||||
configured_value=self.evil_root,
|
||||
source="env",
|
||||
expected_slug=INSTALL_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
)
|
||||
self.assertFalse(conflicting["proven"], conflicting)
|
||||
self.assertTrue(conflicting["block"], conflicting)
|
||||
self.assertIsNone(conflicting["reason_code"], conflicting)
|
||||
self.assertTrue(
|
||||
any("identity mismatch" in r for r in conflicting["reasons"]),
|
||||
"omission must behave exactly like explicit validation",
|
||||
)
|
||||
|
||||
def test_explicit_validation_retains_strict_behavior(self):
|
||||
"""G5a/G5b/G5c."""
|
||||
ok = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root, source="env",
|
||||
expected_slug=TARGET_SLUG, process_project_root=self.install_root,
|
||||
remote="prgs", require_binding=True, mode="validation",
|
||||
)
|
||||
self.assertTrue(ok["proven"], ok)
|
||||
|
||||
mismatch = crr.assess_canonical_repository_root(
|
||||
configured_value=self.evil_root, source="env",
|
||||
expected_slug=INSTALL_SLUG, process_project_root=self.install_root,
|
||||
remote="prgs", require_binding=True, mode="validation",
|
||||
)
|
||||
self.assertTrue(mismatch["block"], mismatch)
|
||||
self.assertTrue(any("identity mismatch" in r for r in mismatch["reasons"]))
|
||||
|
||||
unprovable = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root, source="env",
|
||||
expected_slug=None, process_project_root=self.install_root,
|
||||
remote="prgs", require_binding=True, mode="validation",
|
||||
)
|
||||
self.assertTrue(unprovable["block"], unprovable)
|
||||
self.assertTrue(
|
||||
any("unprovable or missing" in r for r in unprovable["reasons"])
|
||||
)
|
||||
|
||||
def test_explicit_derivation_retains_trusted_derivation(self):
|
||||
"""G5d: derivation still resolves identity with no expected slug."""
|
||||
derived = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root, source="env",
|
||||
expected_slug=None, process_project_root=self.install_root,
|
||||
remote="prgs", require_binding=True, mode="derivation",
|
||||
)
|
||||
self.assertTrue(derived["proven"], derived)
|
||||
self.assertFalse(derived["block"], derived)
|
||||
self.assertIsNone(derived["reason_code"], derived)
|
||||
self.assertEqual(derived["resolved_slug"], TARGET_SLUG)
|
||||
|
||||
def test_derivation_without_resolvable_remote_still_fails_closed(self):
|
||||
no_remote = os.path.join(self.tmp_dir, "no-remote-target")
|
||||
os.makedirs(no_remote)
|
||||
subprocess.run(["git", "init", "-b", "master"], cwd=no_remote, check=True,
|
||||
capture_output=True)
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=no_remote, source="env", expected_slug=None,
|
||||
process_project_root=self.install_root, remote="prgs",
|
||||
require_binding=True, mode="derivation",
|
||||
)
|
||||
self.assertTrue(assessment["block"], assessment)
|
||||
self.assertTrue(
|
||||
any("no resolvable" in r for r in assessment["reasons"]), assessment
|
||||
)
|
||||
|
||||
|
||||
class TestB10SingleRepositoryDefaultPath(_CanonicalRootFixture):
|
||||
"""G6: on the unconfigured path an invalid mode used to be weaker than validation."""
|
||||
|
||||
def _assess(self, mode_kwargs: dict) -> dict:
|
||||
return crr.assess_canonical_repository_root(
|
||||
configured_value=None,
|
||||
source=None,
|
||||
expected_slug=FOREIGN_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=False,
|
||||
**mode_kwargs,
|
||||
)
|
||||
|
||||
def test_validation_blocks_a_foreign_expected_identity(self):
|
||||
"""G6b: the reference behaviour the invalid mode must not undercut."""
|
||||
got = self._assess({"mode": "validation"})
|
||||
self.assertFalse(got["proven"], got)
|
||||
self.assertTrue(got["block"], got)
|
||||
self.assertTrue(any("identity mismatch" in r for r in got["reasons"]))
|
||||
|
||||
def test_invalid_mode_no_longer_passes_where_validation_blocks(self):
|
||||
"""G6a: identical inputs, only the mode differs — must not fail open."""
|
||||
invalid = self._assess({"mode": "invalid_mode"})
|
||||
self.assertRefusedForMode(invalid, "invalid_mode")
|
||||
|
||||
validation = self._assess({"mode": "validation"})
|
||||
self.assertEqual(
|
||||
invalid["proven"], validation["proven"],
|
||||
"an unsupported mode must never be more permissive than validation",
|
||||
)
|
||||
self.assertTrue(invalid["block"] and validation["block"])
|
||||
|
||||
def test_empty_string_mode_on_default_path(self):
|
||||
"""G6c."""
|
||||
self.assertRefusedForMode(self._assess({"mode": ""}), "")
|
||||
|
||||
def test_omitted_mode_on_default_path_still_validates(self):
|
||||
got = self._assess({})
|
||||
self.assertFalse(got["proven"], got)
|
||||
self.assertTrue(any("identity mismatch" in r for r in got["reasons"]), got)
|
||||
|
||||
|
||||
class TestB10RejectionOrdering(_CanonicalRootFixture):
|
||||
"""Refusal must precede every form of candidate-root or Git inspection."""
|
||||
|
||||
def test_no_git_or_identity_discovery_runs_for_an_unsupported_mode(self):
|
||||
with patch.object(crr, "resolve_repo_toplevel") as toplevel, \
|
||||
patch.object(crr, "repository_identity_slug") as identity:
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="invalid_mode",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "invalid_mode")
|
||||
toplevel.assert_not_called()
|
||||
identity.assert_not_called()
|
||||
|
||||
def test_the_same_spies_do_fire_for_a_supported_mode(self):
|
||||
"""Control: proves the previous test's assertions are not vacuous."""
|
||||
with patch.object(crr, "resolve_repo_toplevel",
|
||||
wraps=crr.resolve_repo_toplevel) as toplevel, \
|
||||
patch.object(crr, "repository_identity_slug",
|
||||
wraps=crr.repository_identity_slug) as identity:
|
||||
crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="validation",
|
||||
)
|
||||
toplevel.assert_called()
|
||||
identity.assert_called()
|
||||
|
||||
def test_nonexistent_candidate_root_still_reports_the_mode_refusal(self):
|
||||
"""No mocks: the existence check cannot have run before the refusal."""
|
||||
nonexistent = os.path.join(self.tmp_dir, "no-such-repository")
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=nonexistent,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="invalid_mode",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "invalid_mode")
|
||||
self.assertFalse(
|
||||
any("does not exist" in r for r in assessment["reasons"]), assessment
|
||||
)
|
||||
|
||||
def test_symlinked_candidate_root_is_not_resolved_for_an_unsupported_mode(self):
|
||||
link = os.path.join(self.tmp_dir, "target-alias")
|
||||
os.symlink(self.target_root, link)
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=link,
|
||||
source="env",
|
||||
expected_slug=TARGET_SLUG,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="invalid_mode",
|
||||
)
|
||||
self.assertRefusedForMode(assessment, "invalid_mode")
|
||||
# The refusal payload must not leak a resolved path for the candidate.
|
||||
self.assertIsNone(assessment["canonical_repo_root"], assessment)
|
||||
|
||||
|
||||
class TestB10NoModeInjectionSurface(unittest.TestCase):
|
||||
"""``mode`` must not be reachable from requests, environment, or config."""
|
||||
|
||||
PRODUCTION_MODULES = ("gitea_mcp_server.py", "namespace_workspace_binding.py")
|
||||
|
||||
def _repo_root(self) -> str:
|
||||
return os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
def test_production_call_sites_only_pass_allowlisted_literal_modes(self):
|
||||
"""Valid hardcoded modes reach the correct path; nothing else is passed."""
|
||||
seen: list[tuple[str, str | None]] = []
|
||||
for name in self.PRODUCTION_MODULES:
|
||||
path = os.path.join(self._repo_root(), name)
|
||||
with open(path) as handle:
|
||||
tree = ast.parse(handle.read())
|
||||
for node in ast.walk(tree):
|
||||
if not isinstance(node, ast.Call):
|
||||
continue
|
||||
func = node.func
|
||||
called = (
|
||||
func.attr if isinstance(func, ast.Attribute)
|
||||
else getattr(func, "id", None)
|
||||
)
|
||||
if called != "assess_canonical_repository_root":
|
||||
continue
|
||||
supplied = [k for k in node.keywords if k.arg == "mode"]
|
||||
if not supplied:
|
||||
seen.append((name, None))
|
||||
continue
|
||||
value = supplied[0].value
|
||||
self.assertIsInstance(
|
||||
value, ast.Constant,
|
||||
f"{name}: mode must be a literal, never a variable or expression",
|
||||
)
|
||||
self.assertIn(
|
||||
value.value, crr.SUPPORTED_MODES,
|
||||
f"{name}: unsupported mode literal {value.value!r}",
|
||||
)
|
||||
seen.append((name, value.value))
|
||||
self.assertTrue(seen, "expected production call sites to be found")
|
||||
# Both documented modes are exercised by production, and omission is used.
|
||||
self.assertIn(None, [mode for _, mode in seen])
|
||||
self.assertIn("derivation", [mode for _, mode in seen])
|
||||
|
||||
def test_no_public_entry_point_exposes_a_mode_parameter(self):
|
||||
for func in (
|
||||
nwb.resolve_namespace_mutation_context,
|
||||
nwb.assess_namespace_mutation_workspace,
|
||||
mcp_server._resolve_namespace_mutation_context,
|
||||
mcp_server._enforce_canonical_repository_root,
|
||||
mcp_server._canonical_repository_slug,
|
||||
mcp_server._trusted_session_repository,
|
||||
mcp_server._resolve_expected_repository_slug,
|
||||
):
|
||||
with self.subTest(func=func.__name__):
|
||||
self.assertNotIn("mode", inspect.signature(func).parameters)
|
||||
|
||||
def test_no_environment_key_selects_a_repository_authority_mode(self):
|
||||
for key in gitea_config.RECOGNIZED_GITEA_ENV_KEYS:
|
||||
self.assertNotIn(
|
||||
"CANONICAL_REPOSITORY_MODE", key.upper(),
|
||||
f"{key} would expose a repository-authority mode selector",
|
||||
)
|
||||
# An invented mode-ish variable is simply not consumed by crr.
|
||||
env = {
|
||||
"GITEA_CANONICAL_REPOSITORY_ROOT": "/some/path",
|
||||
"GITEA_CANONICAL_REPOSITORY_MODE": "invalid_mode",
|
||||
}
|
||||
value, source = crr.configured_canonical_root(None, env)
|
||||
self.assertEqual(value, "/some/path")
|
||||
self.assertNotIn("mode", (source or "").lower())
|
||||
self.assertIn(
|
||||
"GITEA_CANONICAL_REPOSITORY_MODE",
|
||||
gitea_config.get_unconsumed_gitea_env_overrides(env),
|
||||
"an unknown GITEA_* key must still be rejected as unrecognised",
|
||||
)
|
||||
|
||||
def test_repository_configuration_carries_no_mode_field(self):
|
||||
profile = {
|
||||
"canonical_repository_root": "/some/path",
|
||||
"mode": "invalid_mode",
|
||||
}
|
||||
value, source = crr.configured_canonical_root(profile, {})
|
||||
self.assertEqual(value, "/some/path")
|
||||
self.assertEqual(source, "profile canonical_repository_root")
|
||||
|
||||
|
||||
class TestB10ProductionEnforcementPaths(_CanonicalRootFixture):
|
||||
"""Production enforcement, mutation-context, reviewer and merger consumers."""
|
||||
|
||||
def _force_invalid_mode(self):
|
||||
"""Simulate a future call site threading an unsupported mode.
|
||||
|
||||
The real ``assess_canonical_repository_root`` still executes — only the
|
||||
caller-side argument is substituted — so the refusal under test is
|
||||
produced by production code, not by a stand-in. No caller-controlled
|
||||
``mode`` parameter is added to any production signature to achieve this.
|
||||
"""
|
||||
real = crr.assess_canonical_repository_root
|
||||
|
||||
def _wrapper(**kwargs):
|
||||
kwargs["mode"] = "invalid_mode"
|
||||
return real(**kwargs)
|
||||
|
||||
return patch.object(crr, "assess_canonical_repository_root", _wrapper)
|
||||
|
||||
def test_mutation_context_fails_closed_under_a_refused_mode(self):
|
||||
with self._force_invalid_mode():
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug=TARGET_SLUG,
|
||||
remote="prgs",
|
||||
)
|
||||
assessment = ctx["canonical_root_assessment"]
|
||||
self.assertFalse(ctx["roots_aligned"], ctx)
|
||||
self.assertTrue(assessment["block"], assessment)
|
||||
self.assertEqual(assessment["reason_code"], crr.DENY_UNKNOWN_MODE)
|
||||
self.assertIsNone(assessment["resolved_slug"])
|
||||
# The refused mode must not yield the candidate root as canonical.
|
||||
self.assertNotEqual(ctx["canonical_repo_root"], self.target_root)
|
||||
self.assertEqual(ctx["canonical_repo_root"], self.install_root)
|
||||
|
||||
def test_mutation_context_unchanged_for_the_supported_default(self):
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug=TARGET_SLUG,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertTrue(ctx["roots_aligned"], ctx)
|
||||
self.assertEqual(ctx["canonical_repo_root"], self.target_root)
|
||||
self.assertIsNone(ctx["canonical_root_assessment"]["reason_code"])
|
||||
|
||||
def test_reviewer_and_merger_authorization_fails_closed_under_a_refused_mode(self):
|
||||
for role in ("reviewer", "merger"):
|
||||
with self.subTest(role=role), self._force_invalid_mode():
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind=role,
|
||||
worktree_path=self.target_worktree,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug=TARGET_SLUG,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertTrue(assessment["block"], assessment)
|
||||
self.assertTrue(
|
||||
any("unsupported repository-authority mode" in r
|
||||
for r in assessment["reasons"]),
|
||||
assessment,
|
||||
)
|
||||
|
||||
def test_reviewer_and_merger_authorization_unchanged_for_supported_modes(self):
|
||||
for role in ("reviewer", "merger"):
|
||||
with self.subTest(role=role):
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind=role,
|
||||
worktree_path=self.target_worktree,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug=TARGET_SLUG,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertFalse(assessment["block"], assessment)
|
||||
|
||||
def test_final_mutation_gate_blocks_when_the_mode_refusal_unaligns_roots(self):
|
||||
with self._force_invalid_mode():
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug=TARGET_SLUG,
|
||||
remote="prgs",
|
||||
)
|
||||
report = stable_control_runtime.build_runtime_report(
|
||||
process_root=self.install_root,
|
||||
checkout_branch="master",
|
||||
runtime_head="abcdef123456",
|
||||
active_task_workspace=ctx["workspace_path"],
|
||||
canonical_repository_root=ctx["canonical_repo_root"],
|
||||
workspace_roots_aligned=ctx["roots_aligned"],
|
||||
)
|
||||
gate = stable_control_runtime.assess_runtime_mutation_gate(report)
|
||||
self.assertTrue(gate["block"], gate)
|
||||
|
||||
def test_enforce_canonical_repository_root_uses_the_validation_default(self):
|
||||
"""B8 boundary intact: a foreign configured root raises through production."""
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server, "_configured_canonical_root",
|
||||
return_value=(self.evil_root, "env")), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context",
|
||||
return_value=None):
|
||||
with self.assertRaises(RuntimeError) as raised:
|
||||
mcp_server._enforce_canonical_repository_root(remote="prgs")
|
||||
self.assertIn("identity mismatch", str(raised.exception))
|
||||
|
||||
def test_enforce_canonical_repository_root_refuses_a_threaded_invalid_mode(self):
|
||||
bound = {"org": "Scaled-Tech-Consulting",
|
||||
"repository": "mcp-control-plane", "remote": "prgs"}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server, "_configured_canonical_root",
|
||||
return_value=(self.target_root, "env")), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context",
|
||||
return_value=bound), \
|
||||
self._force_invalid_mode():
|
||||
with self.assertRaises(RuntimeError) as raised:
|
||||
mcp_server._enforce_canonical_repository_root(remote="prgs")
|
||||
self.assertIn("unsupported repository-authority mode", str(raised.exception))
|
||||
|
||||
def test_enforce_canonical_repository_root_passes_for_a_valid_binding(self):
|
||||
bound = {"org": "Scaled-Tech-Consulting",
|
||||
"repository": "mcp-control-plane", "remote": "prgs"}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server, "_configured_canonical_root",
|
||||
return_value=(self.target_root, "env")), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context",
|
||||
return_value=bound), \
|
||||
patch.object(mcp_server.session_ctx, "assess_session_context",
|
||||
return_value={"block": False, "reasons": []}), \
|
||||
patch.object(mcp_server, "get_profile",
|
||||
return_value={"profile_name": "prgs-reviewer"}):
|
||||
mcp_server._enforce_canonical_repository_root(remote="prgs")
|
||||
|
||||
def test_canonical_repository_slug_derivation_unbroken(self):
|
||||
"""B9 boundary intact: legitimate cross-repository derivation still works."""
|
||||
profile = {
|
||||
"profile_name": "prgs-author",
|
||||
"allowed_repositories": [TARGET_SLUG],
|
||||
"canonical_repository_root": self.target_root,
|
||||
}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context",
|
||||
return_value=None):
|
||||
slug, reasons = mcp_server._canonical_repository_slug(profile, "prgs")
|
||||
self.assertEqual(slug, TARGET_SLUG, reasons)
|
||||
self.assertEqual(reasons, [])
|
||||
|
||||
def test_canonical_repository_slug_fails_closed_under_a_refused_mode(self):
|
||||
profile = {
|
||||
"profile_name": "prgs-author",
|
||||
"allowed_repositories": [TARGET_SLUG],
|
||||
"canonical_repository_root": self.target_root,
|
||||
}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context",
|
||||
return_value=None), \
|
||||
self._force_invalid_mode():
|
||||
slug, reasons = mcp_server._canonical_repository_slug(profile, "prgs")
|
||||
self.assertIsNone(slug)
|
||||
self.assertTrue(
|
||||
any("unsupported repository-authority mode" in r for r in reasons),
|
||||
reasons,
|
||||
)
|
||||
result = mcp_server._trusted_session_repository(
|
||||
profile, "prgs", for_mutation=True
|
||||
)
|
||||
self.assertIsNone(result["org"])
|
||||
self.assertIsNone(result["repository"])
|
||||
self.assertTrue(result["reasons"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -0,0 +1,546 @@
|
||||
"""Regression tests for Issue #973: validated cross-repository canonical roots."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest.mock import patch, MagicMock
|
||||
from pathlib import Path
|
||||
import subprocess
|
||||
|
||||
import gitea_config
|
||||
import namespace_workspace_binding as nwb
|
||||
import canonical_repository_root as crr
|
||||
import stable_control_runtime
|
||||
import gitea_mcp_server as mcp_server
|
||||
|
||||
|
||||
class TestIssue973RecognizedEnvKeys(unittest.TestCase):
|
||||
"""Test recognized environment variable keys under #973."""
|
||||
|
||||
def test_recognized_gitea_env_keys(self):
|
||||
for key in (
|
||||
"GITEA_CANONICAL_REPOSITORY_ROOT",
|
||||
"GITEA_REVIEWER_WORKTREE",
|
||||
"GITEA_MERGER_WORKTREE",
|
||||
"GITEA_MCP_SESSION_STATE_TTL_HOURS",
|
||||
):
|
||||
self.assertIn(key, gitea_config.RECOGNIZED_GITEA_ENV_KEYS)
|
||||
|
||||
def test_get_unconsumed_gitea_env_overrides_ignores_recognized(self):
|
||||
env = {
|
||||
"GITEA_CANONICAL_REPOSITORY_ROOT": "/some/path",
|
||||
"GITEA_REVIEWER_WORKTREE": "/some/reviewer/path",
|
||||
"GITEA_MERGER_WORKTREE": "/some/merger/path",
|
||||
"GITEA_MCP_SESSION_STATE_TTL_HOURS": "24",
|
||||
"GITEA_UNRECOGNIZED_FOO_VAR": "bar",
|
||||
}
|
||||
unconsumed = gitea_config.get_unconsumed_gitea_env_overrides(env)
|
||||
self.assertNotIn("GITEA_CANONICAL_REPOSITORY_ROOT", unconsumed)
|
||||
self.assertNotIn("GITEA_REVIEWER_WORKTREE", unconsumed)
|
||||
self.assertNotIn("GITEA_MERGER_WORKTREE", unconsumed)
|
||||
self.assertNotIn("GITEA_MCP_SESSION_STATE_TTL_HOURS", unconsumed)
|
||||
self.assertIn("GITEA_UNRECOGNIZED_FOO_VAR", unconsumed)
|
||||
|
||||
|
||||
class TestIssue973CrossRepoCanonicalRoots(unittest.TestCase):
|
||||
"""Test workspace binding and canonical root validation for cross-repo namespaces."""
|
||||
|
||||
def setUp(self):
|
||||
self._tmp = tempfile.TemporaryDirectory()
|
||||
self.tmp_dir = os.path.realpath(self._tmp.name)
|
||||
|
||||
# Create simulated installation root
|
||||
self.install_root = os.path.join(self.tmp_dir, "Gitea-Tools")
|
||||
os.makedirs(self.install_root)
|
||||
subprocess.run(["git", "init", "-b", "master"], cwd=self.install_root, check=True)
|
||||
subprocess.run(["git", "config", "user.email", "[email protected]"], cwd=self.install_root, check=True)
|
||||
subprocess.run(["git", "config", "user.name", "Test User"], cwd=self.install_root, check=True)
|
||||
with open(os.path.join(self.install_root, "README.md"), "w") as f:
|
||||
f.write("install\n")
|
||||
subprocess.run(["git", "add", "README.md"], cwd=self.install_root, check=True)
|
||||
subprocess.run(["git", "commit", "-m", "initial"], cwd=self.install_root, check=True)
|
||||
|
||||
# Create simulated target repository root
|
||||
self.target_root = os.path.join(self.tmp_dir, "mcp-control-plane")
|
||||
os.makedirs(self.target_root)
|
||||
subprocess.run(["git", "init", "-b", "master"], cwd=self.target_root, check=True)
|
||||
subprocess.run(["git", "config", "user.email", "[email protected]"], cwd=self.target_root, check=True)
|
||||
subprocess.run(["git", "config", "user.name", "Test User"], cwd=self.target_root, check=True)
|
||||
with open(os.path.join(self.target_root, "README.md"), "w") as f:
|
||||
f.write("target\n")
|
||||
subprocess.run(["git", "add", "README.md"], cwd=self.target_root, check=True)
|
||||
subprocess.run(["git", "commit", "-m", "initial"], cwd=self.target_root, check=True)
|
||||
|
||||
# Add remotes to simulate real git repositories with identities
|
||||
subprocess.run(["git", "remote", "add", "prgs", "https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools.git"], cwd=self.install_root, check=True)
|
||||
subprocess.run(["git", "remote", "add", "prgs", "https://gitea.prgs.cc/Scaled-Tech-Consulting/mcp-control-plane.git"], cwd=self.target_root, check=True)
|
||||
|
||||
# Create simulated foreign repository root
|
||||
self.evil_root = os.path.join(self.tmp_dir, "Evil-Repo")
|
||||
os.makedirs(self.evil_root)
|
||||
subprocess.run(["git", "init", "-b", "master"], cwd=self.evil_root, check=True)
|
||||
subprocess.run(["git", "config", "user.email", "[email protected]"], cwd=self.evil_root, check=True)
|
||||
subprocess.run(["git", "config", "user.name", "Evil User"], cwd=self.evil_root, check=True)
|
||||
with open(os.path.join(self.evil_root, "README.md"), "w") as f:
|
||||
f.write("evil\n")
|
||||
subprocess.run(["git", "add", "README.md"], cwd=self.evil_root, check=True)
|
||||
subprocess.run(["git", "commit", "-m", "initial"], cwd=self.evil_root, check=True)
|
||||
subprocess.run(["git", "remote", "add", "prgs", "https://gitea.prgs.cc/Someone-Else/Evil-Repo.git"], cwd=self.evil_root, check=True)
|
||||
|
||||
# Create branches/ directory and a valid registered worktree in target repository
|
||||
self.target_branches = os.path.join(self.target_root, "branches")
|
||||
self.target_worktree = os.path.join(self.target_branches, "rev-pr-99")
|
||||
subprocess.run(["git", "worktree", "add", "-b", "rev-pr-99", self.target_worktree], cwd=self.target_root, check=True)
|
||||
|
||||
def tearDown(self):
|
||||
self._tmp.cleanup()
|
||||
|
||||
def test_valid_same_repository_configuration(self):
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=None,
|
||||
)
|
||||
self.assertEqual(ctx["canonical_repo_root"], self.install_root)
|
||||
self.assertTrue(ctx["roots_aligned"])
|
||||
self.assertTrue(ctx["canonical_root_assessment"]["proven"])
|
||||
|
||||
def test_valid_cross_repo_canonical_root(self):
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertEqual(ctx["canonical_repo_root"], self.target_root)
|
||||
self.assertTrue(ctx["roots_aligned"])
|
||||
self.assertTrue(ctx["canonical_root_assessment"]["proven"])
|
||||
|
||||
def test_expected_repository_identity_match(self):
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="test",
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertTrue(assessment["proven"])
|
||||
self.assertFalse(assessment["block"])
|
||||
|
||||
def test_foreign_repository_identity_mismatch(self):
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.evil_root,
|
||||
source="test",
|
||||
expected_slug="Scaled-Tech-Consulting/Gitea-Tools",
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertFalse(assessment["proven"])
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("identity mismatch" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_native_repository_binding_mismatch(self):
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.evil_root,
|
||||
expected_slug="Scaled-Tech-Consulting/Gitea-Tools",
|
||||
remote="prgs",
|
||||
)
|
||||
self.assertFalse(ctx["roots_aligned"])
|
||||
self.assertFalse(ctx["canonical_root_assessment"]["proven"])
|
||||
self.assertTrue(any("identity mismatch" in r for r in ctx["canonical_root_assessment"]["reasons"]))
|
||||
|
||||
def test_unpatched_foreign_configured_root_derives_expected_from_process_root_and_blocks(self):
|
||||
"""B8: Production path test where foreign configured root cannot self-authorize."""
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server, "_configured_canonical_root", return_value=(self.evil_root, "env")):
|
||||
expected_slug = mcp_server._resolve_expected_repository_slug("prgs")
|
||||
self.assertEqual(expected_slug, "Scaled-Tech-Consulting/Gitea-Tools")
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.evil_root,
|
||||
source="env",
|
||||
expected_slug=expected_slug,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
)
|
||||
self.assertFalse(assessment["proven"])
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertEqual(assessment["resolved_slug"], "Someone-Else/Evil-Repo")
|
||||
self.assertTrue(any("identity mismatch" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_unpatched_valid_cross_repo_matching_session_context(self):
|
||||
"""B8: Valid cross-repo namespace matches when session context is bound to target repo."""
|
||||
bound_ctx = {"org": "Scaled-Tech-Consulting", "repository": "mcp-control-plane", "remote": "prgs"}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context", return_value=bound_ctx):
|
||||
expected_slug = mcp_server._resolve_expected_repository_slug("prgs")
|
||||
self.assertEqual(expected_slug, "Scaled-Tech-Consulting/mcp-control-plane")
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=expected_slug,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
)
|
||||
self.assertTrue(assessment["proven"])
|
||||
self.assertFalse(assessment["block"])
|
||||
self.assertEqual(assessment["resolved_slug"], "Scaled-Tech-Consulting/mcp-control-plane")
|
||||
|
||||
def test_unprovable_expected_identity_fails_closed(self):
|
||||
"""B8: If expected repository identity is unprovable for a configured root, fail closed."""
|
||||
no_remote_root = os.path.join(self.tmp_dir, "no-remote-process-root")
|
||||
os.makedirs(no_remote_root)
|
||||
subprocess.run(["git", "init", "-b", "master"], cwd=no_remote_root, check=True)
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", no_remote_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context", return_value=None):
|
||||
expected_slug = mcp_server._resolve_expected_repository_slug("prgs")
|
||||
self.assertIsNone(expected_slug)
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=expected_slug,
|
||||
process_project_root=no_remote_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
)
|
||||
self.assertFalse(assessment["proven"])
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("unprovable or missing" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_missing_canonical_root(self):
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="author",
|
||||
worktree_path=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root="",
|
||||
)
|
||||
self.assertEqual(ctx["canonical_repo_root"], self.install_root)
|
||||
self.assertTrue(ctx["roots_aligned"])
|
||||
|
||||
def test_nonexistent_configured_canonical_root(self):
|
||||
nonexistent = os.path.join(self.tmp_dir, "nonexistent-repo")
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=nonexistent,
|
||||
)
|
||||
self.assertFalse(ctx["roots_aligned"])
|
||||
self.assertFalse(ctx["canonical_root_assessment"]["proven"])
|
||||
self.assertTrue(any("does not exist" in r for r in ctx["canonical_root_assessment"]["reasons"]))
|
||||
|
||||
def test_non_git_configured_canonical_root(self):
|
||||
non_git = os.path.join(self.tmp_dir, "non-git-dir")
|
||||
os.makedirs(non_git)
|
||||
ctx = nwb.resolve_namespace_mutation_context(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=non_git,
|
||||
)
|
||||
self.assertFalse(ctx["roots_aligned"])
|
||||
self.assertFalse(ctx["canonical_root_assessment"]["proven"])
|
||||
self.assertTrue(any("not a git repository" in r for r in ctx["canonical_root_assessment"]["reasons"]))
|
||||
|
||||
def test_git_common_directory_membership_matching(self):
|
||||
valid, err = nwb.verify_git_common_directory_membership(
|
||||
self.target_worktree, self.target_root
|
||||
)
|
||||
self.assertTrue(valid, err)
|
||||
self.assertIsNone(err)
|
||||
|
||||
def test_foreign_git_common_directory(self):
|
||||
# Foreign worktree created under install_root
|
||||
install_branches = os.path.join(self.install_root, "branches")
|
||||
foreign_wt = os.path.join(install_branches, "foreign-wt")
|
||||
subprocess.run(["git", "worktree", "add", "-b", "foreign-wt", foreign_wt], cwd=self.install_root, check=True)
|
||||
|
||||
valid, err = nwb.verify_git_common_directory_membership(
|
||||
foreign_wt, self.target_root
|
||||
)
|
||||
self.assertFalse(valid)
|
||||
self.assertIn("does not match", err)
|
||||
|
||||
def test_normalized_path_aliases(self):
|
||||
alias_path = self.target_worktree + "/../rev-pr-99/./"
|
||||
valid, err = nwb.verify_git_common_directory_membership(
|
||||
alias_path, self.target_root
|
||||
)
|
||||
self.assertTrue(valid, err)
|
||||
|
||||
def test_safe_symlink_identity(self):
|
||||
link_path = os.path.join(self.target_branches, "symlink-rev-99")
|
||||
try:
|
||||
os.symlink(self.target_worktree, link_path)
|
||||
valid, err = nwb.verify_git_common_directory_membership(
|
||||
link_path, self.target_root
|
||||
)
|
||||
self.assertTrue(valid, err)
|
||||
finally:
|
||||
if os.path.exists(link_path):
|
||||
os.unlink(link_path)
|
||||
|
||||
def test_symlink_escape_or_foreign_alias(self):
|
||||
outside_dir = os.path.join(self.tmp_dir, "outside-target")
|
||||
os.makedirs(outside_dir)
|
||||
link_escape = os.path.join(self.target_branches, "escape-link")
|
||||
try:
|
||||
os.symlink(outside_dir, link_escape)
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind="reviewer",
|
||||
worktree_path=link_escape,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(assessment["block"])
|
||||
finally:
|
||||
if os.path.exists(link_escape):
|
||||
os.unlink(link_escape)
|
||||
|
||||
def test_reviewer_worktree_registered_and_valid(self):
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind="reviewer",
|
||||
worktree_path=self.target_worktree,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertFalse(assessment["block"])
|
||||
|
||||
def test_reviewer_worktree_unregistered_blocks(self):
|
||||
unreg_wt = os.path.join(self.target_branches, "unregistered-reviewer")
|
||||
os.makedirs(unreg_wt)
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind="reviewer",
|
||||
worktree_path=unreg_wt,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("is not registered in git worktree list" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_merger_worktree_registered_and_valid(self):
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind="merger",
|
||||
worktree_path=self.target_worktree,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertFalse(assessment["block"])
|
||||
|
||||
def test_merger_worktree_unregistered_blocks(self):
|
||||
unreg_wt = os.path.join(self.target_branches, "unregistered-merger")
|
||||
os.makedirs(unreg_wt)
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind="merger",
|
||||
worktree_path=unreg_wt,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("is not registered in git worktree list" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_reviewer_or_merger_worktree_outside_branches_blocks(self):
|
||||
outside_wt = os.path.join(self.target_root, "outside_branches_wt")
|
||||
os.makedirs(outside_wt)
|
||||
for r_kind in ("reviewer", "merger"):
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind=r_kind,
|
||||
worktree_path=outside_wt,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("is not under" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_nonexistent_reviewer_or_merger_worktree_blocks(self):
|
||||
nonexistent_wt = os.path.join(self.target_branches, "nonexistent-wt")
|
||||
for r_kind in ("reviewer", "merger"):
|
||||
assessment = nwb.assess_namespace_mutation_workspace(
|
||||
role_kind=r_kind,
|
||||
worktree_path=nonexistent_wt,
|
||||
worktree=None,
|
||||
process_project_root=self.install_root,
|
||||
env={},
|
||||
configured_canonical_root=self.target_root,
|
||||
expected_slug="Scaled-Tech-Consulting/mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("does not exist" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_safe_and_unsafe_mutation_alignment_outcomes(self):
|
||||
# Safe alignment (same repo)
|
||||
report_safe = stable_control_runtime.build_runtime_report(
|
||||
process_root=self.install_root,
|
||||
checkout_branch="master",
|
||||
runtime_head="abcdef123456",
|
||||
active_task_workspace=self.install_root,
|
||||
canonical_repository_root=self.install_root,
|
||||
workspace_roots_aligned=True,
|
||||
)
|
||||
gate_safe = stable_control_runtime.assess_runtime_mutation_gate(report_safe)
|
||||
self.assertFalse(gate_safe["block"])
|
||||
|
||||
# Unsafe alignment
|
||||
report_unsafe = stable_control_runtime.build_runtime_report(
|
||||
process_root=self.install_root,
|
||||
checkout_branch="master",
|
||||
runtime_head="abcdef123456",
|
||||
active_task_workspace=self.target_worktree,
|
||||
canonical_repository_root=self.target_root,
|
||||
workspace_roots_aligned=False,
|
||||
)
|
||||
gate_unsafe = stable_control_runtime.assess_runtime_mutation_gate(report_unsafe)
|
||||
self.assertTrue(gate_unsafe["block"])
|
||||
|
||||
def test_reviewer_lease_lifecycle_production_path(self):
|
||||
"""B5: Automated regression for reviewer lease acquire and release through production path."""
|
||||
mock_whoami = {
|
||||
"authenticated": True,
|
||||
"username": "sysadmin",
|
||||
"remote": "prgs",
|
||||
"profile": {
|
||||
"profile_name": "prgs-reviewer",
|
||||
"role": "reviewer",
|
||||
"role_kind": "reviewer",
|
||||
"allowed_operations": ["gitea.read", "gitea.pr.comment", "gitea.pr.approve", "gitea.pr.request_changes"],
|
||||
"forbidden_operations": [],
|
||||
},
|
||||
}
|
||||
|
||||
def mock_api_request(method, url, auth=None, json_data=None):
|
||||
if method == "GET":
|
||||
return {"number": 99, "head": {"sha": "abc1234"}, "state": "open", "merged": False, "merged_at": None}
|
||||
elif method == "POST":
|
||||
return {"id": 9999, "body": (json_data or {}).get("body", "")}
|
||||
return {}
|
||||
|
||||
with patch.object(mcp_server, "gitea_whoami", return_value=mock_whoami), \
|
||||
patch.object(mcp_server, "get_profile", return_value=mock_whoami["profile"]), \
|
||||
patch.object(mcp_server, "_effective_workspace_role", return_value="reviewer"), \
|
||||
patch.object(mcp_server, "_configured_canonical_root", return_value=(self.target_root, "env")), \
|
||||
patch.object(mcp_server, "_reviewer_session_worktree", return_value=self.target_worktree), \
|
||||
patch.object(mcp_server, "_auth", return_value="token mock-token"), \
|
||||
patch.object(mcp_server, "_fetch_pr_comments", return_value=[]), \
|
||||
patch.object(mcp_server, "api_request", side_effect=mock_api_request), \
|
||||
patch("reviewer_pr_lease.assess_acquire_lease", return_value={"acquire_allowed": True, "reasons": [], "lease_body": "<!-- LEASE -->"}), \
|
||||
patch("reviewer_pr_lease.find_active_reviewer_lease", return_value={"session_id": "sid-123", "reviewer": "sysadmin"}), \
|
||||
patch("reviewer_pr_lease.get_session_lease", return_value={"session_id": "sid-123", "reviewer": "sysadmin"}), \
|
||||
patch("reviewer_pr_lease.clear_session_lease") as mock_clear:
|
||||
|
||||
acq_res = mcp_server.gitea_acquire_reviewer_pr_lease(
|
||||
pr_number=99,
|
||||
remote="prgs",
|
||||
worktree=self.target_worktree,
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(acq_res.get("success"), acq_res)
|
||||
|
||||
rel_res = mcp_server.gitea_release_reviewer_pr_lease(
|
||||
pr_number=99,
|
||||
worktree=self.target_worktree,
|
||||
remote="prgs",
|
||||
org="Scaled-Tech-Consulting",
|
||||
repo="mcp-control-plane",
|
||||
)
|
||||
self.assertTrue(rel_res.get("success"), rel_res)
|
||||
mock_clear.assert_called_once()
|
||||
|
||||
def test_unbound_session_derives_and_retains_configured_target_repository(self):
|
||||
"""B9: Unbound session with a legitimate configured target root derives and retains target repository."""
|
||||
prof = {
|
||||
"profile_name": "prgs-author",
|
||||
"allowed_repositories": ["Scaled-Tech-Consulting/mcp-control-plane"],
|
||||
"canonical_repository_root": self.target_root,
|
||||
}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context", return_value=None):
|
||||
slug, reasons = mcp_server._canonical_repository_slug(prof, "prgs")
|
||||
self.assertEqual(slug, "Scaled-Tech-Consulting/mcp-control-plane", f"reasons: {reasons}")
|
||||
self.assertEqual(reasons, [])
|
||||
res = mcp_server._trusted_session_repository(prof, "prgs")
|
||||
self.assertEqual(res["org"], "Scaled-Tech-Consulting")
|
||||
self.assertEqual(res["repository"], "mcp-control-plane")
|
||||
|
||||
def test_process_root_a_configured_target_b_retains_b(self):
|
||||
"""B9: Process root A (Gitea-Tools) and configured target B (mcp-control-plane) keep B as expected identity when bound."""
|
||||
bound_ctx = {"org": "Scaled-Tech-Consulting", "repository": "mcp-control-plane", "remote": "prgs"}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context", return_value=bound_ctx):
|
||||
expected = mcp_server._resolve_expected_repository_slug("prgs")
|
||||
self.assertEqual(expected, "Scaled-Tech-Consulting/mcp-control-plane")
|
||||
|
||||
def test_request_supplied_repository_c_cannot_replace_derived_or_bound_repository_b(self):
|
||||
"""B9: Request-supplied repository C (Timesheet) cannot replace bound repository B (mcp-control-plane)."""
|
||||
bound_ctx = {"org": "Scaled-Tech-Consulting", "repository": "mcp-control-plane", "remote": "prgs"}
|
||||
with patch.object(mcp_server, "PROJECT_ROOT", self.install_root), \
|
||||
patch.object(mcp_server.session_ctx, "get_session_context", return_value=bound_ctx):
|
||||
expected = mcp_server._resolve_expected_repository_slug("prgs", org="Scaled-Tech-Consulting", repo="Timesheet")
|
||||
self.assertEqual(expected, "Scaled-Tech-Consulting/mcp-control-plane")
|
||||
|
||||
def test_known_expected_repo_a_plus_candidate_root_b_fails_closed(self):
|
||||
"""B9: Known expected repository A plus candidate root B fails closed in validation mode."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug="Scaled-Tech-Consulting/Gitea-Tools",
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="validation",
|
||||
)
|
||||
self.assertFalse(assessment["proven"])
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("identity mismatch" in r for r in assessment["reasons"]))
|
||||
|
||||
def test_validation_mode_with_unprovable_expected_identity_fails_closed(self):
|
||||
"""B9: Validation mode with expected_slug=None and require_binding=True fails closed."""
|
||||
assessment = crr.assess_canonical_repository_root(
|
||||
configured_value=self.target_root,
|
||||
source="env",
|
||||
expected_slug=None,
|
||||
process_project_root=self.install_root,
|
||||
remote="prgs",
|
||||
require_binding=True,
|
||||
mode="validation",
|
||||
)
|
||||
self.assertFalse(assessment["proven"])
|
||||
self.assertTrue(assessment["block"])
|
||||
self.assertTrue(any("unprovable or missing" in r for r in assessment["reasons"]))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,806 @@
|
||||
"""CAS-protected stale worker retirement (#980).
|
||||
|
||||
Covers the acceptance criteria: the registry CAS token is stable across time
|
||||
and row order but moves on any retirement-relevant change, plan mutates
|
||||
nothing, apply fails closed on drift, every target is revalidated inside the
|
||||
retirement transaction, and live / ambiguous / incomplete rows are preserved.
|
||||
|
||||
The regression test that matters most is
|
||||
``test_snapshot_at_would_have_moved_the_token``: it demonstrates the exact
|
||||
defect the R3-C review found — ``mcp_fleet_snapshot._consistency_token``
|
||||
changes when only the observation time changed — and proves the new token does
|
||||
not.
|
||||
|
||||
All identifiers are synthetic.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import tempfile
|
||||
import unittest
|
||||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Any
|
||||
|
||||
import mcp_fleet_retirement as retire
|
||||
import mcp_fleet_snapshot as fleet
|
||||
import mcp_worker_identity as mwi
|
||||
|
||||
|
||||
NOW = datetime(2026, 7, 30, 7, 0, 0, tzinfo=timezone.utc)
|
||||
TTL = 900.0
|
||||
REPO = "/synthetic/repo/Gitea-Tools"
|
||||
|
||||
_ROW_COLUMNS = (
|
||||
"worker_identity",
|
||||
"client_name",
|
||||
"client_instance_id",
|
||||
"session_id",
|
||||
"generation_id",
|
||||
"role",
|
||||
"profile",
|
||||
"namespace",
|
||||
"remote",
|
||||
"repository_binding",
|
||||
"pid",
|
||||
"process_identity",
|
||||
"transport",
|
||||
"token_fingerprint",
|
||||
"started_at",
|
||||
"last_heartbeat_at",
|
||||
"heartbeat_ttl_seconds",
|
||||
"fencing_epoch",
|
||||
"status",
|
||||
"fleet_run_id",
|
||||
"authenticated_account",
|
||||
"instance_id_provenance",
|
||||
)
|
||||
|
||||
|
||||
def _registry() -> mwi.WorkerRegistry:
|
||||
handle, path = tempfile.mkstemp(suffix=".sqlite3")
|
||||
os.close(handle)
|
||||
os.unlink(path)
|
||||
return mwi.WorkerRegistry(path)
|
||||
|
||||
|
||||
def _row(
|
||||
*,
|
||||
identity: str,
|
||||
pid: int | None,
|
||||
instance: str = "inst-codex-20260730T070000Z-0123456789ab",
|
||||
session: str = "sess-a",
|
||||
generation: str = "gen-a",
|
||||
namespace: str = "author",
|
||||
heartbeat: datetime | None = None,
|
||||
status: str = mwi.STATUS_ACTIVE,
|
||||
repository_binding: str | None = REPO,
|
||||
ttl: float = TTL,
|
||||
**overrides: Any,
|
||||
) -> dict[str, Any]:
|
||||
"""One synthetic ``worker_registrations`` row."""
|
||||
record = {
|
||||
"worker_identity": identity,
|
||||
"client_name": "codex",
|
||||
"client_instance_id": instance,
|
||||
"session_id": session,
|
||||
"generation_id": generation,
|
||||
"role": namespace,
|
||||
"profile": f"prgs-{namespace}",
|
||||
"namespace": namespace,
|
||||
"remote": "prgs",
|
||||
"repository_binding": repository_binding,
|
||||
"pid": pid,
|
||||
"process_identity": f"pid-{pid}" if pid is not None else None,
|
||||
"transport": "stdio",
|
||||
"token_fingerprint": None,
|
||||
"started_at": "2026-07-30T06:00:00Z",
|
||||
"last_heartbeat_at": mwi._ts(heartbeat or (NOW - timedelta(seconds=7200))),
|
||||
"heartbeat_ttl_seconds": ttl,
|
||||
"fencing_epoch": 1,
|
||||
"status": status,
|
||||
"fleet_run_id": "run-canary",
|
||||
"authenticated_account": "synthetic-user",
|
||||
"instance_id_provenance": fleet.INSTANCE_ID_PROVENANCE_TRUSTED,
|
||||
}
|
||||
record.update(overrides)
|
||||
return record
|
||||
|
||||
|
||||
def _dead(_pid: int | None) -> bool:
|
||||
return False
|
||||
|
||||
|
||||
def _alive(_pid: int | None) -> bool:
|
||||
return True
|
||||
|
||||
|
||||
def _selective(alive_pids: set[int]):
|
||||
def probe(pid: int | None) -> bool:
|
||||
return pid in alive_pids
|
||||
|
||||
return probe
|
||||
|
||||
|
||||
def _plan(rows, *, probe=_dead, now=NOW, **kwargs):
|
||||
return retire.plan_stale_worker_retirement(
|
||||
rows,
|
||||
now=now,
|
||||
pid_alive_probe=probe,
|
||||
canonical_repository=REPO,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
|
||||
def _register(registry: mwi.WorkerRegistry, row: dict[str, Any]) -> None:
|
||||
"""Insert a synthetic row directly, bypassing register()'s live-now stamps."""
|
||||
columns = [name for name in _ROW_COLUMNS if name in row]
|
||||
placeholders = ", ".join("?" for _ in columns)
|
||||
with registry._tx() as conn:
|
||||
conn.execute(
|
||||
f"INSERT INTO worker_registrations ({', '.join(columns)}) "
|
||||
f"VALUES ({placeholders})",
|
||||
[row[name] for name in columns],
|
||||
)
|
||||
|
||||
|
||||
class RegistryFingerprintStabilityTests(unittest.TestCase):
|
||||
"""AC: identical contents observed at different times produce one token."""
|
||||
|
||||
def test_same_contents_different_observation_times_same_token(self):
|
||||
rows = [_row(identity="w-1", pid=101), _row(identity="w-2", pid=102)]
|
||||
self.assertEqual(
|
||||
retire.registry_fingerprint(rows), retire.registry_fingerprint(rows)
|
||||
)
|
||||
# Recompute after the clock has moved a full hour: the token is derived
|
||||
# from content only, so it cannot notice.
|
||||
plan_early = _plan(rows, now=NOW)
|
||||
plan_late = _plan(rows, now=NOW + timedelta(hours=1))
|
||||
self.assertEqual(
|
||||
plan_early["registry_fingerprint"], plan_late["registry_fingerprint"]
|
||||
)
|
||||
|
||||
def test_snapshot_at_would_have_moved_the_token(self):
|
||||
"""Regression: the old time-seeded derivation moved, the new one does not."""
|
||||
rows = [_row(identity="w-1", pid=101)]
|
||||
old_early = fleet._consistency_token(rows, "2026-07-30T07:00:00Z")
|
||||
old_late = fleet._consistency_token(rows, "2026-07-30T07:00:01Z")
|
||||
self.assertNotEqual(
|
||||
old_early,
|
||||
old_late,
|
||||
"the #980 defect: one second of observation drift changed the token",
|
||||
)
|
||||
self.assertEqual(
|
||||
retire.registry_fingerprint(rows), retire.registry_fingerprint(rows)
|
||||
)
|
||||
|
||||
def test_row_order_does_not_change_the_token(self):
|
||||
rows = [
|
||||
_row(identity="w-1", pid=101),
|
||||
_row(identity="w-2", pid=102),
|
||||
_row(identity="w-3", pid=103),
|
||||
]
|
||||
self.assertEqual(
|
||||
retire.registry_fingerprint(rows),
|
||||
retire.registry_fingerprint(list(reversed(rows))),
|
||||
)
|
||||
|
||||
def test_numeric_typing_does_not_change_the_token(self):
|
||||
as_float = [_row(identity="w-1", pid=101, heartbeat_ttl_seconds=900.0)]
|
||||
as_int = [_row(identity="w-1", pid=101, heartbeat_ttl_seconds=900)]
|
||||
self.assertEqual(
|
||||
retire.registry_fingerprint(as_float),
|
||||
retire.registry_fingerprint(as_int),
|
||||
)
|
||||
|
||||
def test_retirement_relevant_changes_move_the_token(self):
|
||||
base = [_row(identity="w-1", pid=101)]
|
||||
baseline = retire.registry_fingerprint(base)
|
||||
mutations = {
|
||||
"row added": base + [_row(identity="w-2", pid=102)],
|
||||
"row removed": [],
|
||||
"status changed": [_row(identity="w-1", pid=101, status="released")],
|
||||
"identity changed": [
|
||||
_row(
|
||||
identity="w-1",
|
||||
pid=101,
|
||||
instance="inst-codex-20260730T070000Z-ffffffffffff",
|
||||
)
|
||||
],
|
||||
"ownership changed": [_row(identity="w-1", pid=101, session="sess-other")],
|
||||
"generation changed": [
|
||||
_row(identity="w-1", pid=101, generation="gen-other")
|
||||
],
|
||||
"pid changed": [_row(identity="w-1", pid=999)],
|
||||
"repository binding changed": [
|
||||
_row(identity="w-1", pid=101, repository_binding="/elsewhere")
|
||||
],
|
||||
}
|
||||
for label, rows in mutations.items():
|
||||
with self.subTest(change=label):
|
||||
self.assertNotEqual(baseline, retire.registry_fingerprint(rows))
|
||||
|
||||
def test_heartbeat_change_moves_the_token(self):
|
||||
base = [_row(identity="w-1", pid=101)]
|
||||
moved = [_row(identity="w-1", pid=101, heartbeat=NOW - timedelta(seconds=30))]
|
||||
self.assertNotEqual(
|
||||
retire.registry_fingerprint(base), retire.registry_fingerprint(moved)
|
||||
)
|
||||
|
||||
def test_ttl_change_moves_the_token(self):
|
||||
base = [_row(identity="w-1", pid=101)]
|
||||
moved = [_row(identity="w-1", pid=101, ttl=60.0)]
|
||||
self.assertNotEqual(
|
||||
retire.registry_fingerprint(base), retire.registry_fingerprint(moved)
|
||||
)
|
||||
|
||||
|
||||
class PlanEligibilityTests(unittest.TestCase):
|
||||
"""AC: only conclusively stale orphans are selected; everything else stays."""
|
||||
|
||||
def test_plan_selects_dead_stale_orphan(self):
|
||||
plan = _plan([_row(identity="w-1", pid=101)])
|
||||
self.assertEqual(plan["candidate_count"], 1)
|
||||
self.assertEqual(plan["candidate_worker_identities"], ["w-1"])
|
||||
self.assertEqual(plan["candidates"][0]["reason_code"], retire.REASON_ELIGIBLE)
|
||||
self.assertFalse(plan["mutation_performed"])
|
||||
self.assertTrue(plan["read_only"])
|
||||
|
||||
def test_plan_performs_no_mutation(self):
|
||||
registry = _registry()
|
||||
_register(registry, _row(identity="w-1", pid=101))
|
||||
before = registry.list_workers(status=None)
|
||||
plan = _plan(before)
|
||||
after = registry.list_workers(status=None)
|
||||
self.assertEqual(plan["candidate_count"], 1)
|
||||
self.assertEqual(before, after)
|
||||
self.assertEqual([r["status"] for r in after], [mwi.STATUS_ACTIVE])
|
||||
self.assertFalse(plan["mutation_performed"])
|
||||
|
||||
def test_live_worker_is_preserved(self):
|
||||
row = _row(identity="w-live", pid=101, heartbeat=NOW - timedelta(seconds=10))
|
||||
plan = _plan([row], probe=_alive)
|
||||
self.assertEqual(plan["candidate_count"], 0)
|
||||
self.assertEqual(
|
||||
plan["preserved"][0]["reason_code"], retire.REASON_WORKER_LIVE
|
||||
)
|
||||
|
||||
def test_fresh_heartbeat_with_dead_pid_still_fails_closed(self):
|
||||
"""A dead pid withdraws liveness; the unexpired heartbeat still preserves."""
|
||||
row = _row(identity="w-fresh", pid=101, heartbeat=NOW - timedelta(seconds=10))
|
||||
plan = _plan([row], probe=_dead)
|
||||
self.assertEqual(plan["candidate_count"], 0)
|
||||
self.assertEqual(
|
||||
plan["preserved"][0]["reason_code"], retire.REASON_HEARTBEAT_FRESH
|
||||
)
|
||||
|
||||
def test_unprobeable_pid_is_preserved(self):
|
||||
plan = _plan([_row(identity="w-1", pid=101)], probe=lambda _pid: None)
|
||||
self.assertEqual(plan["candidate_count"], 0)
|
||||
self.assertEqual(
|
||||
plan["preserved"][0]["reason_code"], retire.REASON_PID_UNKNOWN
|
||||
)
|
||||
|
||||
def test_incomplete_legacy_identity_is_preserved(self):
|
||||
"""AC: incomplete legacy identities remain fail-closed."""
|
||||
rows = [
|
||||
_row(identity="w-nopid", pid=None, instance="legacy-pid-27833"),
|
||||
_row(identity="w-nosession", pid=102, session=""),
|
||||
]
|
||||
plan = _plan(rows)
|
||||
self.assertEqual(plan["candidate_count"], 0)
|
||||
self.assertEqual(
|
||||
{p["reason_code"] for p in plan["preserved"]},
|
||||
{retire.REASON_INCOMPLETE_IDENTITY},
|
||||
)
|
||||
detail = next(
|
||||
p["detail"] for p in plan["preserved"] if p["worker_identity"] == "w-nopid"
|
||||
)
|
||||
self.assertIn("pid", detail)
|
||||
|
||||
def test_unparsable_heartbeat_is_preserved(self):
|
||||
rows = [_row(identity="w-1", pid=101, last_heartbeat_at="not-a-stamp")]
|
||||
plan = _plan(rows)
|
||||
self.assertEqual(plan["candidate_count"], 0)
|
||||
self.assertEqual(
|
||||
plan["preserved"][0]["reason_code"], retire.REASON_UNPARSABLE_HEARTBEAT
|
||||
)
|
||||
|
||||
def test_identity_shared_with_live_worker_is_preserved(self):
|
||||
"""Two rows, one live: the dead one shares session evidence, so it stays."""
|
||||
rows = [
|
||||
_row(
|
||||
identity="w-live",
|
||||
pid=101,
|
||||
session="sess-shared",
|
||||
heartbeat=NOW - timedelta(seconds=5),
|
||||
),
|
||||
_row(identity="w-dead", pid=102, session="sess-shared"),
|
||||
]
|
||||
plan = _plan(rows, probe=_selective({101}))
|
||||
self.assertEqual(plan["candidate_count"], 0)
|
||||
codes = {p["reason_code"] for p in plan["preserved"]}
|
||||
self.assertIn(retire.REASON_CONFLICTING_IDENTITY, codes)
|
||||
|
||||
def test_foreign_or_missing_repository_binding_is_preserved(self):
|
||||
rows = [
|
||||
_row(identity="w-foreign", pid=101, repository_binding="/other/repo"),
|
||||
_row(identity="w-unbound", pid=102, repository_binding=None),
|
||||
]
|
||||
plan = _plan(rows)
|
||||
self.assertEqual(plan["candidate_count"], 0)
|
||||
self.assertEqual(
|
||||
{p["reason_code"] for p in plan["preserved"]},
|
||||
{retire.REASON_FOREIGN_REPOSITORY},
|
||||
)
|
||||
|
||||
def test_active_workflow_owner_is_preserved(self):
|
||||
rows = [_row(identity="w-1", pid=101, session="sess-leased")]
|
||||
plan = _plan(rows, protected_session_ids=["sess-leased"])
|
||||
self.assertEqual(plan["candidate_count"], 0)
|
||||
self.assertEqual(
|
||||
plan["preserved"][0]["reason_code"], retire.REASON_PROTECTED_OWNER
|
||||
)
|
||||
|
||||
def test_terminal_rows_are_not_retired_again(self):
|
||||
rows = [_row(identity="w-1", pid=101, status=mwi.STATUS_RELEASED)]
|
||||
plan = _plan(rows)
|
||||
self.assertEqual(plan["candidate_count"], 0)
|
||||
self.assertEqual(
|
||||
plan["preserved"][0]["reason_code"], retire.REASON_ALREADY_TERMINAL
|
||||
)
|
||||
|
||||
def test_mixed_fleet_produces_correct_per_worker_outcomes(self):
|
||||
rows = [
|
||||
_row(identity="w-stale-1", pid=101, session="s1", generation="g1"),
|
||||
_row(identity="w-stale-2", pid=102, session="s2", generation="g2"),
|
||||
_row(
|
||||
identity="w-live",
|
||||
pid=103,
|
||||
session="s3",
|
||||
generation="g3",
|
||||
heartbeat=NOW - timedelta(seconds=5),
|
||||
),
|
||||
_row(identity="w-nopid", pid=None, session="s4", generation="g4"),
|
||||
_row(
|
||||
identity="w-foreign",
|
||||
pid=105,
|
||||
session="s5",
|
||||
generation="g5",
|
||||
repository_binding="/other",
|
||||
),
|
||||
_row(
|
||||
identity="w-terminal",
|
||||
pid=106,
|
||||
session="s6",
|
||||
generation="g6",
|
||||
status=mwi.STATUS_SUPERSEDED,
|
||||
),
|
||||
]
|
||||
plan = _plan(rows, probe=_selective({103}))
|
||||
self.assertEqual(
|
||||
sorted(plan["candidate_worker_identities"]), ["w-stale-1", "w-stale-2"]
|
||||
)
|
||||
by_identity = {
|
||||
p["worker_identity"]: p["reason_code"] for p in plan["preserved"]
|
||||
}
|
||||
self.assertEqual(by_identity["w-live"], retire.REASON_WORKER_LIVE)
|
||||
self.assertEqual(by_identity["w-nopid"], retire.REASON_INCOMPLETE_IDENTITY)
|
||||
self.assertEqual(by_identity["w-foreign"], retire.REASON_FOREIGN_REPOSITORY)
|
||||
self.assertEqual(by_identity["w-terminal"], retire.REASON_ALREADY_TERMINAL)
|
||||
self.assertEqual(plan["assessed_count"], 6)
|
||||
self.assertEqual(plan["preserved_count"], 4)
|
||||
|
||||
def test_plan_is_deterministic_for_identical_contents(self):
|
||||
rows = [_row(identity="w-1", pid=101), _row(identity="w-2", pid=102)]
|
||||
first = _plan(rows)
|
||||
second = _plan(list(reversed(rows)), now=NOW + timedelta(minutes=5))
|
||||
self.assertEqual(first["registry_fingerprint"], second["registry_fingerprint"])
|
||||
self.assertEqual(
|
||||
first["candidate_fingerprint"], second["candidate_fingerprint"]
|
||||
)
|
||||
self.assertEqual(
|
||||
sorted(first["candidate_worker_identities"]),
|
||||
sorted(second["candidate_worker_identities"]),
|
||||
)
|
||||
|
||||
|
||||
class ApplyCasTests(unittest.TestCase):
|
||||
"""AC: apply is compare-and-swap protected and revalidates every target."""
|
||||
|
||||
def setUp(self) -> None:
|
||||
self.registry = _registry()
|
||||
self.probe = _dead
|
||||
|
||||
def _rows(self):
|
||||
return self.registry.list_workers(status=None)
|
||||
|
||||
def _apply(self, plan, *, probe=None, identities=None, **kwargs):
|
||||
chosen = probe or self.probe
|
||||
return self.registry.retire_stale_workers(
|
||||
expected_registry_fingerprint=plan["registry_fingerprint"],
|
||||
expected_candidate_fingerprint=plan["candidate_fingerprint"],
|
||||
worker_identities=(
|
||||
identities
|
||||
if identities is not None
|
||||
else plan["candidate_worker_identities"]
|
||||
),
|
||||
fingerprint_fn=retire.registry_fingerprint,
|
||||
plan_fn=lambda rows: _plan(rows, probe=chosen, **kwargs),
|
||||
retired_by="synthetic-user/prgs-reconciler",
|
||||
now=NOW,
|
||||
)
|
||||
|
||||
def test_matching_token_retires_the_planned_set(self):
|
||||
_register(self.registry, _row(identity="w-1", pid=101, session="s1"))
|
||||
_register(self.registry, _row(identity="w-2", pid=102, session="s2"))
|
||||
plan = _plan(self._rows(), probe=self.probe)
|
||||
result = self._apply(plan)
|
||||
self.assertEqual(result["outcome"], "applied")
|
||||
self.assertTrue(result["mutation_performed"])
|
||||
self.assertEqual(result["retired_count"], 2)
|
||||
statuses = {r["worker_identity"]: r["status"] for r in self._rows()}
|
||||
self.assertEqual(
|
||||
statuses, {"w-1": mwi.STATUS_RETIRED, "w-2": mwi.STATUS_RETIRED}
|
||||
)
|
||||
retired_row = self.registry.get("w-1")
|
||||
self.assertEqual(retired_row["retired_at"], mwi._ts(NOW))
|
||||
self.assertEqual(retired_row["retired_by"], "synthetic-user/prgs-reconciler")
|
||||
|
||||
def test_moved_registry_token_retires_zero_workers(self):
|
||||
_register(self.registry, _row(identity="w-1", pid=101))
|
||||
plan = _plan(self._rows(), probe=self.probe)
|
||||
# An unrelated registration lands between plan and apply.
|
||||
_register(self.registry, _row(identity="w-2", pid=102, session="s2"))
|
||||
result = self._apply(plan)
|
||||
self.assertEqual(result["outcome"], retire.OUTCOME_REGISTRY_MOVED)
|
||||
self.assertEqual(result["retired_count"], 0)
|
||||
self.assertFalse(result["mutation_performed"])
|
||||
self.assertTrue(all(r["status"] == mwi.STATUS_ACTIVE for r in self._rows()))
|
||||
|
||||
def test_moved_candidate_set_retires_zero_workers(self):
|
||||
"""Registry unchanged, but the eligibility verdict is no longer the same."""
|
||||
_register(self.registry, _row(identity="w-1", pid=101))
|
||||
plan = _plan(self._rows(), probe=self.probe)
|
||||
# The row did not change; the process came back (pid probes alive), so
|
||||
# revalidation inside the transaction finds no candidates at all.
|
||||
result = self._apply(plan, probe=_alive)
|
||||
self.assertEqual(result["outcome"], retire.OUTCOME_CANDIDATES_MOVED)
|
||||
self.assertEqual(result["retired_count"], 0)
|
||||
self.assertFalse(result["mutation_performed"])
|
||||
self.assertEqual(self.registry.get("w-1")["status"], mwi.STATUS_ACTIVE)
|
||||
|
||||
def test_worker_that_becomes_live_between_plan_and_apply_is_preserved(self):
|
||||
_register(self.registry, _row(identity="w-dead", pid=101, session="s1"))
|
||||
_register(self.registry, _row(identity="w-back", pid=102, session="s2"))
|
||||
plan = _plan(self._rows(), probe=self.probe)
|
||||
self.assertEqual(len(plan["candidate_worker_identities"]), 2)
|
||||
# w-back's process is alive by the time apply runs, so the candidate
|
||||
# fingerprint moves and nothing at all is retired.
|
||||
result = self._apply(plan, probe=_selective({102}))
|
||||
self.assertEqual(result["outcome"], retire.OUTCOME_CANDIDATES_MOVED)
|
||||
self.assertEqual(result["retired_count"], 0)
|
||||
self.assertEqual(self.registry.get("w-back")["status"], mwi.STATUS_ACTIVE)
|
||||
self.assertEqual(self.registry.get("w-dead")["status"], mwi.STATUS_ACTIVE)
|
||||
|
||||
def test_worker_that_becomes_ambiguous_between_plan_and_apply_is_preserved(self):
|
||||
_register(self.registry, _row(identity="w-1", pid=101, session="s1"))
|
||||
plan = _plan(self._rows(), probe=self.probe)
|
||||
result = self._apply(plan, probe=lambda _pid: None)
|
||||
self.assertEqual(result["outcome"], retire.OUTCOME_CANDIDATES_MOVED)
|
||||
self.assertEqual(result["retired_count"], 0)
|
||||
self.assertEqual(self.registry.get("w-1")["status"], mwi.STATUS_ACTIVE)
|
||||
|
||||
def test_apply_revalidates_and_will_not_narrow_the_approved_set(self):
|
||||
"""A protection appearing after plan moves the CAS, so nothing is retired."""
|
||||
_register(self.registry, _row(identity="w-1", pid=101, session="s1"))
|
||||
_register(
|
||||
self.registry, _row(identity="w-protected", pid=102, session="sess-leased")
|
||||
)
|
||||
unprotected_plan = _plan(self._rows(), probe=self.probe)
|
||||
self.assertEqual(unprotected_plan["candidate_count"], 2)
|
||||
result = self._apply(
|
||||
unprotected_plan, protected_session_ids=["sess-leased"]
|
||||
)
|
||||
self.assertEqual(result["outcome"], retire.OUTCOME_CANDIDATES_MOVED)
|
||||
self.assertEqual(result["retired_count"], 0)
|
||||
self.assertEqual(
|
||||
self.registry.get("w-protected")["status"], mwi.STATUS_ACTIVE
|
||||
)
|
||||
|
||||
def test_target_outside_the_current_plan_is_preserved(self):
|
||||
_register(
|
||||
self.registry,
|
||||
_row(identity="w-1", pid=101, session="s1", generation="g1"),
|
||||
)
|
||||
_register(
|
||||
self.registry,
|
||||
_row(
|
||||
identity="w-live",
|
||||
pid=102,
|
||||
session="s2",
|
||||
generation="g2",
|
||||
heartbeat=NOW - timedelta(seconds=5),
|
||||
),
|
||||
)
|
||||
probe = _selective({102})
|
||||
plan = _plan(self._rows(), probe=probe)
|
||||
self.assertEqual(plan["candidate_worker_identities"], ["w-1"])
|
||||
# A caller that appends a live identity to the approved list gets it
|
||||
# preserved with the live reason code; only the real candidate retires.
|
||||
result = self._apply(plan, probe=probe, identities=["w-1", "w-live"])
|
||||
self.assertEqual(result["outcome"], "applied")
|
||||
self.assertEqual(result["retired_count"], 1)
|
||||
self.assertEqual(result["retired"][0]["worker_identity"], "w-1")
|
||||
preserved = {
|
||||
p["worker_identity"]: p["reason_code"] for p in result["preserved"]
|
||||
}
|
||||
self.assertEqual(preserved["w-live"], retire.REASON_WORKER_LIVE)
|
||||
self.assertEqual(self.registry.get("w-live")["status"], mwi.STATUS_ACTIVE)
|
||||
|
||||
def test_missing_registration_is_reported_not_invented(self):
|
||||
_register(self.registry, _row(identity="w-1", pid=101))
|
||||
plan = _plan(self._rows(), probe=self.probe)
|
||||
result = self._apply(plan, identities=["w-1", "w-ghost"])
|
||||
self.assertEqual(result["retired_count"], 1)
|
||||
preserved = {
|
||||
p["worker_identity"]: p["reason_code"] for p in result["preserved"]
|
||||
}
|
||||
self.assertEqual(preserved["w-ghost"], "registration_missing")
|
||||
|
||||
def test_reapplying_a_completed_plan_is_a_safe_no_op(self):
|
||||
_register(self.registry, _row(identity="w-1", pid=101))
|
||||
plan = _plan(self._rows(), probe=self.probe)
|
||||
first = self._apply(plan)
|
||||
self.assertEqual(first["retired_count"], 1)
|
||||
second = self._apply(plan)
|
||||
self.assertEqual(second["outcome"], retire.OUTCOME_ALREADY_RETIRED)
|
||||
self.assertTrue(second["idempotent"])
|
||||
self.assertEqual(second["retired_count"], 0)
|
||||
self.assertFalse(second["mutation_performed"])
|
||||
self.assertEqual(self.registry.get("w-1")["status"], mwi.STATUS_RETIRED)
|
||||
|
||||
def test_empty_target_list_reports_nothing_requested(self):
|
||||
_register(self.registry, _row(identity="w-1", pid=101))
|
||||
plan = _plan(self._rows(), probe=self.probe)
|
||||
result = self._apply(plan, identities=[])
|
||||
self.assertEqual(result["outcome"], retire.OUTCOME_NOTHING_REQUESTED)
|
||||
self.assertFalse(result["mutation_performed"])
|
||||
self.assertEqual(self.registry.get("w-1")["status"], mwi.STATUS_ACTIVE)
|
||||
|
||||
def test_transaction_failure_cannot_report_partial_success(self):
|
||||
_register(self.registry, _row(identity="w-1", pid=101, session="s1"))
|
||||
_register(self.registry, _row(identity="w-2", pid=102, session="s2"))
|
||||
plan = _plan(self._rows(), probe=self.probe)
|
||||
calls = {"n": 0}
|
||||
|
||||
def exploding_plan(rows):
|
||||
calls["n"] += 1
|
||||
raise RuntimeError("synthetic revalidation failure")
|
||||
|
||||
result = self.registry.retire_stale_workers(
|
||||
expected_registry_fingerprint=plan["registry_fingerprint"],
|
||||
expected_candidate_fingerprint=plan["candidate_fingerprint"],
|
||||
worker_identities=plan["candidate_worker_identities"],
|
||||
fingerprint_fn=retire.registry_fingerprint,
|
||||
plan_fn=exploding_plan,
|
||||
retired_by="synthetic-user/prgs-reconciler",
|
||||
now=NOW,
|
||||
)
|
||||
self.assertEqual(calls["n"], 1)
|
||||
self.assertFalse(result["success"])
|
||||
self.assertEqual(result["outcome"], "transaction_failed")
|
||||
self.assertEqual(result["retired_count"], 0)
|
||||
self.assertFalse(result["mutation_performed"])
|
||||
self.assertTrue(
|
||||
all(r["status"] == mwi.STATUS_ACTIVE for r in self._rows()),
|
||||
"a failed transaction must roll back every retirement",
|
||||
)
|
||||
|
||||
def test_structured_result_fields_are_accurate(self):
|
||||
_register(self.registry, _row(identity="w-1", pid=101))
|
||||
plan = _plan(self._rows(), probe=self.probe)
|
||||
blocked = self._apply(plan, probe=_alive)
|
||||
self.assertFalse(blocked["mutation_performed"])
|
||||
self.assertEqual(blocked["retired"], [])
|
||||
self.assertEqual(blocked["requested_count"], 1)
|
||||
self.assertEqual(
|
||||
blocked["expected_registry_fingerprint"], plan["registry_fingerprint"]
|
||||
)
|
||||
applied = self._apply(plan)
|
||||
self.assertTrue(applied["mutation_performed"])
|
||||
self.assertEqual(applied["acting_identity"], "synthetic-user/prgs-reconciler")
|
||||
self.assertEqual(applied["retired"][0]["reason_code"], retire.REASON_ELIGIBLE)
|
||||
self.assertIn("evidence", applied["retired"][0])
|
||||
|
||||
|
||||
class PostRetirementFleetCompatibilityTests(unittest.TestCase):
|
||||
"""AC: retired rows leave the stale count and never fake fleet safety."""
|
||||
|
||||
def test_retired_rows_are_historical_not_stale(self):
|
||||
registry = _registry()
|
||||
_register(
|
||||
registry, _row(identity="w-1", pid=101, session="s1", generation="g1")
|
||||
)
|
||||
_register(
|
||||
registry,
|
||||
_row(
|
||||
identity="w-live",
|
||||
pid=102,
|
||||
session="s2",
|
||||
generation="g2",
|
||||
instance="legacy-pid-27833",
|
||||
instance_id_provenance=fleet.INSTANCE_ID_PROVENANCE_LEGACY,
|
||||
heartbeat=NOW - timedelta(seconds=5),
|
||||
),
|
||||
)
|
||||
probe = _selective({102})
|
||||
before = fleet.snapshot_instance_fleet(
|
||||
registry.list_workers(status=None),
|
||||
now=NOW,
|
||||
pid_alive_probe=probe,
|
||||
canonical_repository=REPO,
|
||||
)
|
||||
self.assertEqual(before["stale_worker_count"], 1)
|
||||
|
||||
plan = _plan(registry.list_workers(status=None), probe=probe)
|
||||
result = registry.retire_stale_workers(
|
||||
expected_registry_fingerprint=plan["registry_fingerprint"],
|
||||
expected_candidate_fingerprint=plan["candidate_fingerprint"],
|
||||
worker_identities=plan["candidate_worker_identities"],
|
||||
fingerprint_fn=retire.registry_fingerprint,
|
||||
plan_fn=lambda rows: _plan(rows, probe=probe),
|
||||
retired_by="synthetic-user/prgs-reconciler",
|
||||
now=NOW,
|
||||
)
|
||||
self.assertEqual(result["retired_count"], 1)
|
||||
|
||||
after = fleet.snapshot_instance_fleet(
|
||||
registry.list_workers(status=None),
|
||||
now=NOW,
|
||||
pid_alive_probe=probe,
|
||||
canonical_repository=REPO,
|
||||
)
|
||||
self.assertEqual(after["stale_worker_count"], 0)
|
||||
self.assertEqual(after["live_worker_count"], 1)
|
||||
self.assertEqual(after["historical_worker_count"], 1)
|
||||
# The surviving live worker still carries a legacy instance identity, so
|
||||
# the fleet must not be declared safe.
|
||||
self.assertFalse(after["live_fleet_safe"])
|
||||
self.assertIn(
|
||||
fleet.CLASS_LEGACY_INCOMPLETE,
|
||||
{f["classification"] for f in after["active_blockers"]},
|
||||
)
|
||||
|
||||
def test_existing_snapshot_shape_is_unchanged(self):
|
||||
rows = [_row(identity="w-1", pid=101)]
|
||||
snapshot = fleet.snapshot_instance_fleet(
|
||||
rows, now=NOW, pid_alive_probe=_dead, canonical_repository=REPO
|
||||
)
|
||||
for key in (
|
||||
"consistency_token",
|
||||
"registry_revision",
|
||||
"snapshot_at",
|
||||
"live_workers",
|
||||
"stale_workers",
|
||||
"historical_workers",
|
||||
"findings",
|
||||
"live_fleet_safe",
|
||||
):
|
||||
self.assertIn(key, snapshot)
|
||||
self.assertTrue(snapshot["registry_revision"].startswith("fleetrev-"))
|
||||
self.assertTrue(snapshot["read_only"])
|
||||
self.assertFalse(snapshot.get("mutation_performed", False))
|
||||
|
||||
|
||||
class CapabilityExposureTests(unittest.TestCase):
|
||||
"""AC: only controller/reconciler reach the surface; author policy unchanged."""
|
||||
|
||||
def test_capability_map_declares_controller_role(self):
|
||||
import task_capability_map as tcm
|
||||
|
||||
for task in (
|
||||
"plan_stale_worker_retirement",
|
||||
"gitea_plan_stale_worker_retirement",
|
||||
"apply_stale_worker_retirement",
|
||||
"gitea_apply_stale_worker_retirement",
|
||||
):
|
||||
with self.subTest(task=task):
|
||||
self.assertEqual(tcm.required_permission(task), "gitea.read")
|
||||
self.assertEqual(tcm.required_role(task), "controller")
|
||||
|
||||
def test_ordinary_author_capability_resolution_is_unaffected(self):
|
||||
import task_capability_map as tcm
|
||||
|
||||
self.assertEqual(tcm.required_permission("work_issue"), "gitea.pr.create")
|
||||
self.assertEqual(tcm.required_role("work_issue"), "author")
|
||||
self.assertEqual(tcm.required_permission("create_pr"), "gitea.pr.create")
|
||||
self.assertEqual(tcm.required_role("create_pr"), "author")
|
||||
self.assertEqual(tcm.required_permission("commit_files"), "gitea.repo.commit")
|
||||
self.assertEqual(tcm.required_permission("push_branch"), "gitea.branch.push")
|
||||
self.assertEqual(
|
||||
tcm.required_permission("snapshot_instance_fleet"), "gitea.read"
|
||||
)
|
||||
|
||||
def test_no_new_gitea_write_permission_is_introduced(self):
|
||||
import task_capability_map as tcm
|
||||
|
||||
for task in ("plan_stale_worker_retirement", "apply_stale_worker_retirement"):
|
||||
with self.subTest(task=task):
|
||||
self.assertNotIn(
|
||||
tcm.required_permission(task),
|
||||
{
|
||||
"gitea.pr.approve",
|
||||
"gitea.pr.merge",
|
||||
"gitea.pr.review",
|
||||
"gitea.branch.push",
|
||||
"gitea.repo.commit",
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
class SurfaceRegistrationTests(unittest.TestCase):
|
||||
"""AC: the dry-run and apply modes are exposed as sanctioned native tools."""
|
||||
|
||||
def test_tools_are_registered_and_role_restricted(self):
|
||||
import inspect
|
||||
|
||||
import gitea_mcp_server as server
|
||||
|
||||
for name in (
|
||||
"gitea_plan_stale_worker_retirement",
|
||||
"gitea_apply_stale_worker_retirement",
|
||||
):
|
||||
with self.subTest(tool=name):
|
||||
tool = getattr(server, name)
|
||||
self.assertTrue(callable(tool))
|
||||
source = inspect.getsource(tool)
|
||||
# Both tools route their role check through the shared guard.
|
||||
self.assertIn(f'_retirement_role_block("{name}")', source)
|
||||
|
||||
guard = inspect.getsource(server._retirement_role_block)
|
||||
self.assertIn('{"controller", "reconciler"}', guard)
|
||||
self.assertIn("required_roles", guard)
|
||||
|
||||
apply_source = inspect.getsource(server.gitea_apply_stale_worker_retirement)
|
||||
self.assertIn("registry_fingerprint", apply_source)
|
||||
self.assertIn("candidate_fingerprint", apply_source)
|
||||
self.assertIn("_retirement_runtime_block", apply_source)
|
||||
self.assertIn("classify_cohort", apply_source)
|
||||
# Revalidation inside the transaction re-reads lease protection rather
|
||||
# than reusing the pre-transaction snapshot.
|
||||
self.assertIn("_retirement_revalidation_plan", apply_source)
|
||||
revalidation = inspect.getsource(server._retirement_revalidation_plan)
|
||||
self.assertIn("_retirement_protected_owners", revalidation)
|
||||
self.assertIn("raise RuntimeError", revalidation)
|
||||
|
||||
def test_plan_tool_declares_itself_read_only(self):
|
||||
import inspect
|
||||
|
||||
import gitea_mcp_server as server
|
||||
|
||||
source = inspect.getsource(server.gitea_plan_stale_worker_retirement)
|
||||
self.assertIn("read_only", source)
|
||||
self.assertNotIn("retire_stale_workers", source)
|
||||
|
||||
def test_docs_exist(self):
|
||||
root = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
path = os.path.join(root, "docs", "stale-worker-retirement.md")
|
||||
self.assertTrue(os.path.isfile(path))
|
||||
with open(path, encoding="utf-8") as handle:
|
||||
text = handle.read()
|
||||
for token in (
|
||||
"registry_fingerprint",
|
||||
"candidate_fingerprint",
|
||||
"gitea_plan_stale_worker_retirement",
|
||||
"gitea_apply_stale_worker_retirement",
|
||||
"eligible_stale_orphan",
|
||||
"registry_revision_moved",
|
||||
"does not repair untrusted live identity",
|
||||
):
|
||||
with self.subTest(token=token):
|
||||
self.assertIn(token, text)
|
||||
|
||||
|
||||
if __name__ == "__main__": # pragma: no cover
|
||||
unittest.main()
|
||||
@@ -245,10 +245,15 @@ class TestNamespaceWorkspaceIntegration(unittest.TestCase):
|
||||
def test_pr487_style_merge_binds_clean_merger_workspace(
|
||||
self, _exists, _isdir, mock_run
|
||||
):
|
||||
mock_run.return_value = MagicMock(returncode=0, stdout=f"{CONTROL_ROOT}/.git\n")
|
||||
def mock_git(cmd, *args, **kwargs):
|
||||
if "rev-parse" in cmd:
|
||||
return MagicMock(returncode=0, stdout=f"{CONTROL_ROOT}/.git\n")
|
||||
return MagicMock(returncode=0, stdout=f"worktree {CONTROL_ROOT}\nworktree {MERGER_CLEAN}\n")
|
||||
mock_run.side_effect = mock_git
|
||||
os.environ[nwb.AUTHOR_WORKTREE_ENV] = AUTHOR_DIRTY
|
||||
os.environ[nwb.MERGER_WORKTREE_ENV] = MERGER_CLEAN
|
||||
srv._preflight_resolved_role = "reviewer"
|
||||
with mock.patch.object(srv, "PROJECT_ROOT", MCP_PROCESS_ROOT):
|
||||
with mock.patch("gitea_mcp_server.get_profile", return_value=self._merger_profile()):
|
||||
resolved = srv._verify_role_mutation_workspace("prgs")
|
||||
self.assertEqual(resolved, os.path.realpath(MCP_PROCESS_ROOT))
|
||||
self.assertEqual(resolved, os.path.realpath(MERGER_CLEAN))
|
||||
@@ -153,6 +153,12 @@ EXPECTED_ROLE_EXCLUSIVE_TASKS = frozenset(
|
||||
"delete_branch",
|
||||
"cleanup_merged_pr_branch",
|
||||
"reconciliation_cleanup",
|
||||
# #970 review 644 B2: retiring a missing worktree binding is a
|
||||
# control-plane cleanup mutation, so it carries the same reconciler-only
|
||||
# authority as every other reconciliation cleanup. Permission alone must
|
||||
# not authorize it.
|
||||
"reconcile_missing_worktree_bindings",
|
||||
"gitea_reconcile_missing_worktree_bindings",
|
||||
"work_issue",
|
||||
"work-issue",
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user