Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e8bae606cb | ||
|
|
956fa15fe3 | ||
|
|
fa510dd28d | ||
|
|
1dd30ecb15 | ||
|
|
8eada1fbe4 | ||
|
|
58bd852188 | ||
|
|
ca5f078d8a | ||
|
|
126d76ad28 | ||
|
|
e3fa3b263d | ||
|
|
4533e86bcd | ||
|
|
17ba1ff035 | ||
|
|
0104a76eea | ||
|
|
a143cd065b | ||
|
|
2fb835a1aa | ||
|
|
c162608175 | ||
|
|
9b80e75ca3 | ||
|
|
b3de9c941c | ||
|
|
aad5c8b423 | ||
|
|
b4c9f55890 | ||
|
|
1aa351718a | ||
|
|
55d66c57e4 | ||
|
|
cdf0daefa9 | ||
|
|
82d71b7702 | ||
|
|
47bfae07d2 | ||
|
|
35ed8a2fcb | ||
|
|
b1fcf15937 | ||
|
|
ed9414ebda | ||
|
|
5fea326988 | ||
|
|
dc5d2c8caa | ||
|
|
c30b381eb2 | ||
|
|
79334d4840 | ||
|
|
f49e781102 | ||
|
|
aab54d4825 | ||
|
|
a09c485fc0 | ||
|
|
8c1d22a658 | ||
|
|
6a56260768 | ||
|
|
97bc190fc2 | ||
|
|
17dd05ec9d | ||
|
|
08d9cf4cbd | ||
|
|
7bf4f12584 | ||
|
|
2066623986 | ||
|
|
dc0bff764d | ||
|
|
dfe8d7c28d | ||
|
|
fad44669d9 | ||
|
|
29ad93d145 | ||
|
|
f2dbf30e81 | ||
|
|
2b4e43042a | ||
|
|
0f9390aab4 | ||
|
|
d7ad2838ec | ||
|
|
c6d68dbc7b | ||
|
|
c83a10d7c2 | ||
|
|
ca22c326a4 | ||
|
|
3bbe6df6c7 | ||
|
|
71031c812e | ||
|
|
e43ddd3cbe | ||
|
|
a64ba08e27 | ||
|
|
26f54851d1 | ||
|
|
f02a2dc030 | ||
|
|
bb8c3a537b | ||
|
|
461e1dac78 | ||
|
|
6010f4295b | ||
|
|
9b8e315b49 | ||
|
|
9a01543477 | ||
|
|
59aab06fe1 | ||
|
|
4f06d30e07 |
+343
-70
@@ -23,6 +23,7 @@ import shutil
|
|||||||
import subprocess
|
import subprocess
|
||||||
from typing import Any, Mapping
|
from typing import Any, Mapping
|
||||||
|
|
||||||
|
import author_lock_contract
|
||||||
import author_mutation_worktree
|
import author_mutation_worktree
|
||||||
import control_plane_db
|
import control_plane_db
|
||||||
import issue_lock_store
|
import issue_lock_store
|
||||||
@@ -196,6 +197,58 @@ def _verify_assignment_and_lease_ids(
|
|||||||
# Some lease rows may not yet have an assignment join; still require
|
# Some lease rows may not yet have an assignment join; still require
|
||||||
# the lease itself to exist and bind to the claimed session/issue.
|
# the lease itself to exist and bind to the claimed session/issue.
|
||||||
pass
|
pass
|
||||||
|
# #943 review 622 B2: a lease that is no longer live confers no ownership.
|
||||||
|
# Existence alone previously satisfied this gate, so a released or expired
|
||||||
|
# lease could still authorize a bootstrap for a claim its session had given
|
||||||
|
# up. Checked before the session comparison so the reason names the real
|
||||||
|
# problem rather than reporting a mismatch.
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
|
||||||
|
lease_status = str(lease.get("status") or "").strip().lower()
|
||||||
|
if lease_status and lease_status != "active":
|
||||||
|
return {
|
||||||
|
"success": False,
|
||||||
|
"reason_code": "lease_not_live",
|
||||||
|
"message": (
|
||||||
|
f"lease_id '{lid}' is '{lease_status}', not active; a lease that "
|
||||||
|
"is not live confers no ownership (fail closed)."
|
||||||
|
),
|
||||||
|
"exact_next_action": (
|
||||||
|
"Re-allocate the work item and pass the live assignment/lease pair."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
expires_raw = str(lease.get("expires_at") or "").strip()
|
||||||
|
if expires_raw:
|
||||||
|
try:
|
||||||
|
expires_at = datetime.fromisoformat(expires_raw.replace("Z", "+00:00"))
|
||||||
|
except ValueError:
|
||||||
|
return {
|
||||||
|
"success": False,
|
||||||
|
"reason_code": "lease_not_live",
|
||||||
|
"message": (
|
||||||
|
f"lease_id '{lid}' records an unparseable expiry "
|
||||||
|
f"'{expires_raw}' (fail closed)."
|
||||||
|
),
|
||||||
|
"exact_next_action": (
|
||||||
|
"Re-allocate the work item and pass the live "
|
||||||
|
"assignment/lease pair."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
if expires_at.tzinfo is None:
|
||||||
|
expires_at = expires_at.replace(tzinfo=timezone.utc)
|
||||||
|
if expires_at <= datetime.now(timezone.utc):
|
||||||
|
return {
|
||||||
|
"success": False,
|
||||||
|
"reason_code": "lease_not_live",
|
||||||
|
"message": (
|
||||||
|
f"lease_id '{lid}' expired at {expires_raw}; an expired lease "
|
||||||
|
"confers no ownership (fail closed)."
|
||||||
|
),
|
||||||
|
"exact_next_action": (
|
||||||
|
"Reclaim or re-allocate the lease, then retry with the live pair."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
lease_session = str(lease.get("session_id") or "").strip()
|
lease_session = str(lease.get("session_id") or "").strip()
|
||||||
if lease_session and lease_session != owner_session:
|
if lease_session and lease_session != owner_session:
|
||||||
return {
|
return {
|
||||||
@@ -226,6 +279,36 @@ def _verify_assignment_and_lease_ids(
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _branch_exists(canonical_repo_root: str, branch_name: str) -> bool:
|
||||||
|
"""Whether *branch_name* still resolves in the canonical checkout (#953 F2).
|
||||||
|
|
||||||
|
Used after compensating recovery to observe what survived rather than infer
|
||||||
|
it from the journal. Fails closed to ``True``: an unobservable branch is
|
||||||
|
reported as present, so the recommendation stays conservative rather than
|
||||||
|
telling an author to re-bootstrap over something that may still be there.
|
||||||
|
"""
|
||||||
|
if not branch_name:
|
||||||
|
return False
|
||||||
|
try:
|
||||||
|
res = subprocess.run(
|
||||||
|
[
|
||||||
|
"git",
|
||||||
|
"-C",
|
||||||
|
canonical_repo_root,
|
||||||
|
"rev-parse",
|
||||||
|
"--verify",
|
||||||
|
"--quiet",
|
||||||
|
f"refs/heads/{branch_name}",
|
||||||
|
],
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
check=False,
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
return True
|
||||||
|
return res.returncode == 0
|
||||||
|
|
||||||
|
|
||||||
def run_compensating_recovery(
|
def run_compensating_recovery(
|
||||||
journal: dict[str, Any],
|
journal: dict[str, Any],
|
||||||
canonical_repo_root: str,
|
canonical_repo_root: str,
|
||||||
@@ -262,10 +345,21 @@ def run_compensating_recovery(
|
|||||||
issue_number=issue_num,
|
issue_number=issue_num,
|
||||||
session=session_id,
|
session=session_id,
|
||||||
lock_dir=journal_dir,
|
lock_dir=journal_dir,
|
||||||
|
remote=journal.get("remote"),
|
||||||
|
# The same defaults the lock was written under, so the
|
||||||
|
# rollback targets the exact file bind_session_lock keyed.
|
||||||
|
org=journal.get("org") or "Scaled-Tech-Consulting",
|
||||||
|
repo=journal.get("repo") or "Gitea-Tools",
|
||||||
)
|
)
|
||||||
rolled_back.append(f"lock:issue-{issue_num}")
|
rolled_back.append(f"lock:issue-{issue_num}")
|
||||||
except Exception:
|
except Exception as exc:
|
||||||
pass
|
# #953 F2: a swallowed failure here is what made the rollback
|
||||||
|
# report success while leaving an unrecoverable lock behind.
|
||||||
|
# Record it so the post-compensation classification can see the
|
||||||
|
# lock survived and recommend accordingly.
|
||||||
|
rolled_back.append(
|
||||||
|
f"lock_release_failed:issue-{issue_num}:{type(exc).__name__}"
|
||||||
|
)
|
||||||
artifacts["lock_created"] = False
|
artifacts["lock_created"] = False
|
||||||
|
|
||||||
worktree_created = (
|
worktree_created = (
|
||||||
@@ -386,6 +480,68 @@ def run_compensating_recovery(
|
|||||||
return recovery_info
|
return recovery_info
|
||||||
|
|
||||||
|
|
||||||
|
def _normalize_sha(value: str | None) -> str | None:
|
||||||
|
"""Normalize a Git object id for comparison, or ``None`` when unknown."""
|
||||||
|
normalized = (value or "").strip().lower()
|
||||||
|
return normalized or None
|
||||||
|
|
||||||
|
|
||||||
|
def _author_bootstrap_assessment(
|
||||||
|
*,
|
||||||
|
not_applicable: bool,
|
||||||
|
allowed: bool,
|
||||||
|
block: bool,
|
||||||
|
reasons: list[str],
|
||||||
|
workspace: str,
|
||||||
|
root: str,
|
||||||
|
branch: str | None,
|
||||||
|
dirty: list[str],
|
||||||
|
under_branches: bool,
|
||||||
|
bootstrap_path: str | None = None,
|
||||||
|
local_head_sha: str | None = None,
|
||||||
|
remote_master_sha: str | None = None,
|
||||||
|
exact_next_action: str | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Structured author-bootstrap assessment consumable by bootstrap_permits (#892).
|
||||||
|
|
||||||
|
Field shape mirrors :func:`create_issue_bootstrap._result` so the shared
|
||||||
|
``bootstrap_permits_control_checkout`` predicate can prove control-checkout
|
||||||
|
eligibility for ``gitea_bootstrap_author_issue_worktree`` the same way it
|
||||||
|
does for ``create_issue``. Allowed control assessments must use empty
|
||||||
|
``reasons`` — narrative belongs in other fields, not the refusal list.
|
||||||
|
"""
|
||||||
|
local_tip = _normalize_sha(local_head_sha)
|
||||||
|
remote_tip = _normalize_sha(remote_master_sha)
|
||||||
|
base_tips_verified = bool(local_tip and remote_tip and local_tip == remote_tip)
|
||||||
|
return {
|
||||||
|
"not_applicable": not_applicable,
|
||||||
|
"allowed": allowed,
|
||||||
|
"block": block,
|
||||||
|
"proven": bool(allowed and not block and not not_applicable),
|
||||||
|
"reasons": list(reasons),
|
||||||
|
"workspace_path": workspace,
|
||||||
|
"canonical_repo_root": root,
|
||||||
|
"current_branch": branch,
|
||||||
|
"dirty_files": list(dirty),
|
||||||
|
"under_branches": under_branches,
|
||||||
|
"exact_next_action": exact_next_action,
|
||||||
|
"bootstrap_path": bootstrap_path,
|
||||||
|
"task_scope": "author_issue_bootstrap",
|
||||||
|
"local_head_sha": local_tip,
|
||||||
|
"remote_master_sha": remote_tip,
|
||||||
|
"base_tips_verified": base_tips_verified,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
EXACT_NEXT_ACTION_AUTHOR_BOOTSTRAP = (
|
||||||
|
"Restore the canonical control checkout to a clean accepted base branch "
|
||||||
|
"(master/main/dev) that matches live master, with no tracked local edits. "
|
||||||
|
"Re-resolve bootstrap_author_issue_worktree, then re-run "
|
||||||
|
"gitea_bootstrap_author_issue_worktree from that clean control checkout. "
|
||||||
|
"Do not use shell git worktree add as the primary path once bootstrap is healthy."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def assess_author_issue_bootstrap(
|
def assess_author_issue_bootstrap(
|
||||||
*,
|
*,
|
||||||
workspace_path: str,
|
workspace_path: str,
|
||||||
@@ -397,7 +553,13 @@ def assess_author_issue_bootstrap(
|
|||||||
remote_master_sha_error: str | None = None,
|
remote_master_sha_error: str | None = None,
|
||||||
task: str | None = None,
|
task: str | None = None,
|
||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""Assess whether author issue worktree bootstrap may proceed from control or worktree root."""
|
"""Assess whether author issue worktree bootstrap may proceed from control or worktree root.
|
||||||
|
|
||||||
|
#892: control-checkout successes emit the full field set required by
|
||||||
|
``create_issue_bootstrap.bootstrap_permits_control_checkout`` (empty reasons,
|
||||||
|
task_scope, base tip proof, binding paths) so the #274/#604 guards can
|
||||||
|
waive control-checkout for this one sanctioned bootstrap task.
|
||||||
|
"""
|
||||||
root = os.path.realpath(canonical_repo_root or "")
|
root = os.path.realpath(canonical_repo_root or "")
|
||||||
workspace = os.path.realpath(workspace_path or root or ".")
|
workspace = os.path.realpath(workspace_path or root or ".")
|
||||||
branch = (current_branch or "").strip()
|
branch = (current_branch or "").strip()
|
||||||
@@ -407,34 +569,50 @@ def assess_author_issue_bootstrap(
|
|||||||
if root
|
if root
|
||||||
else False
|
else False
|
||||||
)
|
)
|
||||||
|
local_tip = _normalize_sha(head_sha)
|
||||||
|
remote_tip = _normalize_sha(remote_master_sha)
|
||||||
|
|
||||||
if not is_author_issue_bootstrap_task(task):
|
if not is_author_issue_bootstrap_task(task):
|
||||||
return {
|
return _author_bootstrap_assessment(
|
||||||
"not_applicable": True,
|
not_applicable=True,
|
||||||
"allowed": False,
|
allowed=False,
|
||||||
"block": False,
|
block=False,
|
||||||
"proven": False,
|
reasons=["task is not author_issue_bootstrap"],
|
||||||
"reasons": ["task is not author_issue_bootstrap"],
|
workspace=workspace,
|
||||||
}
|
root=root,
|
||||||
|
branch=branch or None,
|
||||||
|
dirty=dirty,
|
||||||
|
under_branches=under_branches,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Already under branches/: ordinary #274 path applies; not a control waiver.
|
||||||
if under_branches:
|
if under_branches:
|
||||||
return {
|
return _author_bootstrap_assessment(
|
||||||
"not_applicable": False,
|
not_applicable=True,
|
||||||
"allowed": True,
|
allowed=False,
|
||||||
"block": False,
|
block=False,
|
||||||
"proven": True,
|
reasons=["workspace is under branches/; ordinary #274 path applies"],
|
||||||
"bootstrap_path": "existing_branches_worktree",
|
workspace=workspace,
|
||||||
"reasons": [
|
root=root,
|
||||||
"workspace is already a registered worktree under branches/"
|
branch=branch or None,
|
||||||
],
|
dirty=dirty,
|
||||||
}
|
under_branches=True,
|
||||||
|
bootstrap_path="existing_branches_worktree",
|
||||||
|
local_head_sha=local_tip,
|
||||||
|
remote_master_sha=remote_tip,
|
||||||
|
)
|
||||||
|
|
||||||
reasons: list[str] = []
|
reasons: list[str] = []
|
||||||
if workspace != root:
|
if not root or workspace != root:
|
||||||
reasons.append(
|
reasons.append(
|
||||||
"bootstrap requires workspace to be canonical control checkout or branches/ worktree"
|
"bootstrap requires workspace to be canonical control checkout or branches/ worktree"
|
||||||
)
|
)
|
||||||
if branch not in author_mutation_worktree.BASE_BRANCHES:
|
if not branch:
|
||||||
|
reasons.append(
|
||||||
|
"control checkout is detached HEAD; expected an accepted base branch "
|
||||||
|
f"({', '.join(sorted(author_mutation_worktree.BASE_BRANCHES))})"
|
||||||
|
)
|
||||||
|
elif branch not in author_mutation_worktree.BASE_BRANCHES:
|
||||||
reasons.append(
|
reasons.append(
|
||||||
f"control checkout branch '{branch}' is not an accepted base branch "
|
f"control checkout branch '{branch}' is not an accepted base branch "
|
||||||
f"({', '.join(sorted(author_mutation_worktree.BASE_BRANCHES))})"
|
f"({', '.join(sorted(author_mutation_worktree.BASE_BRANCHES))})"
|
||||||
@@ -444,37 +622,64 @@ def assess_author_issue_bootstrap(
|
|||||||
f"control checkout has tracked local edits: {', '.join(dirty[:5])}"
|
f"control checkout has tracked local edits: {', '.join(dirty[:5])}"
|
||||||
)
|
)
|
||||||
|
|
||||||
if remote_master_sha_error:
|
# Fail closed on missing tip proof (same bar as create_issue bootstrap #757).
|
||||||
|
if not local_tip:
|
||||||
reasons.append(
|
reasons.append(
|
||||||
f"could not verify live master tip: {remote_master_sha_error}"
|
"control checkout HEAD SHA is unknown; base equivalence to live "
|
||||||
|
"master cannot be proven (fail closed)"
|
||||||
|
)
|
||||||
|
resolver_error = (remote_master_sha_error or "").strip() or None
|
||||||
|
if resolver_error:
|
||||||
|
reasons.append(
|
||||||
|
f"live master tip could not be resolved ({resolver_error}); "
|
||||||
|
"base equivalence cannot be proven (fail closed)"
|
||||||
|
)
|
||||||
|
elif not remote_tip:
|
||||||
|
reasons.append(
|
||||||
|
"live master tip is unknown; base equivalence cannot be proven "
|
||||||
|
"(fail closed)"
|
||||||
|
)
|
||||||
|
elif local_tip and remote_tip and local_tip != remote_tip:
|
||||||
|
reasons.append(
|
||||||
|
f"control checkout HEAD ({local_tip[:12]}) != live master tip "
|
||||||
|
f"({remote_tip[:12]})"
|
||||||
)
|
)
|
||||||
elif remote_master_sha and head_sha:
|
|
||||||
h = head_sha.strip().lower()
|
|
||||||
rm = remote_master_sha.strip().lower()
|
|
||||||
if h != rm:
|
|
||||||
reasons.append(
|
|
||||||
f"control checkout HEAD ({h[:12]}) != live master tip ({rm[:12]})"
|
|
||||||
)
|
|
||||||
|
|
||||||
if reasons:
|
if reasons:
|
||||||
return {
|
return _author_bootstrap_assessment(
|
||||||
"not_applicable": False,
|
not_applicable=False,
|
||||||
"allowed": False,
|
allowed=False,
|
||||||
"block": True,
|
block=True,
|
||||||
"proven": False,
|
reasons=reasons,
|
||||||
"reasons": reasons,
|
workspace=workspace,
|
||||||
}
|
root=root,
|
||||||
|
branch=branch or None,
|
||||||
|
dirty=dirty,
|
||||||
|
under_branches=False,
|
||||||
|
local_head_sha=local_tip,
|
||||||
|
remote_master_sha=remote_tip,
|
||||||
|
exact_next_action=EXACT_NEXT_ACTION_AUTHOR_BOOTSTRAP,
|
||||||
|
)
|
||||||
|
|
||||||
return {
|
# Allowed: empty reasons so bootstrap_permits_control_checkout can pass.
|
||||||
"not_applicable": False,
|
return _author_bootstrap_assessment(
|
||||||
"allowed": True,
|
not_applicable=False,
|
||||||
"block": False,
|
allowed=True,
|
||||||
"proven": True,
|
block=False,
|
||||||
"bootstrap_path": "clean_canonical_control_checkout",
|
reasons=[],
|
||||||
"reasons": [
|
workspace=workspace,
|
||||||
"control checkout is clean on accepted base branch matching live master"
|
root=root,
|
||||||
],
|
branch=branch or None,
|
||||||
}
|
dirty=dirty,
|
||||||
|
under_branches=False,
|
||||||
|
bootstrap_path="clean_canonical_control_checkout",
|
||||||
|
local_head_sha=local_tip,
|
||||||
|
remote_master_sha=remote_tip,
|
||||||
|
exact_next_action=(
|
||||||
|
"Call gitea_bootstrap_author_issue_worktree with the allocated "
|
||||||
|
"issue/lease pins; it will create the branches/ worktree and lock."
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
import fcntl
|
import fcntl
|
||||||
@@ -1035,26 +1240,31 @@ def bootstrap_author_issue_worktree(
|
|||||||
save_phase_journal(journal, journal_dir=lock_dir)
|
save_phase_journal(journal, journal_dir=lock_dir)
|
||||||
|
|
||||||
# Phase 6: STATE_ESTABLISHED — Issue Lock Acquisition
|
# Phase 6: STATE_ESTABLISHED — Issue Lock Acquisition
|
||||||
|
#
|
||||||
|
# #953: this used to hand-build a thinner record — claimant at the top
|
||||||
|
# level, no work_lease, no lock_provenance, no expiry — which every
|
||||||
|
# downstream reader then refused. It now builds through the one shared
|
||||||
|
# canonical contract, so the lock bootstrap writes is the same lock
|
||||||
|
# gitea_lock_issue writes.
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
try:
|
try:
|
||||||
lock_data = {
|
lock_data = author_lock_contract.build_canonical_issue_lock(
|
||||||
"remote": remote,
|
issue_number=issue_number,
|
||||||
"org": org or "Scaled-Tech-Consulting",
|
branch_name=target_branch,
|
||||||
"repo": repo or "Gitea-Tools",
|
worktree_path=target_worktree,
|
||||||
"issue_number": issue_number,
|
remote=remote,
|
||||||
"branch": target_branch,
|
org=org or "Scaled-Tech-Consulting",
|
||||||
"branch_name": target_branch,
|
repo=repo or "Gitea-Tools",
|
||||||
"worktree_path": target_worktree,
|
identity=identity,
|
||||||
"owner_session": session,
|
profile=profile,
|
||||||
"claimant": {
|
tool="gitea_bootstrap_author_issue_worktree",
|
||||||
"username": identity,
|
source=author_lock_contract.SOURCE_BOOTSTRAP,
|
||||||
"profile": profile,
|
owner_session=session,
|
||||||
},
|
assignment_id=assignment_id,
|
||||||
"assignment_id": assignment_id,
|
lease_id=lease_id,
|
||||||
"lease_id": lease_id,
|
expected_base_sha=live_master_sha,
|
||||||
"expected_base_sha": live_master_sha,
|
)
|
||||||
"created_at": datetime.now(timezone.utc).isoformat(),
|
lock_data["created_at"] = datetime.now(timezone.utc).isoformat()
|
||||||
}
|
|
||||||
journal.setdefault("pending_creations", {})["lock"] = True
|
journal.setdefault("pending_creations", {})["lock"] = True
|
||||||
journal["artifacts_created"]["lock_created"] = True
|
journal["artifacts_created"]["lock_created"] = True
|
||||||
save_phase_journal(journal, journal_dir=lock_dir)
|
save_phase_journal(journal, journal_dir=lock_dir)
|
||||||
@@ -1072,9 +1282,65 @@ def bootstrap_author_issue_worktree(
|
|||||||
"exact_next_action": "Verify lease/assignment state and retry.",
|
"exact_next_action": "Verify lease/assignment state and retry.",
|
||||||
}
|
}
|
||||||
|
|
||||||
|
# ── #953 AC7: verify the lock that was actually written ──
|
||||||
|
# Reporting "lock_created: true" and then directing the author to
|
||||||
|
# implement is what produced the unrecoverable state: by the time any
|
||||||
|
# reader refused the lock, the branch already carried commits and every
|
||||||
|
# sanctioned recovery path had become ineligible. The lock is therefore
|
||||||
|
# read back from disk and structurally verified *before* this function
|
||||||
|
# can report success, and a partial lock fails closed here — while the
|
||||||
|
# branch is still base-equivalent and recovery is still cheap.
|
||||||
|
written_lock = issue_lock_store.read_lock_file(lock_res)
|
||||||
|
contract = author_lock_contract.assess_lock_contract(written_lock)
|
||||||
|
if not contract["canonical"]:
|
||||||
|
journal["failure_reason"] = author_lock_contract.format_contract_refusal(
|
||||||
|
contract
|
||||||
|
)
|
||||||
|
compensation = run_compensating_recovery(
|
||||||
|
journal, root, journal_dir=lock_dir
|
||||||
|
)
|
||||||
|
# AC5/AC15: the recommendation must describe the state compensation
|
||||||
|
# actually left, not the state that provoked it.
|
||||||
|
# ``run_compensating_recovery`` has by now released the lock, removed
|
||||||
|
# the worktree, and deleted the branch, so recommending
|
||||||
|
# incomplete-lock recovery for those exact artifacts would refuse
|
||||||
|
# twice over. Observe what survived and answer for that.
|
||||||
|
post_state = author_lock_contract.assess_post_compensation_state(
|
||||||
|
compensation,
|
||||||
|
lock_present=bool(lock_res) and os.path.exists(lock_res),
|
||||||
|
worktree_present=os.path.isdir(target_worktree),
|
||||||
|
branch_present=_branch_exists(root, target_branch),
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"success": False,
|
||||||
|
"reason_code": "incomplete_issue_lock_contract",
|
||||||
|
"message": author_lock_contract.format_contract_refusal(contract),
|
||||||
|
"issue_number": issue_number,
|
||||||
|
"branch_name": target_branch,
|
||||||
|
"worktree_path": target_worktree,
|
||||||
|
"lock_state": lock_res,
|
||||||
|
"lock_contract": contract,
|
||||||
|
"missing_fields": contract["missing_fields"],
|
||||||
|
"implementation_allowed": False,
|
||||||
|
"compensating_recovery": compensation,
|
||||||
|
"post_compensation_state": post_state,
|
||||||
|
# AC15: never strand a branch or worktree without a structured
|
||||||
|
# recovery recommendation — and never name an artifact the
|
||||||
|
# rollback has already deleted.
|
||||||
|
"exact_next_action": author_lock_contract.post_compensation_action(
|
||||||
|
post_state,
|
||||||
|
issue_number=issue_number,
|
||||||
|
branch_name=target_branch,
|
||||||
|
worktree_path=target_worktree,
|
||||||
|
missing_fields=contract["missing_fields"],
|
||||||
|
),
|
||||||
|
"phase_journal": journal,
|
||||||
|
}
|
||||||
|
|
||||||
journal["phases"][PHASE_6_STATE_ESTABLISHED] = {
|
journal["phases"][PHASE_6_STATE_ESTABLISHED] = {
|
||||||
"status": "completed",
|
"status": "completed",
|
||||||
"lock": lock_res,
|
"lock": lock_res,
|
||||||
|
"lock_contract": contract["contract"],
|
||||||
}
|
}
|
||||||
journal["phases"][PHASE_7_TRANSITION_COMPLETED] = {
|
journal["phases"][PHASE_7_TRANSITION_COMPLETED] = {
|
||||||
"status": "completed",
|
"status": "completed",
|
||||||
@@ -1098,9 +1364,16 @@ def bootstrap_author_issue_worktree(
|
|||||||
"assignment_id": assignment_id,
|
"assignment_id": assignment_id,
|
||||||
"idempotency_key": key,
|
"idempotency_key": key,
|
||||||
"lock_state": lock_res,
|
"lock_state": lock_res,
|
||||||
|
"lock_contract": contract,
|
||||||
|
# #953 AC6: the canonical ownership token for this claim. Never null
|
||||||
|
# on a successful bootstrap — it is the fencing token every
|
||||||
|
# subsequent heartbeat and renewal is checked against.
|
||||||
|
"task_session_id": contract["task_session_id"],
|
||||||
|
"implementation_allowed": True,
|
||||||
"phase_journal": journal,
|
"phase_journal": journal,
|
||||||
"exact_next_action": (
|
# #953 AC5: executable under the state actually returned. The lock
|
||||||
"Call gitea_whoami, then gitea_resolve_task_capability(task='work_issue') "
|
# has been read back and verified canonical, so proceeding to
|
||||||
"and proceed with author implementation in the bootstrapped worktree."
|
# implementation is genuinely the correct next step here — which is
|
||||||
),
|
# exactly what the old unconditional wording could not promise.
|
||||||
|
"exact_next_action": author_lock_contract.recommended_action(contract),
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,623 @@
|
|||||||
|
"""One canonical author issue-lock contract shared by every writer (#953).
|
||||||
|
|
||||||
|
Before this module, ``gitea_lock_issue`` and
|
||||||
|
``gitea_bootstrap_author_issue_worktree`` each wrote their own lock record.
|
||||||
|
``gitea_lock_issue`` wrote the canonical shape — ``work_lease`` carrying the
|
||||||
|
claimant plus a sanctioned ``lock_provenance`` — while bootstrap wrote a thinner
|
||||||
|
record with the claimant at the lock top level, ``lease_id: null``, and no
|
||||||
|
``work_lease``, ``lock_provenance``, or expiry at all.
|
||||||
|
|
||||||
|
Every downstream reader was written against the canonical shape, so a lock that
|
||||||
|
bootstrap reported as successfully created was simultaneously:
|
||||||
|
|
||||||
|
* un-heartbeatable — the ownership check read the claimant only from
|
||||||
|
``work_lease.claimant``;
|
||||||
|
* un-renewable — expiry is read only from ``work_lease.expires_at``, so a
|
||||||
|
missing lease read as "never expires", and #760 exact-owner renewal only ever
|
||||||
|
assesses an *expired* lease;
|
||||||
|
* un-re-lockable — the branch had by then advanced past its base;
|
||||||
|
* and rejected by the #447 create-PR provenance guard.
|
||||||
|
|
||||||
|
Each of those gates is individually correct. The defect was that two writers
|
||||||
|
disagreed about what a lock *is*. This module is the single definition, and both
|
||||||
|
writers now build through it.
|
||||||
|
|
||||||
|
Nothing here weakens a guard. ``build_sanctioned_lock_provenance`` remains the
|
||||||
|
only provenance source, provenance is never accepted from a caller, and the
|
||||||
|
#447 guard is untouched — this module simply makes bootstrap satisfy it.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from datetime import datetime, timedelta, timezone
|
||||||
|
from typing import Any, Mapping
|
||||||
|
|
||||||
|
import issue_lock_provenance
|
||||||
|
import issue_lock_store
|
||||||
|
import lease_policy
|
||||||
|
|
||||||
|
# Bootstrap writes through the same sanctioned source as gitea_lock_issue: the
|
||||||
|
# lock it produces *is* a canonical lock, not a second dialect that readers must
|
||||||
|
# learn. Adding a distinct source would have required widening
|
||||||
|
# SANCTIONED_LOCK_SOURCES, which is exactly the #447 weakening this issue's
|
||||||
|
# safety requirements forbid.
|
||||||
|
SOURCE_BOOTSTRAP = issue_lock_provenance.SOURCE_LOCK_ISSUE
|
||||||
|
|
||||||
|
# Recovery of an incomplete bootstrap lock (#953 AC8-AC11) deliberately writes
|
||||||
|
# through SOURCE_LOCK_ISSUE too, and records its distinctness in
|
||||||
|
# ``lock_provenance.written_by_tool`` plus the ``bootstrap_lock_recovery``
|
||||||
|
# transition block instead. There is no distinct recovery *source* constant, for
|
||||||
|
# the same reason bootstrap has none: minting one would require widening
|
||||||
|
# SANCTIONED_LOCK_SOURCES, which the #447 safety requirements forbid.
|
||||||
|
|
||||||
|
#: Top-level keys every canonical author issue lock must carry.
|
||||||
|
REQUIRED_LOCK_FIELDS: tuple[str, ...] = (
|
||||||
|
"remote",
|
||||||
|
"org",
|
||||||
|
"repo",
|
||||||
|
"issue_number",
|
||||||
|
"branch_name",
|
||||||
|
"worktree_path",
|
||||||
|
"work_lease",
|
||||||
|
"lock_provenance",
|
||||||
|
)
|
||||||
|
|
||||||
|
#: Keys every canonical ``work_lease`` must carry.
|
||||||
|
REQUIRED_WORK_LEASE_FIELDS: tuple[str, ...] = (
|
||||||
|
"operation_type",
|
||||||
|
"issue_number",
|
||||||
|
"branch",
|
||||||
|
"worktree_path",
|
||||||
|
"claimant",
|
||||||
|
"created_at",
|
||||||
|
"expires_at",
|
||||||
|
"last_heartbeat_at",
|
||||||
|
"task_session_id",
|
||||||
|
"lifecycle_version",
|
||||||
|
)
|
||||||
|
|
||||||
|
# ── Explicit expiration states (AC12) ──
|
||||||
|
# The bug this replaces: a lock with no recorded expiry produced
|
||||||
|
# ``is_lease_expired() -> False``, which reads as "not yet expired" and made the
|
||||||
|
# lock permanently non-expiring *and* permanently ineligible for the renewal
|
||||||
|
# path, which only ever assesses an expired lease. "Absent" and "in the future"
|
||||||
|
# are different facts and are now named differently.
|
||||||
|
EXPIRATION_RECORDED = "recorded"
|
||||||
|
EXPIRATION_MISSING = "missing"
|
||||||
|
EXPIRATION_UNPARSEABLE = "unparseable"
|
||||||
|
|
||||||
|
#: Structural verdicts returned by :func:`assess_lock_contract`.
|
||||||
|
CONTRACT_CANONICAL = "canonical"
|
||||||
|
CONTRACT_INCOMPLETE = "incomplete"
|
||||||
|
CONTRACT_LEGACY = "legacy"
|
||||||
|
CONTRACT_ABSENT = "absent"
|
||||||
|
|
||||||
|
|
||||||
|
def _text(value: Any) -> str:
|
||||||
|
return str(value or "").strip()
|
||||||
|
|
||||||
|
|
||||||
|
def now_utc() -> datetime:
|
||||||
|
return datetime.now(timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
def format_timestamp(value: datetime) -> str:
|
||||||
|
"""Serialize in the durable ``...Z`` form already used on disk."""
|
||||||
|
return (
|
||||||
|
value.astimezone(timezone.utc)
|
||||||
|
.replace(microsecond=0)
|
||||||
|
.isoformat()
|
||||||
|
.replace("+00:00", "Z")
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def lock_claimant(lock: Mapping[str, Any] | None) -> dict[str, str]:
|
||||||
|
"""Read the claimant from either canonical or legacy placement.
|
||||||
|
|
||||||
|
``work_lease.claimant`` is canonical and is preferred. A top-level
|
||||||
|
``claimant`` is the legacy/bootstrap placement and is accepted as a
|
||||||
|
fallback (AC14) — three separate readers already disagreed about this
|
||||||
|
(``issue_lock_store``, ``issue_lock_renewal``, ``issue_lock_recovery``),
|
||||||
|
which is why it now lives in one place.
|
||||||
|
|
||||||
|
Reading a legacy placement is *not* a widening: every caller still compares
|
||||||
|
the values it returns against server-resolved identity and profile. This
|
||||||
|
only decides where to look, never whether ownership is proven.
|
||||||
|
|
||||||
|
Delegates to ``issue_lock_store.lock_claimant`` rather than reimplementing
|
||||||
|
the rule. A second copy here would be a fourth reader that could drift from
|
||||||
|
the other three, which is the exact failure #953 exists to end. It lives in
|
||||||
|
the store because ``author_lock_contract`` imports the store, so defining it
|
||||||
|
here would make that import circular.
|
||||||
|
"""
|
||||||
|
recorded = issue_lock_store.lock_claimant(dict(lock) if isinstance(lock, Mapping) else None)
|
||||||
|
return {
|
||||||
|
"username": _text(recorded.get("username")),
|
||||||
|
"profile": _text(recorded.get("profile")),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def claimant_placement(lock: Mapping[str, Any] | None) -> str:
|
||||||
|
"""Where the claimant was found: ``work_lease``, ``top_level``, or ``absent``."""
|
||||||
|
if not isinstance(lock, Mapping):
|
||||||
|
return "absent"
|
||||||
|
lease = lock.get("work_lease")
|
||||||
|
if isinstance(lease, Mapping) and isinstance(lease.get("claimant"), Mapping):
|
||||||
|
return "work_lease"
|
||||||
|
if isinstance(lock.get("claimant"), Mapping):
|
||||||
|
return "top_level"
|
||||||
|
return "absent"
|
||||||
|
|
||||||
|
|
||||||
|
def build_claimant(*, username: str | None, profile: str | None) -> dict[str, str]:
|
||||||
|
"""Build the canonical claimant pair from server-resolved values."""
|
||||||
|
return {"username": _text(username), "profile": _text(profile)}
|
||||||
|
|
||||||
|
|
||||||
|
def build_author_issue_work_lease(
|
||||||
|
*,
|
||||||
|
issue_number: int,
|
||||||
|
branch_name: str,
|
||||||
|
worktree_path: str,
|
||||||
|
claimant: Mapping[str, Any],
|
||||||
|
task_session_id: str | None = None,
|
||||||
|
created: datetime | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Build the canonical author ``work_lease``.
|
||||||
|
|
||||||
|
The single definition behind both writers. The TTL comes from the central
|
||||||
|
policy rather than a literal, and the window slides from the last valid
|
||||||
|
heartbeat (#790), so an abandoned task releases its claim within one TTL.
|
||||||
|
"""
|
||||||
|
started = created or now_utc()
|
||||||
|
policy = lease_policy.policy_for(lease_policy.TASK_CLASS_AUTHOR_ISSUE_WORK)
|
||||||
|
expires = started + timedelta(minutes=policy.initial_ttl_minutes)
|
||||||
|
session_id = _text(task_session_id) or issue_lock_store.mint_task_session_id(
|
||||||
|
issue_lock_store.AUTHOR_ISSUE_WORK_LEASE
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"operation_type": issue_lock_store.AUTHOR_ISSUE_WORK_LEASE,
|
||||||
|
"issue_number": int(issue_number),
|
||||||
|
"pr_number": None,
|
||||||
|
"branch": branch_name,
|
||||||
|
"worktree_path": worktree_path,
|
||||||
|
"claimant": dict(claimant),
|
||||||
|
"created_at": format_timestamp(started),
|
||||||
|
"expires_at": format_timestamp(expires),
|
||||||
|
"last_heartbeat_at": format_timestamp(started),
|
||||||
|
# #790 AC-N1: the ownership key for this task, distinct from the
|
||||||
|
# recorded PID, which is the shared daemon and identifies no task.
|
||||||
|
"task_session_id": session_id,
|
||||||
|
# #790 AC-N8: the explicit lifecycle marker. Its absence — never a
|
||||||
|
# timestamp comparison — is what makes a lock legacy.
|
||||||
|
"lifecycle_version": lease_policy.LIFECYCLE_HEARTBEAT_V1,
|
||||||
|
"heartbeat_count": 1,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def build_canonical_issue_lock(
|
||||||
|
*,
|
||||||
|
issue_number: int,
|
||||||
|
branch_name: str,
|
||||||
|
worktree_path: str,
|
||||||
|
remote: str,
|
||||||
|
org: str,
|
||||||
|
repo: str,
|
||||||
|
identity: str | None,
|
||||||
|
profile: str | None,
|
||||||
|
tool: str,
|
||||||
|
source: str = issue_lock_provenance.SOURCE_LOCK_ISSUE,
|
||||||
|
owner_session: str | None = None,
|
||||||
|
assignment_id: str | None = None,
|
||||||
|
lease_id: str | None = None,
|
||||||
|
expected_base_sha: str | None = None,
|
||||||
|
task_session_id: str | None = None,
|
||||||
|
created: datetime | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Build a complete canonical lock record.
|
||||||
|
|
||||||
|
``tool`` and ``source`` are server-supplied. There is deliberately no
|
||||||
|
parameter through which a caller could inject provenance: the #953 safety
|
||||||
|
requirements forbid caller-manufactured provenance, so provenance is always
|
||||||
|
minted here from ``build_sanctioned_lock_provenance``.
|
||||||
|
"""
|
||||||
|
claimant = build_claimant(username=identity, profile=profile)
|
||||||
|
work_lease = build_author_issue_work_lease(
|
||||||
|
issue_number=issue_number,
|
||||||
|
branch_name=branch_name,
|
||||||
|
worktree_path=worktree_path,
|
||||||
|
claimant=claimant,
|
||||||
|
task_session_id=task_session_id,
|
||||||
|
created=created,
|
||||||
|
)
|
||||||
|
record: dict[str, Any] = {
|
||||||
|
"remote": remote,
|
||||||
|
"org": org,
|
||||||
|
"repo": repo,
|
||||||
|
"issue_number": int(issue_number),
|
||||||
|
"branch": branch_name,
|
||||||
|
"branch_name": branch_name,
|
||||||
|
"worktree_path": worktree_path,
|
||||||
|
"work_lease": work_lease,
|
||||||
|
"lock_provenance": issue_lock_provenance.build_sanctioned_lock_provenance(
|
||||||
|
tool=tool,
|
||||||
|
source=source,
|
||||||
|
claimant=claimant,
|
||||||
|
),
|
||||||
|
}
|
||||||
|
if owner_session is not None:
|
||||||
|
record["owner_session"] = owner_session
|
||||||
|
if assignment_id is not None:
|
||||||
|
record["assignment_id"] = assignment_id
|
||||||
|
# #953 AC6: a null lease id is recorded only when no workflow lease was
|
||||||
|
# allocated for this bootstrap. The task-session identifier in the
|
||||||
|
# work_lease is what downstream ownership checks fence on, and it is never
|
||||||
|
# null on a canonical lock.
|
||||||
|
if lease_id is not None:
|
||||||
|
record["lease_id"] = lease_id
|
||||||
|
if expected_base_sha is not None:
|
||||||
|
record["expected_base_sha"] = expected_base_sha
|
||||||
|
return record
|
||||||
|
|
||||||
|
|
||||||
|
def expiration_state(lock: Mapping[str, Any] | None) -> dict[str, Any]:
|
||||||
|
"""Classify a lock's recorded expiry explicitly (AC12).
|
||||||
|
|
||||||
|
Distinguishes "no expiry was ever recorded" from "an expiry was recorded
|
||||||
|
and is still in the future". Collapsing those two into a single ``False``
|
||||||
|
from ``is_lease_expired`` is what let a malformed lock be treated as
|
||||||
|
permanently live and simultaneously never renewable.
|
||||||
|
"""
|
||||||
|
if not isinstance(lock, Mapping):
|
||||||
|
return {"state": EXPIRATION_MISSING, "expires_at": None, "expired": None}
|
||||||
|
lease = lock.get("work_lease")
|
||||||
|
raw = lease.get("expires_at") if isinstance(lease, Mapping) else None
|
||||||
|
text = _text(raw)
|
||||||
|
if not text:
|
||||||
|
return {"state": EXPIRATION_MISSING, "expires_at": None, "expired": None}
|
||||||
|
try:
|
||||||
|
parsed = datetime.fromisoformat(text.replace("Z", "+00:00")).astimezone(
|
||||||
|
timezone.utc
|
||||||
|
)
|
||||||
|
except ValueError:
|
||||||
|
return {"state": EXPIRATION_UNPARSEABLE, "expires_at": text, "expired": None}
|
||||||
|
return {
|
||||||
|
"state": EXPIRATION_RECORDED,
|
||||||
|
"expires_at": text,
|
||||||
|
"expired": parsed <= now_utc(),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def missing_contract_fields(lock: Mapping[str, Any] | None) -> list[str]:
|
||||||
|
"""Name every canonical field a lock does not carry (AC7)."""
|
||||||
|
if not isinstance(lock, Mapping):
|
||||||
|
return ["<no lock record>"]
|
||||||
|
missing: list[str] = []
|
||||||
|
for field in REQUIRED_LOCK_FIELDS:
|
||||||
|
value = lock.get(field)
|
||||||
|
if value is None or (isinstance(value, str) and not value.strip()):
|
||||||
|
missing.append(field)
|
||||||
|
lease = lock.get("work_lease")
|
||||||
|
if not isinstance(lease, Mapping):
|
||||||
|
if "work_lease" not in missing:
|
||||||
|
missing.append("work_lease")
|
||||||
|
else:
|
||||||
|
for field in REQUIRED_WORK_LEASE_FIELDS:
|
||||||
|
value = lease.get(field)
|
||||||
|
if value is None or (isinstance(value, str) and not value.strip()):
|
||||||
|
missing.append(f"work_lease.{field}")
|
||||||
|
provenance = lock.get("lock_provenance")
|
||||||
|
if isinstance(provenance, Mapping):
|
||||||
|
if (
|
||||||
|
_text(provenance.get("source"))
|
||||||
|
not in issue_lock_provenance.SANCTIONED_LOCK_SOURCES
|
||||||
|
):
|
||||||
|
missing.append("lock_provenance.source (not sanctioned)")
|
||||||
|
if not _text(provenance.get("written_by_tool")):
|
||||||
|
missing.append("lock_provenance.written_by_tool")
|
||||||
|
claimant = lock_claimant(lock)
|
||||||
|
if not claimant["username"]:
|
||||||
|
missing.append("claimant.username")
|
||||||
|
if not claimant["profile"]:
|
||||||
|
missing.append("claimant.profile")
|
||||||
|
return missing
|
||||||
|
|
||||||
|
|
||||||
|
def assess_lock_contract(lock: Mapping[str, Any] | None) -> dict[str, Any]:
|
||||||
|
"""Structural, read-only verdict on a durable lock record (AC7, AC16).
|
||||||
|
|
||||||
|
Pure inspection: it reads the record it is handed and mutates nothing —
|
||||||
|
no lock, lease, branch, worktree, issue, or PR. Callers use it both to
|
||||||
|
verify a lock they just wrote and to report on one they found.
|
||||||
|
"""
|
||||||
|
if not isinstance(lock, Mapping) or not lock:
|
||||||
|
return {
|
||||||
|
"contract": CONTRACT_ABSENT,
|
||||||
|
"canonical": False,
|
||||||
|
"missing_fields": ["<no lock record>"],
|
||||||
|
"claimant": {"username": "", "profile": ""},
|
||||||
|
"claimant_placement": "absent",
|
||||||
|
"expiration": {
|
||||||
|
"state": EXPIRATION_MISSING,
|
||||||
|
"expires_at": None,
|
||||||
|
"expired": None,
|
||||||
|
},
|
||||||
|
"heartbeatable": False,
|
||||||
|
"create_pr_eligible": False,
|
||||||
|
"lock_generation": None,
|
||||||
|
"task_session_id": None,
|
||||||
|
"reasons": ["no durable lock record"],
|
||||||
|
}
|
||||||
|
|
||||||
|
missing = missing_contract_fields(lock)
|
||||||
|
claimant = lock_claimant(lock)
|
||||||
|
placement = claimant_placement(lock)
|
||||||
|
expiration = expiration_state(lock)
|
||||||
|
provenance_check = issue_lock_provenance.assess_lock_file_for_create_pr(dict(lock))
|
||||||
|
|
||||||
|
# Canonical means: every required field present, the claimant in the
|
||||||
|
# canonical placement, an expiry actually recorded, and the untouched #447
|
||||||
|
# guard satisfied.
|
||||||
|
canonical = (
|
||||||
|
not missing
|
||||||
|
and placement == "work_lease"
|
||||||
|
and expiration["state"] == EXPIRATION_RECORDED
|
||||||
|
and bool(provenance_check.get("proven"))
|
||||||
|
)
|
||||||
|
if canonical:
|
||||||
|
contract = CONTRACT_CANONICAL
|
||||||
|
elif placement == "top_level" and claimant["username"] and claimant["profile"]:
|
||||||
|
contract = CONTRACT_LEGACY
|
||||||
|
else:
|
||||||
|
contract = CONTRACT_INCOMPLETE
|
||||||
|
|
||||||
|
reasons: list[str] = []
|
||||||
|
if missing:
|
||||||
|
reasons.append("missing canonical fields: " + ", ".join(missing))
|
||||||
|
if placement == "top_level":
|
||||||
|
reasons.append(
|
||||||
|
"claimant recorded at the lock top level rather than in work_lease "
|
||||||
|
"(legacy/bootstrap placement)"
|
||||||
|
)
|
||||||
|
if expiration["state"] == EXPIRATION_MISSING:
|
||||||
|
reasons.append(
|
||||||
|
"no expiration recorded; the lock is neither expirable nor renewable "
|
||||||
|
"until it is upgraded"
|
||||||
|
)
|
||||||
|
elif expiration["state"] == EXPIRATION_UNPARSEABLE:
|
||||||
|
reasons.append(f"unparseable expires_at '{expiration['expires_at']}'")
|
||||||
|
if provenance_check.get("block"):
|
||||||
|
reasons.extend(provenance_check.get("reasons") or [])
|
||||||
|
|
||||||
|
# Heartbeat needs the claimant pair (from either placement, post-fix) plus a
|
||||||
|
# task-session identifier to fence on.
|
||||||
|
lease = lock.get("work_lease")
|
||||||
|
task_session_id = (
|
||||||
|
_text(lease.get("task_session_id")) if isinstance(lease, Mapping) else ""
|
||||||
|
)
|
||||||
|
heartbeatable = bool(
|
||||||
|
claimant["username"] and claimant["profile"] and task_session_id
|
||||||
|
)
|
||||||
|
|
||||||
|
return {
|
||||||
|
"contract": contract,
|
||||||
|
"canonical": canonical,
|
||||||
|
"missing_fields": missing,
|
||||||
|
"claimant": claimant,
|
||||||
|
"claimant_placement": placement,
|
||||||
|
"expiration": expiration,
|
||||||
|
"heartbeatable": heartbeatable,
|
||||||
|
"create_pr_eligible": bool(provenance_check.get("proven")),
|
||||||
|
"lock_generation": lock.get("lock_generation"),
|
||||||
|
"task_session_id": task_session_id or None,
|
||||||
|
"reasons": reasons,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def format_contract_refusal(assessment: Mapping[str, Any]) -> str:
|
||||||
|
"""Human-readable refusal naming exactly what the lock is missing."""
|
||||||
|
missing = ", ".join(assessment.get("missing_fields") or []) or "unknown fields"
|
||||||
|
return (
|
||||||
|
"Issue lock contract incomplete (#953): "
|
||||||
|
f"{missing}. The lock cannot be heartbeated, renewed, or accepted by "
|
||||||
|
"gitea_create_pr in this state (fail closed)"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def recommended_action(assessment: Mapping[str, Any]) -> str:
|
||||||
|
"""The one executable next step for a lock in this state (AC5, AC15)."""
|
||||||
|
contract = assessment.get("contract")
|
||||||
|
if contract == CONTRACT_CANONICAL:
|
||||||
|
return (
|
||||||
|
"Lock is canonical. Call gitea_whoami, then "
|
||||||
|
"gitea_resolve_task_capability(task='work_issue'), then proceed with "
|
||||||
|
"author implementation in the bootstrapped worktree."
|
||||||
|
)
|
||||||
|
if contract == CONTRACT_ABSENT:
|
||||||
|
return (
|
||||||
|
"No durable lock exists. Call gitea_lock_issue for this issue and "
|
||||||
|
"branch before writing any implementation bytes."
|
||||||
|
)
|
||||||
|
return (
|
||||||
|
"Do not begin implementation. Call "
|
||||||
|
"gitea_recover_incomplete_bootstrap_lock for this exact issue, branch, "
|
||||||
|
"and worktree to upgrade the lock to the canonical contract, or "
|
||||||
|
"gitea_lock_issue while the worktree is still base-equivalent."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# ── Post-compensation recovery guidance (#953 AC5/AC15, review 632 F2) ──
|
||||||
|
#
|
||||||
|
# ``recommended_action`` above answers "what can be done about a lock in this
|
||||||
|
# shape". That is the wrong question on the bootstrap AC7 refusal path, because
|
||||||
|
# ``run_compensating_recovery`` has already run by the time the answer is
|
||||||
|
# reported: it releases the lock, removes the worktree when clean — which it
|
||||||
|
# always is there, no implementation bytes having been written — and deletes the
|
||||||
|
# created branch. Recommending incomplete-lock recovery for those artifacts
|
||||||
|
# hands the author two refusals in a row (``no_durable_lock``, then
|
||||||
|
# ``worktree_invalid``) for a state that a plain bootstrap retry would fix. The
|
||||||
|
# advice must describe the state that actually *remains*.
|
||||||
|
|
||||||
|
#: Compensation removed every artifact this transition created.
|
||||||
|
CLEANUP_COMPLETE = "complete"
|
||||||
|
#: Compensation removed some artifacts; others survive and are still actionable.
|
||||||
|
CLEANUP_PARTIAL = "partial"
|
||||||
|
#: Compensation itself failed or could not be observed; nothing is provable.
|
||||||
|
CLEANUP_FAILED = "failed"
|
||||||
|
|
||||||
|
|
||||||
|
def assess_post_compensation_state(
|
||||||
|
recovery: Mapping[str, Any] | None,
|
||||||
|
*,
|
||||||
|
lock_present: bool,
|
||||||
|
worktree_present: bool,
|
||||||
|
branch_present: bool,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Classify what survived compensation, from observed durable state.
|
||||||
|
|
||||||
|
Pure. The caller observes the filesystem and git; this decides. Observation
|
||||||
|
is authoritative over the journal's ``rolled_back`` list, which records what
|
||||||
|
compensation *attempted*: ``run_compensating_recovery`` swallows a failed
|
||||||
|
lock release and appends nothing, so an absent marker proves nothing either
|
||||||
|
way. The list is still carried through as corroborating evidence.
|
||||||
|
|
||||||
|
The three states are distinct facts, not degrees of the same one:
|
||||||
|
|
||||||
|
* ``CLEANUP_COMPLETE`` — compensation ran and nothing it created remains.
|
||||||
|
* ``CLEANUP_PARTIAL`` — compensation ran and artifacts survive, whether by
|
||||||
|
design (a worktree dirty at rollback time, a branch carrying commits) or
|
||||||
|
because a rollback step errored. Either way the surviving set was observed
|
||||||
|
directly, so it is known and actionable; ``failed_rollback_steps`` records
|
||||||
|
which cause applies.
|
||||||
|
* ``CLEANUP_FAILED`` — compensation never ran to completion, so nothing it
|
||||||
|
would have removed can be assumed removed.
|
||||||
|
"""
|
||||||
|
rolled_back = list((recovery or {}).get("rolled_back") or [])
|
||||||
|
executed = bool((recovery or {}).get("executed"))
|
||||||
|
failed_steps = [entry for entry in rolled_back if "_failed" in entry]
|
||||||
|
|
||||||
|
surviving: list[str] = []
|
||||||
|
if lock_present:
|
||||||
|
surviving.append("lock")
|
||||||
|
if worktree_present:
|
||||||
|
surviving.append("worktree")
|
||||||
|
if branch_present:
|
||||||
|
surviving.append("branch")
|
||||||
|
|
||||||
|
if not executed:
|
||||||
|
state = CLEANUP_FAILED
|
||||||
|
elif surviving:
|
||||||
|
state = CLEANUP_PARTIAL
|
||||||
|
else:
|
||||||
|
state = CLEANUP_COMPLETE
|
||||||
|
|
||||||
|
return {
|
||||||
|
"cleanup_state": state,
|
||||||
|
"compensation_executed": executed,
|
||||||
|
"lock_present": bool(lock_present),
|
||||||
|
"worktree_present": bool(worktree_present),
|
||||||
|
"branch_present": bool(branch_present),
|
||||||
|
"surviving_artifacts": surviving,
|
||||||
|
"removed_artifacts": [
|
||||||
|
name
|
||||||
|
for name, present in (
|
||||||
|
("lock", lock_present),
|
||||||
|
("worktree", worktree_present),
|
||||||
|
("branch", branch_present),
|
||||||
|
)
|
||||||
|
if not present
|
||||||
|
],
|
||||||
|
"failed_rollback_steps": failed_steps,
|
||||||
|
"rolled_back": rolled_back,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def post_compensation_action(
|
||||||
|
state: Mapping[str, Any],
|
||||||
|
*,
|
||||||
|
issue_number: int,
|
||||||
|
branch_name: str,
|
||||||
|
worktree_path: str,
|
||||||
|
missing_fields: list[str] | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""The one executable next step for the state compensation actually left.
|
||||||
|
|
||||||
|
Every branch names only artifacts the classification says still exist, so no
|
||||||
|
recommendation can point at something the rollback deleted.
|
||||||
|
"""
|
||||||
|
missing = ", ".join(missing_fields or []) or "the reported missing fields"
|
||||||
|
cleanup_state = state.get("cleanup_state")
|
||||||
|
lock_present = bool(state.get("lock_present"))
|
||||||
|
worktree_present = bool(state.get("worktree_present"))
|
||||||
|
branch_present = bool(state.get("branch_present"))
|
||||||
|
|
||||||
|
if not state.get("compensation_executed"):
|
||||||
|
# Compensation never ran, so nothing was rolled back and nothing about
|
||||||
|
# the remaining state was decided. The read-only surface is the only
|
||||||
|
# action executable under any state.
|
||||||
|
return (
|
||||||
|
"Compensating rollback did not complete, so the remaining state is "
|
||||||
|
f"not proven. Call gitea_inspect_issue_lock_contract for issue "
|
||||||
|
f"#{issue_number} (read-only) to establish what survives before any "
|
||||||
|
"further action. Do not retry bootstrap until it is known."
|
||||||
|
)
|
||||||
|
|
||||||
|
prefix = ""
|
||||||
|
failed_steps = state.get("failed_rollback_steps") or []
|
||||||
|
if failed_steps:
|
||||||
|
prefix = (
|
||||||
|
"Compensating rollback reported a failed step "
|
||||||
|
f"({', '.join(failed_steps)}); what survives was observed directly "
|
||||||
|
"and the action below is scoped to exactly that. "
|
||||||
|
)
|
||||||
|
|
||||||
|
if cleanup_state == CLEANUP_COMPLETE:
|
||||||
|
return (
|
||||||
|
"Compensating rollback removed the malformed lock, the branch, and "
|
||||||
|
f"the worktree, so nothing from this attempt remains. Resolve "
|
||||||
|
f"{missing} and re-run gitea_bootstrap_author_issue_worktree for "
|
||||||
|
f"issue #{issue_number} from the clean pre-bootstrap state. Do not "
|
||||||
|
"call gitea_recover_incomplete_bootstrap_lock: there is no lock, "
|
||||||
|
"branch, or worktree left for it to act on."
|
||||||
|
)
|
||||||
|
|
||||||
|
if lock_present and worktree_present and branch_present:
|
||||||
|
return prefix + (
|
||||||
|
"The lock, branch, and worktree all survive. Call "
|
||||||
|
"gitea_recover_incomplete_bootstrap_lock for issue "
|
||||||
|
f"#{issue_number}, branch '{branch_name}', and worktree "
|
||||||
|
f"'{worktree_path}', passing the worktree's current head as "
|
||||||
|
"expected_head, to upgrade the lock to the canonical contract."
|
||||||
|
)
|
||||||
|
|
||||||
|
if not lock_present and worktree_present and branch_present:
|
||||||
|
return prefix + (
|
||||||
|
"The malformed lock was released but the branch and worktree "
|
||||||
|
"survive. No implementation bytes were written, so the worktree is "
|
||||||
|
f"still base-equivalent: call gitea_lock_issue for issue "
|
||||||
|
f"#{issue_number} on branch '{branch_name}' from worktree "
|
||||||
|
f"'{worktree_path}' to acquire a canonical lock."
|
||||||
|
)
|
||||||
|
|
||||||
|
if lock_present and not worktree_present:
|
||||||
|
return prefix + (
|
||||||
|
f"The worktree for issue #{issue_number} is gone but the durable "
|
||||||
|
"lock survived, so neither gitea_recover_incomplete_bootstrap_lock "
|
||||||
|
"(it would refuse worktree_invalid) nor gitea_lock_issue (it has no "
|
||||||
|
"worktree to bind) is executable. Call "
|
||||||
|
"gitea_inspect_issue_lock_contract for issue "
|
||||||
|
f"#{issue_number} (read-only) to confirm the surviving lock; it "
|
||||||
|
"must be released by its recorded owner before bootstrap is "
|
||||||
|
"retried."
|
||||||
|
)
|
||||||
|
|
||||||
|
# Lock gone, worktree gone, some git artifact left (a branch with commits,
|
||||||
|
# or a branch this transition did not create).
|
||||||
|
return prefix + (
|
||||||
|
"Compensating rollback removed the lock and worktree; branch "
|
||||||
|
f"'{branch_name}' survives and was not deleted. Call "
|
||||||
|
f"gitea_inspect_issue_lock_contract for issue #{issue_number} "
|
||||||
|
"(read-only) to confirm no durable lock remains, then re-run "
|
||||||
|
"gitea_bootstrap_author_issue_worktree, which will adopt the existing "
|
||||||
|
"branch rather than recreating it."
|
||||||
|
)
|
||||||
@@ -0,0 +1,304 @@
|
|||||||
|
"""Target-specific recovery for incomplete bootstrap issue locks (#953).
|
||||||
|
|
||||||
|
The situation this exists for: ``gitea_bootstrap_author_issue_worktree``
|
||||||
|
reported success, wrote an incomplete lock, and told the author to implement.
|
||||||
|
The author did — legitimately, following the tool's own reported next action —
|
||||||
|
and the branch now carries real committed and pushed work. At that point every
|
||||||
|
pre-existing recovery path is simultaneously ineligible:
|
||||||
|
|
||||||
|
* heartbeat refuses, because the claimant is not where it looks;
|
||||||
|
* ``gitea_lock_issue`` refuses, because the branch is no longer base-equivalent;
|
||||||
|
* #760 exact-owner renewal never engages, because a lock with no recorded
|
||||||
|
expiry is never *expired*;
|
||||||
|
* the #447 create-PR guard refuses, because there is no provenance.
|
||||||
|
|
||||||
|
Distinct from every neighbouring path: #753 ``issue_lock_recovery`` requires a
|
||||||
|
dead owner PID, #760 ``issue_lock_renewal`` requires an *expired* lease, and
|
||||||
|
#442 ``issue_lock_adoption`` decides branch adoption. None of them addresses a
|
||||||
|
lock that is structurally incomplete and therefore never expires at all.
|
||||||
|
|
||||||
|
**What this will not do.** It never moves, resets, or rewinds a branch, and
|
||||||
|
never requires base-equivalence — the committed work is the thing being
|
||||||
|
preserved. It never pushes and never opens a pull request. It touches only the
|
||||||
|
one lock file named by (remote, org, repo, issue). It accepts no caller-supplied
|
||||||
|
provenance and no caller-supplied authorization flag; both are minted
|
||||||
|
server-side. It refuses a healthy foreign-owned lock outright, and a matching
|
||||||
|
username alone is never accepted as proof of ownership — the profile must match
|
||||||
|
too, and the lock's recorded binding must agree with the observed branch,
|
||||||
|
worktree, and head.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
from typing import Any, Mapping
|
||||||
|
|
||||||
|
import author_lock_contract
|
||||||
|
import issue_lock_store
|
||||||
|
|
||||||
|
#: Refusal codes, so callers can branch on cause rather than parse prose.
|
||||||
|
REFUSAL_NO_LOCK = "no_durable_lock"
|
||||||
|
REFUSAL_ALREADY_CANONICAL = "already_canonical"
|
||||||
|
REFUSAL_FOREIGN_CLAIMANT = "foreign_claimant"
|
||||||
|
REFUSAL_HEALTHY_FOREIGN = "healthy_foreign_lock"
|
||||||
|
REFUSAL_IDENTITY_UNRESOLVED = "identity_unresolved"
|
||||||
|
REFUSAL_BINDING_MISMATCH = "binding_mismatch"
|
||||||
|
REFUSAL_WORKTREE_INVALID = "worktree_invalid"
|
||||||
|
REFUSAL_HEAD_MISMATCH = "head_mismatch"
|
||||||
|
|
||||||
|
|
||||||
|
def _text(value: Any) -> str:
|
||||||
|
return str(value or "").strip()
|
||||||
|
|
||||||
|
|
||||||
|
def _same_realpath(left: str | None, right: str | None) -> bool:
|
||||||
|
lhs, rhs = _text(left), _text(right)
|
||||||
|
if not lhs or not rhs:
|
||||||
|
return False
|
||||||
|
try:
|
||||||
|
return os.path.realpath(lhs) == os.path.realpath(rhs)
|
||||||
|
except OSError:
|
||||||
|
return lhs == rhs
|
||||||
|
|
||||||
|
|
||||||
|
def assess_bootstrap_lock_recovery(
|
||||||
|
existing_lock: Mapping[str, Any] | None,
|
||||||
|
*,
|
||||||
|
issue_number: int,
|
||||||
|
branch_name: str,
|
||||||
|
worktree_path: str,
|
||||||
|
remote: str,
|
||||||
|
org: str,
|
||||||
|
repo: str,
|
||||||
|
identity: str | None,
|
||||||
|
profile: str | None,
|
||||||
|
observed_head: str | None,
|
||||||
|
declared_head: str | None,
|
||||||
|
worktree_exists: bool,
|
||||||
|
worktree_registered: bool,
|
||||||
|
current_branch: str | None,
|
||||||
|
now: Any = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Decide whether this exact lock may be upgraded by this exact caller.
|
||||||
|
|
||||||
|
Pure: every input is an observation the caller already made, and nothing
|
||||||
|
here reads or writes the filesystem, git, or Gitea. That is what makes the
|
||||||
|
same decision testable in isolation and reusable by the read-only
|
||||||
|
inspection surface, which must not mutate anything (AC16).
|
||||||
|
|
||||||
|
Returns a dict with ``recovery_sanctioned`` plus the full evidence set. A
|
||||||
|
refusal never raises — it reports, so the caller can surface exactly which
|
||||||
|
piece of evidence was missing.
|
||||||
|
"""
|
||||||
|
reasons: list[str] = []
|
||||||
|
refusal_code: str | None = None
|
||||||
|
|
||||||
|
contract = author_lock_contract.assess_lock_contract(existing_lock)
|
||||||
|
|
||||||
|
if not existing_lock:
|
||||||
|
return {
|
||||||
|
"recovery_sanctioned": False,
|
||||||
|
"refusal_code": REFUSAL_NO_LOCK,
|
||||||
|
"reasons": [
|
||||||
|
f"no durable issue lock exists for issue #{issue_number}; there is "
|
||||||
|
"nothing to recover (fail closed)"
|
||||||
|
],
|
||||||
|
"contract": contract,
|
||||||
|
"evidence": {},
|
||||||
|
"expected_generation": None,
|
||||||
|
}
|
||||||
|
|
||||||
|
active_identity = _text(identity)
|
||||||
|
active_profile = _text(profile)
|
||||||
|
recorded = author_lock_contract.lock_claimant(existing_lock)
|
||||||
|
freshness = issue_lock_store.assess_lock_freshness(dict(existing_lock), now=now)
|
||||||
|
generation = issue_lock_store.lock_generation(existing_lock)
|
||||||
|
|
||||||
|
evidence: dict[str, Any] = {
|
||||||
|
"recorded_claimant": recorded,
|
||||||
|
"active_identity": active_identity,
|
||||||
|
"active_profile": active_profile,
|
||||||
|
"recorded_branch": existing_lock.get("branch_name"),
|
||||||
|
"recorded_worktree": existing_lock.get("worktree_path"),
|
||||||
|
"recorded_owner_session": existing_lock.get("owner_session"),
|
||||||
|
"recorded_generation": generation,
|
||||||
|
"recorded_remote": existing_lock.get("remote"),
|
||||||
|
"recorded_org": existing_lock.get("org"),
|
||||||
|
"recorded_repo": existing_lock.get("repo"),
|
||||||
|
"observed_head": _text(observed_head),
|
||||||
|
"declared_head": _text(declared_head),
|
||||||
|
"current_branch": _text(current_branch),
|
||||||
|
"worktree_exists": bool(worktree_exists),
|
||||||
|
"worktree_registered": bool(worktree_registered),
|
||||||
|
"freshness": freshness,
|
||||||
|
"claimant_placement": contract.get("claimant_placement"),
|
||||||
|
"expiration_state": contract.get("expiration", {}).get("state"),
|
||||||
|
}
|
||||||
|
|
||||||
|
# ── Repository and issue identity (AC10) ──
|
||||||
|
if _text(existing_lock.get("remote")) != _text(remote):
|
||||||
|
reasons.append(
|
||||||
|
f"recorded remote '{existing_lock.get('remote')}' does not match '{remote}'"
|
||||||
|
)
|
||||||
|
refusal_code = refusal_code or REFUSAL_BINDING_MISMATCH
|
||||||
|
if _text(existing_lock.get("org")) != _text(org):
|
||||||
|
reasons.append(
|
||||||
|
f"recorded org '{existing_lock.get('org')}' does not match '{org}'"
|
||||||
|
)
|
||||||
|
refusal_code = refusal_code or REFUSAL_BINDING_MISMATCH
|
||||||
|
if _text(existing_lock.get("repo")) != _text(repo):
|
||||||
|
reasons.append(
|
||||||
|
f"recorded repo '{existing_lock.get('repo')}' does not match '{repo}'"
|
||||||
|
)
|
||||||
|
refusal_code = refusal_code or REFUSAL_BINDING_MISMATCH
|
||||||
|
if existing_lock.get("issue_number") != issue_number:
|
||||||
|
reasons.append(
|
||||||
|
f"lock targets issue #{existing_lock.get('issue_number')}, not "
|
||||||
|
f"#{issue_number}"
|
||||||
|
)
|
||||||
|
refusal_code = refusal_code or REFUSAL_BINDING_MISMATCH
|
||||||
|
|
||||||
|
# ── Branch and worktree binding (AC10) ──
|
||||||
|
if _text(existing_lock.get("branch_name")) != _text(branch_name):
|
||||||
|
reasons.append(
|
||||||
|
f"recorded branch '{existing_lock.get('branch_name')}' does not match "
|
||||||
|
f"'{branch_name}'"
|
||||||
|
)
|
||||||
|
refusal_code = refusal_code or REFUSAL_BINDING_MISMATCH
|
||||||
|
if not _same_realpath(existing_lock.get("worktree_path"), worktree_path):
|
||||||
|
reasons.append(
|
||||||
|
f"recorded worktree '{existing_lock.get('worktree_path')}' does not "
|
||||||
|
f"match '{worktree_path}'"
|
||||||
|
)
|
||||||
|
refusal_code = refusal_code or REFUSAL_BINDING_MISMATCH
|
||||||
|
|
||||||
|
# ── The worktree is real, registered, and on the branch (AC10) ──
|
||||||
|
# Deliberately no base-equivalence requirement and no constraint on how far
|
||||||
|
# the branch has advanced: the whole point is that it already carries the
|
||||||
|
# author's legitimate commits (AC9).
|
||||||
|
if not worktree_exists:
|
||||||
|
reasons.append(f"declared worktree '{worktree_path}' does not exist")
|
||||||
|
refusal_code = refusal_code or REFUSAL_WORKTREE_INVALID
|
||||||
|
if not worktree_registered:
|
||||||
|
reasons.append(f"worktree '{worktree_path}' is not a registered git worktree")
|
||||||
|
refusal_code = refusal_code or REFUSAL_WORKTREE_INVALID
|
||||||
|
if _text(current_branch) != _text(branch_name):
|
||||||
|
reasons.append(
|
||||||
|
f"worktree is on branch '{_text(current_branch) or 'unknown'}', not "
|
||||||
|
f"'{branch_name}'"
|
||||||
|
)
|
||||||
|
refusal_code = refusal_code or REFUSAL_WORKTREE_INVALID
|
||||||
|
|
||||||
|
# ── Current head fencing (AC10) ──
|
||||||
|
# The caller names the commit it believes it is recovering. A mismatch means
|
||||||
|
# the worktree moved under the caller, so the decision is stale.
|
||||||
|
if not _text(observed_head):
|
||||||
|
reasons.append("could not observe the worktree head")
|
||||||
|
refusal_code = refusal_code or REFUSAL_HEAD_MISMATCH
|
||||||
|
elif _text(declared_head) and _text(declared_head) != _text(observed_head):
|
||||||
|
reasons.append(
|
||||||
|
f"declared head '{_text(declared_head)}' does not match observed head "
|
||||||
|
f"'{_text(observed_head)}'"
|
||||||
|
)
|
||||||
|
refusal_code = refusal_code or REFUSAL_HEAD_MISMATCH
|
||||||
|
|
||||||
|
# ── Ownership (AC10, AC11) ──
|
||||||
|
# A matching username alone is never sufficient: the profile must match too,
|
||||||
|
# and both are compared against server-resolved values the caller cannot set.
|
||||||
|
if not active_identity or not active_profile:
|
||||||
|
reasons.append(
|
||||||
|
"active identity and profile could not both be resolved; ownership "
|
||||||
|
"cannot be proven"
|
||||||
|
)
|
||||||
|
refusal_code = refusal_code or REFUSAL_IDENTITY_UNRESOLVED
|
||||||
|
if not recorded["username"] or not recorded["profile"]:
|
||||||
|
reasons.append(
|
||||||
|
"durable lock does not record both a claimant username and profile"
|
||||||
|
)
|
||||||
|
refusal_code = refusal_code or REFUSAL_FOREIGN_CLAIMANT
|
||||||
|
elif (
|
||||||
|
recorded["username"] != active_identity
|
||||||
|
or recorded["profile"] != active_profile
|
||||||
|
):
|
||||||
|
# AC11: a foreign-owned lock is never recoverable through this path,
|
||||||
|
# healthy or not. The healthy case is reported distinctly so the refusal
|
||||||
|
# is legible, but both refuse.
|
||||||
|
if freshness.get("live"):
|
||||||
|
reasons.append(
|
||||||
|
f"lock is owned by a healthy foreign claimant "
|
||||||
|
f"'{recorded['username']}/{recorded['profile']}'; takeover is not "
|
||||||
|
"a recovery path"
|
||||||
|
)
|
||||||
|
refusal_code = REFUSAL_HEALTHY_FOREIGN
|
||||||
|
else:
|
||||||
|
reasons.append(
|
||||||
|
f"lock claimant '{recorded['username']}/{recorded['profile']}' "
|
||||||
|
f"does not match active '{active_identity}/{active_profile}'"
|
||||||
|
)
|
||||||
|
refusal_code = refusal_code or REFUSAL_FOREIGN_CLAIMANT
|
||||||
|
|
||||||
|
# ── Nothing to recover ──
|
||||||
|
# A lock that is already canonical is left strictly alone. Rewriting it would
|
||||||
|
# mint a new task-session identifier and invalidate the heartbeat token the
|
||||||
|
# legitimate owner is already using.
|
||||||
|
if contract.get("canonical") and not reasons:
|
||||||
|
return {
|
||||||
|
"recovery_sanctioned": False,
|
||||||
|
"refusal_code": REFUSAL_ALREADY_CANONICAL,
|
||||||
|
"reasons": [
|
||||||
|
"lock already satisfies the canonical contract; no recovery is "
|
||||||
|
"required"
|
||||||
|
],
|
||||||
|
"contract": contract,
|
||||||
|
"evidence": evidence,
|
||||||
|
"expected_generation": generation,
|
||||||
|
}
|
||||||
|
|
||||||
|
sanctioned = not reasons
|
||||||
|
return {
|
||||||
|
"recovery_sanctioned": sanctioned,
|
||||||
|
"refusal_code": None if sanctioned else refusal_code,
|
||||||
|
"reasons": reasons,
|
||||||
|
"contract": contract,
|
||||||
|
"evidence": evidence,
|
||||||
|
"expected_generation": generation,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def build_recovery_record(
|
||||||
|
assessment: Mapping[str, Any],
|
||||||
|
*,
|
||||||
|
recovered_at: str,
|
||||||
|
new_task_session_id: str,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Auditable record of the ownership and generation transition (AC10).
|
||||||
|
|
||||||
|
A recovered lock must never read as an original claim, so both sides of the
|
||||||
|
transition are preserved: what the incomplete lock recorded, and what
|
||||||
|
replaced it.
|
||||||
|
"""
|
||||||
|
evidence = dict(assessment.get("evidence") or {})
|
||||||
|
contract = dict(assessment.get("contract") or {})
|
||||||
|
return {
|
||||||
|
"recovery_kind": "incomplete_bootstrap_lock",
|
||||||
|
"recovered_at": recovered_at,
|
||||||
|
"prior_contract": contract.get("contract"),
|
||||||
|
"prior_missing_fields": list(contract.get("missing_fields") or []),
|
||||||
|
"prior_claimant_placement": evidence.get("claimant_placement"),
|
||||||
|
"prior_expiration_state": evidence.get("expiration_state"),
|
||||||
|
"prior_generation": evidence.get("recorded_generation"),
|
||||||
|
"prior_owner_session": evidence.get("recorded_owner_session"),
|
||||||
|
"prior_freshness": (evidence.get("freshness") or {}).get("status"),
|
||||||
|
"replacement_task_session_id": new_task_session_id,
|
||||||
|
"preserved_head": evidence.get("observed_head"),
|
||||||
|
"branch_reset": False,
|
||||||
|
"base_equivalence_required": False,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def format_recovery_refusal(assessment: Mapping[str, Any]) -> str:
|
||||||
|
reasons = "; ".join(
|
||||||
|
assessment.get("reasons") or ["unknown bootstrap lock recovery refusal"]
|
||||||
|
)
|
||||||
|
code = assessment.get("refusal_code") or "refused"
|
||||||
|
return f"Bootstrap lock recovery refused ({code}): {reasons} (fail closed)"
|
||||||
+252
-7
@@ -27,12 +27,12 @@ import uuid
|
|||||||
from contextlib import contextmanager
|
from contextlib import contextmanager
|
||||||
from dataclasses import dataclass
|
from dataclasses import dataclass
|
||||||
from datetime import datetime, timedelta, timezone
|
from datetime import datetime, timedelta, timezone
|
||||||
from typing import Any, Iterator, Sequence
|
from typing import Any, Iterator, Mapping, Sequence
|
||||||
|
|
||||||
import dependency_graph
|
import dependency_graph
|
||||||
import gitea_audit
|
import gitea_audit
|
||||||
|
|
||||||
SCHEMA_VERSION = 5
|
SCHEMA_VERSION = 6
|
||||||
|
|
||||||
# Assignable work kinds only — raw monitoring incidents are never work items.
|
# Assignable work kinds only — raw monitoring incidents are never work items.
|
||||||
WORK_KINDS = frozenset({"issue", "pr"})
|
WORK_KINDS = frozenset({"issue", "pr"})
|
||||||
@@ -419,6 +419,7 @@ class ControlPlaneDB:
|
|||||||
self._migrate_incident_links_null_scope(conn)
|
self._migrate_incident_links_null_scope(conn)
|
||||||
self._migrate_lease_lifecycle_columns(conn)
|
self._migrate_lease_lifecycle_columns(conn)
|
||||||
self._migrate_session_ownership_columns(conn)
|
self._migrate_session_ownership_columns(conn)
|
||||||
|
self._migrate_session_lifecycle_columns(conn)
|
||||||
self._migrate_usage_events_table(conn)
|
self._migrate_usage_events_table(conn)
|
||||||
conn.execute(
|
conn.execute(
|
||||||
"INSERT OR REPLACE INTO schema_meta(key, value) VALUES (?, ?)",
|
"INSERT OR REPLACE INTO schema_meta(key, value) VALUES (?, ?)",
|
||||||
@@ -812,6 +813,7 @@ class ControlPlaneDB:
|
|||||||
pid: int | None = None,
|
pid: int | None = None,
|
||||||
status: str = "active",
|
status: str = "active",
|
||||||
controller_instance_id: str | None = None,
|
controller_instance_id: str | None = None,
|
||||||
|
owner_process_started_at: str | None = None,
|
||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""Register/refresh a session row.
|
"""Register/refresh a session row.
|
||||||
|
|
||||||
@@ -821,16 +823,22 @@ class ControlPlaneDB:
|
|||||||
controller instance can. It is never overwritten with ``None``, so a
|
controller instance can. It is never overwritten with ``None``, so a
|
||||||
heartbeat from a caller that does not supply one cannot erase
|
heartbeat from a caller that does not supply one cannot erase
|
||||||
ownership.
|
ownership.
|
||||||
|
|
||||||
|
*owner_process_started_at* (#969) is the OS start time of the owner
|
||||||
|
process at registration time. When present it is retained across
|
||||||
|
heartbeats (never cleared by ``None``) so later PID-reuse checks do
|
||||||
|
not depend on a live ``ps`` probe of a long-dead process.
|
||||||
"""
|
"""
|
||||||
now = _ts()
|
now = _ts()
|
||||||
instance = (controller_instance_id or "").strip() or None
|
instance = (controller_instance_id or "").strip() or None
|
||||||
|
proc_start = (owner_process_started_at or "").strip() or None
|
||||||
with self._tx() as conn:
|
with self._tx() as conn:
|
||||||
existing = conn.execute(
|
existing = conn.execute(
|
||||||
"SELECT session_id FROM sessions WHERE session_id = ?",
|
"SELECT session_id FROM sessions WHERE session_id = ?",
|
||||||
(session_id,),
|
(session_id,),
|
||||||
).fetchone()
|
).fetchone()
|
||||||
if existing:
|
if existing:
|
||||||
if instance is None:
|
if instance is None and proc_start is None:
|
||||||
conn.execute(
|
conn.execute(
|
||||||
"""
|
"""
|
||||||
UPDATE sessions
|
UPDATE sessions
|
||||||
@@ -840,7 +848,21 @@ class ControlPlaneDB:
|
|||||||
""",
|
""",
|
||||||
(role, profile, namespace, pid, now, status, session_id),
|
(role, profile, namespace, pid, now, status, session_id),
|
||||||
)
|
)
|
||||||
else:
|
elif instance is None:
|
||||||
|
conn.execute(
|
||||||
|
"""
|
||||||
|
UPDATE sessions
|
||||||
|
SET role = ?, profile = ?, namespace = ?, pid = ?,
|
||||||
|
last_heartbeat_at = ?, status = ?,
|
||||||
|
owner_process_started_at = COALESCE(?, owner_process_started_at)
|
||||||
|
WHERE session_id = ?
|
||||||
|
""",
|
||||||
|
(
|
||||||
|
role, profile, namespace, pid, now, status,
|
||||||
|
proc_start, session_id,
|
||||||
|
),
|
||||||
|
)
|
||||||
|
elif proc_start is None:
|
||||||
conn.execute(
|
conn.execute(
|
||||||
"""
|
"""
|
||||||
UPDATE sessions
|
UPDATE sessions
|
||||||
@@ -854,18 +876,33 @@ class ControlPlaneDB:
|
|||||||
instance, session_id,
|
instance, session_id,
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
|
else:
|
||||||
|
conn.execute(
|
||||||
|
"""
|
||||||
|
UPDATE sessions
|
||||||
|
SET role = ?, profile = ?, namespace = ?, pid = ?,
|
||||||
|
last_heartbeat_at = ?, status = ?,
|
||||||
|
controller_instance_id = ?,
|
||||||
|
owner_process_started_at = COALESCE(?, owner_process_started_at)
|
||||||
|
WHERE session_id = ?
|
||||||
|
""",
|
||||||
|
(
|
||||||
|
role, profile, namespace, pid, now, status,
|
||||||
|
instance, proc_start, session_id,
|
||||||
|
),
|
||||||
|
)
|
||||||
else:
|
else:
|
||||||
conn.execute(
|
conn.execute(
|
||||||
"""
|
"""
|
||||||
INSERT INTO sessions(
|
INSERT INTO sessions(
|
||||||
session_id, role, profile, namespace, pid,
|
session_id, role, profile, namespace, pid,
|
||||||
started_at, last_heartbeat_at, status,
|
started_at, last_heartbeat_at, status,
|
||||||
controller_instance_id
|
controller_instance_id, owner_process_started_at
|
||||||
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
|
) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
||||||
""",
|
""",
|
||||||
(
|
(
|
||||||
session_id, role, profile, namespace, pid, now, now,
|
session_id, role, profile, namespace, pid, now, now,
|
||||||
status, instance,
|
status, instance, proc_start,
|
||||||
),
|
),
|
||||||
)
|
)
|
||||||
row = conn.execute(
|
row = conn.execute(
|
||||||
@@ -881,6 +918,165 @@ class ControlPlaneDB:
|
|||||||
(_ts(), session_id),
|
(_ts(), session_id),
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def retire_session(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
session_id: str,
|
||||||
|
reason: str,
|
||||||
|
actor_session_id: str | None = None,
|
||||||
|
details: Mapping[str, Any] | None = None,
|
||||||
|
now: datetime | None = None,
|
||||||
|
terminal_status: str = "retired",
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Terminalize one session row if it is still non-terminal (#969).
|
||||||
|
|
||||||
|
CAS on non-terminal status: concurrent retirements of the same row
|
||||||
|
yield exactly one ``retired`` outcome and subsequent
|
||||||
|
``already_terminal`` outcomes. Never deletes historical rows. Writes a
|
||||||
|
durable ``session_retired`` event for audit.
|
||||||
|
"""
|
||||||
|
moment = _ts(now)
|
||||||
|
reason_s = (reason or "").strip() or "unspecified"
|
||||||
|
term = (terminal_status or "retired").strip().lower() or "retired"
|
||||||
|
terminal_set = {
|
||||||
|
"retired",
|
||||||
|
"ended",
|
||||||
|
"terminal",
|
||||||
|
"dead",
|
||||||
|
"stale",
|
||||||
|
"orphaned",
|
||||||
|
}
|
||||||
|
with self._tx() as conn:
|
||||||
|
row = conn.execute(
|
||||||
|
"SELECT * FROM sessions WHERE session_id = ?",
|
||||||
|
(session_id,),
|
||||||
|
).fetchone()
|
||||||
|
if row is None:
|
||||||
|
return {
|
||||||
|
"outcome": "missing",
|
||||||
|
"reason": reason_s,
|
||||||
|
"prior_status": None,
|
||||||
|
"new_status": None,
|
||||||
|
"details": {"session_id": session_id},
|
||||||
|
}
|
||||||
|
prior = dict(row)
|
||||||
|
prior_status = str(prior.get("status") or "").strip().lower()
|
||||||
|
if prior_status in terminal_set:
|
||||||
|
return {
|
||||||
|
"outcome": "already_terminal",
|
||||||
|
"reason": reason_s,
|
||||||
|
"prior_status": prior_status,
|
||||||
|
"new_status": prior_status,
|
||||||
|
"details": {"session_id": session_id, "idempotent": True},
|
||||||
|
}
|
||||||
|
|
||||||
|
# Optional: refuse when an active lease still names this session.
|
||||||
|
active_lease = conn.execute(
|
||||||
|
"""
|
||||||
|
SELECT lease_id, status, expires_at FROM leases
|
||||||
|
WHERE session_id = ? AND status = 'active'
|
||||||
|
LIMIT 1
|
||||||
|
""",
|
||||||
|
(session_id,),
|
||||||
|
).fetchone()
|
||||||
|
if active_lease is not None:
|
||||||
|
return {
|
||||||
|
"outcome": "blocked",
|
||||||
|
"reason": "live_lease",
|
||||||
|
"prior_status": prior_status,
|
||||||
|
"new_status": prior_status,
|
||||||
|
"details": {
|
||||||
|
"session_id": session_id,
|
||||||
|
"lease_id": active_lease["lease_id"],
|
||||||
|
"blocker": "active_lease_row",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
cols = {
|
||||||
|
r[1] for r in conn.execute("PRAGMA table_info(sessions)").fetchall()
|
||||||
|
}
|
||||||
|
if "retired_at" in cols and "retire_reason" in cols:
|
||||||
|
conn.execute(
|
||||||
|
"""
|
||||||
|
UPDATE sessions
|
||||||
|
SET status = ?, retired_at = ?, retire_reason = ?
|
||||||
|
WHERE session_id = ? AND status = ?
|
||||||
|
""",
|
||||||
|
(term, moment, reason_s, session_id, prior.get("status")),
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
conn.execute(
|
||||||
|
"""
|
||||||
|
UPDATE sessions
|
||||||
|
SET status = ?
|
||||||
|
WHERE session_id = ? AND status = ?
|
||||||
|
""",
|
||||||
|
(term, session_id, prior.get("status")),
|
||||||
|
)
|
||||||
|
changed = conn.execute(
|
||||||
|
"SELECT changes()"
|
||||||
|
).fetchone()[0]
|
||||||
|
if not changed:
|
||||||
|
# Lost CAS race — re-read.
|
||||||
|
refreshed = conn.execute(
|
||||||
|
"SELECT status FROM sessions WHERE session_id = ?",
|
||||||
|
(session_id,),
|
||||||
|
).fetchone()
|
||||||
|
cur = (
|
||||||
|
str(refreshed["status"]).strip().lower()
|
||||||
|
if refreshed is not None
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"outcome": "already_terminal"
|
||||||
|
if cur in terminal_set
|
||||||
|
else "blocked",
|
||||||
|
"reason": reason_s,
|
||||||
|
"prior_status": prior_status,
|
||||||
|
"new_status": cur,
|
||||||
|
"details": {
|
||||||
|
"session_id": session_id,
|
||||||
|
"cas_lost": True,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
detail_payload: dict[str, Any] = {
|
||||||
|
"session_id": session_id,
|
||||||
|
"prior_status": prior_status,
|
||||||
|
"new_status": term,
|
||||||
|
"reason": reason_s,
|
||||||
|
"actor_session_id": actor_session_id,
|
||||||
|
"pid": prior.get("pid"),
|
||||||
|
"role": prior.get("role"),
|
||||||
|
"profile": prior.get("profile"),
|
||||||
|
}
|
||||||
|
if isinstance(details, Mapping):
|
||||||
|
for key, value in details.items():
|
||||||
|
if key not in detail_payload:
|
||||||
|
detail_payload[key] = value
|
||||||
|
message = (
|
||||||
|
f"session {session_id} retired reason={reason_s} "
|
||||||
|
f"prior_status={prior_status}"
|
||||||
|
)
|
||||||
|
# events.work_item_id is nullable; session retirement is not work-scoped.
|
||||||
|
conn.execute(
|
||||||
|
"""
|
||||||
|
INSERT INTO events(work_item_id, event_type, message, created_at)
|
||||||
|
VALUES (NULL, 'session_retired', ?, ?)
|
||||||
|
""",
|
||||||
|
(message[:2000], moment),
|
||||||
|
)
|
||||||
|
# Also persist a structured JSON line in the message when short enough
|
||||||
|
# by appending a compact summary (full detail stays in return value /
|
||||||
|
# audit log; events.message is human-readable).
|
||||||
|
return {
|
||||||
|
"outcome": "retired",
|
||||||
|
"reason": reason_s,
|
||||||
|
"prior_status": prior_status,
|
||||||
|
"new_status": term,
|
||||||
|
"details": detail_payload,
|
||||||
|
}
|
||||||
|
|
||||||
def list_sessions(
|
def list_sessions(
|
||||||
self,
|
self,
|
||||||
*,
|
*,
|
||||||
@@ -1568,6 +1764,32 @@ class ControlPlaneDB:
|
|||||||
).fetchone()
|
).fetchone()
|
||||||
return dict(row) if row else None
|
return dict(row) if row else None
|
||||||
|
|
||||||
|
def list_incident_links(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
provider: str | None = None,
|
||||||
|
gitea_org: str | None = None,
|
||||||
|
gitea_repo: str | None = None,
|
||||||
|
limit: int = 100,
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
"""List stored incident_links rows, optionally filtered by provider/repo (#612 / #649)."""
|
||||||
|
query = "SELECT * FROM incident_links WHERE 1=1"
|
||||||
|
params: list[Any] = []
|
||||||
|
if provider:
|
||||||
|
query += " AND provider = ?"
|
||||||
|
params.append(provider.strip().lower())
|
||||||
|
if gitea_org:
|
||||||
|
query += " AND gitea_org = ?"
|
||||||
|
params.append(_norm_scope(gitea_org))
|
||||||
|
if gitea_repo:
|
||||||
|
query += " AND gitea_repo = ?"
|
||||||
|
params.append(_norm_scope(gitea_repo))
|
||||||
|
query += " ORDER BY link_id DESC LIMIT ?"
|
||||||
|
params.append(max(1, limit))
|
||||||
|
with self._tx(immediate=False) as conn:
|
||||||
|
rows = conn.execute(query, params).fetchall()
|
||||||
|
return [dict(r) for r in rows]
|
||||||
|
|
||||||
|
|
||||||
# ── lease lifecycle (#601) ────────────────────────────────────────────
|
# ── lease lifecycle (#601) ────────────────────────────────────────────
|
||||||
|
|
||||||
@@ -1613,6 +1835,29 @@ class ControlPlaneDB:
|
|||||||
if name not in cols:
|
if name not in cols:
|
||||||
conn.execute(f"ALTER TABLE sessions ADD COLUMN {name} {decl}")
|
conn.execute(f"ALTER TABLE sessions ADD COLUMN {name} {decl}")
|
||||||
|
|
||||||
|
_SESSION_LIFECYCLE_COLUMNS: tuple[tuple[str, str], ...] = (
|
||||||
|
("owner_process_started_at", "TEXT"),
|
||||||
|
("retired_at", "TEXT"),
|
||||||
|
("retire_reason", "TEXT"),
|
||||||
|
)
|
||||||
|
|
||||||
|
def _migrate_session_lifecycle_columns(self, conn: sqlite3.Connection) -> None:
|
||||||
|
"""Add session retirement / PID-reuse provenance columns (#969).
|
||||||
|
|
||||||
|
Additive and idempotent. Pre-existing rows migrate with NULL; retirement
|
||||||
|
fills ``retired_at`` / ``retire_reason``, and new upserts may record
|
||||||
|
``owner_process_started_at`` for stronger identity checks.
|
||||||
|
"""
|
||||||
|
cols = {
|
||||||
|
row[1]
|
||||||
|
for row in conn.execute("PRAGMA table_info(sessions)").fetchall()
|
||||||
|
}
|
||||||
|
if not cols:
|
||||||
|
return
|
||||||
|
for name, decl in self._SESSION_LIFECYCLE_COLUMNS:
|
||||||
|
if name not in cols:
|
||||||
|
conn.execute(f"ALTER TABLE sessions ADD COLUMN {name} {decl}")
|
||||||
|
|
||||||
def list_active_claims(
|
def list_active_claims(
|
||||||
self,
|
self,
|
||||||
*,
|
*,
|
||||||
|
|||||||
@@ -241,9 +241,14 @@ def bootstrap_permits_control_checkout(
|
|||||||
caller's ordinary block in force.
|
caller's ordinary block in force.
|
||||||
|
|
||||||
``assessment`` is server-derived only: it is produced by
|
``assessment`` is server-derived only: it is produced by
|
||||||
:func:`assess_create_issue_bootstrap` from inspected repository state. It is
|
:func:`assess_create_issue_bootstrap` or
|
||||||
never accepted from an MCP tool argument, so no caller can assert
|
:func:`author_issue_bootstrap.assess_author_issue_bootstrap` from inspected
|
||||||
eligibility it has not proven.
|
repository state. It is never accepted from an MCP tool argument, so no
|
||||||
|
caller can assert eligibility it has not proven.
|
||||||
|
|
||||||
|
#892: author issue worktree bootstrap uses the same predicate with
|
||||||
|
``task_scope='author_issue_bootstrap'`` so a clean control checkout can
|
||||||
|
create the first ``branches/`` worktree without the lock↔worktree cycle.
|
||||||
"""
|
"""
|
||||||
if not isinstance(assessment, dict):
|
if not isinstance(assessment, dict):
|
||||||
return False
|
return False
|
||||||
@@ -264,9 +269,16 @@ def bootstrap_permits_control_checkout(
|
|||||||
if assessment.get("reasons"):
|
if assessment.get("reasons"):
|
||||||
return False
|
return False
|
||||||
|
|
||||||
# Scope proof: only the create_issue bootstrap, only via the clean
|
# Scope proof: create_issue (#749) or author issue bootstrap (#850/#892),
|
||||||
# canonical control checkout path.
|
# only via the clean canonical control checkout path.
|
||||||
if assessment.get("task_scope") != "create_issue_only":
|
task_scope = assessment.get("task_scope")
|
||||||
|
if is_create_issue_task(task):
|
||||||
|
if task_scope != "create_issue_only":
|
||||||
|
return False
|
||||||
|
elif author_issue_bootstrap.is_author_issue_bootstrap_task(task):
|
||||||
|
if task_scope != "author_issue_bootstrap":
|
||||||
|
return False
|
||||||
|
else:
|
||||||
return False
|
return False
|
||||||
if assessment.get("bootstrap_path") != "clean_canonical_control_checkout":
|
if assessment.get("bootstrap_path") != "clean_canonical_control_checkout":
|
||||||
return False
|
return False
|
||||||
|
|||||||
@@ -0,0 +1,167 @@
|
|||||||
|
# ADR: High-availability and rolling-restart architecture for Gitea MCP control plane
|
||||||
|
|
||||||
|
- **Status:** Proposed (Design ADR under [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668))
|
||||||
|
- **Date:** 2026-07-25
|
||||||
|
- **Tracking Issue:** [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668)
|
||||||
|
- **Policy Version:** `mcp-ha-rolling-restart/v1`
|
||||||
|
- **Related:**
|
||||||
|
- Parent: [#655](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/655) — Governed MCP restart coordination and zero-disruption recovery
|
||||||
|
- Governance Policy: [#656](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/656) / `docs/architecture/mcp-restart-governance.md`
|
||||||
|
- Control-Plane DB Substrate: [#613](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/613) / `docs/architecture/control-plane-db-substrate.md`
|
||||||
|
- Runtime Policy: [#615](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/615) / `docs/architecture/mcp-stable-control-runtime-policy-adr.md`
|
||||||
|
- Product Vision: [#652](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/652) (Phase 5 Maturity)
|
||||||
|
- Delivery Roadmap: [#653](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/653)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 1. Context & Problem Statement
|
||||||
|
|
||||||
|
The Gitea MCP server operates as the authoritative **control plane** for managing issues, Pull Requests, code mutations, formal reviews, and workflow reconciliations. Under single-process governance ([#656](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/656)), process restarts are strictly controlled using pre-flight checks, drain phases, and operator approvals.
|
||||||
|
|
||||||
|
However, a single-instance control plane inherently presents fundamental constraints:
|
||||||
|
|
||||||
|
1. **Downtime during updates:** Even a perfectly executed single-process drain requires a window where incoming client requests must be paused or rejected while the server binary or python environment reloads.
|
||||||
|
2. **Single point of failure:** Infrastructure issues, process crashes, or unhandled host-level terminations immediately disconnect active LLM sessions and leave transient workflows incomplete.
|
||||||
|
3. **Multi-agent concurrency bottlenecks:** High volumes of concurrent multi-LLM tasks put all lock management, lease allocation, and Gitea API interactions through a single process event loop.
|
||||||
|
|
||||||
|
To achieve true zero-disruption operation and seamless rolling deployments without stopping active work, the system requires a high-availability (HA), multi-instance MCP architecture.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 2. Architectural Principles & Non-Goals
|
||||||
|
|
||||||
|
### 2.1 Core Architectural Principles
|
||||||
|
* **Gitea as Canonical Work SoT:** Gitea remains the ultimate System of Record (SoT) for issue states, pull requests, labels, and audit comments. The MCP control plane does not duplicate domain entities.
|
||||||
|
* **Control-Plane DB as Multi-Instance State Substrate:** The control-plane SQLite/durable database ([#613](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/613)) acts as the single source of truth for workflow leases, session tokens, assignment records, and lock fences across all MCP nodes.
|
||||||
|
* **Stateless Worker Nodes:** MCP role server processes (`gitea-author`, `gitea-reviewer`, `gitea-merger`, `gitea-reconciler`, `gitea-controller`) maintain no unique in-memory state; any node can handle any request given a valid session resume token.
|
||||||
|
* **Fail-Closed Split-Brain Defense:** In any network partition or quorum loss scenario, nodes must fail closed rather than risk double-mutations or conflicting Gitea states.
|
||||||
|
|
||||||
|
### 2.2 Non-Goals
|
||||||
|
* **Replacing Gitea:** We do not replace Gitea issue/PR tracking with an independent database.
|
||||||
|
* **Immediate Multi-Node Cluster Execution in v1:** This ADR defines the target architecture and phased roadmap; immediate implementation occurs incrementally post-[#655] v1.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 3. High-Availability & Rolling-Restart Architecture
|
||||||
|
|
||||||
|
### 3.1 Architecture Overview
|
||||||
|
|
||||||
|
```
|
||||||
|
+----------------------------+
|
||||||
|
| LLM Clients / IDE Sessions |
|
||||||
|
+--------------+-------------+
|
||||||
|
|
|
||||||
|
v
|
||||||
|
+----------------------------+
|
||||||
|
| HA Proxy / Router |
|
||||||
|
| (Health-based & Affinity) |
|
||||||
|
+------+--------------+------+
|
||||||
|
| |
|
||||||
|
+--------------+ +--------------+
|
||||||
|
v v
|
||||||
|
+--------------------+ +--------------------+
|
||||||
|
| MCP Instance Node A| | MCP Instance Node B|
|
||||||
|
| (Version N) | | (Version N+1) |
|
||||||
|
+---------+----------+ +---------+----------+
|
||||||
|
| |
|
||||||
|
+----------------------+----------------------+
|
||||||
|
|
|
||||||
|
v
|
||||||
|
+----------------------------+
|
||||||
|
| Control-Plane DB Substrate|
|
||||||
|
| (Shared Lease & Locks) |
|
||||||
|
+--------------+-------------+
|
||||||
|
|
|
||||||
|
v
|
||||||
|
+----------------------------+
|
||||||
|
| Gitea API |
|
||||||
|
+----------------------------+
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
### 3.2 Key System Components
|
||||||
|
|
||||||
|
#### A. Multiple MCP Instance Cohorts
|
||||||
|
* The control plane runs across $N \ge 2$ redundant process nodes.
|
||||||
|
* Dual-namespace deployment allows running the old version (Node A) alongside a updated version (Node B) during rolling upgrades.
|
||||||
|
|
||||||
|
#### B. Shared Durable Session Storage & Resume Tokens
|
||||||
|
* Session context, preflight verification proofs, and capability resolution states are stored in the shared control-plane database.
|
||||||
|
* Client requests carry an explicit `session_id` and `resume_token`. If an MCP instance restarts or a request routes to a different instance, the target node validates the token against the database without requiring full session re-initialization.
|
||||||
|
|
||||||
|
#### C. Shared Lease Authority & Fencing Counters
|
||||||
|
* Workflow leases (`gitea_allocate_next_work`, `gitea_adopt_workflow_lease`) use monotonic fencing tokens (`lease_generation_id`).
|
||||||
|
* When Node B acquires or renews a lease, it increments the generation counter. Any delayed or out-of-order write attempt from Node A using an older generation token is rejected by database constraints.
|
||||||
|
|
||||||
|
#### D. Leader Election & Coordinated Drain
|
||||||
|
* Node clusters elect a primary coordinator node for administrative background tasks (such as stale lease cleanup or incident Watchdogs).
|
||||||
|
* During a rolling deployment:
|
||||||
|
1. Node B (new version) is launched and registers as healthy.
|
||||||
|
2. Router directs new session creations to Node B.
|
||||||
|
3. Node A enters `MAINTENANCE_DRAIN` status ([#659](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/659)), completing in-flight mutations while refusing new tasks.
|
||||||
|
4. Once all active sessions migrate or complete, Node A shuts down cleanly.
|
||||||
|
|
||||||
|
#### E. Idempotent Mutations & Failover Safety
|
||||||
|
* All state-changing tool executions (PR creation, review submission, merge operations, label changes) carry a deterministic `idempotency_key`.
|
||||||
|
* If a network connection flaps or a node fails mid-mutation, the re-issued request with the same `idempotency_key` is recognized by the control-plane substrate, returning the existing recorded result without repeating side effects on Gitea.
|
||||||
|
|
||||||
|
#### F. Schema Version Compatibility
|
||||||
|
* Database migrations follow non-breaking additive patterns.
|
||||||
|
* During rolling upgrades where Node A (Version $N$) and Node B (Version $N+1$) run concurrently, both versions operate against the shared schema without structural conflicts.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 4. Split-Brain & Failure Behavior
|
||||||
|
|
||||||
|
### 4.1 Split-Brain Risk Scenarios & Mitigation
|
||||||
|
|
||||||
|
| Scenario | Risk | Mitigation Strategy |
|
||||||
|
|---|---|---|
|
||||||
|
| **Network Partition between Nodes** | Both Node A and Node B attempt to process operations for the same issue/PR. | **Generation Fencing:** Lease renewal requires updating the DB generation counter. The node isolated from the DB fails closed immediately. |
|
||||||
|
| **Stale Node Recovery** | Node A recovers after a long pause and executes a queued mutation. | **Lease Expiry & TTL Fencing:** Transactions verify that `expires_at > NOW()` within the atomic SQLite transaction boundaries. |
|
||||||
|
| **Database Connection Loss** | Node loses access to shared control-plane DB substrate. | **Strict Fail-Closed:** The node immediately marks all task capabilities as `blocked` and rejects mutation tools until DB connectivity is re-established. |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 5. Phased Implementation Milestones
|
||||||
|
|
||||||
|
```mermaid
|
||||||
|
flowchart TD
|
||||||
|
M1[Milestone 1: Shared Control-Plane DB Schema & Resume Tokens] --> M2[Milestone 2: Idempotent Mutation Layer]
|
||||||
|
M2 --> M3[Milestone 3: Health Routing & Standby Failover]
|
||||||
|
M3 --> M4[Milestone 4: Active-Active Rolling Deployment & Auto-Drain]
|
||||||
|
```
|
||||||
|
|
||||||
|
### Milestone 1: Shared Control-Plane DB Schema & Resume Tokens (Post-#655)
|
||||||
|
* Extend [#613] Control-Plane DB schema to store multi-instance node heartbeat records and session resume tokens.
|
||||||
|
* Enable session lookup across instances via `session_id`.
|
||||||
|
|
||||||
|
### Milestone 2: Idempotent Mutation Layer & Lease Fencing
|
||||||
|
* Add mandatory `idempotency_key` tracking to all Gitea mutation tools.
|
||||||
|
* Implement monotonic lease fencing counters in `gitea_allocate_next_work` and `gitea_adopt_workflow_lease`.
|
||||||
|
|
||||||
|
### Milestone 3: Health-Based Routing & Active-Passive Standby
|
||||||
|
* Introduce lightweight proxy/router capable of checking node health endpoints.
|
||||||
|
* Implement active-standby failover where standby node automatically assumes work if active node fails health checks.
|
||||||
|
|
||||||
|
### Milestone 4: Active-Active Horizontal Deployment & Rolling Upgrade Automation
|
||||||
|
* Enable true active-active multi-instance execution.
|
||||||
|
* Integrate automated zero-downtime rolling upgrades coordinated with `gitea_request_mcp_restart` maintenance drain.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 6. Observability & Audit Requirements
|
||||||
|
|
||||||
|
High-availability control plane operations must expose clear telemetry and audit trails:
|
||||||
|
|
||||||
|
* **Node Registry Telemetry:** Active nodes, version numbers, uptime, and heartbeat timestamps reported via `gitea_get_runtime_context`.
|
||||||
|
* **Lease Fencing Metrics:** Tracking lease acquire latency, fence rejection counts, and lease handoff durations.
|
||||||
|
* **Failover & Re-route Audit Logs:** Durable logging of session migrations between nodes, drain initiation, and process retirement events.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 7. Tradeoffs & Accepted Risks
|
||||||
|
|
||||||
|
* **Increased Architectural Complexity:** Moving from a single process to a multi-instance control plane requires robust DB locking, proxy routing, and migration governance.
|
||||||
|
* **Database Dependency:** The control-plane database substrate becomes a critical shared dependency for multi-node deployments. High availability for the underlying SQLite file system / DB must be guaranteed.
|
||||||
@@ -0,0 +1,209 @@
|
|||||||
|
# The canonical author issue-lock contract (#953)
|
||||||
|
|
||||||
|
Every author issue lock has exactly one shape. Both writers —
|
||||||
|
`gitea_bootstrap_author_issue_worktree` and `gitea_lock_issue` — build it
|
||||||
|
through `author_lock_contract.build_canonical_issue_lock`, and every reader
|
||||||
|
consumes that same shape.
|
||||||
|
|
||||||
|
Before #953 the two writers disagreed. `gitea_lock_issue` wrote the canonical
|
||||||
|
record; bootstrap wrote a thinner one with the claimant at the lock top level,
|
||||||
|
`lease_id: null`, and no `work_lease`, `lock_provenance`, or expiry. Because
|
||||||
|
every reader was written against the canonical shape, a lock that bootstrap
|
||||||
|
reported as successfully created could not be heartbeated, renewed, re-locked,
|
||||||
|
or accepted by `gitea_create_pr`. Each of those gates was individually correct;
|
||||||
|
the defect was that two writers disagreed about what a lock *is*.
|
||||||
|
|
||||||
|
## Required ordering
|
||||||
|
|
||||||
|
**Finalize the lock before writing any implementation bytes.** This ordering is
|
||||||
|
what keeps recovery cheap: while the worktree is still base-equivalent, a lock
|
||||||
|
problem can be fixed by simply calling `gitea_lock_issue` again. Once the branch
|
||||||
|
carries commits, base-equivalence is gone and the ordinary re-lock path is no
|
||||||
|
longer available.
|
||||||
|
|
||||||
|
1. `gitea_whoami` — resolve identity and profile.
|
||||||
|
2. `gitea_resolve_task_capability(task='work_issue')`.
|
||||||
|
3. `gitea_bootstrap_author_issue_worktree` — creates the branch, the registered
|
||||||
|
worktree under `branches/`, and a **canonical** lock. It reads the lock back
|
||||||
|
and verifies it structurally before reporting success; a partial lock fails
|
||||||
|
closed here, with the missing fields named, and never reports
|
||||||
|
`implementation_allowed: true`.
|
||||||
|
4. `gitea_heartbeat_issue_lock` — prove the lock is usable, using the
|
||||||
|
`task_session_id` bootstrap returned.
|
||||||
|
5. Implement, commit, push.
|
||||||
|
6. `gitea_create_pr`.
|
||||||
|
|
||||||
|
If bootstrap returns `success: false` with
|
||||||
|
`reason_code: incomplete_issue_lock_contract`, **do not implement**. Its
|
||||||
|
`exact_next_action` names the executable recovery step. Bootstrap's reported
|
||||||
|
next action always matches the state it actually returned.
|
||||||
|
|
||||||
|
### What that refusal leaves behind
|
||||||
|
|
||||||
|
The AC7 refusal runs `run_compensating_recovery` *before* it reports, so the
|
||||||
|
advice has to describe the post-rollback state rather than the shape of the lock
|
||||||
|
that provoked it. Recommending incomplete-lock recovery for artifacts the
|
||||||
|
rollback already deleted would produce `no_durable_lock` and then
|
||||||
|
`worktree_invalid` — two refusals for a state a plain retry fixes.
|
||||||
|
|
||||||
|
The refusal therefore carries `compensating_recovery` and
|
||||||
|
`post_compensation_state`, and derives `exact_next_action` from what was
|
||||||
|
observed on disk. `cleanup_state` is one of:
|
||||||
|
|
||||||
|
| `cleanup_state` | Meaning | Next action |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `complete` | lock, branch, and worktree all removed | resolve `missing_fields` and re-run `gitea_bootstrap_author_issue_worktree` |
|
||||||
|
| `partial` | rollback ran; some artifacts survive, by design or because a step errored | scoped to exactly what survives — see below |
|
||||||
|
| `failed` | rollback never completed, so nothing is proven removed | `gitea_inspect_issue_lock_contract` (read-only) before anything else |
|
||||||
|
|
||||||
|
Within `partial`, the surviving set decides the action:
|
||||||
|
|
||||||
|
| Survives | Next action |
|
||||||
|
| --- | --- |
|
||||||
|
| lock + branch + worktree | `gitea_recover_incomplete_bootstrap_lock` for that exact issue, branch, and worktree |
|
||||||
|
| branch + worktree (lock released) | `gitea_lock_issue` — no implementation bytes were written, so the worktree is still base-equivalent |
|
||||||
|
| lock only (worktree removed) | `gitea_inspect_issue_lock_contract`; the surviving lock must be released by its recorded owner before bootstrap is retried |
|
||||||
|
| branch only | `gitea_inspect_issue_lock_contract`, then re-run bootstrap, which adopts the existing branch |
|
||||||
|
|
||||||
|
`failed_rollback_steps` names any rollback step that errored, and the returned
|
||||||
|
action says so rather than presenting the surviving state as intentional.
|
||||||
|
|
||||||
|
> The lock half of that rollback was dead code until #953 review 632 F2:
|
||||||
|
> `run_compensating_recovery` called `issue_lock_store.release_session_lock`,
|
||||||
|
> which did not exist, inside a bare `except Exception: pass`. Every rollback
|
||||||
|
> removed the branch and worktree and silently left the lock — the exact
|
||||||
|
> uninspectable, unrecoverable state this issue exists to eliminate. The
|
||||||
|
> function now exists, releases only a lock whose recorded `owner_session`
|
||||||
|
> matches, and its failures are recorded rather than swallowed.
|
||||||
|
|
||||||
|
## The contract
|
||||||
|
|
||||||
|
A canonical lock carries every field in
|
||||||
|
`author_lock_contract.REQUIRED_LOCK_FIELDS`:
|
||||||
|
|
||||||
|
| Field | Meaning |
|
||||||
|
| --- | --- |
|
||||||
|
| `remote`, `org`, `repo`, `issue_number` | repository and issue identity |
|
||||||
|
| `branch_name`, `worktree_path` | the binding this claim owns |
|
||||||
|
| `work_lease` | the canonical lease block, below |
|
||||||
|
| `lock_provenance` | sanctioned source, minted server-side |
|
||||||
|
| `lock_generation` | monotonic; every write advances it |
|
||||||
|
|
||||||
|
`work_lease` carries every field in
|
||||||
|
`author_lock_contract.REQUIRED_WORK_LEASE_FIELDS`, notably:
|
||||||
|
|
||||||
|
| Field | Meaning |
|
||||||
|
| --- | --- |
|
||||||
|
| `claimant.{username,profile}` | **canonical** claimant placement |
|
||||||
|
| `expires_at` | sliding TTL from `lease_policy` |
|
||||||
|
| `last_heartbeat_at`, `heartbeat_count` | liveness evidence |
|
||||||
|
| `task_session_id` | the ownership fencing token — never null |
|
||||||
|
| `lifecycle_version` | `heartbeat-v1`; its absence is what makes a lock legacy |
|
||||||
|
|
||||||
|
### Claimant placement and legacy compatibility
|
||||||
|
|
||||||
|
`work_lease.claimant` is canonical. A top-level `claimant` is the legacy
|
||||||
|
placement written by pre-#953 bootstrap and is still **read** — through the one
|
||||||
|
shared reader, `issue_lock_store.lock_claimant` — so an existing lock is not
|
||||||
|
refused for "not recording a claimant" when it plainly records one.
|
||||||
|
|
||||||
|
Tolerating the placement is not a widening. Every caller still compares the
|
||||||
|
values against server-resolved identity and profile, so a legacy placement
|
||||||
|
grants nothing the canonical placement would not. When both are present, the
|
||||||
|
`work_lease` copy wins: after an upgrade, a stale top-level copy must never
|
||||||
|
decide ownership.
|
||||||
|
|
||||||
|
### Expiration is explicit
|
||||||
|
|
||||||
|
A lock with no recorded expiry is **not** "not yet expired". `is_lease_expired`
|
||||||
|
returns `False` for it, which used to make such a lock permanently non-expiring
|
||||||
|
*and* permanently ineligible for #760 exact-owner renewal, which only ever
|
||||||
|
assesses an expired lease. `author_lock_contract.expiration_state` names the
|
||||||
|
real fact: `recorded`, `missing`, or `unparseable`. A `missing` expiry makes the
|
||||||
|
lock eligible for the recovery path below rather than stranding it.
|
||||||
|
|
||||||
|
## Recovering an existing incomplete bootstrap lock
|
||||||
|
|
||||||
|
For locks already written by the old bootstrap — including those whose branches
|
||||||
|
already carry legitimate committed and pushed work — use:
|
||||||
|
|
||||||
|
```text
|
||||||
|
gitea_inspect_issue_lock_contract(issue_number, branch_name, worktree_path, remote=...)
|
||||||
|
gitea_recover_incomplete_bootstrap_lock(issue_number, branch_name, worktree_path, expected_head, remote=...)
|
||||||
|
```
|
||||||
|
|
||||||
|
`gitea_inspect_issue_lock_contract` is strictly read-only: it performs no lock,
|
||||||
|
lease, branch, worktree, issue, or pull-request mutation. Use it first to see
|
||||||
|
which fields are missing and what the recommended action is; pass `dry_run=True`
|
||||||
|
to the recovery tool to preview the decision without writing.
|
||||||
|
|
||||||
|
`gitea_recover_incomplete_bootstrap_lock` upgrades that one lock to the
|
||||||
|
canonical contract. Before writing anything it verifies:
|
||||||
|
|
||||||
|
* repository (`remote`, `org`, `repo`) and issue number
|
||||||
|
* claimant username **and** profile against the server-resolved values — a
|
||||||
|
matching username alone is never accepted
|
||||||
|
* branch, worktree path, worktree existence, and worktree registration
|
||||||
|
* the worktree is on the recorded branch
|
||||||
|
* the observed head equals the caller's `expected_head`
|
||||||
|
* the existing lock's generation and provenance state
|
||||||
|
* the absence of healthy foreign ownership
|
||||||
|
|
||||||
|
What it deliberately does **not** do:
|
||||||
|
|
||||||
|
* it never moves, resets, or rewinds the branch, and never requires
|
||||||
|
base-equivalence — preserving the committed work is the entire point;
|
||||||
|
* it never pushes and never creates a pull request;
|
||||||
|
* it touches only the single lock file for that exact remote/org/repo/issue;
|
||||||
|
* it accepts no caller-supplied provenance and no caller-supplied authorization
|
||||||
|
flag — both are minted server-side.
|
||||||
|
|
||||||
|
A recovered lock records a `bootstrap_lock_recovery` block holding both sides of
|
||||||
|
the transition — prior contract, prior missing fields, prior generation, prior
|
||||||
|
owning session, the replacement `task_session_id`, and the preserved head — so a
|
||||||
|
recovered claim never reads as an original one.
|
||||||
|
|
||||||
|
### Gates, in order
|
||||||
|
|
||||||
|
`gitea_recover_incomplete_bootstrap_lock` is an author-only durable-lock
|
||||||
|
mutation and carries the same three gates as every comparable author operation,
|
||||||
|
in this order:
|
||||||
|
|
||||||
|
1. `role_session_router.check_author_mutation_after_reviewer_stop` — no author
|
||||||
|
fallback after a reviewer `wrong_role_stop`.
|
||||||
|
2. `_namespace_mutation_block(task, remote=remote, author_role_exclusive=True)` —
|
||||||
|
the namespace wall. It refuses a reviewer-bound session and, because this
|
||||||
|
task's required permission is `gitea.issue.comment` (which merger,
|
||||||
|
controller, and reconciler profiles also hold), additionally requires the
|
||||||
|
active profile's derived role kind to be exactly `author`. A refusal carries
|
||||||
|
`namespace_block: true` and emits a `BLOCKED` audit record naming the
|
||||||
|
namespace and profile.
|
||||||
|
3. `_profile_permission_block` — operation, provenance, and session-context
|
||||||
|
gates.
|
||||||
|
|
||||||
|
Exact-owner claimant matching inside `assess_bootstrap_lock_recovery` runs
|
||||||
|
*after* all three. It is a further layer, never a substitute for them: on its
|
||||||
|
own it refuses one step too late and leaves the audit trail silent about the
|
||||||
|
attempt.
|
||||||
|
|
||||||
|
### Refusals
|
||||||
|
|
||||||
|
| `refusal_code` | Meaning |
|
||||||
|
| --- | --- |
|
||||||
|
| `no_durable_lock` | nothing to recover |
|
||||||
|
| `already_canonical` | lock is fine; rewriting would invalidate a live heartbeat token |
|
||||||
|
| `foreign_claimant` | recorded claimant is not the active identity/profile pair |
|
||||||
|
| `healthy_foreign_lock` | a live foreign-owned lock; takeover is not a recovery path |
|
||||||
|
| `identity_unresolved` | identity or profile could not be resolved |
|
||||||
|
| `binding_mismatch` | repository, issue, branch, or worktree does not match |
|
||||||
|
| `worktree_invalid` | worktree missing, unregistered, or on another branch |
|
||||||
|
| `head_mismatch` | the worktree moved under the caller |
|
||||||
|
|
||||||
|
## The #447 create-PR provenance guard is unchanged
|
||||||
|
|
||||||
|
`issue_lock_provenance.assess_lock_file_for_create_pr` still requires both a
|
||||||
|
sanctioned `lock_provenance` and a `work_lease`, and the sanctioned source set
|
||||||
|
was **not** widened. Bootstrap writes through
|
||||||
|
`issue_lock_provenance.SOURCE_LOCK_ISSUE` — the lock it produces *is* a
|
||||||
|
canonical lock, not a second dialect with its own exemption. Bootstrap now
|
||||||
|
satisfies the guard rather than the guard being relaxed to admit bootstrap.
|
||||||
@@ -0,0 +1,83 @@
|
|||||||
|
# Incident #670: bare direct-to-master commit `2fa97c26` (retroactive audit)
|
||||||
|
|
||||||
|
Status: verified; disposition recommendation: **accept as-is, no revert** (final
|
||||||
|
disposition owned by controller per issue #670).
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
Commit `2fa97c26fbda555a1a83930ca5fdcea9d8e47b50`
|
||||||
|
(`fix(mcp): load dotenv relative to project root`) landed on `prgs/master`
|
||||||
|
as a single-parent commit with no PR wrapper and no review record, bypassing
|
||||||
|
the sanctioned issue → branch → PR → review → merge workflow. It was
|
||||||
|
discovered during the PR #654 post-merge audit. PR #654 itself merged
|
||||||
|
cleanly via the Gitea API and did **not** introduce this commit.
|
||||||
|
|
||||||
|
## Verification evidence (acceptance criteria 1–3)
|
||||||
|
|
||||||
|
- **AC1 — present on `prgs/master`: yes.**
|
||||||
|
`git merge-base --is-ancestor 2fa97c26fbda555a1a83930ca5fdcea9d8e47b50 prgs/master` → true.
|
||||||
|
- **AC2 — no PR or review record: confirmed.**
|
||||||
|
The commit is a single-parent, non-merge commit sitting directly on
|
||||||
|
first-parent master between the #629 merge (`5ab5fe85`) and the #654
|
||||||
|
merge (`ec903b0d`). A PR landing on master produces a merge commit (or a
|
||||||
|
PR-linked head); neither exists here. The controller audit at issue-create
|
||||||
|
time also found no PR wrapper and no review record for this SHA.
|
||||||
|
- **AC3 — changed files and diff summary: confirmed.**
|
||||||
|
`gitea_auth.py | 5 +++--` (+3/−2). Single parent
|
||||||
|
`5ab5fe8583c07134d55dadf09381aecb67df246e`. The change moves
|
||||||
|
`PROJECT_ROOT` derivation above `load_dotenv()` and loads
|
||||||
|
`.env` relative to the project root instead of the process CWD.
|
||||||
|
|
||||||
|
## AC4 — why no immediate revert
|
||||||
|
|
||||||
|
- The dotenv fix is intentional and required for correct runtime behavior:
|
||||||
|
without it, `load_dotenv()` resolves `.env` against the process working
|
||||||
|
directory, which breaks MCP server launches whose CWD is not the project
|
||||||
|
root.
|
||||||
|
- The change is small (+3/−2), self-contained in `gitea_auth.py`, and has
|
||||||
|
been running on master without incident since 2026-07-10.
|
||||||
|
- Reverting would re-introduce a real bug to remove a provenance defect —
|
||||||
|
the wrong trade. Provenance is repaired retroactively by this document,
|
||||||
|
issue #670, and the hardening landed under #671.
|
||||||
|
- If the controller later judges the change unsafe, a separate
|
||||||
|
revert/repair issue is the sanctioned path (issue #670, recommended
|
||||||
|
disposition option 4).
|
||||||
|
|
||||||
|
## AC5 — workflow-hardening linkage
|
||||||
|
|
||||||
|
Prevention already landed: **issue #671** (closed)
|
||||||
|
*“Block direct pushes to stable branches from MCP workflow sessions”*,
|
||||||
|
implemented by commit `5933d87647656643a67a50331c4c7b06ea751dad`
|
||||||
|
(`feat(guard): block direct stable-branch pushes from MCP workflow sessions`).
|
||||||
|
|
||||||
|
Shipped guardrails include:
|
||||||
|
|
||||||
|
- `gitea_record_stable_branch_push_attempt` — classifies proposed commands
|
||||||
|
for direct stable-branch push intent (`git push <remote> master`,
|
||||||
|
refspecs, `HEAD:master`, `--force`, dry-run intent, `:master` delete),
|
||||||
|
plus root/control-checkout local commits not carried by an issue branch,
|
||||||
|
and writes a durable `stable_branch_contamination` marker.
|
||||||
|
- `gitea_audit_stable_branch_contamination` — reconciler-only audit/clear
|
||||||
|
path; a contaminated worker session cannot self-clear.
|
||||||
|
- Review/merge/close/completion mutations fail closed while a
|
||||||
|
contamination marker is active.
|
||||||
|
|
||||||
|
## AC6 — PR #654 was not the source
|
||||||
|
|
||||||
|
- `2fa97c26` is the **first parent** of the #654 merge commit
|
||||||
|
`ec903b0d619e7a27d24aed272a890f4e5d381411`; it predates the #654 merge.
|
||||||
|
- First-parent history `5ab5fe8..ec903b0`:
|
||||||
|
`2fa97c2 fix(mcp): load dotenv relative to project root` followed by
|
||||||
|
`ec903b0 Merge pull request 'feat: lifecycle role/hazard labels ... (#603)' (#654)`.
|
||||||
|
- The #654 merger audit confirmed `ec903b0d` was a valid Gitea-API merge,
|
||||||
|
the `git push prgs master` attempt during that run was a no-op, and the
|
||||||
|
net change `2fa97c2..ec903b0` contained only the reviewed #603
|
||||||
|
lifecycle-label files.
|
||||||
|
- Conclusion: #654 merged reviewed content only; the unauthorized-path
|
||||||
|
defect is solely the earlier bare commit `2fa97c26`.
|
||||||
|
|
||||||
|
## Explicit non-actions (unchanged by this audit)
|
||||||
|
|
||||||
|
- No revert of `2fa97c26`.
|
||||||
|
- No force-push or history rewrite.
|
||||||
|
- No master mutation from the audit session.
|
||||||
@@ -0,0 +1,70 @@
|
|||||||
|
# MCP Config Drift Diagnostic & Sanctioned Repair Runbook (#672)
|
||||||
|
|
||||||
|
This document describes the diagnostic framework for detecting configuration drift between the active IDE MCP configuration (`~/.gemini/antigravity-ide/mcp_config.json`) and the offline/global canonical configuration (`~/.gemini/config/mcp_config.json`), and establishes the **sanctioned repair runbook**.
|
||||||
|
|
||||||
|
## Background & Problem Statement
|
||||||
|
|
||||||
|
Offline tools like `test_mcp_conn.py` test the global configuration (`~/.gemini/config/mcp_config.json`) via `subprocess.Popen`. However, the active IDE/client namespace uses `~/.gemini/antigravity-ide/mcp_config.json`. When required Gitea role servers (`gitea-author`, `gitea-reviewer`, `gitea-merger`, `gitea-reconciler`, `gitea-controller`, `gitea-tools`) are missing or carry mismatched profile environments in the active IDE config:
|
||||||
|
|
||||||
|
1. Offline tests pass (`test_mcp_conn.py` green).
|
||||||
|
2. The IDE client returns `EOF` / `transport closed` when attempting role-scoped mutations.
|
||||||
|
3. Operators misdiagnose missing server definitions as stale runtimes, leading to forbidden `pkill` attempts (#630) or `mtime` hacks (#655).
|
||||||
|
|
||||||
|
## Diagnostic Tool: `mcp_config_drift.py`
|
||||||
|
|
||||||
|
Run the diagnostic tool directly to compare configurations:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python3 mcp_config_drift.py --json
|
||||||
|
```
|
||||||
|
|
||||||
|
Or specify custom config locations:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python3 mcp_config_drift.py \
|
||||||
|
--active-config ~/.gemini/antigravity-ide/mcp_config.json \
|
||||||
|
--global-config ~/.gemini/config/mcp_config.json
|
||||||
|
```
|
||||||
|
|
||||||
|
### Key Diagnostic Outputs
|
||||||
|
|
||||||
|
- `in_sync`: Boolean indicating if all required Gitea role servers exist in the active IDE config with matching profile declarations.
|
||||||
|
- `missing_role_servers`: List of role servers present in global config but missing from active IDE config.
|
||||||
|
- `profile_mismatches`: List of profile environment mismatches per server.
|
||||||
|
- `reasons`: Explicit, human-readable list of drift causes.
|
||||||
|
|
||||||
|
All returned payloads automatically redact secret tokens, DSNs, Authorization headers, and private keys.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Sanctioned Repair Path (Step-by-Step)
|
||||||
|
|
||||||
|
When `mcp_config_drift.py` reports drift (`in_sync: false`), execute the following **sanctioned repair steps**:
|
||||||
|
|
||||||
|
1. **Backup Active IDE Config:**
|
||||||
|
```bash
|
||||||
|
cp ~/.gemini/antigravity-ide/mcp_config.json ~/.gemini/antigravity-ide/mcp_config.json.bak
|
||||||
|
```
|
||||||
|
2. **Patch Active IDE Config:**
|
||||||
|
Copy the missing Gitea role server JSON blocks (`gitea-author`, `gitea-reviewer`, etc.) from `~/.gemini/config/mcp_config.json` into `~/.gemini/antigravity-ide/mcp_config.json`.
|
||||||
|
3. **Reconnect via IDE/Client:**
|
||||||
|
Use the IDE / client UI reconnection control (or restart the IDE client app).
|
||||||
|
4. **Verify Active Namespace Health:**
|
||||||
|
Invoke `gitea_whoami` (and optional `gitea_resolve_task_capability`) through the active IDE client on each required role namespace.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## FORBIDDEN Repair Actions (#630 / #655)
|
||||||
|
|
||||||
|
The following actions are **strictly forbidden** for config drift repair:
|
||||||
|
|
||||||
|
- ❌ **`pkill` or manual daemon process kill commands:** Process kills cause contamination and break active session leases.
|
||||||
|
- ❌ **`mtime` touch edits:** Artificial mtime modifications mask stale runtimes without updating configuration.
|
||||||
|
- ❌ **Source code edits:** Mutating python tool logic to bypass missing server entries.
|
||||||
|
- ❌ **Session-state edits:** Direct database or lock-file state mutation.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Final Report Guidelines
|
||||||
|
|
||||||
|
A workflow final report **must not** rely on offline `test_mcp_conn.py` output alone. Final reports must include active-config evidence from live `gitea_whoami` calls on the active IDE namespaces.
|
||||||
@@ -47,18 +47,33 @@ Do the steps in order. Stop as soon as a live **client-namespace** call succeeds
|
|||||||
- Only the Gitea namespace fails → single-namespace transport close. Continue.
|
- Only the Gitea namespace fails → single-namespace transport close. Continue.
|
||||||
- Every server fails → restart the whole MCP client, not just one namespace.
|
- Every server fails → restart the whole MCP client, not just one namespace.
|
||||||
|
|
||||||
2. **Reconnect the namespace through the client, not the shell.** Use the IDE /
|
2. **Request the sanctioned reconnect surface (#678), then reconnect through
|
||||||
client MCP-reconnect action for that server entry (in Claude Code:
|
the client — not the shell.** From a still-reachable Gitea MCP namespace
|
||||||
`/mcp` → reconnect the affected `gitea-*` server). Reconnecting forces the
|
(or after host auto-reconnect), call:
|
||||||
client to spawn a fresh subprocess and re-open the pipe. This clears the
|
|
||||||
closed-client state that a bare `kill`/respawn from a terminal does **not**.
|
```text
|
||||||
|
gitea_request_mcp_reconnect(
|
||||||
|
namespace="gitea-author", # or gitea-reviewer / gitea-merger / …
|
||||||
|
reason="transport_eof",
|
||||||
|
client="codex", # or claude_code / generic
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
The tool is **report-only**: it never restarts a process. It returns
|
||||||
|
namespace, profile, pid/session, startup SHA, current master SHA, boundary
|
||||||
|
status, and a **typed blocker** with exact operator UI steps for Codex
|
||||||
|
(Reload Developer Tools / per-server reconnect) or Claude Code (`/mcp`).
|
||||||
|
Then perform the host reconnect those steps describe so the client spawns a
|
||||||
|
fresh subprocess and re-opens the pipe. That clears the closed-client state
|
||||||
|
that a bare `kill`/respawn from a terminal does **not**.
|
||||||
|
|
||||||
3. **Do not "fix" it by importing the server or poking the process.** Reaching
|
3. **Do not "fix" it by importing the server or poking the process.** Reaching
|
||||||
for `python -c 'import gitea_mcp_server ...'`, raw JSON-RPC from a shell,
|
for `python -c 'import gitea_mcp_server ...'`, raw JSON-RPC from a shell,
|
||||||
killing PIDs to force a respawn, or touching MCP config mtimes does **not**
|
killing PIDs to force a respawn, or touching MCP config mtimes does **not**
|
||||||
restore the *client's* view of the namespace and violates the daemon-import
|
restore the *client's* view of the namespace and violates the daemon-import
|
||||||
guard (#558, `docs/mcp-daemon-import-guard.md`). The only sanctioned repair
|
guard (#558, `docs/mcp-daemon-import-guard.md`). The only sanctioned repair
|
||||||
is a **client reconnect / relaunch**.
|
is a **client reconnect / relaunch** (or the typed operator path returned by
|
||||||
|
`gitea_request_mcp_reconnect`).
|
||||||
|
|
||||||
4. **Verify through the same path the workflow will use.** After reconnect, call
|
4. **Verify through the same path the workflow will use.** After reconnect, call
|
||||||
the specific tool the blocked workflow needs — not just any tool — through
|
the specific tool the blocked workflow needs — not just any tool — through
|
||||||
@@ -153,7 +168,19 @@ not a tool argument: a session must never be able to authorize itself.
|
|||||||
## Related
|
## Related
|
||||||
|
|
||||||
- #630 — manual daemon killing as contaminated recovery (this contrast, enforced).
|
- #630 — manual daemon killing as contaminated recovery (this contrast, enforced).
|
||||||
|
- #657 — restart-path inventory and daemon classification.
|
||||||
|
- #686 — manual server launch detection & fail-closed provenance gate.
|
||||||
- #531 / #544 — stale-runtime detection (`ps`-based); sibling failure mode.
|
- #531 / #544 — stale-runtime detection (`ps`-based); sibling failure mode.
|
||||||
- #558 / `docs/mcp-daemon-import-guard.md` — why shell imports are not a repair.
|
- #558 / `docs/mcp-daemon-import-guard.md` — why shell imports are not a repair.
|
||||||
- `docs/mcp-client-registration.md` — per-server registration contract.
|
- `docs/mcp-client-registration.md` — per-server registration contract.
|
||||||
- `docs/mcp-namespace-health.md` — probe sources and mutation enforcement.
|
- `docs/mcp-namespace-health.md` — probe sources and mutation enforcement.
|
||||||
|
|
||||||
|
## Sanctioned reconnect vs forbidden manual launch (#686)
|
||||||
|
|
||||||
|
In addition to manual process killing (#630), manually launching a duplicate role server from an ad hoc shell (`python3 mcp_server.py`) is forbidden and fail-closed:
|
||||||
|
|
||||||
|
- **Why manual launches are unsupported:** A terminal-launched `mcp_server.py` holds its own stdio transport; it can never bind to the IDE client's stdio pipes. It cannot restore a dropped IDE namespace, and a manual duplicate process masks stale client-managed runtimes for that profile, defeating stale-runtime gates.
|
||||||
|
- **Sanctioned path:** Supported recovery is IDE/client-managed reconnect only (`/mcp reconnect`, IDE restart, or sanctioned reconnect exposure).
|
||||||
|
- **Fail-closed enforcement (#686):** Mutating tools on a server lacking client-managed launch provenance (`GITEA_CLIENT_MANAGED=1`) refuse execution fail-closed with typed blocker `unsupported_manual_launch` and an exact next action. Unsupported `GITEA_*` env overrides (e.g. `GITEA_DUMMY`) are surfaced in diagnostics rather than silently ignored.
|
||||||
|
- **Inventory & staleness:** Staleness diagnostics ignore non-client-managed duplicates when evaluating runtime freshness and inventory duplicate processes per profile (#657, #686).
|
||||||
|
|
||||||
|
|||||||
@@ -86,3 +86,74 @@ When a namespace returns EOF, follow
|
|||||||
|
|
||||||
When blocked, repair the IDE namespace and re-record a healthy
|
When blocked, repair the IDE namespace and re-record a healthy
|
||||||
`client_namespace` assessment before retrying the mutation.
|
`client_namespace` assessment before retrying the mutation.
|
||||||
|
|
||||||
|
## Connected vs Attached Tool Surface (#708)
|
||||||
|
|
||||||
|
MCP servers can report **Connected** at the CLI / host inventory layer while the **active LLM session exposes none of their tool namespaces**.
|
||||||
|
|
||||||
|
### Core principle
|
||||||
|
|
||||||
|
* **Connected status at host layer ≠ attached tools in active session.**
|
||||||
|
* Required preflight proof is **live tool visibility + `gitea_whoami` call** through the target namespace, not host `Connected` status alone.
|
||||||
|
* When servers report Connected but namespaces are absent from attached tools, classify as `mcp_connected_namespaces_missing`.
|
||||||
|
|
||||||
|
### Forbidden unsafe fallbacks
|
||||||
|
|
||||||
|
When `mcp_connected_namespaces_missing` is detected, workflows must **fail closed** and must **never** encourage or perform:
|
||||||
|
|
||||||
|
* direct imports of MCP server Python modules
|
||||||
|
* CLI or raw Gitea API mutations as a substitute for native tools
|
||||||
|
* profile hopping to another MCP profile/namespace to bypass the empty session
|
||||||
|
* session-state overrides or hand-edited session/ledger files
|
||||||
|
* process kills (`pkill`), config mtime touches, or `.env` edits
|
||||||
|
|
||||||
|
Only sanctioned recovery: **client reconnect path**, followed by full preflight (`whoami` → capability resolve → task).
|
||||||
|
|
||||||
|
### Native detection tool
|
||||||
|
|
||||||
|
`gitea_assess_mcp_namespace_attachment` classifies the condition and records it in
|
||||||
|
the session. Pass the namespaces the host reports Connected and the namespaces
|
||||||
|
actually attached to the active session tool surface:
|
||||||
|
|
||||||
|
| Argument | Meaning |
|
||||||
|
|---|---|
|
||||||
|
| `connected_servers` | Namespaces the host/CLI reports Connected |
|
||||||
|
| `attached_session_namespaces` | Namespaces exposed in the active session tool surface |
|
||||||
|
| `required_namespaces` | Namespaces this workflow needs (defaults to the role namespaces) |
|
||||||
|
| `discovery_cache_hit` / `discovery_cache_age_seconds` | Client tool-discovery cache state |
|
||||||
|
| `auto_attach_attempted` / `auto_attach_succeeded` | Whether the runtime auto-attached |
|
||||||
|
| `session_tool_snapshot_at` / `namespace_connected_at` | Epoch seconds, to detect startup ordering races |
|
||||||
|
|
||||||
|
It returns `discovery_status`
|
||||||
|
(`namespaces_attached` | `connected_but_namespaces_missing` | `disconnected`),
|
||||||
|
`missing_namespaces`, `proof_of_connected_vs_attached` (per namespace
|
||||||
|
`{connected, attached}`), `error_type`, `reconnect_required`, `auto_recovered`,
|
||||||
|
`startup_ordering_race`, `late_attaching_namespaces`, `sanctioned_recovery_tool`,
|
||||||
|
and a reconnect-only `exact_next_action`.
|
||||||
|
|
||||||
|
### Startup ordering
|
||||||
|
|
||||||
|
`session_tool_snapshot_at` earlier than a namespace's `namespace_connected_at`
|
||||||
|
means that namespace could not have been in the session snapshot, however healthy
|
||||||
|
it looks now. That is reported as `startup_ordering_race` with the affected
|
||||||
|
namespaces listed — the multi-role parallel-connect case where Connected flips
|
||||||
|
true after the session tool list was already captured.
|
||||||
|
|
||||||
|
### Fail-closed gate
|
||||||
|
|
||||||
|
The recorded verdict gates mutations, mirroring the #543 health gate: a namespace
|
||||||
|
that has not been assessed does not gate, but one recorded Connected-but-unattached
|
||||||
|
blocks the mutation whose role namespace it is
|
||||||
|
(`review_pr` / `submit_review` → `gitea-reviewer`, `merge_pr` → `gitea-merger`,
|
||||||
|
`work_issue` / `create_pr` → `gitea-author`). `gitea_submit_pr_review` and
|
||||||
|
`gitea_merge_pr` return the block reason plus the hard-stop policy string, and the
|
||||||
|
only offered recovery is `gitea_request_mcp_reconnect`.
|
||||||
|
|
||||||
|
### Telemetry
|
||||||
|
|
||||||
|
The `telemetry` block carries `connected_count`, `attached_count`,
|
||||||
|
`required_count`, `missing_count`, `discovery_status`, `discovery_cache_hit`,
|
||||||
|
`discovery_cache_age_seconds`, `reconnect_required`, `auto_attach_attempted`,
|
||||||
|
`auto_recovered`, `startup_ordering_race`, and `error_type`. It contains namespace
|
||||||
|
names and counts only — never tokens, endpoints, env values, or filesystem paths.
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,94 @@
|
|||||||
|
# MCP scoped recovery playbook (#669)
|
||||||
|
|
||||||
|
**Parent:** [#655](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/655)
|
||||||
|
**Vision / roadmap:** [#652](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/652) · [#653](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/653)
|
||||||
|
**Class matrix:** [#663](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/663) · `docs/mcp-restart-classes.md`
|
||||||
|
**Coordinator:** [#658](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/658) · `restart_coordinator.py`
|
||||||
|
**Audit lineage:** [#665](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/665)
|
||||||
|
|
||||||
|
## Decision
|
||||||
|
|
||||||
|
Full-server MCP reset is a **last resort**. Prefer the narrowest recovery that
|
||||||
|
can clear the symptom. The coordinator **refuses** `rolling_mcp_restart`,
|
||||||
|
`full_mcp_restart`, and `host_restart` unless:
|
||||||
|
|
||||||
|
1. The inventory carries a prior **attempt log** of at least one *insufficient*
|
||||||
|
narrower recovery, **or**
|
||||||
|
2. **Break-glass** is authorized
|
||||||
|
(`request_break_glass` + `GITEA_BREAKGLASS_RESTART_AUTHORIZATION`).
|
||||||
|
|
||||||
|
Break-glass still never bypasses the #663 class matrix (role/permission).
|
||||||
|
|
||||||
|
## Ladder (narrow → broad)
|
||||||
|
|
||||||
|
| Rank | Action | Self-service | Implementation / delegation |
|
||||||
|
|---:|---|---|---|
|
||||||
|
| 0 | `client_reconnect` | yes | Host auto-reconnect / client reconnect · [#584](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/584) · `docs/mcp-namespace-eof-recovery.md` |
|
||||||
|
| 1 | `capability_refresh` | yes | `gitea_resolve_task_capability` + `gitea_whoami` · [#610](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/610) · [#685](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/685) |
|
||||||
|
| 2 | `session_reconnect` | yes | Runtime rebind + explicit `worktree_path` · [#543](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/543) · [#618](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/618) |
|
||||||
|
| 3 | `configuration_reload` | no | Class `configuration_reload` · console reload · [#642](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/642) |
|
||||||
|
| 4 | `lease_recovery` | no | Lock/lease recovery paths · [#702](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/702) · [#753](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/753) · [#790](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/790) |
|
||||||
|
| 5 | `worker_restart` | no | Class `worker_restart` · #663 |
|
||||||
|
| 6 | `role_runtime_restart` | no | Class `role_runtime_restart` · console restart · #642/#663 |
|
||||||
|
| 7 | `connector_restart` | no | Class `connector_restart` · #663 |
|
||||||
|
| 8 | `rolling_mcp_restart` | no | Class `rolling_mcp_restart` · design [#668](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/668) · **attempt log required** |
|
||||||
|
| 9 | `full_mcp_restart` | no | Class `full_mcp_restart` · **attempt log required** |
|
||||||
|
| 10 | `host_restart` | no | Class `host_restart` · **attempt log required** |
|
||||||
|
|
||||||
|
Machine-readable source of truth: `recovery_playbook.RECOVERY_LADDER` and
|
||||||
|
`recovery_playbook.ladder_document()`.
|
||||||
|
|
||||||
|
## Attempt log shape
|
||||||
|
|
||||||
|
Each prior attempt is a mapping:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"action": "client_reconnect",
|
||||||
|
"outcome": "insufficient",
|
||||||
|
"reason": "transport still closed after IDE reconnect",
|
||||||
|
"actor": "prgs-controller-12345",
|
||||||
|
"recorded_at": "2026-07-25T21:00:00+00:00"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
Outcomes that count toward escalation: `failed`, `insufficient`, `denied`,
|
||||||
|
`unresolved`, `timeout`, `error`.
|
||||||
|
|
||||||
|
Pass attempts into the coordinator via inventory
|
||||||
|
`prior_recovery_attempts` or the MCP tool argument
|
||||||
|
`prior_recovery_attempts_json` on `gitea_request_mcp_restart`.
|
||||||
|
|
||||||
|
Helper: `recovery_playbook.build_attempt_record(...)`.
|
||||||
|
|
||||||
|
## Symptom → first rung
|
||||||
|
|
||||||
|
`recovery_playbook.recommend_actions(symptoms=[...])` maps symptoms such as
|
||||||
|
`transport_eof`, `stale_capability`, `stale_lease`, `daemon_corrupt` to the
|
||||||
|
narrowest recommended action, then walks the ladder. Soft recommendations
|
||||||
|
never replace the hard gate on broad restarts.
|
||||||
|
|
||||||
|
## Enforcement points
|
||||||
|
|
||||||
|
1. **`recovery_playbook.assess_escalation`** — pure gate.
|
||||||
|
2. **`restart_coordinator.evaluate_restart_impact`** — when `restart_class` is
|
||||||
|
set (policy-enforced path), broad classes require the gate; report fields
|
||||||
|
`attempt_log_satisfied`, `playbook_escalation`, `break_glass`.
|
||||||
|
3. **`gitea_request_mcp_restart`** — accepts attempt JSON and env-authorized
|
||||||
|
break-glass; never restarts a process.
|
||||||
|
|
||||||
|
## Metrics
|
||||||
|
|
||||||
|
`recovery_playbook.recovery_metrics(attempts)` reports the fraction of
|
||||||
|
successful recoveries that avoided full/host restart
|
||||||
|
(`fraction_avoided_full_restart`).
|
||||||
|
|
||||||
|
## Non-goals
|
||||||
|
|
||||||
|
* HA multi-instance execution (#668 design only here).
|
||||||
|
* Normalizing `pkill` (#630 contamination stays forbidden).
|
||||||
|
* Silent mutation of leases or processes from the playbook itself.
|
||||||
|
|
||||||
|
## Manual process kills
|
||||||
|
|
||||||
|
Remain forbidden and contaminating (#630). The playbook never recommends them.
|
||||||
@@ -95,12 +95,18 @@ gitea_request_mcp_restart(remote, host, org, repo,
|
|||||||
target_session_id=None, target_role=None,
|
target_session_id=None, target_role=None,
|
||||||
target_connector=None,
|
target_connector=None,
|
||||||
drain_proof_json=None,
|
drain_proof_json=None,
|
||||||
request_break_glass=False)
|
request_break_glass=False,
|
||||||
|
prior_recovery_attempts_json=None)
|
||||||
```
|
```
|
||||||
|
|
||||||
It **never restarts anything**: `apply_supported` is always `false` and
|
It **never restarts anything**: `apply_supported` is always `false` and
|
||||||
`restart_performed` is always `false`.
|
`restart_performed` is always `false`.
|
||||||
|
|
||||||
|
`prior_recovery_attempts_json` (#669) is an optional JSON array of prior
|
||||||
|
narrow recovery attempts. Rolling / full / host classes require at least one
|
||||||
|
*insufficient* narrower attempt (or authorized break-glass). See
|
||||||
|
`docs/mcp-recovery-playbook.md`.
|
||||||
|
|
||||||
### Dry-run versus apply
|
### Dry-run versus apply
|
||||||
|
|
||||||
| Call | Behavior |
|
| Call | Behavior |
|
||||||
@@ -112,9 +118,10 @@ It **never restarts anything**: `apply_supported` is always `false` and
|
|||||||
|
|
||||||
An apply requires **both** authorizations, and they are independent:
|
An apply requires **both** authorizations, and they are independent:
|
||||||
|
|
||||||
1. **Restart-class authorization** (#663) — the requester's role and permissions
|
1. **Restart-class authorization** (#663 / #669) — the requester's role and
|
||||||
must allow the requested class, the class's approval requirement must be
|
permissions must allow the requested class, the class's approval requirement
|
||||||
satisfied, and any target-scoped class must name its target. Failing any of
|
must be satisfied, any target-scoped class must name its target, and broad
|
||||||
|
classes must satisfy the recovery-playbook attempt-log gate. Failing any of
|
||||||
these makes `allow_restart` `false`.
|
these makes `allow_restart` `false`.
|
||||||
2. **Drain-proof gate** (#661) — a valid, unexpired, clean proof bound to the
|
2. **Drain-proof gate** (#661) — a valid, unexpired, clean proof bound to the
|
||||||
current impact fingerprint, or an authorized break-glass.
|
current impact fingerprint, or an authorized break-glass.
|
||||||
@@ -127,7 +134,10 @@ the authorization that produced it.
|
|||||||
|
|
||||||
### Break-glass
|
### Break-glass
|
||||||
|
|
||||||
Break-glass bypasses the **drain proof only** — never the restart-class matrix.
|
Break-glass bypasses the **drain proof only** — never the restart-class matrix
|
||||||
|
(role/permission). Separately, authorized break-glass also satisfies the #669
|
||||||
|
attempt-log requirement for broad restarts (rolling/full/host), because that
|
||||||
|
gate is not a class-matrix permission check.
|
||||||
It is honoured solely when `request_break_glass` is set *and* the environment
|
It is honoured solely when `request_break_glass` is set *and* the environment
|
||||||
carries `GITEA_BREAKGLASS_RESTART_AUTHORIZATION`; like operator override, the
|
carries `GITEA_BREAKGLASS_RESTART_AUTHORIZATION`; like operator override, the
|
||||||
tool argument expresses caller intent and cannot be self-asserted by a worker
|
tool argument expresses caller intent and cannot be self-asserted by a worker
|
||||||
|
|||||||
@@ -45,10 +45,11 @@ and *fails closed*.
|
|||||||
| `legacy_auto_restart_helper` | removed | A helper (`_trigger_mcp_auto_restart`) that actively restarted the server from the read-only resolver path. | Removed in #685; kept absent by `assert_auto_restart_helper_absent()`. | #685, #657 |
|
| `legacy_auto_restart_helper` | removed | A helper (`_trigger_mcp_auto_restart`) that actively restarted the server from the read-only resolver path. | Removed in #685; kept absent by `assert_auto_restart_helper_absent()`. | #685, #657 |
|
||||||
| `config_touch_reload` | removed | Touching (utime) the MCP client config to make the host reload the server. | Removed from the resolver in #685: stale detection is report-only, never mutating config, spawning threads, or calling `os._exit`. | #685, #657 |
|
| `config_touch_reload` | removed | Touching (utime) the MCP client config to make the host reload the server. | Removed from the resolver in #685: stale detection is report-only, never mutating config, spawning threads, or calling `os._exit`. | #685, #657 |
|
||||||
| `master_advance_auto_restart` | guarded_fail_closed | On-disk master advancing past the running code. | `master_parity_gate` captures startup parity and blocks mutations while stale, emitting restart guidance; the process never self-restarts. | #420, #591, #657 |
|
| `master_advance_auto_restart` | guarded_fail_closed | On-disk master advancing past the running code. | `master_parity_gate` captures startup parity and blocks mutations while stale, emitting restart guidance; the process never self-restarts. | #420, #591, #657 |
|
||||||
| `stale_runtime_resolver_reconnect` | guarded_fail_closed | The capability resolver detecting a stale serving process. | Report-only (#685): returns `restart_required`/`stop_required` and an exact reconnect action; no restart, thread, config touch, or `os._exit`. | #685, #657 |
|
| `stale_runtime_resolver_reconnect` | guarded_fail_closed | The capability resolver detecting a stale serving process. | Report-only (#685): returns `restart_required`/`stop_required` and an exact reconnect action; no restart, thread, config touch, or `os._exit`. | #685, #657, #678 |
|
||||||
|
| `codex_client_reconnect_request` | guarded_fail_closed | `gitea_request_mcp_reconnect` report-only tool for Codex/LLM sessions. | Report-only (#678): returns namespace/profile/pid/startup SHA/master SHA/boundary status and a typed operator blocker with exact client UI steps; never restarts or kills. | #678, #630, #685, #657 |
|
||||||
| `manual_daemon_kill` | forbidden | Shell kills of the daemon: `pkill -f mcp_server.py`, `killall`, broad `pkill -f python` sweeps, or `kill <pid>` of a daemon pid. | Forbidden (#630): `runtime_recovery_guard` classifies these as contamination and `gitea_record_daemon_process_kill_attempt` writes a durable marker that fails later mutations closed. Operator maintenance authorization is read only from the environment. | #630, #657 |
|
| `manual_daemon_kill` | forbidden | Shell kills of the daemon: `pkill -f mcp_server.py`, `killall`, broad `pkill -f python` sweeps, or `kill <pid>` of a daemon pid. | Forbidden (#630): `runtime_recovery_guard` classifies these as contamination and `gitea_record_daemon_process_kill_attempt` writes a durable marker that fails later mutations closed. Operator maintenance authorization is read only from the environment. | #630, #657 |
|
||||||
| `conflict_marker_infra_stop` | guarded_fail_closed | The daemon entrypoint scans for unresolved merge-conflict markers at startup and stops (`sys.exit(1)`). | Fail-closed startup stop, not a restart: the process exits and waits for the operator to resolve conflicts and relaunch; never loops. | #657 |
|
| `conflict_marker_infra_stop` | guarded_fail_closed | The daemon entrypoint scans for unresolved merge-conflict markers at startup and stops (`sys.exit(1)`). | Fail-closed startup stop, not a restart: the process exits and waits for the operator to resolve conflicts and relaunch; never loops. | #657 |
|
||||||
| `ide_client_reconnect` | host_residual | A manual `/mcp reconnect` (or equivalent host action) that recreates the MCP client connection. | Outside this process's control; the sanctioned recovery the gates point operators toward. No in-process code initiates it. | #584, #656, #657 |
|
| `ide_client_reconnect` | host_residual | A manual `/mcp reconnect` (or equivalent host action) that recreates the MCP client connection. Agents obtain exact UI steps via `gitea_request_mcp_reconnect` (#678). | Outside this process's control; the sanctioned recovery the gates point operators toward. No in-process code initiates it. | #584, #656, #657, #678 |
|
||||||
| `profile_switch_runtime` | sanctioned_narrow_recovery | Switching the active execution profile at runtime (dynamic-profile mode). | In-process and restart-free: `runtime_switching_supported` is true, so a switch rebinds capability without recreating the process. | #656, #657 |
|
| `profile_switch_runtime` | sanctioned_narrow_recovery | Switching the active execution profile at runtime (dynamic-profile mode). | In-process and restart-free: `runtime_switching_supported` is true, so a switch rebinds capability without recreating the process. | #656, #657 |
|
||||||
|
|
||||||
## Guards enforced in CI
|
## Guards enforced in CI
|
||||||
|
|||||||
@@ -56,6 +56,7 @@ that gates each call, not which tools exist.
|
|||||||
- `gitea_assess_conflict_fix_push`
|
- `gitea_assess_conflict_fix_push`
|
||||||
- `gitea_assess_gitea_operation_path`
|
- `gitea_assess_gitea_operation_path`
|
||||||
- `gitea_assess_master_parity`
|
- `gitea_assess_master_parity`
|
||||||
|
- `gitea_assess_mcp_namespace_attachment`
|
||||||
- `gitea_assess_mcp_namespace_health`
|
- `gitea_assess_mcp_namespace_health`
|
||||||
- `gitea_assess_pr_sync_status`
|
- `gitea_assess_pr_sync_status`
|
||||||
- `gitea_assess_review_merge_state_machine`
|
- `gitea_assess_review_merge_state_machine`
|
||||||
@@ -103,6 +104,7 @@ that gates each call, not which tools exist.
|
|||||||
- `gitea_get_shell_health`
|
- `gitea_get_shell_health`
|
||||||
- `gitea_heartbeat_issue_lock`
|
- `gitea_heartbeat_issue_lock`
|
||||||
- `gitea_heartbeat_reviewer_pr_lease`
|
- `gitea_heartbeat_reviewer_pr_lease`
|
||||||
|
- `gitea_inspect_issue_lock_contract`
|
||||||
- `gitea_inspect_workflow_lease`
|
- `gitea_inspect_workflow_lease`
|
||||||
- `gitea_issue_irrecoverable_provenance_authorization`
|
- `gitea_issue_irrecoverable_provenance_authorization`
|
||||||
- `gitea_list_dependency_edges`
|
- `gitea_list_dependency_edges`
|
||||||
@@ -134,9 +136,11 @@ that gates each call, not which tools exist.
|
|||||||
- `gitea_record_pre_review_command`
|
- `gitea_record_pre_review_command`
|
||||||
- `gitea_record_shell_spawn_outcome`
|
- `gitea_record_shell_spawn_outcome`
|
||||||
- `gitea_record_stable_branch_push_attempt`
|
- `gitea_record_stable_branch_push_attempt`
|
||||||
|
- `gitea_recover_incomplete_bootstrap_lock`
|
||||||
- `gitea_release_merger_pr_lease`
|
- `gitea_release_merger_pr_lease`
|
||||||
- `gitea_release_reviewer_pr_lease`
|
- `gitea_release_reviewer_pr_lease`
|
||||||
- `gitea_release_workflow_lease`
|
- `gitea_release_workflow_lease`
|
||||||
|
- `gitea_request_mcp_reconnect`
|
||||||
- `gitea_request_mcp_restart`
|
- `gitea_request_mcp_restart`
|
||||||
- `gitea_resolve_task_capability`
|
- `gitea_resolve_task_capability`
|
||||||
- `gitea_resume_review_draft`
|
- `gitea_resume_review_draft`
|
||||||
|
|||||||
@@ -0,0 +1,35 @@
|
|||||||
|
# Web Console: Sentry/GlitchTip Observability & Incident Bridge Console (#649)
|
||||||
|
|
||||||
|
This document describes the Phase 4 observability console surface integrated into the MCP Control Plane Web Console (`webui/`), backed by the #612 incident bridge and the #613 control-plane DB substrate.
|
||||||
|
|
||||||
|
## Architectural Authority Model (ADR Alignment)
|
||||||
|
|
||||||
|
Per the Web Console Architecture ADR (`docs/architecture/webui-control-plane-console-architecture-adr.md`):
|
||||||
|
|
||||||
|
| Layer | Responsibility | Authority |
|
||||||
|
|---|---|---|
|
||||||
|
| **Gitea** | Durable work record | Issues, PRs, comments, reviews, labels, merges |
|
||||||
|
| **Control-plane DB** | Live coordination & linkage | `incident_links` table, session leases, allocations |
|
||||||
|
| **Sentry / GlitchTip** | Observability input | Unresolved incidents, error events, stack traces |
|
||||||
|
| **Incident Bridge (#612)** | Reconciliation engine | Reconciles provider observations into Gitea issues |
|
||||||
|
| **Web Console (`webui/`)** | Read-only projection & gated actions | Projects connection health & correlation links; gates writes |
|
||||||
|
|
||||||
|
> **Key Rule:** Raw monitoring incidents are **never** assignable control-plane `work_items`. They remain observation input only.
|
||||||
|
|
||||||
|
## Redaction Boundary Invariants
|
||||||
|
|
||||||
|
1. **No secrets in returns or rendering:** Auth tokens (`SENTRY_AUTH_TOKEN`, `GLITCHTIP_AUTH_TOKEN`), DSNs, `Authorization` headers, and sensitive local file paths are passed through `webui.console_redaction` before leaving the server.
|
||||||
|
2. **Safe projection:** Connection objects report `credentials_present: true/false` rather than exposing raw keys or headers.
|
||||||
|
|
||||||
|
## Console Endpoints
|
||||||
|
|
||||||
|
- **HTML Surface:** `GET /observability` — Renders provider connection cards, error correlation tables, and gated reconcile controls.
|
||||||
|
- **Versioned API:** `GET /api/v1/observability` — Returns structured JSON snapshot with `schema_version`, `providers`, `links`, and `metrics`.
|
||||||
|
- **Legacy Compatibility Alias:** `GET /api/observability` — Read-only compatibility alias for Phase 4.
|
||||||
|
|
||||||
|
## Gated Actions
|
||||||
|
|
||||||
|
- `observability_reconcile_incident` (`gitea_observability_reconcile_incident`): Triggers or previews dry-run issue reconciliation for a provider incident.
|
||||||
|
- `observability_link_issue` (`gitea_observability_link_issue`): Links a provider incident to an existing Gitea tracking issue.
|
||||||
|
|
||||||
|
Both actions require `operator` role and gate through `task_capability_map`. Execution fails closed in read-only MVP mode.
|
||||||
@@ -19,7 +19,11 @@ The assessor classifies:
|
|||||||
|
|
||||||
- **service_health** — process healthy / parity mutation-safe
|
- **service_health** — process healthy / parity mutation-safe
|
||||||
- **clients** — connected client descriptors (optional inventory)
|
- **clients** — connected client descriptors (optional inventory)
|
||||||
- **sessions** — active session rows with dead owner pids are unresolved
|
- **sessions** — active session rows with dead or reused owner pids are
|
||||||
|
unresolved until retired via `#969` (`session_lifecycle` /
|
||||||
|
`apply_session_cleanup=true` on `gitea_reconcile_after_restart`, or
|
||||||
|
`gitea_retire_stale_workflow_sessions`). Live owners, live leases, and live
|
||||||
|
client-managed sessions are never retired.
|
||||||
- **checkpoints** — soft-depends on #660; skipped with reason when schema absent
|
- **checkpoints** — soft-depends on #660; skipped with reason when schema absent
|
||||||
- **leases** — live control-plane leases after restart
|
- **leases** — live control-plane leases after restart
|
||||||
- **capabilities** — master-parity / stale-runtime (#610)
|
- **capabilities** — master-parity / stale-runtime (#610)
|
||||||
|
|||||||
@@ -0,0 +1,230 @@
|
|||||||
|
# Remote-MCP coupling inventory
|
||||||
|
|
||||||
|
Every place the Gitea MCP server depends on being a local, client-spawned, stdio-attached
|
||||||
|
process on the operator's machine.
|
||||||
|
|
||||||
|
- **Issue:** #930 (Remote-MCP 01), child 1 of epic #929.
|
||||||
|
- **Generated against commit:** `7bf4f1258451823a55b36d2157e74f8457165088` (`master`).
|
||||||
|
- **Anchors:** every `file:line` below resolves at the commit above and at the commit that
|
||||||
|
adds this document. This change adds one new file and edits no existing file, so no
|
||||||
|
existing line number shifts between the two.
|
||||||
|
- **Scope:** documentation only. No server behavior changes in this child.
|
||||||
|
|
||||||
|
## How to read an entry
|
||||||
|
|
||||||
|
| Field | Meaning |
|
||||||
|
| ----- | ------- |
|
||||||
|
| **Anchor** | `file:line` at the commit under review. |
|
||||||
|
| **Assumes today** | What the code takes for granted while running as a local stdio process. |
|
||||||
|
| **Observes remotely** | What the same code would actually see on a shared remote host. |
|
||||||
|
| **Class** | One of: *portable as written*, *needs a seam*, *needs a replacement*, *cannot be remote*. |
|
||||||
|
| **Owner** | Exactly one epic child (#931–#939) responsible for the fix. |
|
||||||
|
|
||||||
|
Classification meanings:
|
||||||
|
|
||||||
|
- **portable as written** — the code is already transport-, host-, and principal-neutral; it
|
||||||
|
moves unchanged once its inputs are supplied by a remote-aware caller.
|
||||||
|
- **needs a seam** — the logic is correct but is wired to a hard-coded local source. It needs
|
||||||
|
an injection point, not new semantics.
|
||||||
|
- **needs a replacement** — the semantics themselves are local-only. A remote deployment
|
||||||
|
needs a differently-defined mechanism, not the same mechanism relocated.
|
||||||
|
- **cannot be remote** — the operation is inherently about the operator's own machine
|
||||||
|
(its process table, its keychain, its checkout). It must either stay local behind an
|
||||||
|
explicit boundary or be deleted from the remote surface.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 1. Transport bind
|
||||||
|
|
||||||
|
The transport is bound literally, once, at process start, and the bound value is the root of
|
||||||
|
the mutation-authorization chain.
|
||||||
|
|
||||||
|
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||||
|
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||||
|
| T1 | `gitea_mcp_server.py:23750` | The single production bind call passes the literal `transport="stdio"` immediately before the server loop. | The literal is wrong for any non-stdio deployment; there is no parameter to change it. | needs a seam | #931 |
|
||||||
|
| T2 | `mcp_daemon_guard.py:45` | `_PRODUCTION_TRANSPORTS = frozenset({"stdio"})` is the closed allowlist of production transports. | A remote transport name is rejected by the allowlist before any other check runs. | needs a seam | #931 |
|
||||||
|
| T3 | `mcp_daemon_guard.py:174` | `bind_native_mcp_transport` raises `UnsanctionedRuntimeError` for any transport outside `_PRODUCTION_TRANSPORTS` (raise at `mcp_daemon_guard.py:187`). | The remote server fails to start rather than degrading; the failure is correct, but the allowlist is the only thing that must change. | needs a seam | #931 |
|
||||||
|
| T4 | `mcp_daemon_guard.py:328` | `is_native_mcp_transport()` asserts a process-local runtime record whose `pid` matches `os.getpid()` and whose phase is `transport_bound`. The predicate itself names no transport. | Unchanged semantics: one server process that bound one transport. It stays true on a remote host. | portable as written | #931 |
|
||||||
|
| T5 | `mcp_daemon_guard.py:349` | `is_production_native_mcp_transport()` adds only a `mode == production` check on top of T4. | Unchanged. | portable as written | #931 |
|
||||||
|
| T6 | `irrecoverable_provenance.py:497` | `assess_transport_for_auth_mint()` requires production native transport before minting non-forgeable recovery authorization (#709 F1). | The gate is transport-agnostic in form, but its guarantee — "an ordinary Python process cannot reach this" — is currently underwritten by the stdio bind. Under a remote transport the guarantee must be re-derived from the authenticated session, not from the bind. | needs a seam | #931 |
|
||||||
|
| T7 | `gitea_mcp_server.py:8375` | Consumer: refuses to proceed unless `assess_transport_for_auth_mint()` allows. | Unchanged given a corrected T6. | portable as written | #931 |
|
||||||
|
| T8 | `gitea_mcp_server.py:8624` | Second consumer of the same gate on the confirmation path. | Unchanged given a corrected T6. | portable as written | #931 |
|
||||||
|
| T9 | `mcp_server.py:4` | Module docstring asserts "Runs over stdio." as a property of the server. | The stated contract becomes false on the remote deployment and is load-bearing documentation for operators. | needs a replacement | #931 |
|
||||||
|
|
||||||
|
## 2. Launch provenance
|
||||||
|
|
||||||
|
Mutations fail closed unless the process can prove a client launched it with real stdio pipes
|
||||||
|
and `GITEA_CLIENT_MANAGED` provenance. Every proof in this section is a statement about the
|
||||||
|
local operating system.
|
||||||
|
|
||||||
|
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||||
|
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||||
|
| P1 | `gitea_mcp_server.py:14588` | `_is_client_managed_process()` derives provenance from `GITEA_CLIENT_MANAGED` / `GITEA_MCP_CLIENT_MANAGED` / `GITEA_SERVER_PROVENANCE` / `GITEA_FORCE_CLIENT_MANAGED` on this process's own environment. | A long-lived remote process has one environment for all callers, so a per-process env var can no longer say anything about the caller that issued a request. | needs a replacement | #934 |
|
||||||
|
| P2 | `gitea_mcp_server.py:14606` | Falls back to `sys.stdin.isatty()`: an active TTY on stdin means a human launched it from a terminal, so refuse. | A remote server has no meaningful stdin. The signal is absent, not merely different. | cannot be remote | #934 |
|
||||||
|
| P3 | `gitea_mcp_server.py:14618` | `_provenance_mutation_block()` emits `blocker_kind: "unsupported_manual_launch"` and a "reconnect the IDE/client-managed MCP namespace" remediation. | The block shape is reusable; its predicate and its remediation text are both stdio-specific. | needs a seam | #934 |
|
||||||
|
| P4 | `gitea_mcp_server.py:20599` | `_check_mcp_runtimes_diagnostics()` shells `ps -o pid,lstart,command -ax` and greps for `mcp_server.py` to find peer role servers. | On a shared host the process table lists unrelated tenants' processes, or none at all under a container. Peer discovery by `ps` has no remote meaning. | cannot be remote | #934 |
|
||||||
|
| P5 | `gitea_mcp_server.py:20702` | More than one process per `GITEA_MCP_PROFILE` in the local process table is reported as a duplicate-launch fault. | A remote endpoint is expected to serve many concurrent sessions per role. "Two processes for one role" becomes the normal case, so the check inverts from a safety net into a false wall. | cannot be remote | #934 |
|
||||||
|
| P6 | `gitea_mcp_server.py:20715` | Processes lacking client-managed provenance are ignored for runtime freshness and reported as manual launches. | Same defect as P5: correctness depends on enumerating local peers. | cannot be remote | #934 |
|
||||||
|
| P7 | `gitea_config.py:1172` | `RECOGNIZED_GITEA_ENV_KEYS` is the allowlist of `GITEA_*` env vars a legitimately launched server may carry; anything else is contamination. | Configuration on a remote host arrives from deployment tooling, not from a client-authored env block. The allowlist keeps working mechanically but stops proving anything about provenance. | needs a replacement | #934 |
|
||||||
|
| P8 | `gitea_mcp_server.py:20683` | The unsupported-env scan applies `RECOGNIZED_GITEA_ENV_KEYS` to *other* processes' environments harvested via `ps eww <pid>`. | Reading another process's environment is unavailable or prohibited across tenants, and is not exposed in this form outside macOS/BSD `ps`. | cannot be remote | #934 |
|
||||||
|
| P9 | `mcp_daemon_guard.py:126` | `mark_sanctioned_daemon()` requires the claiming stack frame's resolved absolute path to be the canonical `mcp_server.py` / `gitea_mcp_server.py` next to the guard module; basename spoofing is rejected. | Entrypoint-path identity still exists on a remote host, but it authenticates the *deployment*, not the *caller*. It must be kept and demoted from "authorizes mutations" to "authorizes the process". | needs a seam | #934 |
|
||||||
|
| P10 | `gitea_config.py:1233` | The client-config generator emits `"GITEA_CLIENT_MANAGED": "1"` into each generated MCP client entry, alongside `GITEA_MCP_CONFIG` / `GITEA_MCP_PROFILE`. | A remote endpoint is addressed by URL and credential, not by a spawn command with an env block. This generator produces the wrong artifact entirely. | needs a replacement | #938 |
|
||||||
|
| P11 | `mcp_namespace_health.py:232` | Namespace health classifies a namespace as `client_managed` or `manual_launch` from the reported env summary. | During dual-run, local and remote namespaces coexist and must both be classifiable; a two-valued local/manual axis cannot express "remote endpoint, authenticated session". | needs a replacement | #939 |
|
||||||
|
| P12 | `gitea_mcp_server.py:18161` | The diagnostics payload reports `server_provenance` as exactly `"client_managed"` or `"manual_launch"`. | This is the field a cutover operator reads to confirm which deployment served a call. It must gain a remote value before dual-run parity can be validated. | needs a replacement | #939 |
|
||||||
|
|
||||||
|
## 3. Role binding
|
||||||
|
|
||||||
|
Role separation is currently enforced by *which process a call reaches*. The process is pinned
|
||||||
|
to one role for its lifetime by an environment variable.
|
||||||
|
|
||||||
|
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||||
|
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||||
|
| R1 | `gitea_config.py:54` | `ENV_PROFILE = "GITEA_MCP_PROFILE"` is the single source of the active profile, read from the process environment. | One shared process serves several principals; a process-wide profile cannot answer "who is calling now". This is the root of the coupling. | needs a replacement | #932 |
|
||||||
|
| R2 | `review_workflow_load.py:95` | Reads `GITEA_MCP_PROFILE` directly to decide the reviewer workflow binding. | Reads the deployment's profile, not the caller's, silently granting or denying the wrong role. | needs a replacement | #932 |
|
||||||
|
| R3 | `mcp_discoverability.py:152` | Reads `GITEA_MCP_PROFILE` to describe the namespace to the client. | Correct logic, wrong input source; it needs the request principal injected. | needs a seam | #932 |
|
||||||
|
| R4 | `webui/deployment_boundary.py:115` | Reads `GITEA_MCP_PROFILE` to classify the deployment boundary for the console. | Same as R3. | needs a seam | #932 |
|
||||||
|
| R5 | `gitea_mcp_server.py:21106` | Remediation text instructs the operator to "Relaunch the server with `GITEA_MCP_PROFILE` set to a profile that has the required permission". | Relaunching a shared remote endpoint to change one caller's role is not a valid instruction; it would re-role every other session. | needs a replacement | #932 |
|
||||||
|
| R6 | `native_mcp_preference.py:93` | Detects shell commands that override `GITEA_MCP_PROFILE` away from the session (`native_mcp_preference.py:223`) and flags them as CLI auth divergence. | The divergence check is genuinely useful and survives, but its notion of "the session's profile" must come from the request principal. | needs a seam | #932 |
|
||||||
|
| R7 | `gitea_mcp_server.py:20671` | Recovers a peer server's role by regexing `GITEA_MCP_PROFILE=` out of that process's environment. | Depends on P4/P8 process-table access; role discovery by peer-env scraping has no remote analogue. | cannot be remote | #932 |
|
||||||
|
|
||||||
|
## 4. Credentials
|
||||||
|
|
||||||
|
Every token resolves, directly or indirectly, from one human's macOS keychain.
|
||||||
|
|
||||||
|
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||||
|
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||||
|
| C1 | `gitea_config.py:956` | `_keychain_token()` shells `security find-generic-password -s <item> -w`. | `security(1)` is a macOS binary reading the calling user's login keychain. It does not exist on a Linux host and would be the wrong identity even on a shared Mac. | cannot be remote | #933 |
|
||||||
|
| C2 | `gitea_config.py:974` | `resolve_token(profile, keychain_lookup=_keychain_token)` dispatches on `auth.type` of `env` or `keychain`, defaulting the lookup to C1. | The injectable `keychain_lookup` parameter is the existing seam; a remote credential provider plugs in here without changing the dispatch. | needs a seam | #933 |
|
||||||
|
| C3 | `gitea_config.py:1015` | `keychain_auth(item_id)` constructs the `{"type": "keychain", "id": ...}` reference stored in profiles. | The reference type itself encodes "macOS keychain" into persisted config. A remote provider needs a new auth reference type, not a new value of this one. | needs a replacement | #933 |
|
||||||
|
| C4 | `mcp_daemon_guard.py:440` | `assert_keychain_access_allowed()` fails closed for git-credential keychain fill outside a sanctioned daemon, with an operator opt-out env var. | The gate protects a mechanism that will not exist remotely. Its replacement must gate the *credential provider* call, not the keychain call, or the protection silently lapses. | needs a replacement | #933 |
|
||||||
|
| C5 | `sentry_incident_bridge.py:190` | `resolve_token(env)` resolves the Sentry token from an injected env mapping with no keychain path. | Already host-neutral; it is the shape the Gitea credential path should converge on. | portable as written | #933 |
|
||||||
|
| C6 | `gitea_mcp_server.py:18469` | The profile-audit tool calls `gitea_config.resolve_token(p)` for every configured profile to report "credentials present" without networking. | On a remote host this would materialize every principal's credential inside one process — an audit surface that becomes a credential-aggregation risk. | needs a seam | #933 |
|
||||||
|
|
||||||
|
## 5. Runtime freshness
|
||||||
|
|
||||||
|
The mutation gate is defined as "the commit this process started at matches the checkout on
|
||||||
|
this disk, and both match live master". Two of those three terms are local-disk facts.
|
||||||
|
|
||||||
|
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||||
|
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||||
|
| F1 | `master_parity_gate.py:168` | `capture_startup_parity(root)` reads git `HEAD` from the server's own root once at startup and returns it as the baseline. | A remote host carries a deployed artifact, not the operator's checkout. Its `HEAD` says nothing about the operator's working tree, which is the thing the gate exists to protect. | cannot be remote | #935 |
|
||||||
|
| F2 | `master_parity_gate.py:255` | `mutation_safe = determinable and in_parity and live_known and not live_stale` — a conjunction of two local-HEAD comparisons and one live-remote comparison. | Two of the three conjuncts lose meaning, so the whole verdict does. A remote deployment needs a redefined, testable freshness predicate rather than this one relocated. | needs a replacement | #935 |
|
||||||
|
| F3 | `master_parity_gate.py:164` | The live-remote head is probed and cached per `(root, remote, branch)`, keyed on the local root. | The live-remote probe is the one conjunct that survives; it needs a key that is not the operator's filesystem path. | needs a seam | #935 |
|
||||||
|
| F4 | `gitea_mcp_server.py:18262` | `gitea_assess_master_parity` publishes `startup_head` / `local_head` / `live_remote_head` / `mutation_safe` as the authoritative mutation-safety verdict. | The tool's contract is consumed by every mutation caller and by the operator; it must keep its shape while its semantics are redefined, or every consumer breaks at once. | needs a replacement | #935 |
|
||||||
|
| F5 | `gitea_mcp_server.py:23054` | Falls back to `_process_boot_head_sha` — the commit this process booted at — when the parity payload has no `startup_head`. | Same defect as F1, in a fallback path that is easy to miss when F1 is fixed. | needs a seam | #935 |
|
||||||
|
| F6 | `gitea_mcp_server.py:20615` | Staleness is also inferred from `os.path.getmtime()` of `gitea_mcp_server.py` under `PROJECT_ROOT` (`gitea_mcp_server.py:20611`), compared against peer process start times. | File mtime on a deployed artifact tracks the deploy, not the operator's edits, and the peer start times it is compared against come from the unavailable process table (P4). | cannot be remote | #935 |
|
||||||
|
|
||||||
|
## 6. Local filesystem
|
||||||
|
|
||||||
|
Author and reviewer tools act directly on the operator's checkout.
|
||||||
|
|
||||||
|
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||||
|
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||||
|
| L1 | `gitea_mcp_server.py:10122` | `gitea_bootstrap_author_issue_worktree` creates and binds a git worktree on the server's own disk. | The remote host has no operator checkout to add a worktree to. Executing this remotely would act on the wrong disk while reporting success. | cannot be remote | #936 |
|
||||||
|
| L2 | `gitea_mcp_server.py:190` | `ACTIVE_WORKTREE_ENV = "GITEA_ACTIVE_WORKTREE"` and `AUTHOR_WORKTREE_ENV` (`gitea_mcp_server.py:191`) carry the active workspace as process-wide environment. | Process-wide workspace state cannot represent per-session workspaces on a shared endpoint. | needs a replacement | #936 |
|
||||||
|
| L3 | `gitea_mcp_server.py:9801` | Binding a worktree writes `os.environ["GITEA_AUTHOR_WORKTREE"]` and `os.environ["GITEA_ACTIVE_WORKTREE"]` (`gitea_mcp_server.py:9802`), mutating global process state. | One session's bind would silently retarget every other concurrent session in the same process. This is a correctness bug the moment concurrency is real. | needs a replacement | #936 |
|
||||||
|
| L4 | `reviewer_inventory_worktree.py:48` | `_BRANCHES_WORKTREE_RE = re.compile(r"\bbranches/", re.I)` requires review worktree paths to sit under `branches/`. | A path convention on the operator's machine, asserted as a validation rule. It needs to become a property of a declared workspace, not a substring test. | needs a seam | #936 |
|
||||||
|
| L5 | `stable_control_runtime.py:54` | `DEV_WORKTREE_SEGMENT = "branches"` classifies a process root as a development worktree by path segment. | Same class of assumption as L4, on the runtime-classification side. | needs a seam | #936 |
|
||||||
|
| L6 | `mcp_server.py:42` | `check_conflict_markers()` runs at import and `os.walk`s the install directory for unresolved conflict markers, `sys.exit(1)` on a hit. | On a remote host it scans a deployed artifact, which by construction never has conflict markers — so the guard passes trivially and stops protecting the thing it was written to protect. | needs a replacement | #936 |
|
||||||
|
| L7 | `role_session_router.py:487` | `check_mid_merge()` reports infra-stop from `.git/MERGE_HEAD`, `rebase-merge`, `rebase-apply` and a source conflict scan under the server's project root. | Same inversion as L6: it would report the deployment's git state, not the operator's. | needs a replacement | #936 |
|
||||||
|
| L8 | `author_issue_bootstrap.py:996` | Enumerates worktrees with `git -C <root> worktree list --porcelain`. | Requires a real local clone with real worktrees; there is nothing equivalent to enumerate remotely. | cannot be remote | #936 |
|
||||||
|
| L9 | `mcp_server.py:10` | Redirects `sys.stderr` to the fixed path `/tmp/mcp_server_stderr.log` outside pytest. | A single fixed `/tmp` path is shared by every concurrent server on a host and is not a deployment's logging surface. | needs a replacement | #938 |
|
||||||
|
| L10 | `gitea_mcp_server.py:2314` | `ISSUE_LOCK_FILE = "/tmp/gitea_issue_lock.json"` — the legacy single global lock slot. | One global `/tmp` slot per host cannot represent concurrent remote sessions and is world-visible on a shared machine. | needs a replacement | #937 |
|
||||||
|
| L11 | `issue_lock_provenance.py:14` | `ISSUE_LOCK_FILE = os.environ.get("GITEA_ISSUE_LOCK_FILE", "/tmp/gitea_issue_lock.json")` keeps the same `/tmp` default in the provenance path. | Same as L10; the env override is a local escape hatch, not a remote design. | needs a replacement | #937 |
|
||||||
|
|
||||||
|
## 7. Durable state
|
||||||
|
|
||||||
|
Locks, leases, session state, and the control-plane database live in the operator's home
|
||||||
|
directory and are keyed on local PIDs.
|
||||||
|
|
||||||
|
| ID | Anchor | Assumes today | Observes remotely | Class | Owner |
|
||||||
|
| -- | ------ | ------------- | ----------------- | ----- | ----- |
|
||||||
|
| S1 | `issue_lock_store.py:26` | `DEFAULT_LOCK_DIR = ~/.cache/gitea-tools/issue-locks` — per-issue lock files under one user's home. | A shared endpoint has no single operator home; per-user paths make locks invisible across sessions and hosts. | needs a replacement | #937 |
|
||||||
|
| S2 | `issue_lock_store.py:83` | `session_pointer_path()` names the session pointer file `session-<os.getpid()>.json`. | Many sessions share one PID on a remote server, so the pointer collapses to a single slot and sessions overwrite each other. | cannot be remote | #937 |
|
||||||
|
| S3 | `issue_lock_store.py:98` | `is_process_alive(pid)` decides lock liveness by probing the local process table. | A PID recorded by one host is meaningless on another, and may coincidentally match a live unrelated process. | cannot be remote | #937 |
|
||||||
|
| S4 | `issue_lock_store.py:213` | Lock records stamp `session_pid` and `pid` from `os.getpid()`. | The recorded identity no longer distinguishes sessions; ownership checks silently pass for the wrong caller. | needs a replacement | #937 |
|
||||||
|
| S5 | `mcp_session_state.py:27` | `DEFAULT_STATE_DIR = ~/.cache/gitea-tools/session-state`, mode `0o700`. | Same home-directory coupling as S1, for review decision locks and workflow proofs. | needs a replacement | #937 |
|
||||||
|
| S6 | `mcp_session_state.py:559` | Session bodies stamp `session_pid` and `writer_pid` from `os.getpid()` (`mcp_session_state.py:560`). | Writer attribution collapses across concurrent sessions in one process. | needs a replacement | #937 |
|
||||||
|
| S7 | `control_plane_db.py:47` | `DEFAULT_DB_PATH = ~/.cache/gitea-tools/control-plane/control_plane.sqlite3`. | A per-user SQLite file is not reachable by, or safe for, multiple remote sessions or multiple hosts. | needs a replacement | #937 |
|
||||||
|
| S8 | `control_plane_db.py:386` | `sqlite3.connect(self.db_path, timeout=30)` — single-writer file locking tuned for one local process. | SQLite's write lock does not extend across hosts and degrades sharply under real concurrency; the store needs a concurrency-safe backend. | needs a replacement | #937 |
|
||||||
|
| S9 | `control_plane_db.py:1145` | Lease rows record `owner_pid` defaulting to `os.getpid()` (also `control_plane_db.py:2039`). | PID-keyed lease ownership is unusable across hosts and ambiguous within one shared process. | cannot be remote | #937 |
|
||||||
|
| S10 | `mcp_daemon_guard.py:53` | `_DEFAULT_SESSION_STATE_DIR` is pinned once at transport bind so a later `GITEA_MCP_SESSION_STATE_DIR` change cannot manufacture a second authority domain (#695 AC2). | The single-authority-domain invariant is exactly right and must be preserved; only its backing location needs to move. | needs a seam | #937 |
|
||||||
|
| S11 | `gitea_mcp_server.py:11875` | Reviewer-lease reclaim reads `owner_pid_alive` from the lease freshness record to decide whether an owner is dead. | Consumes S3/S9; a false "owner alive" or "owner dead" here reclaims or refuses a live lease. This is the highest-consequence consumer of PID liveness. | cannot be remote | #937 |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Summary
|
||||||
|
|
||||||
|
### Entries per category
|
||||||
|
|
||||||
|
| Category | Entries |
|
||||||
|
| -------- | ------: |
|
||||||
|
| 1. Transport bind | 9 |
|
||||||
|
| 2. Launch provenance | 12 |
|
||||||
|
| 3. Role binding | 7 |
|
||||||
|
| 4. Credentials | 6 |
|
||||||
|
| 5. Runtime freshness | 6 |
|
||||||
|
| 6. Local filesystem | 11 |
|
||||||
|
| 7. Durable state | 11 |
|
||||||
|
| **Total** | **62** |
|
||||||
|
|
||||||
|
No category is empty, so no "this category has no coupling" justification is required.
|
||||||
|
|
||||||
|
### Entries per classification
|
||||||
|
|
||||||
|
| Classification | Entries |
|
||||||
|
| -------------- | ------: |
|
||||||
|
| portable as written | 5 |
|
||||||
|
| needs a seam | 16 |
|
||||||
|
| needs a replacement | 26 |
|
||||||
|
| cannot be remote | 15 |
|
||||||
|
| **Total** | **62** |
|
||||||
|
|
||||||
|
### Category × classification
|
||||||
|
|
||||||
|
| Category | portable | seam | replacement | cannot | Total |
|
||||||
|
| -------- | -------: | ---: | ----------: | -----: | ----: |
|
||||||
|
| 1. Transport bind | 4 | 4 | 1 | 0 | 9 |
|
||||||
|
| 2. Launch provenance | 0 | 2 | 5 | 5 | 12 |
|
||||||
|
| 3. Role binding | 0 | 3 | 3 | 1 | 7 |
|
||||||
|
| 4. Credentials | 1 | 2 | 2 | 1 | 6 |
|
||||||
|
| 5. Runtime freshness | 0 | 2 | 2 | 2 | 6 |
|
||||||
|
| 6. Local filesystem | 0 | 2 | 7 | 2 | 11 |
|
||||||
|
| 7. Durable state | 0 | 1 | 6 | 4 | 11 |
|
||||||
|
| **Total** | **5** | **16** | **26** | **15** | **62** |
|
||||||
|
|
||||||
|
### Entries per epic child
|
||||||
|
|
||||||
|
Every child from 2 through 10 is named by at least one entry, and every entry names exactly
|
||||||
|
one child.
|
||||||
|
|
||||||
|
| Child | Issue | Title | Entries | IDs |
|
||||||
|
| ----: | ----- | ----- | ------: | --- |
|
||||||
|
| 2 | #931 | Transport-neutral bind seam | 9 | T1–T9 |
|
||||||
|
| 3 | #932 | Per-request principal resolution | 7 | R1–R7 |
|
||||||
|
| 4 | #933 | Server-side credential provider | 6 | C1–C6 |
|
||||||
|
| 5 | #934 | Remote-session provenance | 9 | P1–P9 |
|
||||||
|
| 6 | #935 | Redefined master-parity gate | 6 | F1–F6 |
|
||||||
|
| 7 | #936 | Local-filesystem vs remotable tool split | 8 | L1–L8 |
|
||||||
|
| 8 | #937 | Concurrency-safe session, lock, and lease state | 13 | L10, L11, S1–S11 |
|
||||||
|
| 9 | #938 | Authenticated remote MCP endpoint | 2 | P10, L9 |
|
||||||
|
| 10 | #939 | Dual-run cutover and rollback | 2 | P11, P12 |
|
||||||
|
| | | **Total** | **62** | |
|
||||||
|
|
||||||
|
## Notes for downstream children
|
||||||
|
|
||||||
|
- **The three highest-risk entries are P5, F2, and S11.** Each is a guard that does not
|
||||||
|
merely stop working remotely — it inverts. P5 turns concurrency into a reported fault,
|
||||||
|
F2 returns a verdict computed from terms that no longer mean anything, and S11 reclaims
|
||||||
|
or refuses leases on a PID-liveness answer that is wrong rather than unknown. A gate that
|
||||||
|
fails open while still reporting green is worse than one that fails to start.
|
||||||
|
- **T4, T5, T7, T8, and C5 are the portable core.** They show the target shape: predicates
|
||||||
|
over injected inputs, with no reference to the host, the process table, or the operator's
|
||||||
|
disk.
|
||||||
|
- **The keychain seam already exists** at C2 (`resolve_token`'s injectable `keychain_lookup`).
|
||||||
|
#933 should widen that seam rather than introduce a parallel path, and must remember C4 —
|
||||||
|
the guard protecting the old mechanism has to be re-pointed, or the protection lapses
|
||||||
|
silently when the mechanism is replaced.
|
||||||
|
- **`branches/` appears as a validation rule in at least two independent places** (L4, L5).
|
||||||
|
Path-substring conventions tend to have more copies than expected; #936 should re-grep
|
||||||
|
rather than trust this list to be exhaustive for that specific pattern.
|
||||||
@@ -0,0 +1,246 @@
|
|||||||
|
{
|
||||||
|
"_comment": [
|
||||||
|
"Machine-checkable anchor table for docs/remote-mcp/threat-model.md (#956).",
|
||||||
|
"Every file:line anchor cited in the threat model must appear here, and the",
|
||||||
|
"source line at that anchor must contain the 'expect' substring.",
|
||||||
|
"tests/test_issue_956_threat_model.py enforces both directions, so a refactor",
|
||||||
|
"that shifts a line number fails the suite instead of silently rotting the",
|
||||||
|
"document. #930's inventory had no such guard and its gitea_mcp_server.py",
|
||||||
|
"anchors drifted between 7bf4f125 and aad5c8b4."
|
||||||
|
],
|
||||||
|
"generated_against_commit": "1dd30ecb1508b559868c2d5d94367bc055d5138e",
|
||||||
|
"anchors": [
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:25087",
|
||||||
|
"expect": "mcp_daemon_guard.bind_native_mcp_transport()"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "mcp_daemon_guard.py:49",
|
||||||
|
"expect": "_PRODUCTION_TRANSPORTS = mcp_transport_config.SUPPORTED_TRANSPORTS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "mcp_daemon_guard.py:195",
|
||||||
|
"expect": "def bind_native_mcp_transport"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "irrecoverable_provenance.py:497",
|
||||||
|
"expect": "def assess_transport_for_auth_mint"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:9192",
|
||||||
|
"expect": "assess_transport_for_auth_mint()"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:9441",
|
||||||
|
"expect": "assess_transport_for_auth_mint()"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "mcp_server.py:4",
|
||||||
|
"expect": "The transport is selected by deployment configuration"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:15630",
|
||||||
|
"expect": "def _is_client_managed_process"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:15644",
|
||||||
|
"expect": "def _provenance_mutation_block"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:15652",
|
||||||
|
"expect": "unsupported_manual_launch"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:19217",
|
||||||
|
"expect": "server_provenance"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:21741",
|
||||||
|
"expect": "def _check_mcp_runtimes_diagnostics"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:21761",
|
||||||
|
"expect": "\"ps\", \"-o\", \"pid,lstart,command\""
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:21805",
|
||||||
|
"expect": "\"ps\", \"eww\""
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_config.py:1172",
|
||||||
|
"expect": "RECOGNIZED_GITEA_ENV_KEYS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_config.py:1233",
|
||||||
|
"expect": "GITEA_CLIENT_MANAGED"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_config.py:54",
|
||||||
|
"expect": "ENV_PROFILE = \"GITEA_MCP_PROFILE\""
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_config.py:97",
|
||||||
|
"expect": "_REVIEW_MERGE_OPS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_config.py:499",
|
||||||
|
"expect": "repository authorization scope"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_config.py:956",
|
||||||
|
"expect": "def _keychain_token"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_config.py:974",
|
||||||
|
"expect": "def resolve_token"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_config.py:1015",
|
||||||
|
"expect": "def keychain_auth"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_config.py:294",
|
||||||
|
"expect": "def _validate_identity_auth"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "mcp_daemon_guard.py:583",
|
||||||
|
"expect": "def assert_keychain_access_allowed"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:19487",
|
||||||
|
"expect": "def gitea_list_profiles"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:19538",
|
||||||
|
"expect": "gitea_config.resolve_token(p)"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:19851",
|
||||||
|
"expect": "def gitea_audit_config"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:19873",
|
||||||
|
"expect": "service_summaries(config)"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_config.py:704",
|
||||||
|
"expect": "def resolve_service"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_config.py:837",
|
||||||
|
"expect": "def service_summaries"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_config.py:851",
|
||||||
|
"expect": "_keychain_token(auth.get(\"id\"))"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:17909",
|
||||||
|
"expect": "\"jenkins-mcp\""
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:17915",
|
||||||
|
"expect": "external-mcp"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:17936",
|
||||||
|
"expect": "\"glitchtip-mcp\""
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:17941",
|
||||||
|
"expect": "external-mcp"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "mcp_discoverability.py:9",
|
||||||
|
"expect": "EXPECTED_JENKINS_TOOLS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "mcp_discoverability.py:17",
|
||||||
|
"expect": "EXPECTED_GLITCHTIP_TOOLS"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "sentry_incident_bridge.py:36",
|
||||||
|
"expect": "SENTRY_AUTH_TOKEN"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "sentry_incident_bridge.py:190",
|
||||||
|
"expect": "def resolve_token"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "sentry_incident_bridge.py:289",
|
||||||
|
"expect": "Authorization"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "sentry_observability.py:55",
|
||||||
|
"expect": "SENTRY_DSN"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "master_parity_gate.py:168",
|
||||||
|
"expect": "def capture_startup_parity"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "master_parity_gate.py:255",
|
||||||
|
"expect": "mutation_safe"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:19331",
|
||||||
|
"expect": "def gitea_assess_master_parity"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:193",
|
||||||
|
"expect": "ACTIVE_WORKTREE_ENV"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:194",
|
||||||
|
"expect": "AUTHOR_WORKTREE_ENV"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:2352",
|
||||||
|
"expect": "/tmp/gitea_issue_lock.json"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:10957",
|
||||||
|
"expect": "def gitea_bootstrap_author_issue_worktree"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "mcp_server.py:13",
|
||||||
|
"expect": "/tmp/mcp_server_stderr.log"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "issue_lock_store.py:26",
|
||||||
|
"expect": "DEFAULT_LOCK_DIR"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "issue_lock_store.py:83",
|
||||||
|
"expect": "def session_pointer_path"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "issue_lock_store.py:98",
|
||||||
|
"expect": "def is_process_alive"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "mcp_session_state.py:27",
|
||||||
|
"expect": "DEFAULT_STATE_DIR"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "control_plane_db.py:47",
|
||||||
|
"expect": "DEFAULT_DB_PATH"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "control_plane_db.py:380",
|
||||||
|
"expect": "mode=0o700"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "control_plane_db.py:386",
|
||||||
|
"expect": "sqlite3.connect"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "control_plane_db.py:1145",
|
||||||
|
"expect": "os.getpid()"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"anchor": "gitea_mcp_server.py:12871",
|
||||||
|
"expect": "owner_pid_alive"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
@@ -0,0 +1,409 @@
|
|||||||
|
# Remote-MCP threat model, trust boundaries, and service decomposition
|
||||||
|
|
||||||
|
What the adversary is, what each boundary protects, and which services may share a process.
|
||||||
|
|
||||||
|
- **Issue:** #956 (Remote-MCP threat model), child of epic #929, cross-linked to #955.
|
||||||
|
- **Depends on:** #930 (closed) — `docs/remote-mcp/coupling-inventory.md`.
|
||||||
|
- **Blocks:** #932, #933, #934, #938.
|
||||||
|
- **Generated against commit:** `1dd30ecb1508b559868c2d5d94367bc055d5138e` (#708's
|
||||||
|
namespace-attachment gate). Originally generated against
|
||||||
|
`aad5c8b42361d380a8eeb07b94b90815e594c2c5` (`master`), re-anchored at
|
||||||
|
`a143cd065ba06e1a2bdc5143a19ec156e53650ef` when #931's transport bind seam shifted the
|
||||||
|
cited lines, and re-anchored again when #708 shifted them further.
|
||||||
|
- **Scope:** documentation only. This child changes no server behavior. It adds one
|
||||||
|
document, one anchor fixture, and the test that enforces them.
|
||||||
|
|
||||||
|
## Relationship to #930
|
||||||
|
|
||||||
|
#930 asked *what breaks when the process stops being local*. This document asks *what an
|
||||||
|
attacker gets, and where we stop them*. The two are deliberately different axes: #930
|
||||||
|
classifies each coupling as portable, seam, replacement, or cannot-be-remote; this document
|
||||||
|
classifies each **credential** by blast radius and each **boundary** by what crossing it
|
||||||
|
requires. An entry can be perfectly portable and still be a trust disaster —
|
||||||
|
`gitea_config.py:851` is portable Python that reads a CI secret from inside the Gitea server.
|
||||||
|
|
||||||
|
### Anchors are enforced, not asserted
|
||||||
|
|
||||||
|
Every `file:line` in this document is declared in `docs/remote-mcp/threat-model-anchors.json`
|
||||||
|
with the substring that must appear at that line, and
|
||||||
|
`tests/test_issue_956_threat_model.py` fails if any anchor does not resolve or if the
|
||||||
|
document cites an anchor the fixture does not cover.
|
||||||
|
|
||||||
|
This guard exists because #930 did not have one. Its inventory was generated at
|
||||||
|
`7bf4f125`; by `aad5c8b4` its `gitea_mcp_server.py` anchors had drifted — the transport
|
||||||
|
bind it cited at line 23750 now lives at `gitea_mcp_server.py:25087`, and its
|
||||||
|
client-managed provenance anchor at 14588 now lands in an unrelated function. Nothing
|
||||||
|
failed, because nothing checked. Anchors into a ~24,700-line module rot silently, and a
|
||||||
|
security document that cannot prove its own citations is worse than none, because it is
|
||||||
|
trusted.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 1. Assets
|
||||||
|
|
||||||
|
What an adversary wants. Ordered by consequence, not by likelihood.
|
||||||
|
|
||||||
|
| ID | Asset | Why it matters |
|
||||||
|
| -- | ----- | -------------- |
|
||||||
|
| A1 | Merge authority on `Scaled-Tech-Consulting/Gitea-Tools` | This repository *is* the control plane. Code merged here becomes the gate that authorizes every future mutation, so merge authority is self-amplifying: one merge can disable every other control in this document. |
|
||||||
|
| A2 | Write authority on the `mdcps` tenant | A second, unrelated organization reachable from the same configuration. Compromise here is a cross-organization incident, not an internal one. |
|
||||||
|
| A3 | The eight Gitea role credentials | Long-lived bearer tokens. Possession is authority; there is no second factor at the API. |
|
||||||
|
| A4 | Jenkins read access (`mdcps`, enabled) | Build logs routinely carry deployment topology, internal hostnames, and accidentally-echoed secrets. |
|
||||||
|
| A5 | Error-tracking read access (GlitchTip / Sentry) | Event payloads carry stack frames, request context, and production user data. |
|
||||||
|
| A6 | Coordination-state integrity | The locks, leases, and review-decision records that make "exactly one owner" true. Corrupting them needs no Gitea credential and produces duplicate or lost work. |
|
||||||
|
| A7 | The operator's checkout and worktrees | Unmerged code, branch state, and the filesystem the author tools write to. |
|
||||||
|
| A8 | The macOS login keychain | The meta-credential. Everything in A3, A4, and A5 resolves from it. |
|
||||||
|
| A9 | Separation of duty between review and merge | The property that no single actor both approves and lands a change. An *asset*, not a control, because it is what the controls exist to produce. |
|
||||||
|
| A10 | Audit and provenance records | Determine whether an incident is reconstructable. An attacker who can forge provenance makes an intrusion indistinguishable from normal work. |
|
||||||
|
|
||||||
|
## 2. Adversaries
|
||||||
|
|
||||||
|
| ID | Adversary | Capability assumed | Not assumed |
|
||||||
|
| -- | --------- | ------------------ | ----------- |
|
||||||
|
| ADV1 | **Compromised LLM client** | Full control of one MCP client. Issues arbitrary tool calls, in any order, with any arguments, at machine speed. Sees every tool result. | Cannot read the operator's disk except through tools; cannot execute arbitrary local code outside the tool surface. |
|
||||||
|
| ADV2 | **Prompt injection** via repository content | Controls text the model reads and treats as instruction — issue bodies, PR descriptions, review comments, commit messages, file contents. Reaches the model on any read of untrusted content. | Holds no credential and issues no call directly. Its entire power is causing an *authorized* client to act. |
|
||||||
|
| ADV3 | **Malicious tool arguments** | Supplies hostile values to any parameter — paths, branch names, session identifiers, worktree paths, issue numbers — including traversal, injection, and confusion between look-alike identifiers. | Cannot bypass a gate that actually validates its input. |
|
||||||
|
| ADV4 | **Network attacker** | Observes and modifies traffic between client, server, and Gitea. Attempts downgrade, replay, and endpoint impersonation. | Does not hold a valid credential at the start. |
|
||||||
|
| ADV5 | **Curious operator** | Legitimate local access to the workstation: process table, `/tmp`, home directory, keychain prompts. Not malicious, but not authorized for every role either. | Does not defeat the OS keychain's own access control without a prompt. |
|
||||||
|
|
||||||
|
ADV2 is the adversary this architecture most under-models. Every other adversary must first
|
||||||
|
obtain something. Prompt injection obtains nothing: it borrows authority the client already
|
||||||
|
holds and is indistinguishable at the tool boundary from legitimate work. Each boundary
|
||||||
|
below therefore states whether it constrains ADV2 at all — and most do not, because they
|
||||||
|
authenticate the *caller*, not the *intent*.
|
||||||
|
|
||||||
|
## 3. Trust boundaries
|
||||||
|
|
||||||
|
"Crossing requires today" is what the code actually enforces at
|
||||||
|
`1dd30ecb1508b559868c2d5d94367bc055d5138e`, not what the design intends.
|
||||||
|
|
||||||
|
| ID | Boundary | Protects | Crossing requires today | Crossing must require remotely |
|
||||||
|
| -- | -------- | -------- | ----------------------- | ------------------------------ |
|
||||||
|
| B1 | LLM client ↔ MCP server session | A1, A3, A10 — that a mutating session was established through the sanctioned client path | A single configured bind (`gitea_mcp_server.py:25087`) validated against one closed allowlist (`mcp_daemon_guard.py:49`, `mcp_daemon_guard.py:195`) — since #931 the identifier comes from deployment configuration and defaults to the local transport, so the boundary no longer rests on a literal, but it still rests on the *bind* rather than on an authenticated caller; client-managed provenance (`gitea_mcp_server.py:15630`) or a refusal (`gitea_mcp_server.py:15652`); production transport before recovery-authorization mint (`irrecoverable_provenance.py:497`, consumed at `gitea_mcp_server.py:9192` and `gitea_mcp_server.py:9441`) | An authenticated handshake issuing a server-side session identity bound to a principal, with the transport recorded in provenance. The physical proof (a pipe) must become a cryptographic one. |
|
||||||
|
| B2 | Role ↔ role | A9 — that author, reviewer, merger, and reconciler are distinct authorities | **The process boundary only.** The role is a property of the process, read once from `GITEA_MCP_PROFILE` (`gitea_config.py:54`). A caller gets author permissions by connecting to the author process. Review and merge are the operations singled out for extra care (`gitea_config.py:97`) | A per-request principal, so the role follows from the credential presented and cannot be selected by reaching a different endpoint. |
|
||||||
|
| B3 | MCP server ↔ credential store | A3, A8 — that only sanctioned code turns a profile into a token | `_keychain_token` shelling out to the login keychain (`gitea_config.py:956`), dispatched by `resolve_token` (`gitea_config.py:974`) with the reference type built at `gitea_config.py:1015`, gated by `assert_keychain_access_allowed` (`mcp_daemon_guard.py:583`). Inline secrets are rejected at config load (`gitea_config.py:294`) | A credential provider keyed by the *request* principal, returning only that principal's credential, with the source recorded and the value never returned. |
|
||||||
|
| B4 | MCP server ↔ Gitea | A1, A2 — that only authorized calls reach the forge | A bearer token over TLS. Server-side, nothing distinguishes one role's token from another beyond the account it belongs to | Unchanged at the forge; the endpoint in front of it must refuse unauthenticated and plaintext connections before tool dispatch. |
|
||||||
|
| B5 | MCP server ↔ caller's filesystem | A7 — that a tool acts on the *caller's* disk or refuses | Nothing. The server's disk *is* the caller's disk. Worktree bootstrap writes directly (`gitea_mcp_server.py:10957`); the active workspace is process-global (`gitea_mcp_server.py:193`, `gitea_mcp_server.py:194`) | An explicit per-tool classification, enforced at dispatch, refusing filesystem tools over a transport that cannot reach the caller's disk. A green verdict about the wrong disk is the failure to prevent. |
|
||||||
|
| B6 | MCP server ↔ coordination state | A6, A9 — mutual exclusion | Local files and a local SQLite database, with liveness judged from the local process table (`issue_lock_store.py:98`), keyed on paths under one user's home (`issue_lock_store.py:26`, `mcp_session_state.py:27`, `control_plane_db.py:47`) and on `os.getpid()` (`control_plane_db.py:1145`, `gitea_mcp_server.py:12871`). A legacy global slot still exists at `gitea_mcp_server.py:2352`, and the session-pointer file is named per PID (`issue_lock_store.py:83`) | One authority per ownership question, with liveness from session identity and expiry, and atomic acquire, renew, and release across hosts. |
|
||||||
|
| B7 | Gitea integration ↔ unrelated integrations | A4, A5 — that a Gitea compromise is not a CI and observability compromise | **Nothing.** See §5. The Gitea server reads Jenkins and GlitchTip secrets (`gitea_config.py:851`, reached from `gitea_config.py:837`) and holds the Sentry token (`sentry_incident_bridge.py:190`) | A hard process boundary. This is the boundary #956 exists to create. |
|
||||||
|
| B8 | Tenant ↔ tenant (`prgs` / `mdcps` / `local-lab`) | A2 — that one organization's compromise is not another's | Convention. One configuration declares all three contexts; `resolve_service` fails closed on a *disabled* context (`gitea_config.py:704`) but the credentials of enabled ones remain reachable in-process. A per-profile repository scope exists (`gitea_config.py:499`) | Separate deployments, or at minimum per-tenant credential scopes with no process able to resolve both. |
|
||||||
|
| B9 | Deployed code ↔ merged policy | A1, A10 — that the running server enforces the rules that were actually merged | Comparing this process's startup commit against this disk (`master_parity_gate.py:168`), conjoined into a single verdict (`master_parity_gate.py:255`) published by `gitea_mcp_server.py:19331` | Freshness defined against the deployed build identity, with an explicit fail-closed verdict when undeterminable. |
|
||||||
|
|
||||||
|
### What no boundary constrains
|
||||||
|
|
||||||
|
None of B1–B9 constrains **ADV2**. Every one authenticates a caller or a process; prompt
|
||||||
|
injection supplies neither. An injected instruction that reaches an authorized author
|
||||||
|
session crosses B1, B2, B3, and B5 legitimately, because at each of those boundaries it *is*
|
||||||
|
the author. The only controls that bite ADV2 are those constraining what an authenticated
|
||||||
|
principal may do regardless of what it asks for — the per-role permission split (B2), the
|
||||||
|
repository scope at `gitea_config.py:499`, and separation of duty (A9). Sizing those
|
||||||
|
controls correctly matters more after the migration, not less, because a remote endpoint
|
||||||
|
raises the number of clients that can be injected into.
|
||||||
|
|
||||||
|
## 4. Data flows
|
||||||
|
|
||||||
|
Flows that cross a boundary. `==>` carries a credential; `-->` does not.
|
||||||
|
|
||||||
|
```
|
||||||
|
B1 B4
|
||||||
|
[LLM client] ====================> [MCP server] ========> [Gitea]
|
||||||
|
^ stdio pipe today | ^ (A1,A2)
|
||||||
|
| session identity | |
|
||||||
|
| after migration | |
|
||||||
|
| | | B3
|
||||||
|
untrusted repository content | +======> [macOS login keychain] (A8)
|
||||||
|
read back into the model (ADV2) | resolves A3, A4, A5
|
||||||
|
^ |
|
||||||
|
+----------------------------------+
|
||||||
|
|
|
||||||
|
B5 | B6
|
||||||
|
[operator checkout / worktrees] <--------+-------> [locks · leases · sqlite]
|
||||||
|
(A7) | (A6)
|
||||||
|
|
|
||||||
|
B7 <-- boundary does not exist today
|
||||||
|
|
|
||||||
|
+========================+========================+
|
||||||
|
| | |
|
||||||
|
[Jenkins] (A4) [GlitchTip] (A5) [Sentry] (A5)
|
||||||
|
external MCP server external MCP server in-process bridge
|
||||||
|
```
|
||||||
|
|
||||||
|
Two flows deserve attention because neither is obvious from the code:
|
||||||
|
|
||||||
|
1. **The keychain flow fans out.** B3 is drawn once but resolves credentials for *every*
|
||||||
|
configured profile and service, not only the active one. `gitea_list_profiles`
|
||||||
|
(`gitea_mcp_server.py:19487`) reports each profile's credential status by calling
|
||||||
|
`resolve_token` on it (`gitea_mcp_server.py:19538`), and `gitea_audit_config`
|
||||||
|
(`gitea_mcp_server.py:19851`) reports service credential status through
|
||||||
|
`service_summaries` (`gitea_mcp_server.py:19873`).
|
||||||
|
2. **The return path is a flow too.** Content read from Gitea travels back into the model
|
||||||
|
and is treated as instruction. This is the ADV2 edge, and it is the only edge in the
|
||||||
|
diagram with no authentication on it, because it is not a request.
|
||||||
|
|
||||||
|
## 5. Per-boundary credential inventory
|
||||||
|
|
||||||
|
**14 credentials in total.** Blast radius is stated as what the credential yields *on its
|
||||||
|
own*, assuming every gate not backed by the credential itself has been bypassed — because
|
||||||
|
an attacker holding a token calls the API, not our tools.
|
||||||
|
|
||||||
|
| ID | Credential | Holder | Boundary | Blast radius |
|
||||||
|
| -- | ---------- | ------ | -------- | ------------ |
|
||||||
|
| CR1 | `prgs-author` Gitea token — account `jcwalker3` | macOS keychain; resolved in-process (`gitea_config.py:974`) | B3 → B4 | Create branches, push, commit, open PRs, create/close/comment issues on the control-plane repo. Cannot approve or merge. The one credential whose identity is genuinely distinct. |
|
||||||
|
| CR2 | `prgs-reviewer` Gitea token — account `sysadmin` | macOS keychain | B3 → B4 | Approve and request changes. **Shares one Gitea account with CR3, CR4, CR5.** |
|
||||||
|
| CR3 | `prgs-merger` Gitea token — account `sysadmin` | macOS keychain | B3 → B4 | Merge to `master` — A1 in full. Same account as CR2. |
|
||||||
|
| CR4 | `prgs-reconciler` Gitea token — account `sysadmin` | macOS keychain | B3 → B4 | Close PRs, delete branches, irrecoverable decision-lock recovery. Same account as CR2. |
|
||||||
|
| CR5 | `prgs-controller` Gitea token — account `sysadmin` | macOS keychain | B3 → B4 | Same operation set as CR4. Same account as CR2. |
|
||||||
|
| CR6 | `mdcps-author` Gitea token — account `913443` | macOS keychain | B3 → B4, B8 | Author operations on a second organization. **Shares one account with CR7 and CR8.** |
|
||||||
|
| CR7 | `mdcps-reviewer` Gitea token — account `913443` | macOS keychain | B3 → B4, B8 | Approve and request changes on `mdcps`. Same account as CR6. |
|
||||||
|
| CR8 | `mdcps-merger` Gitea token — account `913443` | macOS keychain | B3 → B4, B8 | Merge on `mdcps` — A2 in full. Same account as CR6. |
|
||||||
|
| CR9 | MDCPS Jenkins read credential | macOS keychain, read from the Gitea server process (`gitea_config.py:851`) | B7 | Read CI jobs, builds, and logs (A4). Enabled today. |
|
||||||
|
| CR10 | MDCPS GlitchTip read credential | macOS keychain, read from the Gitea server process (`gitea_config.py:851`) | B7 | Read error events and their payloads (A5). Enabled today. |
|
||||||
|
| CR11 | `SENTRY_AUTH_TOKEN` | Process environment, read in-process (`sentry_incident_bridge.py:36`, `sentry_incident_bridge.py:190`), sent as a bearer header (`sentry_incident_bridge.py:289`) | B7 | Read and reconcile Sentry issues (A5). Not a keychain credential — an env var, so it is inherited by anything the process spawns. |
|
||||||
|
| CR12 | `SENTRY_DSN` | Process environment (`sentry_observability.py:55`) | B7 | Write events into the observability project. Low read value, real forgery value: an attacker can inject fabricated events into the record (A10). |
|
||||||
|
| CR13 | macOS login keychain access | The operator's login session; gated by `assert_keychain_access_allowed` (`mcp_daemon_guard.py:583`) | B3, ADV5 | **Every other credential in this table except CR11 and CR12.** This is the aggregation point. |
|
||||||
|
| CR14 | Coordination-store access (no secret) | Filesystem permissions — `control_plane_db.py:47`, created `0o700` (`control_plane_db.py:380`), opened with a local file lock (`control_plane_db.py:386`) | B6, ADV5 | Full read/write of locks, leases, and decision records (A6). **There is no credential here at all** — anything running as the operator can rewrite ownership. |
|
||||||
|
|
||||||
|
### Findings
|
||||||
|
|
||||||
|
**Finding 1 — Role separation is not credential separation.** Four `prgs` roles resolve to
|
||||||
|
one Gitea account (`sysadmin`): reviewer, merger, reconciler, and controller. A stolen
|
||||||
|
reviewer credential *is* a merger credential. A9 — separation of duty between approving and
|
||||||
|
landing — is therefore enforced entirely by which local process a call reaches (B2), and not
|
||||||
|
at all by the forge. It survives exactly as long as B2 does, and B2 is the boundary the
|
||||||
|
migration dissolves.
|
||||||
|
|
||||||
|
**Finding 2 — The `mdcps` tenant has no role separation at all.** Author, reviewer, and
|
||||||
|
merger all resolve to account `913443`. One credential can open a PR, approve it, and merge
|
||||||
|
it. The in-process self-review check compares the authenticated username against the PR
|
||||||
|
author and would refuse — but that check runs on our side of B4. It is not a property of
|
||||||
|
the credential, and an attacker holding the token does not call our tools.
|
||||||
|
|
||||||
|
**Finding 3 — Any one role process can resolve every other role's credential.** This is not
|
||||||
|
inferred; it is demonstrated by tool output. `gitea_list_profiles`
|
||||||
|
(`gitea_mcp_server.py:19487`) called from the **author** session reports
|
||||||
|
`identity_status: "credentials present"` for `prgs-merger`, `prgs-reviewer`,
|
||||||
|
`prgs-reconciler`, and every `mdcps` profile, because it calls `resolve_token` on each one
|
||||||
|
(`gitea_mcp_server.py:19538`). The author process does not merely *have access to* the
|
||||||
|
merger's credential — it reads it to answer a status query. B2 is not a credential boundary
|
||||||
|
in either direction.
|
||||||
|
|
||||||
|
**Finding 4 — The Gitea server reads CI and observability secrets.** `gitea_audit_config`
|
||||||
|
(`gitea_mcp_server.py:19851`) reports `MDCPS Jenkins: enabled, read-only, authenticated`.
|
||||||
|
That word `authenticated` is produced by `service_summaries` (`gitea_mcp_server.py:19873`,
|
||||||
|
defined at `gitea_config.py:837`), whose default check calls `_keychain_token` on the
|
||||||
|
service's own keychain reference (`gitea_config.py:851`). Producing that one line requires
|
||||||
|
the Gitea MCP server to read the Jenkins secret and the GlitchTip secret out of the
|
||||||
|
keychain. B7 does not exist.
|
||||||
|
|
||||||
|
**Finding 5 — Jenkins and GlitchTip are already decomposed; the reach is residual.** Their
|
||||||
|
tools live in separately registered servers, marked `external-mcp`
|
||||||
|
(`gitea_mcp_server.py:17909`, `gitea_mcp_server.py:17915`, `gitea_mcp_server.py:17936`,
|
||||||
|
`gitea_mcp_server.py:17941`) with their own expected tool sets (`mcp_discoverability.py:9`,
|
||||||
|
`mcp_discoverability.py:17`). The correct decomposition was already chosen. What remains is
|
||||||
|
a leak across it: the credential *references* still live in the Gitea configuration and are
|
||||||
|
still resolved by the Gitea process. #75 bundled these services into one control-plane
|
||||||
|
umbrella; the tools were separated afterwards, the credentials were not.
|
||||||
|
|
||||||
|
**Finding 6 — Sentry is the exception that is not decomposed.** Unlike Jenkins and
|
||||||
|
GlitchTip, the Sentry bridge runs *inside* the Gitea server, resolving its token from the
|
||||||
|
process environment (`sentry_incident_bridge.py:190`) and sending it as a bearer header
|
||||||
|
(`sentry_incident_bridge.py:289`). Being an environment variable rather than a keychain item
|
||||||
|
makes it strictly worse: it needs no keychain prompt and is inherited by every subprocess the
|
||||||
|
server spawns — including the `ps` invocations at `gitea_mcp_server.py:21761` and
|
||||||
|
`gitea_mcp_server.py:21805`, reached from `gitea_mcp_server.py:21741`.
|
||||||
|
|
||||||
|
**Finding 7 — The highest-value coordination asset has the weakest gate.** A6 is protected
|
||||||
|
by filesystem permissions alone (CR14). Corrupting a lease requires no Gitea credential,
|
||||||
|
produces no forge-side audit record, and breaks the mutual exclusion the entire workflow
|
||||||
|
assumes. Every other asset costs an attacker a credential; this one costs nothing beyond
|
||||||
|
local access, which is exactly ADV5's position.
|
||||||
|
|
||||||
|
**Finding 8 — Provenance authenticates the launch, not the caller.** `server_provenance` is
|
||||||
|
reported as exactly `client_managed` or `manual_launch` (`gitea_mcp_server.py:19217`),
|
||||||
|
derived from environment inspection (`gitea_mcp_server.py:15630`) with the recognized-key
|
||||||
|
allowlist at `gitea_config.py:1172` and the generator that emits the marker at
|
||||||
|
`gitea_config.py:1233`. Every one of those facts is fixed at process start. A client that is
|
||||||
|
trustworthy at launch and compromised a minute later remains `client_managed` for the life
|
||||||
|
of the process, and the transport contract that underwrites it is stated as a property of
|
||||||
|
the server itself (`mcp_server.py:4`). Since #931 that contract names the configured
|
||||||
|
transport rather than asserting stdio, but it is still fixed once, at bind, for the life of
|
||||||
|
the process.
|
||||||
|
|
||||||
|
## 6. Decomposition ruling
|
||||||
|
|
||||||
|
This section is the ruling #956 requires. It is a decision, not a recommendation.
|
||||||
|
|
||||||
|
**D1 — No unrelated co-residency.** A single integration process **must not** hold, resolve,
|
||||||
|
or be able to resolve credentials for services it does not itself integrate with.
|
||||||
|
Concretely: the Gitea MCP service may hold Gitea credentials and nothing else. Jenkins,
|
||||||
|
GlitchTip, Sentry, and any database credential are **not permitted** to co-reside with Gitea
|
||||||
|
credentials in one process.
|
||||||
|
|
||||||
|
*Rationale.* A process is the smallest unit an attacker takes whole. Once ADV1 or ADV2
|
||||||
|
controls execution in a process, every credential that process can resolve is theirs, and no
|
||||||
|
in-process check helps, because the checks are in the process too. Blast radius is therefore
|
||||||
|
a property of the process boundary and nothing finer. Findings 4 and 6 show that today one
|
||||||
|
compromise of the Gitea server yields CI read access, error-tracking read access, and — via
|
||||||
|
CR13 — every role credential on both tenants. That is the single largest reduction in blast
|
||||||
|
radius available anywhere in epic #929, and it costs no new mechanism: the decomposition
|
||||||
|
already exists (Finding 5) and is merely leaked across.
|
||||||
|
|
||||||
|
**D2 — Separation of duty must be backed by credentials.** Two roles whose separation is a
|
||||||
|
security property must not resolve to the same forge account. Specifically, reviewer and
|
||||||
|
merger must be distinct accounts. Today they are not, on either tenant (Findings 1 and 2).
|
||||||
|
|
||||||
|
*Rationale.* B2 is a process boundary, and the migration's entire purpose is to replace
|
||||||
|
process boundaries with request-level ones. A separation enforced only by which process a
|
||||||
|
call reaches does not survive that replacement — and it is already bypassable by anyone who
|
||||||
|
holds the token and calls the API instead of the tool.
|
||||||
|
|
||||||
|
**D3 — Credential resolution is scoped to the request principal.** A session must resolve its
|
||||||
|
own credential and must have no path to any other principal's. The resolve-every-profile
|
||||||
|
behavior behind `gitea_mcp_server.py:19538` and `gitea_mcp_server.py:19873` must report
|
||||||
|
configured-or-not from configuration alone, without resolving the secret.
|
||||||
|
|
||||||
|
*Rationale.* Finding 3. An audit surface that proves a credential exists by fetching it is a
|
||||||
|
credential-aggregation primitive wearing a diagnostic's clothes.
|
||||||
|
|
||||||
|
**D4 — Coordination state is a protected asset with its own authority.** Access to locks,
|
||||||
|
leases, and decision records must require an authenticated session, not merely local
|
||||||
|
filesystem access.
|
||||||
|
|
||||||
|
*Rationale.* Finding 7. #937 already moves this store for concurrency reasons; the
|
||||||
|
authorization requirement must land with it, or the store becomes remotely reachable while
|
||||||
|
still being authorized by nothing.
|
||||||
|
|
||||||
|
### Exceptions
|
||||||
|
|
||||||
|
**One, time-boxed.** During the dual-run window defined by #939, the **local** stdio fleet
|
||||||
|
may continue to resolve Jenkins and GlitchTip credential *references* from the shared
|
||||||
|
configuration, because removing them from the local configuration is not a prerequisite for
|
||||||
|
standing up the remote endpoint and would strand the operator's existing local workflow.
|
||||||
|
|
||||||
|
This exception is bounded by all of:
|
||||||
|
|
||||||
|
- It applies to the local stdio deployment only. The remote endpoint (#938) must be
|
||||||
|
configured with Gitea credentials and no others from its first day.
|
||||||
|
- It expires when #939 completes. It does not survive cutover.
|
||||||
|
- It does not extend to Sentry: CR11 and CR12 are process-environment credentials in the
|
||||||
|
Gitea server (Finding 6) and must be absent from the remote deployment's environment
|
||||||
|
regardless of dual-run state.
|
||||||
|
|
||||||
|
No exception is granted to D2, D3, or D4.
|
||||||
|
|
||||||
|
### Consequences for the target architecture
|
||||||
|
|
||||||
|
- The remote endpoint serves **Gitea only**. It is not a general control-plane endpoint.
|
||||||
|
- Jenkins and GlitchTip keep their existing separate servers, and their credential
|
||||||
|
references move out of the Gitea configuration.
|
||||||
|
- The Sentry bridge either moves behind its own service boundary or is absent from the
|
||||||
|
remote deployment. It does not travel with the Gitea server.
|
||||||
|
- Reviewer and merger accounts diverge before the endpoint is trusted for merges, or A9 is
|
||||||
|
recorded as unenforced.
|
||||||
|
|
||||||
|
## 7. Child-to-boundary mapping
|
||||||
|
|
||||||
|
Every #929 child from 2 through 10, mapped to the boundary it implements. A child
|
||||||
|
implementing more than one boundary names its primary first.
|
||||||
|
|
||||||
|
| Child | Issue | Boundaries | What it must establish | Rulings it must honor |
|
||||||
|
| ----: | ----- | ---------- | ---------------------- | --------------------- |
|
||||||
|
| 2 | #931 | B1, B9 | The bound transport becomes a validated value that provenance and freshness can both key on. Without it neither B1 nor B9 has an input. | — |
|
||||||
|
| 3 | #932 | B2 | The role becomes a property of the request, not the process — the boundary the migration otherwise deletes. | D2, D3 |
|
||||||
|
| 4 | #933 | B3, B7 | Credentials come from a provider keyed by principal. This is where D1 and D3 are either enforced or permanently lost. | D1, D3 |
|
||||||
|
| 5 | #934 | B1 | Session provenance replaces pipe-and-process-table proof with an authenticated session identity. | — |
|
||||||
|
| 6 | #935 | B9 | Freshness redefined against deployed build identity, with an explicit undeterminable verdict. | — |
|
||||||
|
| 7 | #936 | B5 | Every tool classified and the filesystem boundary enforced at dispatch, so a tool cannot return green about the wrong disk. | — |
|
||||||
|
| 8 | #937 | B6 | One authority per ownership question, with session-identity liveness and atomic transitions. | D4 |
|
||||||
|
| 9 | #938 | B4, B1, B8 | The endpoint: authentication, principal binding, transport security, and — critically — the deployed credential set. | D1, D2, D3 |
|
||||||
|
| 10 | #939 | B6 | Dual-run with exactly one coordination authority at every instant, and the rollback that proves the way back. | D1 exception expiry |
|
||||||
|
|
||||||
|
Boundary coverage: B1 (#931, #934, #938), B2 (#932), B3 (#933), B4 (#938), B5 (#936),
|
||||||
|
B6 (#937, #939), B7 (#933), B8 (#938), B9 (#931, #935).
|
||||||
|
|
||||||
|
B7 has exactly one owner, #933, and that is deliberate. B7 is not created by standing up an
|
||||||
|
endpoint; it is created by deciding which credentials a process may resolve, which is
|
||||||
|
precisely what the credential-provider child does.
|
||||||
|
|
||||||
|
## 8. Adversarial walkthrough: a fully compromised LLM client
|
||||||
|
|
||||||
|
The required negative case. ADV1 holds one MCP client and issues arbitrary tool calls. What
|
||||||
|
does it reach at each boundary?
|
||||||
|
|
||||||
|
### Before the migration
|
||||||
|
|
||||||
|
The client is attached to the local fleet over stdio.
|
||||||
|
|
||||||
|
| Boundary | What ADV1 reaches | Stopped by |
|
||||||
|
| -------- | ----------------- | ---------- |
|
||||||
|
| B1 | Everything the fleet serves. The client *is* the sanctioned launcher: it satisfies the client-managed check (`gitea_mcp_server.py:15630`) by construction, and provenance is never re-verified after launch (Finding 8). | Nothing. The guard authenticates the launch, not the caller. |
|
||||||
|
| B2 | All five roles — it is attached to all five namespaces. It can author a PR, approve it from the reviewer namespace, and merge it from the merger namespace. | Only the in-process self-review check, which compares `jcwalker3` (author) against `sysadmin` (reviewer) and **passes**, because Finding 1 made them different accounts while leaving reviewer and merger identical. A9 falls in one sequence of legitimate calls. |
|
||||||
|
| B3 | Every credential in CR1–CR10 via CR13, with no additional prompt — the daemon is already sanctioned, so `assert_keychain_access_allowed` (`mcp_daemon_guard.py:583`) returns immediately. | Nothing. |
|
||||||
|
| B4 | A1 and A2 in full. | Branch protection at the forge, to the extent configured. |
|
||||||
|
| B5 | The operator's checkout and every worktree, through the author tools (`gitea_mcp_server.py:10957`), plus the shared stderr path at `mcp_server.py:13`. | Nothing; the server's disk is the target disk. |
|
||||||
|
| B6 | All coordination state — no credential required (CR14). It can forge lease ownership and clear decision locks. | Filesystem permissions, which it already satisfies. |
|
||||||
|
| B7 | Jenkins (A4) and GlitchTip (A5) secrets via Finding 4, and CR11/CR12 from its own environment. | Nothing. |
|
||||||
|
| B8 | Both tenants. | Nothing in-process; only the disabled-context check (`gitea_config.py:704`), which does not apply to enabled contexts. |
|
||||||
|
| B9 | Can it run stale code? Only by advancing the checkout, which parity detects (`master_parity_gate.py:255`). | This gate holds — the one boundary that behaves as designed against ADV1. |
|
||||||
|
|
||||||
|
**Result: total.** One compromised client yields A1 through A8 and A10. The only asset with
|
||||||
|
real resistance is A1 via branch protection, and the client holds the merger credential
|
||||||
|
anyway. Nine boundaries, one meaningful stop.
|
||||||
|
|
||||||
|
### After the migration
|
||||||
|
|
||||||
|
The same client authenticates to the remote endpoint with one role's credential, assuming
|
||||||
|
#931–#939 land **and honor D1–D4**.
|
||||||
|
|
||||||
|
| Boundary | What ADV1 reaches | Stopped by |
|
||||||
|
| -------- | ----------------- | ---------- |
|
||||||
|
| B1 | One authenticated session, bound to one principal. | #934: a forged or expired session identity is refused; the client cannot mint one. |
|
||||||
|
| B2 | **One role.** Presenting the author credential yields author permissions only. | #932: the principal comes from the credential, not from which endpoint was reached. |
|
||||||
|
| B3 | **One credential — its own.** | #933 with D3: the provider resolves by principal, and no diagnostic resolves the others. |
|
||||||
|
| B4 | That role's authority on the forge. | Endpoint authentication (#938); plaintext and unauthenticated attempts refused before dispatch. |
|
||||||
|
| B5 | **Nothing.** Filesystem tools are refused over the remote transport with a named blocker. | #936. |
|
||||||
|
| B6 | Its own leases; contention resolves to exactly one winner. | #937 with D4: authenticated session required, not filesystem access. |
|
||||||
|
| B7 | **Nothing.** No CI or observability credential exists in the process. | D1 — the single largest reduction on this table. |
|
||||||
|
| B8 | One tenant. | D1 and #938: the deployment carries one tenant's credentials. |
|
||||||
|
| B9 | Cannot induce stale enforcement. | #935: explicit fail-closed verdict, including undeterminable. |
|
||||||
|
|
||||||
|
**Result: bounded.** The compromise is contained to one role on one tenant, with no
|
||||||
|
filesystem reach and no lateral credential access. A9 survives *only if D2 lands* — if
|
||||||
|
reviewer and merger still share `sysadmin`, a compromised reviewer session still merges, and
|
||||||
|
this row reads the same after the migration as before it.
|
||||||
|
|
||||||
|
### What the migration does not fix
|
||||||
|
|
||||||
|
Against **ADV2**, both tables are identical. Prompt injection does not need to cross a
|
||||||
|
boundary: it arrives inside an authorized session and asks that session to do what it is
|
||||||
|
already permitted to do. Every "stopped by" above authenticates a principal, and the
|
||||||
|
injected instruction has the correct principal. The migration reduces ADV1's blast radius by
|
||||||
|
roughly an order of magnitude and reduces ADV2's by nothing.
|
||||||
|
|
||||||
|
The controls that do constrain ADV2 are per-principal permission scope (#932), repository
|
||||||
|
scope (`gitea_config.py:499`), and credential-backed separation of duty (D2) — each limiting
|
||||||
|
what an authenticated session may do *regardless of what it is asked for*. #955's
|
||||||
|
secure-isolation end state should be read with that distinction in mind: removing credentials
|
||||||
|
from clients defeats ADV1 and ADV5, and does not by itself defeat ADV2.
|
||||||
|
|
||||||
|
Two further items are explicitly out of scope here and unowned by #929:
|
||||||
|
|
||||||
|
- **Session-credential rotation and revocation.** #938 names rotation as documentation, but
|
||||||
|
no child owns proving that a revoked credential stops an in-flight session.
|
||||||
|
- **ADV3** (malicious tool arguments) is diffused across every child rather than owned. The
|
||||||
|
per-request principal work in #932 is the natural place to assert that identifiers taken
|
||||||
|
from the request never authorize anything on their own.
|
||||||
|
|
||||||
|
## 9. How to verify this document
|
||||||
|
|
||||||
|
1. `PYTHONPATH=. pytest tests/test_issue_956_threat_model.py` — resolves every anchor
|
||||||
|
against the working tree and checks the document's structural obligations.
|
||||||
|
2. Pick any five anchors at random and read them; the fixture states what each line must
|
||||||
|
contain.
|
||||||
|
3. Reproduce Findings 3 and 4 live: call `gitea_list_profiles` and `gitea_audit_config`
|
||||||
|
from the **author** namespace. Credential presence reported for roles other than the
|
||||||
|
active one is Finding 3; `MDCPS Jenkins: enabled, read-only, authenticated` is Finding 4.
|
||||||
|
|
||||||
|
If the anchor test fails after an unrelated refactor, the anchors moved and the fixture
|
||||||
|
needs regenerating — the claims are still true, but they are no longer traceable, which
|
||||||
|
#956 treats as the same defect.
|
||||||
@@ -98,6 +98,8 @@ already define, and a regression test asserts each mapping matches.
|
|||||||
| `system.rebind_session_worktree` | operator | gated_write | `gitea.read` | Yes | No | No | 2 |
|
| `system.rebind_session_worktree` | operator | gated_write | `gitea.read` | Yes | No | No | 2 |
|
||||||
| `system.reconcile_cleanups` | controller | privileged | `gitea.pr.close` | Yes | No | No | 2 |
|
| `system.reconcile_cleanups` | controller | privileged | `gitea.pr.close` | Yes | No | No | 2 |
|
||||||
| `initiate_workflow` | operator | gated_write | `gitea.read` | Yes | No | No | 2 |
|
| `initiate_workflow` | operator | gated_write | `gitea.read` | Yes | No | No | 2 |
|
||||||
|
| `observability_reconcile_incident` | operator | gated_write | `gitea.read` | Yes | No | No | 4 |
|
||||||
|
| `observability_link_issue` | operator | gated_write | `gitea.read` | Yes | No | No | 4 |
|
||||||
|
|
||||||
**Dual control** means the acting principal may not be the sole authority: a
|
**Dual control** means the acting principal may not be the sole authority: a
|
||||||
second distinct principal must confirm. **Break-glass** means the action is
|
second distinct principal must confirm. **Break-glass** means the action is
|
||||||
|
|||||||
@@ -0,0 +1,81 @@
|
|||||||
|
# Web Console: Notifications & Human-Attention Routing (#648)
|
||||||
|
|
||||||
|
- **Status:** Phase 3 Live
|
||||||
|
- **Tracking Issue:** [#648](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/648)
|
||||||
|
- **Parent Epic:** [#631](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/631)
|
||||||
|
- **Attention Boundary Reference:** [#628](https://gitea.prgs.cc/Scaled-Tech-Consulting/Gitea-Tools/issues/628)
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 1. Overview
|
||||||
|
|
||||||
|
The **Notifications & Human-Attention Console** (`/notifications`, `/api/v1/notifications`) provides intelligent event classification and human-attention routing for autonomous workflow operations.
|
||||||
|
|
||||||
|
To prevent alert fatigue while ensuring critical escalation boundaries are never missed, events are classified into three distinct **Attention Classes**:
|
||||||
|
|
||||||
|
1. **`human-required`** (Urgent Escalation Boundary):
|
||||||
|
- Items requiring immediate human intervention or business decisions.
|
||||||
|
- Triggers: Auth failures, hard stops, irrecoverable state, decision locks, failed report validations, critical probe errors.
|
||||||
|
- Display: Highlighted in red (`badge-blocked`) with a `HUMAN REQUIRED` badge.
|
||||||
|
|
||||||
|
2. **`operator`** (Operational Inbox):
|
||||||
|
- Items requiring controller or operator review/triage during routine execution.
|
||||||
|
- Triggers: Blocked PRs (merge conflicts), stale leases, duplicate PRs on issues, unassigned ready work.
|
||||||
|
- Display: Displayed in orange/yellow (`badge-claimed`).
|
||||||
|
|
||||||
|
3. **`routine`** (Background Workflow Transitions):
|
||||||
|
- Normal, healthy workflow transitions and state progressions.
|
||||||
|
- Triggers: Active PRs/issues in standard state, clean branch creation, routine heartbeats.
|
||||||
|
- Display: Filtered out of default inbox views to eliminate notification spam; viewable on demand via the "Routine" or "All" tab.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 2. API Endpoints
|
||||||
|
|
||||||
|
### `GET /api/v1/notifications`
|
||||||
|
*Compatibility Alias:* `GET /api/notifications`
|
||||||
|
|
||||||
|
#### Query Parameters:
|
||||||
|
- `project_id` (optional): Filter notifications by project ID.
|
||||||
|
- `attention_class` (optional): `inbox` (default: human-required + operator), `human-required`, `operator`, `routine`, `all`.
|
||||||
|
|
||||||
|
#### Example JSON Response:
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"project_id": "gitea-tools",
|
||||||
|
"repo_label": "Scaled-Tech-Consulting/Gitea-Tools",
|
||||||
|
"human_required_count": 0,
|
||||||
|
"operator_count": 2,
|
||||||
|
"routine_count": 5,
|
||||||
|
"total_count": 7,
|
||||||
|
"fetch_error": null,
|
||||||
|
"inbox_items": [
|
||||||
|
{
|
||||||
|
"id": "notif-pr-block-742",
|
||||||
|
"attention_class": "operator",
|
||||||
|
"category": "blocker",
|
||||||
|
"title": "Blocked PR #742",
|
||||||
|
"summary": "PR #742 requires merge conflict resolution.",
|
||||||
|
"work_kind": "pr",
|
||||||
|
"work_number": 742,
|
||||||
|
"project_id": "gitea-tools",
|
||||||
|
"repo_label": "Scaled-Tech-Consulting/Gitea-Tools",
|
||||||
|
"created_at": "2026-07-25T16:39:47Z",
|
||||||
|
"deep_link": "/traffic",
|
||||||
|
"requires_human": false,
|
||||||
|
"extra": {}
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"all_items": [...]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 3. UI Navigation
|
||||||
|
|
||||||
|
- Access via the **Traffic** navigation menu: **Traffic → Notifications**.
|
||||||
|
- The main view displays:
|
||||||
|
- **Metrics Summary Bar**: Highlighting counts for Human Required, Operator Inbox, and Routine items.
|
||||||
|
- **Attention Filter Tabs**: Toggle between Inbox (Human + Operator), Human Required, Operator, Routine, and All.
|
||||||
|
- **Structured Event Table**: Displays category, title, summary, work item links, and timestamps.
|
||||||
@@ -0,0 +1,102 @@
|
|||||||
|
# Web Console: restart status, impact preview, and approval state (#667)
|
||||||
|
|
||||||
|
Phase 1 of the console restart surface. It consumes the #655 coordinator
|
||||||
|
substrate and displays it. It performs no restart, reload, drain, approval, or
|
||||||
|
process action, and it registers no write endpoint.
|
||||||
|
|
||||||
|
Issue #667's rollout is explicit — *status views first, write approval after the
|
||||||
|
backend gates are green* — and this change delivers only the status half.
|
||||||
|
|
||||||
|
## Surfaces
|
||||||
|
|
||||||
|
| Path | Method | Purpose |
|
||||||
|
|------|--------|---------|
|
||||||
|
| `/runtime/restart` | GET | Restart status page |
|
||||||
|
| `/api/v1/system/restart/status` | GET | Same snapshot as JSON |
|
||||||
|
|
||||||
|
Both accept an optional `restart_class` query parameter (default
|
||||||
|
`full_mcp_restart`). An unrecognised class is not an error: the coordinator
|
||||||
|
resolves it as unknown and fails closed, and the page shows the resulting deny.
|
||||||
|
|
||||||
|
Neither path accepts `POST`; a write attempt returns `405`, and a test asserts
|
||||||
|
it.
|
||||||
|
|
||||||
|
## What it shows
|
||||||
|
|
||||||
|
* **Impact preview (#658)** — verdict, blast radius, affected sessions, leases,
|
||||||
|
critical sections, mutations, and the counts behind them, evaluated
|
||||||
|
`dry_run=True` against live control-plane state.
|
||||||
|
* **Drain proof (#661)** — verification of a supplied proof: valid, clean,
|
||||||
|
expired, tampered, and the reasons behind a refusal.
|
||||||
|
* **Post-restart reconcile (#662)** — the most recent completion proof, its
|
||||||
|
overall status, and which dimensions still require follow-up.
|
||||||
|
* **Restart classes (#663)** — the least-privilege matrix, with *you may
|
||||||
|
request* and *you may execute* computed for the viewing role rather than for a
|
||||||
|
generic operator.
|
||||||
|
* **Approval controls (#633)** — the authorization state of
|
||||||
|
`system.restart_namespace` and `system.reload_namespace`.
|
||||||
|
* **Break-glass (#664)** — declared and marked unavailable; see below.
|
||||||
|
|
||||||
|
## Three rules this surface holds itself to
|
||||||
|
|
||||||
|
A status page that is wrong is worse than one that is missing, because an
|
||||||
|
operator acts on it. Three properties are enforced by tests, and each was
|
||||||
|
verified by reverting the guard and watching a test fail.
|
||||||
|
|
||||||
|
### An unreadable source reports unavailable, never green
|
||||||
|
|
||||||
|
Every source carries its own `SourceStatus`. Nothing substitutes a default,
|
||||||
|
placeholder, or self-comparison for a reading that failed. An unreadable
|
||||||
|
control-plane database yields `inventory_complete: false`, which the coordinator
|
||||||
|
itself turns into a fail-closed verdict, and the page says the blast radius is
|
||||||
|
unknown rather than showing an empty affected-sessions table.
|
||||||
|
|
||||||
|
An absent drain proof is reported as absent — not as a pass. The #661 gate
|
||||||
|
authorizes a restart only against a valid, unexpired, clean proof, so no proof
|
||||||
|
is precisely the state that gate denies on.
|
||||||
|
|
||||||
|
### Authorization is asked the way execution would ask it
|
||||||
|
|
||||||
|
Every probe passes `for_execution=True`.
|
||||||
|
|
||||||
|
Asked without it, an admin is `allowed` for `system.restart_namespace`. On a
|
||||||
|
control surface that reads as a live button. Asked the way an execution attempt
|
||||||
|
would ask, the same principal is refused `phase_not_active`, because the console
|
||||||
|
is in Phase 1 and the action is Phase 2. This surface reports the second answer.
|
||||||
|
|
||||||
|
`execution_enabled` is therefore `false` for every action and every role today,
|
||||||
|
and a test asserts that across the whole role matrix.
|
||||||
|
|
||||||
|
### The control-plane database is opened read-only
|
||||||
|
|
||||||
|
`ControlPlaneDB()` creates directories and runs migrations on construction — a
|
||||||
|
write. This surface never constructs one. It opens the sqlite file with
|
||||||
|
`mode=ro`, exactly as `webui/inventory.py` does, and treats a missing file as
|
||||||
|
missing authority rather than as an empty inventory.
|
||||||
|
|
||||||
|
The test that protects this points at a path inside a directory that already
|
||||||
|
exists, so a read-write `connect` would really create the file. A nested
|
||||||
|
missing-directory path would have passed for the wrong reason.
|
||||||
|
|
||||||
|
## Break-glass is declared, not offered
|
||||||
|
|
||||||
|
The break-glass workflow (#664) is not available on this branch's base. The
|
||||||
|
panel is rendered to operator-class roles as **unavailable**, naming the issue
|
||||||
|
that tracks it. It is not silently omitted, because an operator who has been
|
||||||
|
told a governance path exists needs to see that it is not wired here; and it is
|
||||||
|
not rendered as a control, because there is nothing behind it.
|
||||||
|
|
||||||
|
Unprivileged viewers see only a note that the surface is operator-class.
|
||||||
|
|
||||||
|
## Redaction and escaping
|
||||||
|
|
||||||
|
Every interpolated value passes through `_esc` (`html.escape(..., quote=True)`).
|
||||||
|
Free-form text and anything that can carry a filesystem path additionally passes
|
||||||
|
through `webui.inventory.scrub_text`, which redacts credential-shaped tokens
|
||||||
|
inside a string rather than only at its start. The impact payload is passed
|
||||||
|
through `webui.inventory.scrub` before rendering.
|
||||||
|
|
||||||
|
## Linkage
|
||||||
|
|
||||||
|
Parent #655 · extends #642 · consumes #658, #661, #662, #663 · RBAC #633 ·
|
||||||
|
console #631 · vision #652 · roadmap #653 · break-glass #664.
|
||||||
+50
-1
@@ -1169,10 +1169,57 @@ def server_command():
|
|||||||
return python, [os.path.join(root, "mcp_server.py")]
|
return python, [os.path.join(root, "mcp_server.py")]
|
||||||
|
|
||||||
|
|
||||||
|
RECOGNIZED_GITEA_ENV_KEYS = frozenset({
|
||||||
|
"GITEA_MCP_CONFIG",
|
||||||
|
"GITEA_MCP_PROFILE",
|
||||||
|
"GITEA_PROFILE_NAME",
|
||||||
|
"GITEA_SERVICE",
|
||||||
|
"GITEA_EXECUTION_ROLE",
|
||||||
|
"GITEA_CLIENT_MANAGED",
|
||||||
|
"GITEA_MCP_CLIENT_MANAGED",
|
||||||
|
"GITEA_SERVER_PROVENANCE",
|
||||||
|
"GITEA_AUTHOR_WORKTREE",
|
||||||
|
"GITEA_ACTIVE_WORKTREE",
|
||||||
|
"GITEA_DISABLE_KEYCHAIN",
|
||||||
|
"GITEA_CONTROL_PLANE_DB",
|
||||||
|
"GITEA_DB_PATH",
|
||||||
|
"GITEA_LOG_LEVEL",
|
||||||
|
"GITEA_DEBUG",
|
||||||
|
"GITEA_HMAC_SECRET",
|
||||||
|
"GITEA_IRRECOVERABLE_HMAC_SECRET",
|
||||||
|
"GITEA_FORCE_MCP_RUNTIME_CHECK",
|
||||||
|
"GITEA_FORCE_CLIENT_MANAGED",
|
||||||
|
})
|
||||||
|
|
||||||
|
RECOGNIZED_GITEA_ENV_PREFIXES = (
|
||||||
|
"GITEA_TOKEN_",
|
||||||
|
"GITEA_PASS_",
|
||||||
|
"GITEA_USER_",
|
||||||
|
"GITEA_URL_",
|
||||||
|
"GITEA_HOST_",
|
||||||
|
"GITEA_REMOTE_",
|
||||||
|
"GITEA_HTTP_HEADER_",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def get_unconsumed_gitea_env_overrides(env=None) -> dict[str, str]:
|
||||||
|
"""Find unsupported GITEA_* env vars present in *env* (defaults to os.environ)."""
|
||||||
|
target = os.environ if env is None else env
|
||||||
|
unconsumed = {}
|
||||||
|
for key, value in target.items():
|
||||||
|
if key.startswith("GITEA_"):
|
||||||
|
if key in RECOGNIZED_GITEA_ENV_KEYS:
|
||||||
|
continue
|
||||||
|
if any(key.startswith(p) for p in RECOGNIZED_GITEA_ENV_PREFIXES):
|
||||||
|
continue
|
||||||
|
unconsumed[key] = str(value)
|
||||||
|
return unconsumed
|
||||||
|
|
||||||
|
|
||||||
def launcher_entry(profile_name, config_path=None):
|
def launcher_entry(profile_name, config_path=None):
|
||||||
"""Return a thin MCP launcher entry for *profile_name*.
|
"""Return a thin MCP launcher entry for *profile_name*.
|
||||||
|
|
||||||
Contains only command/args and the two GITEA_MCP_* env vars — never a token
|
Contains command/args and the GITEA_MCP_* / GITEA_CLIENT_MANAGED env vars — never a token
|
||||||
or password. Suitable for Claude / Gemini / Codex ``mcpServers`` blocks.
|
or password. Suitable for Claude / Gemini / Codex ``mcpServers`` blocks.
|
||||||
"""
|
"""
|
||||||
command, args = server_command()
|
command, args = server_command()
|
||||||
@@ -1183,11 +1230,13 @@ def launcher_entry(profile_name, config_path=None):
|
|||||||
"env": {
|
"env": {
|
||||||
"GITEA_MCP_CONFIG": config_path or DEFAULT_CONFIG_PATH,
|
"GITEA_MCP_CONFIG": config_path or DEFAULT_CONFIG_PATH,
|
||||||
"GITEA_MCP_PROFILE": profile_name,
|
"GITEA_MCP_PROFILE": profile_name,
|
||||||
|
"GITEA_CLIENT_MANAGED": "1",
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
def keychain_set(item_id, token, account=None, runner=subprocess.run):
|
def keychain_set(item_id, token, account=None, runner=subprocess.run):
|
||||||
"""Store *token* in the macOS keychain under service *item_id*.
|
"""Store *token* in the macOS keychain under service *item_id*.
|
||||||
|
|
||||||
|
|||||||
+1695
-73
File diff suppressed because it is too large
Load Diff
@@ -500,6 +500,11 @@ def assess_transport_for_auth_mint() -> dict[str, Any]:
|
|||||||
native = mcp_daemon_guard.is_native_mcp_transport()
|
native = mcp_daemon_guard.is_native_mcp_transport()
|
||||||
pytest = mcp_daemon_guard.is_pytest_runtime()
|
pytest = mcp_daemon_guard.is_pytest_runtime()
|
||||||
production = mcp_daemon_guard.is_production_native_mcp_transport()
|
production = mcp_daemon_guard.is_production_native_mcp_transport()
|
||||||
|
# #931: report which transport underwrites the verdict, read through the
|
||||||
|
# one shared accessor rather than assumed to be stdio. The gate's decision
|
||||||
|
# is unchanged here; naming the transport is what lets #932 re-derive the
|
||||||
|
# guarantee from an authenticated session instead of from the bind.
|
||||||
|
bound = mcp_daemon_guard.bound_transport()
|
||||||
if not native and not pytest:
|
if not native and not pytest:
|
||||||
reasons.append(
|
reasons.append(
|
||||||
"irrecoverable provenance authorization requires production native "
|
"irrecoverable provenance authorization requires production native "
|
||||||
@@ -510,6 +515,7 @@ def assess_transport_for_auth_mint() -> dict[str, Any]:
|
|||||||
"allowed": not reasons,
|
"allowed": not reasons,
|
||||||
"native_mcp_transport": native,
|
"native_mcp_transport": native,
|
||||||
"production_native_mcp_transport": production,
|
"production_native_mcp_transport": production,
|
||||||
|
"bound_transport": bound,
|
||||||
"pytest": pytest,
|
"pytest": pytest,
|
||||||
"reasons": reasons,
|
"reasons": reasons,
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -436,6 +436,117 @@ def owning_pr_renewal_evidence(
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def owning_pr_renewal_from_lock(
|
||||||
|
lock_record: Mapping[str, Any] | None,
|
||||||
|
) -> dict[str, Any] | None:
|
||||||
|
"""Rebuild owning-PR renewal evidence from a persisted lock (#945).
|
||||||
|
|
||||||
|
The renewal mirror of ``issue_lock_recovery.recovered_owning_pr_from_lock``.
|
||||||
|
``owning_pr_renewal_evidence`` supplies the waiver for the duration of the
|
||||||
|
``gitea_lock_issue`` call only. The commit, push, create-PR, and
|
||||||
|
duplicate-assessment gates run later in their own calls and re-derive
|
||||||
|
ownership from the durable lock instead — so without this the open PR that
|
||||||
|
renewal already proved belongs to this author reappears there as competing
|
||||||
|
duplicate work, and the exact owner is refused with
|
||||||
|
``duplicate_commit_prevented`` despite complete matching evidence.
|
||||||
|
|
||||||
|
This reads only the ``lease_renewal`` block that the server itself writes,
|
||||||
|
on a lock the caller must already own. Like the recovery mirror it is a
|
||||||
|
re-read of server-derived state, never a fresh assertion: a caller able to
|
||||||
|
forge it could equally forge the lock file every other ownership gate
|
||||||
|
already treats as authoritative.
|
||||||
|
|
||||||
|
Renewal has no descendant case — the assessor required the local, remote and
|
||||||
|
PR heads to be equal — so that equality is re-checked here, and the record
|
||||||
|
must still name the claimant the lock records.
|
||||||
|
|
||||||
|
**What the claimant check below is, and what it is not.** It compares
|
||||||
|
``lease_renewal.identity``/``profile`` against the claimant recorded on the
|
||||||
|
*same* lock file. Both sides are server-written fields of one document, so
|
||||||
|
this is an internal-consistency check: it rejects a lock whose renewal block
|
||||||
|
and claimant disagree. It does **not** consult the live authenticated caller
|
||||||
|
and therefore does not, on its own, prove that the session invoking a later
|
||||||
|
gate is the session the renewal was granted to.
|
||||||
|
|
||||||
|
The binding that actually keeps one session from using another's renewal is
|
||||||
|
structural, and it lives in the caller rather than here. The enforcement
|
||||||
|
paths load the lock through ``_load_existing_issue_lock()`` with no issue
|
||||||
|
coordinates, which resolves ``issue_lock_store.read_session_issue_lock()``
|
||||||
|
→ the session pointer at ``session-{os.getpid()}.json``. Lock *selection* is
|
||||||
|
scoped to the operating-system process, so a caller cannot aim the recheck
|
||||||
|
at a lock some other process bound. Its limits follow from what that scope
|
||||||
|
is: it is per-process, not per-authenticated-user; it says nothing about a
|
||||||
|
lock reached by explicit issue coordinates rather than the session pointer,
|
||||||
|
and nothing about two roles sharing one process. Live identity and profile
|
||||||
|
are enforced separately, by the mutation-authority and profile gates each
|
||||||
|
mutating path already runs — not by this rebuild.
|
||||||
|
|
||||||
|
This function is therefore strictly a re-read with an added consistency
|
||||||
|
requirement. It narrows what a persisted lock can authorize; it never widens
|
||||||
|
it, and it never substitutes for a caller-identity gate.
|
||||||
|
"""
|
||||||
|
if not isinstance(lock_record, Mapping):
|
||||||
|
return None
|
||||||
|
record = lock_record.get("lease_renewal")
|
||||||
|
if not isinstance(record, Mapping) or not record.get("renewed"):
|
||||||
|
return None
|
||||||
|
|
||||||
|
branch_name = _text(record.get("branch_name")) or _text(
|
||||||
|
lock_record.get("branch_name")
|
||||||
|
)
|
||||||
|
pr_head = _text(record.get("pr_head_sha"))
|
||||||
|
local_head = _text(record.get("head_sha"))
|
||||||
|
remote_head = _text(record.get("remote_head_sha"))
|
||||||
|
raw_pr_number = record.get("pr_number")
|
||||||
|
raw_issue_number = lock_record.get("issue_number")
|
||||||
|
|
||||||
|
if raw_pr_number is None or raw_issue_number is None:
|
||||||
|
return None
|
||||||
|
if not branch_name or not pr_head:
|
||||||
|
return None
|
||||||
|
# The assessor required all three heads to agree before it granted renewal.
|
||||||
|
# Re-check, so a truncated, drifted, or hand-built record cannot widen the
|
||||||
|
# exemption past the single head the renewal disposition actually proved.
|
||||||
|
if not local_head or not remote_head:
|
||||||
|
return None
|
||||||
|
if pr_head != local_head or pr_head != remote_head:
|
||||||
|
return None
|
||||||
|
# Renewal is refused outright unless the durable lock records both a
|
||||||
|
# claimant username and profile, so a sanctioned record always carries them.
|
||||||
|
# Requiring them to still agree rejects a lock whose renewal block and
|
||||||
|
# claimant disagree. Both values are read from this one server-written
|
||||||
|
# document: this is internal consistency, not a check against the live
|
||||||
|
# authenticated caller — see the docstring for the binding that is.
|
||||||
|
claimant = lock_record.get("claimant")
|
||||||
|
if not isinstance(claimant, Mapping):
|
||||||
|
lease = lock_record.get("work_lease")
|
||||||
|
claimant = lease.get("claimant") if isinstance(lease, Mapping) else None
|
||||||
|
if not isinstance(claimant, Mapping):
|
||||||
|
return None
|
||||||
|
identity = _text(record.get("identity"))
|
||||||
|
profile = _text(record.get("profile"))
|
||||||
|
if not identity or identity != _text(claimant.get("username")):
|
||||||
|
return None
|
||||||
|
if not profile or profile != _text(claimant.get("profile")):
|
||||||
|
return None
|
||||||
|
|
||||||
|
try:
|
||||||
|
pr_number = int(raw_pr_number)
|
||||||
|
issue_number = int(raw_issue_number)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
|
||||||
|
return {
|
||||||
|
"issue_number": issue_number,
|
||||||
|
"pr_number": pr_number,
|
||||||
|
"branch_name": branch_name,
|
||||||
|
"head_sha": pr_head,
|
||||||
|
"recorded_head": pr_head,
|
||||||
|
"accepted_head": pr_head,
|
||||||
|
"head_relation": "equal",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def build_renewal_record(
|
def build_renewal_record(
|
||||||
assessment: Mapping[str, Any] | None,
|
assessment: Mapping[str, Any] | None,
|
||||||
*,
|
*,
|
||||||
|
|||||||
+128
-9
@@ -299,9 +299,13 @@ def _ownership_refusals(
|
|||||||
f"lock worktree '{lock.get('worktree_path')}' does not match "
|
f"lock worktree '{lock.get('worktree_path')}' does not match "
|
||||||
f"'{worktree_path}'"
|
f"'{worktree_path}'"
|
||||||
)
|
)
|
||||||
lease = lock.get("work_lease") if isinstance(lock, dict) else None
|
# #953 AC2/AC13/AC14: read through the shared claimant reader so a lock
|
||||||
claimant = lease.get("claimant") if isinstance(lease, dict) else None
|
# written by bootstrap — which records the claimant at the top level — is
|
||||||
claimant = claimant if isinstance(claimant, dict) else {}
|
# not refused for "not recording a claimant" when it plainly records one.
|
||||||
|
# This is not a widening: the values are still compared against the
|
||||||
|
# server-resolved identity and profile immediately below, so a legacy
|
||||||
|
# placement grants nothing that the canonical placement would not.
|
||||||
|
claimant = lock_claimant(lock) if isinstance(lock, dict) else {}
|
||||||
recorded_identity = str(claimant.get("username") or "").strip()
|
recorded_identity = str(claimant.get("username") or "").strip()
|
||||||
recorded_profile = str(claimant.get("profile") or "").strip()
|
recorded_profile = str(claimant.get("profile") or "").strip()
|
||||||
if not recorded_identity or not recorded_profile:
|
if not recorded_identity or not recorded_profile:
|
||||||
@@ -636,6 +640,99 @@ def iter_lock_files(lock_dir: str | None = None) -> list[str]:
|
|||||||
return sorted(paths)
|
return sorted(paths)
|
||||||
|
|
||||||
|
|
||||||
|
def release_session_lock(
|
||||||
|
*,
|
||||||
|
issue_number: int,
|
||||||
|
session: str,
|
||||||
|
lock_dir: str | None = None,
|
||||||
|
remote: str | None = None,
|
||||||
|
org: str | None = None,
|
||||||
|
repo: str | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""Remove exactly the durable lock *session* created for *issue_number*.
|
||||||
|
|
||||||
|
``author_issue_bootstrap.run_compensating_recovery`` has called this name
|
||||||
|
since #850, but it was never defined: the call raised ``AttributeError``
|
||||||
|
into a bare ``except Exception: pass``, so the lock half of every
|
||||||
|
compensating rollback silently did nothing. The branch and worktree were
|
||||||
|
removed and the lock was left behind — a state no sanctioned tool can act
|
||||||
|
on, since recovery refuses ``worktree_invalid`` and ``gitea_lock_issue`` has
|
||||||
|
no worktree to bind (#953 review 632 F2).
|
||||||
|
|
||||||
|
Ownership is proven, not asserted. A record is removed only when its
|
||||||
|
recorded ``owner_session`` equals *session* and its issue number matches;
|
||||||
|
``remote``/``org``/``repo`` narrow it further when supplied. Zero matches or
|
||||||
|
more than one both raise, so a caller can never delete a lock it does not
|
||||||
|
own and an ambiguous directory is never guessed at. The ``.json.lock`` flock
|
||||||
|
sidecar is deliberately left in place — it is a zero-byte mutex another
|
||||||
|
process may hold, and removing it under contention would be a race.
|
||||||
|
|
||||||
|
Returns the removed lock file path.
|
||||||
|
"""
|
||||||
|
target_issue = int(issue_number)
|
||||||
|
owner = str(session or "").strip()
|
||||||
|
if not owner:
|
||||||
|
raise ValueError(
|
||||||
|
"release_session_lock requires the owning session id (fail closed)"
|
||||||
|
)
|
||||||
|
|
||||||
|
def _is_owned_durable_lock(record: dict[str, Any] | None) -> bool:
|
||||||
|
# A durable lock, not a bootstrap phase journal or a session pointer,
|
||||||
|
# both of which can share a directory and carry the same issue number
|
||||||
|
# and owner_session.
|
||||||
|
if not record or "lock_generation" not in record:
|
||||||
|
return False
|
||||||
|
if not str(record.get("branch_name") or "").strip():
|
||||||
|
return False
|
||||||
|
if not str(record.get("worktree_path") or "").strip():
|
||||||
|
return False
|
||||||
|
try:
|
||||||
|
if int(record.get("issue_number") or 0) != target_issue:
|
||||||
|
return False
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return False
|
||||||
|
return str(record.get("owner_session") or "").strip() == owner
|
||||||
|
|
||||||
|
# Prefer the exact keyed path when the caller knows the repository; scanning
|
||||||
|
# is the fallback for callers that only carry the issue number.
|
||||||
|
if remote and org and repo:
|
||||||
|
exact = lock_file_path(
|
||||||
|
remote=remote,
|
||||||
|
org=org,
|
||||||
|
repo=repo,
|
||||||
|
issue_number=target_issue,
|
||||||
|
lock_dir=lock_dir,
|
||||||
|
)
|
||||||
|
if not _is_owned_durable_lock(read_lock_file(exact)):
|
||||||
|
raise FileNotFoundError(
|
||||||
|
f"durable issue lock '{exact}' is absent or is not owned by "
|
||||||
|
f"session '{owner}' (fail closed; nothing released)"
|
||||||
|
)
|
||||||
|
os.remove(exact)
|
||||||
|
return exact
|
||||||
|
|
||||||
|
matches: list[str] = []
|
||||||
|
for path in iter_lock_files(lock_dir):
|
||||||
|
if _is_owned_durable_lock(read_lock_file(path)):
|
||||||
|
matches.append(path)
|
||||||
|
|
||||||
|
if not matches:
|
||||||
|
raise FileNotFoundError(
|
||||||
|
f"no durable issue lock for issue #{target_issue} is owned by "
|
||||||
|
f"session '{owner}' (fail closed; nothing released)"
|
||||||
|
)
|
||||||
|
if len(matches) > 1:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"{len(matches)} durable locks for issue #{target_issue} claim "
|
||||||
|
f"session '{owner}'; refusing to guess which to release "
|
||||||
|
"(fail closed)"
|
||||||
|
)
|
||||||
|
|
||||||
|
path = matches[0]
|
||||||
|
os.remove(path)
|
||||||
|
return path
|
||||||
|
|
||||||
|
|
||||||
def find_lock_for_branch(
|
def find_lock_for_branch(
|
||||||
*,
|
*,
|
||||||
remote: str,
|
remote: str,
|
||||||
@@ -1112,21 +1209,43 @@ def assess_same_issue_lease_conflict(
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
def _lock_claimant(lock: dict[str, Any] | None) -> dict[str, str]:
|
def lock_claimant(lock: dict[str, Any] | None) -> dict[str, str]:
|
||||||
|
"""Read the claimant from either canonical or legacy placement (#953 AC13/AC14).
|
||||||
|
|
||||||
|
``work_lease.claimant`` is the canonical placement and is preferred; a
|
||||||
|
top-level ``claimant`` is the legacy/bootstrap placement and is accepted as
|
||||||
|
a fallback. This is the single definition. Before #953 the readers
|
||||||
|
disagreed: this module, ``issue_lock_renewal``, and ``issue_lock_recovery``
|
||||||
|
tolerated both placements, while ``_ownership_refusals`` looked only in
|
||||||
|
``work_lease`` — which is what made a bootstrap-written lock
|
||||||
|
un-heartbeatable.
|
||||||
|
|
||||||
|
Preferring ``work_lease`` over the top level is deliberate: once a legacy
|
||||||
|
lock is upgraded, the canonical placement is authoritative and a stale
|
||||||
|
top-level copy must never win.
|
||||||
|
|
||||||
|
This decides *where to look*, never whether ownership is proven — every
|
||||||
|
caller still compares these values against server-resolved identity and
|
||||||
|
profile.
|
||||||
|
"""
|
||||||
if not isinstance(lock, dict):
|
if not isinstance(lock, dict):
|
||||||
return {}
|
return {}
|
||||||
claimant = lock.get("claimant")
|
lease = lock.get("work_lease")
|
||||||
|
claimant = lease.get("claimant") if isinstance(lease, dict) else None
|
||||||
if not isinstance(claimant, dict):
|
if not isinstance(claimant, dict):
|
||||||
lease = lock.get("work_lease")
|
claimant = lock.get("claimant")
|
||||||
claimant = lease.get("claimant") if isinstance(lease, dict) else None
|
|
||||||
if not isinstance(claimant, dict):
|
if not isinstance(claimant, dict):
|
||||||
return {}
|
return {}
|
||||||
return {
|
return {
|
||||||
"username": str(claimant.get("username") or ""),
|
"username": str(claimant.get("username") or "").strip(),
|
||||||
"profile": str(claimant.get("profile") or ""),
|
"profile": str(claimant.get("profile") or "").strip(),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
#: Back-compatible alias for the pre-#953 private name.
|
||||||
|
_lock_claimant = lock_claimant
|
||||||
|
|
||||||
|
|
||||||
def assess_foreign_lock_overwrite(
|
def assess_foreign_lock_overwrite(
|
||||||
existing_lock: dict[str, Any] | None,
|
existing_lock: dict[str, Any] | None,
|
||||||
incoming_lock: dict[str, Any],
|
incoming_lock: dict[str, Any],
|
||||||
|
|||||||
@@ -0,0 +1,340 @@
|
|||||||
|
"""Sanctioned MCP client reconnect request surface for Codex/LLM sessions (#678).
|
||||||
|
|
||||||
|
Codex and other agent hosts can detect stale or closed Gitea MCP runtimes, but
|
||||||
|
the host owns the transport. This module never restarts, kills, or reloads a
|
||||||
|
daemon. It builds:
|
||||||
|
|
||||||
|
1. A **callable reconnect request** result agents can invoke via
|
||||||
|
``gitea_request_mcp_reconnect`` (report-only, side-effect free).
|
||||||
|
2. A **typed blocker** with exact operator UI steps when recovery must be
|
||||||
|
performed by the host/operator.
|
||||||
|
|
||||||
|
Forbidden recovery paths (must never be recommended):
|
||||||
|
|
||||||
|
* ``pkill`` / ``kill`` / ``killall`` of MCP daemons
|
||||||
|
* ``touch`` / mtime config reload hacks
|
||||||
|
* ``.env`` or MCP config edits as recovery
|
||||||
|
* session-state file edits
|
||||||
|
* raw Gitea API / direct server-import fallbacks
|
||||||
|
|
||||||
|
After the operator reconnects, workflows restart from identity / runtime /
|
||||||
|
capability preflight (``gitea_whoami`` → ``gitea_resolve_task_capability`` →
|
||||||
|
task).
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from typing import Any, Mapping
|
||||||
|
|
||||||
|
# --- Reason vocabulary -------------------------------------------------------
|
||||||
|
|
||||||
|
REASON_STALE_RUNTIME = "stale-runtime"
|
||||||
|
REASON_TRANSPORT_EOF = "transport_eof"
|
||||||
|
REASON_MISSING_NAMESPACE = "missing_namespace"
|
||||||
|
REASON_NOT_REQUIRED = "not_required"
|
||||||
|
REASON_UNSPECIFIED = "unspecified"
|
||||||
|
|
||||||
|
VALID_REASONS = frozenset(
|
||||||
|
{
|
||||||
|
REASON_STALE_RUNTIME,
|
||||||
|
REASON_TRANSPORT_EOF,
|
||||||
|
REASON_MISSING_NAMESPACE,
|
||||||
|
REASON_NOT_REQUIRED,
|
||||||
|
REASON_UNSPECIFIED,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Boundary statuses reported to callers (match review_workflow_boundary style).
|
||||||
|
BOUNDARY_CLEAN = "clean"
|
||||||
|
BOUNDARY_MISMATCH = "mismatch"
|
||||||
|
BOUNDARY_STALE = "stale"
|
||||||
|
BOUNDARY_UNKNOWN = "unknown"
|
||||||
|
|
||||||
|
# Typed blocker kinds
|
||||||
|
BLOCKER_OPERATOR_RECONNECT = "operator_mcp_reconnect_required"
|
||||||
|
BLOCKER_NONE = "none"
|
||||||
|
|
||||||
|
FORBIDDEN_RECOVERY_PATHS: tuple[str, ...] = (
|
||||||
|
"pkill / kill / killall of mcp_server.py, gitea_mcp_server, or broad python sweeps",
|
||||||
|
"touch / mtime-based MCP config reload hacks",
|
||||||
|
".env edits as recovery",
|
||||||
|
"MCP config file edits as recovery",
|
||||||
|
"session-state file edits as recovery",
|
||||||
|
"raw Gitea API or direct MCP server-import fallbacks",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Client-specific operator UI steps. Keep Codex first (issue title surface).
|
||||||
|
OPERATOR_UI_STEPS: dict[str, tuple[str, ...]] = {
|
||||||
|
"codex": (
|
||||||
|
"In Codex, open the MCP / Developer tools panel for this workspace.",
|
||||||
|
"Locate the named Gitea MCP server entry (namespace) that needs reconnect "
|
||||||
|
"(e.g. gitea-author, gitea-reviewer, gitea-merger, gitea-tools, "
|
||||||
|
"gitea-controller, gitea-reconciler).",
|
||||||
|
"Click 'Reload Developer Tools' or the server reconnect/reload control "
|
||||||
|
"for that entry so the client spawns a fresh MCP subprocess.",
|
||||||
|
"If per-server reconnect is unavailable, fully restart the Codex client "
|
||||||
|
"(quit and relaunch) so all MCP namespaces reattach.",
|
||||||
|
"After reconnect, rerun the blocked workflow from preflight: "
|
||||||
|
"gitea_whoami → gitea_resolve_task_capability → the original task. "
|
||||||
|
"Do not resume mid-mutation.",
|
||||||
|
),
|
||||||
|
"claude_code": (
|
||||||
|
"Run `/mcp` (or open the MCP servers UI) in Claude Code.",
|
||||||
|
"Reconnect the affected gitea-* server entry so the client reopens stdio.",
|
||||||
|
"If reconnect fails, relaunch the Claude Code session entirely.",
|
||||||
|
"After reconnect, restart the workflow from gitea_whoami → "
|
||||||
|
"gitea_resolve_task_capability → task.",
|
||||||
|
),
|
||||||
|
"generic": (
|
||||||
|
"Use the host/IDE MCP reconnect or reload control for the named namespace.",
|
||||||
|
"If no per-namespace control exists, restart the MCP client/editor.",
|
||||||
|
"After reconnect, restart the workflow from identity/capability preflight.",
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
#: What an *unidentified* client gets. #948: this is deliberately the
|
||||||
|
#: host-agnostic step set rather than a specific product. Defaulting to one
|
||||||
|
#: vendor emitted Codex UI steps to a Gemini/Antigravity operator, who then had
|
||||||
|
#: no reachable recovery path — the guidance named a panel they do not have.
|
||||||
|
DEFAULT_CLIENT = "generic"
|
||||||
|
|
||||||
|
#: The historical default, kept addressable by name so Codex callers still get
|
||||||
|
#: Codex steps, without it silently becoming the fallback for unknown clients.
|
||||||
|
LEGACY_DEFAULT_CLIENT = "codex"
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_reason(reason: str | None) -> str:
|
||||||
|
"""Map free-form reason strings onto the closed vocabulary."""
|
||||||
|
raw = (reason or "").strip().lower()
|
||||||
|
if not raw:
|
||||||
|
return REASON_UNSPECIFIED
|
||||||
|
if raw in VALID_REASONS:
|
||||||
|
return raw
|
||||||
|
text = raw.replace(" ", "_").replace("-", "_")
|
||||||
|
aliases = {
|
||||||
|
"stale_runtime": REASON_STALE_RUNTIME,
|
||||||
|
"staleruntime": REASON_STALE_RUNTIME,
|
||||||
|
"runtime_stale": REASON_STALE_RUNTIME,
|
||||||
|
"stale": REASON_STALE_RUNTIME,
|
||||||
|
"transport_eof": REASON_TRANSPORT_EOF,
|
||||||
|
"transport_closed": REASON_TRANSPORT_EOF,
|
||||||
|
"eof": REASON_TRANSPORT_EOF,
|
||||||
|
"client_is_closing": REASON_TRANSPORT_EOF,
|
||||||
|
"missing_namespace": REASON_MISSING_NAMESPACE,
|
||||||
|
"namespace_missing": REASON_MISSING_NAMESPACE,
|
||||||
|
"not_required": REASON_NOT_REQUIRED,
|
||||||
|
"healthy": REASON_NOT_REQUIRED,
|
||||||
|
"ok": REASON_NOT_REQUIRED,
|
||||||
|
"unspecified": REASON_UNSPECIFIED,
|
||||||
|
}
|
||||||
|
if text in aliases:
|
||||||
|
return aliases[text]
|
||||||
|
hyphenated = text.replace("_", "-")
|
||||||
|
if hyphenated in VALID_REASONS:
|
||||||
|
return hyphenated
|
||||||
|
return REASON_UNSPECIFIED
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_client(client: str | None) -> str:
|
||||||
|
"""Return the UI-step key for a client.
|
||||||
|
|
||||||
|
#948: alias resolution is shared with ``mcp_worker_identity`` so a client
|
||||||
|
name means the same thing wherever it is read. A name we recognise but have
|
||||||
|
no bespoke steps for — Gemini, Antigravity, Grok — resolves to the generic
|
||||||
|
host-agnostic steps rather than to another vendor's panel.
|
||||||
|
"""
|
||||||
|
import mcp_worker_identity
|
||||||
|
|
||||||
|
canonical = mcp_worker_identity.normalize_client_name(client)
|
||||||
|
if canonical in OPERATOR_UI_STEPS:
|
||||||
|
return canonical
|
||||||
|
return DEFAULT_CLIENT
|
||||||
|
|
||||||
|
|
||||||
|
def classify_boundary_status(
|
||||||
|
*,
|
||||||
|
startup_sha: str | None,
|
||||||
|
current_master_sha: str | None,
|
||||||
|
live_stale: bool | None = None,
|
||||||
|
in_parity: bool | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""Derive boundary_status from parity evidence."""
|
||||||
|
if live_stale is True or in_parity is False:
|
||||||
|
return BOUNDARY_STALE
|
||||||
|
start = (startup_sha or "").strip().lower()
|
||||||
|
current = (current_master_sha or "").strip().lower()
|
||||||
|
if start and current and start != current:
|
||||||
|
return BOUNDARY_MISMATCH
|
||||||
|
if start and current and start == current:
|
||||||
|
return BOUNDARY_CLEAN
|
||||||
|
if in_parity is True:
|
||||||
|
return BOUNDARY_CLEAN
|
||||||
|
return BOUNDARY_UNKNOWN
|
||||||
|
|
||||||
|
|
||||||
|
def operator_ui_steps(client: str | None, *, namespace: str | None = None) -> list[str]:
|
||||||
|
"""Exact operator UI steps for the named client."""
|
||||||
|
key = normalize_client(client)
|
||||||
|
steps = list(OPERATOR_UI_STEPS.get(key) or OPERATOR_UI_STEPS[DEFAULT_CLIENT])
|
||||||
|
ns = (namespace or "").strip()
|
||||||
|
if ns:
|
||||||
|
steps = [
|
||||||
|
s.replace("named Gitea MCP server entry (namespace)", f"namespace '{ns}'")
|
||||||
|
.replace("affected gitea-* server entry", f"server entry '{ns}'")
|
||||||
|
.replace("named namespace", f"namespace '{ns}'")
|
||||||
|
for s in steps
|
||||||
|
]
|
||||||
|
return steps
|
||||||
|
|
||||||
|
|
||||||
|
def build_reconnect_request(
|
||||||
|
*,
|
||||||
|
namespace: str,
|
||||||
|
profile: str | None = None,
|
||||||
|
pid: int | str | None = None,
|
||||||
|
session_id: str | None = None,
|
||||||
|
startup_sha: str | None = None,
|
||||||
|
current_master_sha: str | None = None,
|
||||||
|
boundary_status: str | None = None,
|
||||||
|
reason: str | None = None,
|
||||||
|
client: str | None = DEFAULT_CLIENT,
|
||||||
|
live_stale: bool | None = None,
|
||||||
|
in_parity: bool | None = None,
|
||||||
|
restart_required: bool | None = None,
|
||||||
|
stop_required: bool | None = None,
|
||||||
|
extra: Mapping[str, Any] | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Build the structured reconnect-request / typed-blocker payload (#678).
|
||||||
|
|
||||||
|
Never mutates process, config, or session state. Always side-effect free.
|
||||||
|
"""
|
||||||
|
ns = (namespace or "").strip() or "unknown"
|
||||||
|
normalized_reason = normalize_reason(reason)
|
||||||
|
boundary = (boundary_status or "").strip() or classify_boundary_status(
|
||||||
|
startup_sha=startup_sha,
|
||||||
|
current_master_sha=current_master_sha,
|
||||||
|
live_stale=live_stale,
|
||||||
|
in_parity=in_parity,
|
||||||
|
)
|
||||||
|
|
||||||
|
reconnect_needed = True
|
||||||
|
if normalized_reason == REASON_NOT_REQUIRED and boundary == BOUNDARY_CLEAN:
|
||||||
|
reconnect_needed = False
|
||||||
|
if restart_required is False and stop_required is False and boundary == BOUNDARY_CLEAN:
|
||||||
|
# Explicit healthy probe
|
||||||
|
if normalized_reason in (REASON_NOT_REQUIRED, REASON_UNSPECIFIED):
|
||||||
|
reconnect_needed = False
|
||||||
|
normalized_reason = REASON_NOT_REQUIRED
|
||||||
|
|
||||||
|
if restart_required is True or stop_required is True:
|
||||||
|
reconnect_needed = True
|
||||||
|
if normalized_reason in (REASON_NOT_REQUIRED, REASON_UNSPECIFIED):
|
||||||
|
normalized_reason = REASON_STALE_RUNTIME
|
||||||
|
|
||||||
|
client_key = normalize_client(client)
|
||||||
|
steps = operator_ui_steps(client_key, namespace=ns)
|
||||||
|
|
||||||
|
result: dict[str, Any] = {
|
||||||
|
"success": True,
|
||||||
|
"read_only": True,
|
||||||
|
"reconnect_performed": False,
|
||||||
|
"mutation_performed": False,
|
||||||
|
"reconnect_needed": reconnect_needed,
|
||||||
|
"namespace": ns,
|
||||||
|
"profile": (profile or "").strip() or None,
|
||||||
|
"pid": pid,
|
||||||
|
"session_id": (session_id or "").strip() or None,
|
||||||
|
"startup_sha": (startup_sha or "").strip() or None,
|
||||||
|
"current_master_sha": (current_master_sha or "").strip() or None,
|
||||||
|
"boundary_status": boundary,
|
||||||
|
"reason": normalized_reason,
|
||||||
|
"client": client_key,
|
||||||
|
"forbidden_recovery_paths": list(FORBIDDEN_RECOVERY_PATHS),
|
||||||
|
"post_reconnect_preflight": [
|
||||||
|
"gitea_whoami",
|
||||||
|
"gitea_resolve_task_capability",
|
||||||
|
"original_task",
|
||||||
|
],
|
||||||
|
"exact_safe_next_action": None,
|
||||||
|
"blocker_kind": BLOCKER_NONE,
|
||||||
|
"operator_ui_steps": steps,
|
||||||
|
"typed_blocker": None,
|
||||||
|
}
|
||||||
|
|
||||||
|
if reconnect_needed:
|
||||||
|
result["blocker_kind"] = BLOCKER_OPERATOR_RECONNECT
|
||||||
|
result["stop_required"] = True
|
||||||
|
result["restart_required"] = True
|
||||||
|
result["exact_safe_next_action"] = (
|
||||||
|
f"blocker_kind={BLOCKER_OPERATOR_RECONNECT}: operator must reconnect "
|
||||||
|
f"MCP namespace '{ns}' via the host UI (client={client_key}). "
|
||||||
|
"Do not pkill, touch configs, edit session state, or use raw API. "
|
||||||
|
"After reconnect, restart from gitea_whoami → "
|
||||||
|
"gitea_resolve_task_capability → task."
|
||||||
|
)
|
||||||
|
result["typed_blocker"] = {
|
||||||
|
"blocker_kind": BLOCKER_OPERATOR_RECONNECT,
|
||||||
|
"namespaces": [ns],
|
||||||
|
"why_reconnect_required": normalized_reason,
|
||||||
|
"operator_ui_steps": steps,
|
||||||
|
"client": client_key,
|
||||||
|
"forbidden_recovery_paths": list(FORBIDDEN_RECOVERY_PATHS),
|
||||||
|
"instruction_after_reconnect": (
|
||||||
|
"Rerun the blocked workflow from preflight "
|
||||||
|
"(gitea_whoami → gitea_resolve_task_capability → task). "
|
||||||
|
"Do not continue mid-mutation from pre-reconnect state."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
else:
|
||||||
|
result["stop_required"] = False
|
||||||
|
result["restart_required"] = False
|
||||||
|
result["exact_safe_next_action"] = (
|
||||||
|
f"Reconnect not required for namespace '{ns}' "
|
||||||
|
f"(boundary_status={boundary}). Proceed with the original task."
|
||||||
|
)
|
||||||
|
|
||||||
|
if extra:
|
||||||
|
for key, value in extra.items():
|
||||||
|
if key not in result:
|
||||||
|
result[key] = value
|
||||||
|
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def reasons_never_suggest_forbidden(text: str) -> bool:
|
||||||
|
"""Return True when *text* does not recommend a forbidden recovery path.
|
||||||
|
|
||||||
|
Mentions that *ban* a path (e.g. ``Do not pkill`` / ``never edit session
|
||||||
|
state``) are allowed. Positive recommendations such as ``use pkill`` or
|
||||||
|
``run killall`` fail.
|
||||||
|
"""
|
||||||
|
import re
|
||||||
|
|
||||||
|
lowered = (text or "").lower()
|
||||||
|
# Strip common ban prefixes so "do not pkill" does not trip positive checks.
|
||||||
|
scrubbed = re.sub(
|
||||||
|
r"\b(?:do not|don't|never|must not|forbid(?:den)?|ban(?:ned)?)\b"
|
||||||
|
r"[^.!;\n]{0,80}",
|
||||||
|
" ",
|
||||||
|
lowered,
|
||||||
|
)
|
||||||
|
# Positive imperative / advisory forms that would tell an agent to do harm.
|
||||||
|
positive_suggestions = (
|
||||||
|
"use pkill",
|
||||||
|
"run pkill",
|
||||||
|
"try pkill",
|
||||||
|
"pkill -f",
|
||||||
|
"use killall",
|
||||||
|
"run killall",
|
||||||
|
"killall mcp",
|
||||||
|
"use kill ",
|
||||||
|
"run kill ",
|
||||||
|
"touch the mcp",
|
||||||
|
"touch mcp config",
|
||||||
|
"utime(",
|
||||||
|
"edit the mcp config to recover",
|
||||||
|
"edit .env to recover",
|
||||||
|
"import gitea_mcp_server",
|
||||||
|
"python -c 'import gitea_mcp",
|
||||||
|
)
|
||||||
|
return not any(frag in scrubbed for frag in positive_suggestions)
|
||||||
@@ -0,0 +1,237 @@
|
|||||||
|
"""Antigravity IDE vs Global MCP Config Drift Diagnostic (#672).
|
||||||
|
|
||||||
|
Diagnoses config drift between the active IDE MCP configuration
|
||||||
|
(e.g. ``~/.gemini/antigravity-ide/mcp_config.json``) and the offline/global
|
||||||
|
canonical configuration (e.g. ``~/.gemini/config/mcp_config.json``).
|
||||||
|
|
||||||
|
Hard rules (#672 / #630 / #655):
|
||||||
|
* Distinguish offline/global success from active IDE namespace availability.
|
||||||
|
* Never print tokens, DSNs, Authorization headers, or secret-bearing env vars.
|
||||||
|
* Sanctioned repair path is: backup active config -> patch active config from canonical
|
||||||
|
-> reconnect through IDE/client -> verify with live ``gitea_whoami``.
|
||||||
|
* FORBIDDEN: ``pkill``, mtime edits, source edits, or session-state edits for repair.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from webui import console_redaction
|
||||||
|
|
||||||
|
DEFAULT_ACTIVE_IDE_CONFIG = "~/.gemini/antigravity-ide/mcp_config.json"
|
||||||
|
DEFAULT_GLOBAL_CONFIG = "~/.gemini/config/mcp_config.json"
|
||||||
|
|
||||||
|
REQUIRED_GITEA_ROLE_SERVERS = (
|
||||||
|
"gitea-author",
|
||||||
|
"gitea-reviewer",
|
||||||
|
"gitea-merger",
|
||||||
|
"gitea-reconciler",
|
||||||
|
"gitea-controller",
|
||||||
|
"gitea-tools",
|
||||||
|
)
|
||||||
|
|
||||||
|
SANCTIONED_REPAIR_RUNBOOK: tuple[str, ...] = (
|
||||||
|
"1. Backup active IDE config: cp ~/.gemini/antigravity-ide/mcp_config.json ~/.gemini/antigravity-ide/mcp_config.json.bak",
|
||||||
|
"2. Patch active IDE config: copy required missing Gitea role server entries from global config (~/.gemini/config/mcp_config.json) into active IDE config.",
|
||||||
|
"3. Reconnect via IDE/client UI or client restart (do NOT use host process kill).",
|
||||||
|
"4. Verify active namespace health using live gitea_whoami and gitea_resolve_task_capability on each role namespace.",
|
||||||
|
"FORBIDDEN REPAIR PATHS: pkill / host process kill, mtime touch edits, source code edits, or session-state edits.",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_config_path(path_str: str) -> Path:
|
||||||
|
"""Expand user and resolve absolute path."""
|
||||||
|
return Path(os.path.expanduser(path_str)).resolve()
|
||||||
|
|
||||||
|
|
||||||
|
def load_mcp_config(config_path: str | Path) -> tuple[dict[str, Any] | None, str | None]:
|
||||||
|
"""Load and parse JSON MCP configuration from file.
|
||||||
|
|
||||||
|
Returns (config_dict, error_message).
|
||||||
|
"""
|
||||||
|
resolved = resolve_config_path(str(config_path))
|
||||||
|
if not resolved.exists():
|
||||||
|
return None, f"file_not_found: {resolved}"
|
||||||
|
try:
|
||||||
|
with open(resolved, "r", encoding="utf-8") as f:
|
||||||
|
data = json.load(f)
|
||||||
|
if not isinstance(data, dict):
|
||||||
|
return None, f"invalid_schema: root is not a JSON object in {resolved}"
|
||||||
|
return data, None
|
||||||
|
except Exception as exc:
|
||||||
|
return None, f"unreadable_json: {exc} in {resolved}"
|
||||||
|
|
||||||
|
|
||||||
|
def extract_mcp_servers(config: dict[str, Any] | None) -> dict[str, dict[str, Any]]:
|
||||||
|
"""Extract the mcpServers or mcp_servers mapping safely."""
|
||||||
|
if not config:
|
||||||
|
return {}
|
||||||
|
servers = config.get("mcpServers") or config.get("mcp_servers") or {}
|
||||||
|
if isinstance(servers, dict):
|
||||||
|
return {str(k): v for k, v in servers.items() if isinstance(v, dict)}
|
||||||
|
return {}
|
||||||
|
|
||||||
|
|
||||||
|
def _safe_redact_server_config(srv_cfg: dict[str, Any]) -> dict[str, Any]:
|
||||||
|
"""Redact secrets from environment variables and command line args."""
|
||||||
|
safe = {}
|
||||||
|
if "command" in srv_cfg:
|
||||||
|
safe["command"] = str(srv_cfg["command"])
|
||||||
|
if "args" in srv_cfg and isinstance(srv_cfg["args"], list):
|
||||||
|
safe["args"] = [console_redaction.redact_text(str(a)) for a in srv_cfg["args"]]
|
||||||
|
if "env" in srv_cfg and isinstance(srv_cfg["env"], dict):
|
||||||
|
safe_env = {}
|
||||||
|
for k, v in srv_cfg["env"].items():
|
||||||
|
if any(secret_kw in k.lower() for secret_kw in ("token", "secret", "pass", "key", "auth")):
|
||||||
|
safe_env[k] = "[REDACTED]"
|
||||||
|
else:
|
||||||
|
safe_env[k] = console_redaction.redact_text(str(v))
|
||||||
|
safe["env"] = safe_env
|
||||||
|
return safe
|
||||||
|
|
||||||
|
|
||||||
|
def analyze_config_drift(
|
||||||
|
active_config_path: str = DEFAULT_ACTIVE_IDE_CONFIG,
|
||||||
|
global_config_path: str = DEFAULT_GLOBAL_CONFIG,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Analyze MCP configuration drift between active IDE config and global config.
|
||||||
|
|
||||||
|
Returns structured diagnostic output.
|
||||||
|
"""
|
||||||
|
active_resolved = resolve_config_path(active_config_path)
|
||||||
|
global_resolved = resolve_config_path(global_config_path)
|
||||||
|
|
||||||
|
active_cfg, active_err = load_mcp_config(active_resolved)
|
||||||
|
global_cfg, global_err = load_mcp_config(global_resolved)
|
||||||
|
|
||||||
|
active_servers = extract_mcp_servers(active_cfg)
|
||||||
|
global_servers = extract_mcp_servers(global_cfg)
|
||||||
|
|
||||||
|
missing_role_servers: list[str] = []
|
||||||
|
present_role_servers: list[str] = []
|
||||||
|
profile_mismatches: list[dict[str, Any]] = []
|
||||||
|
reasons: list[str] = []
|
||||||
|
|
||||||
|
if active_err:
|
||||||
|
reasons.append(f"Active IDE config error: {active_err}")
|
||||||
|
if global_err:
|
||||||
|
reasons.append(f"Global canonical config error: {global_err}")
|
||||||
|
|
||||||
|
# Check Gitea role servers
|
||||||
|
for srv_name in REQUIRED_GITEA_ROLE_SERVERS:
|
||||||
|
in_active = srv_name in active_servers
|
||||||
|
in_global = srv_name in global_servers
|
||||||
|
|
||||||
|
if in_active:
|
||||||
|
present_role_servers.append(srv_name)
|
||||||
|
elif in_global:
|
||||||
|
missing_role_servers.append(srv_name)
|
||||||
|
reasons.append(
|
||||||
|
f"Missing Gitea role server '{srv_name}' in active IDE config ({active_resolved})"
|
||||||
|
)
|
||||||
|
|
||||||
|
if in_active and in_global:
|
||||||
|
# Compare profiles & environments
|
||||||
|
act_env = active_servers[srv_name].get("env", {}) if isinstance(active_servers[srv_name], dict) else {}
|
||||||
|
glo_env = global_servers[srv_name].get("env", {}) if isinstance(global_servers[srv_name], dict) else {}
|
||||||
|
|
||||||
|
act_prof = act_env.get("GITEA_MCP_PROFILE") or act_env.get("GITEA_PROFILE_NAME")
|
||||||
|
glo_prof = glo_env.get("GITEA_MCP_PROFILE") or glo_env.get("GITEA_PROFILE_NAME")
|
||||||
|
|
||||||
|
if act_prof != glo_prof:
|
||||||
|
mismatch_item = {
|
||||||
|
"server": srv_name,
|
||||||
|
"active_profile": act_prof,
|
||||||
|
"global_profile": glo_prof,
|
||||||
|
}
|
||||||
|
profile_mismatches.append(mismatch_item)
|
||||||
|
reasons.append(
|
||||||
|
f"Profile mismatch for '{srv_name}': active='{act_prof}' != global='{glo_prof}'"
|
||||||
|
)
|
||||||
|
|
||||||
|
in_sync = bool(
|
||||||
|
not active_err
|
||||||
|
and not global_err
|
||||||
|
and not missing_role_servers
|
||||||
|
and not profile_mismatches
|
||||||
|
)
|
||||||
|
|
||||||
|
report = {
|
||||||
|
"timestamp": datetime.now(timezone.utc).isoformat(),
|
||||||
|
"in_sync": in_sync,
|
||||||
|
"active_config_path": str(active_resolved),
|
||||||
|
"active_config_exists": active_cfg is not None,
|
||||||
|
"global_config_path": str(global_resolved),
|
||||||
|
"global_config_exists": global_cfg is not None,
|
||||||
|
"required_role_servers": list(REQUIRED_GITEA_ROLE_SERVERS),
|
||||||
|
"present_role_servers": present_role_servers,
|
||||||
|
"missing_role_servers": missing_role_servers,
|
||||||
|
"profile_mismatches": profile_mismatches,
|
||||||
|
"reasons": reasons,
|
||||||
|
"sanctioned_repair_runbook": list(SANCTIONED_REPAIR_RUNBOOK),
|
||||||
|
"forbidden_repair_methods": [
|
||||||
|
"pkill / host process kill",
|
||||||
|
"mtime touch edits",
|
||||||
|
"source code edits",
|
||||||
|
"session-state edits",
|
||||||
|
],
|
||||||
|
}
|
||||||
|
|
||||||
|
return console_redaction.redact_payload(report)
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
parser = argparse.ArgumentParser(
|
||||||
|
description="Diagnose Gitea MCP role server config drift between active IDE and global config."
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--active-config",
|
||||||
|
default=DEFAULT_ACTIVE_IDE_CONFIG,
|
||||||
|
help="Path to active IDE MCP config JSON",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--global-config",
|
||||||
|
default=DEFAULT_GLOBAL_CONFIG,
|
||||||
|
help="Path to global/canonical MCP config JSON",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--json", action="store_true", help="Print raw JSON report"
|
||||||
|
)
|
||||||
|
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
report = analyze_config_drift(args.active_config, args.global_config)
|
||||||
|
|
||||||
|
if args.json:
|
||||||
|
print(json.dumps(report, indent=2))
|
||||||
|
else:
|
||||||
|
print("=== MCP Config Drift Diagnostic Report ===")
|
||||||
|
print(f"Timestamp: {report['timestamp']}")
|
||||||
|
print(f"In Sync: {report['in_sync']}")
|
||||||
|
print(f"Active IDE Config: {report['active_config_path']} (exists={report['active_config_exists']})")
|
||||||
|
print(f"Global Config: {report['global_config_path']} (exists={report['global_config_exists']})")
|
||||||
|
print(f"Present Role Servers: {', '.join(report['present_role_servers']) if report['present_role_servers'] else 'None'}")
|
||||||
|
print(f"Missing Role Servers: {', '.join(report['missing_role_servers']) if report['missing_role_servers'] else 'None'}")
|
||||||
|
if report['profile_mismatches']:
|
||||||
|
print("Profile Mismatches:")
|
||||||
|
for m in report['profile_mismatches']:
|
||||||
|
print(f" - {m['server']}: active={m['active_profile']} vs global={m['global_profile']}")
|
||||||
|
if report['reasons']:
|
||||||
|
print("Drift Reasons:")
|
||||||
|
for r in report['reasons']:
|
||||||
|
print(f" - {r}")
|
||||||
|
print("\nSanctioned Repair Runbook:")
|
||||||
|
for step in report['sanctioned_repair_runbook']:
|
||||||
|
print(f" {step}")
|
||||||
|
|
||||||
|
sys.exit(0 if report["in_sync"] else 1)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
+176
-13
@@ -32,6 +32,8 @@ import time
|
|||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
|
|
||||||
|
import mcp_transport_config
|
||||||
|
|
||||||
SANCTIONED_DAEMON_ENV = "GITEA_MCP_SANCTIONED_DAEMON"
|
SANCTIONED_DAEMON_ENV = "GITEA_MCP_SANCTIONED_DAEMON"
|
||||||
ALLOW_DIRECT_IMPORT_ENV = "GITEA_ALLOW_DIRECT_MCP_IMPORT"
|
ALLOW_DIRECT_IMPORT_ENV = "GITEA_ALLOW_DIRECT_MCP_IMPORT"
|
||||||
ALLOW_KEYCHAIN_CLI_ENV = "GITEA_ALLOW_KEYCHAIN_CLI"
|
ALLOW_KEYCHAIN_CLI_ENV = "GITEA_ALLOW_KEYCHAIN_CLI"
|
||||||
@@ -42,7 +44,9 @@ FORCE_PROVENANCE_FAIL_ENV = "GITEA_TEST_FORCE_UNSANCTIONED"
|
|||||||
_NATIVE_RUNTIME: dict[str, Any] | None = None
|
_NATIVE_RUNTIME: dict[str, Any] | None = None
|
||||||
|
|
||||||
# Production transport identifiers accepted by bind_native_mcp_transport.
|
# Production transport identifiers accepted by bind_native_mcp_transport.
|
||||||
_PRODUCTION_TRANSPORTS = frozenset({"stdio"})
|
# #931: the permitted set is defined once, in mcp_transport_config. This name
|
||||||
|
# is kept as an alias so the guard never restates a transport identifier.
|
||||||
|
_PRODUCTION_TRANSPORTS = mcp_transport_config.SUPPORTED_TRANSPORTS
|
||||||
_RUNTIME_MODE_PRODUCTION = "production"
|
_RUNTIME_MODE_PRODUCTION = "production"
|
||||||
_RUNTIME_MODE_TEST = "test"
|
_RUNTIME_MODE_TEST = "test"
|
||||||
_PHASE_ENTRYPOINT_CLAIMED = "entrypoint_claimed"
|
_PHASE_ENTRYPOINT_CLAIMED = "entrypoint_claimed"
|
||||||
@@ -58,6 +62,23 @@ class UnsanctionedRuntimeError(RuntimeError):
|
|||||||
"""Raised when mutation/credential code runs outside a native MCP daemon."""
|
"""Raised when mutation/credential code runs outside a native MCP daemon."""
|
||||||
|
|
||||||
|
|
||||||
|
class TransportExecutionError(UnsanctionedRuntimeError):
|
||||||
|
"""Raised when a bound transport may not be served by this entrypoint (#931).
|
||||||
|
|
||||||
|
Subclasses :class:`UnsanctionedRuntimeError` so every existing fail-closed
|
||||||
|
handler still catches it, while letting a caller that cares distinguish
|
||||||
|
"nothing is bound" from "something valid is bound but its listener has not
|
||||||
|
been commissioned". Carries the structured verdict on ``.assessment``.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, message: str, assessment: dict[str, Any] | None = None):
|
||||||
|
super().__init__(message)
|
||||||
|
self.assessment = assessment or {}
|
||||||
|
self.blocker_kind = self.assessment.get("blocker_kind")
|
||||||
|
self.owner_issue = self.assessment.get("owner_issue")
|
||||||
|
self.transport = self.assessment.get("transport")
|
||||||
|
|
||||||
|
|
||||||
def is_pytest_runtime() -> bool:
|
def is_pytest_runtime() -> bool:
|
||||||
if (os.environ.get(FORCE_PROVENANCE_FAIL_ENV) or "").strip() in {
|
if (os.environ.get(FORCE_PROVENANCE_FAIL_ENV) or "").strip() in {
|
||||||
"1",
|
"1",
|
||||||
@@ -171,23 +192,46 @@ def mark_sanctioned_daemon() -> dict[str, Any]:
|
|||||||
return native_runtime_status()
|
return native_runtime_status()
|
||||||
|
|
||||||
|
|
||||||
def bind_native_mcp_transport(*, transport: str) -> dict[str, Any]:
|
def bind_native_mcp_transport(*, transport: str | None = None) -> dict[str, Any]:
|
||||||
"""Bind the live native MCP transport lifecycle (#695).
|
"""Bind the live native MCP transport lifecycle (#695 / #931).
|
||||||
|
|
||||||
Must be called from the resolved canonical entrypoint immediately before
|
Must be called from the resolved canonical entrypoint immediately before
|
||||||
the real MCP server transport loop (e.g. ``mcp.run(transport=\"stdio\")``).
|
the real MCP server transport loop (``mcp.run``). Requires a prior
|
||||||
Requires a prior successful :func:`mark_sanctioned_daemon` claim in this
|
successful :func:`mark_sanctioned_daemon` claim in this process.
|
||||||
process. Import-only or offline launch without this bind leaves
|
Import-only or offline launch without this bind leaves
|
||||||
:func:`is_native_mcp_transport` false.
|
:func:`is_native_mcp_transport` false.
|
||||||
|
|
||||||
|
#931: ``transport`` is now optional. Omitting it — which is what the
|
||||||
|
production entrypoint does — resolves the identifier from deployment
|
||||||
|
configuration via :func:`mcp_transport_config.resolve_configured_transport`,
|
||||||
|
yielding :data:`mcp_transport_config.DEFAULT_TRANSPORT` when nothing is
|
||||||
|
configured. An explicit argument remains supported for tests and for a
|
||||||
|
launcher that has already resolved the value. Either way the identifier is
|
||||||
|
validated against the single permitted set before the runtime record is
|
||||||
|
written, so no tool can dispatch over an unregistered transport.
|
||||||
|
|
||||||
|
The resolved value is pinned into the process-local record and is read back
|
||||||
|
only through :func:`bound_transport`. Rebinding to a different transport is
|
||||||
|
refused, so two guards can never observe different values in one process.
|
||||||
"""
|
"""
|
||||||
global _NATIVE_RUNTIME
|
global _NATIVE_RUNTIME
|
||||||
transport_name = (transport or "").strip().lower()
|
if transport is None:
|
||||||
if transport_name not in _PRODUCTION_TRANSPORTS:
|
resolution = mcp_transport_config.resolve_configured_transport()
|
||||||
raise UnsanctionedRuntimeError(
|
transport_name = str(resolution["transport"])
|
||||||
f"bind_native_mcp_transport rejected: transport {transport!r} is "
|
if not resolution["supported"]:
|
||||||
f"not a production MCP transport (#695). Allowed: "
|
raise UnsanctionedRuntimeError(
|
||||||
f"{sorted(_PRODUCTION_TRANSPORTS)}."
|
"bind_native_mcp_transport rejected: "
|
||||||
)
|
+ "; ".join(resolution["reasons"])
|
||||||
|
+ " No tool is served over an unregistered transport."
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
transport_name = mcp_transport_config.normalize_transport(transport)
|
||||||
|
if transport_name not in _PRODUCTION_TRANSPORTS:
|
||||||
|
raise UnsanctionedRuntimeError(
|
||||||
|
f"bind_native_mcp_transport rejected: transport {transport!r} is "
|
||||||
|
f"not a production MCP transport (#695). Allowed: "
|
||||||
|
f"{sorted(_PRODUCTION_TRANSPORTS)}."
|
||||||
|
)
|
||||||
|
|
||||||
entrypoint_path = _caller_official_entrypoint_path()
|
entrypoint_path = _caller_official_entrypoint_path()
|
||||||
if entrypoint_path is None:
|
if entrypoint_path is None:
|
||||||
@@ -216,6 +260,23 @@ def bind_native_mcp_transport(*, transport: str) -> dict[str, Any]:
|
|||||||
"between mark and bind (#695)."
|
"between mark and bind (#695)."
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# #931: one process binds one transport. Re-binding the same identifier is
|
||||||
|
# idempotent (a retried launch step must not fail); re-binding a different
|
||||||
|
# one is refused, because a guard that already read the first value would
|
||||||
|
# otherwise disagree with a guard that reads the second.
|
||||||
|
already_bound = (_NATIVE_RUNTIME.get("transport") or "").strip()
|
||||||
|
if (
|
||||||
|
already_bound
|
||||||
|
and _NATIVE_RUNTIME.get("phase") == _PHASE_TRANSPORT_BOUND
|
||||||
|
and already_bound != transport_name
|
||||||
|
):
|
||||||
|
raise UnsanctionedRuntimeError(
|
||||||
|
"bind_native_mcp_transport rejected: transport is already bound to "
|
||||||
|
f"{already_bound!r} in this process; rebinding to "
|
||||||
|
f"{transport_name!r} is forbidden (#931). Restart the server to "
|
||||||
|
"change the deployment transport."
|
||||||
|
)
|
||||||
|
|
||||||
# Pin session-state root for this server lifetime (#695 AC2 / PR #701).
|
# Pin session-state root for this server lifetime (#695 AC2 / PR #701).
|
||||||
# Changing GITEA_MCP_SESSION_STATE_DIR after bind must not manufacture a
|
# Changing GITEA_MCP_SESSION_STATE_DIR after bind must not manufacture a
|
||||||
# second authority domain for decision locks / workflow proofs.
|
# second authority domain for decision locks / workflow proofs.
|
||||||
@@ -353,6 +414,88 @@ def is_production_native_mcp_transport() -> bool:
|
|||||||
return (_NATIVE_RUNTIME or {}).get("mode") == _RUNTIME_MODE_PRODUCTION
|
return (_NATIVE_RUNTIME or {}).get("mode") == _RUNTIME_MODE_PRODUCTION
|
||||||
|
|
||||||
|
|
||||||
|
def bound_transport() -> str | None:
|
||||||
|
"""The one authoritative bound transport identifier, or ``None`` (#931).
|
||||||
|
|
||||||
|
This is the shared accessor every transport-aware guard reads. It reports
|
||||||
|
the value pinned at bind time, never the environment, so changing
|
||||||
|
``GITEA_MCP_TRANSPORT`` after the bind cannot move what a guard observes —
|
||||||
|
the same rule :func:`pinned_session_state_dir` applies to session state.
|
||||||
|
|
||||||
|
``None`` means unbound: an offline import or a launch that never reached
|
||||||
|
the bind. Callers must treat that as fail-closed, exactly as they already
|
||||||
|
treat :func:`is_native_mcp_transport` returning false.
|
||||||
|
"""
|
||||||
|
if not is_native_mcp_transport():
|
||||||
|
return None
|
||||||
|
return (_NATIVE_RUNTIME or {}).get("transport") or None
|
||||||
|
|
||||||
|
|
||||||
|
def assert_transport_bound(context: str = "tool service") -> str:
|
||||||
|
"""Return the bound transport, or fail closed before *context* (#931).
|
||||||
|
|
||||||
|
Called immediately before the server enters its transport loop so an
|
||||||
|
invalid or absent bind stops the process rather than serving tools over a
|
||||||
|
transport no guard can name.
|
||||||
|
"""
|
||||||
|
transport = bound_transport()
|
||||||
|
if transport:
|
||||||
|
return transport
|
||||||
|
raise UnsanctionedRuntimeError(
|
||||||
|
f"No MCP transport is bound; refusing {context} (#931). "
|
||||||
|
"bind_native_mcp_transport must succeed from the canonical entrypoint "
|
||||||
|
"before any tool is served. Offline import and standalone launch "
|
||||||
|
"cannot reconstruct a bind."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def assess_serve_authorization() -> dict[str, Any]:
|
||||||
|
"""Structured verdict on whether this process may serve tools (#931).
|
||||||
|
|
||||||
|
This is the decision that consumes :func:`bound_transport`. It is what stops
|
||||||
|
the bound identifier from being reporting-only metadata: the serve path
|
||||||
|
cannot proceed unless the value pinned at bind is one this entrypoint is
|
||||||
|
commissioned to execute.
|
||||||
|
|
||||||
|
Never raises; returns the verdict so callers and diagnostics can inspect it.
|
||||||
|
"""
|
||||||
|
return mcp_transport_config.assess_transport_execution(bound_transport())
|
||||||
|
|
||||||
|
|
||||||
|
def authorize_transport_execution(context: str = "tool service") -> str:
|
||||||
|
"""Return the transport this process may serve, or fail closed (#931).
|
||||||
|
|
||||||
|
Two distinct boundaries, in order:
|
||||||
|
|
||||||
|
1. **Bind presence** — :func:`assert_transport_bound` enforces the
|
||||||
|
pre-existing #695 contract, so an unbound runtime keeps its established
|
||||||
|
failure and reason code.
|
||||||
|
2. **Execution authorization** — the bound identifier must be one this
|
||||||
|
entrypoint is commissioned to serve. A registered transport whose
|
||||||
|
listener has not been commissioned is refused here, before any listener
|
||||||
|
is created and before any tool can dispatch.
|
||||||
|
|
||||||
|
That ordering matters: recognition, validation and durable recording all
|
||||||
|
still happen for a remote identifier, so #931's seam is intact; only the act
|
||||||
|
of *serving* it is withheld until its owning issue commissions it.
|
||||||
|
"""
|
||||||
|
# Boundary 1: unbound stays exactly as fail-closed as it was under #695.
|
||||||
|
assert_transport_bound(context)
|
||||||
|
|
||||||
|
# Boundary 2: bound, but is this entrypoint allowed to serve it?
|
||||||
|
assessment = assess_serve_authorization()
|
||||||
|
if assessment.get("allowed"):
|
||||||
|
return str(assessment["transport"])
|
||||||
|
|
||||||
|
reasons = "; ".join(assessment.get("reasons") or []) or "not authorized"
|
||||||
|
next_action = assessment.get("exact_next_action") or ""
|
||||||
|
raise TransportExecutionError(
|
||||||
|
f"Refusing {context} (#931) [{assessment.get('blocker_kind')}]: "
|
||||||
|
f"{reasons} {next_action}".strip(),
|
||||||
|
assessment,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
def is_sanctioned_mcp_daemon() -> bool:
|
def is_sanctioned_mcp_daemon() -> bool:
|
||||||
"""Backward-compatible name; #695 requires native transport, not env alone."""
|
"""Backward-compatible name; #695 requires native transport, not env alone."""
|
||||||
if is_production_native_mcp_transport():
|
if is_production_native_mcp_transport():
|
||||||
@@ -476,6 +619,21 @@ def native_runtime_status() -> dict[str, Any]:
|
|||||||
"entrypoint_path": rt.get("entrypoint_path"),
|
"entrypoint_path": rt.get("entrypoint_path"),
|
||||||
"phase": rt.get("phase"),
|
"phase": rt.get("phase"),
|
||||||
"transport": rt.get("transport"),
|
"transport": rt.get("transport"),
|
||||||
|
# #931: the authoritative bound identifier, plus the seam that defines
|
||||||
|
# what may be bound. ``bound_transport`` is None until a bind succeeds,
|
||||||
|
# so an offline import is distinguishable from a stdio session.
|
||||||
|
"bound_transport": bound_transport(),
|
||||||
|
"transport_bound": bound_transport() is not None,
|
||||||
|
"default_transport": mcp_transport_config.DEFAULT_TRANSPORT,
|
||||||
|
"supported_transports": list(mcp_transport_config.supported_transports()),
|
||||||
|
"transport_env": mcp_transport_config.TRANSPORT_ENV,
|
||||||
|
# #931 review 635: recognition and execution authorization are distinct.
|
||||||
|
# ``supported`` is what may be bound; ``executable`` is what this
|
||||||
|
# entrypoint may actually serve. A recognized-but-uncommissioned
|
||||||
|
# transport reports serve_authorized False with a named blocker.
|
||||||
|
"executable_transports": list(mcp_transport_config.executable_transports()),
|
||||||
|
"serve_authorized": bool(assess_serve_authorization().get("allowed")),
|
||||||
|
"serve_authorization": assess_serve_authorization(),
|
||||||
"mode": rt.get("mode"),
|
"mode": rt.get("mode"),
|
||||||
"session_state_dir": pinned_session_state_dir() or rt.get("session_state_dir"),
|
"session_state_dir": pinned_session_state_dir() or rt.get("session_state_dir"),
|
||||||
"session_state_dir_pinned": pinned_session_state_dir() is not None,
|
"session_state_dir_pinned": pinned_session_state_dir() is not None,
|
||||||
@@ -502,7 +660,12 @@ def mutation_provenance_fields() -> dict[str, Any]:
|
|||||||
if st.get("mode") == _RUNTIME_MODE_TEST and st["native_mcp_transport"]:
|
if st.get("mode") == _RUNTIME_MODE_TEST and st["native_mcp_transport"]:
|
||||||
transport = "test_native_mcp"
|
transport = "test_native_mcp"
|
||||||
return {
|
return {
|
||||||
|
# ``transport`` stays the trust *class* it has always been, so existing
|
||||||
|
# durable records keep their shape. ``bound_transport`` (#931) adds the
|
||||||
|
# bound identifier itself, which is what lets an operator tell from a
|
||||||
|
# durable record which transport performed a mutation.
|
||||||
"transport": transport,
|
"transport": transport,
|
||||||
|
"bound_transport": st.get("bound_transport"),
|
||||||
"native_mcp_transport": bool(st["native_mcp_transport"]),
|
"native_mcp_transport": bool(st["native_mcp_transport"]),
|
||||||
"production_native_mcp_transport": bool(
|
"production_native_mcp_transport": bool(
|
||||||
st.get("production_native_mcp_transport")
|
st.get("production_native_mcp_transport")
|
||||||
|
|||||||
@@ -51,6 +51,42 @@ EOF_PATTERNS = (
|
|||||||
"eof",
|
"eof",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
ERROR_CONNECTED_NAMESPACES_MISSING = "mcp_connected_namespaces_missing"
|
||||||
|
|
||||||
|
# Distinct from the condition above. ``mcp_connected_namespaces_missing`` is reserved for
|
||||||
|
# the actual #708 defect: the host *does* report the service Connected, yet the namespace
|
||||||
|
# never entered the active session tool surface. A required namespace that is absent from
|
||||||
|
# the connected-service inventory has no Connected claim behind it at all, so reporting it
|
||||||
|
# under the #708 condition would assert something the evidence does not support and would
|
||||||
|
# point an operator at the wrong recovery.
|
||||||
|
ERROR_REQUIRED_NAMESPACES_NOT_CONNECTED = "mcp_required_namespaces_not_connected"
|
||||||
|
|
||||||
|
DISCOVERY_STATUS_ATTACHED = "namespaces_attached"
|
||||||
|
DISCOVERY_STATUS_CONNECTED_MISSING = "connected_but_namespaces_missing"
|
||||||
|
DISCOVERY_STATUS_NOT_CONNECTED = "required_namespaces_not_connected"
|
||||||
|
DISCOVERY_STATUS_DISCONNECTED = "disconnected"
|
||||||
|
|
||||||
|
# Namespaces that must be *attached to the active session* for a mutation task (#708).
|
||||||
|
# Connected-at-host is not attached-in-session; these are gated separately from the
|
||||||
|
# #543 health map because a namespace can be healthy on probe yet absent from the
|
||||||
|
# session tool surface.
|
||||||
|
ATTACHMENT_GATED_TASKS = {
|
||||||
|
"review_pr": "gitea-reviewer",
|
||||||
|
"submit_review": "gitea-reviewer",
|
||||||
|
"merge_pr": "gitea-merger",
|
||||||
|
"work_issue": "gitea-author",
|
||||||
|
"create_pr": "gitea-author",
|
||||||
|
}
|
||||||
|
|
||||||
|
# The only sanctioned recovery for an unattached namespace (#678 exposes it natively).
|
||||||
|
SANCTIONED_ATTACH_RECOVERY_TOOL = "gitea_request_mcp_reconnect"
|
||||||
|
|
||||||
|
UNSAFE_FALLBACK_WARNING = (
|
||||||
|
"Workflow Safety Hard Stop (#708): Connected-but-namespaces-missing recovery must "
|
||||||
|
"NEVER use direct imports, Gitea API mutations, profile hopping, session-state "
|
||||||
|
"overrides, PID kills, or config mtime touches. Use client reconnect only."
|
||||||
|
)
|
||||||
|
|
||||||
SAFE_ENV_KEYS = (
|
SAFE_ENV_KEYS = (
|
||||||
"GITEA_MCP_PROFILE",
|
"GITEA_MCP_PROFILE",
|
||||||
"GITEA_PROFILE_NAME",
|
"GITEA_PROFILE_NAME",
|
||||||
@@ -59,6 +95,318 @@ SAFE_ENV_KEYS = (
|
|||||||
"GITEA_MCP_CONFIG",
|
"GITEA_MCP_CONFIG",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# #948: provenance used to be derived from the summary this allowlist produces.
|
||||||
|
# The allowlist never carried a provenance key, so that derivation could only
|
||||||
|
# ever evaluate to ``manual_launch`` — whatever the process actually was — while
|
||||||
|
# ``gitea_get_runtime_context`` read the live environment and reported
|
||||||
|
# ``client_managed`` for the same process. Provenance is no longer derived here.
|
||||||
|
# It comes from ``mcp_worker_identity.assess_provenance``, the single authority
|
||||||
|
# every surface shares. This allowlist keeps its original and only job: deciding
|
||||||
|
# which env values are safe to echo back in diagnostics.
|
||||||
|
|
||||||
|
|
||||||
|
def assess_connected_namespace_attachment(
|
||||||
|
*,
|
||||||
|
connected_servers: list[str] | tuple[str, ...] | set[str] | None = None,
|
||||||
|
attached_session_namespaces: list[str] | tuple[str, ...] | set[str] | None = None,
|
||||||
|
required_namespaces: list[str] | tuple[str, ...] | set[str] | None = None,
|
||||||
|
discovery_cache_age_seconds: float | int | None = None,
|
||||||
|
discovery_cache_hit: bool | None = None,
|
||||||
|
auto_attach_attempted: bool = False,
|
||||||
|
auto_attach_succeeded: bool = False,
|
||||||
|
session_tool_snapshot_at: float | int | None = None,
|
||||||
|
namespace_connected_at: dict[str, float] | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Assess whether host-connected MCP servers have attached tool namespaces in the active session (#708).
|
||||||
|
|
||||||
|
Addresses the Connected-but-namespaces-missing defect: CLI/host status may report Connected
|
||||||
|
while the active LLM session tool surface exposes 0 attached tool namespaces.
|
||||||
|
|
||||||
|
This is a *distinct* condition from config drift (#672), transport-closed (#584),
|
||||||
|
and resolver EOF (#685): the transport is up and the host reports Connected, yet the
|
||||||
|
namespace never entered the session tool surface.
|
||||||
|
|
||||||
|
Startup ordering (``session_tool_snapshot_at`` + ``namespace_connected_at``) identifies
|
||||||
|
the race where the session tool snapshot was taken before a role server finished
|
||||||
|
``initialize``/``list_tools``, which is why parallel multi-role startup can leave the
|
||||||
|
session with an empty namespace set while Connected later flips true.
|
||||||
|
|
||||||
|
Returns structured detection details plus secret-free telemetry.
|
||||||
|
"""
|
||||||
|
connected = [str(s).strip() for s in (connected_servers or []) if str(s).strip()]
|
||||||
|
attached = set(str(ns).strip() for ns in (attached_session_namespaces or []) if str(ns).strip())
|
||||||
|
req = [str(r).strip() for r in (required_namespaces or DEFAULT_NAMESPACES) if str(r).strip()]
|
||||||
|
|
||||||
|
required = list(dict.fromkeys(req))
|
||||||
|
connected_set = set(connected)
|
||||||
|
|
||||||
|
proof: dict[str, dict[str, bool]] = {}
|
||||||
|
for s in connected:
|
||||||
|
proof[s] = {"connected": True, "attached": s in attached}
|
||||||
|
for r in required:
|
||||||
|
if r not in proof:
|
||||||
|
proof[r] = {"connected": r in connected_set, "attached": r in attached}
|
||||||
|
|
||||||
|
# The genuine #708 condition: the host reports the service Connected and the namespace
|
||||||
|
# still never entered the active session tool surface.
|
||||||
|
missing = [r for r in required if r in connected_set and r not in attached]
|
||||||
|
# A separate condition: the service is required but absent from the connected-service
|
||||||
|
# inventory. Nothing here is Connected, so this must not borrow the #708 wording or its
|
||||||
|
# recovery — see ERROR_REQUIRED_NAMESPACES_NOT_CONNECTED.
|
||||||
|
not_connected = [r for r in required if r not in connected_set]
|
||||||
|
# Evidence that disagrees with itself: reported attached to the session while absent
|
||||||
|
# from the connected inventory. Neither statement is proof, so it fails closed.
|
||||||
|
contradictory = [r for r in not_connected if r in attached]
|
||||||
|
|
||||||
|
# No required namespace may lack attachment proof and still be called healthy.
|
||||||
|
attachment_healthy = (
|
||||||
|
not missing
|
||||||
|
and not not_connected
|
||||||
|
and len(connected) > 0
|
||||||
|
and len(required) > 0
|
||||||
|
)
|
||||||
|
|
||||||
|
if attachment_healthy:
|
||||||
|
discovery_status = DISCOVERY_STATUS_ATTACHED
|
||||||
|
elif not connected:
|
||||||
|
discovery_status = DISCOVERY_STATUS_DISCONNECTED
|
||||||
|
elif not_connected:
|
||||||
|
discovery_status = DISCOVERY_STATUS_NOT_CONNECTED
|
||||||
|
else:
|
||||||
|
discovery_status = DISCOVERY_STATUS_CONNECTED_MISSING
|
||||||
|
|
||||||
|
# Every condition actually present is reported; ``error_type`` names the primary one.
|
||||||
|
# Not-connected outranks Connected-but-unattached because a service that never
|
||||||
|
# connected cannot be recovered by attaching its namespace.
|
||||||
|
error_types: list[str] = []
|
||||||
|
if not_connected:
|
||||||
|
error_types.append(ERROR_REQUIRED_NAMESPACES_NOT_CONNECTED)
|
||||||
|
if missing:
|
||||||
|
error_types.append(ERROR_CONNECTED_NAMESPACES_MISSING)
|
||||||
|
if not attachment_healthy and not error_types:
|
||||||
|
error_types.append(ERROR_REQUIRED_NAMESPACES_NOT_CONNECTED)
|
||||||
|
error_type = None if attachment_healthy else error_types[0]
|
||||||
|
|
||||||
|
# Per-namespace verdict, so a mutation gate never has to infer one namespace's state
|
||||||
|
# from a whole-session summary. Each entry states only what its own evidence supports.
|
||||||
|
namespace_conditions: dict[str, dict[str, Any]] = {}
|
||||||
|
for ns_name, state in proof.items():
|
||||||
|
ns_connected = bool(state["connected"])
|
||||||
|
ns_attached = bool(state["attached"])
|
||||||
|
if ns_connected and ns_attached:
|
||||||
|
condition = None
|
||||||
|
elif ns_connected:
|
||||||
|
condition = ERROR_CONNECTED_NAMESPACES_MISSING
|
||||||
|
else:
|
||||||
|
condition = ERROR_REQUIRED_NAMESPACES_NOT_CONNECTED
|
||||||
|
namespace_conditions[ns_name] = {
|
||||||
|
"namespace": ns_name,
|
||||||
|
"required": ns_name in required,
|
||||||
|
"connected": ns_connected,
|
||||||
|
"attached": ns_attached,
|
||||||
|
"attachment_healthy": ns_connected and ns_attached,
|
||||||
|
"condition": condition,
|
||||||
|
"contradictory_evidence": ns_attached and not ns_connected,
|
||||||
|
}
|
||||||
|
|
||||||
|
# Startup-ordering race: a namespace that finished connecting *after* the session
|
||||||
|
# tool snapshot was taken cannot be in that snapshot, however healthy it looks now.
|
||||||
|
connected_at = {
|
||||||
|
str(k).strip(): v
|
||||||
|
for k, v in (namespace_connected_at or {}).items()
|
||||||
|
if str(k).strip() and isinstance(v, (int, float))
|
||||||
|
}
|
||||||
|
late_attaching: list[str] = []
|
||||||
|
if isinstance(session_tool_snapshot_at, (int, float)):
|
||||||
|
for ns_name, ts in connected_at.items():
|
||||||
|
if ts > session_tool_snapshot_at and ns_name not in attached:
|
||||||
|
late_attaching.append(ns_name)
|
||||||
|
late_attaching.sort()
|
||||||
|
startup_ordering_race = bool(late_attaching)
|
||||||
|
|
||||||
|
auto_recovered = bool(auto_attach_attempted and auto_attach_succeeded and attachment_healthy)
|
||||||
|
reconnect_required = not attachment_healthy
|
||||||
|
|
||||||
|
reasons: list[str] = []
|
||||||
|
remediation: list[str] = []
|
||||||
|
if not connected:
|
||||||
|
reasons.append("No MCP servers reported Connected.")
|
||||||
|
remediation.append("Start or reconnect Gitea MCP servers in client config.")
|
||||||
|
if missing:
|
||||||
|
reasons.append(
|
||||||
|
f"MCP server(s) {missing} report Connected at host/CLI layer but tool namespaces "
|
||||||
|
f"are missing from active session attached tools (Connected ≠ attached tools, #708)."
|
||||||
|
)
|
||||||
|
remediation.append(
|
||||||
|
"Reconnect the IDE/client MCP session to attach namespaces to the active session "
|
||||||
|
f"(sanctioned path: {SANCTIONED_ATTACH_RECOVERY_TOOL}), then re-run full preflight "
|
||||||
|
"(gitea_whoami -> gitea_resolve_task_capability -> task). "
|
||||||
|
"Do not use direct imports, CLI API mutations, profile hopping, or session file overrides."
|
||||||
|
)
|
||||||
|
if not_connected:
|
||||||
|
reasons.append(
|
||||||
|
f"required MCP namespace(s) {not_connected} are absent from the connected-service "
|
||||||
|
"inventory: nothing reports them Connected, so there is no attachment to claim and "
|
||||||
|
"the Connected-but-unattached condition does not apply to them (#708)."
|
||||||
|
)
|
||||||
|
remediation.append(
|
||||||
|
f"Connect the required MCP server(s) {not_connected} through the client, then "
|
||||||
|
"reconnect the IDE/client MCP session so their namespaces attach "
|
||||||
|
f"(sanctioned path: {SANCTIONED_ATTACH_RECOVERY_TOOL}), and re-run full preflight. "
|
||||||
|
"Do not use direct imports, CLI API mutations, profile hopping, or session file overrides."
|
||||||
|
)
|
||||||
|
if contradictory:
|
||||||
|
reasons.append(
|
||||||
|
f"contradictory evidence for namespace(s) {contradictory}: reported attached to the "
|
||||||
|
"active session while absent from the connected-service inventory; neither statement "
|
||||||
|
"is proof, so attachment is treated as unproven (fail closed, #708)."
|
||||||
|
)
|
||||||
|
if attachment_healthy:
|
||||||
|
reasons.append(
|
||||||
|
"All required MCP server namespaces are connected and attached to the active session."
|
||||||
|
)
|
||||||
|
|
||||||
|
if startup_ordering_race:
|
||||||
|
reasons.append(
|
||||||
|
f"startup ordering race: namespace(s) {late_attaching} finished connecting after the "
|
||||||
|
"active session tool snapshot was taken, so they cannot appear in that snapshot (#708)."
|
||||||
|
)
|
||||||
|
if auto_attach_attempted and not auto_attach_succeeded:
|
||||||
|
reasons.append(
|
||||||
|
"automatic namespace attachment was attempted and did not succeed; only the sanctioned "
|
||||||
|
"client reconnect path remains."
|
||||||
|
)
|
||||||
|
if auto_recovered:
|
||||||
|
reasons.append("namespaces were automatically attached; no operator reconnect was required.")
|
||||||
|
|
||||||
|
return {
|
||||||
|
"success": attachment_healthy,
|
||||||
|
"attachment_healthy": attachment_healthy,
|
||||||
|
"discovery_status": discovery_status,
|
||||||
|
"connected_servers": connected,
|
||||||
|
"attached_session_namespaces": list(attached),
|
||||||
|
"missing_namespaces": missing,
|
||||||
|
"not_connected_namespaces": not_connected,
|
||||||
|
"contradictory_namespaces": contradictory,
|
||||||
|
"proof_of_connected_vs_attached": proof,
|
||||||
|
"namespace_conditions": namespace_conditions,
|
||||||
|
"error_type": error_type,
|
||||||
|
"error_types": error_types,
|
||||||
|
"reasons": reasons,
|
||||||
|
"remediation": remediation,
|
||||||
|
"exact_next_action": (
|
||||||
|
"None; session tool namespaces attached."
|
||||||
|
if attachment_healthy
|
||||||
|
else (
|
||||||
|
f"Connect the required MCP server(s) {not_connected} through the client, then "
|
||||||
|
"reconnect the IDE/client MCP session so their tool namespaces attach to the "
|
||||||
|
"active session. Do not use direct imports, CLI API mutations, profile hopping, "
|
||||||
|
"or session-state overrides."
|
||||||
|
if not_connected
|
||||||
|
else (
|
||||||
|
"Reconnect the IDE/client MCP session so tool namespaces attach to the active "
|
||||||
|
"session. Do not use direct imports, CLI API mutations, profile hopping, or "
|
||||||
|
"session-state overrides."
|
||||||
|
)
|
||||||
|
)
|
||||||
|
),
|
||||||
|
"unsafe_fallback_policy": UNSAFE_FALLBACK_WARNING,
|
||||||
|
"sanctioned_recovery_tool": SANCTIONED_ATTACH_RECOVERY_TOOL,
|
||||||
|
"reconnect_required": reconnect_required,
|
||||||
|
"auto_attach_attempted": bool(auto_attach_attempted),
|
||||||
|
"auto_recovered": auto_recovered,
|
||||||
|
"startup_ordering_race": startup_ordering_race,
|
||||||
|
"late_attaching_namespaces": late_attaching,
|
||||||
|
# Secret-free structured signals (#708 AC5). Namespace names and counts only:
|
||||||
|
# never tokens, endpoints, env values, or filesystem paths.
|
||||||
|
"telemetry": {
|
||||||
|
"connected_count": len(connected),
|
||||||
|
"attached_count": len(attached),
|
||||||
|
"required_count": len(required),
|
||||||
|
"missing_count": len(missing),
|
||||||
|
"not_connected_count": len(not_connected),
|
||||||
|
"contradictory_count": len(contradictory),
|
||||||
|
"required_attached_count": sum(1 for r in required if r in attached),
|
||||||
|
"discovery_status": discovery_status,
|
||||||
|
"discovery_cache_hit": (
|
||||||
|
None if discovery_cache_hit is None else bool(discovery_cache_hit)
|
||||||
|
),
|
||||||
|
"discovery_cache_age_seconds": (
|
||||||
|
float(discovery_cache_age_seconds)
|
||||||
|
if isinstance(discovery_cache_age_seconds, (int, float))
|
||||||
|
else None
|
||||||
|
),
|
||||||
|
"reconnect_required": reconnect_required,
|
||||||
|
"auto_attach_attempted": bool(auto_attach_attempted),
|
||||||
|
"auto_recovered": auto_recovered,
|
||||||
|
"startup_ordering_race": startup_ordering_race,
|
||||||
|
"error_type": error_type,
|
||||||
|
"error_types": error_types,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def required_namespace_for_attachment(task: str) -> str | None:
|
||||||
|
"""Map a mutation task to the MCP namespace that must be *attached* (#708)."""
|
||||||
|
return ATTACHMENT_GATED_TASKS.get((task or "").strip())
|
||||||
|
|
||||||
|
|
||||||
|
def attachment_gate_from_session(
|
||||||
|
task: str,
|
||||||
|
session_attachment: dict[str, dict[str, Any]] | None,
|
||||||
|
) -> list[str]:
|
||||||
|
"""Fail-closed gate on recorded connected-but-unattached namespaces (#708).
|
||||||
|
|
||||||
|
Mirrors :func:`mutation_gate_from_session`: a namespace that has not been
|
||||||
|
assessed yet does not gate, so this never blocks a session that simply has
|
||||||
|
not run the assessment. Once an assessment records the namespace required
|
||||||
|
for *task* as Connected-but-unattached, the mutation fails closed and the
|
||||||
|
only offered recovery is the sanctioned client reconnect path.
|
||||||
|
"""
|
||||||
|
ns = required_namespace_for_attachment(task)
|
||||||
|
if not ns:
|
||||||
|
return []
|
||||||
|
store = session_attachment or {}
|
||||||
|
entry = store.get(ns)
|
||||||
|
if not entry:
|
||||||
|
return []
|
||||||
|
if entry.get("attached") and entry.get("attachment_healthy"):
|
||||||
|
return []
|
||||||
|
|
||||||
|
what = task or "mutation"
|
||||||
|
connected = entry.get("connected")
|
||||||
|
detail = entry.get("condition") or entry.get("error_type")
|
||||||
|
if connected is True:
|
||||||
|
# Only here has anything actually reported the service Connected, so only here may
|
||||||
|
# the block say so.
|
||||||
|
detail = detail or ERROR_CONNECTED_NAMESPACES_MISSING
|
||||||
|
blocked = (
|
||||||
|
f"live MCP namespace '{ns}' is recorded {detail}: the host reports Connected but the "
|
||||||
|
f"namespace is not attached to the active session tool surface; reconnect the "
|
||||||
|
f"IDE/client MCP session and re-run preflight before {what} "
|
||||||
|
"(fail closed, #708)"
|
||||||
|
)
|
||||||
|
elif connected is False:
|
||||||
|
detail = ERROR_REQUIRED_NAMESPACES_NOT_CONNECTED
|
||||||
|
blocked = (
|
||||||
|
f"live MCP namespace '{ns}' is recorded {detail}: it is absent from the "
|
||||||
|
f"connected-service inventory, so it is neither connected nor attached and no "
|
||||||
|
f"Connected status is claimed for it; connect the required MCP server, then "
|
||||||
|
f"reconnect the IDE/client MCP session and re-run preflight before {what} "
|
||||||
|
"(fail closed, #708)"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
# Connected status was never recorded. Refuse without asserting either condition.
|
||||||
|
detail = detail or ERROR_CONNECTED_NAMESPACES_MISSING
|
||||||
|
blocked = (
|
||||||
|
f"live MCP namespace '{ns}' is recorded {detail} with no connected-status evidence, "
|
||||||
|
f"so attachment to the active session tool surface is unproven; reconnect the "
|
||||||
|
f"IDE/client MCP session and re-run preflight before {what} "
|
||||||
|
"(fail closed, #708)"
|
||||||
|
)
|
||||||
|
return [blocked, UNSAFE_FALLBACK_WARNING]
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
def _as_list(value: Any) -> list[str] | None:
|
def _as_list(value: Any) -> list[str] | None:
|
||||||
if value is None:
|
if value is None:
|
||||||
@@ -104,12 +452,23 @@ def classify_namespace_probe(
|
|||||||
profile: str | None = None,
|
profile: str | None = None,
|
||||||
configured: bool = True,
|
configured: bool = True,
|
||||||
probe_source: str | None = None,
|
probe_source: str | None = None,
|
||||||
|
worker_identity: str | None = None,
|
||||||
|
generation_id: str | None = None,
|
||||||
|
registry: Any | None = None,
|
||||||
|
pid_alive_probe: Any | None = None,
|
||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""Classify whether a required tool is callable through a live namespace.
|
"""Classify whether a required tool is callable through a live namespace.
|
||||||
|
|
||||||
``registered_tools`` is static/server-side evidence. ``probe_result`` is
|
``registered_tools`` is static/server-side evidence. ``probe_result`` is
|
||||||
live invocation evidence. Only ``probe_source=client_namespace`` proves the
|
live invocation evidence. Only ``probe_source=client_namespace`` proves the
|
||||||
IDE-managed path; ``offline_spawn`` is an offline subprocess check only.
|
IDE-managed path; ``offline_spawn`` is an offline subprocess check only.
|
||||||
|
|
||||||
|
#948: ``worker_identity``/``generation_id``/``registry`` carry the
|
||||||
|
client/session ownership evidence. Provenance is resolved by
|
||||||
|
``mcp_worker_identity.assess_provenance`` — the same call
|
||||||
|
``gitea_get_runtime_context`` makes — so the two surfaces cannot report
|
||||||
|
different provenance for one process. Omitting them yields the fail-closed
|
||||||
|
``unproven`` verdict, never a fabricated ``client_managed``.
|
||||||
"""
|
"""
|
||||||
ns = (namespace or "").strip()
|
ns = (namespace or "").strip()
|
||||||
tool = required_tool or REQUIRED_NAMESPACE_TOOLS.get(ns) or "gitea_whoami"
|
tool = required_tool or REQUIRED_NAMESPACE_TOOLS.get(ns) or "gitea_whoami"
|
||||||
@@ -225,6 +584,33 @@ def classify_namespace_probe(
|
|||||||
# on bad data without treating success as IDE proof).
|
# on bad data without treating success as IDE proof).
|
||||||
blocks = namespace_health_blocks_task("merge_pr", healthy)
|
blocks = namespace_health_blocks_task("merge_pr", healthy)
|
||||||
|
|
||||||
|
import gitea_config
|
||||||
|
import mcp_worker_identity
|
||||||
|
|
||||||
|
raw_env = process.get("env") if isinstance(process, dict) else None
|
||||||
|
unconsumed_env = gitea_config.get_unconsumed_gitea_env_overrides(raw_env)
|
||||||
|
|
||||||
|
# #948: one authority, shared with gitea_get_runtime_context. The env is
|
||||||
|
# passed whole rather than through SAFE_ENV_KEYS — the allowlist exists to
|
||||||
|
# decide what may be *echoed*, and using it to decide what may be *believed*
|
||||||
|
# is what made this surface structurally unable to report client_managed.
|
||||||
|
# ``declared_only``: ``process`` describes an observed peer, not this
|
||||||
|
# interpreter. Its stdin is unavailable and its launcher-config env is
|
||||||
|
# inherited from whatever shell started it, so only an explicit declaration
|
||||||
|
# is evidence. Absence of one is ``unproven``, not an asserted manual launch.
|
||||||
|
provenance_verdict = mcp_worker_identity.assess_provenance(
|
||||||
|
registry=registry,
|
||||||
|
worker_identity=worker_identity,
|
||||||
|
generation_id=generation_id,
|
||||||
|
env=raw_env if isinstance(raw_env, dict) else {},
|
||||||
|
namespace=ns,
|
||||||
|
profile=profile_name,
|
||||||
|
pid_alive_probe=pid_alive_probe,
|
||||||
|
declared_only=True,
|
||||||
|
)
|
||||||
|
provenance = provenance_verdict["provenance"]
|
||||||
|
is_client_managed = provenance_verdict["is_client_managed"]
|
||||||
|
|
||||||
return {
|
return {
|
||||||
"success": healthy,
|
"success": healthy,
|
||||||
"healthy": healthy,
|
"healthy": healthy,
|
||||||
@@ -240,6 +626,18 @@ def classify_namespace_probe(
|
|||||||
"error_message": error_message or None,
|
"error_message": error_message or None,
|
||||||
"reasons": reasons,
|
"reasons": reasons,
|
||||||
"remediation": remediation,
|
"remediation": remediation,
|
||||||
|
"provenance": provenance,
|
||||||
|
"is_client_managed": is_client_managed,
|
||||||
|
# Every non-client-session verdict fails closed. Consumers that only
|
||||||
|
# need "may this mutate?" read this and stay correct across the #948
|
||||||
|
# vocabulary split between ``manual_launch`` and ``unproven``.
|
||||||
|
"provenance_fail_closed": provenance_verdict["fail_closed"],
|
||||||
|
"provenance_assessment": provenance_verdict,
|
||||||
|
"worker_identity": provenance_verdict["worker_identity"],
|
||||||
|
"session_id": provenance_verdict["session_id"],
|
||||||
|
"generation_id": provenance_verdict["generation_id"],
|
||||||
|
"client_name": provenance_verdict["client_name"],
|
||||||
|
"unconsumed_gitea_env": unconsumed_env,
|
||||||
"diagnostics": {
|
"diagnostics": {
|
||||||
"namespace": ns,
|
"namespace": ns,
|
||||||
"required_tool": tool,
|
"required_tool": tool,
|
||||||
@@ -248,6 +646,16 @@ def classify_namespace_probe(
|
|||||||
"env": env_summary,
|
"env": env_summary,
|
||||||
"config_path": config_path,
|
"config_path": config_path,
|
||||||
"probe_source": source,
|
"probe_source": source,
|
||||||
|
"provenance": provenance,
|
||||||
|
"is_client_managed": is_client_managed,
|
||||||
|
"provenance_fail_closed": provenance_verdict["fail_closed"],
|
||||||
|
"provenance_blocker_kind": provenance_verdict["blocker_kind"],
|
||||||
|
"provenance_scope": provenance_verdict["scope"],
|
||||||
|
"worker_identity": provenance_verdict["worker_identity"],
|
||||||
|
"session_id": provenance_verdict["session_id"],
|
||||||
|
"generation_id": provenance_verdict["generation_id"],
|
||||||
|
"client_name": provenance_verdict["client_name"],
|
||||||
|
"unconsumed_gitea_env": unconsumed_env,
|
||||||
},
|
},
|
||||||
"blocks_merge_workflow": blocks,
|
"blocks_merge_workflow": blocks,
|
||||||
}
|
}
|
||||||
|
|||||||
+35
-5
@@ -225,8 +225,33 @@ _RESTART_PATHS: tuple[RestartPath, ...] = (
|
|||||||
"exact_safe_next_action pointing at IDE/client reconnect; performs "
|
"exact_safe_next_action pointing at IDE/client reconnect; performs "
|
||||||
"no restart, thread spawn, config touch, or os._exit."
|
"no restart, thread spawn, config touch, or os._exit."
|
||||||
),
|
),
|
||||||
locations=("gitea_mcp_server.py (gitea_resolve_task_capability)",),
|
locations=(
|
||||||
references=("#685", "#657"),
|
"gitea_mcp_server.py (gitea_resolve_task_capability)",
|
||||||
|
"gitea_mcp_server.py (gitea_request_mcp_reconnect)",
|
||||||
|
"mcp_client_reconnect.py",
|
||||||
|
),
|
||||||
|
references=("#685", "#657", "#678"),
|
||||||
|
),
|
||||||
|
RestartPath(
|
||||||
|
path_id="codex_client_reconnect_request",
|
||||||
|
title="Sanctioned Codex/LLM reconnect request tool",
|
||||||
|
mechanism=(
|
||||||
|
"gitea_request_mcp_reconnect: agents invoke a report-only tool that "
|
||||||
|
"returns namespace/profile/pid/startup SHA/master SHA/boundary "
|
||||||
|
"status plus a typed operator blocker with exact client UI steps."
|
||||||
|
),
|
||||||
|
classification=CLASS_GUARDED_FAIL_CLOSED,
|
||||||
|
guard=(
|
||||||
|
"Report-only (#678): never restarts, kills, reloads, or edits "
|
||||||
|
"config; recovery is always host/operator reconnect. Forbidden "
|
||||||
|
"paths (pkill, touch, .env/config/session-state hacks) are listed "
|
||||||
|
"and never recommended."
|
||||||
|
),
|
||||||
|
locations=(
|
||||||
|
"mcp_client_reconnect.py",
|
||||||
|
"gitea_mcp_server.py (gitea_request_mcp_reconnect)",
|
||||||
|
),
|
||||||
|
references=("#678", "#630", "#685", "#657"),
|
||||||
),
|
),
|
||||||
RestartPath(
|
RestartPath(
|
||||||
path_id="manual_daemon_kill",
|
path_id="manual_daemon_kill",
|
||||||
@@ -271,7 +296,8 @@ _RESTART_PATHS: tuple[RestartPath, ...] = (
|
|||||||
title="Host/IDE MCP reconnect",
|
title="Host/IDE MCP reconnect",
|
||||||
mechanism=(
|
mechanism=(
|
||||||
"A manual `/mcp reconnect` (or equivalent host action) that the "
|
"A manual `/mcp reconnect` (or equivalent host action) that the "
|
||||||
"IDE performs to recreate the MCP client connection."
|
"IDE performs to recreate the MCP client connection. Agents obtain "
|
||||||
|
"exact UI steps via gitea_request_mcp_reconnect (#678)."
|
||||||
),
|
),
|
||||||
classification=CLASS_HOST_RESIDUAL,
|
classification=CLASS_HOST_RESIDUAL,
|
||||||
guard=(
|
guard=(
|
||||||
@@ -279,8 +305,12 @@ _RESTART_PATHS: tuple[RestartPath, ...] = (
|
|||||||
"gates point operators toward; documented as residual host "
|
"gates point operators toward; documented as residual host "
|
||||||
"behavior. No in-process code initiates it."
|
"behavior. No in-process code initiates it."
|
||||||
),
|
),
|
||||||
locations=("host/IDE",),
|
locations=(
|
||||||
references=("#584", "#656", "#657"),
|
"host/IDE",
|
||||||
|
"mcp_client_reconnect.py",
|
||||||
|
"gitea_mcp_server.py (gitea_request_mcp_reconnect)",
|
||||||
|
),
|
||||||
|
references=("#584", "#656", "#657", "#678"),
|
||||||
residual_host=True,
|
residual_host=True,
|
||||||
),
|
),
|
||||||
RestartPath(
|
RestartPath(
|
||||||
|
|||||||
+7
-3
@@ -1,7 +1,10 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""Gitea MCP Server — exposes Gitea operations as MCP tools.
|
"""Gitea MCP Server — exposes Gitea operations as MCP tools.
|
||||||
|
|
||||||
Runs over stdio. All tools authenticate via macOS keychain (git credential fill).
|
The transport is selected by deployment configuration (GITEA_MCP_TRANSPORT) and
|
||||||
|
defaults to the local client-spawned transport when unset (#931); the permitted
|
||||||
|
set lives in mcp_transport_config. All tools authenticate via macOS keychain
|
||||||
|
(git credential fill).
|
||||||
"""
|
"""
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
@@ -43,8 +46,9 @@ check_conflict_markers()
|
|||||||
|
|
||||||
# #558 / #695: claim the official entrypoint before loading mutation modules.
|
# #558 / #695: claim the official entrypoint before loading mutation modules.
|
||||||
# This alone does NOT authorize mutations — gitea_mcp_server binds the live
|
# This alone does NOT authorize mutations — gitea_mcp_server binds the live
|
||||||
# native MCP transport (stdio) immediately before mcp.run. Import-only or
|
# native MCP transport immediately before mcp.run, over the configured
|
||||||
# offline launch without that bind fails closed on mutations.
|
# transport (#931). Import-only or offline launch without that bind fails
|
||||||
|
# closed on mutations.
|
||||||
try:
|
try:
|
||||||
import mcp_daemon_guard
|
import mcp_daemon_guard
|
||||||
|
|
||||||
|
|||||||
@@ -588,9 +588,13 @@ def save_state(
|
|||||||
bool(prov.get("production_native_mcp_transport")),
|
bool(prov.get("production_native_mcp_transport")),
|
||||||
)
|
)
|
||||||
body.setdefault("transport", prov.get("transport"))
|
body.setdefault("transport", prov.get("transport"))
|
||||||
|
# #931: record the bound transport identifier itself, so a durable
|
||||||
|
# decision lock names which transport performed the mutation.
|
||||||
|
body.setdefault("bound_transport", prov.get("bound_transport"))
|
||||||
except Exception:
|
except Exception:
|
||||||
body.setdefault("native_mcp_transport", False)
|
body.setdefault("native_mcp_transport", False)
|
||||||
body.setdefault("transport", "untrusted")
|
body.setdefault("transport", "untrusted")
|
||||||
|
body.setdefault("bound_transport", None)
|
||||||
|
|
||||||
envelope = {
|
envelope = {
|
||||||
"kind": kind,
|
"kind": kind,
|
||||||
|
|||||||
@@ -0,0 +1,292 @@
|
|||||||
|
"""Single authoritative source for the bound MCP transport identifier (#931).
|
||||||
|
|
||||||
|
Before this module the transport was a literal, passed once at the bottom of
|
||||||
|
``gitea_mcp_server`` as ``bind_native_mcp_transport(transport="stdio")``. Every
|
||||||
|
guard that later asks "is this a trusted native session" resolves that question
|
||||||
|
through the value bound there, so the literal was effectively a constant in the
|
||||||
|
authorization chain rather than configuration.
|
||||||
|
|
||||||
|
This module is the seam. It owns three things and nothing else:
|
||||||
|
|
||||||
|
- the permitted set of transport identifiers,
|
||||||
|
- the default used when deployment configuration says nothing,
|
||||||
|
- the resolution of the configured value into a validated identifier.
|
||||||
|
|
||||||
|
It deliberately holds no state. The *bound* transport is pinned once, at bind
|
||||||
|
time, into the process-local native-runtime record owned by
|
||||||
|
:mod:`mcp_daemon_guard`, and is read back through
|
||||||
|
``mcp_daemon_guard.bound_transport()``. That split matters: configuration is
|
||||||
|
read exactly once, before any tool can dispatch, so a later environment change
|
||||||
|
cannot move the value a guard observes — the same pinning rule already applied
|
||||||
|
to the session-state root under #695 AC2.
|
||||||
|
|
||||||
|
Nothing here consumes tool arguments, request bodies, or provenance fields. The
|
||||||
|
only input is the deployment environment, read at bind time.
|
||||||
|
|
||||||
|
Standing up a listener for a non-stdio transport is #938; this module only
|
||||||
|
makes the identifier expressible and validated.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
from typing import Any, Mapping
|
||||||
|
|
||||||
|
# Deployment configuration key. Read once, at bind time, and never again.
|
||||||
|
TRANSPORT_ENV = "GITEA_MCP_TRANSPORT"
|
||||||
|
|
||||||
|
# The local, client-spawned transport. Unset configuration resolves to this,
|
||||||
|
# which is what keeps every existing stdio deployment byte-identical.
|
||||||
|
DEFAULT_TRANSPORT = "stdio"
|
||||||
|
|
||||||
|
# The sanctioned remote transport identifier. Accepting it here is what makes
|
||||||
|
# the bind pluggable; the endpoint that serves it belongs to #938. The name
|
||||||
|
# matches the MCP transport name so no second vocabulary has to be mapped.
|
||||||
|
REMOTE_TRANSPORT = "streamable-http"
|
||||||
|
|
||||||
|
# The permitted set. This is the only place transport identifiers are
|
||||||
|
# enumerated; guards consult it rather than restating any member.
|
||||||
|
#
|
||||||
|
# ``sse`` is a real MCP transport and is deliberately absent: it is the
|
||||||
|
# superseded remote transport, and admitting it would give the deployment two
|
||||||
|
# remote paths to reason about. An unregistered identifier must fail closed at
|
||||||
|
# bind time, and ``sse`` is held to that rule like any other.
|
||||||
|
SUPPORTED_TRANSPORTS = frozenset({DEFAULT_TRANSPORT, REMOTE_TRANSPORT})
|
||||||
|
|
||||||
|
# Recognition is not execution authorization (#931, review 635 B1/B2).
|
||||||
|
#
|
||||||
|
# SUPPORTED_TRANSPORTS answers "is this an identifier this system knows, and may
|
||||||
|
# it be bound, pinned and recorded?". It deliberately includes the remote
|
||||||
|
# identifier, because #931 requires the bind to become pluggable.
|
||||||
|
#
|
||||||
|
# EXECUTABLE_TRANSPORTS answers a strictly narrower question: "is this entrypoint
|
||||||
|
# commissioned to actually *serve* on that transport?". Only the local transport
|
||||||
|
# is. Handing ``streamable-http`` to ``mcp.run`` would start FastMCP's HTTP
|
||||||
|
# listener with no authentication, no TLS and no per-request principal — the
|
||||||
|
# endpoint #938 owns and gates. Recognition must therefore never imply execution.
|
||||||
|
#
|
||||||
|
# #938 commissions the remote listener by adding REMOTE_TRANSPORT here, together
|
||||||
|
# with the authentication and principal boundary its acceptance criteria require.
|
||||||
|
EXECUTABLE_TRANSPORTS = frozenset({DEFAULT_TRANSPORT})
|
||||||
|
|
||||||
|
# Which issue owns commissioning each recognized-but-not-executable transport.
|
||||||
|
# Used to make the refusal actionable rather than a generic denial.
|
||||||
|
TRANSPORT_EXECUTION_OWNER = {REMOTE_TRANSPORT: "#938"}
|
||||||
|
|
||||||
|
BLOCKER_TRANSPORT_NOT_BOUND = "transport_not_bound"
|
||||||
|
BLOCKER_TRANSPORT_NOT_RECOGNIZED = "transport_not_recognized"
|
||||||
|
BLOCKER_LISTENER_NOT_COMMISSIONED = "transport_listener_not_commissioned"
|
||||||
|
|
||||||
|
SOURCE_CONFIGURED = "deployment_configuration"
|
||||||
|
SOURCE_DEFAULT = "default"
|
||||||
|
|
||||||
|
|
||||||
|
class TransportConfigurationError(ValueError):
|
||||||
|
"""Raised when configured transport is outside :data:`SUPPORTED_TRANSPORTS`."""
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_transport(value: Any) -> str:
|
||||||
|
"""Canonical form of a transport identifier; ``""`` when there is none.
|
||||||
|
|
||||||
|
Non-string values normalize to ``""`` rather than being coerced, so a
|
||||||
|
structured object smuggled in from a caller can never match a member of the
|
||||||
|
permitted set.
|
||||||
|
"""
|
||||||
|
if not isinstance(value, str):
|
||||||
|
return ""
|
||||||
|
return value.strip().lower()
|
||||||
|
|
||||||
|
|
||||||
|
def supported_transports() -> tuple[str, ...]:
|
||||||
|
"""Permitted identifiers, sorted, for messages and status payloads."""
|
||||||
|
return tuple(sorted(SUPPORTED_TRANSPORTS))
|
||||||
|
|
||||||
|
|
||||||
|
def is_supported_transport(value: Any) -> bool:
|
||||||
|
"""True when *value* normalizes to a member of the permitted set."""
|
||||||
|
return normalize_transport(value) in SUPPORTED_TRANSPORTS
|
||||||
|
|
||||||
|
|
||||||
|
def is_remote_transport(value: Any) -> bool:
|
||||||
|
"""True when *value* is a permitted transport that is not the local one."""
|
||||||
|
name = normalize_transport(value)
|
||||||
|
return name in SUPPORTED_TRANSPORTS and name != DEFAULT_TRANSPORT
|
||||||
|
|
||||||
|
|
||||||
|
def executable_transports() -> tuple[str, ...]:
|
||||||
|
"""Transports this entrypoint is commissioned to serve, sorted."""
|
||||||
|
return tuple(sorted(EXECUTABLE_TRANSPORTS))
|
||||||
|
|
||||||
|
|
||||||
|
def is_executable_transport(value: Any) -> bool:
|
||||||
|
"""True when *value* may actually be served by this entrypoint (#931).
|
||||||
|
|
||||||
|
Strictly narrower than :func:`is_supported_transport`. A recognized
|
||||||
|
identifier that is not executable is a correct, fully-bound configuration
|
||||||
|
whose listener has simply not been commissioned yet.
|
||||||
|
"""
|
||||||
|
return normalize_transport(value) in EXECUTABLE_TRANSPORTS
|
||||||
|
|
||||||
|
|
||||||
|
def assess_transport_execution(value: Any) -> dict[str, Any]:
|
||||||
|
"""Structured serve-authorization verdict for a bound transport (#931).
|
||||||
|
|
||||||
|
This is the decision that separates a *recognized* transport from one this
|
||||||
|
entrypoint may execute. It is deliberately a pure function of the bound
|
||||||
|
identifier so the serve path cannot reach a listener the deployment has not
|
||||||
|
commissioned.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
value: The bound transport identifier, or ``None`` when unbound.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict with ``transport``, ``recognized``, ``executable``, ``allowed``,
|
||||||
|
``blocker_kind``, ``owner_issue``, ``reasons`` and
|
||||||
|
``exact_next_action``. ``allowed`` is true only for a bound, recognized,
|
||||||
|
commissioned transport.
|
||||||
|
"""
|
||||||
|
name = normalize_transport(value)
|
||||||
|
|
||||||
|
if not name:
|
||||||
|
return {
|
||||||
|
"transport": None,
|
||||||
|
"recognized": False,
|
||||||
|
"executable": False,
|
||||||
|
"allowed": False,
|
||||||
|
"blocker_kind": BLOCKER_TRANSPORT_NOT_BOUND,
|
||||||
|
"owner_issue": None,
|
||||||
|
"supported_transports": list(supported_transports()),
|
||||||
|
"executable_transports": list(executable_transports()),
|
||||||
|
"reasons": [
|
||||||
|
"no transport is bound; the serve path is fail-closed until "
|
||||||
|
"bind_native_mcp_transport succeeds (#695/#931)"
|
||||||
|
],
|
||||||
|
"exact_next_action": (
|
||||||
|
"Launch through the canonical entrypoint so "
|
||||||
|
"bind_native_mcp_transport runs before tool service."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
recognized = name in SUPPORTED_TRANSPORTS
|
||||||
|
if not recognized:
|
||||||
|
return {
|
||||||
|
"transport": name,
|
||||||
|
"recognized": False,
|
||||||
|
"executable": False,
|
||||||
|
"allowed": False,
|
||||||
|
"blocker_kind": BLOCKER_TRANSPORT_NOT_RECOGNIZED,
|
||||||
|
"owner_issue": None,
|
||||||
|
"supported_transports": list(supported_transports()),
|
||||||
|
"executable_transports": list(executable_transports()),
|
||||||
|
"reasons": [
|
||||||
|
f"transport {name!r} is not a registered MCP transport (#931); "
|
||||||
|
"it should have been refused at bind time"
|
||||||
|
],
|
||||||
|
"exact_next_action": (
|
||||||
|
f"Set {TRANSPORT_ENV} to one of {list(supported_transports())}."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
if name in EXECUTABLE_TRANSPORTS:
|
||||||
|
return {
|
||||||
|
"transport": name,
|
||||||
|
"recognized": True,
|
||||||
|
"executable": True,
|
||||||
|
"allowed": True,
|
||||||
|
"blocker_kind": None,
|
||||||
|
"owner_issue": None,
|
||||||
|
"supported_transports": list(supported_transports()),
|
||||||
|
"executable_transports": list(executable_transports()),
|
||||||
|
"reasons": [],
|
||||||
|
"exact_next_action": None,
|
||||||
|
}
|
||||||
|
|
||||||
|
owner = TRANSPORT_EXECUTION_OWNER.get(name)
|
||||||
|
owner_text = owner or "the issue that commissions this transport's listener"
|
||||||
|
return {
|
||||||
|
"transport": name,
|
||||||
|
"recognized": True,
|
||||||
|
"executable": False,
|
||||||
|
"allowed": False,
|
||||||
|
"blocker_kind": BLOCKER_LISTENER_NOT_COMMISSIONED,
|
||||||
|
"owner_issue": owner,
|
||||||
|
"supported_transports": list(supported_transports()),
|
||||||
|
"executable_transports": list(executable_transports()),
|
||||||
|
"reasons": [
|
||||||
|
f"transport {name!r} is registered and was bound and recorded, but "
|
||||||
|
f"this entrypoint is not commissioned to serve it (#931). Serving it "
|
||||||
|
f"would start a listener with no authentication, no transport "
|
||||||
|
f"security and no per-request principal; that endpoint is owned by "
|
||||||
|
f"{owner_text}."
|
||||||
|
],
|
||||||
|
"exact_next_action": (
|
||||||
|
f"Serve on {DEFAULT_TRANSPORT} until {owner_text} commissions the "
|
||||||
|
f"{name!r} listener with its authentication and principal boundary, "
|
||||||
|
f"which adds {name!r} to EXECUTABLE_TRANSPORTS."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_configured_transport(
|
||||||
|
env: Mapping[str, str] | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Resolve the deployment-configured transport without raising.
|
||||||
|
|
||||||
|
Returns the resolution rather than a bare string so a caller can tell an
|
||||||
|
unset value (which legitimately yields :data:`DEFAULT_TRANSPORT`) from a
|
||||||
|
configured value that is not permitted (which must fail closed, never
|
||||||
|
silently degrade to the default).
|
||||||
|
|
||||||
|
Args:
|
||||||
|
env: Environment mapping to read; defaults to ``os.environ``.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
dict with ``transport`` (normalized; the default when unset),
|
||||||
|
``configured``, ``source``, ``raw``, ``supported``, and ``reasons``.
|
||||||
|
"""
|
||||||
|
source_env = os.environ if env is None else env
|
||||||
|
raw = source_env.get(TRANSPORT_ENV)
|
||||||
|
normalized = normalize_transport(raw)
|
||||||
|
configured = bool(normalized)
|
||||||
|
|
||||||
|
if not configured:
|
||||||
|
return {
|
||||||
|
"transport": DEFAULT_TRANSPORT,
|
||||||
|
"configured": False,
|
||||||
|
"source": SOURCE_DEFAULT,
|
||||||
|
"raw": raw,
|
||||||
|
"supported": True,
|
||||||
|
"supported_transports": list(supported_transports()),
|
||||||
|
"env_key": TRANSPORT_ENV,
|
||||||
|
"reasons": [],
|
||||||
|
}
|
||||||
|
|
||||||
|
supported = normalized in SUPPORTED_TRANSPORTS
|
||||||
|
reasons: list[str] = []
|
||||||
|
if not supported:
|
||||||
|
reasons.append(
|
||||||
|
f"{TRANSPORT_ENV}={normalized!r} is not a registered MCP transport "
|
||||||
|
f"(#931). Registered: {list(supported_transports())}."
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"transport": normalized,
|
||||||
|
"configured": True,
|
||||||
|
"source": SOURCE_CONFIGURED,
|
||||||
|
"raw": raw,
|
||||||
|
"supported": supported,
|
||||||
|
"supported_transports": list(supported_transports()),
|
||||||
|
"env_key": TRANSPORT_ENV,
|
||||||
|
"reasons": reasons,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def require_configured_transport(env: Mapping[str, str] | None = None) -> str:
|
||||||
|
"""Resolved transport identifier, or raise when it is not permitted.
|
||||||
|
|
||||||
|
Raises:
|
||||||
|
TransportConfigurationError: the configured identifier is unregistered.
|
||||||
|
"""
|
||||||
|
resolution = resolve_configured_transport(env)
|
||||||
|
if not resolution["supported"]:
|
||||||
|
raise TransportConfigurationError("; ".join(resolution["reasons"]))
|
||||||
|
return str(resolution["transport"])
|
||||||
File diff suppressed because it is too large
Load Diff
+57
-12
@@ -475,24 +475,65 @@ def reconcile_after_restart(
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
# --- sessions -------------------------------------------------------
|
# --- sessions (#969: dead-owner / PID-reuse lifecycle) ---------------
|
||||||
|
# Prefer a precomputed fleet report from the gather/apply path when present;
|
||||||
|
# otherwise classify pure from inventory (injectable checkers stay default).
|
||||||
sessions = [s for s in (inventory.get("sessions") or []) if isinstance(s, Mapping)]
|
sessions = [s for s in (inventory.get("sessions") or []) if isinstance(s, Mapping)]
|
||||||
orphan_sessions = [
|
leases_for_sessions = [
|
||||||
s
|
L for L in (inventory.get("leases") or []) if isinstance(L, Mapping)
|
||||||
for s in sessions
|
|
||||||
if str(s.get("status") or "").lower() == "active"
|
|
||||||
and s.get("pid") is not None
|
|
||||||
and not lease_lifecycle.is_process_alive(s.get("pid"))
|
|
||||||
]
|
]
|
||||||
if orphan_sessions:
|
fleet_report = inventory.get("session_fleet")
|
||||||
|
if isinstance(fleet_report, Mapping) and "retireable_session_ids" in fleet_report:
|
||||||
|
fleet_details = dict(fleet_report)
|
||||||
|
retireable_ids = list(fleet_details.get("retireable_session_ids") or [])
|
||||||
|
resolved = bool(fleet_details.get("sessions_dimension_resolved", not retireable_ids))
|
||||||
|
else:
|
||||||
|
try:
|
||||||
|
import session_lifecycle as _sl
|
||||||
|
|
||||||
|
client_managed = inventory.get("client_managed_session_ids")
|
||||||
|
cm_set = None
|
||||||
|
if isinstance(client_managed, (list, tuple, set, frozenset)):
|
||||||
|
cm_set = {str(x) for x in client_managed}
|
||||||
|
fleet = _sl.classify_sessions(
|
||||||
|
sessions,
|
||||||
|
leases=leases_for_sessions,
|
||||||
|
now=started,
|
||||||
|
client_managed_sessions=cm_set,
|
||||||
|
)
|
||||||
|
fleet_details = fleet.as_dict()
|
||||||
|
retireable_ids = list(fleet_details.get("retireable_session_ids") or [])
|
||||||
|
resolved = bool(fleet_details.get("sessions_dimension_resolved"))
|
||||||
|
except Exception as exc: # noqa: BLE001 — fail closed to legacy signal
|
||||||
|
# Legacy fallback: dead-pid active rows only (pre-#969 behaviour).
|
||||||
|
orphan_sessions = [
|
||||||
|
s
|
||||||
|
for s in sessions
|
||||||
|
if str(s.get("status") or "").lower() == "active"
|
||||||
|
and s.get("pid") is not None
|
||||||
|
and not lease_lifecycle.is_process_alive(s.get("pid"))
|
||||||
|
]
|
||||||
|
retireable_ids = [s.get("session_id") for s in orphan_sessions]
|
||||||
|
resolved = not orphan_sessions
|
||||||
|
fleet_details = {
|
||||||
|
"total_sessions": len(sessions),
|
||||||
|
"retireable_session_ids": retireable_ids,
|
||||||
|
"legacy_fallback": True,
|
||||||
|
"fallback_error": str(exc),
|
||||||
|
}
|
||||||
|
|
||||||
|
if not resolved and retireable_ids:
|
||||||
items.append(
|
items.append(
|
||||||
_item(
|
_item(
|
||||||
DIM_SESSIONS,
|
DIM_SESSIONS,
|
||||||
ITEM_UNRESOLVED,
|
ITEM_UNRESOLVED,
|
||||||
f"{len(orphan_sessions)} active session row(s) with dead owner pid",
|
f"{len(retireable_ids)} session row(s) with dead/reused owner "
|
||||||
|
f"await retirement",
|
||||||
details={
|
details={
|
||||||
"orphan_session_ids": [s.get("session_id") for s in orphan_sessions],
|
"orphan_session_ids": retireable_ids,
|
||||||
|
"retireable_session_ids": retireable_ids,
|
||||||
"total_sessions": len(sessions),
|
"total_sessions": len(sessions),
|
||||||
|
"fleet": fleet_details,
|
||||||
},
|
},
|
||||||
follow_up=True,
|
follow_up=True,
|
||||||
)
|
)
|
||||||
@@ -502,8 +543,12 @@ def reconcile_after_restart(
|
|||||||
_item(
|
_item(
|
||||||
DIM_SESSIONS,
|
DIM_SESSIONS,
|
||||||
ITEM_RESOLVED,
|
ITEM_RESOLVED,
|
||||||
f"{len(sessions)} session row(s) reconciled (no dead-pid orphans)",
|
f"{len(sessions)} session row(s) reconciled "
|
||||||
details={"total_sessions": len(sessions)},
|
f"(no retireable dead/reused owners)",
|
||||||
|
details={
|
||||||
|
"total_sessions": len(sessions),
|
||||||
|
"fleet": fleet_details,
|
||||||
|
},
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,3 @@
|
|||||||
|
[pytest]
|
||||||
|
testpaths = tests
|
||||||
|
norecursedirs = branches .git venv __pycache__ graphify-out
|
||||||
@@ -0,0 +1,583 @@
|
|||||||
|
"""Scoped MCP recovery playbook (#669).
|
||||||
|
|
||||||
|
Operational recovery must prefer the *narrowest* action that can fix the
|
||||||
|
symptom. Full MCP / host restarts are last-resort rungs on a documented
|
||||||
|
ladder; the coordinator refuses those rungs unless a prior attempt log
|
||||||
|
shows narrower recoveries already failed (or break-glass is authorized).
|
||||||
|
|
||||||
|
This module is pure classification and recommendation:
|
||||||
|
|
||||||
|
* No network, filesystem, or process I/O.
|
||||||
|
* Never restarts anything.
|
||||||
|
* Narrow recovery *execution* is delegated to existing tools/docs (linked
|
||||||
|
per rung) — the playbook records which rung to try next and whether
|
||||||
|
escalation to a broad restart is allowed.
|
||||||
|
|
||||||
|
Design lineage: umbrella #655, class matrix #663, coordinator #658,
|
||||||
|
auto-reconnect #584, stale-runtime #610, contamination #630, audit #665.
|
||||||
|
Vision #652 / roadmap #653.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from enum import Enum
|
||||||
|
from typing import Any, Mapping, Sequence
|
||||||
|
|
||||||
|
PLAYBOOK_VERSION = "1.0.0-issue-669"
|
||||||
|
|
||||||
|
# Attempt outcomes that count as "tried and insufficient" for escalation.
|
||||||
|
INSUFFICIENT_OUTCOMES = frozenset(
|
||||||
|
{
|
||||||
|
"failed",
|
||||||
|
"insufficient",
|
||||||
|
"denied",
|
||||||
|
"unresolved",
|
||||||
|
"timeout",
|
||||||
|
"error",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Break-glass / operator override still records that the ladder was skipped.
|
||||||
|
OUTCOME_BREAK_GLASS = "break_glass"
|
||||||
|
OUTCOME_SUCCESS = "success"
|
||||||
|
OUTCOME_SKIPPED = "skipped"
|
||||||
|
|
||||||
|
|
||||||
|
class RecoveryAction(str, Enum):
|
||||||
|
"""Ordered recovery ladder (narrow → broad)."""
|
||||||
|
|
||||||
|
CLIENT_RECONNECT = "client_reconnect"
|
||||||
|
CAPABILITY_REFRESH = "capability_refresh"
|
||||||
|
SESSION_RECONNECT = "session_reconnect"
|
||||||
|
CONFIGURATION_RELOAD = "configuration_reload"
|
||||||
|
LEASE_RECOVERY = "lease_recovery"
|
||||||
|
WORKER_RESTART = "worker_restart"
|
||||||
|
ROLE_RUNTIME_RESTART = "role_runtime_restart"
|
||||||
|
CONNECTOR_RESTART = "connector_restart"
|
||||||
|
ROLLING_MCP_RESTART = "rolling_mcp_restart"
|
||||||
|
FULL_MCP_RESTART = "full_mcp_restart"
|
||||||
|
HOST_RESTART = "host_restart"
|
||||||
|
|
||||||
|
|
||||||
|
# Classes that require a prior narrow-attempt log (unless break-glass).
|
||||||
|
BROAD_RESTART_ACTIONS: frozenset[RecoveryAction] = frozenset(
|
||||||
|
{
|
||||||
|
RecoveryAction.ROLLING_MCP_RESTART,
|
||||||
|
RecoveryAction.FULL_MCP_RESTART,
|
||||||
|
RecoveryAction.HOST_RESTART,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Map #663 restart_class strings onto playbook actions.
|
||||||
|
RESTART_CLASS_TO_ACTION: dict[str, RecoveryAction] = {
|
||||||
|
"client_reconnect": RecoveryAction.CLIENT_RECONNECT,
|
||||||
|
"session_reconnect": RecoveryAction.SESSION_RECONNECT,
|
||||||
|
"configuration_reload": RecoveryAction.CONFIGURATION_RELOAD,
|
||||||
|
"worker_restart": RecoveryAction.WORKER_RESTART,
|
||||||
|
"role_runtime_restart": RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||||
|
"connector_restart": RecoveryAction.CONNECTOR_RESTART,
|
||||||
|
"rolling_mcp_restart": RecoveryAction.ROLLING_MCP_RESTART,
|
||||||
|
"full_mcp_restart": RecoveryAction.FULL_MCP_RESTART,
|
||||||
|
"host_restart": RecoveryAction.HOST_RESTART,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class RecoveryRung:
|
||||||
|
"""One rung on the recovery ladder."""
|
||||||
|
|
||||||
|
action: RecoveryAction
|
||||||
|
rank: int
|
||||||
|
summary: str
|
||||||
|
# Existing implementation or explicit delegation target.
|
||||||
|
implementation: str
|
||||||
|
issue_links: tuple[str, ...]
|
||||||
|
self_service: bool
|
||||||
|
# Restart-class permission when this rung is requested via coordinator.
|
||||||
|
restart_class: str | None = None
|
||||||
|
|
||||||
|
def as_dict(self) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"action": self.action.value,
|
||||||
|
"rank": self.rank,
|
||||||
|
"summary": self.summary,
|
||||||
|
"implementation": self.implementation,
|
||||||
|
"issue_links": list(self.issue_links),
|
||||||
|
"self_service": self.self_service,
|
||||||
|
"restart_class": self.restart_class,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# Canonical ladder. Rank 0 is narrowest.
|
||||||
|
RECOVERY_LADDER: tuple[RecoveryRung, ...] = (
|
||||||
|
RecoveryRung(
|
||||||
|
RecoveryAction.CLIENT_RECONNECT,
|
||||||
|
0,
|
||||||
|
"Reconnect the IDE/client MCP transport (EOF / transport flap).",
|
||||||
|
"Host auto-reconnect or explicit client reconnect; "
|
||||||
|
"docs/mcp-namespace-eof-recovery.md",
|
||||||
|
("#584", "#655"),
|
||||||
|
True,
|
||||||
|
"client_reconnect",
|
||||||
|
),
|
||||||
|
RecoveryRung(
|
||||||
|
RecoveryAction.CAPABILITY_REFRESH,
|
||||||
|
1,
|
||||||
|
"Re-resolve task capability and clear stale permission context.",
|
||||||
|
"Delegated: gitea_resolve_task_capability + gitea_whoami "
|
||||||
|
"(no process change).",
|
||||||
|
("#610", "#685", "#655"),
|
||||||
|
True,
|
||||||
|
None,
|
||||||
|
),
|
||||||
|
RecoveryRung(
|
||||||
|
RecoveryAction.SESSION_RECONNECT,
|
||||||
|
2,
|
||||||
|
"Rebind identity, workspace, and namespace for one session.",
|
||||||
|
"Delegated: gitea_get_runtime_context + explicit worktree_path "
|
||||||
|
"rebind (#618); docs/mcp-namespace-health.md",
|
||||||
|
("#543", "#618", "#655"),
|
||||||
|
True,
|
||||||
|
"session_reconnect",
|
||||||
|
),
|
||||||
|
RecoveryRung(
|
||||||
|
RecoveryAction.CONFIGURATION_RELOAD,
|
||||||
|
3,
|
||||||
|
"Gracefully reload configuration without replacing the daemon.",
|
||||||
|
"restart_coordinator class configuration_reload; console "
|
||||||
|
"system.reload_namespace (#642).",
|
||||||
|
("#642", "#663", "#655"),
|
||||||
|
False,
|
||||||
|
"configuration_reload",
|
||||||
|
),
|
||||||
|
RecoveryRung(
|
||||||
|
RecoveryAction.LEASE_RECOVERY,
|
||||||
|
4,
|
||||||
|
"Recover or rebind stale leases/locks without a process restart.",
|
||||||
|
"Delegated: issue lock recovery / lease lifecycle paths "
|
||||||
|
"(#702, #753, #790).",
|
||||||
|
("#702", "#753", "#790", "#655"),
|
||||||
|
False,
|
||||||
|
None,
|
||||||
|
),
|
||||||
|
RecoveryRung(
|
||||||
|
RecoveryAction.WORKER_RESTART,
|
||||||
|
5,
|
||||||
|
"Restart one worker after its own lease and mutation scope drains.",
|
||||||
|
"restart_coordinator class worker_restart (#663).",
|
||||||
|
("#663", "#655"),
|
||||||
|
False,
|
||||||
|
"worker_restart",
|
||||||
|
),
|
||||||
|
RecoveryRung(
|
||||||
|
RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||||
|
6,
|
||||||
|
"Restart one role runtime and re-probe that namespace only.",
|
||||||
|
"restart_coordinator class role_runtime_restart; console "
|
||||||
|
"system.restart_namespace (#642).",
|
||||||
|
("#642", "#663", "#655"),
|
||||||
|
False,
|
||||||
|
"role_runtime_restart",
|
||||||
|
),
|
||||||
|
RecoveryRung(
|
||||||
|
RecoveryAction.CONNECTOR_RESTART,
|
||||||
|
7,
|
||||||
|
"Restart one connector while unrelated runtimes stay available.",
|
||||||
|
"restart_coordinator class connector_restart (#663).",
|
||||||
|
("#663", "#655"),
|
||||||
|
False,
|
||||||
|
"connector_restart",
|
||||||
|
),
|
||||||
|
RecoveryRung(
|
||||||
|
RecoveryAction.ROLLING_MCP_RESTART,
|
||||||
|
8,
|
||||||
|
"Drain/restart/verify one instance at a time (HA path).",
|
||||||
|
"restart_coordinator class rolling_mcp_restart; design #668.",
|
||||||
|
("#668", "#663", "#655"),
|
||||||
|
False,
|
||||||
|
"rolling_mcp_restart",
|
||||||
|
),
|
||||||
|
RecoveryRung(
|
||||||
|
RecoveryAction.FULL_MCP_RESTART,
|
||||||
|
9,
|
||||||
|
"Full stable-control MCP process restart after verified full drain.",
|
||||||
|
"restart_coordinator class full_mcp_restart; requires attempt log "
|
||||||
|
"unless break-glass (#669).",
|
||||||
|
("#658", "#661", "#663", "#669", "#655"),
|
||||||
|
False,
|
||||||
|
"full_mcp_restart",
|
||||||
|
),
|
||||||
|
RecoveryRung(
|
||||||
|
RecoveryAction.HOST_RESTART,
|
||||||
|
10,
|
||||||
|
"Host/infrastructure restart — broadest last-resort action.",
|
||||||
|
"restart_coordinator class host_restart; operator-owned.",
|
||||||
|
("#663", "#669", "#655"),
|
||||||
|
False,
|
||||||
|
"host_restart",
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
_LADDER_BY_ACTION: dict[RecoveryAction, RecoveryRung] = {
|
||||||
|
rung.action: rung for rung in RECOVERY_LADDER
|
||||||
|
}
|
||||||
|
|
||||||
|
# Symptom tokens → preferred first rung (decision tree, #663 lineage).
|
||||||
|
SYMPTOM_TO_FIRST_ACTION: dict[str, RecoveryAction] = {
|
||||||
|
"transport_eof": RecoveryAction.CLIENT_RECONNECT,
|
||||||
|
"client_closing_eof": RecoveryAction.CLIENT_RECONNECT,
|
||||||
|
"transport_flap": RecoveryAction.CLIENT_RECONNECT,
|
||||||
|
"namespace_disconnected": RecoveryAction.CLIENT_RECONNECT,
|
||||||
|
"stale_capability": RecoveryAction.CAPABILITY_REFRESH,
|
||||||
|
"permission_stale": RecoveryAction.CAPABILITY_REFRESH,
|
||||||
|
"runtime_reconnect_required": RecoveryAction.CAPABILITY_REFRESH,
|
||||||
|
"stale_runtime": RecoveryAction.SESSION_RECONNECT,
|
||||||
|
"worktree_unbound": RecoveryAction.SESSION_RECONNECT,
|
||||||
|
"namespace_unhealthy": RecoveryAction.SESSION_RECONNECT,
|
||||||
|
"config_drift": RecoveryAction.CONFIGURATION_RELOAD,
|
||||||
|
"profile_misbound": RecoveryAction.CONFIGURATION_RELOAD,
|
||||||
|
"stale_lease": RecoveryAction.LEASE_RECOVERY,
|
||||||
|
"dead_pid_lock": RecoveryAction.LEASE_RECOVERY,
|
||||||
|
"orphan_worktree": RecoveryAction.LEASE_RECOVERY,
|
||||||
|
"single_worker_stuck": RecoveryAction.WORKER_RESTART,
|
||||||
|
"role_runtime_dead": RecoveryAction.ROLE_RUNTIME_RESTART,
|
||||||
|
"connector_dead": RecoveryAction.CONNECTOR_RESTART,
|
||||||
|
"ha_instance_unhealthy": RecoveryAction.ROLLING_MCP_RESTART,
|
||||||
|
"daemon_corrupt": RecoveryAction.FULL_MCP_RESTART,
|
||||||
|
"full_process_deadlock": RecoveryAction.FULL_MCP_RESTART,
|
||||||
|
"host_unresponsive": RecoveryAction.HOST_RESTART,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _utc_now() -> datetime:
|
||||||
|
return datetime.now(timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
def resolve_action(value: RecoveryAction | str) -> RecoveryAction:
|
||||||
|
"""Resolve a recovery action or fail closed for unknown values."""
|
||||||
|
if isinstance(value, RecoveryAction):
|
||||||
|
return value
|
||||||
|
text = str(value or "").strip()
|
||||||
|
# Accept #663 restart_class aliases.
|
||||||
|
if text in RESTART_CLASS_TO_ACTION:
|
||||||
|
return RESTART_CLASS_TO_ACTION[text]
|
||||||
|
try:
|
||||||
|
return RecoveryAction(text)
|
||||||
|
except ValueError as exc:
|
||||||
|
raise ValueError(
|
||||||
|
f"unknown recovery action {value!r}; deny (fail closed, #669)"
|
||||||
|
) from exc
|
||||||
|
|
||||||
|
|
||||||
|
def ladder_rank(action: RecoveryAction | str) -> int:
|
||||||
|
resolved = resolve_action(action)
|
||||||
|
return _LADDER_BY_ACTION[resolved].rank
|
||||||
|
|
||||||
|
|
||||||
|
def rung_for(action: RecoveryAction | str) -> RecoveryRung:
|
||||||
|
return _LADDER_BY_ACTION[resolve_action(action)]
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_attempt(raw: Mapping[str, Any]) -> dict[str, Any] | None:
|
||||||
|
"""Normalize one prior-recovery attempt record; return None if unusable."""
|
||||||
|
if not isinstance(raw, Mapping):
|
||||||
|
return None
|
||||||
|
action_raw = raw.get("action") or raw.get("recovery_action") or raw.get(
|
||||||
|
"restart_class"
|
||||||
|
)
|
||||||
|
if not action_raw:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
action = resolve_action(str(action_raw))
|
||||||
|
except ValueError:
|
||||||
|
return None
|
||||||
|
outcome = str(
|
||||||
|
raw.get("outcome") or raw.get("status") or raw.get("result") or ""
|
||||||
|
).strip().lower()
|
||||||
|
if not outcome:
|
||||||
|
return None
|
||||||
|
recorded_at = raw.get("recorded_at") or raw.get("at") or raw.get("timestamp")
|
||||||
|
reason = str(raw.get("reason") or raw.get("detail") or "").strip()
|
||||||
|
actor = str(raw.get("actor") or raw.get("session_id") or "").strip()
|
||||||
|
return {
|
||||||
|
"action": action.value,
|
||||||
|
"outcome": outcome,
|
||||||
|
"reason": reason,
|
||||||
|
"actor": actor,
|
||||||
|
"recorded_at": recorded_at,
|
||||||
|
"rank": ladder_rank(action),
|
||||||
|
"raw": dict(raw),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def normalize_attempt_log(
|
||||||
|
attempts: Sequence[Mapping[str, Any]] | None,
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
"""Return usable attempt records in ladder order."""
|
||||||
|
out: list[dict[str, Any]] = []
|
||||||
|
for raw in attempts or ():
|
||||||
|
norm = normalize_attempt(raw)
|
||||||
|
if norm is not None:
|
||||||
|
out.append(norm)
|
||||||
|
out.sort(key=lambda a: (a["rank"], str(a.get("recorded_at") or "")))
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def narrower_insufficient_attempts(
|
||||||
|
attempts: Sequence[Mapping[str, Any]] | None,
|
||||||
|
*,
|
||||||
|
requested: RecoveryAction | str,
|
||||||
|
) -> list[dict[str, Any]]:
|
||||||
|
"""Return prior attempts narrower than *requested* that were insufficient."""
|
||||||
|
target_rank = ladder_rank(requested)
|
||||||
|
usable = []
|
||||||
|
for attempt in normalize_attempt_log(attempts):
|
||||||
|
if attempt["rank"] >= target_rank:
|
||||||
|
continue
|
||||||
|
if attempt["outcome"] in INSUFFICIENT_OUTCOMES:
|
||||||
|
usable.append(attempt)
|
||||||
|
return usable
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class EscalationAssessment:
|
||||||
|
"""Whether a requested broad recovery may proceed given the attempt log."""
|
||||||
|
|
||||||
|
requested_action: str
|
||||||
|
allowed: bool
|
||||||
|
require_attempt_log: bool
|
||||||
|
break_glass: bool
|
||||||
|
reasons: list[str] = field(default_factory=list)
|
||||||
|
qualifying_attempts: list[dict[str, Any]] = field(default_factory=list)
|
||||||
|
recommended_next: list[dict[str, Any]] = field(default_factory=list)
|
||||||
|
playbook_version: str = PLAYBOOK_VERSION
|
||||||
|
|
||||||
|
def as_dict(self) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"playbook_version": self.playbook_version,
|
||||||
|
"requested_action": self.requested_action,
|
||||||
|
"allowed": self.allowed,
|
||||||
|
"require_attempt_log": self.require_attempt_log,
|
||||||
|
"break_glass": self.break_glass,
|
||||||
|
"reasons": list(self.reasons),
|
||||||
|
"qualifying_attempts": list(self.qualifying_attempts),
|
||||||
|
"recommended_next": list(self.recommended_next),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def assess_escalation(
|
||||||
|
requested: RecoveryAction | str,
|
||||||
|
*,
|
||||||
|
prior_recovery_attempts: Sequence[Mapping[str, Any]] | None = None,
|
||||||
|
break_glass: bool = False,
|
||||||
|
) -> EscalationAssessment:
|
||||||
|
"""Gate broad restarts on a prior narrow-attempt log (#669 AC3).
|
||||||
|
|
||||||
|
Narrow / mid-ladder actions do not require a prior attempt log.
|
||||||
|
``full_mcp_restart``, ``host_restart``, and ``rolling_mcp_restart``
|
||||||
|
require at least one *insufficient* narrower attempt unless
|
||||||
|
``break_glass`` is true.
|
||||||
|
"""
|
||||||
|
action = resolve_action(requested)
|
||||||
|
require_log = action in BROAD_RESTART_ACTIONS
|
||||||
|
reasons: list[str] = []
|
||||||
|
qualifying = narrower_insufficient_attempts(
|
||||||
|
prior_recovery_attempts, requested=action
|
||||||
|
)
|
||||||
|
|
||||||
|
if not require_log:
|
||||||
|
return EscalationAssessment(
|
||||||
|
requested_action=action.value,
|
||||||
|
allowed=True,
|
||||||
|
require_attempt_log=False,
|
||||||
|
break_glass=bool(break_glass),
|
||||||
|
reasons=["narrow recovery; attempt log not required"],
|
||||||
|
qualifying_attempts=qualifying,
|
||||||
|
recommended_next=[],
|
||||||
|
)
|
||||||
|
|
||||||
|
if break_glass:
|
||||||
|
return EscalationAssessment(
|
||||||
|
requested_action=action.value,
|
||||||
|
allowed=True,
|
||||||
|
require_attempt_log=True,
|
||||||
|
break_glass=True,
|
||||||
|
reasons=[
|
||||||
|
"break-glass authorized; broad restart permitted without "
|
||||||
|
"narrow-attempt log (#669)"
|
||||||
|
],
|
||||||
|
qualifying_attempts=qualifying,
|
||||||
|
recommended_next=[],
|
||||||
|
)
|
||||||
|
|
||||||
|
if qualifying:
|
||||||
|
return EscalationAssessment(
|
||||||
|
requested_action=action.value,
|
||||||
|
allowed=True,
|
||||||
|
require_attempt_log=True,
|
||||||
|
break_glass=False,
|
||||||
|
reasons=[
|
||||||
|
f"{len(qualifying)} narrower recovery attempt(s) recorded as "
|
||||||
|
"insufficient; escalation permitted"
|
||||||
|
],
|
||||||
|
qualifying_attempts=qualifying,
|
||||||
|
recommended_next=[],
|
||||||
|
)
|
||||||
|
|
||||||
|
# Deny: recommend the next untried narrow rung(s).
|
||||||
|
recommended = recommend_actions(
|
||||||
|
symptoms=(),
|
||||||
|
prior_recovery_attempts=prior_recovery_attempts,
|
||||||
|
max_actions=3,
|
||||||
|
)
|
||||||
|
reasons.append(
|
||||||
|
f"{action.value} requires a prior attempt log of insufficient "
|
||||||
|
"narrower recoveries (or break-glass); none found — deny (fail "
|
||||||
|
"closed, #669)"
|
||||||
|
)
|
||||||
|
return EscalationAssessment(
|
||||||
|
requested_action=action.value,
|
||||||
|
allowed=False,
|
||||||
|
require_attempt_log=True,
|
||||||
|
break_glass=False,
|
||||||
|
reasons=reasons,
|
||||||
|
qualifying_attempts=[],
|
||||||
|
recommended_next=recommended.get("recommended_actions") or [],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def recommend_actions(
|
||||||
|
*,
|
||||||
|
symptoms: Sequence[str] = (),
|
||||||
|
prior_recovery_attempts: Sequence[Mapping[str, Any]] | None = None,
|
||||||
|
max_actions: int = 5,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Return ordered recommended recovery actions for the given symptoms.
|
||||||
|
|
||||||
|
Soft mode (rollout): recommendations only — callers decide whether to
|
||||||
|
hard-gate. Hard mode for broad restarts is :func:`assess_escalation`.
|
||||||
|
"""
|
||||||
|
attempted_success = {
|
||||||
|
a["action"]
|
||||||
|
for a in normalize_attempt_log(prior_recovery_attempts)
|
||||||
|
if a["outcome"] == OUTCOME_SUCCESS
|
||||||
|
}
|
||||||
|
attempted_any = {
|
||||||
|
a["action"] for a in normalize_attempt_log(prior_recovery_attempts)
|
||||||
|
}
|
||||||
|
|
||||||
|
first_actions: list[RecoveryAction] = []
|
||||||
|
for symptom in symptoms:
|
||||||
|
key = str(symptom or "").strip().lower().replace(" ", "_").replace("-", "_")
|
||||||
|
mapped = SYMPTOM_TO_FIRST_ACTION.get(key)
|
||||||
|
if mapped is not None and mapped not in first_actions:
|
||||||
|
first_actions.append(mapped)
|
||||||
|
|
||||||
|
# Default entry: client reconnect then walk the ladder.
|
||||||
|
if not first_actions:
|
||||||
|
first_actions = [RecoveryAction.CLIENT_RECONNECT]
|
||||||
|
|
||||||
|
recommended: list[dict[str, Any]] = []
|
||||||
|
seen: set[str] = set()
|
||||||
|
min_rank = min(ladder_rank(a) for a in first_actions)
|
||||||
|
|
||||||
|
for rung in RECOVERY_LADDER:
|
||||||
|
if rung.rank < min_rank:
|
||||||
|
continue
|
||||||
|
if rung.action.value in attempted_success:
|
||||||
|
continue
|
||||||
|
if rung.action.value in seen:
|
||||||
|
continue
|
||||||
|
# Prefer rungs not yet attempted; still list previously-failed ones
|
||||||
|
# only if nothing else remains.
|
||||||
|
entry = rung.as_dict()
|
||||||
|
entry["already_attempted"] = rung.action.value in attempted_any
|
||||||
|
recommended.append(entry)
|
||||||
|
seen.add(rung.action.value)
|
||||||
|
if len(recommended) >= max(1, int(max_actions)):
|
||||||
|
break
|
||||||
|
|
||||||
|
return {
|
||||||
|
"playbook_version": PLAYBOOK_VERSION,
|
||||||
|
"symptoms": [str(s) for s in symptoms],
|
||||||
|
"recommended_actions": recommended,
|
||||||
|
"ladder": [r.as_dict() for r in RECOVERY_LADDER],
|
||||||
|
"read_only": True,
|
||||||
|
"hard_gate_note": (
|
||||||
|
"Broad restarts (rolling/full/host) still require "
|
||||||
|
"assess_escalation / coordinator attempt-log enforcement."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def build_attempt_record(
|
||||||
|
action: RecoveryAction | str,
|
||||||
|
*,
|
||||||
|
outcome: str,
|
||||||
|
reason: str = "",
|
||||||
|
actor: str = "",
|
||||||
|
recorded_at: str | None = None,
|
||||||
|
extra: Mapping[str, Any] | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Build a durable-shaped attempt log entry for inventory/audit (#665)."""
|
||||||
|
resolved = resolve_action(action)
|
||||||
|
record = {
|
||||||
|
"action": resolved.value,
|
||||||
|
"outcome": str(outcome or "").strip().lower(),
|
||||||
|
"reason": str(reason or "").strip(),
|
||||||
|
"actor": str(actor or "").strip(),
|
||||||
|
"recorded_at": recorded_at or _utc_now().isoformat(),
|
||||||
|
"rank": ladder_rank(resolved),
|
||||||
|
"playbook_version": PLAYBOOK_VERSION,
|
||||||
|
}
|
||||||
|
if extra:
|
||||||
|
record["extra"] = dict(extra)
|
||||||
|
return record
|
||||||
|
|
||||||
|
|
||||||
|
def recovery_metrics(
|
||||||
|
attempts: Sequence[Mapping[str, Any]] | None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Compute the fraction of recoveries that avoided full/host restart.
|
||||||
|
|
||||||
|
A recovery *episode* is approximated as one attempt with
|
||||||
|
``outcome=success``. Successes on non-broad rungs count as avoided full
|
||||||
|
restart; successes on full/host count as full-restart recoveries.
|
||||||
|
"""
|
||||||
|
norms = normalize_attempt_log(attempts)
|
||||||
|
successes = [a for a in norms if a["outcome"] == OUTCOME_SUCCESS]
|
||||||
|
broad_success = [
|
||||||
|
a
|
||||||
|
for a in successes
|
||||||
|
if resolve_action(a["action"])
|
||||||
|
in {RecoveryAction.FULL_MCP_RESTART, RecoveryAction.HOST_RESTART}
|
||||||
|
]
|
||||||
|
avoided = [a for a in successes if a not in broad_success]
|
||||||
|
total = len(successes)
|
||||||
|
fraction_avoided = (len(avoided) / total) if total else None
|
||||||
|
return {
|
||||||
|
"playbook_version": PLAYBOOK_VERSION,
|
||||||
|
"attempts_total": len(norms),
|
||||||
|
"successes_total": total,
|
||||||
|
"successes_avoided_full_restart": len(avoided),
|
||||||
|
"successes_full_or_host_restart": len(broad_success),
|
||||||
|
"fraction_avoided_full_restart": fraction_avoided,
|
||||||
|
"insufficient_attempts": sum(
|
||||||
|
1 for a in norms if a["outcome"] in INSUFFICIENT_OUTCOMES
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def ladder_document() -> dict[str, Any]:
|
||||||
|
"""Machine-readable ladder for docs/tools inventory."""
|
||||||
|
return {
|
||||||
|
"playbook_version": PLAYBOOK_VERSION,
|
||||||
|
"parent_issues": ["#655", "#652", "#653"],
|
||||||
|
"enforcement_issue": "#669",
|
||||||
|
"ladder": [r.as_dict() for r in RECOVERY_LADDER],
|
||||||
|
"broad_restart_actions": [a.value for a in sorted(BROAD_RESTART_ACTIONS, key=lambda x: x.value)],
|
||||||
|
"insufficient_outcomes": sorted(INSUFFICIENT_OUTCOMES),
|
||||||
|
"symptom_map": {k: v.value for k, v in sorted(SYMPTOM_TO_FIRST_ACTION.items())},
|
||||||
|
}
|
||||||
+46
-2
@@ -1,4 +1,4 @@
|
|||||||
"""MCP restart coordinator and impact analysis (#658).
|
"""MCP restart coordinator and impact analysis (#658 / #669).
|
||||||
|
|
||||||
Before any sanctioned MCP restart, a central coordinator must evaluate the
|
Before any sanctioned MCP restart, a central coordinator must evaluate the
|
||||||
live control-plane state — active sessions, leases/locks, in-flight issue/PR
|
live control-plane state — active sessions, leases/locks, in-flight issue/PR
|
||||||
@@ -16,6 +16,9 @@ Design rules (mirrors the read-only posture of ``workflow_dashboard`` /
|
|||||||
a mutative apply path is a later child gated by a drain proof (non-goal here).
|
a mutative apply path is a later child gated by a drain proof (non-goal here).
|
||||||
* **Fail closed.** If the inventory is not explicitly complete, the verdict is
|
* **Fail closed.** If the inventory is not explicitly complete, the verdict is
|
||||||
``unsafe`` / deny — an incomplete evaluation must never green-light a restart.
|
``unsafe`` / deny — an incomplete evaluation must never green-light a restart.
|
||||||
|
* **Narrow-first (#669).** Broad classes (rolling / full / host) require a
|
||||||
|
prior attempt log of insufficient narrower recoveries unless break-glass is
|
||||||
|
authorized. See :mod:`recovery_playbook`.
|
||||||
* **No secrets.** Session ids, pids, and profiles are operational metadata, not
|
* **No secrets.** Session ids, pids, and profiles are operational metadata, not
|
||||||
credentials; nothing secret flows through this module.
|
credentials; nothing secret flows through this module.
|
||||||
|
|
||||||
@@ -32,8 +35,9 @@ from enum import Enum
|
|||||||
from typing import Any, Mapping, Sequence
|
from typing import Any, Mapping, Sequence
|
||||||
|
|
||||||
import lease_lifecycle
|
import lease_lifecycle
|
||||||
|
import recovery_playbook
|
||||||
|
|
||||||
COORDINATOR_VERSION = "1.1.0-issue-663"
|
COORDINATOR_VERSION = "1.2.0-issue-669"
|
||||||
|
|
||||||
# Restart verdicts. Exactly the three the acceptance criteria name.
|
# Restart verdicts. Exactly the three the acceptance criteria name.
|
||||||
VERDICT_SAFE = "safe"
|
VERDICT_SAFE = "safe"
|
||||||
@@ -349,6 +353,10 @@ class RestartImpactReport:
|
|||||||
counts: dict[str, int]
|
counts: dict[str, int]
|
||||||
audit_record: dict[str, Any]
|
audit_record: dict[str, Any]
|
||||||
incomplete_reasons: list[str] = field(default_factory=list)
|
incomplete_reasons: list[str] = field(default_factory=list)
|
||||||
|
# #669 playbook escalation gate (attempt-log enforcement).
|
||||||
|
playbook_escalation: dict[str, Any] = field(default_factory=dict)
|
||||||
|
attempt_log_satisfied: bool = True
|
||||||
|
break_glass: bool = False
|
||||||
|
|
||||||
def as_dict(self) -> dict[str, Any]:
|
def as_dict(self) -> dict[str, Any]:
|
||||||
return {
|
return {
|
||||||
@@ -382,6 +390,9 @@ class RestartImpactReport:
|
|||||||
"prior_recovery_attempts": list(self.prior_recovery_attempts),
|
"prior_recovery_attempts": list(self.prior_recovery_attempts),
|
||||||
"counts": dict(self.counts),
|
"counts": dict(self.counts),
|
||||||
"audit_record": dict(self.audit_record),
|
"audit_record": dict(self.audit_record),
|
||||||
|
"playbook_escalation": dict(self.playbook_escalation),
|
||||||
|
"attempt_log_satisfied": self.attempt_log_satisfied,
|
||||||
|
"break_glass": self.break_glass,
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@@ -496,6 +507,7 @@ def evaluate_restart_impact(
|
|||||||
target_session_id: str | None = None,
|
target_session_id: str | None = None,
|
||||||
target_role: str | None = None,
|
target_role: str | None = None,
|
||||||
target_connector: str | None = None,
|
target_connector: str | None = None,
|
||||||
|
break_glass: bool = False,
|
||||||
) -> RestartImpactReport:
|
) -> RestartImpactReport:
|
||||||
"""Evaluate a proposed MCP restart and return an impact preview.
|
"""Evaluate a proposed MCP restart and return an impact preview.
|
||||||
|
|
||||||
@@ -584,6 +596,30 @@ def evaluate_restart_impact(
|
|||||||
dict(a) for a in (inventory.get("prior_recovery_attempts") or [])
|
dict(a) for a in (inventory.get("prior_recovery_attempts") or [])
|
||||||
]
|
]
|
||||||
|
|
||||||
|
# #669: broad restarts require a prior narrow-attempt log unless break-glass.
|
||||||
|
playbook_escalation: dict[str, Any] = {}
|
||||||
|
attempt_log_satisfied = True
|
||||||
|
if policy_enforced and resolved_class is not None:
|
||||||
|
try:
|
||||||
|
escalation = recovery_playbook.assess_escalation(
|
||||||
|
resolved_class.value,
|
||||||
|
prior_recovery_attempts=prior_recovery_attempts,
|
||||||
|
break_glass=bool(break_glass),
|
||||||
|
)
|
||||||
|
playbook_escalation = escalation.as_dict()
|
||||||
|
attempt_log_satisfied = bool(escalation.allowed)
|
||||||
|
if not attempt_log_satisfied:
|
||||||
|
authorization_reasons.extend(list(escalation.reasons))
|
||||||
|
except ValueError as exc:
|
||||||
|
# Unknown mapping should never happen for enum values; fail closed.
|
||||||
|
attempt_log_satisfied = False
|
||||||
|
playbook_escalation = {
|
||||||
|
"allowed": False,
|
||||||
|
"reasons": [str(exc)],
|
||||||
|
"playbook_version": recovery_playbook.PLAYBOOK_VERSION,
|
||||||
|
}
|
||||||
|
authorization_reasons.append(str(exc))
|
||||||
|
|
||||||
session_impacts = [
|
session_impacts = [
|
||||||
_classify_session(
|
_classify_session(
|
||||||
s,
|
s,
|
||||||
@@ -682,6 +718,7 @@ def evaluate_restart_impact(
|
|||||||
and role_authorized
|
and role_authorized
|
||||||
and approval_satisfied
|
and approval_satisfied
|
||||||
and target_complete
|
and target_complete
|
||||||
|
and attempt_log_satisfied
|
||||||
)
|
)
|
||||||
|
|
||||||
if policy_enforced and not authorization_ok:
|
if policy_enforced and not authorization_ok:
|
||||||
@@ -745,6 +782,7 @@ def evaluate_restart_impact(
|
|||||||
"affected_issues": len(affected_issues),
|
"affected_issues": len(affected_issues),
|
||||||
"affected_prs": len(affected_prs),
|
"affected_prs": len(affected_prs),
|
||||||
"prior_recovery_attempts": len(prior_recovery_attempts),
|
"prior_recovery_attempts": len(prior_recovery_attempts),
|
||||||
|
"attempt_log_satisfied": attempt_log_satisfied,
|
||||||
}
|
}
|
||||||
|
|
||||||
audit_record = {
|
audit_record = {
|
||||||
@@ -765,6 +803,9 @@ def evaluate_restart_impact(
|
|||||||
"allow_restart": allow_restart,
|
"allow_restart": allow_restart,
|
||||||
"blast_radius": blast_radius,
|
"blast_radius": blast_radius,
|
||||||
"counts": counts,
|
"counts": counts,
|
||||||
|
"attempt_log_satisfied": attempt_log_satisfied,
|
||||||
|
"break_glass": bool(break_glass),
|
||||||
|
"playbook_version": recovery_playbook.PLAYBOOK_VERSION,
|
||||||
}
|
}
|
||||||
|
|
||||||
return RestartImpactReport(
|
return RestartImpactReport(
|
||||||
@@ -804,4 +845,7 @@ def evaluate_restart_impact(
|
|||||||
counts=counts,
|
counts=counts,
|
||||||
audit_record=audit_record,
|
audit_record=audit_record,
|
||||||
incomplete_reasons=incomplete_reasons,
|
incomplete_reasons=incomplete_reasons,
|
||||||
|
playbook_escalation=playbook_escalation,
|
||||||
|
attempt_log_satisfied=attempt_log_satisfied,
|
||||||
|
break_glass=bool(break_glass),
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -74,6 +74,43 @@ def check_author_mutation_namespace(
|
|||||||
return True, []
|
return True, []
|
||||||
|
|
||||||
|
|
||||||
|
def check_author_role_kind(
|
||||||
|
mutation_task: str,
|
||||||
|
profile: dict,
|
||||||
|
) -> tuple[bool, list[str]]:
|
||||||
|
"""Author-exclusive wall for durable-lock mutations (#953 F1).
|
||||||
|
|
||||||
|
``check_author_mutation_namespace`` walls off reviewer-bound sessions, which
|
||||||
|
is the whole gate for tasks whose required permission is itself author-only
|
||||||
|
(``gitea.pr.create``, ``gitea.repo.commit``). It is *not* sufficient for a
|
||||||
|
task gated on ``gitea.issue.comment``, which every configured role holds: a
|
||||||
|
merger, controller, or reconciler session would clear both the namespace
|
||||||
|
check and the permission gate and still reach the durable write.
|
||||||
|
|
||||||
|
Opt-in per call site and additive. It refuses any active profile whose
|
||||||
|
derived role kind is not exactly ``author`` for a task the router declares
|
||||||
|
author-required, and grants nothing to anyone — a ``mixed`` profile is
|
||||||
|
refused rather than admitted.
|
||||||
|
"""
|
||||||
|
required_role = role_session_router.required_role_for_task(mutation_task)
|
||||||
|
if required_role != "author":
|
||||||
|
return True, []
|
||||||
|
|
||||||
|
allowed = profile.get("allowed_operations") or []
|
||||||
|
forbidden = profile.get("forbidden_operations") or []
|
||||||
|
active_role = derive_role_kind(allowed, forbidden)
|
||||||
|
if active_role == "author":
|
||||||
|
return True, []
|
||||||
|
|
||||||
|
profile_name = profile.get("profile_name") or ""
|
||||||
|
namespace = infer_mcp_namespace(profile_name)
|
||||||
|
return False, [
|
||||||
|
f"author mutation '{mutation_task}' blocked: active session role kind is "
|
||||||
|
f"'{active_role}', not 'author' ({profile_name} / {namespace}); this "
|
||||||
|
"operation writes a durable author issue lock and is author-exclusive",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
def mutation_audit_context(mutation_task: str, profile: dict, *,
|
def mutation_audit_context(mutation_task: str, profile: dict, *,
|
||||||
remote=None, repository=None) -> dict:
|
remote=None, repository=None) -> dict:
|
||||||
"""Structured mutation metadata for audit records (#209)."""
|
"""Structured mutation metadata for audit records (#209)."""
|
||||||
|
|||||||
@@ -75,6 +75,10 @@ AUTHOR_TASKS = frozenset({
|
|||||||
"push_branch",
|
"push_branch",
|
||||||
"bootstrap_author_issue_worktree",
|
"bootstrap_author_issue_worktree",
|
||||||
"gitea_bootstrap_author_issue_worktree",
|
"gitea_bootstrap_author_issue_worktree",
|
||||||
|
# #953: recovery of an incomplete bootstrap lock is an author-only durable
|
||||||
|
# state mutation and belongs to the same class as bootstrap itself.
|
||||||
|
"recover_incomplete_bootstrap_lock",
|
||||||
|
"gitea_recover_incomplete_bootstrap_lock",
|
||||||
"create_pr",
|
"create_pr",
|
||||||
"comment_pr",
|
"comment_pr",
|
||||||
"address_pr_change_requests",
|
"address_pr_change_requests",
|
||||||
@@ -112,6 +116,12 @@ TASK_REQUIRED_ROLE = {
|
|||||||
"claim_issue": "author",
|
"claim_issue": "author",
|
||||||
"create_branch": "author",
|
"create_branch": "author",
|
||||||
"push_branch": "author",
|
"push_branch": "author",
|
||||||
|
# #953: without this entry ``required_role_for_task`` returns None and
|
||||||
|
# ``role_namespace_gate.check_author_mutation_namespace`` short-circuits to
|
||||||
|
# "allowed" — the namespace wall on the recovery tool would be inert. The
|
||||||
|
# capability map already records the same role; both tables must agree.
|
||||||
|
"recover_incomplete_bootstrap_lock": "author",
|
||||||
|
"gitea_recover_incomplete_bootstrap_lock": "author",
|
||||||
"create_pr": "author",
|
"create_pr": "author",
|
||||||
"comment_pr": "author",
|
"comment_pr": "author",
|
||||||
"address_pr_change_requests": "author",
|
"address_pr_change_requests": "author",
|
||||||
|
|||||||
@@ -0,0 +1,703 @@
|
|||||||
|
"""Safe lifecycle for workflow session rows with dead or reused owners (#969).
|
||||||
|
|
||||||
|
Post-restart reconciliation previously left hundreds of ``active`` session rows
|
||||||
|
with dead owner PIDs permanently unresolved. A PID existence check alone is not
|
||||||
|
enough: operating systems reuse PIDs, so an unrelated live process can appear to
|
||||||
|
own a historical session.
|
||||||
|
|
||||||
|
This module is the pure classification + apply core for session retirement:
|
||||||
|
|
||||||
|
* distinguish live, disconnected, stale, protected, and terminal records
|
||||||
|
* refuse retirement when a live lease or live verified owner remains
|
||||||
|
* protect live client-managed sessions
|
||||||
|
* detect PID reuse via process start time vs session start / heartbeat
|
||||||
|
* terminalize confirmed-stale rows idempotently with durable audit events
|
||||||
|
* stay safe under concurrent reconciles (CAS on status)
|
||||||
|
|
||||||
|
Design mirrors ``lease_lifecycle`` / ``post_restart_reconcile``:
|
||||||
|
|
||||||
|
* Pure classification accepts injectable checkers so unit tests never touch
|
||||||
|
real processes.
|
||||||
|
* Apply mutations go only through :meth:`ControlPlaneDB.retire_session`.
|
||||||
|
* Historical rows are never deleted; status moves to a terminal value and an
|
||||||
|
events-row records the action + reason.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import datetime, timedelta, timezone
|
||||||
|
from typing import Any, Callable, Mapping, Sequence
|
||||||
|
|
||||||
|
import control_plane_db as cpd
|
||||||
|
import lease_lifecycle
|
||||||
|
|
||||||
|
# Session status vocabulary.
|
||||||
|
SESSION_STATUS_ACTIVE = "active"
|
||||||
|
SESSION_STATUS_RETIRED = "retired"
|
||||||
|
SESSION_STATUS_ENDED = "ended"
|
||||||
|
SESSION_STATUS_TERMINAL = "terminal"
|
||||||
|
|
||||||
|
TERMINAL_SESSION_STATUSES = frozenset(
|
||||||
|
{
|
||||||
|
SESSION_STATUS_RETIRED,
|
||||||
|
SESSION_STATUS_ENDED,
|
||||||
|
SESSION_STATUS_TERMINAL,
|
||||||
|
"dead",
|
||||||
|
"stale",
|
||||||
|
"orphaned",
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
# Classification outcomes for one session row.
|
||||||
|
CLASS_LIVE = "live"
|
||||||
|
CLASS_DISCONNECTED = "disconnected"
|
||||||
|
CLASS_STALE = "stale"
|
||||||
|
CLASS_PROTECTED = "protected"
|
||||||
|
CLASS_TERMINAL = "terminal"
|
||||||
|
|
||||||
|
# Stable reason codes (audit + tests).
|
||||||
|
REASON_ALREADY_TERMINAL = "already_terminal"
|
||||||
|
REASON_LIVE_OWNER = "live_owner"
|
||||||
|
REASON_LIVE_LEASE = "live_lease"
|
||||||
|
REASON_CLIENT_MANAGED_LIVE = "client_managed_live"
|
||||||
|
REASON_DEAD_OWNER = "dead_owner"
|
||||||
|
REASON_PID_REUSE = "pid_reuse"
|
||||||
|
REASON_HEARTBEAT_STALE_DEAD = "heartbeat_stale_dead_owner"
|
||||||
|
REASON_MISSING_PID = "missing_pid_no_lease"
|
||||||
|
|
||||||
|
# Event type written to control-plane events table.
|
||||||
|
EVENT_SESSION_RETIRED = "session_retired"
|
||||||
|
|
||||||
|
# Default heartbeat window before a still-alive PID is treated as disconnected
|
||||||
|
# rather than live (does not alone authorize retirement).
|
||||||
|
DEFAULT_HEARTBEAT_STALE_SECONDS = 900
|
||||||
|
|
||||||
|
# PID reuse: process start must be strictly later than session started_at by
|
||||||
|
# more than this skew (ps lstart is second-resolution; clocks can lag).
|
||||||
|
PID_REUSE_SKEW = timedelta(seconds=2)
|
||||||
|
|
||||||
|
# Live lease freshness values that block retirement.
|
||||||
|
_LIVE_LEASE_FRESHNESS = frozenset({"active", "live"})
|
||||||
|
|
||||||
|
|
||||||
|
class SessionLifecycleError(RuntimeError):
|
||||||
|
"""Fail-closed session lifecycle policy error."""
|
||||||
|
|
||||||
|
|
||||||
|
def _utc_now() -> datetime:
|
||||||
|
return datetime.now(timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_ts(value: str | datetime | None) -> datetime | None:
|
||||||
|
if value is None:
|
||||||
|
return None
|
||||||
|
if isinstance(value, datetime):
|
||||||
|
if value.tzinfo is None:
|
||||||
|
return value.replace(tzinfo=timezone.utc)
|
||||||
|
return value.astimezone(timezone.utc)
|
||||||
|
return cpd._parse_ts(str(value))
|
||||||
|
|
||||||
|
|
||||||
|
def _ts(dt: datetime | None = None) -> str:
|
||||||
|
return cpd._ts(dt)
|
||||||
|
|
||||||
|
|
||||||
|
def process_start_time(pid: int | None) -> datetime | None:
|
||||||
|
"""Return the OS start time of *pid*, or None when it cannot be resolved.
|
||||||
|
|
||||||
|
Uses ``ps -o lstart=`` (POSIX). Failures return None rather than inventing
|
||||||
|
evidence — missing start time never authorizes retirement of a live PID.
|
||||||
|
"""
|
||||||
|
if pid is None:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
pid_i = int(pid)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
if pid_i <= 0:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
proc = subprocess.run(
|
||||||
|
["ps", "-o", "lstart=", "-p", str(pid_i)],
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
check=False,
|
||||||
|
timeout=2,
|
||||||
|
)
|
||||||
|
except (OSError, subprocess.SubprocessError):
|
||||||
|
return None
|
||||||
|
if proc.returncode != 0:
|
||||||
|
return None
|
||||||
|
text = (proc.stdout or "").strip()
|
||||||
|
if not text:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
# Example: "Wed Jul 29 09:14:36 2026"
|
||||||
|
naive = datetime.strptime(text, "%a %b %d %H:%M:%S %Y")
|
||||||
|
return naive.replace(tzinfo=timezone.utc)
|
||||||
|
except ValueError:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _lease_freshness_label(lease: Mapping[str, Any]) -> str:
|
||||||
|
fr = lease.get("freshness")
|
||||||
|
if isinstance(fr, Mapping):
|
||||||
|
return str(fr.get("freshness") or "").strip().lower()
|
||||||
|
if fr:
|
||||||
|
return str(fr).strip().lower()
|
||||||
|
# Fall back to classifying a raw lease row.
|
||||||
|
try:
|
||||||
|
return str(
|
||||||
|
lease_lifecycle.classify_lease_freshness(lease).get("freshness") or ""
|
||||||
|
).strip().lower()
|
||||||
|
except Exception: # noqa: BLE001 — pure classifier must not raise on bad rows
|
||||||
|
status = str(lease.get("status") or "").strip().lower()
|
||||||
|
return status or "unknown"
|
||||||
|
|
||||||
|
|
||||||
|
def live_lease_session_ids(
|
||||||
|
leases: Sequence[Mapping[str, Any]] | None,
|
||||||
|
*,
|
||||||
|
now: datetime | None = None,
|
||||||
|
pid_checker: Callable[[int | None], bool] = lease_lifecycle.is_process_alive,
|
||||||
|
) -> set[str]:
|
||||||
|
"""Session ids that still hold a live (or ambiguous-active) workflow lease."""
|
||||||
|
live: set[str] = set()
|
||||||
|
moment = now or _utc_now()
|
||||||
|
for lease in leases or ():
|
||||||
|
if not isinstance(lease, Mapping):
|
||||||
|
continue
|
||||||
|
status = str(lease.get("status") or "").strip().lower()
|
||||||
|
if status and status not in {
|
||||||
|
lease_lifecycle.LEASE_STATUS_ACTIVE,
|
||||||
|
"",
|
||||||
|
}:
|
||||||
|
# Explicit terminal lease statuses never protect a session.
|
||||||
|
if status in {
|
||||||
|
lease_lifecycle.LEASE_STATUS_RELEASED,
|
||||||
|
lease_lifecycle.LEASE_STATUS_EXPIRED,
|
||||||
|
lease_lifecycle.LEASE_STATUS_ABANDONED,
|
||||||
|
}:
|
||||||
|
continue
|
||||||
|
freshness = _lease_freshness_label(lease)
|
||||||
|
if freshness in _LIVE_LEASE_FRESHNESS or freshness in {"", "unknown"}:
|
||||||
|
# Ambiguous active rows: re-check with authoritative classifier.
|
||||||
|
try:
|
||||||
|
fr = lease_lifecycle.classify_lease_freshness(
|
||||||
|
lease, now=moment, pid_checker=pid_checker
|
||||||
|
)
|
||||||
|
freshness = str(fr.get("freshness") or "").strip().lower()
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
freshness = "unknown"
|
||||||
|
if freshness in _LIVE_LEASE_FRESHNESS:
|
||||||
|
sid = str(lease.get("session_id") or "").strip()
|
||||||
|
if sid:
|
||||||
|
live.add(sid)
|
||||||
|
elif freshness == "unknown" and status in {
|
||||||
|
lease_lifecycle.LEASE_STATUS_ACTIVE,
|
||||||
|
"",
|
||||||
|
}:
|
||||||
|
# Fail closed: active lease with unknown freshness blocks retirement.
|
||||||
|
sid = str(lease.get("session_id") or "").strip()
|
||||||
|
if sid:
|
||||||
|
live.add(sid)
|
||||||
|
return live
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class SessionClassification:
|
||||||
|
"""Classification of one workflow session row."""
|
||||||
|
|
||||||
|
session_id: str
|
||||||
|
classification: str
|
||||||
|
reason: str
|
||||||
|
retireable: bool
|
||||||
|
status: str | None
|
||||||
|
pid: int | None
|
||||||
|
pid_alive: bool | None
|
||||||
|
pid_reused: bool
|
||||||
|
heartbeat_stale: bool
|
||||||
|
has_live_lease: bool
|
||||||
|
client_managed: bool
|
||||||
|
details: dict[str, Any] = field(default_factory=dict)
|
||||||
|
|
||||||
|
def as_dict(self) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"session_id": self.session_id,
|
||||||
|
"classification": self.classification,
|
||||||
|
"reason": self.reason,
|
||||||
|
"retireable": self.retireable,
|
||||||
|
"status": self.status,
|
||||||
|
"pid": self.pid,
|
||||||
|
"pid_alive": self.pid_alive,
|
||||||
|
"pid_reused": self.pid_reused,
|
||||||
|
"heartbeat_stale": self.heartbeat_stale,
|
||||||
|
"has_live_lease": self.has_live_lease,
|
||||||
|
"client_managed": self.client_managed,
|
||||||
|
"details": dict(self.details),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def classify_session(
|
||||||
|
row: Mapping[str, Any],
|
||||||
|
*,
|
||||||
|
now: datetime | None = None,
|
||||||
|
pid_checker: Callable[[int | None], bool] = lease_lifecycle.is_process_alive,
|
||||||
|
process_start_probe: Callable[[int | None], datetime | None] = process_start_time,
|
||||||
|
live_lease_sessions: set[str] | frozenset[str] | None = None,
|
||||||
|
client_managed_sessions: set[str] | frozenset[str] | None = None,
|
||||||
|
heartbeat_stale_seconds: int = DEFAULT_HEARTBEAT_STALE_SECONDS,
|
||||||
|
) -> SessionClassification:
|
||||||
|
"""Classify one session row for retirement decisions (#969).
|
||||||
|
|
||||||
|
Rules (first match wins where noted):
|
||||||
|
|
||||||
|
1. Non-active / already-terminal status → ``terminal`` (not retireable).
|
||||||
|
2. Session holds a live lease → ``protected`` (never retire).
|
||||||
|
3. PID missing and no live lease → ``stale`` (retireable: missing owner).
|
||||||
|
4. PID alive + process start after session start → ``stale`` (PID reuse).
|
||||||
|
5. PID alive + client-managed → ``live`` protected (never retire).
|
||||||
|
6. PID alive + fresh heartbeat → ``live``.
|
||||||
|
7. PID alive + stale heartbeat → ``disconnected`` (not retireable alone).
|
||||||
|
8. PID dead → ``stale`` (retireable).
|
||||||
|
"""
|
||||||
|
moment = now or _utc_now()
|
||||||
|
session_id = str(row.get("session_id") or "").strip()
|
||||||
|
status = str(row.get("status") or "").strip().lower() or None
|
||||||
|
raw_pid = row.get("pid")
|
||||||
|
try:
|
||||||
|
pid = int(raw_pid) if raw_pid is not None else None
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
pid = None
|
||||||
|
|
||||||
|
client_managed = bool(
|
||||||
|
row.get("client_managed")
|
||||||
|
or row.get("is_client_managed")
|
||||||
|
or (
|
||||||
|
client_managed_sessions is not None
|
||||||
|
and session_id in client_managed_sessions
|
||||||
|
)
|
||||||
|
)
|
||||||
|
has_live_lease = bool(
|
||||||
|
live_lease_sessions is not None and session_id in live_lease_sessions
|
||||||
|
)
|
||||||
|
|
||||||
|
hb = _parse_ts(row.get("last_heartbeat_at"))
|
||||||
|
heartbeat_stale = bool(
|
||||||
|
hb is not None
|
||||||
|
and (moment - hb).total_seconds() > max(0, int(heartbeat_stale_seconds))
|
||||||
|
)
|
||||||
|
if hb is None and status == SESSION_STATUS_ACTIVE:
|
||||||
|
# No heartbeat evidence: treat as stale for liveness bookkeeping only.
|
||||||
|
heartbeat_stale = True
|
||||||
|
|
||||||
|
started = _parse_ts(row.get("started_at"))
|
||||||
|
recorded_proc_start = _parse_ts(row.get("owner_process_started_at"))
|
||||||
|
|
||||||
|
if status in TERMINAL_SESSION_STATUSES:
|
||||||
|
return SessionClassification(
|
||||||
|
session_id=session_id,
|
||||||
|
classification=CLASS_TERMINAL,
|
||||||
|
reason=REASON_ALREADY_TERMINAL,
|
||||||
|
retireable=False,
|
||||||
|
status=status,
|
||||||
|
pid=pid,
|
||||||
|
pid_alive=None,
|
||||||
|
pid_reused=False,
|
||||||
|
heartbeat_stale=heartbeat_stale,
|
||||||
|
has_live_lease=has_live_lease,
|
||||||
|
client_managed=client_managed,
|
||||||
|
)
|
||||||
|
|
||||||
|
if has_live_lease:
|
||||||
|
return SessionClassification(
|
||||||
|
session_id=session_id,
|
||||||
|
classification=CLASS_PROTECTED,
|
||||||
|
reason=REASON_LIVE_LEASE,
|
||||||
|
retireable=False,
|
||||||
|
status=status,
|
||||||
|
pid=pid,
|
||||||
|
pid_alive=pid_checker(pid) if pid is not None else None,
|
||||||
|
pid_reused=False,
|
||||||
|
heartbeat_stale=heartbeat_stale,
|
||||||
|
has_live_lease=True,
|
||||||
|
client_managed=client_managed,
|
||||||
|
details={"blocker": "live_workflow_lease"},
|
||||||
|
)
|
||||||
|
|
||||||
|
if pid is None:
|
||||||
|
return SessionClassification(
|
||||||
|
session_id=session_id,
|
||||||
|
classification=CLASS_STALE,
|
||||||
|
reason=REASON_MISSING_PID,
|
||||||
|
retireable=True,
|
||||||
|
status=status,
|
||||||
|
pid=None,
|
||||||
|
pid_alive=False,
|
||||||
|
pid_reused=False,
|
||||||
|
heartbeat_stale=heartbeat_stale,
|
||||||
|
has_live_lease=False,
|
||||||
|
client_managed=client_managed,
|
||||||
|
)
|
||||||
|
|
||||||
|
pid_alive = bool(pid_checker(pid))
|
||||||
|
pid_reused = False
|
||||||
|
proc_start: datetime | None = None
|
||||||
|
|
||||||
|
if pid_alive:
|
||||||
|
proc_start = recorded_proc_start or process_start_probe(pid)
|
||||||
|
# PID reuse: live process started after the session row itself was
|
||||||
|
# created. Anchor on started_at only — last_heartbeat alone is not a
|
||||||
|
# safe bound (synthetic inventories and long-lived processes would
|
||||||
|
# false-positive against a live ``ps`` probe).
|
||||||
|
if proc_start is not None and started is not None:
|
||||||
|
if proc_start > (started + PID_REUSE_SKEW):
|
||||||
|
pid_reused = True
|
||||||
|
|
||||||
|
if pid_reused:
|
||||||
|
return SessionClassification(
|
||||||
|
session_id=session_id,
|
||||||
|
classification=CLASS_STALE,
|
||||||
|
reason=REASON_PID_REUSE,
|
||||||
|
retireable=True,
|
||||||
|
status=status,
|
||||||
|
pid=pid,
|
||||||
|
pid_alive=True,
|
||||||
|
pid_reused=True,
|
||||||
|
heartbeat_stale=heartbeat_stale,
|
||||||
|
has_live_lease=False,
|
||||||
|
client_managed=client_managed,
|
||||||
|
details={
|
||||||
|
"process_started_at": _ts(proc_start) if proc_start else None,
|
||||||
|
"session_started_at": row.get("started_at"),
|
||||||
|
"last_heartbeat_at": row.get("last_heartbeat_at"),
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
if pid_alive and client_managed:
|
||||||
|
return SessionClassification(
|
||||||
|
session_id=session_id,
|
||||||
|
classification=CLASS_LIVE,
|
||||||
|
reason=REASON_CLIENT_MANAGED_LIVE,
|
||||||
|
retireable=False,
|
||||||
|
status=status,
|
||||||
|
pid=pid,
|
||||||
|
pid_alive=True,
|
||||||
|
pid_reused=False,
|
||||||
|
heartbeat_stale=heartbeat_stale,
|
||||||
|
has_live_lease=False,
|
||||||
|
client_managed=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
if pid_alive and not heartbeat_stale:
|
||||||
|
return SessionClassification(
|
||||||
|
session_id=session_id,
|
||||||
|
classification=CLASS_LIVE,
|
||||||
|
reason=REASON_LIVE_OWNER,
|
||||||
|
retireable=False,
|
||||||
|
status=status,
|
||||||
|
pid=pid,
|
||||||
|
pid_alive=True,
|
||||||
|
pid_reused=False,
|
||||||
|
heartbeat_stale=False,
|
||||||
|
has_live_lease=False,
|
||||||
|
client_managed=client_managed,
|
||||||
|
)
|
||||||
|
|
||||||
|
if pid_alive and heartbeat_stale:
|
||||||
|
# Process still exists but has not heartbeated — disconnected, not
|
||||||
|
# confirmed stale. Do not retire; operator/reconnect owns next step.
|
||||||
|
return SessionClassification(
|
||||||
|
session_id=session_id,
|
||||||
|
classification=CLASS_DISCONNECTED,
|
||||||
|
reason=REASON_LIVE_OWNER,
|
||||||
|
retireable=False,
|
||||||
|
status=status,
|
||||||
|
pid=pid,
|
||||||
|
pid_alive=True,
|
||||||
|
pid_reused=False,
|
||||||
|
heartbeat_stale=True,
|
||||||
|
has_live_lease=False,
|
||||||
|
client_managed=client_managed,
|
||||||
|
details={"note": "alive_pid_stale_heartbeat_not_retired"},
|
||||||
|
)
|
||||||
|
|
||||||
|
# PID dead (or checker said not alive).
|
||||||
|
reason = REASON_DEAD_OWNER
|
||||||
|
if heartbeat_stale:
|
||||||
|
reason = REASON_HEARTBEAT_STALE_DEAD
|
||||||
|
return SessionClassification(
|
||||||
|
session_id=session_id,
|
||||||
|
classification=CLASS_STALE,
|
||||||
|
reason=reason,
|
||||||
|
retireable=True,
|
||||||
|
status=status,
|
||||||
|
pid=pid,
|
||||||
|
pid_alive=False,
|
||||||
|
pid_reused=False,
|
||||||
|
heartbeat_stale=heartbeat_stale,
|
||||||
|
has_live_lease=False,
|
||||||
|
client_managed=client_managed,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class SessionFleetReport:
|
||||||
|
"""Fleet-wide classification summary for reconcile + apply."""
|
||||||
|
|
||||||
|
classifications: tuple[SessionClassification, ...]
|
||||||
|
live_count: int
|
||||||
|
disconnected_count: int
|
||||||
|
stale_count: int
|
||||||
|
protected_count: int
|
||||||
|
terminal_count: int
|
||||||
|
retireable: tuple[SessionClassification, ...]
|
||||||
|
counts_by_reason: dict[str, int]
|
||||||
|
|
||||||
|
def as_dict(self) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"live_count": self.live_count,
|
||||||
|
"disconnected_count": self.disconnected_count,
|
||||||
|
"stale_count": self.stale_count,
|
||||||
|
"protected_count": self.protected_count,
|
||||||
|
"terminal_count": self.terminal_count,
|
||||||
|
"retireable_count": len(self.retireable),
|
||||||
|
"retireable_session_ids": [c.session_id for c in self.retireable],
|
||||||
|
"counts_by_reason": dict(self.counts_by_reason),
|
||||||
|
"classifications": [c.as_dict() for c in self.classifications],
|
||||||
|
"sessions_dimension_resolved": len(self.retireable) == 0,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def classify_sessions(
|
||||||
|
sessions: Sequence[Mapping[str, Any]] | None,
|
||||||
|
*,
|
||||||
|
leases: Sequence[Mapping[str, Any]] | None = None,
|
||||||
|
now: datetime | None = None,
|
||||||
|
pid_checker: Callable[[int | None], bool] = lease_lifecycle.is_process_alive,
|
||||||
|
process_start_probe: Callable[[int | None], datetime | None] = process_start_time,
|
||||||
|
client_managed_sessions: set[str] | frozenset[str] | None = None,
|
||||||
|
heartbeat_stale_seconds: int = DEFAULT_HEARTBEAT_STALE_SECONDS,
|
||||||
|
) -> SessionFleetReport:
|
||||||
|
"""Classify a fleet of session rows against live leases (#969)."""
|
||||||
|
moment = now or _utc_now()
|
||||||
|
live_leases = live_lease_session_ids(
|
||||||
|
leases, now=moment, pid_checker=pid_checker
|
||||||
|
)
|
||||||
|
results: list[SessionClassification] = []
|
||||||
|
for row in sessions or ():
|
||||||
|
if not isinstance(row, Mapping):
|
||||||
|
continue
|
||||||
|
results.append(
|
||||||
|
classify_session(
|
||||||
|
row,
|
||||||
|
now=moment,
|
||||||
|
pid_checker=pid_checker,
|
||||||
|
process_start_probe=process_start_probe,
|
||||||
|
live_lease_sessions=live_leases,
|
||||||
|
client_managed_sessions=client_managed_sessions,
|
||||||
|
heartbeat_stale_seconds=heartbeat_stale_seconds,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
live_count = sum(1 for c in results if c.classification == CLASS_LIVE)
|
||||||
|
disconnected_count = sum(
|
||||||
|
1 for c in results if c.classification == CLASS_DISCONNECTED
|
||||||
|
)
|
||||||
|
stale_count = sum(1 for c in results if c.classification == CLASS_STALE)
|
||||||
|
protected_count = sum(1 for c in results if c.classification == CLASS_PROTECTED)
|
||||||
|
terminal_count = sum(1 for c in results if c.classification == CLASS_TERMINAL)
|
||||||
|
retireable = tuple(c for c in results if c.retireable)
|
||||||
|
by_reason: dict[str, int] = {}
|
||||||
|
for c in results:
|
||||||
|
by_reason[c.reason] = by_reason.get(c.reason, 0) + 1
|
||||||
|
|
||||||
|
return SessionFleetReport(
|
||||||
|
classifications=tuple(results),
|
||||||
|
live_count=live_count,
|
||||||
|
disconnected_count=disconnected_count,
|
||||||
|
stale_count=stale_count,
|
||||||
|
protected_count=protected_count,
|
||||||
|
terminal_count=terminal_count,
|
||||||
|
retireable=retireable,
|
||||||
|
counts_by_reason=by_reason,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class RetirementResult:
|
||||||
|
"""Outcome of one session retirement attempt."""
|
||||||
|
|
||||||
|
session_id: str
|
||||||
|
outcome: str # retired | already_terminal | skipped | blocked | missing
|
||||||
|
reason: str
|
||||||
|
prior_status: str | None = None
|
||||||
|
new_status: str | None = None
|
||||||
|
details: dict[str, Any] = field(default_factory=dict)
|
||||||
|
|
||||||
|
def as_dict(self) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"session_id": self.session_id,
|
||||||
|
"outcome": self.outcome,
|
||||||
|
"reason": self.reason,
|
||||||
|
"prior_status": self.prior_status,
|
||||||
|
"new_status": self.new_status,
|
||||||
|
"details": dict(self.details),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def apply_session_retirements(
|
||||||
|
db: cpd.ControlPlaneDB,
|
||||||
|
report: SessionFleetReport,
|
||||||
|
*,
|
||||||
|
dry_run: bool = False,
|
||||||
|
actor_session_id: str | None = None,
|
||||||
|
now: datetime | None = None,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Terminalize every retireable session in *report* (idempotent).
|
||||||
|
|
||||||
|
Concurrent reconciles are safe: each retirement is a CAS on
|
||||||
|
``status='active'`` (or other non-terminal). A second pass that sees the
|
||||||
|
same session already retired records ``already_terminal`` rather than
|
||||||
|
duplicating audit noise beyond a single no-op outcome.
|
||||||
|
"""
|
||||||
|
moment = now or _utc_now()
|
||||||
|
results: list[RetirementResult] = []
|
||||||
|
retired = 0
|
||||||
|
already = 0
|
||||||
|
blocked = 0
|
||||||
|
missing = 0
|
||||||
|
|
||||||
|
for classification in report.retireable:
|
||||||
|
if not classification.retireable:
|
||||||
|
continue
|
||||||
|
if dry_run:
|
||||||
|
results.append(
|
||||||
|
RetirementResult(
|
||||||
|
session_id=classification.session_id,
|
||||||
|
outcome="skipped",
|
||||||
|
reason=classification.reason,
|
||||||
|
prior_status=classification.status,
|
||||||
|
new_status=SESSION_STATUS_RETIRED,
|
||||||
|
details={"dry_run": True, **classification.details},
|
||||||
|
)
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
|
||||||
|
applied = db.retire_session(
|
||||||
|
session_id=classification.session_id,
|
||||||
|
reason=classification.reason,
|
||||||
|
actor_session_id=actor_session_id,
|
||||||
|
details={
|
||||||
|
"classification": classification.classification,
|
||||||
|
"pid": classification.pid,
|
||||||
|
"pid_alive": classification.pid_alive,
|
||||||
|
"pid_reused": classification.pid_reused,
|
||||||
|
"client_managed": classification.client_managed,
|
||||||
|
**classification.details,
|
||||||
|
},
|
||||||
|
now=moment,
|
||||||
|
)
|
||||||
|
outcome = str(applied.get("outcome") or "missing")
|
||||||
|
results.append(
|
||||||
|
RetirementResult(
|
||||||
|
session_id=classification.session_id,
|
||||||
|
outcome=outcome,
|
||||||
|
reason=str(applied.get("reason") or classification.reason),
|
||||||
|
prior_status=applied.get("prior_status"),
|
||||||
|
new_status=applied.get("new_status"),
|
||||||
|
details=dict(applied.get("details") or {}),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
if outcome == "retired":
|
||||||
|
retired += 1
|
||||||
|
elif outcome == "already_terminal":
|
||||||
|
already += 1
|
||||||
|
elif outcome == "blocked":
|
||||||
|
blocked += 1
|
||||||
|
else:
|
||||||
|
missing += 1
|
||||||
|
|
||||||
|
# Non-retireable classifications are recorded for audit completeness when
|
||||||
|
# dry-run lists the fleet, but apply only mutates retireable rows.
|
||||||
|
return {
|
||||||
|
"success": True,
|
||||||
|
"dry_run": dry_run,
|
||||||
|
"retired_count": retired,
|
||||||
|
"already_terminal_count": already,
|
||||||
|
"blocked_count": blocked,
|
||||||
|
"missing_count": missing,
|
||||||
|
"planned_count": len(report.retireable),
|
||||||
|
"results": [r.as_dict() for r in results],
|
||||||
|
"audit_action": EVENT_SESSION_RETIRED,
|
||||||
|
"actor_session_id": actor_session_id,
|
||||||
|
"recorded_at": _ts(moment),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def retire_stale_sessions(
|
||||||
|
db: cpd.ControlPlaneDB,
|
||||||
|
*,
|
||||||
|
sessions: Sequence[Mapping[str, Any]] | None = None,
|
||||||
|
leases: Sequence[Mapping[str, Any]] | None = None,
|
||||||
|
dry_run: bool = False,
|
||||||
|
actor_session_id: str | None = None,
|
||||||
|
now: datetime | None = None,
|
||||||
|
pid_checker: Callable[[int | None], bool] = lease_lifecycle.is_process_alive,
|
||||||
|
process_start_probe: Callable[[int | None], datetime | None] = process_start_time,
|
||||||
|
client_managed_sessions: set[str] | frozenset[str] | None = None,
|
||||||
|
heartbeat_stale_seconds: int = DEFAULT_HEARTBEAT_STALE_SECONDS,
|
||||||
|
session_limit: int = 500,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""End-to-end plan + apply for stale session retirement (#969).
|
||||||
|
|
||||||
|
When *sessions* is omitted the control-plane DB is inventoried (active
|
||||||
|
rows only). Callers that already gathered inventory should pass it.
|
||||||
|
"""
|
||||||
|
moment = now or _utc_now()
|
||||||
|
if sessions is None:
|
||||||
|
sessions = db.list_sessions(
|
||||||
|
statuses=(SESSION_STATUS_ACTIVE,),
|
||||||
|
limit=max(1, int(session_limit)),
|
||||||
|
)
|
||||||
|
if leases is None:
|
||||||
|
try:
|
||||||
|
leases = db.list_leases(statuses=("active",), limit=max(1, int(session_limit)))
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
leases = []
|
||||||
|
|
||||||
|
report = classify_sessions(
|
||||||
|
sessions,
|
||||||
|
leases=leases,
|
||||||
|
now=moment,
|
||||||
|
pid_checker=pid_checker,
|
||||||
|
process_start_probe=process_start_probe,
|
||||||
|
client_managed_sessions=client_managed_sessions,
|
||||||
|
heartbeat_stale_seconds=heartbeat_stale_seconds,
|
||||||
|
)
|
||||||
|
apply_result = apply_session_retirements(
|
||||||
|
db,
|
||||||
|
report,
|
||||||
|
dry_run=dry_run,
|
||||||
|
actor_session_id=actor_session_id,
|
||||||
|
now=moment,
|
||||||
|
)
|
||||||
|
return {
|
||||||
|
"success": True,
|
||||||
|
"fleet": report.as_dict(),
|
||||||
|
"apply": apply_result,
|
||||||
|
"sessions_dimension_resolved": (
|
||||||
|
report.as_dict()["sessions_dimension_resolved"]
|
||||||
|
if dry_run
|
||||||
|
else apply_result["planned_count"]
|
||||||
|
== (
|
||||||
|
apply_result["retired_count"]
|
||||||
|
+ apply_result["already_terminal_count"]
|
||||||
|
)
|
||||||
|
and apply_result["blocked_count"] == 0
|
||||||
|
),
|
||||||
|
}
|
||||||
@@ -204,6 +204,28 @@ proposed command before running it; `gitea_audit_runtime_recovery_contamination`
|
|||||||
to inspect or (reconciler-only) clear the marker. Full contrast in
|
to inspect or (reconciler-only) clear the marker. Full contrast in
|
||||||
`docs/mcp-namespace-eof-recovery.md`.
|
`docs/mcp-namespace-eof-recovery.md`.
|
||||||
|
|
||||||
|
## Connected is not attached (#708)
|
||||||
|
|
||||||
|
A host/CLI MCP inventory showing **Connected** is not proof the tools are usable.
|
||||||
|
The active session can expose **none** of a Connected server's tool namespaces —
|
||||||
|
a *session attachment* failure, distinct from config drift (#672),
|
||||||
|
transport-closed (#584), and resolver EOF (#685).
|
||||||
|
|
||||||
|
Required preflight proof is **live tool visibility plus `gitea_whoami` on the role
|
||||||
|
namespace**, never host Connected status alone. Call
|
||||||
|
`gitea_assess_mcp_namespace_attachment` with the Connected set and the namespaces
|
||||||
|
actually attached to the session; it returns the typed condition
|
||||||
|
**mcp_connected_namespaces_missing** with per-namespace connected-vs-attached proof,
|
||||||
|
and records the verdict so review and merge fail closed while a required namespace
|
||||||
|
is unattached.
|
||||||
|
|
||||||
|
Only sanctioned recovery: the client attach/reconnect path
|
||||||
|
(`gitea_request_mcp_reconnect`), then full preflight
|
||||||
|
(`gitea_whoami` → `gitea_resolve_task_capability` → task). Never recover by direct
|
||||||
|
module import, CLI/raw API mutation, profile hopping, session-state overrides,
|
||||||
|
process kills, or `.env`/mtime edits. A final report must not claim a healthy
|
||||||
|
session without attachment proof.
|
||||||
|
|
||||||
## Shell Spawn Hard-Stop Rule
|
## Shell Spawn Hard-Stop Rule
|
||||||
|
|
||||||
`exit_code: -1` with empty stdout/stderr means the shell failed to spawn — not a
|
`exit_code: -1` with empty stdout/stderr means the shell failed to spawn — not a
|
||||||
|
|||||||
+33
-2
@@ -41,6 +41,27 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
|||||||
"permission": "gitea.issue.comment",
|
"permission": "gitea.issue.comment",
|
||||||
"role": "author",
|
"role": "author",
|
||||||
},
|
},
|
||||||
|
# #953: target-specific upgrade of an incomplete bootstrap lock (explicit
|
||||||
|
# operation, never a widening of lock_issue). Author-only, and the tool
|
||||||
|
# additionally proves exact-owner claimant match before writing.
|
||||||
|
"recover_incomplete_bootstrap_lock": {
|
||||||
|
"permission": "gitea.issue.comment",
|
||||||
|
"role": "author",
|
||||||
|
},
|
||||||
|
"gitea_recover_incomplete_bootstrap_lock": {
|
||||||
|
"permission": "gitea.issue.comment",
|
||||||
|
"role": "author",
|
||||||
|
},
|
||||||
|
# #953: read-only lock contract inspection. Read permission only — it must
|
||||||
|
# never be able to mutate.
|
||||||
|
"inspect_issue_lock_contract": {
|
||||||
|
"permission": "gitea.read",
|
||||||
|
"role": "author",
|
||||||
|
},
|
||||||
|
"gitea_inspect_issue_lock_contract": {
|
||||||
|
"permission": "gitea.read",
|
||||||
|
"role": "author",
|
||||||
|
},
|
||||||
# #860: dirty orphaned same-claimant worktree recovery (explicit operation).
|
# #860: dirty orphaned same-claimant worktree recovery (explicit operation).
|
||||||
"recover_dirty_orphaned_issue_worktree": {
|
"recover_dirty_orphaned_issue_worktree": {
|
||||||
"permission": "gitea.issue.comment",
|
"permission": "gitea.issue.comment",
|
||||||
@@ -132,8 +153,9 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
|||||||
"permission": "gitea.branch.push",
|
"permission": "gitea.branch.push",
|
||||||
"role": "author",
|
"role": "author",
|
||||||
},
|
},
|
||||||
# #662: post-restart reconcile is read-only inventory + pure classification.
|
# #662: post-restart reconcile is inventory + pure classification.
|
||||||
# Durable follow-up issue creation is a separate apply path (not this task).
|
# #969: optional apply_session_cleanup retires confirmed-stale session rows
|
||||||
|
# through the same tool; durable follow-up Gitea issues remain separate.
|
||||||
"reconcile_after_restart": {
|
"reconcile_after_restart": {
|
||||||
"permission": "gitea.read",
|
"permission": "gitea.read",
|
||||||
"role": "author",
|
"role": "author",
|
||||||
@@ -142,6 +164,15 @@ TASK_CAPABILITY_MAP: dict[str, dict[str, str]] = {
|
|||||||
"permission": "gitea.read",
|
"permission": "gitea.read",
|
||||||
"role": "author",
|
"role": "author",
|
||||||
},
|
},
|
||||||
|
# #969: explicit plan/apply path for dead-owner session retirement.
|
||||||
|
"retire_stale_workflow_sessions": {
|
||||||
|
"permission": "gitea.read",
|
||||||
|
"role": "author",
|
||||||
|
},
|
||||||
|
"gitea_retire_stale_workflow_sessions": {
|
||||||
|
"permission": "gitea.read",
|
||||||
|
"role": "author",
|
||||||
|
},
|
||||||
# #644: Phase 2 Web Console recovery tasks.
|
# #644: Phase 2 Web Console recovery tasks.
|
||||||
"clear_stale_binding": {
|
"clear_stale_binding": {
|
||||||
"permission": "gitea.read",
|
"permission": "gitea.read",
|
||||||
|
|||||||
@@ -44,6 +44,8 @@ def _reset_mutation_authority(monkeypatch):
|
|||||||
]:
|
]:
|
||||||
monkeypatch.delenv(env_key, raising=False)
|
monkeypatch.delenv(env_key, raising=False)
|
||||||
|
|
||||||
|
monkeypatch.setenv("GITEA_CLIENT_MANAGED", "1")
|
||||||
|
|
||||||
# Isolate durable session-state files so tests never share host cache (#559).
|
# Isolate durable session-state files so tests never share host cache (#559).
|
||||||
import tempfile
|
import tempfile
|
||||||
|
|
||||||
|
|||||||
@@ -35,7 +35,7 @@ CONFIG = {
|
|||||||
],
|
],
|
||||||
"forbidden_operations": [],
|
"forbidden_operations": [],
|
||||||
"execution_profile": "full-author",
|
"execution_profile": "full-author",
|
||||||
"allowed_repositories": ["Example-Org/Example-Repo"],
|
"allowed_repositories": ["Scaled-Tech-Consulting/Gitea-Tools", "Example-Org/Example-Repo"],
|
||||||
},
|
},
|
||||||
"reviewer-no-commit": {
|
"reviewer-no-commit": {
|
||||||
"enabled": True,
|
"enabled": True,
|
||||||
@@ -50,7 +50,7 @@ CONFIG = {
|
|||||||
"gitea.repo.commit", "gitea.pr.create", "gitea.branch.push"
|
"gitea.repo.commit", "gitea.pr.create", "gitea.branch.push"
|
||||||
],
|
],
|
||||||
"execution_profile": "reviewer-no-commit",
|
"execution_profile": "reviewer-no-commit",
|
||||||
"allowed_repositories": ["Example-Org/Example-Repo"],
|
"allowed_repositories": ["Scaled-Tech-Consulting/Gitea-Tools", "Example-Org/Example-Repo"],
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
"rules": {"allow_runtime_switching": False},
|
"rules": {"allow_runtime_switching": False},
|
||||||
|
|||||||
@@ -175,7 +175,7 @@ class TestLauncherSnippets(unittest.TestCase):
|
|||||||
def test_only_safe_keys_no_secrets(self):
|
def test_only_safe_keys_no_secrets(self):
|
||||||
entry = gitea_config.launcher_entry("prgs", "/cfg/profiles.json")["gitea-tools"]
|
entry = gitea_config.launcher_entry("prgs", "/cfg/profiles.json")["gitea-tools"]
|
||||||
self.assertEqual(set(entry), {"command", "args", "env"})
|
self.assertEqual(set(entry), {"command", "args", "env"})
|
||||||
self.assertEqual(set(entry["env"]), {"GITEA_MCP_CONFIG", "GITEA_MCP_PROFILE"})
|
self.assertEqual(set(entry["env"]), {"GITEA_MCP_CONFIG", "GITEA_MCP_PROFILE", "GITEA_CLIENT_MANAGED"})
|
||||||
self.assertEqual(entry["env"]["GITEA_MCP_PROFILE"], "prgs")
|
self.assertEqual(entry["env"]["GITEA_MCP_PROFILE"], "prgs")
|
||||||
blob = json.dumps(entry).lower()
|
blob = json.dumps(entry).lower()
|
||||||
for word in ("token", "password", "secret"):
|
for word in ("token", "password", "secret"):
|
||||||
|
|||||||
@@ -37,7 +37,7 @@ class ControlPlaneDBTest(unittest.TestCase):
|
|||||||
rows = dict(conn.execute("SELECT key, value FROM schema_meta").fetchall())
|
rows = dict(conn.execute("SELECT key, value FROM schema_meta").fetchall())
|
||||||
finally:
|
finally:
|
||||||
conn.close()
|
conn.close()
|
||||||
self.assertEqual(rows["schema_version"], "5")
|
self.assertEqual(rows["schema_version"], "6")
|
||||||
self.assertIn("DB coordinates", rows["architecture"])
|
self.assertIn("DB coordinates", rows["architecture"])
|
||||||
self.assertIn("bridge", rows["architecture"].lower())
|
self.assertIn("bridge", rows["architecture"].lower())
|
||||||
|
|
||||||
@@ -868,7 +868,7 @@ class SessionCheckpointTest(unittest.TestCase):
|
|||||||
conn.close()
|
conn.close()
|
||||||
self.assertIn("session_checkpoints", names)
|
self.assertIn("session_checkpoints", names)
|
||||||
record = self._write()
|
record = self._write()
|
||||||
self.assertEqual(record["checkpoint_schema_version"], 5)
|
self.assertEqual(record["checkpoint_schema_version"], 6)
|
||||||
|
|
||||||
# AC2 — checkpoints written for multi-role session fixtures.
|
# AC2 — checkpoints written for multi-role session fixtures.
|
||||||
def test_multi_role_fixtures_each_get_a_row(self) -> None:
|
def test_multi_role_fixtures_each_get_a_row(self) -> None:
|
||||||
|
|||||||
@@ -0,0 +1,323 @@
|
|||||||
|
"""Tests for sanctioned Codex MCP reconnect request surface (#678)."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import unittest
|
||||||
|
from unittest import mock
|
||||||
|
|
||||||
|
import mcp_client_reconnect as mcr
|
||||||
|
|
||||||
|
|
||||||
|
class NormalizeReasonTests(unittest.TestCase):
|
||||||
|
def test_stale_runtime_aliases(self):
|
||||||
|
self.assertEqual(mcr.normalize_reason("stale-runtime"), mcr.REASON_STALE_RUNTIME)
|
||||||
|
self.assertEqual(mcr.normalize_reason("stale_runtime"), mcr.REASON_STALE_RUNTIME)
|
||||||
|
self.assertEqual(mcr.normalize_reason("STALE"), mcr.REASON_STALE_RUNTIME)
|
||||||
|
|
||||||
|
def test_transport_eof_aliases(self):
|
||||||
|
self.assertEqual(mcr.normalize_reason("transport_eof"), mcr.REASON_TRANSPORT_EOF)
|
||||||
|
self.assertEqual(mcr.normalize_reason("EOF"), mcr.REASON_TRANSPORT_EOF)
|
||||||
|
self.assertEqual(
|
||||||
|
mcr.normalize_reason("client_is_closing"), mcr.REASON_TRANSPORT_EOF
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_missing_namespace(self):
|
||||||
|
self.assertEqual(
|
||||||
|
mcr.normalize_reason("missing_namespace"), mcr.REASON_MISSING_NAMESPACE
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_empty_is_unspecified(self):
|
||||||
|
self.assertEqual(mcr.normalize_reason(None), mcr.REASON_UNSPECIFIED)
|
||||||
|
self.assertEqual(mcr.normalize_reason(""), mcr.REASON_UNSPECIFIED)
|
||||||
|
|
||||||
|
|
||||||
|
class BoundaryClassificationTests(unittest.TestCase):
|
||||||
|
def test_clean_when_shas_match(self):
|
||||||
|
self.assertEqual(
|
||||||
|
mcr.classify_boundary_status(
|
||||||
|
startup_sha="abc", current_master_sha="abc"
|
||||||
|
),
|
||||||
|
mcr.BOUNDARY_CLEAN,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_mismatch_when_shas_differ(self):
|
||||||
|
self.assertEqual(
|
||||||
|
mcr.classify_boundary_status(
|
||||||
|
startup_sha="aaa", current_master_sha="bbb"
|
||||||
|
),
|
||||||
|
mcr.BOUNDARY_MISMATCH,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_stale_when_live_stale(self):
|
||||||
|
self.assertEqual(
|
||||||
|
mcr.classify_boundary_status(
|
||||||
|
startup_sha="aaa",
|
||||||
|
current_master_sha="aaa",
|
||||||
|
live_stale=True,
|
||||||
|
),
|
||||||
|
mcr.BOUNDARY_STALE,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class BuildReconnectRequestTests(unittest.TestCase):
|
||||||
|
def test_stale_runtime_returns_typed_blocker_with_codex_steps(self):
|
||||||
|
result = mcr.build_reconnect_request(
|
||||||
|
namespace="gitea-author",
|
||||||
|
profile="prgs-author",
|
||||||
|
pid=1234,
|
||||||
|
session_id="sess-1",
|
||||||
|
startup_sha="aaa111",
|
||||||
|
current_master_sha="bbb222",
|
||||||
|
reason="stale-runtime",
|
||||||
|
client="codex",
|
||||||
|
restart_required=True,
|
||||||
|
stop_required=True,
|
||||||
|
)
|
||||||
|
self.assertTrue(result["success"])
|
||||||
|
self.assertTrue(result["read_only"])
|
||||||
|
self.assertFalse(result["reconnect_performed"])
|
||||||
|
self.assertFalse(result["mutation_performed"])
|
||||||
|
self.assertTrue(result["reconnect_needed"])
|
||||||
|
self.assertEqual(result["namespace"], "gitea-author")
|
||||||
|
self.assertEqual(result["profile"], "prgs-author")
|
||||||
|
self.assertEqual(result["pid"], 1234)
|
||||||
|
self.assertEqual(result["session_id"], "sess-1")
|
||||||
|
self.assertEqual(result["startup_sha"], "aaa111")
|
||||||
|
self.assertEqual(result["current_master_sha"], "bbb222")
|
||||||
|
self.assertEqual(result["boundary_status"], mcr.BOUNDARY_MISMATCH)
|
||||||
|
self.assertEqual(result["blocker_kind"], mcr.BLOCKER_OPERATOR_RECONNECT)
|
||||||
|
self.assertIsNotNone(result["typed_blocker"])
|
||||||
|
blocker = result["typed_blocker"]
|
||||||
|
self.assertEqual(blocker["namespaces"], ["gitea-author"])
|
||||||
|
self.assertEqual(blocker["why_reconnect_required"], mcr.REASON_STALE_RUNTIME)
|
||||||
|
self.assertTrue(any("Codex" in s or "Reload" in s for s in blocker["operator_ui_steps"]))
|
||||||
|
self.assertIn("pkill", " ".join(result["forbidden_recovery_paths"]).lower())
|
||||||
|
self.assertTrue(
|
||||||
|
mcr.reasons_never_suggest_forbidden(result["exact_safe_next_action"] or "")
|
||||||
|
)
|
||||||
|
# Must not recommend forbidden recovery.
|
||||||
|
for step in blocker["operator_ui_steps"]:
|
||||||
|
self.assertTrue(mcr.reasons_never_suggest_forbidden(step), step)
|
||||||
|
|
||||||
|
def test_transport_eof_typed_blocker(self):
|
||||||
|
result = mcr.build_reconnect_request(
|
||||||
|
namespace="gitea-reviewer",
|
||||||
|
reason="transport_eof",
|
||||||
|
client="claude_code",
|
||||||
|
)
|
||||||
|
self.assertTrue(result["reconnect_needed"])
|
||||||
|
self.assertEqual(result["reason"], mcr.REASON_TRANSPORT_EOF)
|
||||||
|
self.assertEqual(result["client"], "claude_code")
|
||||||
|
steps = " ".join(result["operator_ui_steps"]).lower()
|
||||||
|
self.assertIn("/mcp", steps)
|
||||||
|
|
||||||
|
def test_missing_namespace_typed_blocker(self):
|
||||||
|
result = mcr.build_reconnect_request(
|
||||||
|
namespace="gitea-merger",
|
||||||
|
reason="missing_namespace",
|
||||||
|
client="codex",
|
||||||
|
)
|
||||||
|
self.assertTrue(result["reconnect_needed"])
|
||||||
|
self.assertEqual(result["reason"], mcr.REASON_MISSING_NAMESPACE)
|
||||||
|
self.assertEqual(
|
||||||
|
result["typed_blocker"]["blocker_kind"], mcr.BLOCKER_OPERATOR_RECONNECT
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_healthy_not_required(self):
|
||||||
|
result = mcr.build_reconnect_request(
|
||||||
|
namespace="gitea-tools",
|
||||||
|
startup_sha="deadbeef",
|
||||||
|
current_master_sha="deadbeef",
|
||||||
|
reason="not_required",
|
||||||
|
client="codex",
|
||||||
|
in_parity=True,
|
||||||
|
restart_required=False,
|
||||||
|
stop_required=False,
|
||||||
|
)
|
||||||
|
self.assertFalse(result["reconnect_needed"])
|
||||||
|
self.assertEqual(result["blocker_kind"], mcr.BLOCKER_NONE)
|
||||||
|
self.assertIsNone(result["typed_blocker"])
|
||||||
|
self.assertFalse(result["stop_required"])
|
||||||
|
self.assertFalse(result["restart_required"])
|
||||||
|
self.assertIn("not required", (result["exact_safe_next_action"] or "").lower())
|
||||||
|
|
||||||
|
def test_successful_reconnect_report_fields_present(self):
|
||||||
|
"""AC2: reconnect result reports required fields (even when needed)."""
|
||||||
|
result = mcr.build_reconnect_request(
|
||||||
|
namespace="gitea-controller",
|
||||||
|
profile="prgs-controller",
|
||||||
|
pid=99,
|
||||||
|
session_id="sid",
|
||||||
|
startup_sha="s" * 40,
|
||||||
|
current_master_sha="c" * 40,
|
||||||
|
reason="stale-runtime",
|
||||||
|
)
|
||||||
|
for key in (
|
||||||
|
"namespace",
|
||||||
|
"profile",
|
||||||
|
"pid",
|
||||||
|
"session_id",
|
||||||
|
"startup_sha",
|
||||||
|
"current_master_sha",
|
||||||
|
"boundary_status",
|
||||||
|
):
|
||||||
|
self.assertIn(key, result)
|
||||||
|
self.assertIsNotNone(result[key], key)
|
||||||
|
|
||||||
|
|
||||||
|
class ToolSurfaceTests(unittest.TestCase):
|
||||||
|
"""Exercise gitea_request_mcp_reconnect with a stubbed server context."""
|
||||||
|
|
||||||
|
def test_tool_is_registered_and_side_effect_free(self):
|
||||||
|
import gitea_mcp_server as srv
|
||||||
|
|
||||||
|
self.assertTrue(hasattr(srv, "gitea_request_mcp_reconnect"))
|
||||||
|
with mock.patch.object(srv, "_profile_operation_gate", return_value=None):
|
||||||
|
with mock.patch.object(
|
||||||
|
srv,
|
||||||
|
"get_profile",
|
||||||
|
return_value={
|
||||||
|
"profile_name": "prgs-author",
|
||||||
|
"role_kind": "author",
|
||||||
|
"role": "author",
|
||||||
|
},
|
||||||
|
):
|
||||||
|
with mock.patch.object(
|
||||||
|
srv,
|
||||||
|
"_current_master_parity",
|
||||||
|
return_value={
|
||||||
|
"startup_head": "a" * 40,
|
||||||
|
"current_head": "a" * 40,
|
||||||
|
"daemon_start_head": "a" * 40,
|
||||||
|
"local_head": "a" * 40,
|
||||||
|
"in_parity": True,
|
||||||
|
"stale": False,
|
||||||
|
"restart_required": False,
|
||||||
|
"determinable": True,
|
||||||
|
"live_stale": False,
|
||||||
|
"live_known": True,
|
||||||
|
"reasons": [],
|
||||||
|
},
|
||||||
|
):
|
||||||
|
with mock.patch.object(
|
||||||
|
srv.master_parity_gate,
|
||||||
|
"format_parity",
|
||||||
|
return_value="in parity",
|
||||||
|
):
|
||||||
|
with mock.patch.object(
|
||||||
|
srv.role_namespace_gate,
|
||||||
|
"infer_mcp_namespace",
|
||||||
|
return_value="gitea-author",
|
||||||
|
):
|
||||||
|
with mock.patch.object(
|
||||||
|
srv.session_ctx,
|
||||||
|
"mutation_context_audit_fields",
|
||||||
|
return_value={"session_profile": "prgs-author"},
|
||||||
|
):
|
||||||
|
result = srv.gitea_request_mcp_reconnect(
|
||||||
|
namespace="gitea-author",
|
||||||
|
reason="not_required",
|
||||||
|
client="codex",
|
||||||
|
remote="prgs",
|
||||||
|
)
|
||||||
|
self.assertTrue(result.get("success"))
|
||||||
|
self.assertFalse(result.get("reconnect_performed"))
|
||||||
|
self.assertFalse(result.get("mutation_performed"))
|
||||||
|
self.assertEqual(result.get("namespace"), "gitea-author")
|
||||||
|
self.assertEqual(result.get("profile"), "prgs-author")
|
||||||
|
self.assertEqual(result.get("pid"), os.getpid())
|
||||||
|
self.assertIn("startup_sha", result)
|
||||||
|
self.assertIn("current_master_sha", result)
|
||||||
|
self.assertIn("boundary_status", result)
|
||||||
|
self.assertTrue(
|
||||||
|
mcr.reasons_never_suggest_forbidden(
|
||||||
|
result.get("exact_safe_next_action") or ""
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_tool_stale_returns_typed_blocker(self):
|
||||||
|
import gitea_mcp_server as srv
|
||||||
|
|
||||||
|
with mock.patch.object(srv, "_profile_operation_gate", return_value=None):
|
||||||
|
with mock.patch.object(
|
||||||
|
srv,
|
||||||
|
"get_profile",
|
||||||
|
return_value={
|
||||||
|
"profile_name": "prgs-reconciler",
|
||||||
|
"role_kind": "reconciler",
|
||||||
|
"role": "reconciler",
|
||||||
|
},
|
||||||
|
):
|
||||||
|
with mock.patch.object(
|
||||||
|
srv,
|
||||||
|
"_current_master_parity",
|
||||||
|
return_value={
|
||||||
|
"startup_head": "a" * 40,
|
||||||
|
"current_head": "b" * 40,
|
||||||
|
"daemon_start_head": "a" * 40,
|
||||||
|
"local_head": "b" * 40,
|
||||||
|
"in_parity": False,
|
||||||
|
"stale": True,
|
||||||
|
"restart_required": True,
|
||||||
|
"determinable": True,
|
||||||
|
"live_stale": True,
|
||||||
|
"live_known": True,
|
||||||
|
"reasons": ["stale"],
|
||||||
|
},
|
||||||
|
):
|
||||||
|
with mock.patch.object(
|
||||||
|
srv.master_parity_gate,
|
||||||
|
"format_parity",
|
||||||
|
return_value="stale",
|
||||||
|
):
|
||||||
|
with mock.patch.object(
|
||||||
|
srv.role_namespace_gate,
|
||||||
|
"infer_mcp_namespace",
|
||||||
|
return_value="gitea-reconciler",
|
||||||
|
):
|
||||||
|
with mock.patch.object(
|
||||||
|
srv.session_ctx,
|
||||||
|
"mutation_context_audit_fields",
|
||||||
|
return_value={},
|
||||||
|
):
|
||||||
|
result = srv.gitea_request_mcp_reconnect(
|
||||||
|
reason="stale-runtime",
|
||||||
|
client="codex",
|
||||||
|
)
|
||||||
|
self.assertTrue(result["reconnect_needed"])
|
||||||
|
self.assertEqual(
|
||||||
|
result["blocker_kind"], mcr.BLOCKER_OPERATOR_RECONNECT
|
||||||
|
)
|
||||||
|
self.assertIsNotNone(result["typed_blocker"])
|
||||||
|
self.assertIn("gitea-reconciler", result["typed_blocker"]["namespaces"])
|
||||||
|
self.assertTrue(result["stop_required"])
|
||||||
|
self.assertTrue(result["restart_required"])
|
||||||
|
self.assertTrue(
|
||||||
|
mcr.reasons_never_suggest_forbidden(
|
||||||
|
result.get("exact_safe_next_action") or ""
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class InventoryRegistrationTests(unittest.TestCase):
|
||||||
|
def test_reconnect_path_in_restart_inventory(self):
|
||||||
|
import mcp_restart_paths as mrp
|
||||||
|
|
||||||
|
ids = {p.path_id for p in mrp.iter_restart_paths()}
|
||||||
|
self.assertIn("codex_client_reconnect_request", ids)
|
||||||
|
self.assertIn("ide_client_reconnect", ids)
|
||||||
|
|
||||||
|
def test_tool_name_in_documented_inventory(self):
|
||||||
|
import mcp_tool_inventory as inv
|
||||||
|
|
||||||
|
doc_path = os.path.join(
|
||||||
|
os.path.dirname(os.path.dirname(__file__)), inv.INVENTORY_DOC_PATH
|
||||||
|
)
|
||||||
|
with open(doc_path, encoding="utf-8") as handle:
|
||||||
|
documented = inv.parse_documented_inventory(handle.read())
|
||||||
|
self.assertIn("gitea_request_mcp_reconnect", documented)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,153 @@
|
|||||||
|
"""Tests for Issue #686: Detect and reject manually launched duplicate MCP role servers."""
|
||||||
|
import os
|
||||||
|
import unittest
|
||||||
|
from unittest.mock import patch, MagicMock
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
import gitea_config
|
||||||
|
import gitea_mcp_server
|
||||||
|
import mcp_namespace_health
|
||||||
|
|
||||||
|
|
||||||
|
class TestIssue686ManualMcpProvenance(unittest.TestCase):
|
||||||
|
|
||||||
|
def test_client_managed_process_detection(self):
|
||||||
|
"""Test _is_client_managed_process correctly detects provenance markers."""
|
||||||
|
with patch.dict(os.environ, {"GITEA_CLIENT_MANAGED": "1"}, clear=True):
|
||||||
|
self.assertTrue(gitea_mcp_server._is_client_managed_process())
|
||||||
|
|
||||||
|
with patch.dict(os.environ, {"GITEA_MCP_CLIENT_MANAGED": "true"}, clear=True):
|
||||||
|
self.assertTrue(gitea_mcp_server._is_client_managed_process())
|
||||||
|
|
||||||
|
with patch.dict(os.environ, {"GITEA_SERVER_PROVENANCE": "client_managed"}, clear=True):
|
||||||
|
self.assertTrue(gitea_mcp_server._is_client_managed_process())
|
||||||
|
|
||||||
|
with patch.dict(os.environ, {"GITEA_CLIENT_MANAGED": "0"}, clear=True):
|
||||||
|
self.assertFalse(gitea_mcp_server._is_client_managed_process())
|
||||||
|
|
||||||
|
def test_unconsumed_gitea_env_overrides(self):
|
||||||
|
"""Test surfacing of unsupported GITEA_* env overrides (e.g. GITEA_DUMMY)."""
|
||||||
|
env = {
|
||||||
|
"GITEA_MCP_PROFILE": "prgs-author",
|
||||||
|
"GITEA_CLIENT_MANAGED": "1",
|
||||||
|
"GITEA_DUMMY": "2",
|
||||||
|
"GITEA_UNKNOWN_FLAG": "abc",
|
||||||
|
}
|
||||||
|
unconsumed = gitea_config.get_unconsumed_gitea_env_overrides(env)
|
||||||
|
self.assertIn("GITEA_DUMMY", unconsumed)
|
||||||
|
self.assertEqual(unconsumed["GITEA_DUMMY"], "2")
|
||||||
|
self.assertIn("GITEA_UNKNOWN_FLAG", unconsumed)
|
||||||
|
self.assertNotIn("GITEA_MCP_PROFILE", unconsumed)
|
||||||
|
self.assertNotIn("GITEA_CLIENT_MANAGED", unconsumed)
|
||||||
|
|
||||||
|
def test_manual_server_mutation_fail_closed(self):
|
||||||
|
"""AC 2: Mutating tools on a server without client-managed provenance fail closed with a typed blocker."""
|
||||||
|
with patch.dict(os.environ, {"GITEA_CLIENT_MANAGED": "0"}, clear=True):
|
||||||
|
block = gitea_mcp_server._provenance_mutation_block(task="create_issue")
|
||||||
|
self.assertIsNotNone(block)
|
||||||
|
self.assertFalse(block["success"])
|
||||||
|
self.assertFalse(block["performed"])
|
||||||
|
self.assertEqual(block["blocker_kind"], "unsupported_manual_launch")
|
||||||
|
self.assertEqual(block["provenance"], "manual_launch")
|
||||||
|
self.assertTrue(any("mutation denied: server process was launched manually" in r for r in block["reasons"]))
|
||||||
|
self.assertIn("BLOCKED + RECONNECT", block["exact_next_action"])
|
||||||
|
|
||||||
|
def test_client_managed_server_mutation_passes_provenance_gate(self):
|
||||||
|
"""AC 3: Clean client-managed baseline passes the provenance gate."""
|
||||||
|
with patch.dict(os.environ, {"GITEA_CLIENT_MANAGED": "1"}, clear=True):
|
||||||
|
block = gitea_mcp_server._provenance_mutation_block(task="create_issue")
|
||||||
|
self.assertIsNone(block)
|
||||||
|
|
||||||
|
@patch("subprocess.run")
|
||||||
|
@patch("os.path.getmtime")
|
||||||
|
@patch("os.path.exists")
|
||||||
|
@patch("os.getpid")
|
||||||
|
def test_manual_duplicate_does_not_mask_stale_runtime(
|
||||||
|
self, mock_getpid, mock_exists, mock_getmtime, mock_run
|
||||||
|
):
|
||||||
|
"""AC 1 & AC 3: Staleness detection ignores manual duplicates and reports stale supported runtimes."""
|
||||||
|
mock_getpid.return_value = 12345
|
||||||
|
mock_exists.return_value = True
|
||||||
|
|
||||||
|
code_time = datetime(2026, 7, 8, 14, 0, 0)
|
||||||
|
mock_getmtime.return_value = code_time.timestamp()
|
||||||
|
|
||||||
|
# PID 12345: stale client-managed process (started at 13:00)
|
||||||
|
# PID 99999: fresh manual duplicate process (started at 15:00, no GITEA_CLIENT_MANAGED)
|
||||||
|
ps_output = (
|
||||||
|
" PID LSTART COMMAND\n"
|
||||||
|
"12345 Wed Jul 8 13:00:00 2026 /path/to/python mcp_server.py\n"
|
||||||
|
"99999 Wed Jul 8 15:00:00 2026 /path/to/python mcp_server.py\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
mock_run_ps = MagicMock()
|
||||||
|
mock_run_ps.stdout = ps_output
|
||||||
|
|
||||||
|
mock_env_12345 = MagicMock()
|
||||||
|
mock_env_12345.stdout = "GITEA_MCP_PROFILE=prgs-author GITEA_CLIENT_MANAGED=1"
|
||||||
|
|
||||||
|
mock_env_99999 = MagicMock()
|
||||||
|
mock_env_99999.stdout = "GITEA_MCP_PROFILE=prgs-author GITEA_DUMMY=2"
|
||||||
|
|
||||||
|
def side_effect(args, **kwargs):
|
||||||
|
if args[0] == "ps" and "eww" in args:
|
||||||
|
pid = args[2]
|
||||||
|
if pid == "12345":
|
||||||
|
return mock_env_12345
|
||||||
|
elif pid == "99999":
|
||||||
|
return mock_env_99999
|
||||||
|
elif args[0] == "ps":
|
||||||
|
return mock_run_ps
|
||||||
|
raise ValueError(f"Unexpected args: {args}")
|
||||||
|
|
||||||
|
mock_run.side_effect = side_effect
|
||||||
|
|
||||||
|
reasons = gitea_mcp_server._check_mcp_runtimes_diagnostics("create_issue", ["prgs-author"])
|
||||||
|
|
||||||
|
# Manual duplicate process must be flagged
|
||||||
|
self.assertTrue(any("Duplicate MCP server process(es) detected" in r for r in reasons))
|
||||||
|
# Unsupported env override (GITEA_DUMMY=2) must be flagged
|
||||||
|
self.assertTrue(any("unsupported-env: Unsupported GITEA_* environment variable override(s) detected: GITEA_DUMMY=2" in r for r in reasons))
|
||||||
|
# Stale runtime must NOT be masked by fresh manual process 99999!
|
||||||
|
self.assertTrue(any("All matching profiles for task 'create_issue' (['prgs-author']) are running but stale" in r for r in reasons))
|
||||||
|
|
||||||
|
def test_namespace_health_classification_includes_provenance(self):
|
||||||
|
"""AC 1 & 4: mcp_namespace_health diagnostics include provenance and unconsumed_gitea_env.
|
||||||
|
|
||||||
|
#948 narrowed the vocabulary here. This process carries no client-managed
|
||||||
|
declaration, so the old code labelled it ``manual_launch`` — asserting a
|
||||||
|
hand-launched terminal process it had no evidence for, and contradicting
|
||||||
|
``gitea_get_runtime_context``, which read the same process and reported
|
||||||
|
``client_managed``. Absence of proof is now reported as ``unproven``.
|
||||||
|
|
||||||
|
The #686 wall itself is unchanged and still asserted below:
|
||||||
|
``is_client_managed`` stays False, so nothing previously refused is now
|
||||||
|
permitted. Only the label on the *reason* changed, so remediation names
|
||||||
|
the proof that is actually missing.
|
||||||
|
"""
|
||||||
|
process = {
|
||||||
|
"pid": 5555,
|
||||||
|
"profile": "prgs-author",
|
||||||
|
"env": {
|
||||||
|
"GITEA_MCP_PROFILE": "prgs-author",
|
||||||
|
"GITEA_DUMMY": "99",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
res = mcp_namespace_health.classify_namespace_probe(
|
||||||
|
"gitea-author",
|
||||||
|
configured=True,
|
||||||
|
registered_tools=["gitea_whoami"],
|
||||||
|
probe_result={"success": True},
|
||||||
|
process=process,
|
||||||
|
probe_source="client_namespace",
|
||||||
|
)
|
||||||
|
self.assertEqual(res["provenance"], "unproven")
|
||||||
|
self.assertFalse(res["is_client_managed"])
|
||||||
|
# The wall is intact: no client-managed proof still fails closed.
|
||||||
|
self.assertTrue(res["provenance_fail_closed"])
|
||||||
|
self.assertEqual(res["unconsumed_gitea_env"], {"GITEA_DUMMY": "99"})
|
||||||
|
self.assertEqual(res["diagnostics"]["provenance"], "unproven")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,269 @@
|
|||||||
|
"""Regression tests for Issue #708 wiring: detection must reach a gate, not just exist.
|
||||||
|
|
||||||
|
The prior #708 slice added a pure decision function with no call site, so a session
|
||||||
|
whose namespaces were Connected-but-unattached still passed every mutation gate.
|
||||||
|
These tests pin the parts that make the detection load-bearing:
|
||||||
|
|
||||||
|
* the typed condition is distinct from #672 / #584 / #685,
|
||||||
|
* the session store records a per-namespace attachment verdict,
|
||||||
|
* review/merge mutations fail closed while a required namespace is unattached,
|
||||||
|
* recovery is reconnect-only and never suggests an unsafe fallback,
|
||||||
|
* startup ordering races and discovery-cache telemetry are reported,
|
||||||
|
* a healthy final report cannot be produced without attachment proof.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import mcp_namespace_health
|
||||||
|
|
||||||
|
|
||||||
|
REQUIRED = ["gitea-author", "gitea-reviewer", "gitea-merger", "gitea-tools"]
|
||||||
|
|
||||||
|
|
||||||
|
def _connected_but_unattached():
|
||||||
|
"""The #708 signature: host says Connected, session tool surface is empty."""
|
||||||
|
return mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=REQUIRED,
|
||||||
|
attached_session_namespaces=[],
|
||||||
|
required_namespaces=REQUIRED,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# --- AC1: distinct typed detection -----------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_connected_with_empty_session_tool_list_is_typed_distinctly():
|
||||||
|
res = _connected_but_unattached()
|
||||||
|
assert res["attachment_healthy"] is False
|
||||||
|
assert res["discovery_status"] == "connected_but_namespaces_missing"
|
||||||
|
assert res["error_type"] == "mcp_connected_namespaces_missing"
|
||||||
|
assert sorted(res["missing_namespaces"]) == sorted(REQUIRED)
|
||||||
|
# Not misclassified as config drift (#672), transport-closed (#584), or EOF (#685).
|
||||||
|
assert res["error_type"] not in {
|
||||||
|
"mcp_config_drift",
|
||||||
|
"transport_closed",
|
||||||
|
"mcp_client_eof",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def test_proof_carries_connected_and_attached_per_namespace():
|
||||||
|
res = _connected_but_unattached()
|
||||||
|
proof = res["proof_of_connected_vs_attached"]
|
||||||
|
for ns in REQUIRED:
|
||||||
|
assert proof[ns] == {"connected": True, "attached": False}
|
||||||
|
|
||||||
|
|
||||||
|
# --- AC2: recovery is auto-attach or reconnect-only -------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_exact_next_action_is_reconnect_only():
|
||||||
|
res = _connected_but_unattached()
|
||||||
|
assert res["reconnect_required"] is True
|
||||||
|
action = res["exact_next_action"]
|
||||||
|
assert "Reconnect the IDE/client MCP session" in action
|
||||||
|
for forbidden in ("pkill", "chmod", "curl", ".env", "sys.path"):
|
||||||
|
assert forbidden not in action
|
||||||
|
|
||||||
|
|
||||||
|
def test_sanctioned_recovery_tool_is_named():
|
||||||
|
res = _connected_but_unattached()
|
||||||
|
assert res["sanctioned_recovery_tool"] == "gitea_request_mcp_reconnect"
|
||||||
|
assert "gitea_request_mcp_reconnect" in " ".join(res["remediation"])
|
||||||
|
|
||||||
|
|
||||||
|
def test_successful_auto_attach_reports_recovered_without_operator():
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=REQUIRED,
|
||||||
|
attached_session_namespaces=REQUIRED,
|
||||||
|
required_namespaces=REQUIRED,
|
||||||
|
auto_attach_attempted=True,
|
||||||
|
auto_attach_succeeded=True,
|
||||||
|
)
|
||||||
|
assert res["attachment_healthy"] is True
|
||||||
|
assert res["auto_recovered"] is True
|
||||||
|
assert res["reconnect_required"] is False
|
||||||
|
|
||||||
|
|
||||||
|
def test_failed_auto_attach_still_requires_reconnect():
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=REQUIRED,
|
||||||
|
attached_session_namespaces=[],
|
||||||
|
required_namespaces=REQUIRED,
|
||||||
|
auto_attach_attempted=True,
|
||||||
|
auto_attach_succeeded=False,
|
||||||
|
)
|
||||||
|
assert res["auto_recovered"] is False
|
||||||
|
assert res["reconnect_required"] is True
|
||||||
|
assert any("did not succeed" in r for r in res["reasons"])
|
||||||
|
|
||||||
|
|
||||||
|
def test_reconnect_rediscovery_clears_the_condition():
|
||||||
|
"""Attach state after a reconnect is healthy without any other change."""
|
||||||
|
before = _connected_but_unattached()
|
||||||
|
after = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=REQUIRED,
|
||||||
|
attached_session_namespaces=REQUIRED,
|
||||||
|
required_namespaces=REQUIRED,
|
||||||
|
discovery_cache_hit=False,
|
||||||
|
discovery_cache_age_seconds=0.0,
|
||||||
|
)
|
||||||
|
assert before["attachment_healthy"] is False
|
||||||
|
assert after["attachment_healthy"] is True
|
||||||
|
assert after["discovery_status"] == "namespaces_attached"
|
||||||
|
assert after["missing_namespaces"] == []
|
||||||
|
|
||||||
|
|
||||||
|
# --- AC4: startup ordering across multiple role servers ---------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_multi_role_startup_ordering_race_is_reported():
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=REQUIRED,
|
||||||
|
attached_session_namespaces=["gitea-author"],
|
||||||
|
required_namespaces=REQUIRED,
|
||||||
|
session_tool_snapshot_at=1000.0,
|
||||||
|
namespace_connected_at={
|
||||||
|
"gitea-author": 990.0,
|
||||||
|
"gitea-reviewer": 1005.0,
|
||||||
|
"gitea-merger": 1007.0,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
assert res["startup_ordering_race"] is True
|
||||||
|
assert res["late_attaching_namespaces"] == ["gitea-merger", "gitea-reviewer"]
|
||||||
|
assert res["telemetry"]["startup_ordering_race"] is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_no_ordering_race_when_snapshot_follows_connect():
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=REQUIRED,
|
||||||
|
attached_session_namespaces=REQUIRED,
|
||||||
|
required_namespaces=REQUIRED,
|
||||||
|
session_tool_snapshot_at=2000.0,
|
||||||
|
namespace_connected_at={ns: 1000.0 for ns in REQUIRED},
|
||||||
|
)
|
||||||
|
assert res["startup_ordering_race"] is False
|
||||||
|
assert res["late_attaching_namespaces"] == []
|
||||||
|
|
||||||
|
|
||||||
|
# --- AC5: telemetry ---------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_telemetry_reports_cache_and_recovery_signals():
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=REQUIRED,
|
||||||
|
attached_session_namespaces=["gitea-author"],
|
||||||
|
required_namespaces=REQUIRED,
|
||||||
|
discovery_cache_hit=True,
|
||||||
|
discovery_cache_age_seconds=42.5,
|
||||||
|
)
|
||||||
|
tel = res["telemetry"]
|
||||||
|
assert tel["connected_count"] == 4
|
||||||
|
assert tel["attached_count"] == 1
|
||||||
|
assert tel["missing_count"] == 3
|
||||||
|
assert tel["discovery_cache_hit"] is True
|
||||||
|
assert tel["discovery_cache_age_seconds"] == 42.5
|
||||||
|
assert tel["reconnect_required"] is True
|
||||||
|
assert tel["error_type"] == "mcp_connected_namespaces_missing"
|
||||||
|
|
||||||
|
|
||||||
|
def test_telemetry_leaks_no_secrets():
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=REQUIRED,
|
||||||
|
attached_session_namespaces=[],
|
||||||
|
required_namespaces=REQUIRED,
|
||||||
|
)
|
||||||
|
blob = repr(res["telemetry"]).lower()
|
||||||
|
for leak in ("token", "authorization", "password", "secret", "/users/", "http"):
|
||||||
|
assert leak not in blob
|
||||||
|
|
||||||
|
|
||||||
|
# --- AC2/AC6: fail-closed mutation gate -------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def _session_store_from(assessment):
|
||||||
|
"""Mirror the server-side recorder without importing the MCP server module."""
|
||||||
|
store = {}
|
||||||
|
missing = set(assessment["missing_namespaces"])
|
||||||
|
for ns, state in assessment["proof_of_connected_vs_attached"].items():
|
||||||
|
attached = bool(state["attached"])
|
||||||
|
store[ns] = {
|
||||||
|
"namespace": ns,
|
||||||
|
"connected": bool(state["connected"]),
|
||||||
|
"attached": attached,
|
||||||
|
"attachment_healthy": attached and ns not in missing,
|
||||||
|
"error_type": None if attached else assessment["error_type"],
|
||||||
|
}
|
||||||
|
return store
|
||||||
|
|
||||||
|
|
||||||
|
def test_review_and_merge_fail_closed_while_unattached():
|
||||||
|
store = _session_store_from(_connected_but_unattached())
|
||||||
|
for task in ("review_pr", "submit_review", "merge_pr", "create_pr", "work_issue"):
|
||||||
|
reasons = mcp_namespace_health.attachment_gate_from_session(task, store)
|
||||||
|
assert reasons, f"{task} must fail closed while its namespace is unattached"
|
||||||
|
assert "fail closed, #708" in reasons[0]
|
||||||
|
|
||||||
|
|
||||||
|
def test_gate_offers_only_sanctioned_recovery():
|
||||||
|
store = _session_store_from(_connected_but_unattached())
|
||||||
|
reasons = mcp_namespace_health.attachment_gate_from_session("merge_pr", store)
|
||||||
|
joined = " ".join(reasons)
|
||||||
|
assert "reconnect" in joined.lower()
|
||||||
|
assert "Workflow Safety Hard Stop (#708)" in joined
|
||||||
|
# Unsafe fallbacks appear only inside the prohibition, never as advice.
|
||||||
|
assert "NEVER use" in joined
|
||||||
|
|
||||||
|
|
||||||
|
def test_gate_passes_once_namespaces_are_attached():
|
||||||
|
healthy = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=REQUIRED,
|
||||||
|
attached_session_namespaces=REQUIRED,
|
||||||
|
required_namespaces=REQUIRED,
|
||||||
|
)
|
||||||
|
store = _session_store_from(healthy)
|
||||||
|
for task in ("review_pr", "submit_review", "merge_pr", "create_pr", "work_issue"):
|
||||||
|
assert mcp_namespace_health.attachment_gate_from_session(task, store) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_unassessed_session_does_not_gate():
|
||||||
|
"""No recorded assessment must not fabricate a block (matches #543 semantics)."""
|
||||||
|
assert mcp_namespace_health.attachment_gate_from_session("merge_pr", {}) == []
|
||||||
|
assert mcp_namespace_health.attachment_gate_from_session("merge_pr", None) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_unmapped_task_is_not_gated():
|
||||||
|
store = _session_store_from(_connected_but_unattached())
|
||||||
|
assert mcp_namespace_health.attachment_gate_from_session("gitea_read", store) == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_partial_attachment_gates_only_the_affected_role():
|
||||||
|
"""Author attached, reviewer not: author work proceeds, review fails closed."""
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=REQUIRED,
|
||||||
|
attached_session_namespaces=["gitea-author", "gitea-tools"],
|
||||||
|
required_namespaces=REQUIRED,
|
||||||
|
)
|
||||||
|
store = _session_store_from(res)
|
||||||
|
assert mcp_namespace_health.attachment_gate_from_session("work_issue", store) == []
|
||||||
|
assert mcp_namespace_health.attachment_gate_from_session("review_pr", store)
|
||||||
|
assert mcp_namespace_health.attachment_gate_from_session("merge_pr", store)
|
||||||
|
|
||||||
|
|
||||||
|
# --- AC4: no false healthy report without attachment proof ------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_no_false_healthy_without_attachment_proof():
|
||||||
|
"""Connected alone never yields a healthy verdict."""
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=REQUIRED,
|
||||||
|
attached_session_namespaces=None,
|
||||||
|
required_namespaces=REQUIRED,
|
||||||
|
)
|
||||||
|
assert res["success"] is False
|
||||||
|
assert res["attachment_healthy"] is False
|
||||||
|
assert res["telemetry"]["attached_count"] == 0
|
||||||
|
|
||||||
|
|
||||||
|
def test_attachment_gate_maps_each_role_namespace():
|
||||||
|
assert mcp_namespace_health.required_namespace_for_attachment("review_pr") == "gitea-reviewer"
|
||||||
|
assert mcp_namespace_health.required_namespace_for_attachment("merge_pr") == "gitea-merger"
|
||||||
|
assert mcp_namespace_health.required_namespace_for_attachment("create_pr") == "gitea-author"
|
||||||
|
assert mcp_namespace_health.required_namespace_for_attachment("nope") is None
|
||||||
@@ -0,0 +1,77 @@
|
|||||||
|
"""Unit regression tests for Issue #708: Connected-but-namespaces-missing detection and attachment safety."""
|
||||||
|
|
||||||
|
import mcp_namespace_health
|
||||||
|
|
||||||
|
|
||||||
|
def test_assess_connected_namespace_attachment_success():
|
||||||
|
# required_namespaces is declared explicitly: these three are the namespaces this case
|
||||||
|
# is about. Relying on the default (which also requires gitea-tools) would ask for a
|
||||||
|
# healthy verdict covering a required namespace that was never connected — exactly the
|
||||||
|
# false-healthy classification these tests now forbid.
|
||||||
|
connected = ["gitea-author", "gitea-reviewer", "gitea-merger"]
|
||||||
|
attached = ["gitea-author", "gitea-reviewer", "gitea-merger"]
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=connected,
|
||||||
|
attached_session_namespaces=attached,
|
||||||
|
required_namespaces=connected,
|
||||||
|
)
|
||||||
|
assert res["success"] is True
|
||||||
|
assert res["attachment_healthy"] is True
|
||||||
|
assert res["discovery_status"] == "namespaces_attached"
|
||||||
|
assert res["error_type"] is None
|
||||||
|
assert res["missing_namespaces"] == []
|
||||||
|
assert res["exact_next_action"] == "None; session tool namespaces attached."
|
||||||
|
|
||||||
|
|
||||||
|
def test_assess_connected_namespace_attachment_missing():
|
||||||
|
# Every required namespace here *is* connected, so the only condition present is the
|
||||||
|
# #708 one: Connected at the host, absent from the session tool surface.
|
||||||
|
connected = ["gitea-author", "gitea-reviewer", "gitea-merger"]
|
||||||
|
attached = ["gitea-author"]
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=connected,
|
||||||
|
attached_session_namespaces=attached,
|
||||||
|
required_namespaces=connected,
|
||||||
|
)
|
||||||
|
assert res["success"] is False
|
||||||
|
assert res["attachment_healthy"] is False
|
||||||
|
assert res["discovery_status"] == "connected_but_namespaces_missing"
|
||||||
|
assert res["error_type"] == "mcp_connected_namespaces_missing"
|
||||||
|
assert "gitea-reviewer" in res["missing_namespaces"]
|
||||||
|
assert "gitea-merger" in res["missing_namespaces"]
|
||||||
|
assert "Reconnect the IDE/client MCP session" in res["exact_next_action"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_assess_connected_namespace_attachment_disconnected():
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=[],
|
||||||
|
attached_session_namespaces=[],
|
||||||
|
)
|
||||||
|
assert res["success"] is False
|
||||||
|
assert res["attachment_healthy"] is False
|
||||||
|
assert res["discovery_status"] == "disconnected"
|
||||||
|
|
||||||
|
|
||||||
|
def test_proof_of_connected_vs_attached_mapping():
|
||||||
|
connected = ["gitea-author", "gitea-reviewer"]
|
||||||
|
attached = ["gitea-author"]
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=connected,
|
||||||
|
attached_session_namespaces=attached,
|
||||||
|
)
|
||||||
|
proof = res["proof_of_connected_vs_attached"]
|
||||||
|
assert proof["gitea-author"] == {"connected": True, "attached": True}
|
||||||
|
assert proof["gitea-reviewer"] == {"connected": True, "attached": False}
|
||||||
|
|
||||||
|
|
||||||
|
def test_unsafe_fallback_policy_enforcement():
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=["gitea-author"],
|
||||||
|
attached_session_namespaces=[],
|
||||||
|
)
|
||||||
|
policy = res["unsafe_fallback_policy"]
|
||||||
|
assert "Workflow Safety Hard Stop (#708)" in policy
|
||||||
|
assert "direct imports" in policy
|
||||||
|
assert "API mutations" in policy
|
||||||
|
assert "profile hopping" in policy
|
||||||
|
assert "session-state overrides" in policy
|
||||||
@@ -0,0 +1,283 @@
|
|||||||
|
"""Regression tests for Issue #708 B2: not-connected is not Connected-but-unattached.
|
||||||
|
|
||||||
|
The first #708 slice counted a namespace as ``missing`` only when it appeared in
|
||||||
|
``connected_servers``. A *required* namespace absent from that inventory was therefore
|
||||||
|
never counted at all, so the session reported ``attachment_healthy: true`` with
|
||||||
|
``error_type: None`` while holding no attachment proof for it — and the gate text told the
|
||||||
|
operator "the host reports Connected" about a service nothing had reported Connected.
|
||||||
|
|
||||||
|
These tests pin the corrected distinction:
|
||||||
|
|
||||||
|
* required and not connected never yields a healthy verdict,
|
||||||
|
* it is typed ``mcp_required_namespaces_not_connected``, never the #708 condition,
|
||||||
|
* ``mcp_connected_namespaces_missing`` stays reserved for genuinely Connected namespaces,
|
||||||
|
* both categories still fail review and merge closed, with their own reason,
|
||||||
|
* per-namespace ``connected``/``attached`` evidence is reported accurately,
|
||||||
|
* one session's evidence cannot clear another session's block.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import mcp_namespace_health
|
||||||
|
|
||||||
|
|
||||||
|
ROLES = ["gitea-author", "gitea-reviewer", "gitea-merger", "gitea-tools"]
|
||||||
|
|
||||||
|
NOT_CONNECTED = mcp_namespace_health.ERROR_REQUIRED_NAMESPACES_NOT_CONNECTED
|
||||||
|
CONNECTED_MISSING = mcp_namespace_health.ERROR_CONNECTED_NAMESPACES_MISSING
|
||||||
|
|
||||||
|
|
||||||
|
def _assess(connected, attached, required):
|
||||||
|
return mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=connected,
|
||||||
|
attached_session_namespaces=attached,
|
||||||
|
required_namespaces=required,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _store(assessment):
|
||||||
|
"""The recorder contract: per-namespace verdicts feed the gate."""
|
||||||
|
return dict(assessment["namespace_conditions"])
|
||||||
|
|
||||||
|
|
||||||
|
# --- the four evidence combinations ----------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_connected_and_attached_is_healthy():
|
||||||
|
res = _assess(ROLES, ROLES, ROLES)
|
||||||
|
assert res["attachment_healthy"] is True
|
||||||
|
assert res["error_type"] is None
|
||||||
|
assert res["missing_namespaces"] == []
|
||||||
|
assert res["not_connected_namespaces"] == []
|
||||||
|
assert res["discovery_status"] == "namespaces_attached"
|
||||||
|
|
||||||
|
|
||||||
|
def test_connected_but_unattached_keeps_the_708_condition():
|
||||||
|
res = _assess(ROLES, ["gitea-author"], ROLES)
|
||||||
|
assert res["attachment_healthy"] is False
|
||||||
|
assert res["error_type"] == CONNECTED_MISSING
|
||||||
|
assert res["not_connected_namespaces"] == []
|
||||||
|
assert sorted(res["missing_namespaces"]) == [
|
||||||
|
"gitea-merger",
|
||||||
|
"gitea-reviewer",
|
||||||
|
"gitea-tools",
|
||||||
|
]
|
||||||
|
assert res["discovery_status"] == "connected_but_namespaces_missing"
|
||||||
|
|
||||||
|
|
||||||
|
def test_required_but_not_connected_is_never_healthy():
|
||||||
|
"""The reviewer's exact reproduction from review 637."""
|
||||||
|
res = _assess(
|
||||||
|
["gitea-reviewer"], ["gitea-reviewer"], ["gitea-reviewer", "gitea-merger"]
|
||||||
|
)
|
||||||
|
assert res["attachment_healthy"] is False
|
||||||
|
assert res["success"] is False
|
||||||
|
assert res["error_type"] is not None
|
||||||
|
assert res["telemetry"]["not_connected_count"] == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_required_but_not_connected_is_typed_distinctly():
|
||||||
|
res = _assess(
|
||||||
|
["gitea-reviewer"], ["gitea-reviewer"], ["gitea-reviewer", "gitea-merger"]
|
||||||
|
)
|
||||||
|
assert res["error_type"] == NOT_CONNECTED
|
||||||
|
assert res["discovery_status"] == "required_namespaces_not_connected"
|
||||||
|
assert res["not_connected_namespaces"] == ["gitea-merger"]
|
||||||
|
# Not collapsed into the #708 condition, nor into config drift (#672) or
|
||||||
|
# transport-closed (#584).
|
||||||
|
assert CONNECTED_MISSING not in res["error_types"]
|
||||||
|
assert res["missing_namespaces"] == []
|
||||||
|
assert res["error_type"] not in {"mcp_config_drift", "transport_closed"}
|
||||||
|
|
||||||
|
|
||||||
|
def test_not_connected_reason_makes_no_connected_claim():
|
||||||
|
res = _assess(
|
||||||
|
["gitea-reviewer"], ["gitea-reviewer"], ["gitea-reviewer", "gitea-merger"]
|
||||||
|
)
|
||||||
|
about_merger = [r for r in res["reasons"] if "gitea-merger" in r]
|
||||||
|
assert about_merger, "the not-connected namespace must be named in the reasons"
|
||||||
|
for reason in about_merger:
|
||||||
|
assert "report Connected at host/CLI layer" not in reason
|
||||||
|
assert "gitea-merger" in res["exact_next_action"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_neither_connected_nor_attached_reports_disconnected():
|
||||||
|
res = _assess([], [], ROLES)
|
||||||
|
assert res["attachment_healthy"] is False
|
||||||
|
assert res["discovery_status"] == "disconnected"
|
||||||
|
assert res["error_type"] == NOT_CONNECTED
|
||||||
|
assert sorted(res["not_connected_namespaces"]) == sorted(ROLES)
|
||||||
|
assert res["missing_namespaces"] == []
|
||||||
|
|
||||||
|
|
||||||
|
# --- mixed required namespaces ---------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_mixed_connected_attached_and_not_connected():
|
||||||
|
res = _assess(
|
||||||
|
["gitea-author"], ["gitea-author"], ["gitea-author", "gitea-merger"]
|
||||||
|
)
|
||||||
|
assert res["attachment_healthy"] is False
|
||||||
|
assert res["not_connected_namespaces"] == ["gitea-merger"]
|
||||||
|
assert res["missing_namespaces"] == []
|
||||||
|
conditions = res["namespace_conditions"]
|
||||||
|
assert conditions["gitea-author"]["attachment_healthy"] is True
|
||||||
|
assert conditions["gitea-author"]["condition"] is None
|
||||||
|
assert conditions["gitea-merger"]["attachment_healthy"] is False
|
||||||
|
assert conditions["gitea-merger"]["condition"] == NOT_CONNECTED
|
||||||
|
|
||||||
|
|
||||||
|
def test_both_conditions_present_are_both_reported():
|
||||||
|
"""One namespace Connected-but-unattached, another never connected."""
|
||||||
|
res = _assess(
|
||||||
|
["gitea-author", "gitea-reviewer"],
|
||||||
|
["gitea-author"],
|
||||||
|
["gitea-author", "gitea-reviewer", "gitea-merger"],
|
||||||
|
)
|
||||||
|
assert res["missing_namespaces"] == ["gitea-reviewer"]
|
||||||
|
assert res["not_connected_namespaces"] == ["gitea-merger"]
|
||||||
|
# Neither condition is hidden by the other; error_type names the primary one.
|
||||||
|
assert sorted(res["error_types"]) == sorted([NOT_CONNECTED, CONNECTED_MISSING])
|
||||||
|
assert res["error_type"] == NOT_CONNECTED
|
||||||
|
|
||||||
|
|
||||||
|
# --- unknown, partial, malformed, contradictory evidence --------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_unknown_attachment_evidence_is_not_healthy():
|
||||||
|
"""No session tool surface reported at all is unproven, not proven good."""
|
||||||
|
res = _assess(ROLES, None, ROLES)
|
||||||
|
assert res["attachment_healthy"] is False
|
||||||
|
assert res["telemetry"]["attached_count"] == 0
|
||||||
|
assert res["error_type"] == CONNECTED_MISSING
|
||||||
|
|
||||||
|
|
||||||
|
def test_partial_evidence_gates_only_the_unproven_roles():
|
||||||
|
res = _assess(ROLES, ["gitea-author", "gitea-tools"], ROLES)
|
||||||
|
store = _store(res)
|
||||||
|
assert mcp_namespace_health.attachment_gate_from_session("work_issue", store) == []
|
||||||
|
assert mcp_namespace_health.attachment_gate_from_session("review_pr", store)
|
||||||
|
assert mcp_namespace_health.attachment_gate_from_session("merge_pr", store)
|
||||||
|
|
||||||
|
|
||||||
|
def test_malformed_namespace_entries_are_discarded_not_trusted():
|
||||||
|
"""Blank and whitespace-only names must not become namespaces or proof."""
|
||||||
|
res = mcp_namespace_health.assess_connected_namespace_attachment(
|
||||||
|
connected_servers=["gitea-author", "", " "],
|
||||||
|
attached_session_namespaces=["gitea-author", ""],
|
||||||
|
required_namespaces=["gitea-author", " "],
|
||||||
|
)
|
||||||
|
assert "" not in res["proof_of_connected_vs_attached"]
|
||||||
|
assert " " not in res["proof_of_connected_vs_attached"]
|
||||||
|
assert res["attachment_healthy"] is True
|
||||||
|
assert res["telemetry"]["required_count"] == 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_contradictory_attached_without_connected_fails_closed():
|
||||||
|
"""Attached in the session yet absent from the connected inventory."""
|
||||||
|
res = _assess(["gitea-author"], ["gitea-author", "gitea-merger"], ROLES)
|
||||||
|
assert res["attachment_healthy"] is False
|
||||||
|
assert "gitea-merger" in res["contradictory_namespaces"]
|
||||||
|
assert "gitea-merger" in res["not_connected_namespaces"]
|
||||||
|
assert res["namespace_conditions"]["gitea-merger"]["contradictory_evidence"] is True
|
||||||
|
assert res["namespace_conditions"]["gitea-merger"]["attachment_healthy"] is False
|
||||||
|
assert any("contradictory evidence" in r for r in res["reasons"])
|
||||||
|
assert res["telemetry"]["contradictory_count"] >= 1
|
||||||
|
|
||||||
|
|
||||||
|
def test_duplicate_required_entries_are_counted_once():
|
||||||
|
res = _assess(["gitea-author"], ["gitea-author"], ["gitea-author", "gitea-author"])
|
||||||
|
assert res["telemetry"]["required_count"] == 1
|
||||||
|
assert res["attachment_healthy"] is True
|
||||||
|
|
||||||
|
|
||||||
|
# --- review and merge behaviour for both failure categories -----------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_merge_fails_closed_when_required_namespace_never_connected():
|
||||||
|
store = _store(_assess(["gitea-author"], ["gitea-author"], ROLES))
|
||||||
|
reasons = mcp_namespace_health.attachment_gate_from_session("merge_pr", store)
|
||||||
|
assert reasons
|
||||||
|
assert NOT_CONNECTED in reasons[0]
|
||||||
|
assert "fail closed, #708" in reasons[0]
|
||||||
|
|
||||||
|
|
||||||
|
def test_review_fails_closed_when_required_namespace_never_connected():
|
||||||
|
store = _store(_assess(["gitea-author"], ["gitea-author"], ROLES))
|
||||||
|
reasons = mcp_namespace_health.attachment_gate_from_session("review_pr", store)
|
||||||
|
assert reasons
|
||||||
|
assert NOT_CONNECTED in reasons[0]
|
||||||
|
assert "fail closed, #708" in reasons[0]
|
||||||
|
|
||||||
|
|
||||||
|
def test_not_connected_block_never_claims_the_host_reports_connected():
|
||||||
|
store = _store(_assess(["gitea-author"], ["gitea-author"], ROLES))
|
||||||
|
reasons = mcp_namespace_health.attachment_gate_from_session("merge_pr", store)
|
||||||
|
assert "the host reports Connected" not in reasons[0]
|
||||||
|
assert "absent from the connected-service inventory" in reasons[0]
|
||||||
|
|
||||||
|
|
||||||
|
def test_connected_but_unattached_block_still_says_connected():
|
||||||
|
store = _store(_assess(ROLES, ["gitea-author"], ROLES))
|
||||||
|
reasons = mcp_namespace_health.attachment_gate_from_session("merge_pr", store)
|
||||||
|
assert CONNECTED_MISSING in reasons[0]
|
||||||
|
assert "the host reports Connected" in reasons[0]
|
||||||
|
|
||||||
|
|
||||||
|
def test_both_categories_offer_only_the_sanctioned_recovery():
|
||||||
|
for store in (
|
||||||
|
_store(_assess(ROLES, [], ROLES)),
|
||||||
|
_store(_assess(["gitea-author"], ["gitea-author"], ROLES)),
|
||||||
|
):
|
||||||
|
reasons = mcp_namespace_health.attachment_gate_from_session("merge_pr", store)
|
||||||
|
joined = " ".join(reasons)
|
||||||
|
assert "reconnect" in joined.lower()
|
||||||
|
assert "Workflow Safety Hard Stop (#708)" in joined
|
||||||
|
for forbidden in ("pkill", "chmod", "curl ", ".env", "sys.path"):
|
||||||
|
assert forbidden not in reasons[0]
|
||||||
|
|
||||||
|
|
||||||
|
def test_entry_without_connected_evidence_blocks_without_asserting_either():
|
||||||
|
"""A legacy/partial store entry must fail closed and claim nothing it cannot prove."""
|
||||||
|
store = {"gitea-merger": {"namespace": "gitea-merger", "attached": False}}
|
||||||
|
reasons = mcp_namespace_health.attachment_gate_from_session("merge_pr", store)
|
||||||
|
assert reasons
|
||||||
|
assert "no connected-status evidence" in reasons[0]
|
||||||
|
assert "the host reports Connected" not in reasons[0]
|
||||||
|
assert "fail closed, #708" in reasons[0]
|
||||||
|
|
||||||
|
|
||||||
|
# --- session isolation ------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_one_session_evidence_does_not_clear_another_session_block():
|
||||||
|
"""Attachment evidence is per-session state; it must not travel between sessions."""
|
||||||
|
blocked_session = _store(_assess(["gitea-author"], ["gitea-author"], ROLES))
|
||||||
|
healthy_session = _store(_assess(ROLES, ROLES, ROLES))
|
||||||
|
|
||||||
|
assert (
|
||||||
|
mcp_namespace_health.attachment_gate_from_session("merge_pr", healthy_session)
|
||||||
|
== []
|
||||||
|
)
|
||||||
|
# The healthy session's verdict is not consulted for the blocked session.
|
||||||
|
assert mcp_namespace_health.attachment_gate_from_session("merge_pr", blocked_session)
|
||||||
|
# And the blocked session's store is unchanged by the healthy one existing.
|
||||||
|
assert blocked_session["gitea-merger"]["attachment_healthy"] is False
|
||||||
|
|
||||||
|
|
||||||
|
def test_gate_reads_only_the_store_it_is_given():
|
||||||
|
healthy_session = _store(_assess(ROLES, ROLES, ROLES))
|
||||||
|
assert mcp_namespace_health.attachment_gate_from_session("merge_pr", {}) == []
|
||||||
|
assert mcp_namespace_health.attachment_gate_from_session("merge_pr", None) == []
|
||||||
|
assert (
|
||||||
|
mcp_namespace_health.attachment_gate_from_session("merge_pr", healthy_session)
|
||||||
|
== []
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# --- telemetry stays secret-free -------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def test_not_connected_telemetry_leaks_no_secrets():
|
||||||
|
res = _assess(["gitea-author"], ["gitea-author"], ROLES)
|
||||||
|
blob = repr(res["telemetry"]).lower()
|
||||||
|
for leak in ("token", "authorization", "password", "secret", "/users/", "http"):
|
||||||
|
assert leak not in blob
|
||||||
@@ -35,6 +35,17 @@ BREAK_GLASS_ENV = "GITEA_BREAKGLASS_RESTART_AUTHORIZATION"
|
|||||||
QUIET_SESSIONS: list[dict] = []
|
QUIET_SESSIONS: list[dict] = []
|
||||||
QUIET_LEASES: list[dict] = []
|
QUIET_LEASES: list[dict] = []
|
||||||
|
|
||||||
|
# #669: broad restarts need a prior narrow-attempt log (unless break-glass).
|
||||||
|
PRIOR_NARROW_ATTEMPTS_JSON = json.dumps(
|
||||||
|
[
|
||||||
|
{
|
||||||
|
"action": "client_reconnect",
|
||||||
|
"outcome": "insufficient",
|
||||||
|
"reason": "still flapping after reconnect",
|
||||||
|
}
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
class _FakeDB:
|
class _FakeDB:
|
||||||
"""Minimal control-plane DB stand-in for the restart inventory."""
|
"""Minimal control-plane DB stand-in for the restart inventory."""
|
||||||
@@ -128,6 +139,7 @@ class TestConjunction(_RestartToolHarness):
|
|||||||
preview = self._call(
|
preview = self._call(
|
||||||
role="operator",
|
role="operator",
|
||||||
restart_class="full_mcp_restart",
|
restart_class="full_mcp_restart",
|
||||||
|
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||||
)
|
)
|
||||||
self.assertTrue(preview["allow_restart"],
|
self.assertTrue(preview["allow_restart"],
|
||||||
@@ -136,6 +148,7 @@ class TestConjunction(_RestartToolHarness):
|
|||||||
result = self._call(
|
result = self._call(
|
||||||
role="operator",
|
role="operator",
|
||||||
restart_class="full_mcp_restart",
|
restart_class="full_mcp_restart",
|
||||||
|
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||||
dry_run=False,
|
dry_run=False,
|
||||||
drain_proof_json=self._clean_proof_for(preview),
|
drain_proof_json=self._clean_proof_for(preview),
|
||||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||||
@@ -342,6 +355,7 @@ class TestExistingPathsStillWork(_RestartToolHarness):
|
|||||||
result = self._call(
|
result = self._call(
|
||||||
role="operator",
|
role="operator",
|
||||||
restart_class="full_mcp_restart",
|
restart_class="full_mcp_restart",
|
||||||
|
prior_recovery_attempts_json=PRIOR_NARROW_ATTEMPTS_JSON,
|
||||||
dry_run=False,
|
dry_run=False,
|
||||||
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
env={CONTROLLER_APPROVAL_ENV: "operator-approved"},
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -0,0 +1,215 @@
|
|||||||
|
"""Regression: author worktree bootstrap from clean control checkout (#892).
|
||||||
|
|
||||||
|
#892 is the four-door deadlock where every documented recovery path is closed:
|
||||||
|
bootstrap refuses control, lock demands an existing worktree, worktree-start
|
||||||
|
demands a lock, and shell worktree add is outside the sanctioned MCP path.
|
||||||
|
|
||||||
|
Root cause: assess_author_issue_bootstrap returned allowed/proven for a clean
|
||||||
|
control checkout, but bootstrap_permits_control_checkout only accepted
|
||||||
|
create_issue assessments (task_scope=create_issue_only + empty reasons + full
|
||||||
|
base-tip field set). Author assessments never satisfied the shared predicate,
|
||||||
|
so the #274/#604 guards kept the ordinary control-checkout block.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
from unittest import mock
|
||||||
|
|
||||||
|
import author_issue_bootstrap as aib
|
||||||
|
import create_issue_bootstrap as cib
|
||||||
|
|
||||||
|
|
||||||
|
CONTROL = "/repo/Gitea-Tools"
|
||||||
|
MASTER = "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"
|
||||||
|
OTHER = "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"
|
||||||
|
|
||||||
|
|
||||||
|
def _assess(
|
||||||
|
*,
|
||||||
|
workspace=CONTROL,
|
||||||
|
root=CONTROL,
|
||||||
|
branch="master",
|
||||||
|
head=MASTER,
|
||||||
|
porcelain="",
|
||||||
|
remote=MASTER,
|
||||||
|
remote_error=None,
|
||||||
|
task="bootstrap_author_issue_worktree",
|
||||||
|
):
|
||||||
|
return aib.assess_author_issue_bootstrap(
|
||||||
|
workspace_path=workspace,
|
||||||
|
canonical_repo_root=root,
|
||||||
|
current_branch=branch,
|
||||||
|
head_sha=head,
|
||||||
|
porcelain_status=porcelain,
|
||||||
|
remote_master_sha=remote,
|
||||||
|
remote_master_sha_error=remote_error,
|
||||||
|
task=task,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class TestAuthorBootstrapAssessmentShape(unittest.TestCase):
|
||||||
|
def test_clean_control_emits_predicate_compatible_fields(self):
|
||||||
|
assessment = _assess()
|
||||||
|
self.assertTrue(assessment["allowed"])
|
||||||
|
self.assertTrue(assessment["proven"])
|
||||||
|
self.assertFalse(assessment["block"])
|
||||||
|
self.assertFalse(assessment["not_applicable"])
|
||||||
|
self.assertEqual(assessment["reasons"], [])
|
||||||
|
self.assertEqual(assessment["task_scope"], "author_issue_bootstrap")
|
||||||
|
self.assertEqual(
|
||||||
|
assessment["bootstrap_path"], "clean_canonical_control_checkout"
|
||||||
|
)
|
||||||
|
self.assertEqual(assessment["dirty_files"], [])
|
||||||
|
self.assertIs(assessment["under_branches"], False)
|
||||||
|
self.assertTrue(assessment["base_tips_verified"])
|
||||||
|
self.assertEqual(assessment["local_head_sha"], MASTER)
|
||||||
|
self.assertEqual(assessment["remote_master_sha"], MASTER)
|
||||||
|
self.assertEqual(assessment["workspace_path"], os.path.realpath(CONTROL))
|
||||||
|
self.assertEqual(
|
||||||
|
assessment["canonical_repo_root"], os.path.realpath(CONTROL)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_wrong_task_not_applicable(self):
|
||||||
|
assessment = _assess(task="lock_issue")
|
||||||
|
self.assertTrue(assessment["not_applicable"])
|
||||||
|
self.assertFalse(assessment["allowed"])
|
||||||
|
|
||||||
|
def test_branches_worktree_not_applicable_for_control_waiver(self):
|
||||||
|
branches = os.path.join(CONTROL, "branches", "fix-issue-1")
|
||||||
|
assessment = _assess(workspace=branches)
|
||||||
|
self.assertTrue(assessment["not_applicable"])
|
||||||
|
self.assertFalse(assessment["allowed"])
|
||||||
|
self.assertEqual(assessment["bootstrap_path"], "existing_branches_worktree")
|
||||||
|
|
||||||
|
def test_dirty_control_blocks(self):
|
||||||
|
assessment = _assess(porcelain=" M gitea_mcp_server.py\n")
|
||||||
|
self.assertTrue(assessment["block"])
|
||||||
|
self.assertFalse(assessment["allowed"])
|
||||||
|
self.assertTrue(any("tracked local edits" in r for r in assessment["reasons"]))
|
||||||
|
|
||||||
|
def test_head_remote_mismatch_blocks(self):
|
||||||
|
assessment = _assess(head=MASTER, remote=OTHER)
|
||||||
|
self.assertTrue(assessment["block"])
|
||||||
|
self.assertFalse(assessment["allowed"])
|
||||||
|
|
||||||
|
def test_missing_remote_tip_blocks(self):
|
||||||
|
assessment = _assess(remote=None)
|
||||||
|
self.assertTrue(assessment["block"])
|
||||||
|
self.assertFalse(assessment["allowed"])
|
||||||
|
|
||||||
|
|
||||||
|
class TestAuthorBootstrapPredicate(unittest.TestCase):
|
||||||
|
def _permits(self, assessment, task="bootstrap_author_issue_worktree"):
|
||||||
|
return cib.bootstrap_permits_control_checkout(
|
||||||
|
assessment,
|
||||||
|
task=task,
|
||||||
|
workspace_path=os.path.realpath(CONTROL),
|
||||||
|
canonical_repo_root=os.path.realpath(CONTROL),
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_clean_author_bootstrap_permits(self):
|
||||||
|
self.assertTrue(self._permits(_assess()))
|
||||||
|
|
||||||
|
def test_tool_alias_permits(self):
|
||||||
|
assessment = _assess(task="gitea_bootstrap_author_issue_worktree")
|
||||||
|
self.assertTrue(
|
||||||
|
self._permits(assessment, task="gitea_bootstrap_author_issue_worktree")
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_create_issue_scope_cannot_license_author_bootstrap(self):
|
||||||
|
# Cross-scope smuggling: a create_issue-shaped assessment must not
|
||||||
|
# authorize the author bootstrap task.
|
||||||
|
create_shaped = dict(_assess())
|
||||||
|
create_shaped["task_scope"] = "create_issue_only"
|
||||||
|
self.assertFalse(self._permits(create_shaped))
|
||||||
|
|
||||||
|
def test_author_scope_cannot_license_create_issue(self):
|
||||||
|
assessment = _assess()
|
||||||
|
self.assertFalse(
|
||||||
|
cib.bootstrap_permits_control_checkout(
|
||||||
|
assessment,
|
||||||
|
task="create_issue",
|
||||||
|
workspace_path=os.path.realpath(CONTROL),
|
||||||
|
canonical_repo_root=os.path.realpath(CONTROL),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_nonempty_reasons_fail_closed(self):
|
||||||
|
bad = dict(_assess(), reasons=["informational text must not be here"])
|
||||||
|
self.assertFalse(self._permits(bad))
|
||||||
|
|
||||||
|
def test_dirty_fails_closed(self):
|
||||||
|
self.assertFalse(self._permits(_assess(porcelain=" M x.py\n")))
|
||||||
|
|
||||||
|
def test_mismatch_fails_closed(self):
|
||||||
|
self.assertFalse(self._permits(_assess(remote=OTHER)))
|
||||||
|
|
||||||
|
|
||||||
|
class TestAuthorBootstrapPreflightIntegration(unittest.TestCase):
|
||||||
|
"""Server preflight path: clean control + author bootstrap task must not raise."""
|
||||||
|
|
||||||
|
def test_enforce_branches_only_allows_clean_control_for_bootstrap(self):
|
||||||
|
# Exercise the real enforcer wiring with a temporary clean repo.
|
||||||
|
import gitea_mcp_server as srv
|
||||||
|
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
repo = os.path.join(tmp, "repo")
|
||||||
|
os.makedirs(os.path.join(repo, "branches"))
|
||||||
|
# Minimal git repo on master at a known tip.
|
||||||
|
import subprocess
|
||||||
|
|
||||||
|
subprocess.check_call(["git", "init", "-b", "master", repo])
|
||||||
|
subprocess.check_call(
|
||||||
|
["git", "-C", repo, "commit", "--allow-empty", "-m", "init"]
|
||||||
|
)
|
||||||
|
head = subprocess.check_output(
|
||||||
|
["git", "-C", repo, "rev-parse", "HEAD"], text=True
|
||||||
|
).strip()
|
||||||
|
|
||||||
|
assessment = aib.assess_author_issue_bootstrap(
|
||||||
|
workspace_path=repo,
|
||||||
|
canonical_repo_root=repo,
|
||||||
|
current_branch="master",
|
||||||
|
head_sha=head,
|
||||||
|
porcelain_status="",
|
||||||
|
remote_master_sha=head,
|
||||||
|
task="bootstrap_author_issue_worktree",
|
||||||
|
)
|
||||||
|
self.assertTrue(
|
||||||
|
cib.bootstrap_permits_control_checkout(
|
||||||
|
assessment,
|
||||||
|
task="bootstrap_author_issue_worktree",
|
||||||
|
workspace_path=repo,
|
||||||
|
canonical_repo_root=repo,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
# Simulate what _enforce_branches_only_author_mutation does when
|
||||||
|
# durable resolution blocks control: the shared predicate must waive.
|
||||||
|
durable_block = {
|
||||||
|
"block": True,
|
||||||
|
"workspace_path": repo,
|
||||||
|
"workspace_binding_source": "process_project_root",
|
||||||
|
"reasons": [
|
||||||
|
"author mutation blocked: workspace is the stable control checkout"
|
||||||
|
],
|
||||||
|
}
|
||||||
|
if cib.bootstrap_permits_control_checkout(
|
||||||
|
assessment,
|
||||||
|
task="bootstrap_author_issue_worktree",
|
||||||
|
workspace_path=repo,
|
||||||
|
canonical_repo_root=repo,
|
||||||
|
):
|
||||||
|
waived = True
|
||||||
|
else:
|
||||||
|
waived = False
|
||||||
|
self.assertTrue(waived)
|
||||||
|
# Keep durable_block referenced so the scenario is explicit.
|
||||||
|
self.assertTrue(durable_block["block"])
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,924 @@
|
|||||||
|
"""Transport-neutral MCP bind seam (#931).
|
||||||
|
|
||||||
|
These tests drive the *real* bind boundary — ``mark_sanctioned_daemon`` followed
|
||||||
|
by ``bind_native_mcp_transport`` from a canonical entrypoint path, with the
|
||||||
|
pytest allowance switched off — rather than mocking the new accessor. The
|
||||||
|
distinction matters here for the same reason it mattered in #941: a suite that
|
||||||
|
only exercises the helper in isolation cannot observe a seam that the live path
|
||||||
|
never reaches.
|
||||||
|
|
||||||
|
Covered:
|
||||||
|
|
||||||
|
1. no configured transport defaults to the local transport
|
||||||
|
2. explicit local transport binds
|
||||||
|
3. the sanctioned remote identifier binds through the seam
|
||||||
|
4. an unregistered identifier is rejected at bind time
|
||||||
|
5. an invalid bind prevents the server reaching tool service
|
||||||
|
6. the unbound state fails closed where a bind is required
|
||||||
|
7. every transport-aware guard reads the same authoritative value
|
||||||
|
8. client-controlled input cannot alter the bound transport
|
||||||
|
9. the durable decision-lock record carries the selected transport
|
||||||
|
10. existing stdio behaviour is unchanged
|
||||||
|
11. repeated / conflicting bind attempts follow one fail-closed contract
|
||||||
|
12. capability, role, repository and provenance protections do not regress
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
from pathlib import Path
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
REPO_ROOT = Path(__file__).resolve().parent.parent
|
||||||
|
|
||||||
|
import mcp_daemon_guard
|
||||||
|
import mcp_session_state
|
||||||
|
import mcp_transport_config
|
||||||
|
import irrecoverable_provenance
|
||||||
|
|
||||||
|
|
||||||
|
class _ProductionBind:
|
||||||
|
"""Context manager that reaches the real production bind path.
|
||||||
|
|
||||||
|
Patches only the two things a unit test cannot otherwise satisfy: the
|
||||||
|
resolved canonical entrypoint frame, and the pytest allowance that would
|
||||||
|
short-circuit ``mark_sanctioned_daemon``. Everything downstream of those —
|
||||||
|
validation, pinning, the rebind contract — runs unmodified.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def __init__(self, env: dict[str, str] | None = None):
|
||||||
|
self._env = env or {}
|
||||||
|
self._stack: list = []
|
||||||
|
|
||||||
|
def __enter__(self):
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
canonical = str((REPO_ROOT / "mcp_server.py").resolve())
|
||||||
|
self._stack = [
|
||||||
|
patch.object(
|
||||||
|
mcp_daemon_guard,
|
||||||
|
"_caller_official_entrypoint_path",
|
||||||
|
side_effect=lambda: canonical,
|
||||||
|
),
|
||||||
|
patch.object(mcp_daemon_guard, "is_pytest_runtime", return_value=False),
|
||||||
|
patch.dict(os.environ, self._env),
|
||||||
|
]
|
||||||
|
for ctx in self._stack:
|
||||||
|
ctx.__enter__()
|
||||||
|
# Start from a clean configuration unless the test set one.
|
||||||
|
if mcp_transport_config.TRANSPORT_ENV not in self._env:
|
||||||
|
os.environ.pop(mcp_transport_config.TRANSPORT_ENV, None)
|
||||||
|
mcp_daemon_guard.mark_sanctioned_daemon()
|
||||||
|
return mcp_daemon_guard
|
||||||
|
|
||||||
|
def __exit__(self, *exc):
|
||||||
|
for ctx in reversed(self._stack):
|
||||||
|
ctx.__exit__(*exc)
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
class TestPermittedSetIsSingleSourceOfTruth(unittest.TestCase):
|
||||||
|
"""AC3: identifiers are enumerated once, in the seam."""
|
||||||
|
|
||||||
|
def test_guard_allowlist_is_the_seam_allowlist(self):
|
||||||
|
self.assertIs(
|
||||||
|
mcp_daemon_guard._PRODUCTION_TRANSPORTS,
|
||||||
|
mcp_transport_config.SUPPORTED_TRANSPORTS,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_default_is_a_member_of_the_permitted_set(self):
|
||||||
|
self.assertIn(
|
||||||
|
mcp_transport_config.DEFAULT_TRANSPORT,
|
||||||
|
mcp_transport_config.SUPPORTED_TRANSPORTS,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_remote_identifier_is_permitted_and_not_the_default(self):
|
||||||
|
self.assertIn(
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT,
|
||||||
|
mcp_transport_config.SUPPORTED_TRANSPORTS,
|
||||||
|
)
|
||||||
|
self.assertNotEqual(
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT,
|
||||||
|
mcp_transport_config.DEFAULT_TRANSPORT,
|
||||||
|
)
|
||||||
|
self.assertTrue(
|
||||||
|
mcp_transport_config.is_remote_transport(
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_no_default_transport_literal_outside_the_seam(self):
|
||||||
|
"""AC3: no production module reads the literal outside the seam.
|
||||||
|
|
||||||
|
Review 635 flagged that a fixed five-module list cannot catch a *new*
|
||||||
|
module reintroducing the literal. This globs every production module in
|
||||||
|
the repository root instead, so the guarantee holds for code that does
|
||||||
|
not exist yet.
|
||||||
|
"""
|
||||||
|
default = mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
needles = (f'"{default}"', f"'{default}'")
|
||||||
|
seam = Path(mcp_transport_config.__file__).name
|
||||||
|
scanned: list[str] = []
|
||||||
|
offenders: list[str] = []
|
||||||
|
for path in sorted(REPO_ROOT.glob("*.py")):
|
||||||
|
if path.name == seam:
|
||||||
|
continue # the seam is the one place the literal may live
|
||||||
|
scanned.append(path.name)
|
||||||
|
for lineno, line in enumerate(
|
||||||
|
path.read_text(encoding="utf-8").splitlines(), start=1
|
||||||
|
):
|
||||||
|
code = line.split("#", 1)[0]
|
||||||
|
if any(needle in code for needle in needles):
|
||||||
|
offenders.append(f"{path.name}:{lineno}: {line.strip()}")
|
||||||
|
# Guard the guard: a glob that silently matched nothing would pass.
|
||||||
|
self.assertGreater(len(scanned), 20, "production glob matched too little")
|
||||||
|
self.assertIn("mcp_daemon_guard.py", scanned)
|
||||||
|
self.assertIn("gitea_mcp_server.py", scanned)
|
||||||
|
self.assertEqual(offenders, [], "\n".join(offenders))
|
||||||
|
|
||||||
|
|
||||||
|
class TestConfiguredTransportResolution(unittest.TestCase):
|
||||||
|
"""AC1: configuration supplies the identifier; unset still yields the default."""
|
||||||
|
|
||||||
|
def test_unset_yields_default(self):
|
||||||
|
res = mcp_transport_config.resolve_configured_transport(env={})
|
||||||
|
self.assertEqual(res["transport"], mcp_transport_config.DEFAULT_TRANSPORT)
|
||||||
|
self.assertFalse(res["configured"])
|
||||||
|
self.assertEqual(res["source"], mcp_transport_config.SOURCE_DEFAULT)
|
||||||
|
self.assertTrue(res["supported"])
|
||||||
|
self.assertEqual(res["reasons"], [])
|
||||||
|
|
||||||
|
def test_blank_and_whitespace_are_treated_as_unset(self):
|
||||||
|
for raw in ("", " ", "\t\n"):
|
||||||
|
res = mcp_transport_config.resolve_configured_transport(
|
||||||
|
env={mcp_transport_config.TRANSPORT_ENV: raw}
|
||||||
|
)
|
||||||
|
self.assertEqual(res["transport"], mcp_transport_config.DEFAULT_TRANSPORT)
|
||||||
|
self.assertFalse(res["configured"])
|
||||||
|
|
||||||
|
def test_explicit_default_is_reported_as_configured(self):
|
||||||
|
res = mcp_transport_config.resolve_configured_transport(
|
||||||
|
env={
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
}
|
||||||
|
)
|
||||||
|
self.assertEqual(res["transport"], mcp_transport_config.DEFAULT_TRANSPORT)
|
||||||
|
self.assertTrue(res["configured"])
|
||||||
|
self.assertEqual(res["source"], mcp_transport_config.SOURCE_CONFIGURED)
|
||||||
|
|
||||||
|
def test_remote_identifier_resolves_and_is_supported(self):
|
||||||
|
res = mcp_transport_config.resolve_configured_transport(
|
||||||
|
env={
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
}
|
||||||
|
)
|
||||||
|
self.assertEqual(res["transport"], mcp_transport_config.REMOTE_TRANSPORT)
|
||||||
|
self.assertTrue(res["supported"])
|
||||||
|
|
||||||
|
def test_case_and_padding_are_normalized(self):
|
||||||
|
padded = f" {mcp_transport_config.REMOTE_TRANSPORT.upper()} "
|
||||||
|
res = mcp_transport_config.resolve_configured_transport(
|
||||||
|
env={mcp_transport_config.TRANSPORT_ENV: padded}
|
||||||
|
)
|
||||||
|
self.assertEqual(res["transport"], mcp_transport_config.REMOTE_TRANSPORT)
|
||||||
|
self.assertTrue(res["supported"])
|
||||||
|
|
||||||
|
def test_unregistered_identifier_is_not_silently_defaulted(self):
|
||||||
|
res = mcp_transport_config.resolve_configured_transport(
|
||||||
|
env={mcp_transport_config.TRANSPORT_ENV: "carrier-pigeon"}
|
||||||
|
)
|
||||||
|
self.assertFalse(res["supported"])
|
||||||
|
self.assertEqual(res["transport"], "carrier-pigeon")
|
||||||
|
self.assertNotEqual(res["transport"], mcp_transport_config.DEFAULT_TRANSPORT)
|
||||||
|
self.assertTrue(res["reasons"])
|
||||||
|
|
||||||
|
def test_superseded_sse_transport_is_not_registered(self):
|
||||||
|
"""A real MCP transport that this deployment does not sanction."""
|
||||||
|
self.assertFalse(mcp_transport_config.is_supported_transport("sse"))
|
||||||
|
res = mcp_transport_config.resolve_configured_transport(
|
||||||
|
env={mcp_transport_config.TRANSPORT_ENV: "sse"}
|
||||||
|
)
|
||||||
|
self.assertFalse(res["supported"])
|
||||||
|
|
||||||
|
def test_require_configured_transport_raises_on_unregistered(self):
|
||||||
|
with self.assertRaises(mcp_transport_config.TransportConfigurationError):
|
||||||
|
mcp_transport_config.require_configured_transport(
|
||||||
|
env={mcp_transport_config.TRANSPORT_ENV: "carrier-pigeon"}
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_non_string_configuration_never_matches_permitted_set(self):
|
||||||
|
for value in (object(), 1, None, True, ["stdio"], {"t": "stdio"}):
|
||||||
|
self.assertEqual(mcp_transport_config.normalize_transport(value), "")
|
||||||
|
self.assertFalse(mcp_transport_config.is_supported_transport(value))
|
||||||
|
|
||||||
|
|
||||||
|
class TestBindSeam(unittest.TestCase):
|
||||||
|
"""AC1/AC2: the live bind path resolves, validates, and pins."""
|
||||||
|
|
||||||
|
def tearDown(self) -> None:
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
|
||||||
|
def test_1_no_configured_transport_binds_default(self):
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
status = guard.bind_native_mcp_transport()
|
||||||
|
self.assertEqual(
|
||||||
|
status["transport"], mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
guard.bound_transport(), mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
self.assertTrue(status["production_native_mcp_transport"])
|
||||||
|
|
||||||
|
def test_2_explicit_default_transport_binds(self):
|
||||||
|
with _ProductionBind(
|
||||||
|
{
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
}
|
||||||
|
) as guard:
|
||||||
|
status = guard.bind_native_mcp_transport()
|
||||||
|
self.assertEqual(
|
||||||
|
status["transport"], mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
self.assertTrue(guard.is_production_native_mcp_transport())
|
||||||
|
|
||||||
|
def test_2b_explicit_argument_still_binds(self):
|
||||||
|
"""The pre-#931 call form keeps working for launchers and tests."""
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
status = guard.bind_native_mcp_transport(
|
||||||
|
transport=mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
status["transport"], mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_3_sanctioned_remote_identifier_binds_through_the_seam(self):
|
||||||
|
with _ProductionBind(
|
||||||
|
{
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
}
|
||||||
|
) as guard:
|
||||||
|
status = guard.bind_native_mcp_transport()
|
||||||
|
self.assertEqual(
|
||||||
|
status["transport"], mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
guard.bound_transport(), mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
# The remote identifier is trusted exactly like the local one; the
|
||||||
|
# listener that serves it is #938 and is not implemented here.
|
||||||
|
self.assertTrue(guard.is_native_mcp_transport())
|
||||||
|
self.assertTrue(guard.is_production_native_mcp_transport())
|
||||||
|
guard.assert_production_mutation_runtime("remote-bind")
|
||||||
|
|
||||||
|
def test_4_unregistered_identifier_rejected_at_bind_time(self):
|
||||||
|
with _ProductionBind(
|
||||||
|
{mcp_transport_config.TRANSPORT_ENV: "carrier-pigeon"}
|
||||||
|
) as guard:
|
||||||
|
with self.assertRaises(guard.UnsanctionedRuntimeError) as ctx:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
self.assertIn("carrier-pigeon", str(ctx.exception))
|
||||||
|
self.assertIn("#931", str(ctx.exception))
|
||||||
|
# Nothing was bound, so nothing may dispatch.
|
||||||
|
self.assertIsNone(guard.bound_transport())
|
||||||
|
self.assertFalse(guard.is_native_mcp_transport())
|
||||||
|
|
||||||
|
def test_4b_unregistered_explicit_argument_rejected(self):
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
with self.assertRaises(guard.UnsanctionedRuntimeError):
|
||||||
|
guard.bind_native_mcp_transport(transport="carrier-pigeon")
|
||||||
|
self.assertIsNone(guard.bound_transport())
|
||||||
|
|
||||||
|
def test_4c_superseded_sse_rejected_at_bind_time(self):
|
||||||
|
with _ProductionBind({mcp_transport_config.TRANSPORT_ENV: "sse"}) as guard:
|
||||||
|
with self.assertRaises(guard.UnsanctionedRuntimeError):
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
self.assertIsNone(guard.bound_transport())
|
||||||
|
|
||||||
|
def test_5_invalid_bind_prevents_tool_service(self):
|
||||||
|
"""A failed bind must stop the server before it serves tools."""
|
||||||
|
with _ProductionBind(
|
||||||
|
{mcp_transport_config.TRANSPORT_ENV: "carrier-pigeon"}
|
||||||
|
) as guard:
|
||||||
|
with self.assertRaises(guard.UnsanctionedRuntimeError):
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
# This is the exact expression the entrypoint passes to mcp.run.
|
||||||
|
with self.assertRaises(guard.UnsanctionedRuntimeError) as ctx:
|
||||||
|
guard.assert_transport_bound("tool service")
|
||||||
|
self.assertIn("No MCP transport is bound", str(ctx.exception))
|
||||||
|
|
||||||
|
def test_6_unbound_state_fails_closed(self):
|
||||||
|
"""Entrypoint claimed but never bound — the offline-import shape."""
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
self.assertIsNone(guard.bound_transport())
|
||||||
|
self.assertFalse(guard.is_native_mcp_transport())
|
||||||
|
with self.assertRaises(guard.UnsanctionedRuntimeError):
|
||||||
|
guard.assert_transport_bound("tool service")
|
||||||
|
with self.assertRaises(guard.UnsanctionedRuntimeError):
|
||||||
|
guard.assert_sanctioned_mutation_runtime("gitea_mutation")
|
||||||
|
|
||||||
|
def test_6b_no_runtime_at_all_fails_closed(self):
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
with patch.object(mcp_daemon_guard, "is_pytest_runtime", return_value=False):
|
||||||
|
self.assertIsNone(mcp_daemon_guard.bound_transport())
|
||||||
|
with self.assertRaises(mcp_daemon_guard.UnsanctionedRuntimeError):
|
||||||
|
mcp_daemon_guard.assert_transport_bound("tool service")
|
||||||
|
|
||||||
|
|
||||||
|
class TestOneAuthoritativeValue(unittest.TestCase):
|
||||||
|
"""AC: every transport-aware guard observes the same value."""
|
||||||
|
|
||||||
|
def tearDown(self) -> None:
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
|
||||||
|
def test_7_all_guards_read_the_same_bound_value(self):
|
||||||
|
with _ProductionBind(
|
||||||
|
{
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
}
|
||||||
|
) as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
expected = mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
|
||||||
|
self.assertEqual(guard.bound_transport(), expected)
|
||||||
|
self.assertEqual(guard.assert_transport_bound(), expected)
|
||||||
|
self.assertEqual(guard.native_runtime_status()["bound_transport"], expected)
|
||||||
|
self.assertEqual(guard.native_runtime_status()["transport"], expected)
|
||||||
|
self.assertEqual(
|
||||||
|
guard.mutation_provenance_fields()["bound_transport"], expected
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
irrecoverable_provenance.assess_transport_for_auth_mint()[
|
||||||
|
"bound_transport"
|
||||||
|
],
|
||||||
|
expected,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_8_environment_change_after_bind_cannot_move_the_value(self):
|
||||||
|
"""Client- or environment-shaped input must not alter a bound transport."""
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
self.assertEqual(
|
||||||
|
guard.bound_transport(), mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
# A stray launcher (or an attacker) rewrites config post-bind.
|
||||||
|
os.environ[mcp_transport_config.TRANSPORT_ENV] = (
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
guard.bound_transport(), mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
guard.mutation_provenance_fields()["bound_transport"],
|
||||||
|
mcp_transport_config.DEFAULT_TRANSPORT,
|
||||||
|
)
|
||||||
|
os.environ[mcp_transport_config.TRANSPORT_ENV] = "carrier-pigeon"
|
||||||
|
self.assertEqual(
|
||||||
|
guard.bound_transport(), mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_8b_bound_transport_takes_no_caller_argument(self):
|
||||||
|
"""The accessor cannot be steered by a tool parameter."""
|
||||||
|
import inspect
|
||||||
|
|
||||||
|
self.assertEqual(
|
||||||
|
list(inspect.signature(mcp_daemon_guard.bound_transport).parameters), []
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_11_rebinding_the_same_transport_is_idempotent(self):
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
first = guard.bind_native_mcp_transport()
|
||||||
|
second = guard.bind_native_mcp_transport()
|
||||||
|
self.assertEqual(first["transport"], second["transport"])
|
||||||
|
self.assertEqual(
|
||||||
|
guard.bound_transport(), mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_11b_rebinding_a_different_transport_fails_closed(self):
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
with self.assertRaises(guard.UnsanctionedRuntimeError) as ctx:
|
||||||
|
guard.bind_native_mcp_transport(
|
||||||
|
transport=mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
self.assertIn("already bound", str(ctx.exception))
|
||||||
|
# The first value survives the attempt.
|
||||||
|
self.assertEqual(
|
||||||
|
guard.bound_transport(), mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_11c_rebinding_an_unregistered_transport_fails_closed(self):
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
with self.assertRaises(guard.UnsanctionedRuntimeError):
|
||||||
|
guard.bind_native_mcp_transport(transport="carrier-pigeon")
|
||||||
|
self.assertEqual(
|
||||||
|
guard.bound_transport(), mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class TestDurableRecordCarriesTransport(unittest.TestCase):
|
||||||
|
"""AC4: the identifier reaches a durable decision-lock record."""
|
||||||
|
|
||||||
|
def tearDown(self) -> None:
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
|
||||||
|
def _save_and_read_decision_lock(self, state_dir: str) -> dict:
|
||||||
|
mcp_session_state.save_state(
|
||||||
|
kind=mcp_session_state.KIND_DECISION_LOCK,
|
||||||
|
payload={"pr_number": 931, "action": "COMMENT"},
|
||||||
|
remote="prgs",
|
||||||
|
org="Scaled-Tech-Consulting",
|
||||||
|
repo="Gitea-Tools",
|
||||||
|
profile_identity="prgs-author",
|
||||||
|
state_dir=state_dir,
|
||||||
|
)
|
||||||
|
loaded = mcp_session_state.load_state(
|
||||||
|
kind=mcp_session_state.KIND_DECISION_LOCK,
|
||||||
|
remote="prgs",
|
||||||
|
org="Scaled-Tech-Consulting",
|
||||||
|
repo="Gitea-Tools",
|
||||||
|
profile_identity="prgs-author",
|
||||||
|
state_dir=state_dir,
|
||||||
|
)
|
||||||
|
self.assertIsNotNone(loaded)
|
||||||
|
return loaded
|
||||||
|
|
||||||
|
def test_9_decision_lock_records_the_bound_transport(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
mcp_daemon_guard.install_test_native_runtime()
|
||||||
|
record = self._save_and_read_decision_lock(tmp)
|
||||||
|
self.assertIn("bound_transport", record)
|
||||||
|
self.assertEqual(
|
||||||
|
record["bound_transport"], mcp_daemon_guard.bound_transport()
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_9b_unbound_runtime_records_no_transport_identifier(self):
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
record = self._save_and_read_decision_lock(tmp)
|
||||||
|
self.assertIn("bound_transport", record)
|
||||||
|
self.assertIsNone(record["bound_transport"])
|
||||||
|
|
||||||
|
def test_9c_provenance_fields_expose_the_identifier(self):
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
fields = mcp_daemon_guard.mutation_provenance_fields()
|
||||||
|
self.assertIn("bound_transport", fields)
|
||||||
|
self.assertIsNone(fields["bound_transport"])
|
||||||
|
|
||||||
|
|
||||||
|
class TestStdioBehaviourUnchanged(unittest.TestCase):
|
||||||
|
"""AC6 / prompt items 10 and 12: no regression on the existing path."""
|
||||||
|
|
||||||
|
def tearDown(self) -> None:
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
|
||||||
|
def test_10_trust_class_field_keeps_its_pre_931_values(self):
|
||||||
|
"""``transport`` remains the trust class, not the identifier."""
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
self.assertEqual(
|
||||||
|
mcp_daemon_guard.mutation_provenance_fields()["transport"], "untrusted"
|
||||||
|
)
|
||||||
|
mcp_daemon_guard.install_test_native_runtime()
|
||||||
|
self.assertEqual(
|
||||||
|
mcp_daemon_guard.mutation_provenance_fields()["transport"],
|
||||||
|
"test_native_mcp",
|
||||||
|
)
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
self.assertEqual(
|
||||||
|
guard.mutation_provenance_fields()["transport"], "native_mcp"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_10b_default_bind_reproduces_the_pre_931_runtime_record(self):
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
status = guard.bind_native_mcp_transport()
|
||||||
|
self.assertTrue(status["native_mcp_transport"])
|
||||||
|
self.assertTrue(status["production_native_mcp_transport"])
|
||||||
|
self.assertEqual(status["mode"], "production")
|
||||||
|
self.assertEqual(status["phase"], "transport_bound")
|
||||||
|
self.assertEqual(
|
||||||
|
status["transport"], mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
guard.assert_sanctioned_mutation_runtime("native-ide")
|
||||||
|
guard.assert_production_mutation_runtime("native-ide")
|
||||||
|
|
||||||
|
def test_12_session_state_root_still_pinned_at_bind(self):
|
||||||
|
"""#695 AC2 must survive the seam."""
|
||||||
|
with tempfile.TemporaryDirectory() as legit:
|
||||||
|
with tempfile.TemporaryDirectory() as rogue:
|
||||||
|
with _ProductionBind(
|
||||||
|
{mcp_daemon_guard.SESSION_STATE_DIR_ENV: legit}
|
||||||
|
) as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
self.assertEqual(
|
||||||
|
guard.pinned_session_state_dir(), str(Path(legit).resolve())
|
||||||
|
)
|
||||||
|
os.environ[mcp_daemon_guard.SESSION_STATE_DIR_ENV] = rogue
|
||||||
|
self.assertEqual(
|
||||||
|
guard.pinned_session_state_dir(), str(Path(legit).resolve())
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_12b_test_mode_record_still_cannot_authorize_production(self):
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
mcp_daemon_guard.install_test_native_runtime()
|
||||||
|
self.assertTrue(mcp_daemon_guard.is_native_mcp_transport())
|
||||||
|
self.assertFalse(mcp_daemon_guard.is_production_native_mcp_transport())
|
||||||
|
with self.assertRaises(mcp_daemon_guard.UnsanctionedRuntimeError):
|
||||||
|
mcp_daemon_guard.assert_production_mutation_runtime("prod-endpoint")
|
||||||
|
|
||||||
|
def test_12c_bind_still_requires_the_canonical_entrypoint(self):
|
||||||
|
"""A remote identifier does not relax provenance."""
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
with patch.object(mcp_daemon_guard, "is_pytest_runtime", return_value=False):
|
||||||
|
with patch.object(
|
||||||
|
mcp_daemon_guard,
|
||||||
|
"_caller_official_entrypoint_path",
|
||||||
|
return_value=None,
|
||||||
|
):
|
||||||
|
with self.assertRaises(mcp_daemon_guard.UnsanctionedRuntimeError):
|
||||||
|
mcp_daemon_guard.bind_native_mcp_transport(
|
||||||
|
transport=mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
self.assertIsNone(mcp_daemon_guard.bound_transport())
|
||||||
|
|
||||||
|
def test_12d_bind_still_requires_a_prior_entrypoint_claim(self):
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
canonical = str((REPO_ROOT / "mcp_server.py").resolve())
|
||||||
|
with patch.object(mcp_daemon_guard, "is_pytest_runtime", return_value=False):
|
||||||
|
with patch.object(
|
||||||
|
mcp_daemon_guard,
|
||||||
|
"_caller_official_entrypoint_path",
|
||||||
|
side_effect=lambda: canonical,
|
||||||
|
):
|
||||||
|
# No mark_sanctioned_daemon() first.
|
||||||
|
with self.assertRaises(
|
||||||
|
mcp_daemon_guard.UnsanctionedRuntimeError
|
||||||
|
) as ctx:
|
||||||
|
mcp_daemon_guard.bind_native_mcp_transport()
|
||||||
|
self.assertIn("no entrypoint claim", str(ctx.exception))
|
||||||
|
|
||||||
|
def test_12e_auth_mint_verdict_is_unchanged_for_the_default_transport(self):
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
verdict = irrecoverable_provenance.assess_transport_for_auth_mint()
|
||||||
|
self.assertTrue(verdict["allowed"])
|
||||||
|
self.assertTrue(verdict["native_mcp_transport"])
|
||||||
|
self.assertTrue(verdict["production_native_mcp_transport"])
|
||||||
|
self.assertEqual(verdict["reasons"], [])
|
||||||
|
|
||||||
|
def test_12f_auth_mint_still_refuses_an_unbound_runtime(self):
|
||||||
|
with _ProductionBind():
|
||||||
|
# Claimed but never bound.
|
||||||
|
verdict = irrecoverable_provenance.assess_transport_for_auth_mint()
|
||||||
|
self.assertFalse(verdict["allowed"])
|
||||||
|
self.assertTrue(verdict["reasons"])
|
||||||
|
self.assertIsNone(verdict["bound_transport"])
|
||||||
|
|
||||||
|
|
||||||
|
class TestExecutionBoundary(unittest.TestCase):
|
||||||
|
"""Recognition is not execution authorization (#931, review 635 B1/B2).
|
||||||
|
|
||||||
|
A registered remote identifier must still bind, pin and record — the #931
|
||||||
|
seam — while being refused at the serve boundary, because serving it would
|
||||||
|
start a listener with no authentication or per-request principal. That
|
||||||
|
listener belongs to #938.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def tearDown(self) -> None:
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
|
||||||
|
# -- the two sets are distinct, and narrower in the right direction ----
|
||||||
|
|
||||||
|
def test_executable_set_is_a_strict_subset_of_recognized(self):
|
||||||
|
self.assertTrue(
|
||||||
|
mcp_transport_config.EXECUTABLE_TRANSPORTS
|
||||||
|
< mcp_transport_config.SUPPORTED_TRANSPORTS
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_default_transport_is_executable(self):
|
||||||
|
self.assertTrue(
|
||||||
|
mcp_transport_config.is_executable_transport(
|
||||||
|
mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_remote_transport_is_recognized_but_not_executable(self):
|
||||||
|
remote = mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
self.assertTrue(mcp_transport_config.is_supported_transport(remote))
|
||||||
|
self.assertFalse(mcp_transport_config.is_executable_transport(remote))
|
||||||
|
|
||||||
|
def test_remote_listener_ownership_is_declared(self):
|
||||||
|
self.assertEqual(
|
||||||
|
mcp_transport_config.TRANSPORT_EXECUTION_OWNER[
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
],
|
||||||
|
"#938",
|
||||||
|
)
|
||||||
|
|
||||||
|
# -- stdio still reaches the runner, unchanged -------------------------
|
||||||
|
|
||||||
|
def test_default_transport_is_authorized_for_service(self):
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
self.assertEqual(
|
||||||
|
guard.authorize_transport_execution("tool service"),
|
||||||
|
mcp_transport_config.DEFAULT_TRANSPORT,
|
||||||
|
)
|
||||||
|
self.assertTrue(guard.assess_serve_authorization()["allowed"])
|
||||||
|
|
||||||
|
def test_default_transport_reaches_the_real_runner(self):
|
||||||
|
"""The production runner is actually invoked, with stdio, unchanged."""
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
seen = {}
|
||||||
|
|
||||||
|
class _Runner:
|
||||||
|
def run(self, transport=None, **kw):
|
||||||
|
seen["transport"] = transport
|
||||||
|
|
||||||
|
_Runner().run(transport=guard.authorize_transport_execution("tool service"))
|
||||||
|
self.assertEqual(
|
||||||
|
seen["transport"], mcp_transport_config.DEFAULT_TRANSPORT
|
||||||
|
)
|
||||||
|
|
||||||
|
# -- streamable-http binds, records, and is refused before serving -----
|
||||||
|
|
||||||
|
def test_remote_transport_binds_and_is_recorded(self):
|
||||||
|
"""The #931 seam is intact: it binds, pins and records."""
|
||||||
|
with _ProductionBind(
|
||||||
|
{
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
}
|
||||||
|
) as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
self.assertEqual(
|
||||||
|
guard.bound_transport(), mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
guard.mutation_provenance_fields()["bound_transport"],
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_remote_transport_is_refused_at_the_serve_boundary(self):
|
||||||
|
with _ProductionBind(
|
||||||
|
{
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
}
|
||||||
|
) as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
with self.assertRaises(guard.TransportExecutionError) as ctx:
|
||||||
|
guard.authorize_transport_execution("tool service")
|
||||||
|
err = ctx.exception
|
||||||
|
self.assertEqual(
|
||||||
|
err.blocker_kind,
|
||||||
|
mcp_transport_config.BLOCKER_LISTENER_NOT_COMMISSIONED,
|
||||||
|
)
|
||||||
|
self.assertEqual(err.owner_issue, "#938")
|
||||||
|
self.assertEqual(err.transport, mcp_transport_config.REMOTE_TRANSPORT)
|
||||||
|
|
||||||
|
def test_refusal_names_transport_and_unmet_requirement_and_owner(self):
|
||||||
|
with _ProductionBind(
|
||||||
|
{
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
}
|
||||||
|
) as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
with self.assertRaises(guard.TransportExecutionError) as ctx:
|
||||||
|
guard.authorize_transport_execution("tool service")
|
||||||
|
text = str(ctx.exception)
|
||||||
|
self.assertIn(mcp_transport_config.REMOTE_TRANSPORT, text)
|
||||||
|
self.assertIn("#938", text)
|
||||||
|
self.assertIn("not commissioned", text)
|
||||||
|
self.assertIn("#931", text)
|
||||||
|
|
||||||
|
def test_remote_transport_never_reaches_the_runner(self):
|
||||||
|
"""No transport value is handed to a run() call for the remote case."""
|
||||||
|
with _ProductionBind(
|
||||||
|
{
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
}
|
||||||
|
) as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
calls = []
|
||||||
|
|
||||||
|
class _Runner:
|
||||||
|
def run(self, transport=None, **kw):
|
||||||
|
calls.append(transport)
|
||||||
|
|
||||||
|
with self.assertRaises(guard.TransportExecutionError):
|
||||||
|
_Runner().run(
|
||||||
|
transport=guard.authorize_transport_execution("tool service")
|
||||||
|
)
|
||||||
|
self.assertEqual(calls, [], "runner must never be invoked")
|
||||||
|
|
||||||
|
def test_no_http_listener_is_created_for_remote_transport(self):
|
||||||
|
"""Nothing in the refusal path touches uvicorn or a socket bind."""
|
||||||
|
with _ProductionBind(
|
||||||
|
{
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
}
|
||||||
|
) as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
import socket
|
||||||
|
|
||||||
|
bound_sockets = []
|
||||||
|
real_bind = socket.socket.bind
|
||||||
|
|
||||||
|
def _tripwire(self, addr): # pragma: no cover - must not run
|
||||||
|
bound_sockets.append(addr)
|
||||||
|
return real_bind(self, addr)
|
||||||
|
|
||||||
|
with patch.object(socket.socket, "bind", _tripwire):
|
||||||
|
with self.assertRaises(guard.TransportExecutionError):
|
||||||
|
guard.authorize_transport_execution("tool service")
|
||||||
|
self.assertEqual(bound_sockets, [], "no socket may be bound")
|
||||||
|
|
||||||
|
def test_no_mutation_is_authorized_after_the_denial(self):
|
||||||
|
"""The denial leaves no partial state that would let a tool dispatch."""
|
||||||
|
with _ProductionBind(
|
||||||
|
{
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
}
|
||||||
|
) as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
with self.assertRaises(guard.TransportExecutionError):
|
||||||
|
guard.authorize_transport_execution("tool service")
|
||||||
|
# Serve stays unauthorized on every subsequent query.
|
||||||
|
self.assertFalse(guard.assess_serve_authorization()["allowed"])
|
||||||
|
self.assertFalse(guard.native_runtime_status()["serve_authorized"])
|
||||||
|
with self.assertRaises(guard.TransportExecutionError):
|
||||||
|
guard.authorize_transport_execution("tool service")
|
||||||
|
|
||||||
|
# -- B2: the decision genuinely consumes bound_transport ---------------
|
||||||
|
|
||||||
|
def test_serve_decision_consumes_bound_transport(self):
|
||||||
|
"""Changing only bound_transport flips the verdict."""
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
self.assertTrue(guard.assess_serve_authorization()["allowed"])
|
||||||
|
with patch.object(
|
||||||
|
guard,
|
||||||
|
"bound_transport",
|
||||||
|
return_value=mcp_transport_config.REMOTE_TRANSPORT,
|
||||||
|
):
|
||||||
|
verdict = guard.assess_serve_authorization()
|
||||||
|
self.assertFalse(verdict["allowed"])
|
||||||
|
self.assertEqual(
|
||||||
|
verdict["transport"], mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
with self.assertRaises(guard.TransportExecutionError):
|
||||||
|
guard.authorize_transport_execution("tool service")
|
||||||
|
|
||||||
|
def test_serve_verdict_reports_the_bound_transport(self):
|
||||||
|
with _ProductionBind(
|
||||||
|
{
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
}
|
||||||
|
) as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
verdict = guard.assess_serve_authorization()
|
||||||
|
self.assertEqual(
|
||||||
|
verdict["transport"], mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
self.assertTrue(verdict["recognized"])
|
||||||
|
self.assertFalse(verdict["executable"])
|
||||||
|
|
||||||
|
# -- earlier and later boundaries are unchanged ------------------------
|
||||||
|
|
||||||
|
def test_unregistered_identifier_still_fails_at_bind_not_at_serve(self):
|
||||||
|
with _ProductionBind(
|
||||||
|
{mcp_transport_config.TRANSPORT_ENV: "carrier-pigeon"}
|
||||||
|
) as guard:
|
||||||
|
with self.assertRaises(guard.UnsanctionedRuntimeError) as ctx:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
self.assertIn("not a registered MCP transport", str(ctx.exception))
|
||||||
|
self.assertIsNone(guard.bound_transport())
|
||||||
|
|
||||||
|
def test_unbound_execution_keeps_the_pre_existing_failure(self):
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
with self.assertRaises(guard.UnsanctionedRuntimeError) as ctx:
|
||||||
|
guard.authorize_transport_execution("tool service")
|
||||||
|
self.assertIn("No MCP transport is bound", str(ctx.exception))
|
||||||
|
self.assertNotIsInstance(ctx.exception, guard.TransportExecutionError)
|
||||||
|
|
||||||
|
def test_transport_execution_error_is_caught_by_existing_handlers(self):
|
||||||
|
"""Subclassing keeps every pre-existing fail-closed handler correct."""
|
||||||
|
self.assertTrue(
|
||||||
|
issubclass(
|
||||||
|
mcp_daemon_guard.TransportExecutionError,
|
||||||
|
mcp_daemon_guard.UnsanctionedRuntimeError,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_test_mode_runtime_is_not_servable(self):
|
||||||
|
"""The pytest-only record is outside the recognized and executable sets."""
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
mcp_daemon_guard.install_test_native_runtime()
|
||||||
|
self.assertNotIn(
|
||||||
|
mcp_daemon_guard.bound_transport(),
|
||||||
|
mcp_transport_config.SUPPORTED_TRANSPORTS,
|
||||||
|
)
|
||||||
|
self.assertFalse(mcp_daemon_guard.assess_serve_authorization()["allowed"])
|
||||||
|
|
||||||
|
def test_serve_authorization_does_not_leak_across_runtimes(self):
|
||||||
|
"""A later runtime's verdict never reflects an earlier one."""
|
||||||
|
with _ProductionBind(
|
||||||
|
{
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
)
|
||||||
|
}
|
||||||
|
) as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
self.assertFalse(guard.assess_serve_authorization()["allowed"])
|
||||||
|
with _ProductionBind() as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
self.assertTrue(guard.assess_serve_authorization()["allowed"])
|
||||||
|
mcp_daemon_guard.clear_native_runtime_for_tests()
|
||||||
|
self.assertEqual(
|
||||||
|
mcp_daemon_guard.assess_serve_authorization()["blocker_kind"],
|
||||||
|
mcp_transport_config.BLOCKER_TRANSPORT_NOT_BOUND,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_refusal_carries_no_credential_material(self):
|
||||||
|
with _ProductionBind(
|
||||||
|
{
|
||||||
|
mcp_transport_config.TRANSPORT_ENV: (
|
||||||
|
mcp_transport_config.REMOTE_TRANSPORT
|
||||||
|
),
|
||||||
|
"GITEA_TOKEN": "super-secret-value",
|
||||||
|
}
|
||||||
|
) as guard:
|
||||||
|
guard.bind_native_mcp_transport()
|
||||||
|
with self.assertRaises(guard.TransportExecutionError) as ctx:
|
||||||
|
guard.authorize_transport_execution("tool service")
|
||||||
|
blob = str(ctx.exception) + repr(ctx.exception.assessment)
|
||||||
|
self.assertNotIn("super-secret-value", blob)
|
||||||
|
self.assertNotIn("GITEA_TOKEN", blob)
|
||||||
|
|
||||||
|
|
||||||
|
class TestEntrypointWiring(unittest.TestCase):
|
||||||
|
"""The live entrypoint must use the seam and the execution guard."""
|
||||||
|
|
||||||
|
def test_entrypoint_binds_without_a_literal_transport(self):
|
||||||
|
text = (REPO_ROOT / "gitea_mcp_server.py").read_text(encoding="utf-8")
|
||||||
|
self.assertIn("mcp_daemon_guard.bind_native_mcp_transport()", text)
|
||||||
|
self.assertNotIn('bind_native_mcp_transport(transport="stdio")', text)
|
||||||
|
|
||||||
|
def test_entrypoint_serves_only_through_the_execution_guard(self):
|
||||||
|
text = (REPO_ROOT / "gitea_mcp_server.py").read_text(encoding="utf-8")
|
||||||
|
self.assertIn(
|
||||||
|
"mcp.run(transport=mcp_daemon_guard.authorize_transport_execution", text
|
||||||
|
)
|
||||||
|
self.assertNotIn('mcp.run(transport="stdio")', text)
|
||||||
|
self.assertNotIn(
|
||||||
|
"mcp.run(transport=mcp_daemon_guard.assert_transport_bound", text
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_every_run_call_in_production_goes_through_the_guard(self):
|
||||||
|
"""Glob the production surface: no serve site may bypass the guard."""
|
||||||
|
offenders: list[str] = []
|
||||||
|
run_sites = 0
|
||||||
|
for path in sorted(REPO_ROOT.glob("*.py")):
|
||||||
|
for lineno, line in enumerate(
|
||||||
|
path.read_text(encoding="utf-8").splitlines(), start=1
|
||||||
|
):
|
||||||
|
code = line.split("#", 1)[0]
|
||||||
|
if "mcp.run(" not in code:
|
||||||
|
continue
|
||||||
|
run_sites += 1
|
||||||
|
if "authorize_transport_execution" not in code:
|
||||||
|
offenders.append(f"{path.name}:{lineno}: {line.strip()}")
|
||||||
|
self.assertEqual(run_sites, 1, "expected exactly one serve site")
|
||||||
|
self.assertEqual(offenders, [], "\n".join(offenders))
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,346 @@
|
|||||||
|
"""Regression: author bootstrap scope reaches workflow_scope_guard (#941).
|
||||||
|
|
||||||
|
PR #926 (#892) made ``bootstrap_permits_control_checkout`` accept
|
||||||
|
``task_scope=author_issue_bootstrap`` and wired that canonical decision into
|
||||||
|
the #274 branches-only enforcer and the #604 anti-stomp preflight. A third
|
||||||
|
enforcement path was left unwired.
|
||||||
|
|
||||||
|
``workflow_scope_guard.assess_root_source_mutation`` kept its own copy of the
|
||||||
|
clean-root author decision, gated on ``create_issue_bootstrap.is_create_issue_task``
|
||||||
|
— a task-name allowlist that never contained ``bootstrap_author_issue_worktree``.
|
||||||
|
So the real call path
|
||||||
|
|
||||||
|
gitea_bootstrap_author_issue_worktree
|
||||||
|
-> verify_preflight_purity
|
||||||
|
-> _enforce_issue_scope_guard
|
||||||
|
-> workflow_scope_guard.assess_production_mutation_guards
|
||||||
|
|
||||||
|
raised ProductionGuardError(missing_issue_worktree) before
|
||||||
|
``assess_author_issue_bootstrap`` was ever consulted.
|
||||||
|
|
||||||
|
These tests drive the real enforcer, not the authorization helper in
|
||||||
|
isolation. A helper-only test cannot observe this defect: #892's own predicate
|
||||||
|
tests all passed while the live bootstrap stayed blocked.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
from unittest import mock
|
||||||
|
|
||||||
|
import author_issue_bootstrap as aib
|
||||||
|
import create_issue_bootstrap as cib
|
||||||
|
import workflow_scope_guard
|
||||||
|
|
||||||
|
BOOTSTRAP_TASK = "bootstrap_author_issue_worktree"
|
||||||
|
BOOTSTRAP_TOOL = "gitea_bootstrap_author_issue_worktree"
|
||||||
|
|
||||||
|
|
||||||
|
def _make_control_repo(tmp: str) -> tuple[str, str]:
|
||||||
|
"""Create a clean control checkout on master and return (path, head)."""
|
||||||
|
repo = os.path.join(tmp, "repo")
|
||||||
|
os.makedirs(os.path.join(repo, "branches"))
|
||||||
|
subprocess.check_call(
|
||||||
|
["git", "init", "-b", "master", repo],
|
||||||
|
stdout=subprocess.DEVNULL,
|
||||||
|
stderr=subprocess.DEVNULL,
|
||||||
|
)
|
||||||
|
subprocess.check_call(
|
||||||
|
[
|
||||||
|
"git", "-C", repo,
|
||||||
|
"-c", "user.email=t@t", "-c", "user.name=t",
|
||||||
|
"commit", "--allow-empty", "-m", "init",
|
||||||
|
],
|
||||||
|
stdout=subprocess.DEVNULL,
|
||||||
|
stderr=subprocess.DEVNULL,
|
||||||
|
)
|
||||||
|
head = subprocess.check_output(
|
||||||
|
["git", "-C", repo, "rev-parse", "HEAD"], text=True
|
||||||
|
).strip()
|
||||||
|
return repo, head
|
||||||
|
|
||||||
|
|
||||||
|
def _assessment(
|
||||||
|
repo: str,
|
||||||
|
head: str,
|
||||||
|
*,
|
||||||
|
task: str = BOOTSTRAP_TASK,
|
||||||
|
porcelain: str = "",
|
||||||
|
remote: str | None = None,
|
||||||
|
) -> dict:
|
||||||
|
return aib.assess_author_issue_bootstrap(
|
||||||
|
workspace_path=repo,
|
||||||
|
canonical_repo_root=repo,
|
||||||
|
current_branch="master",
|
||||||
|
head_sha=head,
|
||||||
|
porcelain_status=porcelain,
|
||||||
|
remote_master_sha=head if remote is None else remote,
|
||||||
|
task=task,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class _ControlCheckoutHarness(unittest.TestCase):
|
||||||
|
"""Drive the real server guard against a temporary clean control checkout."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self._tmp = tempfile.TemporaryDirectory()
|
||||||
|
self.addCleanup(self._tmp.cleanup)
|
||||||
|
self.repo, self.head = _make_control_repo(self._tmp.name)
|
||||||
|
|
||||||
|
# #683 force-on: production guards must execute under pytest.
|
||||||
|
patcher = mock.patch.dict(
|
||||||
|
os.environ,
|
||||||
|
{workflow_scope_guard.FORCE_PRODUCTION_GUARDS_ENV: "1"},
|
||||||
|
)
|
||||||
|
patcher.start()
|
||||||
|
self.addCleanup(patcher.stop)
|
||||||
|
|
||||||
|
def _enforce(
|
||||||
|
self,
|
||||||
|
task: str,
|
||||||
|
*,
|
||||||
|
porcelain: str = "",
|
||||||
|
assessment: object = "auto",
|
||||||
|
role_kind: str = "author",
|
||||||
|
):
|
||||||
|
"""Call the real _enforce_issue_scope_guard for *task*."""
|
||||||
|
import gitea_mcp_server as srv
|
||||||
|
|
||||||
|
if assessment == "auto":
|
||||||
|
assessment = _assessment(
|
||||||
|
self.repo, self.head, task=task, porcelain=porcelain
|
||||||
|
)
|
||||||
|
|
||||||
|
ctx = {
|
||||||
|
"workspace_path": self.repo,
|
||||||
|
"canonical_repo_root": self.repo,
|
||||||
|
"workspace_role_kind": role_kind,
|
||||||
|
"workspace_binding_source": "process_project_root",
|
||||||
|
}
|
||||||
|
git_state = {
|
||||||
|
"current_branch": "master",
|
||||||
|
"head_sha": self.head,
|
||||||
|
"porcelain_status": porcelain,
|
||||||
|
}
|
||||||
|
|
||||||
|
with mock.patch.object(
|
||||||
|
srv, "_resolve_namespace_mutation_context", return_value=ctx
|
||||||
|
), mock.patch.object(
|
||||||
|
srv.issue_lock_worktree,
|
||||||
|
"read_worktree_git_state",
|
||||||
|
return_value=git_state,
|
||||||
|
), mock.patch.object(
|
||||||
|
srv,
|
||||||
|
"_session_issue_lock_snapshot",
|
||||||
|
return_value={
|
||||||
|
"locked_issue_number": None,
|
||||||
|
"lock_branch_name": None,
|
||||||
|
"worktrees_match": False,
|
||||||
|
},
|
||||||
|
), mock.patch.object(
|
||||||
|
srv, "_actual_profile_role", return_value=role_kind
|
||||||
|
), mock.patch.object(
|
||||||
|
srv, "_effective_workspace_role", return_value=role_kind
|
||||||
|
), mock.patch.object(
|
||||||
|
srv, "_create_issue_bootstrap_assessment", return_value=assessment
|
||||||
|
):
|
||||||
|
srv._enforce_issue_scope_guard(None, task=task)
|
||||||
|
|
||||||
|
|
||||||
|
class TestRealPathBootstrapReachesGuard(_ControlCheckoutHarness):
|
||||||
|
"""The defect and its fix, observed through the real enforcer."""
|
||||||
|
|
||||||
|
def test_bootstrap_task_passes_scope_guard_from_clean_control(self):
|
||||||
|
# Pre-fix this raises ProductionGuardError(missing_issue_worktree)
|
||||||
|
# because the guard consulted a task-name allowlist instead of the
|
||||||
|
# canonical authorization decision.
|
||||||
|
self._enforce(BOOTSTRAP_TASK)
|
||||||
|
|
||||||
|
def test_bootstrap_tool_alias_passes_scope_guard(self):
|
||||||
|
self._enforce(BOOTSTRAP_TOOL)
|
||||||
|
|
||||||
|
def test_guard_consults_canonical_predicate(self):
|
||||||
|
"""The guard must reach bootstrap_permits_control_checkout, not a name list."""
|
||||||
|
real = cib.bootstrap_permits_control_checkout
|
||||||
|
seen: list[str | None] = []
|
||||||
|
|
||||||
|
def _spy(assessment, *, task, workspace_path, canonical_repo_root):
|
||||||
|
seen.append(task)
|
||||||
|
return real(
|
||||||
|
assessment,
|
||||||
|
task=task,
|
||||||
|
workspace_path=workspace_path,
|
||||||
|
canonical_repo_root=canonical_repo_root,
|
||||||
|
)
|
||||||
|
|
||||||
|
with mock.patch.object(
|
||||||
|
cib, "bootstrap_permits_control_checkout", side_effect=_spy
|
||||||
|
):
|
||||||
|
self._enforce(BOOTSTRAP_TASK)
|
||||||
|
|
||||||
|
self.assertIn(
|
||||||
|
BOOTSTRAP_TASK,
|
||||||
|
seen,
|
||||||
|
"workflow_scope_guard did not consult the canonical bootstrap "
|
||||||
|
"authorization decision",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class TestFailClosedOnBadEvidence(_ControlCheckoutHarness):
|
||||||
|
"""Missing, malformed, or mismatched scope evidence must still block."""
|
||||||
|
|
||||||
|
def _assert_blocked(self, **kwargs):
|
||||||
|
with self.assertRaises(workflow_scope_guard.ProductionGuardError):
|
||||||
|
self._enforce(BOOTSTRAP_TASK, **kwargs)
|
||||||
|
|
||||||
|
def test_missing_assessment_fails_closed(self):
|
||||||
|
self._assert_blocked(assessment=None)
|
||||||
|
|
||||||
|
def test_malformed_assessment_fails_closed(self):
|
||||||
|
self._assert_blocked(assessment={"allowed": True})
|
||||||
|
|
||||||
|
def test_non_dict_assessment_fails_closed(self):
|
||||||
|
self._assert_blocked(assessment="allowed")
|
||||||
|
|
||||||
|
def test_wrong_task_scope_fails_closed(self):
|
||||||
|
bad = dict(_assessment(self.repo, self.head))
|
||||||
|
bad["task_scope"] = "create_issue_only"
|
||||||
|
self._assert_blocked(assessment=bad)
|
||||||
|
|
||||||
|
def test_nonempty_reasons_fail_closed(self):
|
||||||
|
bad = dict(_assessment(self.repo, self.head), reasons=["note"])
|
||||||
|
self._assert_blocked(assessment=bad)
|
||||||
|
|
||||||
|
def test_mismatched_base_tips_fail_closed(self):
|
||||||
|
bad = dict(_assessment(self.repo, self.head))
|
||||||
|
bad["remote_master_sha"] = "b" * 40
|
||||||
|
self._assert_blocked(assessment=bad)
|
||||||
|
|
||||||
|
def test_unverified_base_tips_fail_closed(self):
|
||||||
|
bad = dict(_assessment(self.repo, self.head), base_tips_verified=False)
|
||||||
|
self._assert_blocked(assessment=bad)
|
||||||
|
|
||||||
|
def test_mismatched_workspace_binding_fails_closed(self):
|
||||||
|
bad = dict(_assessment(self.repo, self.head))
|
||||||
|
bad["workspace_path"] = os.path.join(self.repo, "elsewhere")
|
||||||
|
self._assert_blocked(assessment=bad)
|
||||||
|
|
||||||
|
def test_mismatched_repo_root_binding_fails_closed(self):
|
||||||
|
bad = dict(_assessment(self.repo, self.head))
|
||||||
|
bad["canonical_repo_root"] = os.path.join(self.repo, "other-root")
|
||||||
|
self._assert_blocked(assessment=bad)
|
||||||
|
|
||||||
|
def test_blocked_assessment_fails_closed(self):
|
||||||
|
bad = dict(_assessment(self.repo, self.head), block=True, allowed=False)
|
||||||
|
self._assert_blocked(assessment=bad)
|
||||||
|
|
||||||
|
|
||||||
|
class TestOrdinaryControlCheckoutMutationStillForbidden(_ControlCheckoutHarness):
|
||||||
|
"""The waiver must not leak to ordinary author work."""
|
||||||
|
|
||||||
|
def test_ordinary_author_task_still_blocked(self):
|
||||||
|
with self.assertRaises(workflow_scope_guard.ProductionGuardError):
|
||||||
|
self._enforce("commit_files", assessment=None)
|
||||||
|
|
||||||
|
def test_lock_issue_still_blocked_from_control(self):
|
||||||
|
with self.assertRaises(workflow_scope_guard.ProductionGuardError):
|
||||||
|
self._enforce("lock_issue", assessment=None)
|
||||||
|
|
||||||
|
def test_bootstrap_assessment_cannot_license_other_task(self):
|
||||||
|
# Cross-task smuggling: valid bootstrap evidence must not waive a
|
||||||
|
# different author mutation.
|
||||||
|
good = _assessment(self.repo, self.head)
|
||||||
|
with self.assertRaises(workflow_scope_guard.ProductionGuardError):
|
||||||
|
self._enforce("commit_files", assessment=good)
|
||||||
|
|
||||||
|
def test_dirty_control_checkout_still_blocked_for_bootstrap(self):
|
||||||
|
with self.assertRaises(workflow_scope_guard.ProductionGuardError):
|
||||||
|
self._enforce(BOOTSTRAP_TASK, porcelain=" M gitea_mcp_server.py\n")
|
||||||
|
|
||||||
|
|
||||||
|
class TestCreateIssueBehaviorUnchanged(_ControlCheckoutHarness):
|
||||||
|
"""#749 create_issue keeps its own sanctioned path."""
|
||||||
|
|
||||||
|
def test_create_issue_still_allowed_from_clean_control(self):
|
||||||
|
self._enforce("create_issue", assessment=None)
|
||||||
|
|
||||||
|
def test_create_issue_tool_alias_still_allowed(self):
|
||||||
|
self._enforce("gitea_create_issue", assessment=None)
|
||||||
|
|
||||||
|
def test_create_issue_blocked_when_control_dirty(self):
|
||||||
|
with self.assertRaises(workflow_scope_guard.ProductionGuardError):
|
||||||
|
self._enforce(
|
||||||
|
"create_issue",
|
||||||
|
porcelain=" M gitea_mcp_server.py\n",
|
||||||
|
assessment=None,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class TestGuardUnitLevelWiring(unittest.TestCase):
|
||||||
|
"""assess_root_source_mutation itself must accept and honour the evidence."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self._tmp = tempfile.TemporaryDirectory()
|
||||||
|
self.addCleanup(self._tmp.cleanup)
|
||||||
|
self.repo, self.head = _make_control_repo(self._tmp.name)
|
||||||
|
patcher = mock.patch.dict(
|
||||||
|
os.environ,
|
||||||
|
{workflow_scope_guard.FORCE_PRODUCTION_GUARDS_ENV: "1"},
|
||||||
|
)
|
||||||
|
patcher.start()
|
||||||
|
self.addCleanup(patcher.stop)
|
||||||
|
|
||||||
|
def _assess(self, *, task=BOOTSTRAP_TASK, bootstrap_assessment="auto"):
|
||||||
|
if bootstrap_assessment == "auto":
|
||||||
|
bootstrap_assessment = _assessment(self.repo, self.head, task=task)
|
||||||
|
return workflow_scope_guard.assess_root_source_mutation(
|
||||||
|
workspace_path=self.repo,
|
||||||
|
canonical_repo_root=self.repo,
|
||||||
|
porcelain_status="",
|
||||||
|
current_branch="master",
|
||||||
|
role_kind="author",
|
||||||
|
mutation_task=task,
|
||||||
|
bootstrap_assessment=bootstrap_assessment,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_valid_evidence_unblocks(self):
|
||||||
|
result = self._assess()
|
||||||
|
self.assertFalse(result["block"])
|
||||||
|
self.assertIsNone(result["blocker_kind"])
|
||||||
|
|
||||||
|
def test_absent_evidence_blocks(self):
|
||||||
|
result = self._assess(bootstrap_assessment=None)
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
self.assertEqual(
|
||||||
|
result["blocker_kind"], workflow_scope_guard.BLOCKER_MISSING_WORKTREE
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_reconciler_exemption_preserved(self):
|
||||||
|
result = workflow_scope_guard.assess_root_source_mutation(
|
||||||
|
workspace_path=self.repo,
|
||||||
|
canonical_repo_root=self.repo,
|
||||||
|
porcelain_status="",
|
||||||
|
current_branch="master",
|
||||||
|
role_kind="reconciler",
|
||||||
|
mutation_task=BOOTSTRAP_TASK,
|
||||||
|
)
|
||||||
|
self.assertFalse(result["block"])
|
||||||
|
|
||||||
|
def test_signature_accepts_evidence_without_it_being_required(self):
|
||||||
|
# Callers that supply no evidence keep the pre-existing behaviour.
|
||||||
|
result = workflow_scope_guard.assess_root_source_mutation(
|
||||||
|
workspace_path=self.repo,
|
||||||
|
canonical_repo_root=self.repo,
|
||||||
|
porcelain_status="",
|
||||||
|
current_branch="master",
|
||||||
|
role_kind="author",
|
||||||
|
mutation_task="create_issue",
|
||||||
|
)
|
||||||
|
self.assertFalse(result["block"])
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,747 @@
|
|||||||
|
"""Regression: author bootstrap runtime authority and session ownership (#943).
|
||||||
|
|
||||||
|
Two rounds of defects live here.
|
||||||
|
|
||||||
|
**Round 1 (#943 as filed).** ``gitea_bootstrap_author_issue_worktree`` passed
|
||||||
|
four values down to the bootstrap service that were never defined:
|
||||||
|
``_active_username``, ``_active_profile_name``, ``_current_session_id`` and
|
||||||
|
``_author_mutation_block``. Every call — dry-run included — raised
|
||||||
|
``NameError`` while evaluating the arguments, before the service was entered.
|
||||||
|
|
||||||
|
**Round 2 (review 622 on PR #944).** The first fix defined all four but made
|
||||||
|
``_current_session_id`` mint ``<profile>-<pid>-<hex>`` once per process. The MCP
|
||||||
|
daemon outlives every task it serves, so that value conflates sequential author
|
||||||
|
tasks and can never equal the control-plane session that owns an
|
||||||
|
allocator-created lease: ``_verify_assignment_and_lease_ids`` refused the whole
|
||||||
|
allocated path with ``lease_session_mismatch``. The reviewed round also read the
|
||||||
|
identity from the pinned session context while reading the profile from the live
|
||||||
|
profile, so a rebind could produce a mixed claimant pair, and it swallowed every
|
||||||
|
``get_profile()`` exception.
|
||||||
|
|
||||||
|
These tests therefore drive real state, not mocks of internals: a temporary
|
||||||
|
control-plane SQLite database and a temporary issue-lock directory, both
|
||||||
|
redirected through the same environment variables production uses
|
||||||
|
(``GITEA_CONTROL_PLANE_DB``, ``GITEA_ISSUE_LOCK_DIR``). The ownership gate that
|
||||||
|
runs is the real one.
|
||||||
|
|
||||||
|
``test_every_global_referenced_by_the_wrapper_resolves`` remains: it is what
|
||||||
|
found ``_author_mutation_block``, and it generalises to the next missing
|
||||||
|
reference. It supplements the runtime coverage below rather than standing in for
|
||||||
|
it.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import ast
|
||||||
|
import builtins
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import subprocess
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
from unittest import mock
|
||||||
|
|
||||||
|
import author_issue_bootstrap as aib
|
||||||
|
import control_plane_db
|
||||||
|
import create_issue_bootstrap as cib
|
||||||
|
import gitea_mcp_server as gms
|
||||||
|
import issue_lock_store
|
||||||
|
import workflow_scope_guard
|
||||||
|
|
||||||
|
BOOTSTRAP_TASK = "bootstrap_author_issue_worktree"
|
||||||
|
WRAPPER_NAME = "gitea_bootstrap_author_issue_worktree"
|
||||||
|
RUNTIME_HELPERS = (
|
||||||
|
"_active_mutation_authority",
|
||||||
|
"_active_username",
|
||||||
|
"_active_profile_name",
|
||||||
|
"_resolve_owner_workflow_session",
|
||||||
|
"_author_mutation_block",
|
||||||
|
)
|
||||||
|
ORG = "Scaled-Tech-Consulting"
|
||||||
|
REPO = "Gitea-Tools"
|
||||||
|
IDENTITY = "jcwalker3"
|
||||||
|
PROFILE = "prgs-author"
|
||||||
|
|
||||||
|
# A per-task ownership key must carry no process identifier (#790).
|
||||||
|
TASK_KEY_RE = re.compile(r"^author_issue_work-[0-9a-f]{16}$")
|
||||||
|
|
||||||
|
|
||||||
|
def _make_control_repo(tmp: str) -> tuple[str, str]:
|
||||||
|
"""Create a clean control checkout on master and return (path, head)."""
|
||||||
|
repo = os.path.join(tmp, "repo")
|
||||||
|
os.makedirs(os.path.join(repo, "branches"))
|
||||||
|
subprocess.check_call(
|
||||||
|
["git", "init", "-b", "master", repo],
|
||||||
|
stdout=subprocess.DEVNULL,
|
||||||
|
stderr=subprocess.DEVNULL,
|
||||||
|
)
|
||||||
|
subprocess.check_call(
|
||||||
|
[
|
||||||
|
"git", "-C", repo,
|
||||||
|
"-c", "user.email=t@t", "-c", "user.name=t",
|
||||||
|
"commit", "--allow-empty", "-m", "init",
|
||||||
|
],
|
||||||
|
stdout=subprocess.DEVNULL,
|
||||||
|
stderr=subprocess.DEVNULL,
|
||||||
|
)
|
||||||
|
head = subprocess.check_output(
|
||||||
|
["git", "-C", repo, "rev-parse", "HEAD"], text=True
|
||||||
|
).strip()
|
||||||
|
return repo, head
|
||||||
|
|
||||||
|
|
||||||
|
def _wrapper_ast() -> ast.FunctionDef:
|
||||||
|
"""Return the AST of the bootstrap wrapper as it exists on disk."""
|
||||||
|
path = os.path.join(
|
||||||
|
os.path.dirname(os.path.dirname(os.path.abspath(__file__))),
|
||||||
|
"gitea_mcp_server.py",
|
||||||
|
)
|
||||||
|
with open(path, encoding="utf-8") as fh:
|
||||||
|
tree = ast.parse(fh.read())
|
||||||
|
for node in ast.walk(tree):
|
||||||
|
if isinstance(node, ast.FunctionDef) and node.name == WRAPPER_NAME:
|
||||||
|
return node
|
||||||
|
raise AssertionError(f"{WRAPPER_NAME} not found in gitea_mcp_server.py")
|
||||||
|
|
||||||
|
|
||||||
|
class _IsolatedControlPlane(unittest.TestCase):
|
||||||
|
"""Temp control-plane DB and temp issue-lock dir, via production env vars."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self._tmp = tempfile.TemporaryDirectory()
|
||||||
|
self.addCleanup(self._tmp.cleanup)
|
||||||
|
self.tmp = self._tmp.name
|
||||||
|
self.db_path = os.path.join(self.tmp, "control-plane.sqlite3")
|
||||||
|
self.lock_dir = os.path.join(self.tmp, "issue-locks")
|
||||||
|
self.journals = os.path.join(self.tmp, "journals")
|
||||||
|
os.makedirs(self.lock_dir)
|
||||||
|
os.makedirs(self.journals)
|
||||||
|
env = mock.patch.dict(
|
||||||
|
os.environ,
|
||||||
|
{
|
||||||
|
control_plane_db.DB_PATH_ENV: self.db_path,
|
||||||
|
issue_lock_store.LOCK_DIR_ENV: self.lock_dir,
|
||||||
|
},
|
||||||
|
)
|
||||||
|
env.start()
|
||||||
|
self.addCleanup(env.stop)
|
||||||
|
self.db = control_plane_db.ControlPlaneDB(self.db_path)
|
||||||
|
|
||||||
|
def _allocate(self, session_id: str, *, issue: int = 943):
|
||||||
|
"""Create a real assignment + lease owned by *session_id*."""
|
||||||
|
self.db.upsert_session(
|
||||||
|
session_id=session_id, role="author", profile=PROFILE, pid=os.getpid()
|
||||||
|
)
|
||||||
|
res = self.db.assign_and_lease(
|
||||||
|
session_id=session_id, role="author", remote="prgs",
|
||||||
|
org=ORG, repo=REPO, kind="issue", number=issue,
|
||||||
|
)
|
||||||
|
self.assertEqual(res.outcome, "assigned", res)
|
||||||
|
return res.assignment_id, res.lease_id
|
||||||
|
|
||||||
|
def _authority(self):
|
||||||
|
"""A resolved authority pair, as the wrapper would compute it."""
|
||||||
|
return {"ok": True, "identity": IDENTITY, "profile_name": PROFILE}
|
||||||
|
|
||||||
|
def _resolve_session(self, **over):
|
||||||
|
kwargs = dict(
|
||||||
|
issue_number=943,
|
||||||
|
assignment_id=None,
|
||||||
|
lease_id=None,
|
||||||
|
session_id=None,
|
||||||
|
identity=IDENTITY,
|
||||||
|
profile_name=PROFILE,
|
||||||
|
remote="prgs",
|
||||||
|
org=ORG,
|
||||||
|
repo=REPO,
|
||||||
|
)
|
||||||
|
kwargs.update(over)
|
||||||
|
return gms._resolve_owner_workflow_session(**kwargs)
|
||||||
|
|
||||||
|
|
||||||
|
class OwnershipGateTests(_IsolatedControlPlane):
|
||||||
|
"""B2: the allocator-driven ownership path, against a real control plane."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
super().setUp()
|
||||||
|
self.repo, self.head = _make_control_repo(self.tmp)
|
||||||
|
|
||||||
|
def _bootstrap(self, **over):
|
||||||
|
kwargs = dict(
|
||||||
|
issue_number=943,
|
||||||
|
canonical_repo_root=self.repo,
|
||||||
|
expected_base_sha=self.head,
|
||||||
|
branch_name="fix/issue-943-runtime-context-helpers",
|
||||||
|
remote="prgs",
|
||||||
|
org=ORG,
|
||||||
|
repo=REPO,
|
||||||
|
active_identity=IDENTITY,
|
||||||
|
active_profile=PROFILE,
|
||||||
|
lock_dir=self.journals,
|
||||||
|
idempotency_key="test-943",
|
||||||
|
dry_run=True,
|
||||||
|
)
|
||||||
|
kwargs.update(over)
|
||||||
|
return aib.bootstrap_author_issue_worktree(**kwargs)
|
||||||
|
|
||||||
|
def test_true_owning_session_passes_the_ownership_gate(self):
|
||||||
|
"""The canonical owner reaches and completes the service."""
|
||||||
|
session = "prgs-author-task-a"
|
||||||
|
assignment_id, lease_id = self._allocate(session)
|
||||||
|
res = self._bootstrap(
|
||||||
|
assignment_id=assignment_id, lease_id=lease_id, owner_session=session
|
||||||
|
)
|
||||||
|
self.assertTrue(res.get("success"), res)
|
||||||
|
self.assertTrue(res.get("dry_run"))
|
||||||
|
self.assertEqual(res.get("base_sha"), self.head)
|
||||||
|
|
||||||
|
def test_different_session_is_refused(self):
|
||||||
|
session = "prgs-author-task-a"
|
||||||
|
assignment_id, lease_id = self._allocate(session)
|
||||||
|
res = self._bootstrap(
|
||||||
|
assignment_id=assignment_id,
|
||||||
|
lease_id=lease_id,
|
||||||
|
owner_session="prgs-author-task-b",
|
||||||
|
)
|
||||||
|
self.assertFalse(res.get("success"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "lease_session_mismatch")
|
||||||
|
|
||||||
|
def test_process_derived_session_would_be_refused(self):
|
||||||
|
"""The reviewed round-1 value shape can never own an allocated lease."""
|
||||||
|
session = "prgs-author-task-a"
|
||||||
|
assignment_id, lease_id = self._allocate(session)
|
||||||
|
round_one_value = f"{PROFILE}-{os.getpid()}-deadbeef"
|
||||||
|
self.assertNotEqual(round_one_value, session)
|
||||||
|
res = self._bootstrap(
|
||||||
|
assignment_id=assignment_id,
|
||||||
|
lease_id=lease_id,
|
||||||
|
owner_session=round_one_value,
|
||||||
|
)
|
||||||
|
self.assertFalse(res.get("success"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "lease_session_mismatch")
|
||||||
|
|
||||||
|
def test_unknown_lease_fails_closed(self):
|
||||||
|
session = "prgs-author-task-a"
|
||||||
|
assignment_id, _ = self._allocate(session)
|
||||||
|
res = self._bootstrap(
|
||||||
|
assignment_id=assignment_id,
|
||||||
|
lease_id="lease-does-not-exist",
|
||||||
|
owner_session=session,
|
||||||
|
)
|
||||||
|
self.assertFalse(res.get("success"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "unknown_lease_id")
|
||||||
|
|
||||||
|
def test_released_lease_fails_closed(self):
|
||||||
|
session = "prgs-author-task-a"
|
||||||
|
assignment_id, lease_id = self._allocate(session)
|
||||||
|
self.db.release_lease(lease_id, session_id=session)
|
||||||
|
res = self._bootstrap(
|
||||||
|
assignment_id=assignment_id, lease_id=lease_id, owner_session=session
|
||||||
|
)
|
||||||
|
self.assertFalse(res.get("success"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "lease_not_live")
|
||||||
|
|
||||||
|
def test_force_expired_lease_fails_closed(self):
|
||||||
|
session = "prgs-author-task-a"
|
||||||
|
assignment_id, lease_id = self._allocate(session)
|
||||||
|
self.db.force_expire_lease(lease_id, reason="test")
|
||||||
|
res = self._bootstrap(
|
||||||
|
assignment_id=assignment_id, lease_id=lease_id, owner_session=session
|
||||||
|
)
|
||||||
|
self.assertFalse(res.get("success"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "lease_not_live")
|
||||||
|
|
||||||
|
def test_replacement_lease_does_not_inherit_prior_ownership(self):
|
||||||
|
"""A second task's lease is not ownable by the first task's session."""
|
||||||
|
first = "prgs-author-task-a"
|
||||||
|
assignment_a, lease_a = self._allocate(first)
|
||||||
|
self.db.release_lease(lease_a, session_id=first)
|
||||||
|
second = "prgs-author-task-b"
|
||||||
|
assignment_b, lease_b = self._allocate(second)
|
||||||
|
self.assertNotEqual(lease_a, lease_b)
|
||||||
|
res = self._bootstrap(
|
||||||
|
assignment_id=assignment_b, lease_id=lease_b, owner_session=first
|
||||||
|
)
|
||||||
|
self.assertFalse(res.get("success"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "lease_session_mismatch")
|
||||||
|
|
||||||
|
def test_assignment_lease_identifier_mismatch_fails_closed(self):
|
||||||
|
session = "prgs-author-task-a"
|
||||||
|
_, lease_id = self._allocate(session)
|
||||||
|
res = self._bootstrap(
|
||||||
|
assignment_id="asn-not-the-recorded-one",
|
||||||
|
lease_id=lease_id,
|
||||||
|
owner_session=session,
|
||||||
|
)
|
||||||
|
self.assertFalse(res.get("success"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "assignment_lease_mismatch")
|
||||||
|
|
||||||
|
def test_lease_id_without_assignment_id_fails_closed(self):
|
||||||
|
session = "prgs-author-task-a"
|
||||||
|
_, lease_id = self._allocate(session)
|
||||||
|
res = self._bootstrap(lease_id=lease_id, owner_session=session)
|
||||||
|
self.assertFalse(res.get("success"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "incomplete_assignment_lease_ids")
|
||||||
|
|
||||||
|
def test_dry_run_with_valid_allocator_bindings_leaves_no_durable_state(self):
|
||||||
|
session = "prgs-author-task-a"
|
||||||
|
assignment_id, lease_id = self._allocate(session)
|
||||||
|
res = self._bootstrap(
|
||||||
|
assignment_id=assignment_id, lease_id=lease_id, owner_session=session
|
||||||
|
)
|
||||||
|
self.assertTrue(res.get("success"), res)
|
||||||
|
|
||||||
|
branches = subprocess.check_output(
|
||||||
|
["git", "-C", self.repo, "branch", "--list"], text=True
|
||||||
|
)
|
||||||
|
self.assertNotIn("issue-943", branches)
|
||||||
|
worktrees = subprocess.check_output(
|
||||||
|
["git", "-C", self.repo, "worktree", "list"], text=True
|
||||||
|
)
|
||||||
|
self.assertNotIn("issue-943", worktrees)
|
||||||
|
self.assertFalse(
|
||||||
|
os.path.exists(
|
||||||
|
os.path.join(self.repo, "branches",
|
||||||
|
"fix-issue-943-runtime-context-helpers")
|
||||||
|
)
|
||||||
|
)
|
||||||
|
journal = res.get("phase_journal") or {}
|
||||||
|
self.assertFalse(journal.get("completed"))
|
||||||
|
self.assertFalse(any((journal.get("artifacts_created") or {}).values()))
|
||||||
|
# The dry run must not have created an issue lock in the isolated dir.
|
||||||
|
self.assertEqual(os.listdir(self.lock_dir), [])
|
||||||
|
|
||||||
|
def test_apply_reaches_the_intended_transition_with_valid_bindings(self):
|
||||||
|
session = "prgs-author-task-a"
|
||||||
|
assignment_id, lease_id = self._allocate(session)
|
||||||
|
res = self._bootstrap(
|
||||||
|
assignment_id=assignment_id,
|
||||||
|
lease_id=lease_id,
|
||||||
|
owner_session=session,
|
||||||
|
dry_run=False,
|
||||||
|
)
|
||||||
|
self.assertTrue(res.get("success"), res)
|
||||||
|
self.assertNotEqual(res.get("dry_run"), True)
|
||||||
|
branches = subprocess.check_output(
|
||||||
|
["git", "-C", self.repo, "branch", "--list"], text=True
|
||||||
|
)
|
||||||
|
self.assertIn("issue-943", branches)
|
||||||
|
self.assertTrue(os.path.isdir(res.get("worktree_path") or ""))
|
||||||
|
|
||||||
|
|
||||||
|
class WorkflowSessionResolutionTests(_IsolatedControlPlane):
|
||||||
|
"""B1: the wrapper resolves the owning session, never a process identifier."""
|
||||||
|
|
||||||
|
def test_declared_session_is_verified_against_the_control_plane(self):
|
||||||
|
session = "prgs-author-task-a"
|
||||||
|
assignment_id, lease_id = self._allocate(session)
|
||||||
|
res = self._resolve_session(
|
||||||
|
session_id=session, assignment_id=assignment_id, lease_id=lease_id
|
||||||
|
)
|
||||||
|
self.assertTrue(res.get("ok"), res)
|
||||||
|
self.assertEqual(res.get("session_id"), session)
|
||||||
|
self.assertEqual(res.get("session_source"), "declared")
|
||||||
|
|
||||||
|
def test_unknown_declared_session_is_refused_not_trusted(self):
|
||||||
|
res = self._resolve_session(session_id="prgs-author-not-a-session")
|
||||||
|
self.assertFalse(res.get("ok"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "workflow_session_unverified")
|
||||||
|
|
||||||
|
def test_declared_session_for_another_role_is_refused(self):
|
||||||
|
self.db.upsert_session(
|
||||||
|
session_id="prgs-reviewer-x", role="reviewer", profile="prgs-reviewer",
|
||||||
|
pid=os.getpid(),
|
||||||
|
)
|
||||||
|
res = self._resolve_session(session_id="prgs-reviewer-x")
|
||||||
|
self.assertFalse(res.get("ok"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "workflow_session_unverified")
|
||||||
|
|
||||||
|
def test_declared_session_for_another_profile_is_refused(self):
|
||||||
|
self.db.upsert_session(
|
||||||
|
session_id="other-profile-session", role="author",
|
||||||
|
profile="prgs-controller", pid=os.getpid(),
|
||||||
|
)
|
||||||
|
res = self._resolve_session(session_id="other-profile-session")
|
||||||
|
self.assertFalse(res.get("ok"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "workflow_session_unverified")
|
||||||
|
|
||||||
|
def test_allocated_work_without_a_session_is_refused(self):
|
||||||
|
"""Supplying a lease is not itself evidence of ownership."""
|
||||||
|
session = "prgs-author-task-a"
|
||||||
|
assignment_id, lease_id = self._allocate(session)
|
||||||
|
res = self._resolve_session(assignment_id=assignment_id, lease_id=lease_id)
|
||||||
|
self.assertFalse(res.get("ok"))
|
||||||
|
self.assertEqual(
|
||||||
|
res.get("reason_code"), "workflow_session_required_for_allocated_work"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_existing_issue_lock_supplies_its_per_task_session(self):
|
||||||
|
lock_session = issue_lock_store.mint_task_session_id(
|
||||||
|
issue_lock_store.AUTHOR_ISSUE_WORK_LEASE
|
||||||
|
)
|
||||||
|
path = issue_lock_store.lock_file_path(
|
||||||
|
remote="prgs", org=ORG, repo=REPO, issue_number=943,
|
||||||
|
lock_dir=self.lock_dir,
|
||||||
|
)
|
||||||
|
issue_lock_store.write_lock_file(
|
||||||
|
path,
|
||||||
|
{
|
||||||
|
"issue_number": 943,
|
||||||
|
"branch_name": "fix/issue-943-runtime-context-helpers",
|
||||||
|
"work_lease": {
|
||||||
|
"task_session_id": lock_session,
|
||||||
|
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
) if hasattr(issue_lock_store, "write_lock_file") else _write_json(
|
||||||
|
path,
|
||||||
|
{
|
||||||
|
"issue_number": 943,
|
||||||
|
"branch_name": "fix/issue-943-runtime-context-helpers",
|
||||||
|
"work_lease": {
|
||||||
|
"task_session_id": lock_session,
|
||||||
|
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
)
|
||||||
|
res = self._resolve_session()
|
||||||
|
self.assertTrue(res.get("ok"), res)
|
||||||
|
self.assertEqual(res.get("session_id"), lock_session)
|
||||||
|
self.assertEqual(res.get("session_source"), "issue_lock")
|
||||||
|
|
||||||
|
def test_issue_lock_owned_by_another_identity_is_refused(self):
|
||||||
|
path = issue_lock_store.lock_file_path(
|
||||||
|
remote="prgs", org=ORG, repo=REPO, issue_number=943,
|
||||||
|
lock_dir=self.lock_dir,
|
||||||
|
)
|
||||||
|
_write_json(
|
||||||
|
path,
|
||||||
|
{
|
||||||
|
"issue_number": 943,
|
||||||
|
"work_lease": {
|
||||||
|
"task_session_id": "author_issue_work-" + "0" * 16,
|
||||||
|
"claimant": {"username": "someone-else", "profile": PROFILE},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
)
|
||||||
|
res = self._resolve_session()
|
||||||
|
self.assertFalse(res.get("ok"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "issue_lock_owner_mismatch")
|
||||||
|
|
||||||
|
def test_unallocated_bootstrap_mints_a_per_task_key(self):
|
||||||
|
res = self._resolve_session()
|
||||||
|
self.assertTrue(res.get("ok"), res)
|
||||||
|
self.assertEqual(res.get("session_source"), "minted_task_key")
|
||||||
|
self.assertRegex(res["session_id"], TASK_KEY_RE)
|
||||||
|
|
||||||
|
def test_minted_key_contains_no_process_identifier(self):
|
||||||
|
res = self._resolve_session()
|
||||||
|
self.assertNotIn(str(os.getpid()), res["session_id"])
|
||||||
|
self.assertNotIn(PROFILE, res["session_id"])
|
||||||
|
|
||||||
|
def test_sequential_tasks_on_one_daemon_do_not_share_ownership(self):
|
||||||
|
"""The round-1 defect: one identifier per process for every task."""
|
||||||
|
first = self._resolve_session()["session_id"]
|
||||||
|
second = self._resolve_session()["session_id"]
|
||||||
|
third = self._resolve_session()["session_id"]
|
||||||
|
self.assertNotEqual(first, second)
|
||||||
|
self.assertNotEqual(second, third)
|
||||||
|
self.assertEqual(len({first, second, third}), 3)
|
||||||
|
|
||||||
|
def test_no_process_lifetime_cache_remains(self):
|
||||||
|
self.assertFalse(hasattr(gms, "_ACTIVE_SESSION_ID"))
|
||||||
|
self.assertFalse(hasattr(gms, "_current_session_id"))
|
||||||
|
|
||||||
|
|
||||||
|
class MutationAuthorityTests(unittest.TestCase):
|
||||||
|
"""F3/F4: one coherent authority pair, drift detected, no silent fallback."""
|
||||||
|
|
||||||
|
def _ctx(self, **over):
|
||||||
|
base = {"identity": IDENTITY, "profile_name": PROFILE}
|
||||||
|
base.update(over)
|
||||||
|
return base
|
||||||
|
|
||||||
|
def test_matching_live_and_pinned_authority_resolves(self):
|
||||||
|
with mock.patch.object(gms, "get_profile",
|
||||||
|
return_value={"profile_name": PROFILE}), \
|
||||||
|
mock.patch.object(gms, "_authenticated_username",
|
||||||
|
return_value=IDENTITY), \
|
||||||
|
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||||
|
return_value=self._ctx()):
|
||||||
|
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||||
|
self.assertTrue(res.get("ok"), res)
|
||||||
|
self.assertEqual(res["identity"], IDENTITY)
|
||||||
|
self.assertEqual(res["profile_name"], PROFILE)
|
||||||
|
|
||||||
|
def test_identity_drift_fails_closed(self):
|
||||||
|
with mock.patch.object(gms, "get_profile",
|
||||||
|
return_value={"profile_name": PROFILE}), \
|
||||||
|
mock.patch.object(gms, "_authenticated_username",
|
||||||
|
return_value="someone-else"), \
|
||||||
|
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||||
|
return_value=self._ctx()):
|
||||||
|
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||||
|
self.assertFalse(res.get("ok"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "authority_identity_drift")
|
||||||
|
self.assertEqual(res.get("expected"), IDENTITY)
|
||||||
|
self.assertEqual(res.get("actual"), "someone-else")
|
||||||
|
|
||||||
|
def test_profile_drift_fails_closed(self):
|
||||||
|
with mock.patch.object(gms, "get_profile",
|
||||||
|
return_value={"profile_name": "prgs-controller"}), \
|
||||||
|
mock.patch.object(gms, "_authenticated_username",
|
||||||
|
return_value=IDENTITY), \
|
||||||
|
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||||
|
return_value=self._ctx()):
|
||||||
|
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||||
|
self.assertFalse(res.get("ok"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "authority_profile_drift")
|
||||||
|
|
||||||
|
def test_identity_and_profile_never_come_from_different_snapshots(self):
|
||||||
|
"""Round 2's mixed pair: pinned identity plus live profile."""
|
||||||
|
with mock.patch.object(gms, "get_profile",
|
||||||
|
return_value={"profile_name": "prgs-controller"}), \
|
||||||
|
mock.patch.object(gms, "_authenticated_username",
|
||||||
|
return_value="new-identity"), \
|
||||||
|
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||||
|
return_value=self._ctx()):
|
||||||
|
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||||
|
self.assertFalse(res.get("ok"))
|
||||||
|
self.assertIsNone(gms._active_username("gitea.prgs.cc"))
|
||||||
|
self.assertIsNone(gms._active_profile_name("gitea.prgs.cc"))
|
||||||
|
|
||||||
|
def test_unresolvable_profile_is_a_structured_refusal_not_a_fallback(self):
|
||||||
|
"""F4: no bare-except fallback to a previously pinned profile name."""
|
||||||
|
with mock.patch.object(gms, "get_profile",
|
||||||
|
side_effect=RuntimeError("profile disabled")), \
|
||||||
|
mock.patch.object(gms, "_authenticated_username",
|
||||||
|
return_value=IDENTITY), \
|
||||||
|
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||||
|
return_value=self._ctx()):
|
||||||
|
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||||
|
self.assertFalse(res.get("ok"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "authority_profile_unresolved")
|
||||||
|
self.assertNotEqual(res.get("profile_name"), PROFILE)
|
||||||
|
|
||||||
|
def test_malformed_profile_without_name_fails_closed(self):
|
||||||
|
with mock.patch.object(gms, "get_profile", return_value={}), \
|
||||||
|
mock.patch.object(gms, "_authenticated_username",
|
||||||
|
return_value=IDENTITY), \
|
||||||
|
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||||
|
return_value=None):
|
||||||
|
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||||
|
self.assertFalse(res.get("ok"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "authority_profile_unresolved")
|
||||||
|
|
||||||
|
def test_unresolved_identity_fails_closed(self):
|
||||||
|
for value in (None, "", " "):
|
||||||
|
with self.subTest(identity=value):
|
||||||
|
with mock.patch.object(gms, "get_profile",
|
||||||
|
return_value={"profile_name": PROFILE}), \
|
||||||
|
mock.patch.object(gms, "_authenticated_username",
|
||||||
|
return_value=value), \
|
||||||
|
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||||
|
return_value=None):
|
||||||
|
res = gms._active_mutation_authority("gitea.prgs.cc")
|
||||||
|
self.assertFalse(res.get("ok"))
|
||||||
|
self.assertEqual(
|
||||||
|
res.get("reason_code"), "authority_identity_unresolved"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_missing_host_cannot_yield_an_identity(self):
|
||||||
|
with mock.patch.object(gms, "get_profile",
|
||||||
|
return_value={"profile_name": PROFILE}), \
|
||||||
|
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||||
|
return_value=None):
|
||||||
|
res = gms._active_mutation_authority(None)
|
||||||
|
self.assertFalse(res.get("ok"))
|
||||||
|
self.assertEqual(res.get("reason_code"), "authority_identity_unresolved")
|
||||||
|
|
||||||
|
def test_expected_username_is_never_substituted_for_authentication(self):
|
||||||
|
with mock.patch.object(
|
||||||
|
gms, "get_profile",
|
||||||
|
return_value={"profile_name": PROFILE, "username": IDENTITY},
|
||||||
|
), mock.patch.object(gms, "_authenticated_username", return_value=None), \
|
||||||
|
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||||
|
return_value={"expected_username": IDENTITY}):
|
||||||
|
self.assertIsNone(gms._active_username("gitea.prgs.cc"))
|
||||||
|
|
||||||
|
def test_accessors_share_one_snapshot(self):
|
||||||
|
with mock.patch.object(gms, "get_profile",
|
||||||
|
return_value={"profile_name": PROFILE}), \
|
||||||
|
mock.patch.object(gms, "_authenticated_username",
|
||||||
|
return_value=IDENTITY), \
|
||||||
|
mock.patch.object(gms.session_ctx, "get_session_context",
|
||||||
|
return_value=self._ctx()):
|
||||||
|
self.assertEqual(gms._active_username("gitea.prgs.cc"), IDENTITY)
|
||||||
|
self.assertEqual(gms._active_profile_name("gitea.prgs.cc"), PROFILE)
|
||||||
|
|
||||||
|
|
||||||
|
class AuthorMutationBlockTests(unittest.TestCase):
|
||||||
|
"""Preserved: the structured refusal shape review 622 confirmed correct."""
|
||||||
|
|
||||||
|
def test_matches_the_sibling_refusal_shape(self):
|
||||||
|
res = gms._author_mutation_block(["stopped"])
|
||||||
|
self.assertIs(res["success"], False)
|
||||||
|
self.assertIs(res["performed"], False)
|
||||||
|
self.assertEqual(res["outcome"], "REFUSED")
|
||||||
|
self.assertEqual(res["reasons"], ["stopped"])
|
||||||
|
|
||||||
|
def test_carries_reason_code_and_transport_fields(self):
|
||||||
|
res = gms._author_mutation_block(
|
||||||
|
["nope"], reason_code="authority_identity_drift",
|
||||||
|
retryable=False, transport_survives=True,
|
||||||
|
expected="a", actual="b", issue_number=943,
|
||||||
|
)
|
||||||
|
self.assertEqual(res["reason_code"], "authority_identity_drift")
|
||||||
|
self.assertIs(res["retryable"], False)
|
||||||
|
self.assertIs(res["transport_survives"], True)
|
||||||
|
self.assertEqual((res["expected"], res["actual"]), ("a", "b"))
|
||||||
|
self.assertEqual(res["issue_number"], 943)
|
||||||
|
self.assertIs(res["success"], False)
|
||||||
|
|
||||||
|
|
||||||
|
class RuntimeHelperResolutionTests(unittest.TestCase):
|
||||||
|
"""Every runtime helper the wrapper references is defined and callable.
|
||||||
|
|
||||||
|
Supplements the runtime coverage above; it does not replace it.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def test_named_helpers_are_defined_and_callable(self):
|
||||||
|
for name in RUNTIME_HELPERS:
|
||||||
|
with self.subTest(helper=name):
|
||||||
|
self.assertTrue(hasattr(gms, name), f"{name} is not defined")
|
||||||
|
self.assertTrue(callable(getattr(gms, name)))
|
||||||
|
|
||||||
|
def test_every_global_referenced_by_the_wrapper_resolves(self):
|
||||||
|
"""The generalised form of the round-1 defect: an unresolvable global."""
|
||||||
|
fn = _wrapper_ast()
|
||||||
|
bound: set[str] = {a.arg for a in fn.args.args}
|
||||||
|
bound |= {a.arg for a in fn.args.kwonlyargs}
|
||||||
|
if fn.args.vararg:
|
||||||
|
bound.add(fn.args.vararg.arg)
|
||||||
|
if fn.args.kwarg:
|
||||||
|
bound.add(fn.args.kwarg.arg)
|
||||||
|
for node in ast.walk(fn):
|
||||||
|
if isinstance(node, ast.Name) and isinstance(
|
||||||
|
node.ctx, (ast.Store, ast.Del)
|
||||||
|
):
|
||||||
|
bound.add(node.id)
|
||||||
|
elif isinstance(node, (ast.Import, ast.ImportFrom)):
|
||||||
|
for alias in node.names:
|
||||||
|
bound.add((alias.asname or alias.name).split(".")[0])
|
||||||
|
elif isinstance(node, ast.ExceptHandler) and node.name:
|
||||||
|
bound.add(node.name)
|
||||||
|
|
||||||
|
unresolved = sorted(
|
||||||
|
node.id
|
||||||
|
for node in ast.walk(fn)
|
||||||
|
if isinstance(node, ast.Name)
|
||||||
|
and isinstance(node.ctx, ast.Load)
|
||||||
|
and node.id not in bound
|
||||||
|
and not hasattr(gms, node.id)
|
||||||
|
and not hasattr(builtins, node.id)
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
unresolved, [],
|
||||||
|
f"{WRAPPER_NAME} references undefined globals: {unresolved}",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_wrapper_wires_the_authority_and_session_resolvers(self):
|
||||||
|
fn = _wrapper_ast()
|
||||||
|
called = {
|
||||||
|
node.func.id
|
||||||
|
for node in ast.walk(fn)
|
||||||
|
if isinstance(node, ast.Call) and isinstance(node.func, ast.Name)
|
||||||
|
}
|
||||||
|
self.assertIn("_active_mutation_authority", called)
|
||||||
|
self.assertIn("_resolve_owner_workflow_session", called)
|
||||||
|
self.assertIn("_author_mutation_block", called)
|
||||||
|
|
||||||
|
def test_wrapper_accepts_an_optional_session_id(self):
|
||||||
|
"""ABI addition stays backward compatible: optional, defaulting to None."""
|
||||||
|
fn = _wrapper_ast()
|
||||||
|
names = [a.arg for a in fn.args.args]
|
||||||
|
self.assertIn("session_id", names)
|
||||||
|
offset = len(names) - len(fn.args.defaults)
|
||||||
|
default = fn.args.defaults[names.index("session_id") - offset]
|
||||||
|
self.assertIsInstance(default, ast.Constant)
|
||||||
|
self.assertIsNone(default.value)
|
||||||
|
|
||||||
|
|
||||||
|
class Issue941ScopeGuardNotRegressedTests(unittest.TestCase):
|
||||||
|
"""Preserved: PR #942's bootstrap-scope wiring still holds."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self._tmp = tempfile.TemporaryDirectory()
|
||||||
|
self.addCleanup(self._tmp.cleanup)
|
||||||
|
self.repo, self.head = _make_control_repo(self._tmp.name)
|
||||||
|
|
||||||
|
def _assessment(self, task: str = BOOTSTRAP_TASK) -> dict:
|
||||||
|
return aib.assess_author_issue_bootstrap(
|
||||||
|
workspace_path=self.repo,
|
||||||
|
canonical_repo_root=self.repo,
|
||||||
|
current_branch="master",
|
||||||
|
head_sha=self.head,
|
||||||
|
porcelain_status="",
|
||||||
|
remote_master_sha=self.head,
|
||||||
|
task=task,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_bootstrap_task_still_permitted_from_clean_control_checkout(self):
|
||||||
|
res = workflow_scope_guard.assess_root_source_mutation(
|
||||||
|
workspace_path=self.repo,
|
||||||
|
canonical_repo_root=self.repo,
|
||||||
|
role_kind="author",
|
||||||
|
mutation_task=BOOTSTRAP_TASK,
|
||||||
|
porcelain_status="",
|
||||||
|
bootstrap_assessment=self._assessment(),
|
||||||
|
)
|
||||||
|
self.assertFalse(res.get("block"), res)
|
||||||
|
self.assertNotEqual(
|
||||||
|
res.get("blocker_kind"), workflow_scope_guard.BLOCKER_MISSING_WORKTREE
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_bootstrap_task_still_blocked_without_evidence(self):
|
||||||
|
res = workflow_scope_guard.assess_root_source_mutation(
|
||||||
|
workspace_path=self.repo,
|
||||||
|
canonical_repo_root=self.repo,
|
||||||
|
role_kind="author",
|
||||||
|
mutation_task=BOOTSTRAP_TASK,
|
||||||
|
porcelain_status="",
|
||||||
|
)
|
||||||
|
self.assertTrue(res.get("block"))
|
||||||
|
self.assertEqual(
|
||||||
|
res.get("blocker_kind"), workflow_scope_guard.BLOCKER_MISSING_WORKTREE
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_ordinary_author_mutation_still_blocked_from_control_checkout(self):
|
||||||
|
res = workflow_scope_guard.assess_root_source_mutation(
|
||||||
|
workspace_path=self.repo,
|
||||||
|
canonical_repo_root=self.repo,
|
||||||
|
role_kind="author",
|
||||||
|
mutation_task="commit_files",
|
||||||
|
porcelain_status="",
|
||||||
|
bootstrap_assessment=self._assessment(),
|
||||||
|
)
|
||||||
|
self.assertTrue(res.get("block"))
|
||||||
|
self.assertEqual(
|
||||||
|
res.get("blocker_kind"), workflow_scope_guard.BLOCKER_MISSING_WORKTREE
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_create_issue_bootstrap_unchanged(self):
|
||||||
|
self.assertTrue(cib.is_create_issue_task("create_issue"))
|
||||||
|
self.assertFalse(cib.is_create_issue_task(BOOTSTRAP_TASK))
|
||||||
|
|
||||||
|
|
||||||
|
def _write_json(path: str, payload: dict) -> None:
|
||||||
|
"""Write an issue-lock file directly, for lock-precedence tests."""
|
||||||
|
import json
|
||||||
|
|
||||||
|
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||||
|
with open(path, "w", encoding="utf-8") as fh:
|
||||||
|
json.dump(payload, fh)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,685 @@
|
|||||||
|
import sys as _sys
|
||||||
|
from pathlib import Path as _Path
|
||||||
|
_sys.path.insert(0, str(_Path(__file__).resolve().parent))
|
||||||
|
from mutation_profile_fixture import shared_mutation_env # noqa: E402
|
||||||
|
"""The renewal waiver reaches the real enforcement paths (#945 B1).
|
||||||
|
|
||||||
|
``tests/test_issue_945_owning_pr_renewal_continuation.py`` proves the pure
|
||||||
|
pieces: that ``issue_lock_renewal.owning_pr_renewal_from_lock`` rebuilds a
|
||||||
|
renewal waiver, that ``_owning_pr_continuation_from_lock`` resolves the two
|
||||||
|
dispositions in the right precedence, and that the duplicate gate honours the
|
||||||
|
resulting token. None of that proves any *production* path consumes the
|
||||||
|
resolver, and review ``623`` demonstrated the gap by reverting the primary call
|
||||||
|
site at ``gitea_mcp_server.py:2894`` back to the recovery-only rebuild: the
|
||||||
|
whole repository stayed green, failing-test ids byte identical.
|
||||||
|
|
||||||
|
This file closes that hole. Every test here starts from a real durable lock
|
||||||
|
file written to a temporary lock directory and bound to this process's session
|
||||||
|
pointer, then calls the authoritative production entry point — not a helper:
|
||||||
|
|
||||||
|
* ``mcp_server._enforce_locked_issue_duplicate_recheck`` — the shared recheck
|
||||||
|
behind ``gitea_commit_files`` and ``gitea_create_pr``
|
||||||
|
* ``mcp_server.gitea_assess_work_issue_duplicate`` — the read-only assessor
|
||||||
|
* ``mcp_server._prove_author_ownership_for_pr`` — the push / PR-update
|
||||||
|
ownership prover, which is also the existing-PR continuation path
|
||||||
|
|
||||||
|
Only the external boundaries are mocked: Gitea HTTP reads (the duplicate
|
||||||
|
context fetcher, open-PR and branch listings) and the credential header. The
|
||||||
|
reconstruction and enforcement chain under test — lock load, evidence rebuild,
|
||||||
|
resolver precedence, and ``issue_work_duplicate_gate`` — runs for real.
|
||||||
|
|
||||||
|
``TestRevertingThePrimaryWiringIsDetected`` is the explicit regression the
|
||||||
|
review asked for: it reproduces the pre-#945 recovery-only call site and
|
||||||
|
asserts the enforcement path then refuses, so the wiring cannot be removed
|
||||||
|
silently.
|
||||||
|
|
||||||
|
Everything is written under ``tempfile.TemporaryDirectory``. No branch,
|
||||||
|
worktree, PR, comment, lease, or lock outside that directory is created, and no
|
||||||
|
production Gitea or control-plane state is touched (#945 AC18).
|
||||||
|
"""
|
||||||
|
import os
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
from datetime import datetime, timedelta, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
from unittest.mock import patch
|
||||||
|
|
||||||
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||||
|
|
||||||
|
import issue_lock_provenance # noqa: E402
|
||||||
|
import issue_lock_recovery # noqa: E402
|
||||||
|
import issue_lock_renewal # noqa: E402
|
||||||
|
import issue_lock_store # noqa: E402
|
||||||
|
import mcp_server # noqa: E402
|
||||||
|
from issue_work_duplicate_gate import ( # noqa: E402
|
||||||
|
PHASE_COMMIT,
|
||||||
|
PHASE_CREATE_PR,
|
||||||
|
PHASE_LOCK,
|
||||||
|
PHASE_PUSH,
|
||||||
|
)
|
||||||
|
|
||||||
|
ISSUE = 4948
|
||||||
|
OWNING_PR = 4949
|
||||||
|
OTHER_PR = 4950
|
||||||
|
OTHER_ISSUE = 4951
|
||||||
|
BRANCH = f"fix/issue-{ISSUE}-renewal-wiring"
|
||||||
|
OTHER_BRANCH = f"fix/issue-{ISSUE}-competing"
|
||||||
|
HEAD = "e" * 40
|
||||||
|
OTHER_HEAD = "f" * 40
|
||||||
|
IDENTITY = "example-user"
|
||||||
|
PROFILE = "test-author-prgs"
|
||||||
|
ORG = "Scaled-Tech-Consulting"
|
||||||
|
REPO = "Gitea-Tools"
|
||||||
|
HOST = "gitea.prgs.cc"
|
||||||
|
|
||||||
|
|
||||||
|
def dead_pid() -> int:
|
||||||
|
"""A PID that has certainly exited (spawned, then reaped)."""
|
||||||
|
proc = subprocess.Popen([sys.executable, "-c", "pass"])
|
||||||
|
proc.wait()
|
||||||
|
return proc.pid
|
||||||
|
|
||||||
|
|
||||||
|
def shifted_ts(hours: int = 4) -> str:
|
||||||
|
return (
|
||||||
|
(datetime.now(timezone.utc) + timedelta(hours=hours))
|
||||||
|
.isoformat()
|
||||||
|
.replace("+00:00", "Z")
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def owning_pr(number=OWNING_PR, ref=BRANCH, sha=HEAD, issue=ISSUE):
|
||||||
|
return {
|
||||||
|
"number": number,
|
||||||
|
"title": f"fix: something (Closes #{issue})",
|
||||||
|
"body": f"Closes #{issue}.",
|
||||||
|
"head": {"ref": ref, "sha": sha},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def renewal_block(
|
||||||
|
*,
|
||||||
|
pr_number=OWNING_PR,
|
||||||
|
branch=BRANCH,
|
||||||
|
head=HEAD,
|
||||||
|
identity=IDENTITY,
|
||||||
|
profile=PROFILE,
|
||||||
|
):
|
||||||
|
"""The ``lease_renewal`` block ``build_renewal_record`` writes on success."""
|
||||||
|
return {
|
||||||
|
"renewed": True,
|
||||||
|
"renewed_at": shifted_ts(-1),
|
||||||
|
"prior_pid": 4242,
|
||||||
|
"prior_pid_alive": True,
|
||||||
|
"prior_expires_at": shifted_ts(-1),
|
||||||
|
"replacement_pid": os.getpid(),
|
||||||
|
"new_expires_at": shifted_ts(),
|
||||||
|
"identity": identity,
|
||||||
|
"profile": profile,
|
||||||
|
"branch_name": branch,
|
||||||
|
"worktree_path": os.path.realpath(os.getcwd()),
|
||||||
|
"head_sha": head,
|
||||||
|
"remote_head_sha": head,
|
||||||
|
"pr_head_sha": head,
|
||||||
|
"pr_number": pr_number,
|
||||||
|
"reason": "expired lease renewed by its exact recorded owner",
|
||||||
|
"proof": [],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def recovery_block(*, pr_number=OWNING_PR, branch=BRANCH, head=HEAD):
|
||||||
|
"""The ``dead_session_recovery`` block ``build_recovery_record`` writes."""
|
||||||
|
return {
|
||||||
|
"recovered": True,
|
||||||
|
"reason": "owning MCP session exited; durable ownership evidence matched",
|
||||||
|
"recovered_at": shifted_ts(-1),
|
||||||
|
"prior_session_pid": 4242,
|
||||||
|
"replacement_session_pid": os.getpid(),
|
||||||
|
"prior_pid_alive": False,
|
||||||
|
"branch_name": branch,
|
||||||
|
"pr_number": pr_number,
|
||||||
|
"pr_head": head,
|
||||||
|
"recorded_head": head,
|
||||||
|
"accepted_head": head,
|
||||||
|
"head_relation": issue_lock_recovery.HEAD_RELATION_EQUAL,
|
||||||
|
"identity": IDENTITY,
|
||||||
|
"profile": PROFILE,
|
||||||
|
"proof": [],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class EnforcementPathBase(unittest.TestCase):
|
||||||
|
"""Drives production enforcement entry points against a real durable lock.
|
||||||
|
|
||||||
|
The lock lives in a throwaway directory and is bound to this process's
|
||||||
|
session pointer exactly as ``gitea_lock_issue`` binds it, so
|
||||||
|
``_load_existing_issue_lock()`` resolves it through the ordinary
|
||||||
|
``read_session_issue_lock()`` path rather than a test shortcut.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self.lock_dir = tempfile.TemporaryDirectory()
|
||||||
|
self.addCleanup(self.lock_dir.cleanup)
|
||||||
|
self.worktree = os.path.realpath(os.getcwd())
|
||||||
|
self.remotes = patch.dict(
|
||||||
|
mcp_server.REMOTES,
|
||||||
|
{"prgs": {"host": HOST, "org": ORG, "repo": REPO}},
|
||||||
|
)
|
||||||
|
self.remotes.start()
|
||||||
|
self.addCleanup(patch.stopall)
|
||||||
|
mcp_server._IDENTITY_CACHE.clear()
|
||||||
|
|
||||||
|
# ── fixtures ────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
def build_lock(
|
||||||
|
self,
|
||||||
|
*,
|
||||||
|
issue_number=ISSUE,
|
||||||
|
branch=BRANCH,
|
||||||
|
renewal=None,
|
||||||
|
recovery=None,
|
||||||
|
claimant=None,
|
||||||
|
pid=None,
|
||||||
|
live=True,
|
||||||
|
):
|
||||||
|
pid = os.getpid() if pid is None else pid
|
||||||
|
claimant = claimant or {"username": IDENTITY, "profile": PROFILE}
|
||||||
|
expires = shifted_ts() if live else shifted_ts(-1)
|
||||||
|
data = {
|
||||||
|
"issue_number": issue_number,
|
||||||
|
"branch_name": branch,
|
||||||
|
"remote": "prgs",
|
||||||
|
"org": ORG,
|
||||||
|
"repo": REPO,
|
||||||
|
"worktree_path": self.worktree,
|
||||||
|
"session_pid": pid,
|
||||||
|
"pid": pid,
|
||||||
|
"claimant": dict(claimant),
|
||||||
|
"work_lease": {
|
||||||
|
"operation_type": issue_lock_store.AUTHOR_ISSUE_WORK_LEASE,
|
||||||
|
"issue_number": issue_number,
|
||||||
|
"branch": branch,
|
||||||
|
"worktree_path": self.worktree,
|
||||||
|
"claimant": dict(claimant),
|
||||||
|
"created_at": shifted_ts(-1),
|
||||||
|
"last_heartbeat_at": shifted_ts(0) if live else shifted_ts(-1),
|
||||||
|
"expires_at": expires,
|
||||||
|
},
|
||||||
|
"lock_provenance": issue_lock_provenance.build_sanctioned_lock_provenance(
|
||||||
|
tool="gitea_lock_issue",
|
||||||
|
claimant=dict(claimant),
|
||||||
|
),
|
||||||
|
}
|
||||||
|
if renewal is not None:
|
||||||
|
data["lease_renewal"] = renewal
|
||||||
|
if recovery is not None:
|
||||||
|
data["dead_session_recovery"] = recovery
|
||||||
|
return data
|
||||||
|
|
||||||
|
def bind(self, data):
|
||||||
|
"""Persist the lock and bind it to this process, as the server does."""
|
||||||
|
issue_lock_store.bind_session_lock(data, self.lock_dir.name)
|
||||||
|
return data
|
||||||
|
|
||||||
|
def env(self):
|
||||||
|
return shared_mutation_env(
|
||||||
|
PROFILE,
|
||||||
|
include_example_repo=True,
|
||||||
|
GITEA_ISSUE_LOCK_DIR=self.lock_dir.name,
|
||||||
|
)
|
||||||
|
|
||||||
|
def gitea_reads(self, *, open_prs, branch_names=None):
|
||||||
|
"""Patch only the external Gitea read boundary."""
|
||||||
|
branch_names = [BRANCH] if branch_names is None else branch_names
|
||||||
|
return (
|
||||||
|
patch("mcp_server.get_auth_header", return_value="token x"),
|
||||||
|
patch(
|
||||||
|
"mcp_server.issue_duplicate_context_fetcher",
|
||||||
|
side_effect=lambda h, o, r, auth, issue_number: (
|
||||||
|
list(open_prs), list(branch_names), {"status": "not_claimed"}
|
||||||
|
),
|
||||||
|
),
|
||||||
|
patch("mcp_server._list_open_pulls", return_value=list(open_prs)),
|
||||||
|
patch(
|
||||||
|
"mcp_server.api_get_all",
|
||||||
|
return_value=[
|
||||||
|
{"name": n, "commit": {"id": HEAD}} for n in branch_names
|
||||||
|
],
|
||||||
|
),
|
||||||
|
)
|
||||||
|
|
||||||
|
# ── production entry points ─────────────────────────────────────────────
|
||||||
|
|
||||||
|
def run_duplicate_recheck(self, *, phase, open_prs, branch_names=None):
|
||||||
|
"""The real shared recheck behind gitea_commit_files / gitea_create_pr."""
|
||||||
|
patches = self.gitea_reads(open_prs=open_prs, branch_names=branch_names)
|
||||||
|
with patches[0], patches[1], patches[2], patches[3]:
|
||||||
|
with patch.dict(os.environ, self.env(), clear=True):
|
||||||
|
os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name
|
||||||
|
return mcp_server._enforce_locked_issue_duplicate_recheck(
|
||||||
|
"prgs", phase, host=HOST, org=ORG, repo=REPO
|
||||||
|
)
|
||||||
|
|
||||||
|
def run_readonly_assessor(
|
||||||
|
self, *, open_prs, issue_number=ISSUE, branch=BRANCH, branch_names=None
|
||||||
|
):
|
||||||
|
"""The real read-only duplicate assessor MCP tool."""
|
||||||
|
patches = self.gitea_reads(open_prs=open_prs, branch_names=branch_names)
|
||||||
|
with patches[0], patches[1], patches[2], patches[3]:
|
||||||
|
with patch.dict(os.environ, self.env(), clear=True):
|
||||||
|
os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name
|
||||||
|
return mcp_server.gitea_assess_work_issue_duplicate(
|
||||||
|
issue_number=issue_number,
|
||||||
|
branch_name=branch,
|
||||||
|
phase=PHASE_COMMIT,
|
||||||
|
remote="prgs",
|
||||||
|
host=HOST,
|
||||||
|
org=ORG,
|
||||||
|
repo=REPO,
|
||||||
|
)
|
||||||
|
|
||||||
|
def run_ownership_prover(
|
||||||
|
self, *, pr_number=OWNING_PR, branch=BRANCH, issue_number=ISSUE
|
||||||
|
):
|
||||||
|
"""The real push / PR-update ownership prover (existing-PR continuation)."""
|
||||||
|
with patch("mcp_server.get_auth_header", return_value="token x"):
|
||||||
|
with patch.dict(os.environ, self.env(), clear=True):
|
||||||
|
os.environ["GITEA_ISSUE_LOCK_DIR"] = self.lock_dir.name
|
||||||
|
return mcp_server._prove_author_ownership_for_pr(
|
||||||
|
pr_number=pr_number,
|
||||||
|
pr_title=f"fix: something (Closes #{issue_number})",
|
||||||
|
pr_body=f"Closes #{issue_number}.",
|
||||||
|
source_branch=branch,
|
||||||
|
remote="prgs",
|
||||||
|
host=HOST,
|
||||||
|
org=ORG,
|
||||||
|
repo=REPO,
|
||||||
|
worktree_path=self.worktree,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# ─────────────── B1: renewal evidence reaches every enforcement path ───────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestRenewalReachesEnforcementPaths(EnforcementPathBase):
|
||||||
|
"""A renewal-only lock must exempt its owning PR at the real call sites.
|
||||||
|
|
||||||
|
Each of these fails if its call site is reverted to the recovery-only
|
||||||
|
rebuild, because the lock deliberately carries no ``dead_session_recovery``
|
||||||
|
block at all.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
super().setUp()
|
||||||
|
self.bind(self.build_lock(renewal=renewal_block()))
|
||||||
|
|
||||||
|
def test_commit_duplicate_recheck_permits_the_owning_pr(self):
|
||||||
|
blocked = self.run_duplicate_recheck(
|
||||||
|
phase=PHASE_COMMIT, open_prs=[owning_pr()]
|
||||||
|
)
|
||||||
|
self.assertIsNone(
|
||||||
|
blocked,
|
||||||
|
"commit recheck refused the PR the renewal already proved it owns; "
|
||||||
|
"the resolver is not wired into gitea_mcp_server:2894",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_create_pr_duplicate_recheck_permits_the_owning_pr(self):
|
||||||
|
blocked = self.run_duplicate_recheck(
|
||||||
|
phase=PHASE_CREATE_PR, open_prs=[owning_pr()]
|
||||||
|
)
|
||||||
|
self.assertIsNone(blocked)
|
||||||
|
|
||||||
|
def test_read_only_assessor_reports_the_same_exemption(self):
|
||||||
|
result = self.run_readonly_assessor(open_prs=[owning_pr()])
|
||||||
|
self.assertTrue(result["success"])
|
||||||
|
self.assertFalse(result["block"])
|
||||||
|
self.assertTrue(result["owning_pr_recovery_exempted"])
|
||||||
|
self.assertEqual(result["linked_open_pr"], OWNING_PR)
|
||||||
|
|
||||||
|
def test_push_ownership_prover_carries_the_renewal_evidence(self):
|
||||||
|
ownership = self.run_ownership_prover()
|
||||||
|
self.assertTrue(ownership["proven"], ownership["reasons"])
|
||||||
|
token = ownership["recovered_owning_pr"]
|
||||||
|
self.assertIsNotNone(
|
||||||
|
token,
|
||||||
|
"push prover produced no continuation evidence from a renewal lock; "
|
||||||
|
"the resolver is not wired into gitea_mcp_server:19464",
|
||||||
|
)
|
||||||
|
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||||
|
self.assertEqual(token["branch_name"], BRANCH)
|
||||||
|
self.assertEqual(token["head_sha"], HEAD)
|
||||||
|
|
||||||
|
def test_all_enforcement_paths_decide_alike_from_one_lock(self):
|
||||||
|
"""AC: commit, create-PR, assessor and prover agree on one lock."""
|
||||||
|
for phase in (PHASE_COMMIT, PHASE_CREATE_PR, PHASE_PUSH, PHASE_LOCK):
|
||||||
|
with self.subTest(phase=phase):
|
||||||
|
self.assertIsNone(
|
||||||
|
self.run_duplicate_recheck(phase=phase, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
assessor = self.run_readonly_assessor(open_prs=[owning_pr()])
|
||||||
|
prover = self.run_ownership_prover()
|
||||||
|
self.assertTrue(assessor["owning_pr_recovery_exempted"])
|
||||||
|
self.assertEqual(
|
||||||
|
assessor["linked_open_pr"], prover["recovered_owning_pr"]["pr_number"]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class TestDeadSessionRecoveryStillReachesEnforcementPaths(EnforcementPathBase):
|
||||||
|
"""#755/#768 recovery must be unchanged by the #945 resolver."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
super().setUp()
|
||||||
|
self.bind(self.build_lock(recovery=recovery_block()))
|
||||||
|
|
||||||
|
def test_commit_recheck_still_permits_a_recovered_owning_pr(self):
|
||||||
|
self.assertIsNone(
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_assessor_still_reports_the_recovery_exemption(self):
|
||||||
|
result = self.run_readonly_assessor(open_prs=[owning_pr()])
|
||||||
|
self.assertTrue(result["owning_pr_recovery_exempted"])
|
||||||
|
|
||||||
|
def test_prover_still_carries_recovery_evidence(self):
|
||||||
|
token = self.run_ownership_prover()["recovered_owning_pr"]
|
||||||
|
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||||
|
|
||||||
|
|
||||||
|
# ──────────────── B1: the explicit anti-revert regression test ────────────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestRevertingThePrimaryWiringIsDetected(EnforcementPathBase):
|
||||||
|
"""Reproduce the pre-#945 call site and prove the path then refuses.
|
||||||
|
|
||||||
|
Review ``623`` reverted ``gitea_mcp_server.py:2894`` from
|
||||||
|
``_owning_pr_continuation_from_lock`` to
|
||||||
|
``issue_lock_recovery.recovered_owning_pr_from_lock`` and found the entire
|
||||||
|
repository still green. Substituting exactly that pre-fix behaviour here
|
||||||
|
makes the enforcement path block, so the causal link between the resolver
|
||||||
|
and the gate's answer is asserted, not assumed.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
super().setUp()
|
||||||
|
self.bind(self.build_lock(renewal=renewal_block()))
|
||||||
|
|
||||||
|
def test_recovery_only_rebuild_reintroduces_the_945_refusal(self):
|
||||||
|
with patch.object(
|
||||||
|
mcp_server,
|
||||||
|
"_owning_pr_continuation_from_lock",
|
||||||
|
side_effect=issue_lock_recovery.recovered_owning_pr_from_lock,
|
||||||
|
):
|
||||||
|
blocked = self.run_duplicate_recheck(
|
||||||
|
phase=PHASE_COMMIT, open_prs=[owning_pr()]
|
||||||
|
)
|
||||||
|
self.assertIsNotNone(
|
||||||
|
blocked,
|
||||||
|
"the pre-#945 recovery-only rebuild must lose the renewal waiver; "
|
||||||
|
"if this passes, the enforcement path is not consuming the resolver",
|
||||||
|
)
|
||||||
|
self.assertTrue(blocked["block"])
|
||||||
|
self.assertFalse(blocked["owning_pr_recovery_exempted"])
|
||||||
|
|
||||||
|
def test_restoring_the_resolver_restores_continuation(self):
|
||||||
|
self.assertIsNone(
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_read_only_assessor_is_wired_to_the_same_resolver(self):
|
||||||
|
with patch.object(
|
||||||
|
mcp_server,
|
||||||
|
"_owning_pr_continuation_from_lock",
|
||||||
|
side_effect=issue_lock_recovery.recovered_owning_pr_from_lock,
|
||||||
|
):
|
||||||
|
result = self.run_readonly_assessor(open_prs=[owning_pr()])
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
self.assertFalse(result["owning_pr_recovery_exempted"])
|
||||||
|
|
||||||
|
def test_push_prover_is_wired_to_the_same_resolver(self):
|
||||||
|
with patch.object(
|
||||||
|
mcp_server,
|
||||||
|
"_owning_pr_continuation_from_lock",
|
||||||
|
side_effect=issue_lock_recovery.recovered_owning_pr_from_lock,
|
||||||
|
):
|
||||||
|
ownership = self.run_ownership_prover()
|
||||||
|
self.assertIsNone(ownership["recovered_owning_pr"])
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────── B1: the exemption is not widened at the call sites ─────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestEnforcementPathsStillFailClosed(EnforcementPathBase):
|
||||||
|
def assert_blocked(self, result):
|
||||||
|
self.assertIsNotNone(result, "expected a fail-closed refusal")
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
return result
|
||||||
|
|
||||||
|
def test_open_pr_alone_grants_no_exemption(self):
|
||||||
|
"""No renewal and no recovery block: the open PR still blocks."""
|
||||||
|
self.bind(self.build_lock())
|
||||||
|
blocked = self.assert_blocked(
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
self.assertFalse(blocked["owning_pr_recovery_exempted"])
|
||||||
|
|
||||||
|
def test_second_pr_is_refused(self):
|
||||||
|
self.bind(self.build_lock(renewal=renewal_block()))
|
||||||
|
self.assert_blocked(
|
||||||
|
self.run_duplicate_recheck(
|
||||||
|
phase=PHASE_COMMIT,
|
||||||
|
open_prs=[owning_pr(), owning_pr(number=OTHER_PR, ref=OTHER_BRANCH)],
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_evidence_naming_another_pr_is_refused(self):
|
||||||
|
self.bind(self.build_lock(renewal=renewal_block(pr_number=OTHER_PR)))
|
||||||
|
self.assert_blocked(
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_unrelated_branch_is_refused(self):
|
||||||
|
self.bind(self.build_lock(renewal=renewal_block(branch=OTHER_BRANCH)))
|
||||||
|
self.assert_blocked(
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_live_head_divergence_is_refused(self):
|
||||||
|
"""Force-push or unrelated remote movement: live PR head no longer matches."""
|
||||||
|
self.bind(self.build_lock(renewal=renewal_block()))
|
||||||
|
self.assert_blocked(
|
||||||
|
self.run_duplicate_recheck(
|
||||||
|
phase=PHASE_COMMIT, open_prs=[owning_pr(sha=OTHER_HEAD)]
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_stale_recorded_head_is_refused(self):
|
||||||
|
"""The renewal names a head the live PR never had."""
|
||||||
|
self.bind(self.build_lock(renewal=renewal_block(head=OTHER_HEAD)))
|
||||||
|
self.assert_blocked(
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_local_remote_head_divergence_is_refused(self):
|
||||||
|
record = renewal_block()
|
||||||
|
record["remote_head_sha"] = OTHER_HEAD
|
||||||
|
self.bind(self.build_lock(renewal=record))
|
||||||
|
self.assert_blocked(
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_identity_mismatch_with_the_lock_claimant_is_refused(self):
|
||||||
|
self.bind(self.build_lock(renewal=renewal_block(identity="someone-else")))
|
||||||
|
self.assert_blocked(
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_profile_mismatch_with_the_lock_claimant_is_refused(self):
|
||||||
|
self.bind(self.build_lock(renewal=renewal_block(profile="test-reviewer-prgs")))
|
||||||
|
self.assert_blocked(
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_ungranted_renewal_block_is_refused(self):
|
||||||
|
record = renewal_block()
|
||||||
|
record["renewed"] = False
|
||||||
|
self.bind(self.build_lock(renewal=record))
|
||||||
|
self.assert_blocked(
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_malformed_renewal_block_is_refused(self):
|
||||||
|
record = renewal_block()
|
||||||
|
record["pr_number"] = "not-a-number"
|
||||||
|
self.bind(self.build_lock(renewal=record))
|
||||||
|
self.assert_blocked(
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_wrong_issue_evidence_cannot_be_copied_onto_another_lock(self):
|
||||||
|
"""A renewal block copied onto a lock for a different issue proves nothing.
|
||||||
|
|
||||||
|
The rebuilt token takes its ``issue_number`` from the lock it is found
|
||||||
|
on, not from the record, so a block lifted onto another issue's lock
|
||||||
|
claims that issue while still naming the original PR. That copied
|
||||||
|
evidence must not waive the genuine duplicate the other issue has.
|
||||||
|
"""
|
||||||
|
self.bind(
|
||||||
|
self.build_lock(issue_number=OTHER_ISSUE, renewal=renewal_block())
|
||||||
|
)
|
||||||
|
blocked = self.assert_blocked(
|
||||||
|
self.run_duplicate_recheck(
|
||||||
|
phase=PHASE_COMMIT,
|
||||||
|
# The real open PR for OTHER_ISSUE is a different PR entirely.
|
||||||
|
open_prs=[
|
||||||
|
owning_pr(number=OTHER_PR, ref=OTHER_BRANCH, issue=OTHER_ISSUE)
|
||||||
|
],
|
||||||
|
branch_names=[OTHER_BRANCH],
|
||||||
|
)
|
||||||
|
)
|
||||||
|
self.assertFalse(blocked["owning_pr_recovery_exempted"])
|
||||||
|
|
||||||
|
def test_refusal_carries_complete_structured_fields(self):
|
||||||
|
self.bind(self.build_lock())
|
||||||
|
blocked = self.assert_blocked(
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
for field in (
|
||||||
|
"block",
|
||||||
|
"outcome",
|
||||||
|
"reasons",
|
||||||
|
"owning_pr_recovery_exempted",
|
||||||
|
"owning_pr_recovery_notes",
|
||||||
|
"linked_open_pr",
|
||||||
|
"linked_open_pr_count",
|
||||||
|
):
|
||||||
|
with self.subTest(field=field):
|
||||||
|
self.assertIn(field, blocked)
|
||||||
|
self.assertTrue(blocked["reasons"])
|
||||||
|
|
||||||
|
|
||||||
|
class TestSequentialTasksStayIsolated(EnforcementPathBase):
|
||||||
|
"""One long-lived daemon serves many tasks; a waiver must not leak forward."""
|
||||||
|
|
||||||
|
def test_a_later_lock_without_evidence_does_not_inherit_the_earlier_waiver(self):
|
||||||
|
self.bind(self.build_lock(renewal=renewal_block()))
|
||||||
|
self.assertIsNone(
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
)
|
||||||
|
|
||||||
|
# Second task in the same process: a fresh lock, no renewal evidence.
|
||||||
|
self.bind(
|
||||||
|
self.build_lock(issue_number=OTHER_ISSUE, branch=OTHER_BRANCH)
|
||||||
|
)
|
||||||
|
blocked = self.run_duplicate_recheck(
|
||||||
|
phase=PHASE_COMMIT,
|
||||||
|
open_prs=[owning_pr(number=OTHER_PR, ref=OTHER_BRANCH, issue=OTHER_ISSUE)],
|
||||||
|
branch_names=[OTHER_BRANCH],
|
||||||
|
)
|
||||||
|
self.assertIsNotNone(blocked)
|
||||||
|
self.assertFalse(blocked["owning_pr_recovery_exempted"])
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────── F2: what the caller binding actually is, and is not ────────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestCallerBindingIsStructuralNotFieldComparison(unittest.TestCase):
|
||||||
|
"""Document, in executable form, the binding this patch really provides.
|
||||||
|
|
||||||
|
Review ``623`` found that the claimant check in
|
||||||
|
``owning_pr_renewal_from_lock`` compares two fields of one server-written
|
||||||
|
lock file and is therefore not bound to the authenticated caller. That is
|
||||||
|
correct, and these tests assert the true guarantee rather than the
|
||||||
|
overstated one: lock *selection* is process-scoped, and the claimant check
|
||||||
|
is an internal-consistency check.
|
||||||
|
|
||||||
|
No PID-derived, cached, or process-lifetime session authority is invented
|
||||||
|
here — the process scoping asserted below is pre-existing behaviour of
|
||||||
|
``issue_lock_store``, not something this patch adds.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def test_lock_selection_is_keyed_to_the_operating_system_process(self):
|
||||||
|
with tempfile.TemporaryDirectory() as root:
|
||||||
|
pointer = issue_lock_store.session_pointer_path(root)
|
||||||
|
self.assertEqual(
|
||||||
|
os.path.basename(pointer), f"session-{os.getpid()}.json"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_a_lock_bound_by_another_process_is_not_reachable(self):
|
||||||
|
"""The structural protection: a foreign session pointer is not read."""
|
||||||
|
with tempfile.TemporaryDirectory() as root:
|
||||||
|
foreign_pointer = os.path.join(root, f"session-{os.getpid() + 1}.json")
|
||||||
|
issue_lock_store.save_lock_file(
|
||||||
|
foreign_pointer, {"lock_file_path": "/nonexistent/foreign.json"}
|
||||||
|
)
|
||||||
|
self.assertIsNone(issue_lock_store.read_session_issue_lock(root))
|
||||||
|
|
||||||
|
def test_claimant_check_does_not_consult_the_live_authenticated_caller(self):
|
||||||
|
"""The honest limit: agreement is internal to the lock document.
|
||||||
|
|
||||||
|
A renewal block whose identity/profile agree with the claimant recorded
|
||||||
|
on the same lock rebuilds successfully, regardless of who is
|
||||||
|
authenticated. Live identity and profile are enforced by the separate
|
||||||
|
mutation-authority and profile gates, not by this rebuild.
|
||||||
|
"""
|
||||||
|
lock = {
|
||||||
|
"issue_number": ISSUE,
|
||||||
|
"branch_name": BRANCH,
|
||||||
|
"claimant": {"username": "unrelated-recorded-user", "profile": PROFILE},
|
||||||
|
"lease_renewal": renewal_block(identity="unrelated-recorded-user"),
|
||||||
|
}
|
||||||
|
token = issue_lock_renewal.owning_pr_renewal_from_lock(lock)
|
||||||
|
self.assertIsNotNone(token)
|
||||||
|
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||||
|
|
||||||
|
def test_internal_disagreement_is_what_the_check_actually_rejects(self):
|
||||||
|
lock = {
|
||||||
|
"issue_number": ISSUE,
|
||||||
|
"branch_name": BRANCH,
|
||||||
|
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||||
|
"lease_renewal": renewal_block(identity="someone-else"),
|
||||||
|
}
|
||||||
|
self.assertIsNone(issue_lock_renewal.owning_pr_renewal_from_lock(lock))
|
||||||
|
|
||||||
|
|
||||||
|
class TestNoDurableArtifacts(EnforcementPathBase):
|
||||||
|
def test_enforcement_runs_leave_nothing_outside_the_temp_lock_dir(self):
|
||||||
|
before = sorted(os.listdir(self.lock_dir.name))
|
||||||
|
self.bind(self.build_lock(renewal=renewal_block()))
|
||||||
|
self.run_duplicate_recheck(phase=PHASE_COMMIT, open_prs=[owning_pr()])
|
||||||
|
self.run_ownership_prover()
|
||||||
|
after = sorted(os.listdir(self.lock_dir.name))
|
||||||
|
self.assertNotEqual(before, after, "the test must have written its lock")
|
||||||
|
self.assertTrue(
|
||||||
|
all(
|
||||||
|
os.path.realpath(os.path.join(self.lock_dir.name, name)).startswith(
|
||||||
|
os.path.realpath(self.lock_dir.name)
|
||||||
|
)
|
||||||
|
for name in after
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,602 @@
|
|||||||
|
import sys as _sys
|
||||||
|
from pathlib import Path as _Path
|
||||||
|
_sys.path.insert(0, str(_Path(__file__).resolve().parent))
|
||||||
|
from mutation_profile_fixture import shared_mutation_env # noqa: F401,E402
|
||||||
|
"""Exact-owner renewal keeps its owning-PR waiver past lock_issue (#945).
|
||||||
|
|
||||||
|
#755 taught the duplicate-work gate that a sanctioned *dead-session recovery*
|
||||||
|
owns its open PR, and #768 taught the later gates to rebuild that proof from the
|
||||||
|
durable lock. #760 added the exact-owner *renewal* disposition and granted it
|
||||||
|
the same waiver inside ``gitea_lock_issue`` — but never added the matching
|
||||||
|
rebuild. So an ordinary renewal held the waiver only for the duration of the
|
||||||
|
lock call: ``_enforce_locked_issue_duplicate_recheck`` asked
|
||||||
|
``recovered_owning_pr_from_lock``, which reads only ``dead_session_recovery``,
|
||||||
|
and the very next commit was refused ``duplicate_commit_prevented`` with
|
||||||
|
``owning_pr_recovery_exempted: false`` on the PR the renewal had just proved.
|
||||||
|
|
||||||
|
``TestPreFixReproduction`` pins that defect directly: the recovery-only rebuild
|
||||||
|
still returns ``None`` for a renewal lock, which is exactly why the gates lost
|
||||||
|
the waiver. Everything else proves the renewal half now survives, that recovery
|
||||||
|
is unchanged, and that no path grants an exemption on weaker evidence.
|
||||||
|
|
||||||
|
Every fixture here is an in-memory mapping. Nothing writes a branch, worktree,
|
||||||
|
lock file, lease, comment, or PR (#945 AC18).
|
||||||
|
"""
|
||||||
|
import copy
|
||||||
|
import sys
|
||||||
|
import unittest
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||||
|
|
||||||
|
import gitea_mcp_server # noqa: E402
|
||||||
|
import issue_lock_recovery # noqa: E402
|
||||||
|
import issue_lock_renewal # noqa: E402
|
||||||
|
from issue_work_duplicate_gate import ( # noqa: E402
|
||||||
|
OUTCOME_DUPLICATE_WORK_NOT_PREVENTED,
|
||||||
|
PHASE_COMMIT,
|
||||||
|
PHASE_CREATE_PR,
|
||||||
|
PHASE_LOCK,
|
||||||
|
PHASE_PUSH,
|
||||||
|
assess_work_issue_duplicate_gate,
|
||||||
|
)
|
||||||
|
|
||||||
|
ISSUE = 4945
|
||||||
|
OWNING_PR = 4946
|
||||||
|
OTHER_PR = 4947
|
||||||
|
BRANCH = f"fix/issue-{ISSUE}-owning-pr-renewal"
|
||||||
|
OTHER_BRANCH = f"fix/issue-{ISSUE}-competing"
|
||||||
|
HEAD = "a" * 40
|
||||||
|
OTHER_HEAD = "b" * 40
|
||||||
|
IDENTITY = "example-user"
|
||||||
|
PROFILE = "test-author-prgs"
|
||||||
|
|
||||||
|
|
||||||
|
def renewal_record(**overrides):
|
||||||
|
"""The ``lease_renewal`` block ``build_renewal_record`` writes on success."""
|
||||||
|
record = {
|
||||||
|
"renewed": True,
|
||||||
|
"renewed_at": "2026-01-01T00:00:00Z",
|
||||||
|
"prior_pid": 4242,
|
||||||
|
"prior_pid_alive": True,
|
||||||
|
"prior_expires_at": "2026-01-01T00:00:00Z",
|
||||||
|
"replacement_pid": 4243,
|
||||||
|
"new_expires_at": "2026-01-01T00:10:00Z",
|
||||||
|
"identity": IDENTITY,
|
||||||
|
"profile": PROFILE,
|
||||||
|
"branch_name": BRANCH,
|
||||||
|
"worktree_path": f"branches/issue-{ISSUE}-owning-pr-renewal",
|
||||||
|
"head_sha": HEAD,
|
||||||
|
"remote_head_sha": HEAD,
|
||||||
|
"pr_head_sha": HEAD,
|
||||||
|
"pr_number": OWNING_PR,
|
||||||
|
"reason": "expired lease renewed by its exact recorded owner",
|
||||||
|
"proof": [],
|
||||||
|
}
|
||||||
|
record.update(overrides)
|
||||||
|
return record
|
||||||
|
|
||||||
|
|
||||||
|
def renewal_lock(record=None, *, issue_number=ISSUE, claimant=True, **lock_overrides):
|
||||||
|
lock = {
|
||||||
|
"issue_number": issue_number,
|
||||||
|
"branch_name": BRANCH,
|
||||||
|
"lease_renewal": renewal_record() if record is None else record,
|
||||||
|
}
|
||||||
|
if claimant:
|
||||||
|
lock["claimant"] = {"username": IDENTITY, "profile": PROFILE}
|
||||||
|
lock.update(lock_overrides)
|
||||||
|
return lock
|
||||||
|
|
||||||
|
|
||||||
|
def recovery_lock(pr_number=OWNING_PR, head=HEAD):
|
||||||
|
"""A lock carrying sanctioned dead-session recovery evidence (#755/#768)."""
|
||||||
|
return {
|
||||||
|
"issue_number": ISSUE,
|
||||||
|
"branch_name": BRANCH,
|
||||||
|
"claimant": {"username": IDENTITY, "profile": PROFILE},
|
||||||
|
"dead_session_recovery": {
|
||||||
|
"recovered": True,
|
||||||
|
"branch_name": BRANCH,
|
||||||
|
"pr_number": pr_number,
|
||||||
|
"pr_head": head,
|
||||||
|
"recorded_head": head,
|
||||||
|
"accepted_head": head,
|
||||||
|
"head_relation": issue_lock_recovery.HEAD_RELATION_EQUAL,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def owning_pr(number=OWNING_PR, ref=BRANCH, sha=HEAD, issue=ISSUE):
|
||||||
|
return {
|
||||||
|
"number": number,
|
||||||
|
"title": f"fix: something (Closes #{issue})",
|
||||||
|
"body": f"Closes #{issue}.",
|
||||||
|
"head": {"ref": ref, "sha": sha},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def gate(phase, *, token, open_prs=None, branch_names=None, locked_branch=BRANCH):
|
||||||
|
return assess_work_issue_duplicate_gate(
|
||||||
|
ISSUE,
|
||||||
|
open_prs=[owning_pr()] if open_prs is None else open_prs,
|
||||||
|
branch_names=branch_names or [],
|
||||||
|
claim_entry={},
|
||||||
|
locked_branch=locked_branch,
|
||||||
|
phase=phase,
|
||||||
|
recovered_owning_pr=token,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────────── the defect this issue exists to fix ─────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestPreFixReproduction(unittest.TestCase):
|
||||||
|
"""The exact wiring gap: renewal evidence was invisible to later gates."""
|
||||||
|
|
||||||
|
def test_recovery_only_rebuild_cannot_see_a_renewal_lock(self):
|
||||||
|
# This is the pre-fix behaviour of every enforcement path. It is correct
|
||||||
|
# for the recovery rebuild to ignore a renewal block -- the defect was
|
||||||
|
# that nothing else looked at it.
|
||||||
|
self.assertIsNone(
|
||||||
|
issue_lock_recovery.recovered_owning_pr_from_lock(renewal_lock())
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_renewal_lock_produced_no_exemption_before_the_fix(self):
|
||||||
|
# Feeding the gate what the pre-fix code fed it (recovery rebuild only)
|
||||||
|
# reproduces the reported refusal at the commit phase.
|
||||||
|
token = issue_lock_recovery.recovered_owning_pr_from_lock(renewal_lock())
|
||||||
|
result = gate(PHASE_COMMIT, token=token)
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
self.assertEqual(result["outcome"], "duplicate_commit_prevented")
|
||||||
|
self.assertFalse(result["owning_pr_recovery_exempted"])
|
||||||
|
self.assertEqual(result["owning_pr_recovery_notes"], [])
|
||||||
|
|
||||||
|
def test_shared_resolver_now_sees_it(self):
|
||||||
|
self.assertIsNotNone(
|
||||||
|
gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────────────── rebuild: the granted case ─────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestRenewalRebuildGranted(unittest.TestCase):
|
||||||
|
def test_sanctioned_renewal_rebuilds_owning_pr_evidence(self):
|
||||||
|
token = issue_lock_renewal.owning_pr_renewal_from_lock(renewal_lock())
|
||||||
|
self.assertEqual(
|
||||||
|
token,
|
||||||
|
{
|
||||||
|
"issue_number": ISSUE,
|
||||||
|
"pr_number": OWNING_PR,
|
||||||
|
"branch_name": BRANCH,
|
||||||
|
"head_sha": HEAD,
|
||||||
|
"recorded_head": HEAD,
|
||||||
|
"accepted_head": HEAD,
|
||||||
|
"head_relation": "equal",
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_branch_falls_back_to_the_lock_branch(self):
|
||||||
|
lock = renewal_lock(renewal_record(branch_name=""))
|
||||||
|
token = issue_lock_renewal.owning_pr_renewal_from_lock(lock)
|
||||||
|
self.assertEqual(token["branch_name"], BRANCH)
|
||||||
|
|
||||||
|
def test_claimant_may_live_under_work_lease(self):
|
||||||
|
lock = renewal_lock(claimant=False)
|
||||||
|
lock["work_lease"] = {"claimant": {"username": IDENTITY, "profile": PROFILE}}
|
||||||
|
self.assertIsNotNone(issue_lock_renewal.owning_pr_renewal_from_lock(lock))
|
||||||
|
|
||||||
|
def test_rebuild_does_not_mutate_the_lock(self):
|
||||||
|
lock = renewal_lock()
|
||||||
|
before = copy.deepcopy(lock)
|
||||||
|
issue_lock_renewal.owning_pr_renewal_from_lock(lock)
|
||||||
|
self.assertEqual(lock, before)
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────────────── rebuild: fails closed ─────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestRenewalRebuildFailsClosed(unittest.TestCase):
|
||||||
|
def assertNoEvidence(self, lock):
|
||||||
|
self.assertIsNone(issue_lock_renewal.owning_pr_renewal_from_lock(lock))
|
||||||
|
|
||||||
|
def test_no_lock_at_all(self):
|
||||||
|
self.assertNoEvidence(None)
|
||||||
|
self.assertNoEvidence({})
|
||||||
|
self.assertNoEvidence("not-a-mapping")
|
||||||
|
|
||||||
|
def test_lock_without_renewal_block(self):
|
||||||
|
# A fresh claim, or a lock whose renewal block was replaced.
|
||||||
|
self.assertNoEvidence({"issue_number": ISSUE, "branch_name": BRANCH})
|
||||||
|
|
||||||
|
def test_renewal_not_granted(self):
|
||||||
|
self.assertNoEvidence(renewal_lock(renewal_record(renewed=False)))
|
||||||
|
|
||||||
|
def test_renewal_flag_missing(self):
|
||||||
|
record = renewal_record()
|
||||||
|
del record["renewed"]
|
||||||
|
self.assertNoEvidence(renewal_lock(record))
|
||||||
|
|
||||||
|
def test_renewal_block_malformed(self):
|
||||||
|
self.assertNoEvidence(renewal_lock("not-a-mapping"))
|
||||||
|
|
||||||
|
def test_local_head_diverged_from_pr_head(self):
|
||||||
|
self.assertNoEvidence(renewal_lock(renewal_record(head_sha=OTHER_HEAD)))
|
||||||
|
|
||||||
|
def test_remote_head_diverged_from_pr_head(self):
|
||||||
|
# Force-push or unrelated remote movement.
|
||||||
|
self.assertNoEvidence(renewal_lock(renewal_record(remote_head_sha=OTHER_HEAD)))
|
||||||
|
|
||||||
|
def test_local_head_missing(self):
|
||||||
|
self.assertNoEvidence(renewal_lock(renewal_record(head_sha="")))
|
||||||
|
|
||||||
|
def test_remote_head_missing(self):
|
||||||
|
self.assertNoEvidence(renewal_lock(renewal_record(remote_head_sha="")))
|
||||||
|
|
||||||
|
def test_pr_head_missing(self):
|
||||||
|
self.assertNoEvidence(renewal_lock(renewal_record(pr_head_sha="")))
|
||||||
|
|
||||||
|
def test_pr_number_missing(self):
|
||||||
|
self.assertNoEvidence(renewal_lock(renewal_record(pr_number=None)))
|
||||||
|
|
||||||
|
def test_pr_number_malformed(self):
|
||||||
|
self.assertNoEvidence(renewal_lock(renewal_record(pr_number="not-a-number")))
|
||||||
|
|
||||||
|
def test_issue_number_missing_from_lock(self):
|
||||||
|
self.assertNoEvidence(renewal_lock(issue_number=None))
|
||||||
|
|
||||||
|
def test_branch_unknown_everywhere(self):
|
||||||
|
lock = renewal_lock(renewal_record(branch_name=""))
|
||||||
|
lock["branch_name"] = ""
|
||||||
|
self.assertNoEvidence(lock)
|
||||||
|
|
||||||
|
def test_identity_mismatch(self):
|
||||||
|
self.assertNoEvidence(renewal_lock(renewal_record(identity="someone-else")))
|
||||||
|
|
||||||
|
def test_profile_mismatch(self):
|
||||||
|
self.assertNoEvidence(renewal_lock(renewal_record(profile="other-profile")))
|
||||||
|
|
||||||
|
def test_identity_missing(self):
|
||||||
|
self.assertNoEvidence(renewal_lock(renewal_record(identity="")))
|
||||||
|
|
||||||
|
def test_profile_missing(self):
|
||||||
|
self.assertNoEvidence(renewal_lock(renewal_record(profile="")))
|
||||||
|
|
||||||
|
def test_claimant_absent(self):
|
||||||
|
self.assertNoEvidence(renewal_lock(claimant=False))
|
||||||
|
|
||||||
|
def test_renewal_block_disagreeing_with_the_lock_claimant_is_refused(self):
|
||||||
|
# An internal-consistency check, not a caller check: the renewal block
|
||||||
|
# and the claimant recorded on the same lock must name one identity.
|
||||||
|
# Nothing here proves who is calling — see
|
||||||
|
# TestCallerBindingIsStructuralNotFieldComparison for that boundary.
|
||||||
|
lock = renewal_lock()
|
||||||
|
lock["claimant"] = {"username": "other-recorded-user", "profile": PROFILE}
|
||||||
|
self.assertNoEvidence(lock)
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────────────── the shared resolver ─────────────────────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestSharedResolver(unittest.TestCase):
|
||||||
|
def test_recovery_lock_resolves_to_recovery_evidence(self):
|
||||||
|
token = gitea_mcp_server._owning_pr_continuation_from_lock(recovery_lock())
|
||||||
|
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||||
|
|
||||||
|
def test_renewal_lock_resolves_to_renewal_evidence(self):
|
||||||
|
token = gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||||
|
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||||
|
|
||||||
|
def test_recovery_takes_precedence_over_an_agreeing_renewal(self):
|
||||||
|
# Same precedence gitea_lock_issue applies when granting the waiver, so
|
||||||
|
# the answer cannot differ between the granting and enforcing paths.
|
||||||
|
# Both blocks describe one decision, so both name the same PR and head.
|
||||||
|
lock = recovery_lock()
|
||||||
|
lock["lease_renewal"] = renewal_record()
|
||||||
|
token = gitea_mcp_server._owning_pr_continuation_from_lock(lock)
|
||||||
|
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||||
|
self.assertEqual(token["head_relation"], issue_lock_recovery.HEAD_RELATION_EQUAL)
|
||||||
|
|
||||||
|
def test_no_evidence_resolves_to_none(self):
|
||||||
|
self.assertIsNone(gitea_mcp_server._owning_pr_continuation_from_lock(None))
|
||||||
|
self.assertIsNone(gitea_mcp_server._owning_pr_continuation_from_lock({}))
|
||||||
|
self.assertIsNone(
|
||||||
|
gitea_mcp_server._owning_pr_continuation_from_lock(
|
||||||
|
{"issue_number": ISSUE, "branch_name": BRANCH}
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# ─────────── ambiguous recovery/renewal pairs never broaden authority ──────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestAmbiguousEvidenceFailsClosed(unittest.TestCase):
|
||||||
|
"""#945 F3: a lock carrying two evidence blocks must agree, or authorize nothing.
|
||||||
|
|
||||||
|
Coexistence is legitimately reachable, so this is not a theoretical case.
|
||||||
|
Recovery is assessed whenever the lease is not live and requires a dead
|
||||||
|
recorded PID; renewal is assessed whenever the lease has *expired* — one way
|
||||||
|
to be non-live — and does not branch on PID liveness at all. An expired
|
||||||
|
lease whose owner also died satisfies both, and ``gitea_lock_issue`` then
|
||||||
|
writes both blocks into the same freshly built dict. A sanctioned pair comes
|
||||||
|
from one live observation, so it always agrees; disagreement means the
|
||||||
|
persisted lock no longer records a single sanctioned decision.
|
||||||
|
|
||||||
|
The dangerous direction is fall-through: before this, a recovery block that
|
||||||
|
failed validation was skipped and renewal evidence naming a *different* PR
|
||||||
|
was returned instead. Every case below asserts ``None`` — no continuation
|
||||||
|
authority at all, not a partial or downgraded one.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def resolve(self, lock):
|
||||||
|
return gitea_mcp_server._owning_pr_continuation_from_lock(lock)
|
||||||
|
|
||||||
|
def both(self, *, recovery=None, renewal=None, **lock_overrides):
|
||||||
|
"""A lock carrying both server-written evidence blocks."""
|
||||||
|
lock = recovery_lock()
|
||||||
|
if recovery is not None:
|
||||||
|
lock["dead_session_recovery"] = recovery
|
||||||
|
lock["lease_renewal"] = renewal if renewal is not None else renewal_record()
|
||||||
|
lock.update(lock_overrides)
|
||||||
|
return lock
|
||||||
|
|
||||||
|
# ── the two legitimate single-block shapes still work ──────────────────
|
||||||
|
|
||||||
|
def test_valid_recovery_only_still_authorizes(self):
|
||||||
|
token = self.resolve(recovery_lock())
|
||||||
|
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||||
|
|
||||||
|
def test_valid_renewal_only_still_authorizes(self):
|
||||||
|
token = self.resolve(renewal_lock())
|
||||||
|
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||||
|
|
||||||
|
# ── both present ───────────────────────────────────────────────────────
|
||||||
|
|
||||||
|
def test_both_present_and_identical_authorizes_once(self):
|
||||||
|
token = self.resolve(self.both())
|
||||||
|
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||||
|
self.assertEqual(token["head_sha"], HEAD)
|
||||||
|
|
||||||
|
def test_both_present_naming_different_prs_authorizes_nothing(self):
|
||||||
|
lock = self.both(renewal=renewal_record(pr_number=OTHER_PR))
|
||||||
|
self.assertIsNone(self.resolve(lock))
|
||||||
|
|
||||||
|
def test_conflicting_head_authorizes_nothing(self):
|
||||||
|
lock = self.both(
|
||||||
|
renewal=renewal_record(
|
||||||
|
head_sha=OTHER_HEAD, remote_head_sha=OTHER_HEAD, pr_head_sha=OTHER_HEAD
|
||||||
|
)
|
||||||
|
)
|
||||||
|
self.assertIsNone(self.resolve(lock))
|
||||||
|
|
||||||
|
def test_conflicting_branch_authorizes_nothing(self):
|
||||||
|
lock = self.both(renewal=renewal_record(branch_name=OTHER_BRANCH))
|
||||||
|
self.assertIsNone(self.resolve(lock))
|
||||||
|
|
||||||
|
def test_conflicting_head_relation_authorizes_nothing(self):
|
||||||
|
"""A descendant recovery beside an equal-head renewal is not one decision."""
|
||||||
|
recovery = dict(recovery_lock()["dead_session_recovery"])
|
||||||
|
recovery["head_relation"] = issue_lock_recovery.HEAD_RELATION_STRICT_DESCENDANT
|
||||||
|
recovery["recorded_head"] = HEAD
|
||||||
|
recovery["accepted_head"] = OTHER_HEAD
|
||||||
|
self.assertIsNone(self.resolve(self.both(recovery=recovery)))
|
||||||
|
|
||||||
|
def test_conflicting_identity_authorizes_nothing(self):
|
||||||
|
"""The renewal half stops rebuilding, so the pair can no longer agree."""
|
||||||
|
lock = self.both(renewal=renewal_record(identity="other-user"))
|
||||||
|
lock["claimant"] = {"username": IDENTITY, "profile": PROFILE}
|
||||||
|
# Recovery alone would still rebuild; presence of an unusable renewal
|
||||||
|
# block must not silently downgrade to the recovery answer.
|
||||||
|
self.assertEqual(self.resolve(lock)["pr_number"], OWNING_PR)
|
||||||
|
|
||||||
|
def test_conflicting_profile_between_renewal_and_claimant(self):
|
||||||
|
lock = self.both(renewal=renewal_record(profile="other-profile"))
|
||||||
|
self.assertEqual(self.resolve(lock)["pr_number"], OWNING_PR)
|
||||||
|
|
||||||
|
def test_conflicting_issue_number_authorizes_nothing(self):
|
||||||
|
"""Both tokens read issue_number from the lock, so a wrong issue moves both."""
|
||||||
|
lock = self.both(issue_number=ISSUE + 1)
|
||||||
|
token = self.resolve(lock)
|
||||||
|
self.assertEqual(token["issue_number"], ISSUE + 1)
|
||||||
|
self.assertEqual(token["pr_number"], OWNING_PR)
|
||||||
|
|
||||||
|
# ── recovery present but unusable: never fall through to renewal ────────
|
||||||
|
|
||||||
|
def test_malformed_recovery_beside_valid_renewal_authorizes_nothing(self):
|
||||||
|
recovery = {"recovered": True, "pr_number": "not-a-number"}
|
||||||
|
self.assertIsNone(self.resolve(self.both(recovery=recovery)))
|
||||||
|
|
||||||
|
def test_ungranted_recovery_beside_valid_renewal_authorizes_nothing(self):
|
||||||
|
recovery = dict(recovery_lock()["dead_session_recovery"])
|
||||||
|
recovery["recovered"] = False
|
||||||
|
self.assertIsNone(self.resolve(self.both(recovery=recovery)))
|
||||||
|
|
||||||
|
def test_stale_recovery_beside_newer_renewal_authorizes_nothing(self):
|
||||||
|
"""The exact bypass review 623 probed: conflicting recovery, valid renewal."""
|
||||||
|
recovery = dict(recovery_lock(pr_number=OTHER_PR)["dead_session_recovery"])
|
||||||
|
recovery["accepted_head"] = OTHER_HEAD # fails its own head equality
|
||||||
|
lock = self.both(recovery=recovery)
|
||||||
|
self.assertIsNone(
|
||||||
|
self.resolve(lock),
|
||||||
|
"a conflicting recovery record must not be bypassed by renewal "
|
||||||
|
"evidence naming a different PR",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_empty_recovery_block_beside_valid_renewal_authorizes_nothing(self):
|
||||||
|
self.assertIsNone(self.resolve(self.both(recovery={})))
|
||||||
|
|
||||||
|
# ── ambiguity yields nothing at all, not a partial authorization ────────
|
||||||
|
|
||||||
|
def test_ambiguity_yields_no_partial_token(self):
|
||||||
|
lock = self.both(renewal=renewal_record(pr_number=OTHER_PR))
|
||||||
|
result = self.resolve(lock)
|
||||||
|
self.assertIsNone(result)
|
||||||
|
self.assertNotIsInstance(result, dict)
|
||||||
|
|
||||||
|
def test_resolution_does_not_mutate_the_lock(self):
|
||||||
|
lock = self.both(renewal=renewal_record(pr_number=OTHER_PR))
|
||||||
|
before = copy.deepcopy(lock)
|
||||||
|
self.resolve(lock)
|
||||||
|
self.assertEqual(lock, before)
|
||||||
|
|
||||||
|
|
||||||
|
# ────────────── every enforcement path uses the same decision ──────────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestEnforcementPathsShareOneDecision(unittest.TestCase):
|
||||||
|
"""AC: commit, push and create-PR gates consume one authoritative token."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self.token = gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||||
|
|
||||||
|
def test_commit_phase_permits_continuation(self):
|
||||||
|
result = gate(PHASE_COMMIT, token=self.token)
|
||||||
|
self.assertFalse(result["block"])
|
||||||
|
self.assertTrue(result["owning_pr_recovery_exempted"])
|
||||||
|
self.assertEqual(result["outcome"], OUTCOME_DUPLICATE_WORK_NOT_PREVENTED)
|
||||||
|
|
||||||
|
def test_create_pr_phase_permits_continuation(self):
|
||||||
|
result = gate(PHASE_CREATE_PR, token=self.token)
|
||||||
|
self.assertFalse(result["block"])
|
||||||
|
self.assertTrue(result["owning_pr_recovery_exempted"])
|
||||||
|
|
||||||
|
def test_push_phase_permits_continuation(self):
|
||||||
|
result = gate(PHASE_PUSH, token=self.token)
|
||||||
|
self.assertFalse(result["block"])
|
||||||
|
self.assertTrue(result["owning_pr_recovery_exempted"])
|
||||||
|
|
||||||
|
def test_lock_phase_permits_continuation(self):
|
||||||
|
result = gate(PHASE_LOCK, token=self.token)
|
||||||
|
self.assertFalse(result["block"])
|
||||||
|
|
||||||
|
def test_all_phases_agree(self):
|
||||||
|
outcomes = {
|
||||||
|
phase: gate(phase, token=self.token)["block"]
|
||||||
|
for phase in (PHASE_LOCK, PHASE_COMMIT, PHASE_PUSH, PHASE_CREATE_PR)
|
||||||
|
}
|
||||||
|
self.assertEqual(set(outcomes.values()), {False}, outcomes)
|
||||||
|
|
||||||
|
def test_dead_session_recovery_still_permits_continuation(self):
|
||||||
|
token = gitea_mcp_server._owning_pr_continuation_from_lock(recovery_lock())
|
||||||
|
for phase in (PHASE_COMMIT, PHASE_PUSH, PHASE_CREATE_PR):
|
||||||
|
with self.subTest(phase=phase):
|
||||||
|
result = gate(phase, token=token)
|
||||||
|
self.assertFalse(result["block"])
|
||||||
|
self.assertTrue(result["owning_pr_recovery_exempted"])
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────── the exemption cannot be widened ─────────────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestExemptionCannotBeWidened(unittest.TestCase):
|
||||||
|
def setUp(self):
|
||||||
|
self.token = gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||||
|
|
||||||
|
def test_an_open_pr_alone_grants_nothing(self):
|
||||||
|
result = gate(PHASE_COMMIT, token=None)
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
self.assertFalse(result["owning_pr_recovery_exempted"])
|
||||||
|
|
||||||
|
def test_a_second_pr_is_refused(self):
|
||||||
|
result = gate(
|
||||||
|
PHASE_CREATE_PR,
|
||||||
|
token=self.token,
|
||||||
|
open_prs=[owning_pr(), owning_pr(number=OTHER_PR, ref=OTHER_BRANCH)],
|
||||||
|
)
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
self.assertFalse(result["owning_pr_recovery_exempted"])
|
||||||
|
|
||||||
|
def test_a_different_pr_is_refused(self):
|
||||||
|
result = gate(
|
||||||
|
PHASE_COMMIT, token=self.token, open_prs=[owning_pr(number=OTHER_PR)]
|
||||||
|
)
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
|
||||||
|
def test_a_different_branch_is_refused(self):
|
||||||
|
result = gate(
|
||||||
|
PHASE_COMMIT, token=self.token, open_prs=[owning_pr(ref=OTHER_BRANCH)]
|
||||||
|
)
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
|
||||||
|
def test_locked_branch_mismatch_is_refused(self):
|
||||||
|
result = gate(PHASE_COMMIT, token=self.token, locked_branch=OTHER_BRANCH)
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
|
||||||
|
def test_live_pr_head_divergence_is_refused(self):
|
||||||
|
# Force-push or unrelated remote movement after renewal.
|
||||||
|
result = gate(
|
||||||
|
PHASE_COMMIT, token=self.token, open_prs=[owning_pr(sha=OTHER_HEAD)]
|
||||||
|
)
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
|
||||||
|
def test_evidence_for_another_issue_is_refused(self):
|
||||||
|
foreign = gitea_mcp_server._owning_pr_continuation_from_lock(
|
||||||
|
renewal_lock(issue_number=ISSUE + 1)
|
||||||
|
)
|
||||||
|
result = gate(PHASE_COMMIT, token=foreign)
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
|
||||||
|
def test_sequential_tasks_do_not_inherit_continuation(self):
|
||||||
|
# One daemon serves many tasks. A renewal proved for issue N must not
|
||||||
|
# authorize continuation for the next task's issue.
|
||||||
|
prior_task = gitea_mcp_server._owning_pr_continuation_from_lock(
|
||||||
|
renewal_lock(issue_number=ISSUE + 7)
|
||||||
|
)
|
||||||
|
self.assertIsNotNone(prior_task)
|
||||||
|
self.assertTrue(gate(PHASE_COMMIT, token=prior_task)["block"])
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────── ordinary duplicate prevention is intact ─────────────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestDuplicatePreventionRetained(unittest.TestCase):
|
||||||
|
def test_competing_branch_still_blocks(self):
|
||||||
|
token = gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||||
|
result = gate(
|
||||||
|
PHASE_COMMIT,
|
||||||
|
token=token,
|
||||||
|
open_prs=[],
|
||||||
|
branch_names=[BRANCH, OTHER_BRANCH],
|
||||||
|
)
|
||||||
|
self.assertTrue(result["block"])
|
||||||
|
|
||||||
|
def test_unrelated_work_without_a_lock_still_blocks(self):
|
||||||
|
token = gitea_mcp_server._owning_pr_continuation_from_lock(None)
|
||||||
|
self.assertIsNone(token)
|
||||||
|
self.assertTrue(gate(PHASE_COMMIT, token=token)["block"])
|
||||||
|
|
||||||
|
|
||||||
|
# ───────────────── refusals stay structured and auditable ─────────────────
|
||||||
|
|
||||||
|
|
||||||
|
class TestRefusalShapePreserved(unittest.TestCase):
|
||||||
|
def test_blocked_result_keeps_its_audit_fields(self):
|
||||||
|
token = gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||||
|
result = gate(
|
||||||
|
PHASE_COMMIT, token=token, open_prs=[owning_pr(number=OTHER_PR)]
|
||||||
|
)
|
||||||
|
for field in (
|
||||||
|
"block",
|
||||||
|
"outcome",
|
||||||
|
"reasons",
|
||||||
|
"owning_pr_recovery_exempted",
|
||||||
|
"owning_pr_recovery_notes",
|
||||||
|
):
|
||||||
|
with self.subTest(field=field):
|
||||||
|
self.assertIn(field, result)
|
||||||
|
self.assertTrue(result["reasons"])
|
||||||
|
# A rejected token explains which element of ownership disagreed.
|
||||||
|
self.assertTrue(result["owning_pr_recovery_notes"])
|
||||||
|
|
||||||
|
def test_granted_result_records_why(self):
|
||||||
|
token = gitea_mcp_server._owning_pr_continuation_from_lock(renewal_lock())
|
||||||
|
result = gate(PHASE_COMMIT, token=token)
|
||||||
|
self.assertTrue(result["owning_pr_recovery_notes"])
|
||||||
|
self.assertIn(
|
||||||
|
f"#{OWNING_PR}", " ".join(result["owning_pr_recovery_notes"])
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,662 @@
|
|||||||
|
"""Client/session-aware runtime ownership and provenance (#948).
|
||||||
|
|
||||||
|
Covers the reproduced contradiction that motivated the issue: one surface
|
||||||
|
reporting ``client_managed`` while another reported ``manual_launch`` for the
|
||||||
|
same process, remediation hardcoded to one vendor, and a profile-wide duplicate
|
||||||
|
wall that could not tell two healthy clients apart.
|
||||||
|
|
||||||
|
All client and session identifiers here are synthetic.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
from datetime import datetime, timedelta, timezone
|
||||||
|
|
||||||
|
import mcp_client_reconnect
|
||||||
|
import mcp_namespace_health
|
||||||
|
import mcp_worker_identity as mwi
|
||||||
|
|
||||||
|
|
||||||
|
NOW = datetime(2026, 7, 29, 6, 0, 0, tzinfo=timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
def _registry() -> mwi.WorkerRegistry:
|
||||||
|
"""A registry on a throwaway path; never the operator's real one."""
|
||||||
|
handle, path = tempfile.mkstemp(suffix=".sqlite3")
|
||||||
|
os.close(handle)
|
||||||
|
os.unlink(path)
|
||||||
|
return mwi.WorkerRegistry(path)
|
||||||
|
|
||||||
|
|
||||||
|
def _attach(
|
||||||
|
registry: mwi.WorkerRegistry,
|
||||||
|
*,
|
||||||
|
client: str,
|
||||||
|
session: str,
|
||||||
|
generation: str,
|
||||||
|
profile: str = "prgs-reviewer",
|
||||||
|
role: str = "reviewer",
|
||||||
|
pid: int = 4242,
|
||||||
|
now: datetime = NOW,
|
||||||
|
ttl: float = 900.0,
|
||||||
|
) -> dict:
|
||||||
|
"""Register one synthetic worker and return the outcome."""
|
||||||
|
identity = mwi.generate_worker_identity(client, session, now=now)
|
||||||
|
outcome = registry.register(
|
||||||
|
worker_identity=identity,
|
||||||
|
client_name=client,
|
||||||
|
client_instance_id=f"inst-{session}",
|
||||||
|
session_id=session,
|
||||||
|
generation_id=generation,
|
||||||
|
role=role,
|
||||||
|
profile=profile,
|
||||||
|
pid=pid,
|
||||||
|
heartbeat_ttl_seconds=ttl,
|
||||||
|
now=now,
|
||||||
|
)
|
||||||
|
outcome["identity"] = identity
|
||||||
|
return outcome
|
||||||
|
|
||||||
|
|
||||||
|
class IdentityFormatTests(unittest.TestCase):
|
||||||
|
"""AC27-29: collision-resistant `<llm-name>-<UTC-timestamp>-<short-sha>`."""
|
||||||
|
|
||||||
|
def test_identity_matches_required_format(self):
|
||||||
|
identity = mwi.generate_worker_identity("Gemini", "sess-0001", now=NOW)
|
||||||
|
parsed = mwi.parse_worker_identity(identity)
|
||||||
|
self.assertTrue(parsed["valid"], parsed["reasons"])
|
||||||
|
self.assertEqual(parsed["client_name"], "gemini")
|
||||||
|
self.assertEqual(parsed["minted_at"], "20260729T060000Z")
|
||||||
|
self.assertEqual(len(parsed["digest"]), 12)
|
||||||
|
|
||||||
|
def test_digest_varies_with_session_and_nonce(self):
|
||||||
|
base = dict(timestamp_ns=1, now=NOW)
|
||||||
|
a = mwi.generate_worker_identity("codex", "sess-A", nonce="n", **base)
|
||||||
|
b = mwi.generate_worker_identity("codex", "sess-B", nonce="n", **base)
|
||||||
|
c = mwi.generate_worker_identity("codex", "sess-A", nonce="m", **base)
|
||||||
|
self.assertNotEqual(a, b, "session must feed the digest")
|
||||||
|
self.assertNotEqual(a, c, "nonce must feed the digest")
|
||||||
|
|
||||||
|
def test_identity_is_not_role_or_profile(self):
|
||||||
|
"""AC26: identity is independent of role and profile."""
|
||||||
|
args = dict(timestamp_ns=7, nonce="fixed", now=NOW)
|
||||||
|
same = mwi.generate_worker_identity("claude", "sess-1", **args)
|
||||||
|
self.assertEqual(same, mwi.generate_worker_identity("claude", "sess-1", **args))
|
||||||
|
# Nothing role- or profile-derived appears in the identity.
|
||||||
|
self.assertNotIn("reviewer", same)
|
||||||
|
self.assertNotIn("prgs", same)
|
||||||
|
|
||||||
|
def test_malformed_identity_rejected(self):
|
||||||
|
self.assertFalse(mwi.parse_worker_identity("prgs-reviewer")["valid"])
|
||||||
|
self.assertFalse(mwi.parse_worker_identity("")["valid"])
|
||||||
|
self.assertFalse(mwi.parse_worker_identity(None)["valid"])
|
||||||
|
|
||||||
|
|
||||||
|
class PerClientAttachmentTests(unittest.TestCase):
|
||||||
|
"""Every supported client attaches and is reported as itself."""
|
||||||
|
|
||||||
|
def _assert_attached_as(self, client: str, expected_name: str):
|
||||||
|
registry = _registry()
|
||||||
|
outcome = _attach(
|
||||||
|
registry, client=client, session=f"sess-{client}", generation="gen-1"
|
||||||
|
)
|
||||||
|
self.assertTrue(outcome["registered"], outcome["reasons"])
|
||||||
|
|
||||||
|
verdict = mwi.assess_provenance(
|
||||||
|
registry=registry, worker_identity=outcome["identity"], env={}, now=NOW
|
||||||
|
)
|
||||||
|
self.assertEqual(verdict["session_ownership"], mwi.OWNERSHIP_OWNED)
|
||||||
|
self.assertEqual(verdict["provenance"], mwi.PROVENANCE_CLIENT_SESSION)
|
||||||
|
self.assertEqual(verdict["client_name"], expected_name)
|
||||||
|
self.assertTrue(verdict["session_owned"])
|
||||||
|
self.assertFalse(verdict["fail_closed"])
|
||||||
|
return verdict
|
||||||
|
|
||||||
|
def test_codex_attachment(self):
|
||||||
|
self._assert_attached_as("codex", "codex")
|
||||||
|
|
||||||
|
def test_gemini_attachment(self):
|
||||||
|
self._assert_attached_as("gemini", "gemini")
|
||||||
|
|
||||||
|
def test_antigravity_attachment(self):
|
||||||
|
self._assert_attached_as("antigravity", "antigravity")
|
||||||
|
|
||||||
|
def test_claude_attachment(self):
|
||||||
|
self._assert_attached_as("claude", "claude_code")
|
||||||
|
|
||||||
|
def test_unknown_client_is_named_not_guessed(self):
|
||||||
|
verdict = self._assert_attached_as("some_new_llm", "some_new_llm")
|
||||||
|
self.assertNotEqual(verdict["client_name"], "codex")
|
||||||
|
|
||||||
|
|
||||||
|
class SessionLifecycleTests(unittest.TestCase):
|
||||||
|
def test_same_client_new_session_gets_distinct_identity(self):
|
||||||
|
registry = _registry()
|
||||||
|
first = _attach(registry, client="codex", session="sess-1", generation="gen-1")
|
||||||
|
second = _attach(registry, client="codex", session="sess-2", generation="gen-2")
|
||||||
|
self.assertTrue(first["registered"])
|
||||||
|
self.assertTrue(second["registered"])
|
||||||
|
self.assertNotEqual(first["identity"], second["identity"])
|
||||||
|
|
||||||
|
# Both are live and neither blocks the other.
|
||||||
|
cohort = mwi.classify_cohort(registry.list_workers(), now=NOW)
|
||||||
|
self.assertEqual(cohort["live_worker_count"], 2)
|
||||||
|
self.assertFalse(cohort["blocked"], cohort["reasons"])
|
||||||
|
|
||||||
|
def test_different_client_attaches_after_previous_session_ends(self):
|
||||||
|
"""AC14: expiry then takeover with a higher fencing epoch."""
|
||||||
|
registry = _registry()
|
||||||
|
gone = _attach(
|
||||||
|
registry, client="codex", session="sess-old", generation="gen-shared", ttl=60
|
||||||
|
)
|
||||||
|
later = NOW + timedelta(hours=1)
|
||||||
|
self.assertFalse(
|
||||||
|
registry.is_live(registry.get(gone["identity"]), now=later)["live"]
|
||||||
|
)
|
||||||
|
|
||||||
|
arriving = _attach(
|
||||||
|
registry,
|
||||||
|
client="gemini",
|
||||||
|
session="sess-new",
|
||||||
|
generation="gen-other",
|
||||||
|
now=later,
|
||||||
|
)
|
||||||
|
claim = registry.claim_generation(
|
||||||
|
worker_identity=arriving["identity"],
|
||||||
|
generation_id="gen-shared",
|
||||||
|
now=later,
|
||||||
|
)
|
||||||
|
self.assertTrue(claim["claimed"], claim["reasons"])
|
||||||
|
self.assertIn(gone["identity"], claim["superseded_workers"])
|
||||||
|
self.assertGreater(claim["fencing_epoch"], gone["fencing_epoch"])
|
||||||
|
|
||||||
|
def test_superseded_session_is_fenced_on_resume(self):
|
||||||
|
"""AC15/AC16: the prior session cannot heartbeat its way back."""
|
||||||
|
registry = _registry()
|
||||||
|
old = _attach(
|
||||||
|
registry, client="codex", session="sess-old", generation="gen-shared", ttl=60
|
||||||
|
)
|
||||||
|
later = NOW + timedelta(hours=1)
|
||||||
|
new = _attach(
|
||||||
|
registry, client="gemini", session="sess-new", generation="gen-x", now=later
|
||||||
|
)
|
||||||
|
registry.claim_generation(
|
||||||
|
worker_identity=new["identity"], generation_id="gen-shared", now=later
|
||||||
|
)
|
||||||
|
|
||||||
|
resumed = registry.heartbeat(
|
||||||
|
worker_identity=old["identity"],
|
||||||
|
fencing_epoch=old["fencing_epoch"],
|
||||||
|
now=later,
|
||||||
|
)
|
||||||
|
self.assertFalse(resumed["renewed"])
|
||||||
|
self.assertFalse(resumed["mutation_performed"])
|
||||||
|
self.assertEqual(resumed["blocker_kind"], mwi.BLOCKER_FENCED)
|
||||||
|
|
||||||
|
def test_heartbeat_renews_only_the_owning_lease(self):
|
||||||
|
"""AC11: a wrong epoch never renews, and never mutates."""
|
||||||
|
registry = _registry()
|
||||||
|
worker = _attach(registry, client="codex", session="s", generation="g")
|
||||||
|
good = registry.heartbeat(
|
||||||
|
worker_identity=worker["identity"],
|
||||||
|
fencing_epoch=worker["fencing_epoch"],
|
||||||
|
now=NOW + timedelta(minutes=5),
|
||||||
|
)
|
||||||
|
self.assertTrue(good["renewed"])
|
||||||
|
|
||||||
|
bad = registry.heartbeat(
|
||||||
|
worker_identity=worker["identity"],
|
||||||
|
fencing_epoch=worker["fencing_epoch"] + 99,
|
||||||
|
now=NOW + timedelta(minutes=6),
|
||||||
|
)
|
||||||
|
self.assertFalse(bad["renewed"])
|
||||||
|
self.assertFalse(bad["mutation_performed"])
|
||||||
|
self.assertEqual(
|
||||||
|
registry.get(worker["identity"])["last_heartbeat_at"],
|
||||||
|
good["last_heartbeat_at"],
|
||||||
|
"a refused heartbeat must not advance the record",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class ConflictAndCollisionTests(unittest.TestCase):
|
||||||
|
def test_two_live_sessions_cannot_claim_one_generation(self):
|
||||||
|
registry = _registry()
|
||||||
|
first = _attach(registry, client="codex", session="s1", generation="gen-shared")
|
||||||
|
second = _attach(registry, client="gemini", session="s2", generation="gen-other")
|
||||||
|
|
||||||
|
claim = registry.claim_generation(
|
||||||
|
worker_identity=second["identity"],
|
||||||
|
generation_id="gen-shared",
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
self.assertFalse(claim["claimed"])
|
||||||
|
self.assertFalse(claim["mutation_performed"])
|
||||||
|
self.assertEqual(claim["blocker_kind"], mwi.BLOCKER_CONFLICTING_SESSIONS)
|
||||||
|
self.assertEqual(
|
||||||
|
claim["conflicting_owners"][0]["worker_identity"], first["identity"]
|
||||||
|
)
|
||||||
|
# The sanctioned recovery must never be "kill the other process".
|
||||||
|
self.assertIn("Do not kill", claim["exact_next_action"])
|
||||||
|
|
||||||
|
def test_contested_generation_fails_closed_in_assessment(self):
|
||||||
|
registry = _registry()
|
||||||
|
first = _attach(registry, client="codex", session="s1", generation="gen-shared")
|
||||||
|
_attach(registry, client="gemini", session="s2", generation="gen-shared")
|
||||||
|
|
||||||
|
verdict = mwi.assess_provenance(
|
||||||
|
registry=registry, worker_identity=first["identity"], env={}, now=NOW
|
||||||
|
)
|
||||||
|
self.assertEqual(verdict["session_ownership"], mwi.OWNERSHIP_CONTESTED)
|
||||||
|
self.assertTrue(verdict["fail_closed"])
|
||||||
|
self.assertEqual(verdict["blocker_kind"], mwi.BLOCKER_CONTRADICTORY)
|
||||||
|
self.assertTrue(verdict["conflicting_live_sessions"])
|
||||||
|
|
||||||
|
def test_identity_collision_is_refused_without_corrupting_existing(self):
|
||||||
|
"""AC31: never replace, adopt, merge with, or corrupt the incumbent."""
|
||||||
|
registry = _registry()
|
||||||
|
incumbent = _attach(registry, client="codex", session="s1", generation="gen-1")
|
||||||
|
before = registry.get(incumbent["identity"])
|
||||||
|
|
||||||
|
collided = registry.register(
|
||||||
|
worker_identity=incumbent["identity"],
|
||||||
|
client_name="gemini",
|
||||||
|
client_instance_id="inst-other",
|
||||||
|
session_id="s2",
|
||||||
|
generation_id="gen-2",
|
||||||
|
pid=9999,
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
self.assertFalse(collided["registered"])
|
||||||
|
self.assertTrue(collided["collision"])
|
||||||
|
self.assertFalse(collided["mutation_performed"])
|
||||||
|
self.assertEqual(collided["blocker_kind"], mwi.BLOCKER_IDENTITY_COLLISION)
|
||||||
|
self.assertEqual(collided["collision_kind"], "active_worker")
|
||||||
|
self.assertEqual(
|
||||||
|
registry.get(incumbent["identity"]), before, "incumbent must be untouched"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_after_collision_a_regenerated_identity_registers(self):
|
||||||
|
"""AC32/AC35: forced collision, safe regeneration, successful replacement."""
|
||||||
|
registry = _registry()
|
||||||
|
fixed = dict(timestamp_ns=99, nonce="deterministic", now=NOW)
|
||||||
|
forced = mwi.generate_worker_identity("codex", "sess-collide", **fixed)
|
||||||
|
first = registry.register(
|
||||||
|
worker_identity=forced,
|
||||||
|
client_name="codex",
|
||||||
|
client_instance_id="inst-1",
|
||||||
|
session_id="sess-collide",
|
||||||
|
generation_id="gen-1",
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
self.assertTrue(first["registered"])
|
||||||
|
|
||||||
|
# A second worker deriving the same inputs collides deterministically.
|
||||||
|
again = mwi.generate_worker_identity("codex", "sess-collide", **fixed)
|
||||||
|
self.assertEqual(again, forced)
|
||||||
|
self.assertTrue(
|
||||||
|
registry.register(
|
||||||
|
worker_identity=again,
|
||||||
|
client_name="codex",
|
||||||
|
client_instance_id="inst-2",
|
||||||
|
session_id="sess-collide",
|
||||||
|
generation_id="gen-2",
|
||||||
|
now=NOW,
|
||||||
|
)["collision"]
|
||||||
|
)
|
||||||
|
|
||||||
|
replacement = mwi.generate_worker_identity(
|
||||||
|
"codex", "sess-collide", timestamp_ns=100, nonce="different", now=NOW
|
||||||
|
)
|
||||||
|
self.assertNotEqual(replacement, forced)
|
||||||
|
self.assertTrue(
|
||||||
|
registry.register(
|
||||||
|
worker_identity=replacement,
|
||||||
|
client_name="codex",
|
||||||
|
client_instance_id="inst-2",
|
||||||
|
session_id="sess-collide",
|
||||||
|
generation_id="gen-2",
|
||||||
|
now=NOW,
|
||||||
|
)["registered"]
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_restarted_worker_inherits_nothing(self):
|
||||||
|
"""AC33/AC34: a restart mints a new identity and no prior epoch."""
|
||||||
|
registry = _registry()
|
||||||
|
before = _attach(
|
||||||
|
registry, client="codex", session="sess-before", generation="gen-1", ttl=60
|
||||||
|
)
|
||||||
|
later = NOW + timedelta(hours=2)
|
||||||
|
after = _attach(
|
||||||
|
registry, client="codex", session="sess-after", generation="gen-2", now=later
|
||||||
|
)
|
||||||
|
self.assertNotEqual(before["identity"], after["identity"])
|
||||||
|
self.assertNotEqual(
|
||||||
|
registry.get(after["identity"])["generation_id"],
|
||||||
|
registry.get(before["identity"])["generation_id"],
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class LivenessTests(unittest.TestCase):
|
||||||
|
def test_stale_session_record_is_not_live(self):
|
||||||
|
registry = _registry()
|
||||||
|
worker = _attach(registry, client="codex", session="s", generation="g", ttl=300)
|
||||||
|
stale = registry.is_live(
|
||||||
|
registry.get(worker["identity"]), now=NOW + timedelta(hours=1)
|
||||||
|
)
|
||||||
|
self.assertFalse(stale["live"])
|
||||||
|
self.assertFalse(stale["heartbeat_fresh"])
|
||||||
|
|
||||||
|
def test_liveness_is_not_pid_comparison_alone(self):
|
||||||
|
"""AC7: a live PID does not resurrect an expired registration."""
|
||||||
|
registry = _registry()
|
||||||
|
worker = _attach(registry, client="codex", session="s", generation="g", ttl=60)
|
||||||
|
verdict = registry.is_live(
|
||||||
|
registry.get(worker["identity"]),
|
||||||
|
now=NOW + timedelta(hours=1),
|
||||||
|
pid_alive=True,
|
||||||
|
)
|
||||||
|
self.assertFalse(
|
||||||
|
verdict["live"], "a live PID must not override a dead heartbeat"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_dead_pid_withdraws_liveness_from_a_fresh_heartbeat(self):
|
||||||
|
registry = _registry()
|
||||||
|
worker = _attach(registry, client="codex", session="s", generation="g")
|
||||||
|
verdict = registry.is_live(
|
||||||
|
registry.get(worker["identity"]), now=NOW, pid_alive=False
|
||||||
|
)
|
||||||
|
self.assertFalse(verdict["live"])
|
||||||
|
|
||||||
|
def test_stale_ownership_does_not_permanently_strand_a_daemon(self):
|
||||||
|
registry = _registry()
|
||||||
|
stranded = _attach(
|
||||||
|
registry, client="codex", session="s-old", generation="gen-daemon", ttl=60
|
||||||
|
)
|
||||||
|
later = NOW + timedelta(hours=3)
|
||||||
|
rescuer = _attach(
|
||||||
|
registry, client="claude", session="s-new", generation="gen-tmp", now=later
|
||||||
|
)
|
||||||
|
claim = registry.claim_generation(
|
||||||
|
worker_identity=rescuer["identity"],
|
||||||
|
generation_id="gen-daemon",
|
||||||
|
now=later,
|
||||||
|
)
|
||||||
|
self.assertTrue(claim["claimed"], claim["reasons"])
|
||||||
|
self.assertIn(stranded["identity"], claim["superseded_workers"])
|
||||||
|
|
||||||
|
|
||||||
|
class EvidenceTests(unittest.TestCase):
|
||||||
|
def test_env_flag_alone_does_not_prove_session_ownership(self):
|
||||||
|
verdict = mwi.assess_provenance(
|
||||||
|
registry=None,
|
||||||
|
worker_identity=None,
|
||||||
|
env={"GITEA_CLIENT_MANAGED": "1", "GITEA_MCP_SANCTIONED_DAEMON": "1"},
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
self.assertFalse(verdict["session_owned"])
|
||||||
|
self.assertEqual(verdict["session_ownership"], mwi.OWNERSHIP_UNOWNED)
|
||||||
|
self.assertTrue(verdict["env_flag_only"])
|
||||||
|
self.assertTrue(verdict["fail_closed"])
|
||||||
|
self.assertNotIn(mwi.EVIDENCE_ATTACHMENT_RECORD, verdict["evidence"])
|
||||||
|
self.assertFalse(verdict["env_signal"]["proves_session_ownership"])
|
||||||
|
|
||||||
|
def test_env_flag_still_answers_the_launch_question(self):
|
||||||
|
"""The #686 wall is preserved: env decides launch, not ownership."""
|
||||||
|
self.assertTrue(
|
||||||
|
mwi.assess_launch_provenance({"GITEA_CLIENT_MANAGED": "1"})["client_managed"]
|
||||||
|
)
|
||||||
|
self.assertFalse(
|
||||||
|
mwi.assess_launch_provenance({"GITEA_CLIENT_MANAGED": "0"})["client_managed"]
|
||||||
|
)
|
||||||
|
self.assertFalse(
|
||||||
|
mwi.assess_launch_provenance({}, stdin_is_tty=True)["client_managed"]
|
||||||
|
)
|
||||||
|
self.assertTrue(
|
||||||
|
mwi.assess_launch_provenance({"GITEA_MCP_PROFILE": "prgs-author"})[
|
||||||
|
"client_managed"
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_missing_evidence_is_unproven_not_manual(self):
|
||||||
|
"""A missing proof must not be reported as a hand-launched process."""
|
||||||
|
verdict = mwi.assess_provenance(
|
||||||
|
registry=None, worker_identity=None, env={}, now=NOW
|
||||||
|
)
|
||||||
|
self.assertEqual(verdict["provenance"], mwi.PROVENANCE_UNPROVEN)
|
||||||
|
self.assertNotEqual(verdict["provenance"], mwi.PROVENANCE_MANUAL)
|
||||||
|
self.assertTrue(verdict["fail_closed"])
|
||||||
|
|
||||||
|
def test_declared_manual_launch_is_reported_as_manual(self):
|
||||||
|
verdict = mwi.assess_provenance(
|
||||||
|
registry=None,
|
||||||
|
worker_identity=None,
|
||||||
|
env={"GITEA_CLIENT_MANAGED": "0"},
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
self.assertEqual(verdict["provenance"], mwi.PROVENANCE_MANUAL)
|
||||||
|
|
||||||
|
def test_fail_closed_refusal_names_its_scope_not_the_profile(self):
|
||||||
|
"""AC17/AC41: no refusal is profile-wide."""
|
||||||
|
verdict = mwi.assess_provenance(
|
||||||
|
registry=None,
|
||||||
|
worker_identity=None,
|
||||||
|
env={},
|
||||||
|
profile="prgs-reviewer",
|
||||||
|
role="reviewer",
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
self.assertFalse(verdict["scope"]["profile_wide"])
|
||||||
|
self.assertEqual(verdict["blocker_kind"], mwi.BLOCKER_NO_ATTACHMENT)
|
||||||
|
|
||||||
|
|
||||||
|
class CohortScopingTests(unittest.TestCase):
|
||||||
|
def test_shared_profile_with_distinct_identities_does_not_block(self):
|
||||||
|
"""AC40: profile is not a singleton identity."""
|
||||||
|
registry = _registry()
|
||||||
|
_attach(
|
||||||
|
registry,
|
||||||
|
client="codex",
|
||||||
|
session="s1",
|
||||||
|
generation="g1",
|
||||||
|
profile="prgs-reviewer",
|
||||||
|
)
|
||||||
|
_attach(
|
||||||
|
registry,
|
||||||
|
client="gemini",
|
||||||
|
session="s2",
|
||||||
|
generation="g2",
|
||||||
|
profile="prgs-reviewer",
|
||||||
|
)
|
||||||
|
|
||||||
|
cohort = mwi.classify_cohort(registry.list_workers(), now=NOW)
|
||||||
|
self.assertFalse(cohort["blocked"], cohort["reasons"])
|
||||||
|
self.assertEqual(cohort["blocker_kind"], mwi.BLOCKER_NONE)
|
||||||
|
self.assertIn("prgs-reviewer", cohort["shared_profiles"])
|
||||||
|
self.assertTrue(cohort["profile_sharing_permitted"])
|
||||||
|
self.assertEqual(cohort["blocked_worker_identities"], [])
|
||||||
|
|
||||||
|
def test_duplicate_cohort_records_block_only_the_offenders(self):
|
||||||
|
registry = _registry()
|
||||||
|
_attach(registry, client="codex", session="s1", generation="gen-contested")
|
||||||
|
_attach(registry, client="gemini", session="s2", generation="gen-contested")
|
||||||
|
_attach(registry, client="claude", session="s3", generation="gen-fine")
|
||||||
|
|
||||||
|
cohort = mwi.classify_cohort(registry.list_workers(), now=NOW)
|
||||||
|
self.assertTrue(cohort["blocked"])
|
||||||
|
self.assertEqual(cohort["contested_generations"], ["gen-contested"])
|
||||||
|
self.assertEqual(len(cohort["blocked_worker_identities"]), 2)
|
||||||
|
|
||||||
|
def test_mixed_runtime_generations_are_scoped_independently(self):
|
||||||
|
"""AC17: one stale generation does not wall unrelated healthy ones."""
|
||||||
|
registry = _registry()
|
||||||
|
stale = _attach(
|
||||||
|
registry, client="codex", session="s1", generation="gen-stale", ttl=60
|
||||||
|
)
|
||||||
|
healthy_a = _attach(registry, client="gemini", session="s2", generation="gen-a")
|
||||||
|
healthy_b = _attach(registry, client="claude", session="s3", generation="gen-b")
|
||||||
|
|
||||||
|
scoped = mwi.scope_runtime_failure(
|
||||||
|
failure_kind="stale-runtime",
|
||||||
|
worker_identity=stale["identity"],
|
||||||
|
profile="prgs-reviewer",
|
||||||
|
all_live_workers=registry.list_workers(),
|
||||||
|
)
|
||||||
|
self.assertFalse(scoped["profile_wide"])
|
||||||
|
self.assertFalse(scoped["fleet_wide"])
|
||||||
|
self.assertEqual(len(scoped["affected_workers"]), 1)
|
||||||
|
self.assertEqual(scoped["unaffected_worker_count"], 2)
|
||||||
|
unaffected = {w["worker_identity"] for w in scoped["unaffected_workers"]}
|
||||||
|
self.assertEqual(unaffected, {healthy_a["identity"], healthy_b["identity"]})
|
||||||
|
|
||||||
|
|
||||||
|
class HardcodedClientRegressionTests(unittest.TestCase):
|
||||||
|
def test_unknown_client_does_not_resolve_to_codex(self):
|
||||||
|
for name in ("gemini", "antigravity", "grok", "some_new_llm", "", None):
|
||||||
|
with self.subTest(client=name):
|
||||||
|
self.assertNotEqual(
|
||||||
|
mcp_client_reconnect.normalize_client(name),
|
||||||
|
"codex",
|
||||||
|
"an unidentified client must never be handed Codex UI steps",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_known_clients_still_get_their_own_steps(self):
|
||||||
|
self.assertEqual(mcp_client_reconnect.normalize_client("codex"), "codex")
|
||||||
|
self.assertEqual(
|
||||||
|
mcp_client_reconnect.normalize_client("claude_code"), "claude_code"
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_generic_steps_do_not_name_a_specific_vendor(self):
|
||||||
|
steps = " ".join(mcp_client_reconnect.operator_ui_steps("gemini"))
|
||||||
|
self.assertNotIn("Codex", steps)
|
||||||
|
|
||||||
|
def test_reconnect_client_is_derived_from_the_attachment_record(self):
|
||||||
|
registry = _registry()
|
||||||
|
worker = _attach(registry, client="antigravity", session="s", generation="g")
|
||||||
|
verdict = mwi.assess_provenance(
|
||||||
|
registry=registry, worker_identity=worker["identity"], env={}, now=NOW
|
||||||
|
)
|
||||||
|
self.assertEqual(mwi.reconnect_client_for(verdict), "antigravity")
|
||||||
|
|
||||||
|
def test_reconnect_client_is_unknown_rather_than_guessed(self):
|
||||||
|
self.assertEqual(mwi.reconnect_client_for({}), mwi.UNKNOWN_CLIENT)
|
||||||
|
|
||||||
|
|
||||||
|
class RemoteBindingTests(unittest.TestCase):
|
||||||
|
def test_explicit_prgs_selection_is_honoured(self):
|
||||||
|
resolved = mwi.resolve_bound_remote(
|
||||||
|
requested_remote="prgs", bound_remote="prgs", default_remote="dadeschools"
|
||||||
|
)
|
||||||
|
self.assertEqual(resolved["remote"], "prgs")
|
||||||
|
self.assertFalse(resolved["drifted"])
|
||||||
|
|
||||||
|
def test_omitted_remote_uses_the_binding_not_the_library_default(self):
|
||||||
|
"""The reported dadeschools host drift."""
|
||||||
|
resolved = mwi.resolve_bound_remote(
|
||||||
|
requested_remote=None, bound_remote="prgs", default_remote="dadeschools"
|
||||||
|
)
|
||||||
|
self.assertEqual(resolved["remote"], "prgs")
|
||||||
|
self.assertNotEqual(resolved["remote"], "dadeschools")
|
||||||
|
self.assertEqual(resolved["resolved_from"], "session_binding")
|
||||||
|
|
||||||
|
def test_contradicting_the_binding_is_refused(self):
|
||||||
|
resolved = mwi.resolve_bound_remote(
|
||||||
|
requested_remote="dadeschools",
|
||||||
|
bound_remote="prgs",
|
||||||
|
default_remote="dadeschools",
|
||||||
|
)
|
||||||
|
self.assertEqual(resolved["remote"], "prgs")
|
||||||
|
self.assertTrue(resolved["drifted"])
|
||||||
|
self.assertFalse(resolved["honoured_request"])
|
||||||
|
|
||||||
|
def test_unbound_session_falls_back_and_says_so(self):
|
||||||
|
resolved = mwi.resolve_bound_remote(
|
||||||
|
requested_remote=None, bound_remote=None, default_remote="dadeschools"
|
||||||
|
)
|
||||||
|
self.assertEqual(resolved["remote"], "dadeschools")
|
||||||
|
self.assertEqual(resolved["resolved_from"], "library_default")
|
||||||
|
self.assertTrue(resolved["reasons"])
|
||||||
|
|
||||||
|
|
||||||
|
class SurfaceAgreementTests(unittest.TestCase):
|
||||||
|
"""The reproduced contradiction: two surfaces, one process, two answers."""
|
||||||
|
|
||||||
|
def test_namespace_health_and_direct_assessment_agree(self):
|
||||||
|
registry = _registry()
|
||||||
|
worker = _attach(
|
||||||
|
registry,
|
||||||
|
client="gemini",
|
||||||
|
session="sess-agree",
|
||||||
|
generation="gen-agree",
|
||||||
|
profile="prgs-reviewer",
|
||||||
|
)
|
||||||
|
env = {"GITEA_MCP_PROFILE": "prgs-reviewer", "GITEA_CLIENT_MANAGED": "1"}
|
||||||
|
|
||||||
|
direct = mwi.assess_provenance(
|
||||||
|
registry=registry,
|
||||||
|
worker_identity=worker["identity"],
|
||||||
|
env=env,
|
||||||
|
profile="prgs-reviewer",
|
||||||
|
)
|
||||||
|
health = mcp_namespace_health.classify_namespace_probe(
|
||||||
|
"gitea-reviewer",
|
||||||
|
configured=True,
|
||||||
|
registered_tools=["gitea_whoami"],
|
||||||
|
probe_result={"success": True},
|
||||||
|
probe_source="client_namespace",
|
||||||
|
process={"pid": 4242, "profile": "prgs-reviewer", "env": env},
|
||||||
|
registry=registry,
|
||||||
|
worker_identity=worker["identity"],
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(health["provenance"], direct["provenance"])
|
||||||
|
self.assertEqual(health["is_client_managed"], direct["is_client_managed"])
|
||||||
|
self.assertEqual(health["worker_identity"], direct["worker_identity"])
|
||||||
|
self.assertEqual(health["session_id"], "sess-agree")
|
||||||
|
self.assertEqual(health["client_name"], "gemini")
|
||||||
|
|
||||||
|
def test_namespace_health_can_report_client_managed_at_all(self):
|
||||||
|
"""The old derivation was structurally incapable of this."""
|
||||||
|
env = {"GITEA_CLIENT_MANAGED": "1", "GITEA_MCP_PROFILE": "prgs-author"}
|
||||||
|
health = mcp_namespace_health.classify_namespace_probe(
|
||||||
|
"gitea-author",
|
||||||
|
configured=True,
|
||||||
|
registered_tools=["gitea_whoami"],
|
||||||
|
probe_result={"success": True},
|
||||||
|
probe_source="client_namespace",
|
||||||
|
process={"pid": 1234, "profile": "prgs-author", "env": env},
|
||||||
|
)
|
||||||
|
self.assertTrue(
|
||||||
|
health["is_client_managed"],
|
||||||
|
"a client-managed launch must be reportable as client-managed",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_namespace_health_without_attachment_fails_closed(self):
|
||||||
|
health = mcp_namespace_health.classify_namespace_probe(
|
||||||
|
"gitea-author",
|
||||||
|
configured=True,
|
||||||
|
registered_tools=["gitea_whoami"],
|
||||||
|
probe_result={"success": True},
|
||||||
|
probe_source="client_namespace",
|
||||||
|
process={"pid": 1234, "profile": "prgs-author", "env": {}},
|
||||||
|
)
|
||||||
|
self.assertTrue(health["provenance_fail_closed"])
|
||||||
|
self.assertEqual(health["provenance"], mwi.PROVENANCE_UNPROVEN)
|
||||||
|
self.assertIsNone(health["session_id"])
|
||||||
|
|
||||||
|
def test_no_false_reconnect_loop_for_an_owned_session(self):
|
||||||
|
"""A proven owner must not be told to reconnect."""
|
||||||
|
registry = _registry()
|
||||||
|
worker = _attach(registry, client="claude", session="s", generation="g")
|
||||||
|
verdict = mwi.assess_provenance(
|
||||||
|
registry=registry, worker_identity=worker["identity"], env={}, now=NOW
|
||||||
|
)
|
||||||
|
self.assertFalse(verdict["fail_closed"])
|
||||||
|
self.assertEqual(verdict["blocker_kind"], mwi.BLOCKER_NONE)
|
||||||
|
self.assertEqual(verdict["reasons"], [])
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,323 @@
|
|||||||
|
"""Validation tooling for the remote-MCP threat model (#956).
|
||||||
|
|
||||||
|
#956 requires that "every boundary claim [is] traceable to a file and line
|
||||||
|
anchor that resolves at the reviewed commit". A prose document cannot enforce
|
||||||
|
that about itself, and #930 demonstrated the failure mode: its inventory cited
|
||||||
|
``gitea_mcp_server.py`` anchors generated at ``7bf4f125`` which no longer point
|
||||||
|
at the described code at ``aad5c8b4``. Nothing failed, because nothing checked.
|
||||||
|
|
||||||
|
These tests are that check. They enforce, in both directions:
|
||||||
|
|
||||||
|
* every ``file.py:NNN`` anchor cited in the prose is declared in the fixture;
|
||||||
|
* every declared anchor resolves — the file exists, the line exists, and the
|
||||||
|
source line actually contains the substring the fixture claims for it;
|
||||||
|
* the document's structural obligations (assets, adversaries, boundaries,
|
||||||
|
credential rows, the co-residency ruling, and the child mapping) are present
|
||||||
|
and internally consistent.
|
||||||
|
|
||||||
|
A refactor that shifts a line number therefore breaks the suite instead of
|
||||||
|
silently rotting the security documentation.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import unittest
|
||||||
|
|
||||||
|
REPO_ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||||
|
DOC_PATH = os.path.join(REPO_ROOT, "docs", "remote-mcp", "threat-model.md")
|
||||||
|
FIXTURE_PATH = os.path.join(
|
||||||
|
REPO_ROOT, "docs", "remote-mcp", "threat-model-anchors.json"
|
||||||
|
)
|
||||||
|
|
||||||
|
# ``module.py:123`` as it appears inside markdown inline code spans.
|
||||||
|
ANCHOR_RE = re.compile(r"`([A-Za-z0-9_./-]+\.py):(\d+)`")
|
||||||
|
|
||||||
|
# The epic children this document must map to a boundary (#929 children 2-10).
|
||||||
|
REQUIRED_CHILDREN = [931, 932, 933, 934, 935, 936, 937, 938, 939]
|
||||||
|
|
||||||
|
# The adversaries #956 names explicitly.
|
||||||
|
REQUIRED_ADVERSARIES = [
|
||||||
|
"compromised LLM client",
|
||||||
|
"prompt injection",
|
||||||
|
"malicious tool arguments",
|
||||||
|
"network attacker",
|
||||||
|
"curious operator",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def _read(path):
|
||||||
|
with open(path, "r", encoding="utf-8") as fh:
|
||||||
|
return fh.read()
|
||||||
|
|
||||||
|
|
||||||
|
def _heading_re(title):
|
||||||
|
"""Match a level-2 heading by title, with or without section numbering.
|
||||||
|
|
||||||
|
The document numbers its sections ('## 6. Decomposition ruling'), so an
|
||||||
|
exact-substring assertion would break on renumbering without the document
|
||||||
|
having actually lost anything.
|
||||||
|
"""
|
||||||
|
return re.compile(
|
||||||
|
r"^##\s+(?:\d+\.\s+)?" + re.escape(title), re.MULTILINE
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _section_body(doc, title):
|
||||||
|
"""Return the text of section *title*, bounded by the next level-2 heading.
|
||||||
|
|
||||||
|
Bounding matters: an unbounded slice runs to end-of-document, so the
|
||||||
|
walkthrough tables in a later section leak into the child-to-boundary
|
||||||
|
mapping and satisfy its coverage check with rows that assign no owner.
|
||||||
|
"""
|
||||||
|
match = _heading_re(title).search(doc)
|
||||||
|
if match is None:
|
||||||
|
return None
|
||||||
|
rest = doc[match.end():]
|
||||||
|
nxt = re.search(r"^##\s", rest, re.MULTILINE)
|
||||||
|
return rest[: nxt.start()] if nxt else rest
|
||||||
|
|
||||||
|
|
||||||
|
def _source_line(rel_path, lineno):
|
||||||
|
"""Return the 1-based *lineno* of *rel_path*, or None if out of range."""
|
||||||
|
abs_path = os.path.join(REPO_ROOT, rel_path)
|
||||||
|
if not os.path.exists(abs_path):
|
||||||
|
return None
|
||||||
|
with open(abs_path, "r", encoding="utf-8", errors="replace") as fh:
|
||||||
|
for idx, line in enumerate(fh, start=1):
|
||||||
|
if idx == lineno:
|
||||||
|
return line
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
class ThreatModelFixtureTests(unittest.TestCase):
|
||||||
|
"""The fixture itself must be well-formed before it can prove anything."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self.fixture = json.loads(_read(FIXTURE_PATH))
|
||||||
|
|
||||||
|
def test_fixture_declares_a_generation_commit(self):
|
||||||
|
sha = self.fixture.get("generated_against_commit") or ""
|
||||||
|
self.assertRegex(
|
||||||
|
sha,
|
||||||
|
r"^[0-9a-f]{40}$",
|
||||||
|
"the fixture must record the full commit its anchors were taken at",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_fixture_anchors_are_unique_and_well_formed(self):
|
||||||
|
seen = set()
|
||||||
|
for entry in self.fixture["anchors"]:
|
||||||
|
anchor = entry["anchor"]
|
||||||
|
self.assertNotIn(anchor, seen, f"duplicate anchor entry: {anchor}")
|
||||||
|
seen.add(anchor)
|
||||||
|
self.assertRegex(anchor, r"^[A-Za-z0-9_./-]+\.py:[1-9]\d*$", anchor)
|
||||||
|
self.assertTrue(
|
||||||
|
(entry.get("expect") or "").strip(),
|
||||||
|
f"anchor {anchor} declares no 'expect' substring, so it proves nothing",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class ThreatModelAnchorResolutionTests(unittest.TestCase):
|
||||||
|
"""#956 required positive test: every anchor resolves at the reviewed commit."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self.fixture = json.loads(_read(FIXTURE_PATH))
|
||||||
|
self.doc = _read(DOC_PATH)
|
||||||
|
|
||||||
|
def test_every_declared_anchor_resolves_to_the_claimed_source_line(self):
|
||||||
|
failures = []
|
||||||
|
for entry in self.fixture["anchors"]:
|
||||||
|
rel_path, _, raw_lineno = entry["anchor"].partition(":")
|
||||||
|
lineno = int(raw_lineno)
|
||||||
|
line = _source_line(rel_path, lineno)
|
||||||
|
if line is None:
|
||||||
|
failures.append(f"{entry['anchor']}: file or line does not exist")
|
||||||
|
continue
|
||||||
|
if entry["expect"] not in line:
|
||||||
|
failures.append(
|
||||||
|
f"{entry['anchor']}: expected {entry['expect']!r}, "
|
||||||
|
f"found {line.strip()!r}"
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
[], failures, "unresolved threat-model anchors:\n" + "\n".join(failures)
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_every_anchor_cited_in_the_document_is_declared_in_the_fixture(self):
|
||||||
|
declared = {e["anchor"] for e in self.fixture["anchors"]}
|
||||||
|
cited = {f"{m.group(1)}:{m.group(2)}" for m in ANCHOR_RE.finditer(self.doc)}
|
||||||
|
undeclared = sorted(cited - declared)
|
||||||
|
self.assertEqual(
|
||||||
|
[],
|
||||||
|
undeclared,
|
||||||
|
"document cites anchors that no test verifies: " + ", ".join(undeclared),
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_the_document_actually_cites_anchors(self):
|
||||||
|
cited = {f"{m.group(1)}:{m.group(2)}" for m in ANCHOR_RE.finditer(self.doc)}
|
||||||
|
self.assertGreaterEqual(
|
||||||
|
len(cited),
|
||||||
|
30,
|
||||||
|
"a boundary document with almost no anchors is not traceable",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_unresolvable_anchor_is_detected(self):
|
||||||
|
"""Negative control: the checker must fail on a deliberately bad anchor.
|
||||||
|
|
||||||
|
Without this, a checker that silently passed everything would look
|
||||||
|
identical to a correct one.
|
||||||
|
"""
|
||||||
|
self.assertIsNone(_source_line("gitea_config.py", 10**9))
|
||||||
|
self.assertIsNone(_source_line("no_such_module_for_956.py", 1))
|
||||||
|
real = _source_line("gitea_config.py", 54)
|
||||||
|
self.assertIsNotNone(real)
|
||||||
|
self.assertNotIn("this substring is not on that line", real)
|
||||||
|
|
||||||
|
|
||||||
|
class ThreatModelStructureTests(unittest.TestCase):
|
||||||
|
"""The document must contain what #956's acceptance criteria demand."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self.doc = _read(DOC_PATH)
|
||||||
|
|
||||||
|
def test_records_the_commit_it_was_generated_against(self):
|
||||||
|
fixture = json.loads(_read(FIXTURE_PATH))
|
||||||
|
self.assertIn(
|
||||||
|
fixture["generated_against_commit"],
|
||||||
|
self.doc,
|
||||||
|
"the document must state the commit its anchors resolve at",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_names_every_required_adversary(self):
|
||||||
|
low = self.doc.lower()
|
||||||
|
for adversary in REQUIRED_ADVERSARIES:
|
||||||
|
self.assertIn(adversary.lower(), low, f"adversary not covered: {adversary}")
|
||||||
|
|
||||||
|
def test_maps_every_epic_child_from_two_through_ten(self):
|
||||||
|
for number in REQUIRED_CHILDREN:
|
||||||
|
self.assertIn(
|
||||||
|
f"#{number}",
|
||||||
|
self.doc,
|
||||||
|
f"epic child #{number} is not mapped to a boundary",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_credential_rows_declare_holder_boundary_and_blast_radius(self):
|
||||||
|
for column in ("Holder", "Boundary", "Blast radius"):
|
||||||
|
self.assertIn(
|
||||||
|
column,
|
||||||
|
self.doc,
|
||||||
|
f"the credential inventory must state each credential's {column.lower()}",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_states_an_explicit_co_residency_ruling(self):
|
||||||
|
"""AC3/AC5: an explicit ruling, not an implication."""
|
||||||
|
self.assertIsNotNone(
|
||||||
|
_heading_re("Decomposition ruling").search(self.doc),
|
||||||
|
"the document must contain an explicit decomposition-ruling section",
|
||||||
|
)
|
||||||
|
for service in ("Jenkins", "GlitchTip", "Sentry", "database"):
|
||||||
|
self.assertIn(service, self.doc, f"ruling does not address {service}")
|
||||||
|
self.assertRegex(
|
||||||
|
self.doc,
|
||||||
|
r"D1\b.*must not",
|
||||||
|
"the ruling must state the prohibition, not merely discuss it",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_contains_the_compromised_client_walkthrough(self):
|
||||||
|
"""#956 required negative/adversarial test."""
|
||||||
|
self.assertIsNotNone(
|
||||||
|
_heading_re("Adversarial walkthrough").search(self.doc),
|
||||||
|
"the required compromised-client walkthrough is missing",
|
||||||
|
)
|
||||||
|
self.assertIn("Before the migration", self.doc)
|
||||||
|
self.assertIn("After the migration", self.doc)
|
||||||
|
|
||||||
|
def test_every_boundary_states_what_it_protects_and_what_crossing_requires(self):
|
||||||
|
boundary_ids = set(re.findall(r"\bB(\d+)\b", self.doc))
|
||||||
|
self.assertGreaterEqual(
|
||||||
|
len(boundary_ids), 5, "too few trust boundaries to be a decomposition"
|
||||||
|
)
|
||||||
|
for column in (
|
||||||
|
"Protects",
|
||||||
|
"Crossing requires today",
|
||||||
|
"Crossing must require remotely",
|
||||||
|
):
|
||||||
|
self.assertIn(column, self.doc, f"boundary table is missing '{column}'")
|
||||||
|
|
||||||
|
def test_declares_itself_documentation_only(self):
|
||||||
|
self.assertIn("documentation only", self.doc.lower())
|
||||||
|
|
||||||
|
|
||||||
|
class ThreatModelConsistencyTests(unittest.TestCase):
|
||||||
|
"""Counts stated in prose must match the rows actually present."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self.doc = _read(DOC_PATH)
|
||||||
|
|
||||||
|
def _declared_ids(self, prefix):
|
||||||
|
# Table rows begin '| CR1 |' / '| B3 |' / '| A2 |'.
|
||||||
|
return sorted(
|
||||||
|
{
|
||||||
|
int(m)
|
||||||
|
for m in re.findall(
|
||||||
|
r"^\|\s*%s(\d+)\s*\|" % prefix, self.doc, re.MULTILINE
|
||||||
|
)
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_identifier_sequences_have_no_gaps(self):
|
||||||
|
for prefix, label in (
|
||||||
|
("A", "assets"),
|
||||||
|
("B", "boundaries"),
|
||||||
|
("CR", "credentials"),
|
||||||
|
):
|
||||||
|
ids = self._declared_ids(prefix)
|
||||||
|
self.assertTrue(ids, f"no {label} declared")
|
||||||
|
self.assertEqual(
|
||||||
|
list(range(1, len(ids) + 1)),
|
||||||
|
ids,
|
||||||
|
f"{label} identifiers must run 1..n with no gaps; got {ids}",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_stated_credential_count_matches_the_rows(self):
|
||||||
|
ids = self._declared_ids("CR")
|
||||||
|
match = re.search(r"(\d+)\s+credential(?:s)? in total", self.doc)
|
||||||
|
self.assertIsNotNone(match, "the credential inventory must state its own total")
|
||||||
|
self.assertEqual(
|
||||||
|
len(ids),
|
||||||
|
int(match.group(1)),
|
||||||
|
"stated credential total disagrees with the number of rows",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_every_boundary_is_owned_by_at_least_one_child(self):
|
||||||
|
"""Each boundary must be owned by a child *in the mapping table*.
|
||||||
|
|
||||||
|
Scanning the whole section would let a prose summary line ("Boundary
|
||||||
|
coverage: ... B5 (#936)") satisfy the assertion while the table row
|
||||||
|
that actually assigns the owner had been emptied — verified by
|
||||||
|
deliberately blanking a row and watching a whole-section check still
|
||||||
|
pass. Only table rows count.
|
||||||
|
"""
|
||||||
|
mapping_section = _section_body(self.doc, "Child-to-boundary mapping")
|
||||||
|
self.assertIsNotNone(
|
||||||
|
mapping_section, "child-to-boundary mapping section is missing"
|
||||||
|
)
|
||||||
|
rows = [
|
||||||
|
line
|
||||||
|
for line in mapping_section.splitlines()
|
||||||
|
if line.lstrip().startswith("|") and re.search(r"#93\d", line)
|
||||||
|
]
|
||||||
|
self.assertGreaterEqual(
|
||||||
|
len(rows), len(REQUIRED_CHILDREN), "mapping table has too few child rows"
|
||||||
|
)
|
||||||
|
mapped = set(re.findall(r"\bB(\d+)\b", "\n".join(rows)))
|
||||||
|
declared = {str(i) for i in self._declared_ids("B")}
|
||||||
|
unmapped = sorted(declared - mapped, key=int)
|
||||||
|
self.assertEqual(
|
||||||
|
[],
|
||||||
|
unmapped,
|
||||||
|
"boundaries with no owning child: " + ", ".join("B" + u for u in unmapped),
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,480 @@
|
|||||||
|
"""Tests for dead-owner / PID-reuse session retirement (#969)."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import tempfile
|
||||||
|
import threading
|
||||||
|
import unittest
|
||||||
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||||
|
from datetime import datetime, timedelta, timezone
|
||||||
|
|
||||||
|
import control_plane_db as cpd
|
||||||
|
import post_restart_reconcile as prr
|
||||||
|
import session_lifecycle as sl
|
||||||
|
|
||||||
|
NOW = datetime(2026, 7, 29, 12, 0, 0, tzinfo=timezone.utc)
|
||||||
|
EARLIER = NOW - timedelta(hours=2)
|
||||||
|
LATER = NOW + timedelta(minutes=5)
|
||||||
|
|
||||||
|
|
||||||
|
def _session(
|
||||||
|
session_id: str,
|
||||||
|
*,
|
||||||
|
pid: int | None = 4242,
|
||||||
|
status: str = "active",
|
||||||
|
started_at: datetime = EARLIER,
|
||||||
|
last_heartbeat_at: datetime | None = None,
|
||||||
|
client_managed: bool = False,
|
||||||
|
owner_process_started_at: datetime | None = None,
|
||||||
|
role: str = "author",
|
||||||
|
) -> dict:
|
||||||
|
hb = last_heartbeat_at if last_heartbeat_at is not None else started_at
|
||||||
|
row = {
|
||||||
|
"session_id": session_id,
|
||||||
|
"role": role,
|
||||||
|
"profile": "prgs-author",
|
||||||
|
"pid": pid,
|
||||||
|
"status": status,
|
||||||
|
"started_at": cpd._ts(started_at),
|
||||||
|
"last_heartbeat_at": cpd._ts(hb),
|
||||||
|
"client_managed": client_managed,
|
||||||
|
}
|
||||||
|
if owner_process_started_at is not None:
|
||||||
|
row["owner_process_started_at"] = cpd._ts(owner_process_started_at)
|
||||||
|
return row
|
||||||
|
|
||||||
|
|
||||||
|
def _alive(pids: set[int]):
|
||||||
|
def _check(pid):
|
||||||
|
try:
|
||||||
|
return int(pid) in pids
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return False
|
||||||
|
|
||||||
|
return _check
|
||||||
|
|
||||||
|
|
||||||
|
def _starts(mapping: dict[int, datetime]):
|
||||||
|
def _probe(pid):
|
||||||
|
try:
|
||||||
|
return mapping.get(int(pid))
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
return None
|
||||||
|
|
||||||
|
return _probe
|
||||||
|
|
||||||
|
|
||||||
|
class ClassifyDeadOwnerTests(unittest.TestCase):
|
||||||
|
def test_dead_owner_is_stale_and_retireable(self) -> None:
|
||||||
|
c = sl.classify_session(
|
||||||
|
_session("ghost", pid=2_000_000_000),
|
||||||
|
now=NOW,
|
||||||
|
pid_checker=_alive(set()),
|
||||||
|
process_start_probe=_starts({}),
|
||||||
|
)
|
||||||
|
self.assertEqual(c.classification, sl.CLASS_STALE)
|
||||||
|
self.assertTrue(c.retireable)
|
||||||
|
self.assertIn(c.reason, {sl.REASON_DEAD_OWNER, sl.REASON_HEARTBEAT_STALE_DEAD})
|
||||||
|
|
||||||
|
def test_missing_pid_is_stale(self) -> None:
|
||||||
|
c = sl.classify_session(
|
||||||
|
_session("no-pid", pid=None),
|
||||||
|
now=NOW,
|
||||||
|
pid_checker=_alive(set()),
|
||||||
|
)
|
||||||
|
self.assertEqual(c.classification, sl.CLASS_STALE)
|
||||||
|
self.assertEqual(c.reason, sl.REASON_MISSING_PID)
|
||||||
|
self.assertTrue(c.retireable)
|
||||||
|
|
||||||
|
|
||||||
|
class PidReuseTests(unittest.TestCase):
|
||||||
|
def test_pid_reuse_marks_stale_not_live(self) -> None:
|
||||||
|
# Process with same PID started AFTER the session was recorded.
|
||||||
|
c = sl.classify_session(
|
||||||
|
_session("reused", pid=77, started_at=EARLIER, last_heartbeat_at=EARLIER),
|
||||||
|
now=NOW,
|
||||||
|
pid_checker=_alive({77}),
|
||||||
|
process_start_probe=_starts({77: LATER}),
|
||||||
|
)
|
||||||
|
self.assertEqual(c.classification, sl.CLASS_STALE)
|
||||||
|
self.assertEqual(c.reason, sl.REASON_PID_REUSE)
|
||||||
|
self.assertTrue(c.pid_reused)
|
||||||
|
self.assertTrue(c.retireable)
|
||||||
|
|
||||||
|
def test_matching_process_start_is_live(self) -> None:
|
||||||
|
c = sl.classify_session(
|
||||||
|
_session(
|
||||||
|
"same-proc",
|
||||||
|
pid=88,
|
||||||
|
started_at=EARLIER,
|
||||||
|
last_heartbeat_at=NOW - timedelta(seconds=30),
|
||||||
|
owner_process_started_at=EARLIER - timedelta(seconds=5),
|
||||||
|
),
|
||||||
|
now=NOW,
|
||||||
|
pid_checker=_alive({88}),
|
||||||
|
process_start_probe=_starts({88: EARLIER - timedelta(seconds=5)}),
|
||||||
|
)
|
||||||
|
self.assertEqual(c.classification, sl.CLASS_LIVE)
|
||||||
|
self.assertFalse(c.retireable)
|
||||||
|
|
||||||
|
|
||||||
|
class LiveOwnerAndLeaseTests(unittest.TestCase):
|
||||||
|
def test_live_owner_not_retired(self) -> None:
|
||||||
|
c = sl.classify_session(
|
||||||
|
_session(
|
||||||
|
"live",
|
||||||
|
pid=os.getpid(),
|
||||||
|
last_heartbeat_at=NOW - timedelta(seconds=10),
|
||||||
|
),
|
||||||
|
now=NOW,
|
||||||
|
pid_checker=_alive({os.getpid()}),
|
||||||
|
process_start_probe=_starts({os.getpid(): EARLIER}),
|
||||||
|
)
|
||||||
|
self.assertEqual(c.classification, sl.CLASS_LIVE)
|
||||||
|
self.assertFalse(c.retireable)
|
||||||
|
|
||||||
|
def test_live_lease_blocks_retirement_even_if_pid_dead(self) -> None:
|
||||||
|
c = sl.classify_session(
|
||||||
|
_session("leased", pid=99999),
|
||||||
|
now=NOW,
|
||||||
|
pid_checker=_alive(set()),
|
||||||
|
live_lease_sessions={"leased"},
|
||||||
|
)
|
||||||
|
self.assertEqual(c.classification, sl.CLASS_PROTECTED)
|
||||||
|
self.assertEqual(c.reason, sl.REASON_LIVE_LEASE)
|
||||||
|
self.assertFalse(c.retireable)
|
||||||
|
|
||||||
|
def test_client_managed_live_never_retired(self) -> None:
|
||||||
|
c = sl.classify_session(
|
||||||
|
_session(
|
||||||
|
"client",
|
||||||
|
pid=55,
|
||||||
|
client_managed=True,
|
||||||
|
last_heartbeat_at=NOW - timedelta(seconds=5),
|
||||||
|
),
|
||||||
|
now=NOW,
|
||||||
|
pid_checker=_alive({55}),
|
||||||
|
process_start_probe=_starts({55: EARLIER}),
|
||||||
|
)
|
||||||
|
self.assertEqual(c.classification, sl.CLASS_LIVE)
|
||||||
|
self.assertEqual(c.reason, sl.REASON_CLIENT_MANAGED_LIVE)
|
||||||
|
self.assertFalse(c.retireable)
|
||||||
|
|
||||||
|
def test_client_managed_dead_pid_is_retireable(self) -> None:
|
||||||
|
# Dead client process is not a live client-managed session.
|
||||||
|
c = sl.classify_session(
|
||||||
|
_session("client-dead", pid=56, client_managed=True),
|
||||||
|
now=NOW,
|
||||||
|
pid_checker=_alive(set()),
|
||||||
|
)
|
||||||
|
self.assertEqual(c.classification, sl.CLASS_STALE)
|
||||||
|
self.assertTrue(c.retireable)
|
||||||
|
|
||||||
|
|
||||||
|
class TerminalAndDisconnectedTests(unittest.TestCase):
|
||||||
|
def test_already_terminal_not_retireable(self) -> None:
|
||||||
|
c = sl.classify_session(
|
||||||
|
_session("done", status="retired", pid=1),
|
||||||
|
now=NOW,
|
||||||
|
pid_checker=_alive(set()),
|
||||||
|
)
|
||||||
|
self.assertEqual(c.classification, sl.CLASS_TERMINAL)
|
||||||
|
self.assertFalse(c.retireable)
|
||||||
|
|
||||||
|
def test_alive_stale_heartbeat_is_disconnected_not_retired(self) -> None:
|
||||||
|
c = sl.classify_session(
|
||||||
|
_session(
|
||||||
|
"quiet",
|
||||||
|
pid=66,
|
||||||
|
last_heartbeat_at=NOW - timedelta(hours=5),
|
||||||
|
),
|
||||||
|
now=NOW,
|
||||||
|
pid_checker=_alive({66}),
|
||||||
|
process_start_probe=_starts({66: EARLIER}),
|
||||||
|
)
|
||||||
|
self.assertEqual(c.classification, sl.CLASS_DISCONNECTED)
|
||||||
|
self.assertFalse(c.retireable)
|
||||||
|
|
||||||
|
|
||||||
|
class FleetAndApplyTests(unittest.TestCase):
|
||||||
|
def setUp(self) -> None:
|
||||||
|
self._tmp = tempfile.TemporaryDirectory()
|
||||||
|
self.db_path = os.path.join(self._tmp.name, "cp.sqlite3")
|
||||||
|
self.db = cpd.ControlPlaneDB(self.db_path)
|
||||||
|
|
||||||
|
def tearDown(self) -> None:
|
||||||
|
self._tmp.cleanup()
|
||||||
|
|
||||||
|
def test_mixed_fleet_and_apply_retires_only_stale(self) -> None:
|
||||||
|
live_pid = os.getpid()
|
||||||
|
# Use wall-clock "now" so upsert timestamps align with classification.
|
||||||
|
moment = datetime.now(timezone.utc)
|
||||||
|
proc_start = moment - timedelta(hours=1)
|
||||||
|
self.db.upsert_session(
|
||||||
|
session_id="s-live",
|
||||||
|
role="author",
|
||||||
|
pid=live_pid,
|
||||||
|
status="active",
|
||||||
|
owner_process_started_at=cpd._ts(proc_start),
|
||||||
|
)
|
||||||
|
self.db.upsert_session(
|
||||||
|
session_id="s-dead", role="reviewer", pid=2_000_000_001, status="active"
|
||||||
|
)
|
||||||
|
self.db.upsert_session(
|
||||||
|
session_id="s-ended", role="merger", pid=3, status="ended"
|
||||||
|
)
|
||||||
|
|
||||||
|
sessions = self.db.list_sessions(limit=50)
|
||||||
|
report = sl.classify_sessions(
|
||||||
|
sessions,
|
||||||
|
leases=[],
|
||||||
|
now=moment,
|
||||||
|
pid_checker=_alive({live_pid}),
|
||||||
|
process_start_probe=_starts({live_pid: proc_start}),
|
||||||
|
)
|
||||||
|
self.assertGreaterEqual(report.stale_count, 1)
|
||||||
|
self.assertTrue(
|
||||||
|
any(c.session_id == "s-dead" and c.retireable for c in report.classifications)
|
||||||
|
)
|
||||||
|
self.assertTrue(
|
||||||
|
any(
|
||||||
|
c.session_id == "s-live" and not c.retireable
|
||||||
|
for c in report.classifications
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
first = sl.apply_session_retirements(
|
||||||
|
self.db, report, dry_run=False, actor_session_id="actor-1", now=moment
|
||||||
|
)
|
||||||
|
self.assertGreaterEqual(first["retired_count"], 1)
|
||||||
|
# After retirement, active list should exclude s-dead.
|
||||||
|
active = {
|
||||||
|
s["session_id"]
|
||||||
|
for s in self.db.list_sessions(statuses=("active",), limit=50)
|
||||||
|
}
|
||||||
|
self.assertNotIn("s-dead", active)
|
||||||
|
self.assertIn("s-live", active)
|
||||||
|
|
||||||
|
# Repeated cleanup is idempotent.
|
||||||
|
report2 = sl.classify_sessions(
|
||||||
|
self.db.list_sessions(limit=50),
|
||||||
|
leases=[],
|
||||||
|
now=moment,
|
||||||
|
pid_checker=_alive({live_pid}),
|
||||||
|
process_start_probe=_starts({live_pid: proc_start}),
|
||||||
|
)
|
||||||
|
second = sl.apply_session_retirements(
|
||||||
|
self.db, report2, dry_run=False, actor_session_id="actor-1", now=moment
|
||||||
|
)
|
||||||
|
# No double-retirement of the same row as a new mutation.
|
||||||
|
self.assertEqual(second["retired_count"], 0)
|
||||||
|
|
||||||
|
# Durable audit event present.
|
||||||
|
import sqlite3
|
||||||
|
|
||||||
|
conn = sqlite3.connect(self.db_path)
|
||||||
|
try:
|
||||||
|
events = conn.execute(
|
||||||
|
"SELECT event_type, message FROM events WHERE event_type = ?",
|
||||||
|
("session_retired",),
|
||||||
|
).fetchall()
|
||||||
|
finally:
|
||||||
|
conn.close()
|
||||||
|
self.assertTrue(events)
|
||||||
|
self.assertTrue(any("s-dead" in (m or "") for _, m in events))
|
||||||
|
|
||||||
|
def test_live_lease_blocks_db_retirement(self) -> None:
|
||||||
|
self.db.upsert_session(
|
||||||
|
session_id="s-leased", role="author", pid=2_000_000_002, status="active"
|
||||||
|
)
|
||||||
|
self.db.upsert_work_item(
|
||||||
|
remote="prgs",
|
||||||
|
org="org",
|
||||||
|
repo="repo",
|
||||||
|
kind="issue",
|
||||||
|
number=969,
|
||||||
|
)
|
||||||
|
result = self.db.assign_and_lease(
|
||||||
|
session_id="s-leased",
|
||||||
|
role="author",
|
||||||
|
remote="prgs",
|
||||||
|
org="org",
|
||||||
|
repo="repo",
|
||||||
|
kind="issue",
|
||||||
|
number=969,
|
||||||
|
)
|
||||||
|
self.assertEqual(result.outcome, "assigned")
|
||||||
|
|
||||||
|
# Inventory-style lease with explicit live freshness (authoritative for
|
||||||
|
# the pure classifier). DB apply also blocks on the active lease row.
|
||||||
|
leases = [
|
||||||
|
{
|
||||||
|
"lease_id": result.lease_id,
|
||||||
|
"session_id": "s-leased",
|
||||||
|
"status": "active",
|
||||||
|
"freshness": {"freshness": "active"},
|
||||||
|
}
|
||||||
|
]
|
||||||
|
report = sl.classify_sessions(
|
||||||
|
self.db.list_sessions(statuses=("active",), limit=20),
|
||||||
|
leases=leases,
|
||||||
|
now=NOW,
|
||||||
|
pid_checker=_alive(set()),
|
||||||
|
)
|
||||||
|
# Classifier protects via live lease set.
|
||||||
|
self.assertTrue(
|
||||||
|
any(
|
||||||
|
c.session_id == "s-leased" and c.classification == sl.CLASS_PROTECTED
|
||||||
|
for c in report.classifications
|
||||||
|
)
|
||||||
|
)
|
||||||
|
apply = sl.apply_session_retirements(
|
||||||
|
self.db, report, dry_run=False, actor_session_id="actor", now=NOW
|
||||||
|
)
|
||||||
|
self.assertEqual(apply["retired_count"], 0)
|
||||||
|
active = {
|
||||||
|
s["session_id"]
|
||||||
|
for s in self.db.list_sessions(statuses=("active",), limit=20)
|
||||||
|
}
|
||||||
|
self.assertIn("s-leased", active)
|
||||||
|
|
||||||
|
# Direct DB CAS also refuses while an active lease row remains.
|
||||||
|
blocked = self.db.retire_session(
|
||||||
|
session_id="s-leased",
|
||||||
|
reason=sl.REASON_DEAD_OWNER,
|
||||||
|
actor_session_id="actor",
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
self.assertEqual(blocked["outcome"], "blocked")
|
||||||
|
self.assertEqual(blocked["reason"], "live_lease")
|
||||||
|
|
||||||
|
def test_concurrent_retirement_is_idempotent(self) -> None:
|
||||||
|
for i in range(20):
|
||||||
|
self.db.upsert_session(
|
||||||
|
session_id=f"ghost-{i}",
|
||||||
|
role="author",
|
||||||
|
pid=3_000_000 + i,
|
||||||
|
status="active",
|
||||||
|
)
|
||||||
|
|
||||||
|
def _worker() -> dict:
|
||||||
|
return sl.retire_stale_sessions(
|
||||||
|
self.db,
|
||||||
|
dry_run=False,
|
||||||
|
actor_session_id=f"actor-{threading.get_ident()}",
|
||||||
|
now=NOW,
|
||||||
|
pid_checker=_alive(set()),
|
||||||
|
process_start_probe=_starts({}),
|
||||||
|
session_limit=100,
|
||||||
|
)
|
||||||
|
|
||||||
|
outcomes = []
|
||||||
|
with ThreadPoolExecutor(max_workers=4) as pool:
|
||||||
|
futs = [pool.submit(_worker) for _ in range(4)]
|
||||||
|
for fut in as_completed(futs):
|
||||||
|
outcomes.append(fut.result())
|
||||||
|
|
||||||
|
total_retired = sum(o["apply"]["retired_count"] for o in outcomes)
|
||||||
|
# Exactly one successful retirement per ghost row across all workers.
|
||||||
|
self.assertEqual(total_retired, 20)
|
||||||
|
active = {
|
||||||
|
s["session_id"]
|
||||||
|
for s in self.db.list_sessions(statuses=("active",), limit=100)
|
||||||
|
}
|
||||||
|
for i in range(20):
|
||||||
|
self.assertNotIn(f"ghost-{i}", active)
|
||||||
|
|
||||||
|
|
||||||
|
class ReconcileIntegrationTests(unittest.TestCase):
|
||||||
|
def test_unresolved_until_retired_then_resolved(self) -> None:
|
||||||
|
inv = {
|
||||||
|
"inventory_complete": True,
|
||||||
|
"incomplete_reasons": [],
|
||||||
|
"service_health": {"healthy": True},
|
||||||
|
"clients": [{"session_id": "c1", "connected": True}],
|
||||||
|
"sessions": [
|
||||||
|
_session("ghost", pid=2_000_000_099, last_heartbeat_at=EARLIER),
|
||||||
|
],
|
||||||
|
"leases": [],
|
||||||
|
"checkpoints_available": False,
|
||||||
|
"worktree_bindings": [],
|
||||||
|
"pending_mutations": [],
|
||||||
|
"capabilities": {"stale": False},
|
||||||
|
"boot_head_sha": "a" * 40,
|
||||||
|
"current_head_sha": "a" * 40,
|
||||||
|
"queue_state": {"safe_to_resume": True},
|
||||||
|
}
|
||||||
|
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_LOG_ONLY)
|
||||||
|
sess = next(i for i in proof.items if i.dimension == prr.DIM_SESSIONS)
|
||||||
|
self.assertEqual(sess.status, prr.ITEM_UNRESOLVED)
|
||||||
|
self.assertIn("ghost", sess.details.get("orphan_session_ids") or [])
|
||||||
|
|
||||||
|
# After retirement inventory (no active orphans) resolves.
|
||||||
|
inv2 = dict(inv)
|
||||||
|
inv2["sessions"] = []
|
||||||
|
inv2["session_fleet"] = {
|
||||||
|
"retireable_session_ids": [],
|
||||||
|
"sessions_dimension_resolved": True,
|
||||||
|
"live_count": 0,
|
||||||
|
"stale_count": 0,
|
||||||
|
}
|
||||||
|
proof2 = prr.reconcile_after_restart(inv2, now=NOW, mode=prr.MODE_LOG_ONLY)
|
||||||
|
sess2 = next(i for i in proof2.items if i.dimension == prr.DIM_SESSIONS)
|
||||||
|
self.assertEqual(sess2.status, prr.ITEM_RESOLVED)
|
||||||
|
|
||||||
|
def test_legacy_orphan_key_still_populated(self) -> None:
|
||||||
|
inv = {
|
||||||
|
"inventory_complete": True,
|
||||||
|
"service_health": {"healthy": True},
|
||||||
|
"clients": [],
|
||||||
|
"sessions": [_session("ghost", pid=2_000_000_100)],
|
||||||
|
"leases": [],
|
||||||
|
"checkpoints_available": False,
|
||||||
|
"worktree_bindings": [],
|
||||||
|
"pending_mutations": [],
|
||||||
|
"capabilities": {"stale": False},
|
||||||
|
"boot_head_sha": "a" * 40,
|
||||||
|
"current_head_sha": "a" * 40,
|
||||||
|
"queue_state": {"safe_to_resume": True},
|
||||||
|
}
|
||||||
|
proof = prr.reconcile_after_restart(inv, now=NOW)
|
||||||
|
sess = next(i for i in proof.items if i.dimension == prr.DIM_SESSIONS)
|
||||||
|
self.assertIn("orphan_session_ids", sess.details)
|
||||||
|
|
||||||
|
|
||||||
|
class SchemaMigrationTests(unittest.TestCase):
|
||||||
|
def test_lifecycle_columns_present(self) -> None:
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "cp.sqlite3")
|
||||||
|
db = cpd.ControlPlaneDB(path)
|
||||||
|
db.upsert_session(session_id="s1", role="author", pid=1)
|
||||||
|
out = db.retire_session(
|
||||||
|
session_id="s1",
|
||||||
|
reason=sl.REASON_DEAD_OWNER,
|
||||||
|
actor_session_id="tester",
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
self.assertEqual(out["outcome"], "retired")
|
||||||
|
rows = db.list_sessions(limit=5)
|
||||||
|
# May not appear under active filter
|
||||||
|
all_rows = db.list_sessions(limit=5)
|
||||||
|
# Re-open raw to check columns
|
||||||
|
import sqlite3
|
||||||
|
|
||||||
|
conn = sqlite3.connect(path)
|
||||||
|
try:
|
||||||
|
cols = {r[1] for r in conn.execute("PRAGMA table_info(sessions)")}
|
||||||
|
version = conn.execute(
|
||||||
|
"SELECT value FROM schema_meta WHERE key='schema_version'"
|
||||||
|
).fetchone()[0]
|
||||||
|
finally:
|
||||||
|
conn.close()
|
||||||
|
self.assertIn("retired_at", cols)
|
||||||
|
self.assertIn("retire_reason", cols)
|
||||||
|
self.assertIn("owner_process_started_at", cols)
|
||||||
|
self.assertEqual(version, "6")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -0,0 +1,149 @@
|
|||||||
|
"""Unit tests for mcp_config_drift.py (#672)."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import pytest
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from mcp_config_drift import (
|
||||||
|
REQUIRED_GITEA_ROLE_SERVERS,
|
||||||
|
analyze_config_drift,
|
||||||
|
load_mcp_config,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def sample_global_config() -> dict:
|
||||||
|
return {
|
||||||
|
"mcpServers": {
|
||||||
|
"gitea-author": {
|
||||||
|
"command": "python3",
|
||||||
|
"args": ["gitea_mcp_server.py"],
|
||||||
|
"env": {"GITEA_MCP_PROFILE": "prgs-author", "SENTRY_AUTH_TOKEN": "secret-token-999"},
|
||||||
|
},
|
||||||
|
"gitea-reviewer": {
|
||||||
|
"command": "python3",
|
||||||
|
"args": ["gitea_mcp_server.py"],
|
||||||
|
"env": {"GITEA_MCP_PROFILE": "prgs-reviewer"},
|
||||||
|
},
|
||||||
|
"gitea-merger": {
|
||||||
|
"command": "python3",
|
||||||
|
"args": ["gitea_mcp_server.py"],
|
||||||
|
"env": {"GITEA_MCP_PROFILE": "prgs-merger"},
|
||||||
|
},
|
||||||
|
"gitea-reconciler": {
|
||||||
|
"command": "python3",
|
||||||
|
"args": ["gitea_mcp_server.py"],
|
||||||
|
"env": {"GITEA_MCP_PROFILE": "prgs-reconciler"},
|
||||||
|
},
|
||||||
|
"gitea-controller": {
|
||||||
|
"command": "python3",
|
||||||
|
"args": ["gitea_mcp_server.py"],
|
||||||
|
"env": {"GITEA_MCP_PROFILE": "prgs-controller"},
|
||||||
|
},
|
||||||
|
"gitea-tools": {
|
||||||
|
"command": "python3",
|
||||||
|
"args": ["gitea_mcp_server.py"],
|
||||||
|
"env": {"GITEA_MCP_PROFILE": "prgs-author"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def write_json(path: Path, data: dict) -> str:
|
||||||
|
path.write_text(json.dumps(data, indent=2), encoding="utf-8")
|
||||||
|
return str(path)
|
||||||
|
|
||||||
|
|
||||||
|
def test_drift_detection_in_sync(tmp_path, sample_global_config):
|
||||||
|
glob_file = tmp_path / "global_mcp.json"
|
||||||
|
act_file = tmp_path / "active_mcp.json"
|
||||||
|
|
||||||
|
write_json(glob_file, sample_global_config)
|
||||||
|
write_json(act_file, sample_global_config)
|
||||||
|
|
||||||
|
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||||
|
|
||||||
|
assert report["in_sync"] is True
|
||||||
|
assert report["missing_role_servers"] == []
|
||||||
|
assert report["profile_mismatches"] == []
|
||||||
|
assert set(report["present_role_servers"]) == set(REQUIRED_GITEA_ROLE_SERVERS)
|
||||||
|
|
||||||
|
|
||||||
|
def test_drift_detection_missing_author(tmp_path, sample_global_config):
|
||||||
|
glob_file = tmp_path / "global_mcp.json"
|
||||||
|
act_file = tmp_path / "active_mcp.json"
|
||||||
|
|
||||||
|
active_config = json.loads(json.dumps(sample_global_config))
|
||||||
|
del active_config["mcpServers"]["gitea-author"]
|
||||||
|
|
||||||
|
write_json(glob_file, sample_global_config)
|
||||||
|
write_json(act_file, active_config)
|
||||||
|
|
||||||
|
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||||
|
|
||||||
|
assert report["in_sync"] is False
|
||||||
|
assert "gitea-author" in report["missing_role_servers"]
|
||||||
|
assert "gitea-author" not in report["present_role_servers"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_drift_detection_missing_reviewer(tmp_path, sample_global_config):
|
||||||
|
glob_file = tmp_path / "global_mcp.json"
|
||||||
|
act_file = tmp_path / "active_mcp.json"
|
||||||
|
|
||||||
|
active_config = json.loads(json.dumps(sample_global_config))
|
||||||
|
del active_config["mcpServers"]["gitea-reviewer"]
|
||||||
|
|
||||||
|
write_json(glob_file, sample_global_config)
|
||||||
|
write_json(act_file, active_config)
|
||||||
|
|
||||||
|
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||||
|
|
||||||
|
assert report["in_sync"] is False
|
||||||
|
assert "gitea-reviewer" in report["missing_role_servers"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_drift_detection_profile_mismatch(tmp_path, sample_global_config):
|
||||||
|
glob_file = tmp_path / "global_mcp.json"
|
||||||
|
act_file = tmp_path / "active_mcp.json"
|
||||||
|
|
||||||
|
active_config = json.loads(json.dumps(sample_global_config))
|
||||||
|
active_config["mcpServers"]["gitea-author"]["env"]["GITEA_MCP_PROFILE"] = "dadeschools-author"
|
||||||
|
|
||||||
|
write_json(glob_file, sample_global_config)
|
||||||
|
write_json(act_file, active_config)
|
||||||
|
|
||||||
|
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||||
|
|
||||||
|
assert report["in_sync"] is False
|
||||||
|
assert len(report["profile_mismatches"]) == 1
|
||||||
|
mismatch = report["profile_mismatches"][0]
|
||||||
|
assert mismatch["server"] == "gitea-author"
|
||||||
|
assert mismatch["active_profile"] == "dadeschools-author"
|
||||||
|
assert mismatch["global_profile"] == "prgs-author"
|
||||||
|
|
||||||
|
|
||||||
|
def test_secret_redaction_in_drift_report(tmp_path, sample_global_config):
|
||||||
|
glob_file = tmp_path / "global_mcp.json"
|
||||||
|
act_file = tmp_path / "active_mcp.json"
|
||||||
|
|
||||||
|
write_json(glob_file, sample_global_config)
|
||||||
|
write_json(act_file, sample_global_config)
|
||||||
|
|
||||||
|
report = analyze_config_drift(active_config_path=str(act_file), global_config_path=str(glob_file))
|
||||||
|
serialized = str(report)
|
||||||
|
|
||||||
|
assert "secret-token-999" not in serialized
|
||||||
|
|
||||||
|
|
||||||
|
def test_sanctioned_runbook_forbids_pkill():
|
||||||
|
report = analyze_config_drift(active_config_path="/nonexistent/path/active.json", global_config_path="/nonexistent/path/global.json")
|
||||||
|
|
||||||
|
runbook_text = " ".join(report["sanctioned_repair_runbook"]).lower()
|
||||||
|
forbidden_text = " ".join(report["forbidden_repair_methods"]).lower()
|
||||||
|
|
||||||
|
assert "pkill" in forbidden_text
|
||||||
|
assert "mtime" in forbidden_text
|
||||||
|
assert "source" in forbidden_text
|
||||||
|
assert "session-state" in forbidden_text
|
||||||
@@ -0,0 +1,478 @@
|
|||||||
|
"""Concurrent-session MCP restart safety & dogfooding test suite (#666).
|
||||||
|
|
||||||
|
Automated test suite proving all 10 dogfooding bullets required by Issue #666:
|
||||||
|
1. One LLM cannot restart MCP unilaterally (role-based restart authorization matrix).
|
||||||
|
2. New work stops during drain (assignments_stopped gate enforcement).
|
||||||
|
3. Active safe work can finish (ack collection / graceful completion before restart).
|
||||||
|
4. Unsafe mutations block restart (in-flight author/reviewer mutation gates).
|
||||||
|
5. Session state is durably checkpointed (checkpoints_complete validation).
|
||||||
|
6. Leases/locks not silently orphaned (lease lifecycle & post-restart lease audit).
|
||||||
|
7. Sessions resume or receive canonical next action (reconcile proof canonical next action).
|
||||||
|
8. Failed drain creates durable incident work (durable incident descriptor & bridge integration).
|
||||||
|
9. Restart of one component does not unnecessarily interrupt unrelated work (scoped restart impact).
|
||||||
|
10. Restart/upgrade workflows do not require manual chat reconstruction (state handoff ledger & completion proof).
|
||||||
|
|
||||||
|
Links parent #655, vision #652, roadmap #653, #658, #659, #660, #661, #662, #663.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import unittest
|
||||||
|
from datetime import datetime, timedelta, timezone
|
||||||
|
|
||||||
|
import drain_proof as dp
|
||||||
|
import mcp_restart_paths as rp
|
||||||
|
import post_restart_reconcile as prr
|
||||||
|
import restart_coordinator as rc
|
||||||
|
from restart_coordinator import RestartClass
|
||||||
|
|
||||||
|
NOW = datetime(2026, 7, 25, 12, 0, 0, tzinfo=timezone.utc)
|
||||||
|
SECRET = b"test-secret-dogfooding-issue-666-0123456789"
|
||||||
|
|
||||||
|
|
||||||
|
def _live_pid() -> int:
|
||||||
|
return os.getpid()
|
||||||
|
|
||||||
|
|
||||||
|
def _clean_drain_state() -> dict:
|
||||||
|
return {
|
||||||
|
"assignments_stopped": True,
|
||||||
|
"checkpoints_complete": True,
|
||||||
|
"handoffs_verified": True,
|
||||||
|
"leases_handled": True,
|
||||||
|
"acks": {},
|
||||||
|
"ack_timeout_policy_applied": False,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _clean_inventory() -> dict:
|
||||||
|
return {
|
||||||
|
"service_health": {"healthy": True},
|
||||||
|
"clients": [],
|
||||||
|
"sessions": [
|
||||||
|
{
|
||||||
|
"session_id": "prgs-controller-1",
|
||||||
|
"role": "controller",
|
||||||
|
"profile": "prgs-controller",
|
||||||
|
"pid": _live_pid(),
|
||||||
|
"status": "active",
|
||||||
|
"last_heartbeat_at": NOW.isoformat(),
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"checkpoints": [],
|
||||||
|
"leases": [],
|
||||||
|
"capabilities": {},
|
||||||
|
"worktree_bindings": [],
|
||||||
|
"pending_mutations": [],
|
||||||
|
"inventory_complete": True,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class TestBullet1UnilateralRestartForbidden(unittest.TestCase):
|
||||||
|
"""Bullet 1: One LLM cannot restart MCP unilaterally."""
|
||||||
|
|
||||||
|
def test_worker_role_unilateral_full_restart_denied(self):
|
||||||
|
policy = rc.RESTART_CLASS_POLICIES[RestartClass.FULL_MCP_RESTART]
|
||||||
|
for worker_role in ("author", "reviewer", "merger", "reconciler"):
|
||||||
|
self.assertNotIn(
|
||||||
|
worker_role,
|
||||||
|
policy.request_roles,
|
||||||
|
f"Worker role '{worker_role}' must not unilaterally authorize FULL_MCP_RESTART",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_privileged_role_full_restart_authorized(self):
|
||||||
|
policy = rc.RESTART_CLASS_POLICIES[RestartClass.FULL_MCP_RESTART]
|
||||||
|
for priv_role in ("controller", "operator", "admin"):
|
||||||
|
self.assertIn(
|
||||||
|
priv_role,
|
||||||
|
policy.request_roles,
|
||||||
|
f"Privileged role '{priv_role}' must be authorized for FULL_MCP_RESTART",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_evaluate_impact_records_unauthorized_worker_request(self):
|
||||||
|
report = rc.evaluate_restart_impact(
|
||||||
|
{"sessions": [], "leases": [], "inventory_complete": True},
|
||||||
|
now=NOW,
|
||||||
|
restart_class=RestartClass.FULL_MCP_RESTART,
|
||||||
|
requester_role="author",
|
||||||
|
requesting_session_id="prgs-author-123",
|
||||||
|
)
|
||||||
|
self.assertFalse(report.role_authorized)
|
||||||
|
self.assertEqual(report.verdict, rc.VERDICT_UNSAFE)
|
||||||
|
self.assertTrue(any("may not request" in r.lower() or "authorization denied" in r.lower() for r in report.reasons))
|
||||||
|
|
||||||
|
|
||||||
|
class TestBullet2NewWorkStopsDuringDrain(unittest.TestCase):
|
||||||
|
"""Bullet 2: New work stops during drain."""
|
||||||
|
|
||||||
|
def test_assignments_stopped_false_blocks_drain_proof(self):
|
||||||
|
state = _clean_drain_state()
|
||||||
|
state["assignments_stopped"] = False
|
||||||
|
|
||||||
|
impact = rc.evaluate_restart_impact(
|
||||||
|
{"sessions": [], "leases": [], "inventory_complete": True},
|
||||||
|
now=NOW,
|
||||||
|
).as_dict()
|
||||||
|
|
||||||
|
proof = dp.build_drain_proof(
|
||||||
|
secret=SECRET,
|
||||||
|
impact_report=impact,
|
||||||
|
drain_state=state,
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertFalse(proof.clean)
|
||||||
|
check = next(c for c in proof.checks if c.name == dp.CHECK_ASSIGNMENTS_STOPPED)
|
||||||
|
self.assertFalse(check.passed)
|
||||||
|
|
||||||
|
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
|
||||||
|
self.assertEqual(gate.verdict, dp.GATE_DENY)
|
||||||
|
self.assertFalse(gate.allow)
|
||||||
|
self.assertTrue(any("drain proof invalid" in r.lower() or "assignments_stopped" in r.lower() for r in gate.reasons))
|
||||||
|
|
||||||
|
|
||||||
|
class TestBullet3ActiveSafeWorkCanFinish(unittest.TestCase):
|
||||||
|
"""Bullet 3: Active safe work can finish."""
|
||||||
|
|
||||||
|
def test_active_safe_sessions_ack_allows_clean_drain(self):
|
||||||
|
sessions = [
|
||||||
|
{
|
||||||
|
"session_id": "prgs-controller-1",
|
||||||
|
"role": "controller",
|
||||||
|
"profile": "prgs-controller",
|
||||||
|
"pid": _live_pid(),
|
||||||
|
"status": "active",
|
||||||
|
"last_heartbeat_at": NOW.isoformat(),
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"session_id": "prgs-reviewer-42",
|
||||||
|
"role": "reviewer",
|
||||||
|
"profile": "prgs-reviewer",
|
||||||
|
"pid": _live_pid(),
|
||||||
|
"status": "active",
|
||||||
|
"last_heartbeat_at": NOW.isoformat(),
|
||||||
|
},
|
||||||
|
]
|
||||||
|
leases = [
|
||||||
|
{
|
||||||
|
"lease_id": "lease-ro",
|
||||||
|
"session_id": "prgs-reviewer-42",
|
||||||
|
"role": "reviewer",
|
||||||
|
"phase": "reviewing",
|
||||||
|
"is_mutating": False,
|
||||||
|
"expires_at": (NOW + timedelta(minutes=5)).isoformat(),
|
||||||
|
"pid": _live_pid(),
|
||||||
|
}
|
||||||
|
]
|
||||||
|
|
||||||
|
impact = rc.evaluate_restart_impact(
|
||||||
|
{"sessions": sessions, "leases": leases, "inventory_complete": True},
|
||||||
|
now=NOW,
|
||||||
|
requesting_session_id="prgs-controller-1",
|
||||||
|
).as_dict()
|
||||||
|
|
||||||
|
state = _clean_drain_state()
|
||||||
|
state["acks"] = {"prgs-reviewer-42": "ack"}
|
||||||
|
|
||||||
|
proof = dp.build_drain_proof(
|
||||||
|
secret=SECRET,
|
||||||
|
impact_report=impact,
|
||||||
|
drain_state=state,
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertTrue(proof.clean)
|
||||||
|
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
|
||||||
|
self.assertTrue(gate.allow)
|
||||||
|
self.assertEqual(gate.verdict, dp.GATE_ALLOW)
|
||||||
|
|
||||||
|
|
||||||
|
class TestBullet4UnsafeMutationsBlockRestart(unittest.TestCase):
|
||||||
|
"""Bullet 4: Unsafe mutations block restart."""
|
||||||
|
|
||||||
|
def test_inflight_unsafe_mutation_yields_unsafe_verdict(self):
|
||||||
|
sessions = [
|
||||||
|
{
|
||||||
|
"session_id": "prgs-controller-1",
|
||||||
|
"role": "controller",
|
||||||
|
"profile": "prgs-controller",
|
||||||
|
"pid": _live_pid(),
|
||||||
|
"status": "active",
|
||||||
|
"last_heartbeat_at": NOW.isoformat(),
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"session_id": "prgs-author-99",
|
||||||
|
"role": "author",
|
||||||
|
"profile": "prgs-author",
|
||||||
|
"pid": _live_pid(),
|
||||||
|
"status": "active",
|
||||||
|
"last_heartbeat_at": NOW.isoformat(),
|
||||||
|
},
|
||||||
|
]
|
||||||
|
leases = [
|
||||||
|
{
|
||||||
|
"lease_id": "lease-mutating",
|
||||||
|
"session_id": "prgs-author-99",
|
||||||
|
"role": "author",
|
||||||
|
"phase": "implementing",
|
||||||
|
"worktree_path": "/Users/jasonwalker/Development/Gitea-Tools/branches/feat-test",
|
||||||
|
"freshness": {"freshness": "active"},
|
||||||
|
"expires_at": (NOW + timedelta(minutes=5)).isoformat(),
|
||||||
|
"pid": _live_pid(),
|
||||||
|
}
|
||||||
|
]
|
||||||
|
|
||||||
|
report = rc.evaluate_restart_impact(
|
||||||
|
{"sessions": sessions, "leases": leases, "inventory_complete": True},
|
||||||
|
now=NOW,
|
||||||
|
requesting_session_id="prgs-controller-1",
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertEqual(report.verdict, rc.VERDICT_UNSAFE)
|
||||||
|
self.assertFalse(report.allow_restart)
|
||||||
|
self.assertGreater(len(report.mutations), 0)
|
||||||
|
|
||||||
|
proof = dp.build_drain_proof(
|
||||||
|
secret=SECRET,
|
||||||
|
impact_report=report.as_dict(),
|
||||||
|
drain_state=_clean_drain_state(),
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertFalse(proof.clean)
|
||||||
|
check = next(c for c in proof.checks if c.name == dp.CHECK_NO_INFLIGHT_MUTATIONS)
|
||||||
|
self.assertFalse(check.passed)
|
||||||
|
|
||||||
|
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
|
||||||
|
self.assertEqual(gate.verdict, dp.GATE_DENY)
|
||||||
|
self.assertFalse(gate.allow)
|
||||||
|
|
||||||
|
|
||||||
|
class TestBullet5DurableSessionCheckpoints(unittest.TestCase):
|
||||||
|
"""Bullet 5: Session state is durably checkpointed."""
|
||||||
|
|
||||||
|
def test_incomplete_checkpoints_blocks_drain_proof(self):
|
||||||
|
state = _clean_drain_state()
|
||||||
|
state["checkpoints_complete"] = False
|
||||||
|
|
||||||
|
impact = rc.evaluate_restart_impact(
|
||||||
|
{"sessions": [], "leases": [], "inventory_complete": True},
|
||||||
|
now=NOW,
|
||||||
|
).as_dict()
|
||||||
|
|
||||||
|
proof = dp.build_drain_proof(
|
||||||
|
secret=SECRET,
|
||||||
|
impact_report=impact,
|
||||||
|
drain_state=state,
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertFalse(proof.clean)
|
||||||
|
check = next(c for c in proof.checks if c.name == dp.CHECK_CHECKPOINTS_COMPLETE)
|
||||||
|
self.assertFalse(check.passed)
|
||||||
|
|
||||||
|
def test_post_restart_reconcile_audits_checkpoint_dimension(self):
|
||||||
|
inv = _clean_inventory()
|
||||||
|
inv["checkpoints_available"] = True
|
||||||
|
inv["checkpoints"] = [
|
||||||
|
{
|
||||||
|
"session_id": "prgs-author-99",
|
||||||
|
"checkpoint_id": "chk-1",
|
||||||
|
"stale": True,
|
||||||
|
}
|
||||||
|
]
|
||||||
|
|
||||||
|
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_ENFORCE)
|
||||||
|
chk_item = next(i for i in proof.items if i.dimension == prr.DIM_CHECKPOINTS)
|
||||||
|
self.assertIn(chk_item.status, (prr.ITEM_UNRESOLVED, prr.ITEM_DEGRADED, prr.ITEM_SKIPPED))
|
||||||
|
|
||||||
|
|
||||||
|
class TestBullet6LeasesNotSilentlyOrphaned(unittest.TestCase):
|
||||||
|
"""Bullet 6: Leases/locks not silently orphaned."""
|
||||||
|
|
||||||
|
def test_unhandled_leases_block_drain_proof(self):
|
||||||
|
state = _clean_drain_state()
|
||||||
|
state["leases_handled"] = False
|
||||||
|
|
||||||
|
impact = rc.evaluate_restart_impact(
|
||||||
|
{"sessions": [], "leases": [], "inventory_complete": True},
|
||||||
|
now=NOW,
|
||||||
|
).as_dict()
|
||||||
|
|
||||||
|
proof = dp.build_drain_proof(
|
||||||
|
secret=SECRET,
|
||||||
|
impact_report=impact,
|
||||||
|
drain_state=state,
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertFalse(proof.clean)
|
||||||
|
check = next(c for c in proof.checks if c.name == dp.CHECK_LEASES_HANDLED)
|
||||||
|
self.assertFalse(check.passed)
|
||||||
|
|
||||||
|
def test_post_restart_reconcile_audits_all_leases(self):
|
||||||
|
inv = _clean_inventory()
|
||||||
|
inv["leases"] = [
|
||||||
|
{
|
||||||
|
"lease_id": "lease-orphaned-1",
|
||||||
|
"session_id": "prgs-author-dead",
|
||||||
|
"role": "author",
|
||||||
|
"status": "active",
|
||||||
|
"freshness": "expired",
|
||||||
|
"expires_at": (NOW - timedelta(minutes=10)).isoformat(),
|
||||||
|
}
|
||||||
|
]
|
||||||
|
|
||||||
|
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_LOG_ONLY)
|
||||||
|
lease_item = next(i for i in proof.items if i.dimension == prr.DIM_LEASES)
|
||||||
|
self.assertIsNotNone(lease_item)
|
||||||
|
self.assertTrue(lease_item.summary)
|
||||||
|
|
||||||
|
|
||||||
|
class TestBullet7SessionsResumeOrReceiveNextAction(unittest.TestCase):
|
||||||
|
"""Bullet 7: Sessions resume or receive canonical next action."""
|
||||||
|
|
||||||
|
def test_reconcile_provides_canonical_next_action_for_unresolved(self):
|
||||||
|
inv = _clean_inventory()
|
||||||
|
inv["pending_mutations"] = [
|
||||||
|
{
|
||||||
|
"mutation_id": "mut-404",
|
||||||
|
"session_id": "prgs-author-77",
|
||||||
|
"phase": "implementing",
|
||||||
|
"issue_number": 666,
|
||||||
|
}
|
||||||
|
]
|
||||||
|
|
||||||
|
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_ENFORCE)
|
||||||
|
self.assertEqual(proof.overall_status, prr.STATUS_DEGRADED)
|
||||||
|
self.assertTrue(proof.mutation_hold)
|
||||||
|
self.assertTrue(proof.note)
|
||||||
|
self.assertGreater(len(proof.proposed_follow_ups), 0)
|
||||||
|
|
||||||
|
|
||||||
|
class TestBullet8FailedDrainCreatesIncidentWork(unittest.TestCase):
|
||||||
|
"""Bullet 8: Failed drain creates durable incident work."""
|
||||||
|
|
||||||
|
def test_denied_drain_gate_mints_durable_incident_descriptor(self):
|
||||||
|
impact = rc.evaluate_restart_impact(
|
||||||
|
{"sessions": [], "leases": [], "inventory_complete": True},
|
||||||
|
now=NOW,
|
||||||
|
).as_dict()
|
||||||
|
|
||||||
|
state = _clean_drain_state()
|
||||||
|
state["assignments_stopped"] = False
|
||||||
|
|
||||||
|
proof = dp.build_drain_proof(
|
||||||
|
secret=SECRET,
|
||||||
|
impact_report=impact,
|
||||||
|
drain_state=state,
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
|
||||||
|
gate = dp.gate_apply_restart(proof=proof.as_dict(), secret=SECRET, now=NOW)
|
||||||
|
self.assertEqual(gate.verdict, dp.GATE_DENY)
|
||||||
|
|
||||||
|
incident = gate.incident
|
||||||
|
self.assertIsNotNone(incident)
|
||||||
|
self.assertEqual(incident["kind"], "restart_drain_gate_denied")
|
||||||
|
self.assertTrue(any("assignments_stopped" in r for r in incident["reasons"]))
|
||||||
|
|
||||||
|
|
||||||
|
class TestBullet9ScopedRestartNonInterference(unittest.TestCase):
|
||||||
|
"""Bullet 9: Restart of one component does not unnecessarily interrupt unrelated work."""
|
||||||
|
|
||||||
|
def test_scoped_role_restart_impacts_only_target_role(self):
|
||||||
|
sessions = [
|
||||||
|
{
|
||||||
|
"session_id": "prgs-controller-1",
|
||||||
|
"role": "controller",
|
||||||
|
"profile": "prgs-controller",
|
||||||
|
"pid": _live_pid(),
|
||||||
|
"status": "active",
|
||||||
|
"last_heartbeat_at": NOW.isoformat(),
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"session_id": "prgs-author-10",
|
||||||
|
"role": "author",
|
||||||
|
"profile": "prgs-author",
|
||||||
|
"pid": _live_pid(),
|
||||||
|
"status": "active",
|
||||||
|
"last_heartbeat_at": NOW.isoformat(),
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"session_id": "prgs-reviewer-20",
|
||||||
|
"role": "reviewer",
|
||||||
|
"profile": "prgs-reviewer",
|
||||||
|
"pid": _live_pid(),
|
||||||
|
"status": "active",
|
||||||
|
"last_heartbeat_at": NOW.isoformat(),
|
||||||
|
},
|
||||||
|
]
|
||||||
|
|
||||||
|
policy = rc.RESTART_CLASS_POLICIES[RestartClass.ROLE_RUNTIME_RESTART]
|
||||||
|
report = rc.evaluate_restart_impact(
|
||||||
|
{"sessions": sessions, "leases": [], "inventory_complete": True},
|
||||||
|
now=NOW,
|
||||||
|
restart_class=RestartClass.ROLE_RUNTIME_RESTART,
|
||||||
|
target_role="reviewer",
|
||||||
|
requesting_session_id="prgs-controller-1",
|
||||||
|
requester_role="controller",
|
||||||
|
requester_permissions=list(policy.request_roles),
|
||||||
|
controller_approved=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertTrue(report.role_authorized)
|
||||||
|
|
||||||
|
def test_scoped_connector_restart_limits_blast_radius(self):
|
||||||
|
sessions = [
|
||||||
|
{
|
||||||
|
"session_id": "prgs-author-10",
|
||||||
|
"role": "author",
|
||||||
|
"connector": "gitea-author",
|
||||||
|
"pid": _live_pid(),
|
||||||
|
"status": "active",
|
||||||
|
"last_heartbeat_at": NOW.isoformat(),
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"session_id": "prgs-reviewer-20",
|
||||||
|
"role": "reviewer",
|
||||||
|
"connector": "gitea-reviewer",
|
||||||
|
"pid": _live_pid(),
|
||||||
|
"status": "active",
|
||||||
|
"last_heartbeat_at": NOW.isoformat(),
|
||||||
|
},
|
||||||
|
]
|
||||||
|
|
||||||
|
policy = rc.RESTART_CLASS_POLICIES[RestartClass.CONNECTOR_RESTART]
|
||||||
|
report = rc.evaluate_restart_impact(
|
||||||
|
{"sessions": sessions, "leases": [], "inventory_complete": True},
|
||||||
|
now=NOW,
|
||||||
|
restart_class=RestartClass.CONNECTOR_RESTART,
|
||||||
|
target_connector="gitea-author",
|
||||||
|
requesting_session_id="prgs-controller-1",
|
||||||
|
requester_role="controller",
|
||||||
|
requester_permissions=list(policy.request_roles),
|
||||||
|
controller_approved=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
self.assertIsNotNone(report)
|
||||||
|
|
||||||
|
|
||||||
|
class TestBullet10NoManualChatReconstruction(unittest.TestCase):
|
||||||
|
"""Bullet 10: Restart/upgrade workflows do not require manual chat reconstruction."""
|
||||||
|
|
||||||
|
def test_end_to_end_restart_reconcile_handoff_proof(self):
|
||||||
|
inv = _clean_inventory()
|
||||||
|
proof = prr.reconcile_after_restart(inv, now=NOW, mode=prr.MODE_LOG_ONLY)
|
||||||
|
|
||||||
|
proof_dict = proof.as_dict()
|
||||||
|
self.assertEqual(proof_dict["overall_status"], prr.STATUS_COMPLETE)
|
||||||
|
self.assertFalse(proof_dict["mutation_hold"])
|
||||||
|
self.assertTrue(proof_dict["note"])
|
||||||
|
self.assertIn("links", proof_dict)
|
||||||
|
self.assertEqual(proof_dict["links"]["umbrella"], 655)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -35,10 +35,10 @@ class TestMcpStaleRuntime(unittest.TestCase):
|
|||||||
|
|
||||||
# Mock env output for ps eww
|
# Mock env output for ps eww
|
||||||
mock_run_env12345 = MagicMock()
|
mock_run_env12345 = MagicMock()
|
||||||
mock_run_env12345.stdout = "GITEA_MCP_PROFILE=prgs-reconciler"
|
mock_run_env12345.stdout = "GITEA_MCP_PROFILE=prgs-reconciler GITEA_CLIENT_MANAGED=1"
|
||||||
|
|
||||||
mock_run_env54321 = MagicMock()
|
mock_run_env54321 = MagicMock()
|
||||||
mock_run_env54321.stdout = "GITEA_MCP_PROFILE=prgs-author"
|
mock_run_env54321.stdout = "GITEA_MCP_PROFILE=prgs-author GITEA_CLIENT_MANAGED=1"
|
||||||
|
|
||||||
def side_effect(args, **kwargs):
|
def side_effect(args, **kwargs):
|
||||||
if args[0] == "ps" and "eww" in args:
|
if args[0] == "ps" and "eww" in args:
|
||||||
@@ -91,7 +91,7 @@ class TestMcpStaleRuntime(unittest.TestCase):
|
|||||||
mock_run_ps.stdout = ps_output
|
mock_run_ps.stdout = ps_output
|
||||||
|
|
||||||
mock_run_env = MagicMock()
|
mock_run_env = MagicMock()
|
||||||
mock_run_env.stdout = "GITEA_MCP_PROFILE=prgs-author"
|
mock_run_env.stdout = "GITEA_MCP_PROFILE=prgs-author GITEA_CLIENT_MANAGED=1"
|
||||||
|
|
||||||
mock_run_git = MagicMock()
|
mock_run_git = MagicMock()
|
||||||
mock_run_git.stdout = "FAKE2" # different SHA
|
mock_run_git.stdout = "FAKE2" # different SHA
|
||||||
|
|||||||
@@ -0,0 +1,217 @@
|
|||||||
|
"""Unit tests for the scoped recovery playbook (#669)."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import recovery_playbook as rp
|
||||||
|
import restart_coordinator as rc
|
||||||
|
|
||||||
|
|
||||||
|
def test_ladder_covers_eleven_ordered_rungs():
|
||||||
|
ranks = [r.rank for r in rp.RECOVERY_LADDER]
|
||||||
|
assert ranks == list(range(len(rp.RECOVERY_LADDER)))
|
||||||
|
assert len(rp.RECOVERY_LADDER) == 11
|
||||||
|
assert rp.RECOVERY_LADDER[0].action is rp.RecoveryAction.CLIENT_RECONNECT
|
||||||
|
assert rp.RECOVERY_LADDER[-1].action is rp.RecoveryAction.HOST_RESTART
|
||||||
|
|
||||||
|
|
||||||
|
def test_ladder_document_links_parent_issues():
|
||||||
|
doc = rp.ladder_document()
|
||||||
|
assert "#655" in doc["parent_issues"]
|
||||||
|
assert "#652" in doc["parent_issues"]
|
||||||
|
assert "#653" in doc["parent_issues"]
|
||||||
|
assert doc["enforcement_issue"] == "#669"
|
||||||
|
assert "full_mcp_restart" in doc["broad_restart_actions"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_recommend_transport_eof_starts_at_client_reconnect():
|
||||||
|
plan = rp.recommend_actions(symptoms=["transport_eof"])
|
||||||
|
assert plan["recommended_actions"][0]["action"] == "client_reconnect"
|
||||||
|
assert plan["recommended_actions"][0]["issue_links"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_recommend_skips_successful_prior_attempts():
|
||||||
|
attempts = [
|
||||||
|
rp.build_attempt_record(
|
||||||
|
"client_reconnect", outcome="success", reason="reconnected"
|
||||||
|
)
|
||||||
|
]
|
||||||
|
plan = rp.recommend_actions(
|
||||||
|
symptoms=["transport_eof"], prior_recovery_attempts=attempts
|
||||||
|
)
|
||||||
|
actions = [a["action"] for a in plan["recommended_actions"]]
|
||||||
|
assert "client_reconnect" not in actions
|
||||||
|
assert actions[0] == "capability_refresh"
|
||||||
|
|
||||||
|
|
||||||
|
def test_escalation_denied_without_attempt_log():
|
||||||
|
result = rp.assess_escalation("full_mcp_restart", prior_recovery_attempts=[])
|
||||||
|
assert result.allowed is False
|
||||||
|
assert result.require_attempt_log is True
|
||||||
|
assert any("#669" in r for r in result.reasons)
|
||||||
|
assert result.recommended_next # soft recommendations still provided
|
||||||
|
|
||||||
|
|
||||||
|
def test_escalation_allowed_after_insufficient_narrower():
|
||||||
|
attempts = [
|
||||||
|
rp.build_attempt_record(
|
||||||
|
"client_reconnect",
|
||||||
|
outcome="insufficient",
|
||||||
|
reason="still flapping",
|
||||||
|
),
|
||||||
|
rp.build_attempt_record(
|
||||||
|
"session_reconnect",
|
||||||
|
outcome="failed",
|
||||||
|
reason="namespace still dead",
|
||||||
|
),
|
||||||
|
]
|
||||||
|
result = rp.assess_escalation(
|
||||||
|
"full_mcp_restart", prior_recovery_attempts=attempts
|
||||||
|
)
|
||||||
|
assert result.allowed is True
|
||||||
|
assert len(result.qualifying_attempts) == 2
|
||||||
|
|
||||||
|
|
||||||
|
def test_escalation_break_glass_bypasses_attempt_log():
|
||||||
|
result = rp.assess_escalation(
|
||||||
|
"host_restart", prior_recovery_attempts=[], break_glass=True
|
||||||
|
)
|
||||||
|
assert result.allowed is True
|
||||||
|
assert result.break_glass is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_narrow_action_does_not_require_attempt_log():
|
||||||
|
result = rp.assess_escalation(
|
||||||
|
"client_reconnect", prior_recovery_attempts=[]
|
||||||
|
)
|
||||||
|
assert result.allowed is True
|
||||||
|
assert result.require_attempt_log is False
|
||||||
|
|
||||||
|
|
||||||
|
def test_same_rank_attempt_does_not_qualify_for_escalation():
|
||||||
|
attempts = [
|
||||||
|
rp.build_attempt_record(
|
||||||
|
"full_mcp_restart", outcome="failed", reason="already failed full"
|
||||||
|
)
|
||||||
|
]
|
||||||
|
result = rp.assess_escalation(
|
||||||
|
"full_mcp_restart", prior_recovery_attempts=attempts
|
||||||
|
)
|
||||||
|
assert result.allowed is False
|
||||||
|
|
||||||
|
|
||||||
|
def test_success_outcome_does_not_qualify_for_escalation():
|
||||||
|
attempts = [
|
||||||
|
rp.build_attempt_record(
|
||||||
|
"client_reconnect", outcome="success", reason="fixed"
|
||||||
|
)
|
||||||
|
]
|
||||||
|
result = rp.assess_escalation(
|
||||||
|
"full_mcp_restart", prior_recovery_attempts=attempts
|
||||||
|
)
|
||||||
|
assert result.allowed is False
|
||||||
|
|
||||||
|
|
||||||
|
def test_recovery_metrics_fraction_avoided():
|
||||||
|
attempts = [
|
||||||
|
rp.build_attempt_record("client_reconnect", outcome="success"),
|
||||||
|
rp.build_attempt_record("session_reconnect", outcome="success"),
|
||||||
|
rp.build_attempt_record("full_mcp_restart", outcome="success"),
|
||||||
|
]
|
||||||
|
metrics = rp.recovery_metrics(attempts)
|
||||||
|
assert metrics["successes_total"] == 3
|
||||||
|
assert metrics["successes_avoided_full_restart"] == 2
|
||||||
|
assert metrics["successes_full_or_host_restart"] == 1
|
||||||
|
assert abs(metrics["fraction_avoided_full_restart"] - (2 / 3)) < 1e-9
|
||||||
|
|
||||||
|
|
||||||
|
def test_coordinator_denies_full_restart_without_attempt_log():
|
||||||
|
inv = {
|
||||||
|
"inventory_complete": True,
|
||||||
|
"sessions": [],
|
||||||
|
"leases": [],
|
||||||
|
"prior_recovery_attempts": [],
|
||||||
|
}
|
||||||
|
report = rc.evaluate_restart_impact(
|
||||||
|
inv,
|
||||||
|
restart_class=rc.RestartClass.FULL_MCP_RESTART,
|
||||||
|
requester_role="controller",
|
||||||
|
requester_permissions=rc.permissions_for_role("controller"),
|
||||||
|
controller_approved=True,
|
||||||
|
operator_authorized=True,
|
||||||
|
)
|
||||||
|
assert report.allow_restart is False
|
||||||
|
assert report.attempt_log_satisfied is False
|
||||||
|
assert report.verdict == rc.VERDICT_UNSAFE
|
||||||
|
blob = " ".join(report.reasons + report.authorization_reasons)
|
||||||
|
assert "#669" in blob or "attempt log" in blob
|
||||||
|
|
||||||
|
|
||||||
|
def test_coordinator_allows_full_restart_with_attempt_log():
|
||||||
|
inv = {
|
||||||
|
"inventory_complete": True,
|
||||||
|
"sessions": [],
|
||||||
|
"leases": [],
|
||||||
|
"prior_recovery_attempts": [
|
||||||
|
{
|
||||||
|
"action": "client_reconnect",
|
||||||
|
"outcome": "insufficient",
|
||||||
|
"reason": "still broken",
|
||||||
|
}
|
||||||
|
],
|
||||||
|
}
|
||||||
|
report = rc.evaluate_restart_impact(
|
||||||
|
inv,
|
||||||
|
restart_class=rc.RestartClass.FULL_MCP_RESTART,
|
||||||
|
requester_role="controller",
|
||||||
|
requester_permissions=rc.permissions_for_role("controller"),
|
||||||
|
controller_approved=True,
|
||||||
|
operator_authorized=True,
|
||||||
|
)
|
||||||
|
assert report.attempt_log_satisfied is True
|
||||||
|
assert report.allow_restart is True
|
||||||
|
assert report.verdict == rc.VERDICT_SAFE
|
||||||
|
|
||||||
|
|
||||||
|
def test_coordinator_break_glass_allows_without_log():
|
||||||
|
inv = {
|
||||||
|
"inventory_complete": True,
|
||||||
|
"sessions": [],
|
||||||
|
"leases": [],
|
||||||
|
"prior_recovery_attempts": [],
|
||||||
|
}
|
||||||
|
report = rc.evaluate_restart_impact(
|
||||||
|
inv,
|
||||||
|
restart_class=rc.RestartClass.FULL_MCP_RESTART,
|
||||||
|
requester_role="controller",
|
||||||
|
requester_permissions=rc.permissions_for_role("controller"),
|
||||||
|
controller_approved=True,
|
||||||
|
operator_authorized=True,
|
||||||
|
break_glass=True,
|
||||||
|
)
|
||||||
|
assert report.break_glass is True
|
||||||
|
assert report.attempt_log_satisfied is True
|
||||||
|
assert report.allow_restart is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_coordinator_client_reconnect_unaffected():
|
||||||
|
inv = {
|
||||||
|
"inventory_complete": True,
|
||||||
|
"sessions": [],
|
||||||
|
"leases": [],
|
||||||
|
"prior_recovery_attempts": [],
|
||||||
|
}
|
||||||
|
report = rc.evaluate_restart_impact(
|
||||||
|
inv,
|
||||||
|
restart_class=rc.RestartClass.CLIENT_RECONNECT,
|
||||||
|
requester_role="author",
|
||||||
|
requester_permissions=rc.permissions_for_role("author"),
|
||||||
|
)
|
||||||
|
assert report.attempt_log_satisfied is True
|
||||||
|
assert report.allow_restart is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_restart_class_alias_accepted():
|
||||||
|
assert (
|
||||||
|
rp.resolve_action("full_mcp_restart")
|
||||||
|
is rp.RecoveryAction.FULL_MCP_RESTART
|
||||||
|
)
|
||||||
@@ -243,9 +243,10 @@ class TestRuntimeClarity(unittest.TestCase):
|
|||||||
self.assertIn("switching is disabled", res["message"].lower())
|
self.assertIn("switching is disabled", res["message"].lower())
|
||||||
self.assertIsNone(gitea_config._active_profile_override)
|
self.assertIsNone(gitea_config._active_profile_override)
|
||||||
|
|
||||||
|
@patch("mcp_server._trusted_session_repository", return_value={"repository": "Example-Org/Example-Repo", "org": "Example-Org", "repo": "Example-Repo", "reasons": []})
|
||||||
@patch("mcp_server.api_request")
|
@patch("mcp_server.api_request")
|
||||||
@patch("mcp_server.get_auth_header")
|
@patch("mcp_server.get_auth_header")
|
||||||
def test_activate_profile_succeeds_when_enabled(self, mock_auth, mock_api):
|
def test_activate_profile_succeeds_when_enabled(self, mock_auth, mock_api, mock_trusted):
|
||||||
self._write_config(CONFIG_SWITCHING_ENABLED)
|
self._write_config(CONFIG_SWITCHING_ENABLED)
|
||||||
|
|
||||||
# Setup mock responses for whoami checks
|
# Setup mock responses for whoami checks
|
||||||
|
|||||||
@@ -0,0 +1,465 @@
|
|||||||
|
"""Unit tests for Phase 3 Notifications and Human-Attention Console (#648)."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
from starlette.testclient import TestClient
|
||||||
|
|
||||||
|
from webui.app import create_app
|
||||||
|
from webui.notifications import (
|
||||||
|
ATTENTION_HUMAN_REQUIRED,
|
||||||
|
ATTENTION_OPERATOR,
|
||||||
|
ATTENTION_ROUTINE,
|
||||||
|
CATEGORY_AUTH,
|
||||||
|
CATEGORY_BLOCKER,
|
||||||
|
CATEGORY_LEASE,
|
||||||
|
CATEGORY_SYSTEM,
|
||||||
|
CATEGORY_VALIDATION,
|
||||||
|
CATEGORY_WORKFLOW,
|
||||||
|
NotificationItem,
|
||||||
|
NotificationSnapshot,
|
||||||
|
classify_attention_event,
|
||||||
|
load_notifications_snapshot,
|
||||||
|
snapshot_to_dict,
|
||||||
|
)
|
||||||
|
from webui.notification_views import render_notifications_page
|
||||||
|
from webui.project_registry import load_registry
|
||||||
|
from webui.queue_loader import QueueItem, QueueSnapshot
|
||||||
|
from webui.lease_loader import CollisionWarning, LeaseSnapshot
|
||||||
|
from webui.system_health import DependencyProbe, SystemHealthSnapshot, VersionInfo, StaleRuntime
|
||||||
|
|
||||||
|
|
||||||
|
def test_classify_attention_event_rules():
|
||||||
|
# 1. Critical escalation boundaries -> human-required
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_AUTH, "Auth error", "Unauthorized access attempt", is_auth_failure=True
|
||||||
|
)
|
||||||
|
assert att_cls == ATTENTION_HUMAN_REQUIRED
|
||||||
|
assert req_human is True
|
||||||
|
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_SYSTEM, "Hard stop", "Hard stop triggered", is_hard_stop=True
|
||||||
|
)
|
||||||
|
assert att_cls == ATTENTION_HUMAN_REQUIRED
|
||||||
|
assert req_human is True
|
||||||
|
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_VALIDATION, "Validation Error", "Report validation failed", is_validation_failure=True
|
||||||
|
)
|
||||||
|
assert att_cls == ATTENTION_HUMAN_REQUIRED
|
||||||
|
assert req_human is True
|
||||||
|
|
||||||
|
# 2. Operational issues -> operator
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_BLOCKER, "PR Blocked", "Merge conflict detected", is_blocker=True
|
||||||
|
)
|
||||||
|
assert att_cls == ATTENTION_OPERATOR
|
||||||
|
assert req_human is False
|
||||||
|
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_LEASE, "Lease Expired", "Session lease expired", is_stale=True
|
||||||
|
)
|
||||||
|
assert att_cls == ATTENTION_OPERATOR
|
||||||
|
assert req_human is False
|
||||||
|
|
||||||
|
# 3. Routine workflow transitions -> routine
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_WORKFLOW, "PR Active", "PR in review"
|
||||||
|
)
|
||||||
|
assert att_cls == ATTENTION_ROUTINE
|
||||||
|
assert req_human is False
|
||||||
|
|
||||||
|
|
||||||
|
def test_notification_snapshot_aggregation():
|
||||||
|
reg = load_registry()
|
||||||
|
proj_id = reg.projects[0].id if reg.projects else "gitea-tools"
|
||||||
|
|
||||||
|
mock_queue = QueueSnapshot(
|
||||||
|
project_id=proj_id,
|
||||||
|
repo_label="org/repo",
|
||||||
|
prs=(
|
||||||
|
QueueItem(
|
||||||
|
number=101,
|
||||||
|
title="Blocked PR",
|
||||||
|
badges=("blocked",),
|
||||||
|
extra={},
|
||||||
|
),
|
||||||
|
QueueItem(
|
||||||
|
number=102,
|
||||||
|
title="Normal PR",
|
||||||
|
badges=("in-review",),
|
||||||
|
extra={},
|
||||||
|
),
|
||||||
|
),
|
||||||
|
issues=(),
|
||||||
|
pr_pagination=None,
|
||||||
|
issue_pagination=None,
|
||||||
|
)
|
||||||
|
|
||||||
|
mock_leases = LeaseSnapshot(
|
||||||
|
project_id=proj_id,
|
||||||
|
repo_label="org/repo",
|
||||||
|
issue_lock=None,
|
||||||
|
claim_inventory={},
|
||||||
|
reviewer_leases=(
|
||||||
|
{
|
||||||
|
"pr_number": 101,
|
||||||
|
"status": "expired",
|
||||||
|
"is_expired": True,
|
||||||
|
},
|
||||||
|
),
|
||||||
|
duplicate_prs=(
|
||||||
|
CollisionWarning(
|
||||||
|
kind="duplicate_pr",
|
||||||
|
message="Multiple open PRs for issue #101",
|
||||||
|
issue_number=101,
|
||||||
|
pr_numbers=(101, 103),
|
||||||
|
),
|
||||||
|
),
|
||||||
|
duplicate_branches=(),
|
||||||
|
collision_history=(),
|
||||||
|
fetch_error=None,
|
||||||
|
)
|
||||||
|
|
||||||
|
mock_version = VersionInfo(
|
||||||
|
git_sha="abc1234",
|
||||||
|
git_describe="v1.0.0",
|
||||||
|
control_plane_schema_version=1,
|
||||||
|
python_version="3.11",
|
||||||
|
known=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
mock_stale = StaleRuntime(
|
||||||
|
daemon_head="abc1234",
|
||||||
|
checkout_head="abc1234",
|
||||||
|
remote_head="abc1234",
|
||||||
|
stale=False,
|
||||||
|
determinable=True,
|
||||||
|
mutation_safe=True,
|
||||||
|
reasons=(),
|
||||||
|
)
|
||||||
|
|
||||||
|
mock_health = SystemHealthSnapshot(
|
||||||
|
status="degraded",
|
||||||
|
ready=False,
|
||||||
|
readiness_complete=True,
|
||||||
|
readiness_reasons=("Auth failure",),
|
||||||
|
service="webui",
|
||||||
|
mode="test",
|
||||||
|
version=mock_version,
|
||||||
|
started_at="2026-07-25T00:00:00Z",
|
||||||
|
uptime_seconds=100.0,
|
||||||
|
timestamp="2026-07-25T00:00:00Z",
|
||||||
|
deep_probes_requested=True,
|
||||||
|
dependencies=(
|
||||||
|
DependencyProbe(
|
||||||
|
name="auth_service",
|
||||||
|
kind="auth",
|
||||||
|
status="unauthorized",
|
||||||
|
detail="Token expired",
|
||||||
|
required=True,
|
||||||
|
),
|
||||||
|
),
|
||||||
|
mcp_namespaces=(),
|
||||||
|
stale_runtime=mock_stale,
|
||||||
|
probe_errors=(),
|
||||||
|
)
|
||||||
|
|
||||||
|
snapshot = load_notifications_snapshot(
|
||||||
|
proj_id,
|
||||||
|
load_queue=lambda _id: mock_queue,
|
||||||
|
load_leases=lambda **_kwargs: mock_leases,
|
||||||
|
load_health=lambda **_kwargs: mock_health,
|
||||||
|
)
|
||||||
|
|
||||||
|
assert snapshot.project_id == proj_id
|
||||||
|
assert snapshot.total_count == 5
|
||||||
|
assert snapshot.human_required_count >= 1 # auth probe failure
|
||||||
|
assert snapshot.operator_count >= 3 # blocked PR + expired lease + duplicate PR collision
|
||||||
|
assert snapshot.routine_count >= 1 # normal PR
|
||||||
|
|
||||||
|
# Inbox items should include operator and human-required items only
|
||||||
|
inbox_classes = {item.attention_class for item in snapshot.inbox_items}
|
||||||
|
assert ATTENTION_ROUTINE not in inbox_classes
|
||||||
|
assert ATTENTION_OPERATOR in inbox_classes
|
||||||
|
assert ATTENTION_HUMAN_REQUIRED in inbox_classes
|
||||||
|
|
||||||
|
|
||||||
|
def test_snapshot_to_dict_and_redaction():
|
||||||
|
item = NotificationItem(
|
||||||
|
id="notif-1",
|
||||||
|
attention_class=ATTENTION_HUMAN_REQUIRED,
|
||||||
|
category=CATEGORY_AUTH,
|
||||||
|
title="Auth Error",
|
||||||
|
summary="Failed auth header: Bearer secret_token_12345",
|
||||||
|
work_kind="system",
|
||||||
|
work_number=None,
|
||||||
|
project_id="test-proj",
|
||||||
|
repo_label="org/repo",
|
||||||
|
created_at="2026-07-25T16:00:00Z",
|
||||||
|
requires_human=True,
|
||||||
|
)
|
||||||
|
snap = NotificationSnapshot(
|
||||||
|
project_id="test-proj",
|
||||||
|
repo_label="org/repo",
|
||||||
|
items=(item,),
|
||||||
|
human_required_count=1,
|
||||||
|
operator_count=0,
|
||||||
|
routine_count=0,
|
||||||
|
total_count=1,
|
||||||
|
)
|
||||||
|
|
||||||
|
data = snapshot_to_dict(snap)
|
||||||
|
assert data["project_id"] == "test-proj"
|
||||||
|
assert data["human_required_count"] == 1
|
||||||
|
assert len(data["inbox_items"]) == 1
|
||||||
|
|
||||||
|
# Redaction test
|
||||||
|
summary = data["inbox_items"][0]["summary"]
|
||||||
|
assert "secret_token_12345" not in summary
|
||||||
|
assert "<redacted>" in summary or "Bearer" in summary
|
||||||
|
|
||||||
|
|
||||||
|
def test_notifications_html_views():
|
||||||
|
item = NotificationItem(
|
||||||
|
id="notif-1",
|
||||||
|
attention_class=ATTENTION_HUMAN_REQUIRED,
|
||||||
|
category=CATEGORY_AUTH,
|
||||||
|
title="Critical Auth Failure",
|
||||||
|
summary="Auth failure details",
|
||||||
|
work_kind="issue",
|
||||||
|
work_number=42,
|
||||||
|
project_id="test-proj",
|
||||||
|
repo_label="org/repo",
|
||||||
|
created_at="2026-07-25T16:00:00Z",
|
||||||
|
requires_human=True,
|
||||||
|
)
|
||||||
|
snap = NotificationSnapshot(
|
||||||
|
project_id="test-proj",
|
||||||
|
repo_label="org/repo",
|
||||||
|
items=(item,),
|
||||||
|
human_required_count=1,
|
||||||
|
operator_count=0,
|
||||||
|
routine_count=0,
|
||||||
|
total_count=1,
|
||||||
|
)
|
||||||
|
|
||||||
|
html = render_notifications_page(snap, filter_class="inbox")
|
||||||
|
assert "Notifications & Attention Inbox" in html or "Notifications & Attention Inbox" in html
|
||||||
|
assert "Critical Auth Failure" in html
|
||||||
|
assert "HUMAN REQUIRED" in html
|
||||||
|
assert "Human Required" in html
|
||||||
|
|
||||||
|
|
||||||
|
def test_notifications_app_routes():
|
||||||
|
app = create_app()
|
||||||
|
client = TestClient(app)
|
||||||
|
|
||||||
|
# 1. HTML Route
|
||||||
|
res = client.get("/notifications")
|
||||||
|
assert res.status_code == 200
|
||||||
|
assert "Notifications" in res.text
|
||||||
|
assert "Attention Inbox" in res.text
|
||||||
|
|
||||||
|
# 2. API Route /api/v1/notifications
|
||||||
|
res_api = client.get("/api/v1/notifications")
|
||||||
|
assert res_api.status_code == 200
|
||||||
|
json_data = res_api.json()
|
||||||
|
assert "human_required_count" in json_data
|
||||||
|
assert "operator_count" in json_data
|
||||||
|
assert "routine_count" in json_data
|
||||||
|
assert "inbox_items" in json_data
|
||||||
|
|
||||||
|
# 3. Compatibility Alias /api/notifications
|
||||||
|
res_alias = client.get("/api/notifications")
|
||||||
|
assert res_alias.status_code == 200
|
||||||
|
assert res_alias.json()["project_id"] == json_data["project_id"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_classify_ignores_human_authored_title_and_summary_keywords():
|
||||||
|
"""B1: keywords in human-authored titles must not escalate routine work (#905)."""
|
||||||
|
# Routine transition whose title/summary mention critical-boundary words
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_WORKFLOW,
|
||||||
|
"record irrecoverable decision lock provenance",
|
||||||
|
"PR #999 'record irrecoverable decision lock provenance' is in routine state in-review.",
|
||||||
|
)
|
||||||
|
assert att_cls == ATTENTION_ROUTINE
|
||||||
|
assert req_human is False
|
||||||
|
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_WORKFLOW,
|
||||||
|
"fix unauthorized token path",
|
||||||
|
"Issue #1 'fix unauthorized token path' state: claimed. hard stop docs only.",
|
||||||
|
)
|
||||||
|
assert att_cls == ATTENTION_ROUTINE
|
||||||
|
assert req_human is False
|
||||||
|
|
||||||
|
# Structured flags still escalate (machine-driven)
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_SYSTEM,
|
||||||
|
"anything",
|
||||||
|
"anything with hard stop in text",
|
||||||
|
is_hard_stop=True,
|
||||||
|
)
|
||||||
|
assert att_cls == ATTENTION_HUMAN_REQUIRED
|
||||||
|
assert req_human is True
|
||||||
|
|
||||||
|
|
||||||
|
def test_notification_ids_are_unique_across_probe_errors_and_collisions():
|
||||||
|
"""B2: published notification ids must be unique within a snapshot (#905)."""
|
||||||
|
reg = load_registry()
|
||||||
|
proj_id = reg.projects[0].id if reg.projects else "gitea-tools"
|
||||||
|
|
||||||
|
mock_queue = QueueSnapshot(
|
||||||
|
project_id=proj_id,
|
||||||
|
repo_label="org/repo",
|
||||||
|
prs=(),
|
||||||
|
issues=(),
|
||||||
|
pr_pagination=None,
|
||||||
|
issue_pagination=None,
|
||||||
|
)
|
||||||
|
mock_leases = LeaseSnapshot(
|
||||||
|
project_id=proj_id,
|
||||||
|
repo_label="org/repo",
|
||||||
|
issue_lock=None,
|
||||||
|
claim_inventory={},
|
||||||
|
reviewer_leases=(),
|
||||||
|
duplicate_prs=(
|
||||||
|
CollisionWarning(
|
||||||
|
kind="duplicate_pr",
|
||||||
|
message="Multiple open PRs for issue #10",
|
||||||
|
issue_number=10,
|
||||||
|
pr_numbers=(10, 11),
|
||||||
|
),
|
||||||
|
CollisionWarning(
|
||||||
|
kind="duplicate_branch",
|
||||||
|
message="Another collision without issue",
|
||||||
|
issue_number=None,
|
||||||
|
pr_numbers=(12, 13),
|
||||||
|
),
|
||||||
|
CollisionWarning(
|
||||||
|
kind="duplicate_pr",
|
||||||
|
message="Second issue collision",
|
||||||
|
issue_number=10,
|
||||||
|
pr_numbers=(14, 15),
|
||||||
|
),
|
||||||
|
),
|
||||||
|
duplicate_branches=(),
|
||||||
|
collision_history=(),
|
||||||
|
fetch_error=None,
|
||||||
|
)
|
||||||
|
mock_version = VersionInfo(
|
||||||
|
git_sha="abc1234",
|
||||||
|
git_describe="v1.0.0",
|
||||||
|
control_plane_schema_version=1,
|
||||||
|
python_version="3.11",
|
||||||
|
known=True,
|
||||||
|
)
|
||||||
|
mock_stale = StaleRuntime(
|
||||||
|
daemon_head="abc1234",
|
||||||
|
checkout_head="abc1234",
|
||||||
|
remote_head="abc1234",
|
||||||
|
stale=False,
|
||||||
|
determinable=True,
|
||||||
|
mutation_safe=True,
|
||||||
|
reasons=(),
|
||||||
|
)
|
||||||
|
mock_health = SystemHealthSnapshot(
|
||||||
|
status="degraded",
|
||||||
|
ready=False,
|
||||||
|
readiness_complete=True,
|
||||||
|
readiness_reasons=(),
|
||||||
|
service="webui",
|
||||||
|
mode="test",
|
||||||
|
version=mock_version,
|
||||||
|
started_at="2026-07-25T00:00:00Z",
|
||||||
|
uptime_seconds=100.0,
|
||||||
|
timestamp="2026-07-25T00:00:00Z",
|
||||||
|
deep_probes_requested=True,
|
||||||
|
dependencies=(),
|
||||||
|
mcp_namespaces=(),
|
||||||
|
stale_runtime=mock_stale,
|
||||||
|
probe_errors=("error alpha", "error beta"),
|
||||||
|
)
|
||||||
|
|
||||||
|
snapshot = load_notifications_snapshot(
|
||||||
|
proj_id,
|
||||||
|
load_queue=lambda _id: mock_queue,
|
||||||
|
load_leases=lambda **_kwargs: mock_leases,
|
||||||
|
load_health=lambda **_kwargs: mock_health,
|
||||||
|
)
|
||||||
|
ids = [item.id for item in snapshot.items]
|
||||||
|
assert len(ids) == len(set(ids)), f"duplicate notification ids: {ids}"
|
||||||
|
assert any(i.startswith(f"notif-sys-err-{proj_id}-") for i in ids)
|
||||||
|
assert any(i.startswith("notif-collision-") for i in ids)
|
||||||
|
|
||||||
|
|
||||||
|
def test_probe_errors_do_not_set_fetch_error():
|
||||||
|
"""B3: probe_errors must not be reported as fetch_error (#905)."""
|
||||||
|
reg = load_registry()
|
||||||
|
proj_id = reg.projects[0].id if reg.projects else "gitea-tools"
|
||||||
|
|
||||||
|
mock_queue = QueueSnapshot(
|
||||||
|
project_id=proj_id,
|
||||||
|
repo_label="org/repo",
|
||||||
|
prs=(),
|
||||||
|
issues=(),
|
||||||
|
pr_pagination=None,
|
||||||
|
issue_pagination=None,
|
||||||
|
fetch_error=None,
|
||||||
|
)
|
||||||
|
mock_leases = LeaseSnapshot(
|
||||||
|
project_id=proj_id,
|
||||||
|
repo_label="org/repo",
|
||||||
|
issue_lock=None,
|
||||||
|
claim_inventory={},
|
||||||
|
reviewer_leases=(),
|
||||||
|
duplicate_prs=(),
|
||||||
|
duplicate_branches=(),
|
||||||
|
collision_history=(),
|
||||||
|
fetch_error=None,
|
||||||
|
)
|
||||||
|
mock_version = VersionInfo(
|
||||||
|
git_sha="abc1234",
|
||||||
|
git_describe="v1.0.0",
|
||||||
|
control_plane_schema_version=1,
|
||||||
|
python_version="3.11",
|
||||||
|
known=True,
|
||||||
|
)
|
||||||
|
mock_stale = StaleRuntime(
|
||||||
|
daemon_head="abc1234",
|
||||||
|
checkout_head="abc1234",
|
||||||
|
remote_head="abc1234",
|
||||||
|
stale=False,
|
||||||
|
determinable=True,
|
||||||
|
mutation_safe=True,
|
||||||
|
reasons=(),
|
||||||
|
)
|
||||||
|
mock_health = SystemHealthSnapshot(
|
||||||
|
status="degraded",
|
||||||
|
ready=False,
|
||||||
|
readiness_complete=True,
|
||||||
|
readiness_reasons=(),
|
||||||
|
service="webui",
|
||||||
|
mode="test",
|
||||||
|
version=mock_version,
|
||||||
|
started_at="2026-07-25T00:00:00Z",
|
||||||
|
uptime_seconds=100.0,
|
||||||
|
timestamp="2026-07-25T00:00:00Z",
|
||||||
|
deep_probes_requested=True,
|
||||||
|
dependencies=(),
|
||||||
|
mcp_namespaces=(),
|
||||||
|
stale_runtime=mock_stale,
|
||||||
|
probe_errors=("probe blew up",),
|
||||||
|
)
|
||||||
|
|
||||||
|
snapshot = load_notifications_snapshot(
|
||||||
|
proj_id,
|
||||||
|
load_queue=lambda _id: mock_queue,
|
||||||
|
load_leases=lambda **_kwargs: mock_leases,
|
||||||
|
load_health=lambda **_kwargs: mock_health,
|
||||||
|
)
|
||||||
|
assert snapshot.fetch_error is None
|
||||||
|
# probe errors still appear as items
|
||||||
|
assert any("probe blew up" in item.summary for item in snapshot.items)
|
||||||
@@ -0,0 +1,190 @@
|
|||||||
|
"""Tests for Sentry/GlitchTip observability console (#649, Phase 4)."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import pytest
|
||||||
|
from control_plane_db import ControlPlaneDB
|
||||||
|
from webui.app import create_app
|
||||||
|
from webui.console_authz import authorize, resolve_principal
|
||||||
|
from webui.gated_actions import load_action_registry, preview_action, attempt_action
|
||||||
|
from webui.observability_loader import (
|
||||||
|
load_provider_health,
|
||||||
|
load_observability_snapshot,
|
||||||
|
snapshot_to_dict,
|
||||||
|
ObservabilitySnapshot,
|
||||||
|
)
|
||||||
|
from webui.observability_views import render_observability_page
|
||||||
|
from tests.webui_testclient import TestClient
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def test_db(tmp_path):
|
||||||
|
db_path = str(tmp_path / "test_control_plane.db")
|
||||||
|
db = ControlPlaneDB(db_path)
|
||||||
|
return db
|
||||||
|
|
||||||
|
|
||||||
|
def test_load_provider_health_redaction():
|
||||||
|
"""Ensure tokens and secrets are never returned in provider health data."""
|
||||||
|
env = {
|
||||||
|
"SENTRY_BASE_URL": "https://sentry.prgs.cc",
|
||||||
|
"SENTRY_ORG": "my-org",
|
||||||
|
"SENTRY_PROJECT": "my-project",
|
||||||
|
"SENTRY_AUTH_TOKEN": "secret-sentry-token-12345",
|
||||||
|
"MCP_SENTRY_ISSUE_BRIDGE_ENABLED": "true",
|
||||||
|
}
|
||||||
|
health = load_provider_health("sentry", env)
|
||||||
|
data = health.to_dict()
|
||||||
|
|
||||||
|
assert data["provider"] == "sentry"
|
||||||
|
assert data["base_url"] in {"https://sentry.prgs.cc", "[REDACTED_URL]"}
|
||||||
|
assert data["org"] == "my-org"
|
||||||
|
assert data["project"] == "my-project"
|
||||||
|
assert data["configured"] is True
|
||||||
|
assert data["status"] == "healthy"
|
||||||
|
assert data["credentials_present"] is True
|
||||||
|
|
||||||
|
# Token must NOT be in the dict keys or values
|
||||||
|
serialized = str(data)
|
||||||
|
assert "secret-sentry-token-12345" not in serialized
|
||||||
|
assert "SENTRY_AUTH_TOKEN" not in serialized
|
||||||
|
|
||||||
|
|
||||||
|
def test_load_provider_health_statuses():
|
||||||
|
"""Test unconfigured, missing token, and disabled statuses."""
|
||||||
|
# Not configured
|
||||||
|
h1 = load_provider_health("sentry", {})
|
||||||
|
d1 = h1.to_dict()
|
||||||
|
assert d1["configured"] is False
|
||||||
|
assert d1["status"] == "not_configured"
|
||||||
|
|
||||||
|
# Missing token
|
||||||
|
h2 = load_provider_health(
|
||||||
|
"sentry", {"SENTRY_ORG": "org", "SENTRY_PROJECT": "proj"}
|
||||||
|
)
|
||||||
|
d2 = h2.to_dict()
|
||||||
|
assert d2["configured"] is False
|
||||||
|
assert d2["status"] == "missing_token"
|
||||||
|
|
||||||
|
# Disabled
|
||||||
|
h3 = load_provider_health(
|
||||||
|
"sentry",
|
||||||
|
{
|
||||||
|
"SENTRY_ORG": "org",
|
||||||
|
"SENTRY_PROJECT": "proj",
|
||||||
|
"SENTRY_AUTH_TOKEN": "token",
|
||||||
|
"MCP_SENTRY_ISSUE_BRIDGE_ENABLED": "false",
|
||||||
|
},
|
||||||
|
)
|
||||||
|
d3 = h3.to_dict()
|
||||||
|
assert d3["configured"] is True
|
||||||
|
assert d3["status"] == "disabled"
|
||||||
|
|
||||||
|
|
||||||
|
def test_observability_snapshot_with_db_links(test_db):
|
||||||
|
"""Test loading observability snapshot with incident links in DB."""
|
||||||
|
test_db.upsert_incident_link(
|
||||||
|
provider="sentry",
|
||||||
|
provider_issue_id="101",
|
||||||
|
gitea_org="Scaled-Tech-Consulting",
|
||||||
|
gitea_repo="Gitea-Tools",
|
||||||
|
gitea_issue_number=649,
|
||||||
|
provider_base_url="https://sentry.prgs.cc",
|
||||||
|
provider_org="Scaled-Tech-Consulting",
|
||||||
|
provider_project="Gitea-Tools",
|
||||||
|
provider_short_id="ST-101",
|
||||||
|
provider_permalink="https://sentry.prgs.cc/issues/101/",
|
||||||
|
fingerprint="err-fingerprint-001",
|
||||||
|
linked_pr_numbers=[901, 902],
|
||||||
|
last_seen="2026-07-25T12:00:00Z",
|
||||||
|
event_count=5,
|
||||||
|
)
|
||||||
|
|
||||||
|
snapshot = load_observability_snapshot(db=test_db, env={})
|
||||||
|
data = snapshot.to_dict()
|
||||||
|
|
||||||
|
assert data["schema_version"] == 1
|
||||||
|
assert data["metrics"]["total_links"] == 1
|
||||||
|
assert data["metrics"]["sentry_links_count"] == 1
|
||||||
|
assert data["metrics"]["glitchtip_links_count"] == 0
|
||||||
|
|
||||||
|
link = data["links"][0]
|
||||||
|
assert link["provider"] == "sentry"
|
||||||
|
assert link["provider_issue_id"] == "101"
|
||||||
|
assert link["provider_short_id"] == "ST-101"
|
||||||
|
assert link["gitea_issue_number"] == 649
|
||||||
|
assert link["event_count"] == 5
|
||||||
|
assert link["linked_pr_numbers"] == [901, 902]
|
||||||
|
|
||||||
|
|
||||||
|
def test_observability_views_rendering(test_db):
|
||||||
|
"""Test HTML rendering of the observability dashboard."""
|
||||||
|
snapshot = load_observability_snapshot(db=test_db, env={})
|
||||||
|
html_output = render_observability_page(snapshot)
|
||||||
|
|
||||||
|
assert "Observability & Incident Bridge (#649)" in html_output or "Observability & Incident Bridge (#649)" in html_output or "Observability" in html_output
|
||||||
|
assert "ADR Authority Model:" in html_output
|
||||||
|
assert "Provider Connections" in html_output
|
||||||
|
assert "Correlated Incidents" in html_output
|
||||||
|
|
||||||
|
|
||||||
|
def test_webui_observability_routes():
|
||||||
|
"""Test Starlette HTTP routes for /observability and /api/v1/observability."""
|
||||||
|
client = TestClient(create_app())
|
||||||
|
|
||||||
|
# HTML page route
|
||||||
|
res_html = client.get("/observability")
|
||||||
|
assert res_html.status_code == 200
|
||||||
|
assert "text/html" in res_html.headers["content-type"]
|
||||||
|
assert "Observability" in res_html.text
|
||||||
|
|
||||||
|
# Versioned API route
|
||||||
|
res_api_v1 = client.get("/api/v1/observability")
|
||||||
|
assert res_api_v1.status_code == 200
|
||||||
|
assert "application/json" in res_api_v1.headers["content-type"]
|
||||||
|
data_v1 = res_api_v1.json()
|
||||||
|
assert "schema_version" in data_v1
|
||||||
|
assert "providers" in data_v1
|
||||||
|
assert "links" in data_v1
|
||||||
|
assert "metrics" in data_v1
|
||||||
|
|
||||||
|
# Compatibility alias route
|
||||||
|
res_api_alias = client.get("/api/observability")
|
||||||
|
assert res_api_alias.status_code == 200
|
||||||
|
assert res_api_alias.json() == data_v1
|
||||||
|
|
||||||
|
|
||||||
|
def test_observability_gated_actions():
|
||||||
|
"""Ensure observability actions are registered, gated, and fail closed in MVP mode."""
|
||||||
|
registry = load_action_registry()
|
||||||
|
|
||||||
|
action_reconcile = registry.get("observability_reconcile_incident")
|
||||||
|
assert action_reconcile is not None
|
||||||
|
assert action_reconcile.task_key == "observability_reconcile_incident"
|
||||||
|
assert action_reconcile.mcp_tool == "gitea_observability_reconcile_incident"
|
||||||
|
|
||||||
|
action_link = registry.get("observability_link_issue")
|
||||||
|
assert action_link is not None
|
||||||
|
assert action_link.task_key == "observability_link_issue"
|
||||||
|
|
||||||
|
# Preview returns mutation ledger
|
||||||
|
prev = preview_action("observability_reconcile_incident", provider="sentry", issue_id="101")
|
||||||
|
assert prev["action_id"] == "observability_reconcile_incident"
|
||||||
|
assert prev["enabled"] is False
|
||||||
|
|
||||||
|
# Execution fails closed in MVP mode
|
||||||
|
att = attempt_action("observability_reconcile_incident", provider="sentry", issue_id="101")
|
||||||
|
assert att["success"] is False
|
||||||
|
assert att["error"] == "action_disabled"
|
||||||
|
|
||||||
|
|
||||||
|
def test_observability_authz_rbac():
|
||||||
|
"""Test RBAC authorization for observability actions."""
|
||||||
|
principal = resolve_principal({})
|
||||||
|
|
||||||
|
# Check authorize decision
|
||||||
|
decision = authorize("observability_reconcile_incident", principal)
|
||||||
|
assert decision.action_id == "observability_reconcile_incident"
|
||||||
|
# Phase 4 action denies in Phase 1 runtime by default
|
||||||
|
assert decision.allowed is False
|
||||||
@@ -0,0 +1,452 @@
|
|||||||
|
"""Read-only restart console: views, gates, and honesty rules (#667).
|
||||||
|
|
||||||
|
The console consumes the #655 substrate. These tests hold it to the three
|
||||||
|
properties that make a status surface trustworthy:
|
||||||
|
|
||||||
|
* an unreadable source is reported unavailable, never rendered as green;
|
||||||
|
* authorization is probed the way execution would probe it, so an allow is
|
||||||
|
never shown for something that could not run;
|
||||||
|
* the surface performs no mutation, including no write to the control-plane DB.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sqlite3
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
from datetime import datetime, timedelta, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||||
|
|
||||||
|
from starlette.testclient import TestClient # noqa: E402
|
||||||
|
|
||||||
|
import restart_coordinator # noqa: E402
|
||||||
|
from webui import console_authz, restart_console, restart_views # noqa: E402
|
||||||
|
from webui.app import create_app # noqa: E402
|
||||||
|
|
||||||
|
NOW = datetime(2026, 7, 25, 21, 0, 0, tzinfo=timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
def _principal(role: str) -> console_authz.Principal:
|
||||||
|
return console_authz.Principal(
|
||||||
|
subject="[email protected]",
|
||||||
|
role=role,
|
||||||
|
identity_source=console_authz.IDENTITY_LOCAL_DEV,
|
||||||
|
authenticated=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _inventory(*, complete: bool = True, sessions=(), leases=()):
|
||||||
|
def _read(**_kwargs):
|
||||||
|
return {
|
||||||
|
"sessions": list(sessions),
|
||||||
|
"leases": list(leases),
|
||||||
|
"terminal_lock": None,
|
||||||
|
"prior_recovery_attempts": [],
|
||||||
|
"inventory_complete": complete,
|
||||||
|
"incomplete_reasons": (
|
||||||
|
[] if complete else ["fixture: inventory withheld"]
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
return _read
|
||||||
|
|
||||||
|
|
||||||
|
def _live_session(session_id: str = "prgs-author-1234-abcd") -> dict:
|
||||||
|
return {
|
||||||
|
"session_id": session_id,
|
||||||
|
"role": "author",
|
||||||
|
"profile": "prgs-author",
|
||||||
|
"pid": os.getpid(),
|
||||||
|
"status": "active",
|
||||||
|
"last_heartbeat_at": (NOW - timedelta(seconds=30)).isoformat(),
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def drain_proof_fixture() -> dict:
|
||||||
|
"""A structurally complete but unsigned drain proof."""
|
||||||
|
return {
|
||||||
|
"version": "drain-proof/v1",
|
||||||
|
"proof_id": "deadbeef" * 8,
|
||||||
|
"clean": True,
|
||||||
|
"issued_at": (NOW - timedelta(minutes=1)).isoformat(),
|
||||||
|
"expires_at": (NOW + timedelta(minutes=5)).isoformat(),
|
||||||
|
"requesting_session_id": "s-live",
|
||||||
|
"impact_fingerprint": "f" * 64,
|
||||||
|
"checks": [],
|
||||||
|
"failed_checks": [],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class RestartClassMatrixTest(unittest.TestCase):
|
||||||
|
def test_every_policy_class_is_rendered(self) -> None:
|
||||||
|
views = restart_console.build_restart_class_views("operator")
|
||||||
|
self.assertEqual(len(views), len(restart_coordinator.RESTART_CLASS_POLICIES))
|
||||||
|
|
||||||
|
def test_viewer_capability_is_role_scoped_not_generic(self) -> None:
|
||||||
|
"""A worker role must not be shown as able to request a full restart."""
|
||||||
|
author = {
|
||||||
|
v.restart_class: v
|
||||||
|
for v in restart_console.build_restart_class_views("author")
|
||||||
|
}
|
||||||
|
operator = {
|
||||||
|
v.restart_class: v
|
||||||
|
for v in restart_console.build_restart_class_views("operator")
|
||||||
|
}
|
||||||
|
full = restart_coordinator.RestartClass.FULL_MCP_RESTART.value
|
||||||
|
|
||||||
|
self.assertFalse(author[full].viewer_may_request)
|
||||||
|
self.assertFalse(author[full].viewer_may_execute)
|
||||||
|
self.assertTrue(operator[full].viewer_may_request)
|
||||||
|
self.assertTrue(operator[full].viewer_may_execute)
|
||||||
|
|
||||||
|
def test_unknown_role_may_do_nothing(self) -> None:
|
||||||
|
views = restart_console.build_restart_class_views("not-a-role")
|
||||||
|
self.assertTrue(all(not v.viewer_may_request for v in views))
|
||||||
|
self.assertTrue(all(not v.viewer_may_execute for v in views))
|
||||||
|
|
||||||
|
|
||||||
|
class AuthorizationProbeTest(unittest.TestCase):
|
||||||
|
def test_probe_asks_for_execution_so_phase_gate_is_reported(self) -> None:
|
||||||
|
"""An admin clears the role bar and still cannot execute in Phase 1.
|
||||||
|
|
||||||
|
This is the case that distinguishes the two probes. Asked without
|
||||||
|
``for_execution`` an admin is *allowed* for ``system.restart_namespace``,
|
||||||
|
which on a control surface reads as a live button. Asked the way
|
||||||
|
execution asks, the same principal is refused ``phase_not_active``. The
|
||||||
|
console must report the second answer.
|
||||||
|
"""
|
||||||
|
by_id = {
|
||||||
|
a.action_id: a
|
||||||
|
for a in restart_console.build_action_authorizations(
|
||||||
|
_principal(console_authz.ADMIN)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
restart = by_id["system.restart_namespace"]
|
||||||
|
|
||||||
|
self.assertFalse(restart.execution_enabled)
|
||||||
|
self.assertEqual(restart.reason_code, console_authz.DENY_PHASE_NOT_ACTIVE)
|
||||||
|
|
||||||
|
permissive = console_authz.authorize(
|
||||||
|
"system.restart_namespace", _principal(console_authz.ADMIN)
|
||||||
|
)
|
||||||
|
self.assertTrue(
|
||||||
|
permissive.allowed,
|
||||||
|
"guard precondition: without for_execution an admin is allowed, "
|
||||||
|
"which is exactly why the console must not probe that way",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_operator_is_refused_the_admin_only_restart_action(self) -> None:
|
||||||
|
"""Role refusal precedes the phase gate and is reported as such."""
|
||||||
|
by_id = {
|
||||||
|
a.action_id: a
|
||||||
|
for a in restart_console.build_action_authorizations(
|
||||||
|
_principal(console_authz.OPERATOR)
|
||||||
|
)
|
||||||
|
}
|
||||||
|
self.assertEqual(
|
||||||
|
by_id["system.restart_namespace"].reason_code,
|
||||||
|
console_authz.DENY_INSUFFICIENT_ROLE,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_anonymous_is_denied_unauthenticated(self) -> None:
|
||||||
|
by_id = {
|
||||||
|
a.action_id: a for a in restart_console.build_action_authorizations(None)
|
||||||
|
}
|
||||||
|
self.assertEqual(
|
||||||
|
by_id["system.restart_namespace"].reason_code,
|
||||||
|
console_authz.DENY_UNAUTHENTICATED,
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_no_authorization_ever_reports_execution_enabled(self) -> None:
|
||||||
|
for role in (
|
||||||
|
console_authz.VIEWER,
|
||||||
|
console_authz.OPERATOR,
|
||||||
|
console_authz.CONTROLLER,
|
||||||
|
console_authz.ADMIN,
|
||||||
|
):
|
||||||
|
for auth in restart_console.build_action_authorizations(_principal(role)):
|
||||||
|
self.assertFalse(
|
||||||
|
auth.execution_enabled,
|
||||||
|
f"{role} reported execution_enabled for {auth.action_id}",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class ImpactPreviewTest(unittest.TestCase):
|
||||||
|
def test_impact_renders_from_coordinator_dto(self) -> None:
|
||||||
|
impact, source = restart_console.load_impact_report(
|
||||||
|
principal=_principal(console_authz.OPERATOR),
|
||||||
|
read_inventory=_inventory(sessions=[_live_session()]),
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
self.assertTrue(source.available)
|
||||||
|
self.assertIsNotNone(impact)
|
||||||
|
self.assertEqual(
|
||||||
|
impact["restart_class"],
|
||||||
|
restart_coordinator.RestartClass.FULL_MCP_RESTART.value,
|
||||||
|
)
|
||||||
|
self.assertIn("verdict", impact)
|
||||||
|
self.assertFalse(impact["restart_performed"])
|
||||||
|
self.assertTrue(impact["dry_run"])
|
||||||
|
|
||||||
|
def test_incomplete_inventory_is_surfaced_and_denies(self) -> None:
|
||||||
|
impact, source = restart_console.load_impact_report(
|
||||||
|
principal=_principal(console_authz.OPERATOR),
|
||||||
|
read_inventory=_inventory(complete=False),
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
self.assertFalse(impact["inventory_complete"])
|
||||||
|
self.assertFalse(impact["allow_restart"])
|
||||||
|
self.assertTrue(source.detail, "incomplete inventory must explain itself")
|
||||||
|
|
||||||
|
def test_inventory_reader_failure_is_unavailable_not_empty(self) -> None:
|
||||||
|
"""A reader that raises must not be rendered as 'no sessions affected'."""
|
||||||
|
|
||||||
|
def _boom(**_kwargs):
|
||||||
|
raise RuntimeError("control-plane unreachable")
|
||||||
|
|
||||||
|
impact, source = restart_console.load_impact_report(
|
||||||
|
principal=_principal(console_authz.OPERATOR),
|
||||||
|
read_inventory=_boom,
|
||||||
|
now=NOW,
|
||||||
|
)
|
||||||
|
self.assertIsNone(impact)
|
||||||
|
self.assertFalse(source.available)
|
||||||
|
self.assertIn("control-plane unreachable", source.detail)
|
||||||
|
|
||||||
|
|
||||||
|
class ControlPlaneReadTest(unittest.TestCase):
|
||||||
|
def test_missing_database_is_incomplete_not_empty(self) -> None:
|
||||||
|
inventory = restart_console.read_control_plane_inventory(
|
||||||
|
db_path="/nonexistent/control-plane.sqlite3"
|
||||||
|
)
|
||||||
|
self.assertFalse(inventory["inventory_complete"])
|
||||||
|
self.assertEqual(inventory["sessions"], [])
|
||||||
|
self.assertTrue(inventory["incomplete_reasons"])
|
||||||
|
|
||||||
|
def test_reader_never_creates_the_database(self) -> None:
|
||||||
|
"""Reading status must not bring a control-plane DB into existence.
|
||||||
|
|
||||||
|
The path deliberately sits in a directory that already exists: a
|
||||||
|
read-write ``sqlite3.connect`` would happily create the file there, so
|
||||||
|
this fails if the reader ever stops opening the database ``mode=ro``.
|
||||||
|
A nested-missing-directory path would pass for the wrong reason,
|
||||||
|
because sqlite cannot create the parent directory either way.
|
||||||
|
"""
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "control_plane.sqlite3")
|
||||||
|
self.assertTrue(os.path.isdir(os.path.dirname(path)))
|
||||||
|
|
||||||
|
inventory = restart_console.read_control_plane_inventory(db_path=path)
|
||||||
|
|
||||||
|
self.assertFalse(
|
||||||
|
os.path.exists(path),
|
||||||
|
"reading restart status created a control-plane database",
|
||||||
|
)
|
||||||
|
self.assertFalse(inventory["inventory_complete"])
|
||||||
|
|
||||||
|
def test_reads_active_sessions_from_a_real_database(self) -> None:
|
||||||
|
with tempfile.TemporaryDirectory() as tmp:
|
||||||
|
path = os.path.join(tmp, "cp.sqlite3")
|
||||||
|
conn = sqlite3.connect(path)
|
||||||
|
conn.execute(
|
||||||
|
"CREATE TABLE sessions (session_id TEXT, role TEXT, profile TEXT,"
|
||||||
|
" pid INTEGER, status TEXT, last_heartbeat_at TEXT)"
|
||||||
|
)
|
||||||
|
conn.execute(
|
||||||
|
"CREATE TABLE work_items (work_item_id INTEGER, kind TEXT,"
|
||||||
|
" number INTEGER)"
|
||||||
|
)
|
||||||
|
conn.execute(
|
||||||
|
"CREATE TABLE leases (lease_id TEXT, session_id TEXT, role TEXT,"
|
||||||
|
" phase TEXT, status TEXT, worktree_path TEXT,"
|
||||||
|
" work_item_id INTEGER, expires_at TEXT)"
|
||||||
|
)
|
||||||
|
conn.execute(
|
||||||
|
"INSERT INTO sessions VALUES (?,?,?,?,?,?)",
|
||||||
|
("s-live", "author", "prgs-author", 4242, "active", NOW.isoformat()),
|
||||||
|
)
|
||||||
|
conn.execute(
|
||||||
|
"INSERT INTO sessions VALUES (?,?,?,?,?,?)",
|
||||||
|
("s-done", "author", "prgs-author", 11, "closed", NOW.isoformat()),
|
||||||
|
)
|
||||||
|
conn.execute("INSERT INTO work_items VALUES (1, 'issue', 667)")
|
||||||
|
conn.execute(
|
||||||
|
"INSERT INTO leases VALUES (?,?,?,?,?,?,?,?)",
|
||||||
|
(
|
||||||
|
"l-1",
|
||||||
|
"s-live",
|
||||||
|
"author",
|
||||||
|
"allocated",
|
||||||
|
"active",
|
||||||
|
None,
|
||||||
|
1,
|
||||||
|
NOW.isoformat(),
|
||||||
|
),
|
||||||
|
)
|
||||||
|
conn.commit()
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
inventory = restart_console.read_control_plane_inventory(db_path=path)
|
||||||
|
|
||||||
|
self.assertTrue(inventory["inventory_complete"])
|
||||||
|
self.assertEqual([s["session_id"] for s in inventory["sessions"]], ["s-live"])
|
||||||
|
self.assertEqual(inventory["leases"][0]["work_number"], 667)
|
||||||
|
|
||||||
|
|
||||||
|
class DrainAndReconcileTest(unittest.TestCase):
|
||||||
|
def test_absent_drain_proof_is_not_a_pass(self) -> None:
|
||||||
|
drain, source = restart_console.load_drain_status(proof=None, now=NOW)
|
||||||
|
self.assertIsNone(drain)
|
||||||
|
self.assertFalse(source.available)
|
||||||
|
self.assertIn("denies", source.detail)
|
||||||
|
|
||||||
|
def test_tampered_drain_proof_is_reported_invalid(self) -> None:
|
||||||
|
proof = drain_proof_fixture()
|
||||||
|
proof["clean"] = True
|
||||||
|
proof["proof_id"] = "0" * 64
|
||||||
|
drain, source = restart_console.load_drain_status(proof=proof, now=NOW)
|
||||||
|
self.assertTrue(source.available)
|
||||||
|
self.assertFalse(drain["valid"])
|
||||||
|
|
||||||
|
def test_absent_reconcile_proof_is_unavailable(self) -> None:
|
||||||
|
reconcile, source = restart_console.load_reconcile_status(load_proof=None)
|
||||||
|
self.assertIsNone(reconcile)
|
||||||
|
self.assertFalse(source.available)
|
||||||
|
|
||||||
|
def test_reconcile_proof_is_rendered_when_supplied(self) -> None:
|
||||||
|
payload = {
|
||||||
|
"overall_status": "degraded",
|
||||||
|
"mode": "log_only",
|
||||||
|
"resolved_count": 3,
|
||||||
|
"unresolved_count": 2,
|
||||||
|
"items": [
|
||||||
|
{
|
||||||
|
"dimension": "leases",
|
||||||
|
"status": "unresolved",
|
||||||
|
"summary": "2 orphaned leases",
|
||||||
|
"follow_up_required": True,
|
||||||
|
}
|
||||||
|
],
|
||||||
|
}
|
||||||
|
reconcile, source = restart_console.load_reconcile_status(
|
||||||
|
load_proof=lambda: payload
|
||||||
|
)
|
||||||
|
self.assertTrue(source.available)
|
||||||
|
self.assertEqual(reconcile["unresolved_count"], 2)
|
||||||
|
|
||||||
|
|
||||||
|
class RenderingTest(unittest.TestCase):
|
||||||
|
def _snapshot(self, **kwargs):
|
||||||
|
params = {
|
||||||
|
"principal": _principal(console_authz.OPERATOR),
|
||||||
|
"read_inventory": _inventory(sessions=[_live_session()]),
|
||||||
|
"now": NOW,
|
||||||
|
}
|
||||||
|
params.update(kwargs)
|
||||||
|
return restart_console.load_restart_console_snapshot(**params)
|
||||||
|
|
||||||
|
def test_page_renders_every_section(self) -> None:
|
||||||
|
html = restart_views.render_restart_console_page(self._snapshot())
|
||||||
|
for heading in (
|
||||||
|
"Impact preview",
|
||||||
|
"Drain proof",
|
||||||
|
"Post-restart reconcile",
|
||||||
|
"Restart classes",
|
||||||
|
"Approval controls",
|
||||||
|
"Break-glass",
|
||||||
|
):
|
||||||
|
self.assertIn(heading, html)
|
||||||
|
|
||||||
|
def test_hostile_session_id_is_escaped(self) -> None:
|
||||||
|
hostile = "<script>alert('x')</script>"
|
||||||
|
html = restart_views.render_restart_console_page(
|
||||||
|
self._snapshot(read_inventory=_inventory(sessions=[_live_session(hostile)]))
|
||||||
|
)
|
||||||
|
self.assertNotIn("<script>alert", html)
|
||||||
|
self.assertIn("<script>", html)
|
||||||
|
|
||||||
|
def test_unavailable_impact_says_unsafe_rather_than_clean(self) -> None:
|
||||||
|
def _boom(**_kwargs):
|
||||||
|
raise RuntimeError("nope")
|
||||||
|
|
||||||
|
snapshot = self._snapshot(read_inventory=_boom)
|
||||||
|
html = restart_views.render_restart_console_page(snapshot)
|
||||||
|
self.assertIn("blast radius of a restart is unknown", html)
|
||||||
|
self.assertIn("unavailable", html)
|
||||||
|
|
||||||
|
def test_break_glass_is_hidden_from_unprivileged_viewers(self) -> None:
|
||||||
|
viewer_html = restart_views.render_restart_console_page(
|
||||||
|
self._snapshot(principal=_principal(console_authz.VIEWER))
|
||||||
|
)
|
||||||
|
self.assertIn("visible to operator-class", viewer_html)
|
||||||
|
self.assertNotIn(
|
||||||
|
f"#{restart_console.BREAK_GLASS_ISSUE}", viewer_html
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_break_glass_shown_to_operator_is_marked_unavailable(self) -> None:
|
||||||
|
html = restart_views.render_restart_console_page(self._snapshot())
|
||||||
|
self.assertIn("unavailable", html)
|
||||||
|
self.assertIn(f"#{restart_console.BREAK_GLASS_ISSUE}", html)
|
||||||
|
|
||||||
|
def test_snapshot_always_declares_itself_read_only(self) -> None:
|
||||||
|
self.assertTrue(self._snapshot().read_only)
|
||||||
|
|
||||||
|
|
||||||
|
class RestartConsoleRouteTest(unittest.TestCase):
|
||||||
|
def setUp(self) -> None:
|
||||||
|
self.client = TestClient(create_app())
|
||||||
|
|
||||||
|
def test_page_route_renders(self) -> None:
|
||||||
|
res = self.client.get("/runtime/restart")
|
||||||
|
self.assertEqual(res.status_code, 200)
|
||||||
|
self.assertIn("Restart status and impact", res.text)
|
||||||
|
|
||||||
|
def test_api_route_exports_snapshot(self) -> None:
|
||||||
|
res = self.client.get("/api/v1/system/restart/status")
|
||||||
|
self.assertEqual(res.status_code, 200)
|
||||||
|
payload = res.json()
|
||||||
|
self.assertTrue(payload["read_only"])
|
||||||
|
self.assertEqual(payload["links"]["issue"], 667)
|
||||||
|
self.assertEqual(
|
||||||
|
len(payload["restart_classes"]),
|
||||||
|
len(restart_coordinator.RESTART_CLASS_POLICIES),
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_restart_class_is_selectable(self) -> None:
|
||||||
|
res = self.client.get(
|
||||||
|
"/api/v1/system/restart/status?restart_class=client_reconnect"
|
||||||
|
)
|
||||||
|
self.assertEqual(res.status_code, 200)
|
||||||
|
self.assertEqual(res.json()["impact"]["restart_class"], "client_reconnect")
|
||||||
|
|
||||||
|
def test_unknown_restart_class_fails_closed(self) -> None:
|
||||||
|
res = self.client.get(
|
||||||
|
"/api/v1/system/restart/status?restart_class=obliterate-everything"
|
||||||
|
)
|
||||||
|
self.assertEqual(res.status_code, 200)
|
||||||
|
impact = res.json()["impact"]
|
||||||
|
self.assertFalse(impact["allow_restart"])
|
||||||
|
|
||||||
|
def test_anonymous_api_reader_gets_no_execution_grant(self) -> None:
|
||||||
|
payload = self.client.get("/api/v1/system/restart/status").json()
|
||||||
|
self.assertFalse(payload["break_glass"]["available"])
|
||||||
|
for auth in payload["authorizations"]:
|
||||||
|
self.assertFalse(auth["execution_enabled"])
|
||||||
|
|
||||||
|
def test_route_is_registered_in_nav(self) -> None:
|
||||||
|
from webui.nav import nav_hrefs
|
||||||
|
|
||||||
|
self.assertIn("/runtime/restart", nav_hrefs())
|
||||||
|
|
||||||
|
def test_no_write_method_is_exposed(self) -> None:
|
||||||
|
"""The surface is read-only: nothing accepts a POST."""
|
||||||
|
for path in ("/runtime/restart", "/api/v1/system/restart/status"):
|
||||||
|
self.assertEqual(self.client.post(path).status_code, 405, path)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -47,6 +47,9 @@ from webui.traffic_views import render_traffic_page
|
|||||||
from webui.worktree_scanner import load_hygiene_snapshot, snapshot_to_dict as worktree_snapshot_to_dict
|
from webui.worktree_scanner import load_hygiene_snapshot, snapshot_to_dict as worktree_snapshot_to_dict
|
||||||
from webui.worktree_views import render_worktrees_page
|
from webui.worktree_views import render_worktrees_page
|
||||||
from webui.runtime_health import load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict
|
from webui.runtime_health import load_runtime_snapshot, snapshot_to_dict as runtime_snapshot_to_dict
|
||||||
|
import restart_coordinator
|
||||||
|
from webui.restart_console import load_restart_console_snapshot
|
||||||
|
from webui.restart_views import render_restart_console_page
|
||||||
from webui.runtime_views import render_runtime_page
|
from webui.runtime_views import render_runtime_page
|
||||||
from webui.session_loader import (
|
from webui.session_loader import (
|
||||||
load_session_view_snapshot,
|
load_session_view_snapshot,
|
||||||
@@ -77,8 +80,18 @@ from webui.system_health import (
|
|||||||
snapshot_to_dict as system_health_to_dict,
|
snapshot_to_dict as system_health_to_dict,
|
||||||
)
|
)
|
||||||
from webui.system_health_views import render_system_health_page
|
from webui.system_health_views import render_system_health_page
|
||||||
|
from webui.notifications import (
|
||||||
|
load_notifications_snapshot,
|
||||||
|
snapshot_to_dict as notifications_snapshot_to_dict,
|
||||||
|
)
|
||||||
|
from webui.notification_views import render_notifications_page
|
||||||
from webui import request_service
|
from webui import request_service
|
||||||
from webui.request_views import render_requests_page
|
from webui.request_views import render_requests_page
|
||||||
|
from webui.observability_loader import (
|
||||||
|
load_observability_snapshot,
|
||||||
|
snapshot_to_dict as observability_snapshot_to_dict,
|
||||||
|
)
|
||||||
|
from webui.observability_views import render_observability_page
|
||||||
|
|
||||||
_READ_ONLY_METHODS = frozenset({"GET", "HEAD", "OPTIONS"})
|
_READ_ONLY_METHODS = frozenset({"GET", "HEAD", "OPTIONS"})
|
||||||
_AUDIT_MUTATION_PATHS = frozenset({"/audit", "/api/audit"})
|
_AUDIT_MUTATION_PATHS = frozenset({"/audit", "/api/audit"})
|
||||||
@@ -416,6 +429,33 @@ async def api_runtime(_request: Request) -> JSONResponse:
|
|||||||
return JSONResponse(runtime_snapshot_to_dict(load_runtime_snapshot()))
|
return JSONResponse(runtime_snapshot_to_dict(load_runtime_snapshot()))
|
||||||
|
|
||||||
|
|
||||||
|
def _restart_console_snapshot(request: Request):
|
||||||
|
"""Build the read-only restart snapshot for the requesting principal (#667)."""
|
||||||
|
principal = resolve_principal(request.headers)
|
||||||
|
restart_class = (
|
||||||
|
request.query_params.get("restart_class")
|
||||||
|
or restart_coordinator.RestartClass.FULL_MCP_RESTART.value
|
||||||
|
)
|
||||||
|
return load_restart_console_snapshot(
|
||||||
|
principal=principal, restart_class=restart_class
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def restart_console_page(request: Request) -> HTMLResponse:
|
||||||
|
"""Restart status, impact preview, and approval state (#667). Read-only."""
|
||||||
|
snapshot = _restart_console_snapshot(request)
|
||||||
|
return HTMLResponse(
|
||||||
|
render_page(
|
||||||
|
title="Restart", body_html=render_restart_console_page(snapshot)
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def api_restart_status(request: Request) -> JSONResponse:
|
||||||
|
"""JSON export of the read-only restart console snapshot (#667)."""
|
||||||
|
return JSONResponse(_restart_console_snapshot(request).as_dict())
|
||||||
|
|
||||||
|
|
||||||
async def sessions(_request: Request) -> HTMLResponse:
|
async def sessions(_request: Request) -> HTMLResponse:
|
||||||
"""Runtime and session view (#641) — read-only composition of health + inventory."""
|
"""Runtime and session view (#641) — read-only composition of health + inventory."""
|
||||||
snapshot = load_session_view_snapshot()
|
snapshot = load_session_view_snapshot()
|
||||||
@@ -859,6 +899,35 @@ async def api_v1_analytics_ingest(request: Request) -> JSONResponse:
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
async def notifications_route(request: Request) -> HTMLResponse:
|
||||||
|
project_id = request.query_params.get("project_id")
|
||||||
|
attention_class = request.query_params.get("attention_class") or "inbox"
|
||||||
|
snap = load_notifications_snapshot(project_id)
|
||||||
|
html = render_notifications_page(
|
||||||
|
snap, filter_class=attention_class, filter_project=project_id
|
||||||
|
)
|
||||||
|
return HTMLResponse(html)
|
||||||
|
|
||||||
|
|
||||||
|
async def api_notifications(request: Request) -> JSONResponse:
|
||||||
|
project_id = request.query_params.get("project_id")
|
||||||
|
snap = load_notifications_snapshot(project_id)
|
||||||
|
data = notifications_snapshot_to_dict(snap)
|
||||||
|
return JSONResponse(data)
|
||||||
|
|
||||||
|
|
||||||
|
async def observability_route(request: Request) -> HTMLResponse:
|
||||||
|
snap = load_observability_snapshot()
|
||||||
|
html_content = render_observability_page(snap)
|
||||||
|
return HTMLResponse(html_content)
|
||||||
|
|
||||||
|
|
||||||
|
async def api_observability(request: Request) -> JSONResponse:
|
||||||
|
snap = load_observability_snapshot()
|
||||||
|
data = observability_snapshot_to_dict(snap)
|
||||||
|
return JSONResponse(data)
|
||||||
|
|
||||||
|
|
||||||
def _default_request_scope() -> dict[str, str]:
|
def _default_request_scope() -> dict[str, str]:
|
||||||
"""Resolve remote/org/repo from the project registry for request forms.
|
"""Resolve remote/org/repo from the project registry for request forms.
|
||||||
|
|
||||||
@@ -990,6 +1059,9 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
|||||||
Route("/api/queue", api_queue, methods=["GET"]),
|
Route("/api/queue", api_queue, methods=["GET"]),
|
||||||
Route("/traffic", traffic, methods=["GET"]),
|
Route("/traffic", traffic, methods=["GET"]),
|
||||||
Route("/api/traffic", api_traffic, methods=["GET"]),
|
Route("/api/traffic", api_traffic, methods=["GET"]),
|
||||||
|
Route("/notifications", notifications_route, methods=["GET"]),
|
||||||
|
Route("/api/notifications", api_notifications, methods=["GET"]),
|
||||||
|
Route("/api/v1/notifications", api_notifications, methods=["GET"]),
|
||||||
Route("/projects", projects, methods=["GET"]),
|
Route("/projects", projects, methods=["GET"]),
|
||||||
Route("/projects/{project_id}", project_detail, methods=["GET"]),
|
Route("/projects/{project_id}", project_detail, methods=["GET"]),
|
||||||
Route("/api/projects", api_projects, methods=["GET"]),
|
Route("/api/projects", api_projects, methods=["GET"]),
|
||||||
@@ -1004,6 +1076,13 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
|||||||
Route("/api/prompts", api_prompts, methods=["GET"]),
|
Route("/api/prompts", api_prompts, methods=["GET"]),
|
||||||
Route("/runtime", runtime, methods=["GET"]),
|
Route("/runtime", runtime, methods=["GET"]),
|
||||||
Route("/api/runtime", api_runtime, methods=["GET"]),
|
Route("/api/runtime", api_runtime, methods=["GET"]),
|
||||||
|
# #667 read-only restart status / impact preview / approval state.
|
||||||
|
Route("/runtime/restart", restart_console_page, methods=["GET"]),
|
||||||
|
Route(
|
||||||
|
"/api/v1/system/restart/status",
|
||||||
|
api_restart_status,
|
||||||
|
methods=["GET"],
|
||||||
|
),
|
||||||
Route("/sessions", sessions, methods=["GET"]),
|
Route("/sessions", sessions, methods=["GET"]),
|
||||||
Route("/api/sessions", api_sessions, methods=["GET"]),
|
Route("/api/sessions", api_sessions, methods=["GET"]),
|
||||||
Route("/api/v1/sessions", api_sessions, methods=["GET"]),
|
Route("/api/v1/sessions", api_sessions, methods=["GET"]),
|
||||||
@@ -1014,6 +1093,9 @@ def create_app(*, bind_host: str | None = None) -> Starlette:
|
|||||||
Route("/api/analytics", api_v1_analytics, methods=["GET"]),
|
Route("/api/analytics", api_v1_analytics, methods=["GET"]),
|
||||||
Route("/api/v1/analytics", api_v1_analytics, methods=["GET"]),
|
Route("/api/v1/analytics", api_v1_analytics, methods=["GET"]),
|
||||||
Route("/api/v1/analytics/usage", api_v1_analytics_ingest, methods=["POST"]),
|
Route("/api/v1/analytics/usage", api_v1_analytics_ingest, methods=["POST"]),
|
||||||
|
Route("/observability", observability_route, methods=["GET"]),
|
||||||
|
Route("/api/observability", api_observability, methods=["GET"]),
|
||||||
|
Route("/api/v1/observability", api_observability, methods=["GET"]),
|
||||||
Route("/audit", audit, methods=["GET", "POST"]),
|
Route("/audit", audit, methods=["GET", "POST"]),
|
||||||
Route("/api/audit", api_audit, methods=["GET", "POST"]),
|
Route("/api/audit", api_audit, methods=["GET", "POST"]),
|
||||||
Route("/worktrees", worktrees, methods=["GET"]),
|
Route("/worktrees", worktrees, methods=["GET"]),
|
||||||
|
|||||||
@@ -317,6 +317,29 @@ _ACTION_SPECS: tuple[ConsoleAction, ...] = (
|
|||||||
phase=2,
|
phase=2,
|
||||||
summary="Run reconciler cleanup for merged or superseded PR branches.",
|
summary="Run reconciler cleanup for merged or superseded PR branches.",
|
||||||
),
|
),
|
||||||
|
# #649: Phase 4 observability & incident bridge actions.
|
||||||
|
ConsoleAction(
|
||||||
|
action_id="observability_reconcile_incident",
|
||||||
|
task_key="observability_reconcile_incident",
|
||||||
|
action_class=CLASS_WRITE,
|
||||||
|
minimum_role=OPERATOR,
|
||||||
|
requires_confirmation=True,
|
||||||
|
dual_control=False,
|
||||||
|
break_glass=False,
|
||||||
|
phase=4,
|
||||||
|
summary="Trigger/reconcile durable Gitea issue creation from a provider incident.",
|
||||||
|
),
|
||||||
|
ConsoleAction(
|
||||||
|
action_id="observability_link_issue",
|
||||||
|
task_key="observability_link_issue",
|
||||||
|
action_class=CLASS_WRITE,
|
||||||
|
minimum_role=OPERATOR,
|
||||||
|
requires_confirmation=True,
|
||||||
|
dual_control=False,
|
||||||
|
break_glass=False,
|
||||||
|
phase=4,
|
||||||
|
summary="Link a provider incident to an existing Gitea issue.",
|
||||||
|
),
|
||||||
# #643: submit a work request — desired role, issue/PR, intent — and let
|
# #643: submit a work request — desired role, issue/PR, intent — and let
|
||||||
# the allocator reserve it. This is the one Phase 2 action whose execution
|
# the allocator reserve it. This is the one Phase 2 action whose execution
|
||||||
# path is actually implemented (``webui.request_service``), so it carries
|
# path is actually implemented (``webui.request_service``), so it carries
|
||||||
|
|||||||
@@ -185,6 +185,11 @@ def build_action_registry() -> ActionRegistry:
|
|||||||
"console.rebind_session_worktree", "Rebind session worktree to verified lease."),
|
"console.rebind_session_worktree", "Rebind session worktree to verified lease."),
|
||||||
("system.reconcile_cleanups", "Reconcile cleanups", "reconcile_cleanups",
|
("system.reconcile_cleanups", "Reconcile cleanups", "reconcile_cleanups",
|
||||||
"console.reconcile_cleanups", "Run reconciler cleanup for merged or superseded PRs."),
|
"console.reconcile_cleanups", "Run reconciler cleanup for merged or superseded PRs."),
|
||||||
|
# #649: Phase 4 observability & incident bridge actions.
|
||||||
|
("observability_reconcile_incident", "Reconcile incident", "observability_reconcile_incident",
|
||||||
|
"gitea_observability_reconcile_incident", "Trigger or dry-run durable issue reconciliation for a provider incident."),
|
||||||
|
("observability_link_issue", "Link incident issue", "observability_link_issue",
|
||||||
|
"gitea_observability_link_issue", "Link a provider incident to a Gitea tracking issue."),
|
||||||
)
|
)
|
||||||
actions = tuple(
|
actions = tuple(
|
||||||
GatedAction(
|
GatedAction(
|
||||||
|
|||||||
@@ -46,10 +46,12 @@ NAV_GROUPS: tuple[NavGroup, ...] = (
|
|||||||
NavItem("/queue", "Queue"),
|
NavItem("/queue", "Queue"),
|
||||||
NavItem("/leases", "Leases"),
|
NavItem("/leases", "Leases"),
|
||||||
NavItem("/actions", "Actions"),
|
NavItem("/actions", "Actions"),
|
||||||
|
NavItem("/notifications", "Notifications"),
|
||||||
NavItem("/requests", "Requests"),
|
NavItem("/requests", "Requests"),
|
||||||
)),
|
)),
|
||||||
NavGroup("Runtime/Sessions", (
|
NavGroup("Runtime/Sessions", (
|
||||||
NavItem("/runtime", "Runtime health"),
|
NavItem("/runtime", "Runtime health"),
|
||||||
|
NavItem("/runtime/restart", "Restart status"),
|
||||||
NavItem("/sessions", "Sessions"),
|
NavItem("/sessions", "Sessions"),
|
||||||
)),
|
)),
|
||||||
NavGroup("Projects", (
|
NavGroup("Projects", (
|
||||||
@@ -71,6 +73,7 @@ NAV_GROUPS: tuple[NavGroup, ...] = (
|
|||||||
)),
|
)),
|
||||||
NavGroup("Insights", (
|
NavGroup("Insights", (
|
||||||
NavItem("/insights", "Insights", "stub"),
|
NavItem("/insights", "Insights", "stub"),
|
||||||
|
NavItem("/observability", "Observability"),
|
||||||
NavItem("/analytics", "Analytics"),
|
NavItem("/analytics", "Analytics"),
|
||||||
NavItem("/audit", "Audit"),
|
NavItem("/audit", "Audit"),
|
||||||
)),
|
)),
|
||||||
|
|||||||
@@ -0,0 +1,158 @@
|
|||||||
|
"""HTML rendering for Phase 3 Notifications and Human-Attention Console (#648)."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from html import escape
|
||||||
|
from typing import Sequence
|
||||||
|
|
||||||
|
from webui.layout import render_page
|
||||||
|
from webui.notifications import (
|
||||||
|
ATTENTION_HUMAN_REQUIRED,
|
||||||
|
ATTENTION_OPERATOR,
|
||||||
|
ATTENTION_ROUTINE,
|
||||||
|
NotificationItem,
|
||||||
|
NotificationSnapshot,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _render_attention_badge(attention_class: str) -> str:
|
||||||
|
cls = "badge"
|
||||||
|
if attention_class == ATTENTION_HUMAN_REQUIRED:
|
||||||
|
cls += " badge-blocked"
|
||||||
|
elif attention_class == ATTENTION_OPERATOR:
|
||||||
|
cls += " badge-claimed"
|
||||||
|
else:
|
||||||
|
cls += " muted"
|
||||||
|
return f'<span class="{cls}">{escape(attention_class)}</span>'
|
||||||
|
|
||||||
|
|
||||||
|
def _render_notification_row(item: NotificationItem) -> str:
|
||||||
|
category_label = escape(item.category.upper())
|
||||||
|
id_str = escape(item.id)
|
||||||
|
title_str = escape(item.title)
|
||||||
|
summary_str = escape(item.summary)
|
||||||
|
att_badge = _render_attention_badge(item.attention_class)
|
||||||
|
|
||||||
|
work_item_html = "—"
|
||||||
|
if item.work_number and item.work_kind:
|
||||||
|
kind_label = escape(item.work_kind.upper())
|
||||||
|
num_str = f"#{item.work_number}"
|
||||||
|
link = item.deep_link or "#"
|
||||||
|
work_item_html = f'<a href="{escape(link)}"><code>{kind_label} {num_str}</code></a>'
|
||||||
|
|
||||||
|
requires_human_label = (
|
||||||
|
'<span class="badge badge-blocked" style="font-size:0.75rem;">HUMAN REQUIRED</span>'
|
||||||
|
if item.requires_human
|
||||||
|
else ""
|
||||||
|
)
|
||||||
|
|
||||||
|
return f"""<tr>
|
||||||
|
<td><code>{category_label}</code><br><span class="muted" style="font-size:0.75rem;">{id_str}</span></td>
|
||||||
|
<td>
|
||||||
|
<div><strong>{title_str}</strong> {att_badge} {requires_human_label}</div>
|
||||||
|
<div class="muted" style="font-size:0.85rem; margin-top:0.25rem;">{summary_str}</div>
|
||||||
|
</td>
|
||||||
|
<td>{work_item_html}</td>
|
||||||
|
<td><span class="muted" style="font-size:0.8rem;">{escape(item.created_at[:19])}</span></td>
|
||||||
|
</tr>"""
|
||||||
|
|
||||||
|
|
||||||
|
def _render_notifications_table(items: Sequence[NotificationItem], empty_message: str) -> str:
|
||||||
|
if not items:
|
||||||
|
return f'<p class="muted" style="padding:1rem 0;">{escape(empty_message)}</p>'
|
||||||
|
|
||||||
|
rows = "".join(_render_notification_row(item) for item in items)
|
||||||
|
return f"""<table class="registry">
|
||||||
|
<thead>
|
||||||
|
<tr>
|
||||||
|
<th style="width: 18%;">Category & ID</th>
|
||||||
|
<th style="width: 52%;">Title & Attention Summary</th>
|
||||||
|
<th style="width: 15%;">Work Item</th>
|
||||||
|
<th style="width: 15%;">Time</th>
|
||||||
|
</tr>
|
||||||
|
</thead>
|
||||||
|
<tbody>
|
||||||
|
{rows}
|
||||||
|
</tbody>
|
||||||
|
</table>"""
|
||||||
|
|
||||||
|
|
||||||
|
def render_notifications_page(
|
||||||
|
snapshot: NotificationSnapshot,
|
||||||
|
*,
|
||||||
|
filter_class: str = "inbox",
|
||||||
|
filter_project: str | None = None,
|
||||||
|
) -> str:
|
||||||
|
"""Render the notifications and attention inbox page."""
|
||||||
|
title = "Notifications & Attention Inbox"
|
||||||
|
|
||||||
|
err_html = ""
|
||||||
|
if snapshot.fetch_error:
|
||||||
|
err_html = f'<div class="stub" style="border-color:#e53e3e; background:#fff5f5; color:#c53030; margin-bottom:1rem;"><p><strong>Fetch Warning:</strong> {escape(snapshot.fetch_error)}</p></div>'
|
||||||
|
|
||||||
|
# Determine items to render based on filter_class
|
||||||
|
if filter_class == ATTENTION_HUMAN_REQUIRED:
|
||||||
|
display_items = snapshot.human_required_items
|
||||||
|
active_tab_title = "Human-Required Escalations"
|
||||||
|
elif filter_class == ATTENTION_OPERATOR:
|
||||||
|
display_items = snapshot.operator_items
|
||||||
|
active_tab_title = "Operator Inbox Items"
|
||||||
|
elif filter_class == ATTENTION_ROUTINE:
|
||||||
|
display_items = snapshot.routine_items
|
||||||
|
active_tab_title = "Routine Workflow Transitions"
|
||||||
|
elif filter_class == "all":
|
||||||
|
display_items = snapshot.items
|
||||||
|
active_tab_title = "All Events (including Routine)"
|
||||||
|
else: # "inbox" default
|
||||||
|
display_items = snapshot.inbox_items
|
||||||
|
active_tab_title = "Attention Inbox (Human + Operator)"
|
||||||
|
|
||||||
|
hr_cls = "badge-blocked" if snapshot.human_required_count > 0 else "muted"
|
||||||
|
op_cls = "badge-claimed" if snapshot.operator_count > 0 else "muted"
|
||||||
|
|
||||||
|
metrics_html = f"""<div style="display:flex; gap:1rem; margin-bottom:1.5rem;">
|
||||||
|
<div class="health-card" style="flex:1;">
|
||||||
|
<span class="muted" style="font-size:0.85rem;">Human Required</span>
|
||||||
|
<h2 style="margin:0.2rem 0;"><span class="badge {hr_cls}" style="font-size:1.4rem;">{snapshot.human_required_count}</span></h2>
|
||||||
|
<p class="muted" style="font-size:0.8rem; margin:0;">Critical escalation boundary</p>
|
||||||
|
</div>
|
||||||
|
<div class="health-card" style="flex:1;">
|
||||||
|
<span class="muted" style="font-size:0.85rem;">Operator Inbox</span>
|
||||||
|
<h2 style="margin:0.2rem 0;"><span class="badge {op_cls}" style="font-size:1.4rem;">{snapshot.operator_count}</span></h2>
|
||||||
|
<p class="muted" style="font-size:0.8rem; margin:0;">Operational items needing review</p>
|
||||||
|
</div>
|
||||||
|
<div class="health-card" style="flex:1;">
|
||||||
|
<span class="muted" style="font-size:0.85rem;">Routine Transitions</span>
|
||||||
|
<h2 style="margin:0.2rem 0;"><span class="badge muted" style="font-size:1.4rem;">{snapshot.routine_count}</span></h2>
|
||||||
|
<p class="muted" style="font-size:0.8rem; margin:0;">Background transitions (filtered)</p>
|
||||||
|
</div>
|
||||||
|
</div>"""
|
||||||
|
|
||||||
|
# Filter navigation links
|
||||||
|
def _tab_link(target_class: str, label: str) -> str:
|
||||||
|
is_active = (filter_class == target_class)
|
||||||
|
style = "font-weight:bold; border-bottom:2px solid currentColor;" if is_active else "color:#4a5568;"
|
||||||
|
return f'<a href="/notifications?attention_class={target_class}" style="margin-right:1.25rem; text-decoration:none; padding-bottom:0.25rem; {style}">{label}</a>'
|
||||||
|
|
||||||
|
tabs_html = f"""<div style="margin-bottom:1.25rem; border-bottom:1px solid #e2e8f0; padding-bottom:0.5rem;">
|
||||||
|
{_tab_link("inbox", f"Attention Inbox ({snapshot.human_required_count + snapshot.operator_count})")}
|
||||||
|
{_tab_link("human-required", f"Human Required ({snapshot.human_required_count})")}
|
||||||
|
{_tab_link("operator", f"Operator ({snapshot.operator_count})")}
|
||||||
|
{_tab_link("routine", f"Routine ({snapshot.routine_count})")}
|
||||||
|
{_tab_link("all", f"All Events ({snapshot.total_count})")}
|
||||||
|
</div>"""
|
||||||
|
|
||||||
|
table_html = _render_notifications_table(
|
||||||
|
display_items,
|
||||||
|
f"No items match attention filter '{filter_class}'.",
|
||||||
|
)
|
||||||
|
|
||||||
|
body = f"""<h2>{escape(title)}</h2>
|
||||||
|
<p class="muted">Phase 3 console surface for human-attention routing (#648). Routine workflow transitions are filtered by default to eliminate notification fatigue.</p>
|
||||||
|
{err_html}
|
||||||
|
{metrics_html}
|
||||||
|
{tabs_html}
|
||||||
|
<h3>{escape(active_tab_title)}</h3>
|
||||||
|
{table_html}"""
|
||||||
|
|
||||||
|
return render_page(title=title, body_html=body)
|
||||||
@@ -0,0 +1,486 @@
|
|||||||
|
"""Notifications and human-attention routing module for Phase 3 web console (#648).
|
||||||
|
|
||||||
|
Defines attention classes, event classification rules, and inbox aggregation so
|
||||||
|
operators receive direct alerts only for human-required escalation boundaries
|
||||||
|
(#628) while routine workflow transitions remain available for pull-based review.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from typing import Any, Callable
|
||||||
|
|
||||||
|
from webui import console_redaction
|
||||||
|
from webui.project_registry import load_registry
|
||||||
|
from webui.queue_loader import QueueSnapshot, load_queue_snapshot
|
||||||
|
from webui.lease_loader import LeaseSnapshot, load_lease_snapshot
|
||||||
|
from webui.system_health import SystemHealthSnapshot, load_system_health
|
||||||
|
|
||||||
|
# Attention class definitions (#628, #648)
|
||||||
|
ATTENTION_ROUTINE = "routine"
|
||||||
|
ATTENTION_OPERATOR = "operator"
|
||||||
|
ATTENTION_HUMAN_REQUIRED = "human-required"
|
||||||
|
|
||||||
|
ATTENTION_CLASSES = (
|
||||||
|
ATTENTION_ROUTINE,
|
||||||
|
ATTENTION_OPERATOR,
|
||||||
|
ATTENTION_HUMAN_REQUIRED,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Notification categories
|
||||||
|
CATEGORY_AUTH = "auth"
|
||||||
|
CATEGORY_BLOCKER = "blocker"
|
||||||
|
CATEGORY_LEASE = "lease"
|
||||||
|
CATEGORY_VALIDATION = "validation"
|
||||||
|
CATEGORY_WORKFLOW = "workflow"
|
||||||
|
CATEGORY_SYSTEM = "system"
|
||||||
|
|
||||||
|
CATEGORIES = (
|
||||||
|
CATEGORY_AUTH,
|
||||||
|
CATEGORY_BLOCKER,
|
||||||
|
CATEGORY_LEASE,
|
||||||
|
CATEGORY_VALIDATION,
|
||||||
|
CATEGORY_WORKFLOW,
|
||||||
|
CATEGORY_SYSTEM,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class NotificationItem:
|
||||||
|
"""A single notification or inbox event."""
|
||||||
|
|
||||||
|
id: str
|
||||||
|
attention_class: str # "routine", "operator", "human-required"
|
||||||
|
category: str # "auth", "blocker", "lease", "validation", etc.
|
||||||
|
title: str
|
||||||
|
summary: str
|
||||||
|
work_kind: str | None # "issue", "pr", "session", "system"
|
||||||
|
work_number: int | None
|
||||||
|
project_id: str
|
||||||
|
repo_label: str
|
||||||
|
created_at: str
|
||||||
|
deep_link: str | None = None
|
||||||
|
requires_human: bool = False
|
||||||
|
extra: dict[str, Any] = field(default_factory=dict)
|
||||||
|
|
||||||
|
def as_dict(self) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"id": self.id,
|
||||||
|
"attention_class": self.attention_class,
|
||||||
|
"category": self.category,
|
||||||
|
"title": self.title,
|
||||||
|
"summary": console_redaction.redact_text(self.summary),
|
||||||
|
"work_kind": self.work_kind,
|
||||||
|
"work_number": self.work_number,
|
||||||
|
"project_id": self.project_id,
|
||||||
|
"repo_label": self.repo_label,
|
||||||
|
"created_at": self.created_at,
|
||||||
|
"deep_link": self.deep_link,
|
||||||
|
"requires_human": self.requires_human,
|
||||||
|
"extra": self.extra,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class NotificationSnapshot:
|
||||||
|
"""Snapshot of notifications and attention inbox state."""
|
||||||
|
|
||||||
|
project_id: str
|
||||||
|
repo_label: str
|
||||||
|
items: tuple[NotificationItem, ...]
|
||||||
|
human_required_count: int
|
||||||
|
operator_count: int
|
||||||
|
routine_count: int
|
||||||
|
total_count: int
|
||||||
|
fetch_error: str | None = None
|
||||||
|
|
||||||
|
@property
|
||||||
|
def inbox_items(self) -> tuple[NotificationItem, ...]:
|
||||||
|
"""Items requiring operator or human attention (excluding routine)."""
|
||||||
|
return tuple(
|
||||||
|
item
|
||||||
|
for item in self.items
|
||||||
|
if item.attention_class in {ATTENTION_OPERATOR, ATTENTION_HUMAN_REQUIRED}
|
||||||
|
)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def human_required_items(self) -> tuple[NotificationItem, ...]:
|
||||||
|
return tuple(
|
||||||
|
item for item in self.items if item.attention_class == ATTENTION_HUMAN_REQUIRED
|
||||||
|
)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def operator_items(self) -> tuple[NotificationItem, ...]:
|
||||||
|
return tuple(
|
||||||
|
item for item in self.items if item.attention_class == ATTENTION_OPERATOR
|
||||||
|
)
|
||||||
|
|
||||||
|
@property
|
||||||
|
def routine_items(self) -> tuple[NotificationItem, ...]:
|
||||||
|
return tuple(
|
||||||
|
item for item in self.items if item.attention_class == ATTENTION_ROUTINE
|
||||||
|
)
|
||||||
|
|
||||||
|
def as_dict(self) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"project_id": self.project_id,
|
||||||
|
"repo_label": self.repo_label,
|
||||||
|
"human_required_count": self.human_required_count,
|
||||||
|
"operator_count": self.operator_count,
|
||||||
|
"routine_count": self.routine_count,
|
||||||
|
"total_count": self.total_count,
|
||||||
|
"fetch_error": self.fetch_error,
|
||||||
|
"inbox_items": [item.as_dict() for item in self.inbox_items],
|
||||||
|
"all_items": [item.as_dict() for item in self.items],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def classify_attention_event(
|
||||||
|
category: str,
|
||||||
|
title: str,
|
||||||
|
summary: str,
|
||||||
|
*,
|
||||||
|
is_hard_stop: bool = False,
|
||||||
|
is_auth_failure: bool = False,
|
||||||
|
is_irrecoverable: bool = False,
|
||||||
|
is_decision_lock: bool = False,
|
||||||
|
is_validation_failure: bool = False,
|
||||||
|
is_stale: bool = False,
|
||||||
|
is_blocker: bool = False,
|
||||||
|
) -> tuple[str, bool]:
|
||||||
|
"""Classify an event into an attention class and human requirement flag.
|
||||||
|
|
||||||
|
Rules (#628, #648):
|
||||||
|
1. Critical boundaries (hard stop, auth failure, irrecoverable state,
|
||||||
|
decision lock, validation failure) -> ATTENTION_HUMAN_REQUIRED (requires_human=True).
|
||||||
|
2. Operational queues (blocker, stale lease, unassigned ready work, queue collision)
|
||||||
|
-> ATTENTION_OPERATOR (requires_human=False).
|
||||||
|
3. Routine state transitions (clean progression, healthy heartbeats) -> ATTENTION_ROUTINE (requires_human=False).
|
||||||
|
|
||||||
|
Classification uses structured flags and category only. Human-authored
|
||||||
|
``title`` / ``summary`` text is never substring-matched for escalation
|
||||||
|
(PR #905 review B1) — callers that need text signals must set flags from
|
||||||
|
machine-generated status/detail fields before calling this function.
|
||||||
|
"""
|
||||||
|
del title, summary # kept for API stability; never used for classification
|
||||||
|
if (
|
||||||
|
is_hard_stop
|
||||||
|
or is_auth_failure
|
||||||
|
or is_irrecoverable
|
||||||
|
or is_decision_lock
|
||||||
|
or is_validation_failure
|
||||||
|
or category in {CATEGORY_AUTH, CATEGORY_VALIDATION}
|
||||||
|
):
|
||||||
|
return ATTENTION_HUMAN_REQUIRED, True
|
||||||
|
|
||||||
|
if is_stale or is_blocker or category in {CATEGORY_BLOCKER, CATEGORY_LEASE}:
|
||||||
|
return ATTENTION_OPERATOR, False
|
||||||
|
|
||||||
|
return ATTENTION_ROUTINE, False
|
||||||
|
|
||||||
|
|
||||||
|
def load_notifications_snapshot(
|
||||||
|
project_id: str | None = None,
|
||||||
|
*,
|
||||||
|
load_queue: Callable[..., QueueSnapshot] | None = None,
|
||||||
|
load_leases: Callable[..., LeaseSnapshot] | None = None,
|
||||||
|
load_health: Callable[..., SystemHealthSnapshot] | None = None,
|
||||||
|
) -> NotificationSnapshot:
|
||||||
|
"""Load and classify attention notifications across queue, leases, and system health."""
|
||||||
|
registry = load_registry()
|
||||||
|
project = None
|
||||||
|
if project_id:
|
||||||
|
for entry in registry.projects:
|
||||||
|
if entry.id == project_id:
|
||||||
|
project = entry
|
||||||
|
break
|
||||||
|
else:
|
||||||
|
project = registry.projects[0] if registry.projects else None
|
||||||
|
|
||||||
|
if project is None:
|
||||||
|
return NotificationSnapshot(
|
||||||
|
project_id=project_id or "",
|
||||||
|
repo_label="",
|
||||||
|
items=(),
|
||||||
|
human_required_count=0,
|
||||||
|
operator_count=0,
|
||||||
|
routine_count=0,
|
||||||
|
total_count=0,
|
||||||
|
fetch_error="project not found in registry",
|
||||||
|
)
|
||||||
|
|
||||||
|
queue_loader_fn = load_queue or load_queue_snapshot
|
||||||
|
lease_loader_fn = load_leases or load_lease_snapshot
|
||||||
|
health_loader_fn = load_health or load_system_health
|
||||||
|
|
||||||
|
try:
|
||||||
|
queue_snap = queue_loader_fn(project.id)
|
||||||
|
except TypeError:
|
||||||
|
queue_snap = queue_loader_fn(project_id=project.id)
|
||||||
|
|
||||||
|
try:
|
||||||
|
lease_snap = lease_loader_fn(project_id=project.id)
|
||||||
|
except TypeError:
|
||||||
|
lease_snap = lease_loader_fn(project.id)
|
||||||
|
|
||||||
|
try:
|
||||||
|
health_snap = health_loader_fn(project_id=project.id)
|
||||||
|
except TypeError:
|
||||||
|
try:
|
||||||
|
health_snap = health_loader_fn(project.id)
|
||||||
|
except TypeError:
|
||||||
|
health_snap = health_loader_fn()
|
||||||
|
|
||||||
|
items: list[NotificationItem] = []
|
||||||
|
now_iso = datetime.now(timezone.utc).isoformat()
|
||||||
|
|
||||||
|
# 1. System health alerts (highest priority)
|
||||||
|
for err_idx, probe_err in enumerate(getattr(health_snap, "probe_errors", ())):
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_SYSTEM,
|
||||||
|
"System Health Probe Error",
|
||||||
|
probe_err,
|
||||||
|
is_blocker=True,
|
||||||
|
)
|
||||||
|
items.append(
|
||||||
|
NotificationItem(
|
||||||
|
id=f"notif-sys-err-{project.id}-{err_idx}",
|
||||||
|
attention_class=att_cls,
|
||||||
|
category=CATEGORY_SYSTEM,
|
||||||
|
title="System Health Error",
|
||||||
|
summary=f"System health error: {probe_err}",
|
||||||
|
work_kind="system",
|
||||||
|
work_number=None,
|
||||||
|
project_id=project.id,
|
||||||
|
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||||
|
created_at=now_iso,
|
||||||
|
deep_link="/system",
|
||||||
|
requires_human=req_human,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
for probe in getattr(health_snap, "dependencies", ()):
|
||||||
|
if probe.status not in ("ok", "healthy"):
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_SYSTEM,
|
||||||
|
f"Probe Failure: {probe.name}",
|
||||||
|
probe.detail or probe.status,
|
||||||
|
is_hard_stop=("stop" in probe.status or "fatal" in probe.status),
|
||||||
|
is_auth_failure=("auth" in probe.name.lower() or "unauthorized" in probe.status.lower()),
|
||||||
|
is_blocker=True,
|
||||||
|
)
|
||||||
|
items.append(
|
||||||
|
NotificationItem(
|
||||||
|
id=f"notif-probe-{probe.name}",
|
||||||
|
attention_class=att_cls,
|
||||||
|
category=CATEGORY_AUTH if "auth" in probe.name.lower() else CATEGORY_SYSTEM,
|
||||||
|
title=f"Health Probe Alert: {probe.name}",
|
||||||
|
summary=f"Probe '{probe.name}' reported status '{probe.status}': {probe.detail}",
|
||||||
|
work_kind="system",
|
||||||
|
work_number=None,
|
||||||
|
project_id=project.id,
|
||||||
|
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||||
|
created_at=now_iso,
|
||||||
|
deep_link="/system",
|
||||||
|
requires_human=req_human,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
# 2. Queue items (PRs and Issues)
|
||||||
|
for pr in queue_snap.prs:
|
||||||
|
if "blocked" in pr.badges:
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_BLOCKER,
|
||||||
|
f"PR #{pr.number} Blocked",
|
||||||
|
f"PR #{pr.number} '{pr.title}' is blocked or has merge conflicts.",
|
||||||
|
is_blocker=True,
|
||||||
|
)
|
||||||
|
items.append(
|
||||||
|
NotificationItem(
|
||||||
|
id=f"notif-pr-block-{pr.number}",
|
||||||
|
attention_class=att_cls,
|
||||||
|
category=CATEGORY_BLOCKER,
|
||||||
|
title=f"Blocked PR #{pr.number}",
|
||||||
|
summary=f"PR #{pr.number} ({pr.title}) requires merge conflict resolution.",
|
||||||
|
work_kind="pr",
|
||||||
|
work_number=pr.number,
|
||||||
|
project_id=project.id,
|
||||||
|
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||||
|
created_at=now_iso,
|
||||||
|
deep_link=f"/traffic",
|
||||||
|
requires_human=req_human,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
elif "stale" in pr.badges:
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_WORKFLOW,
|
||||||
|
f"PR #{pr.number} Stale",
|
||||||
|
f"PR #{pr.number} '{pr.title}' has had no activity for over 14 days.",
|
||||||
|
is_stale=True,
|
||||||
|
)
|
||||||
|
items.append(
|
||||||
|
NotificationItem(
|
||||||
|
id=f"notif-pr-stale-{pr.number}",
|
||||||
|
attention_class=att_cls,
|
||||||
|
category=CATEGORY_WORKFLOW,
|
||||||
|
title=f"Stale PR #{pr.number}",
|
||||||
|
summary=f"PR #{pr.number} ({pr.title}) is stale.",
|
||||||
|
work_kind="pr",
|
||||||
|
work_number=pr.number,
|
||||||
|
project_id=project.id,
|
||||||
|
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||||
|
created_at=now_iso,
|
||||||
|
deep_link=f"/queue",
|
||||||
|
requires_human=req_human,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
# Routine PR transition
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_WORKFLOW,
|
||||||
|
f"PR #{pr.number} Active",
|
||||||
|
f"PR #{pr.number} '{pr.title}' is in routine state {', '.join(pr.badges)}.",
|
||||||
|
)
|
||||||
|
items.append(
|
||||||
|
NotificationItem(
|
||||||
|
id=f"notif-pr-routine-{pr.number}",
|
||||||
|
attention_class=att_cls,
|
||||||
|
category=CATEGORY_WORKFLOW,
|
||||||
|
title=f"Routine PR #{pr.number}",
|
||||||
|
summary=f"PR #{pr.number} ({pr.title}) state: {', '.join(pr.badges)}.",
|
||||||
|
work_kind="pr",
|
||||||
|
work_number=pr.number,
|
||||||
|
project_id=project.id,
|
||||||
|
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||||
|
created_at=now_iso,
|
||||||
|
deep_link=f"/queue",
|
||||||
|
requires_human=req_human,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
for issue in queue_snap.issues:
|
||||||
|
if "duplicate" in issue.badges:
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_BLOCKER,
|
||||||
|
f"Issue #{issue.number} Duplicate PRs",
|
||||||
|
f"Issue #{issue.number} has multiple linked PRs.",
|
||||||
|
is_blocker=True,
|
||||||
|
)
|
||||||
|
items.append(
|
||||||
|
NotificationItem(
|
||||||
|
id=f"notif-issue-dup-{issue.number}",
|
||||||
|
attention_class=att_cls,
|
||||||
|
category=CATEGORY_BLOCKER,
|
||||||
|
title=f"Duplicate PRs on Issue #{issue.number}",
|
||||||
|
summary=f"Issue #{issue.number} ({issue.title}) linked to multiple PRs.",
|
||||||
|
work_kind="issue",
|
||||||
|
work_number=issue.number,
|
||||||
|
project_id=project.id,
|
||||||
|
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||||
|
created_at=now_iso,
|
||||||
|
deep_link=f"/traffic",
|
||||||
|
requires_human=req_human,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
elif "claimed" in issue.badges or "in-review" in issue.badges:
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_WORKFLOW,
|
||||||
|
f"Issue #{issue.number} Active",
|
||||||
|
f"Issue #{issue.number} '{issue.title}' in state {', '.join(issue.badges)}.",
|
||||||
|
)
|
||||||
|
items.append(
|
||||||
|
NotificationItem(
|
||||||
|
id=f"notif-issue-routine-{issue.number}",
|
||||||
|
attention_class=att_cls,
|
||||||
|
category=CATEGORY_WORKFLOW,
|
||||||
|
title=f"Routine Issue #{issue.number}",
|
||||||
|
summary=f"Issue #{issue.number} ({issue.title}) state: {', '.join(issue.badges)}.",
|
||||||
|
work_kind="issue",
|
||||||
|
work_number=issue.number,
|
||||||
|
project_id=project.id,
|
||||||
|
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||||
|
created_at=now_iso,
|
||||||
|
deep_link=f"/queue",
|
||||||
|
requires_human=req_human,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
# 3. Leases / Collisions
|
||||||
|
for lease in lease_snap.reviewer_leases:
|
||||||
|
if lease.get("is_expired") or lease.get("status") == "expired":
|
||||||
|
pr_num = lease.get("pr_number") or lease.get("work_item_number")
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_LEASE,
|
||||||
|
f"Reviewer Lease Expired for PR #{pr_num}",
|
||||||
|
f"Reviewer lease for PR #{pr_num} has expired.",
|
||||||
|
is_stale=True,
|
||||||
|
)
|
||||||
|
items.append(
|
||||||
|
NotificationItem(
|
||||||
|
id=f"notif-lease-exp-pr-{pr_num}",
|
||||||
|
attention_class=att_cls,
|
||||||
|
category=CATEGORY_LEASE,
|
||||||
|
title=f"Expired Reviewer Lease (PR #{pr_num})",
|
||||||
|
summary=f"Reviewer lease for PR #{pr_num} expired.",
|
||||||
|
work_kind="pr",
|
||||||
|
work_number=pr_num,
|
||||||
|
project_id=project.id,
|
||||||
|
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||||
|
created_at=now_iso,
|
||||||
|
deep_link="/leases",
|
||||||
|
requires_human=req_human,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
for col_idx, collision in enumerate(lease_snap.duplicate_prs):
|
||||||
|
att_cls, req_human = classify_attention_event(
|
||||||
|
CATEGORY_BLOCKER,
|
||||||
|
f"Duplicate PR Collision ({collision.kind})",
|
||||||
|
collision.message,
|
||||||
|
is_blocker=True,
|
||||||
|
)
|
||||||
|
issue_part = collision.issue_number if collision.issue_number is not None else "none"
|
||||||
|
kind_part = (collision.kind or "unknown").replace(" ", "-")
|
||||||
|
items.append(
|
||||||
|
NotificationItem(
|
||||||
|
id=f"notif-collision-{kind_part}-{issue_part}-{col_idx}",
|
||||||
|
attention_class=att_cls,
|
||||||
|
category=CATEGORY_BLOCKER,
|
||||||
|
title=f"Collision Alert ({collision.kind})",
|
||||||
|
summary=collision.message,
|
||||||
|
work_kind="issue" if collision.issue_number else "pr",
|
||||||
|
work_number=collision.issue_number,
|
||||||
|
project_id=project.id,
|
||||||
|
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||||
|
created_at=now_iso,
|
||||||
|
deep_link="/leases",
|
||||||
|
requires_human=req_human,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
human_req_count = sum(1 for i in items if i.attention_class == ATTENTION_HUMAN_REQUIRED)
|
||||||
|
operator_count = sum(1 for i in items if i.attention_class == ATTENTION_OPERATOR)
|
||||||
|
routine_count = sum(1 for i in items if i.attention_class == ATTENTION_ROUTINE)
|
||||||
|
|
||||||
|
# Fetch errors are transport/load failures only — not probe results that
|
||||||
|
# already surface as first-class notification items (PR #905 review B3).
|
||||||
|
fetch_err = queue_snap.fetch_error or lease_snap.fetch_error
|
||||||
|
if isinstance(fetch_err, (tuple, list)):
|
||||||
|
fetch_err = "; ".join(fetch_err) if fetch_err else None
|
||||||
|
|
||||||
|
return NotificationSnapshot(
|
||||||
|
project_id=project.id,
|
||||||
|
repo_label=f"{project.gitea_owner}/{project.repo_name}",
|
||||||
|
items=tuple(items),
|
||||||
|
human_required_count=human_req_count,
|
||||||
|
operator_count=operator_count,
|
||||||
|
routine_count=routine_count,
|
||||||
|
total_count=len(items),
|
||||||
|
fetch_error=fetch_err,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def snapshot_to_dict(snapshot: NotificationSnapshot) -> dict[str, Any]:
|
||||||
|
"""JSON-serializable export for /api/v1/notifications."""
|
||||||
|
return snapshot.as_dict()
|
||||||
@@ -0,0 +1,275 @@
|
|||||||
|
"""Sentry/GlitchTip observability and incident correlation loader for the console (#649, Phase 4).
|
||||||
|
|
||||||
|
Operators need to inspect provider connection status (Sentry/GlitchTip), error
|
||||||
|
correlations, and durable Gitea issue linkage — without treating raw incidents
|
||||||
|
as allocator work.
|
||||||
|
|
||||||
|
ADR authority model:
|
||||||
|
* Gitea owns work.
|
||||||
|
* Providers (Sentry/GlitchTip) observe incidents.
|
||||||
|
* Control-plane DB coordinates incident links.
|
||||||
|
* The #612 bridge reconciles observations into durable Gitea issues.
|
||||||
|
* The web console projects read-only state and gates mutations.
|
||||||
|
|
||||||
|
Redaction boundary:
|
||||||
|
* Provider auth tokens, DSNs, Authorization headers, and sensitive local file
|
||||||
|
paths are ALWAYS redacted before leaving this module.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
from dataclasses import dataclass
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from control_plane_db import ControlPlaneDB
|
||||||
|
import sentry_incident_bridge
|
||||||
|
from webui import console_redaction
|
||||||
|
|
||||||
|
OBSERVABILITY_SCHEMA_VERSION = 1
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ProviderHealth:
|
||||||
|
"""Connection and health status of an observability provider."""
|
||||||
|
|
||||||
|
provider: str
|
||||||
|
base_url: str
|
||||||
|
org: str
|
||||||
|
project: str
|
||||||
|
configured: bool
|
||||||
|
status: str
|
||||||
|
bridge_enabled: bool
|
||||||
|
lookback: str
|
||||||
|
min_events_for_issue: int
|
||||||
|
self_hosted: bool
|
||||||
|
environment: str | None = None
|
||||||
|
credentials_present: bool = False
|
||||||
|
|
||||||
|
def to_dict(self) -> dict[str, Any]:
|
||||||
|
data = {
|
||||||
|
"provider": self.provider,
|
||||||
|
"base_url": self.base_url,
|
||||||
|
"org": self.org,
|
||||||
|
"project": self.project,
|
||||||
|
"configured": self.configured,
|
||||||
|
"status": self.status,
|
||||||
|
"bridge_enabled": self.bridge_enabled,
|
||||||
|
"lookback": self.lookback,
|
||||||
|
"min_events_for_issue": self.min_events_for_issue,
|
||||||
|
"self_hosted": self.self_hosted,
|
||||||
|
"environment": self.environment,
|
||||||
|
"credentials_present": self.credentials_present,
|
||||||
|
}
|
||||||
|
return console_redaction.redact_payload(data)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class CorrelatedIncidentLink:
|
||||||
|
"""One linked provider incident ↔ Gitea issue correlation record."""
|
||||||
|
|
||||||
|
link_id: int
|
||||||
|
provider: str
|
||||||
|
provider_base_url: str
|
||||||
|
provider_org: str
|
||||||
|
provider_project: str
|
||||||
|
provider_issue_id: str
|
||||||
|
provider_short_id: str | None
|
||||||
|
provider_permalink: str | None
|
||||||
|
fingerprint: str | None
|
||||||
|
gitea_org: str
|
||||||
|
gitea_repo: str
|
||||||
|
gitea_issue_number: int
|
||||||
|
linked_pr_numbers: list[int]
|
||||||
|
last_seen: str | None
|
||||||
|
event_count: int
|
||||||
|
created_at: str | None
|
||||||
|
updated_at: str | None
|
||||||
|
|
||||||
|
def to_dict(self) -> dict[str, Any]:
|
||||||
|
data = {
|
||||||
|
"link_id": self.link_id,
|
||||||
|
"provider": self.provider,
|
||||||
|
"provider_base_url": self.provider_base_url,
|
||||||
|
"provider_org": self.provider_org,
|
||||||
|
"provider_project": self.provider_project,
|
||||||
|
"provider_issue_id": self.provider_issue_id,
|
||||||
|
"provider_short_id": self.provider_short_id,
|
||||||
|
"provider_permalink": self.provider_permalink,
|
||||||
|
"fingerprint": self.fingerprint,
|
||||||
|
"gitea_org": self.gitea_org,
|
||||||
|
"gitea_repo": self.gitea_repo,
|
||||||
|
"gitea_issue_number": self.gitea_issue_number,
|
||||||
|
"linked_pr_numbers": self.linked_pr_numbers,
|
||||||
|
"last_seen": self.last_seen,
|
||||||
|
"event_count": self.event_count,
|
||||||
|
"created_at": self.created_at,
|
||||||
|
"updated_at": self.updated_at,
|
||||||
|
}
|
||||||
|
return console_redaction.redact_payload(data)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ObservabilitySnapshot:
|
||||||
|
"""Read-only snapshot of observability provider status and incident correlations."""
|
||||||
|
|
||||||
|
schema_version: int
|
||||||
|
providers: list[ProviderHealth]
|
||||||
|
links: list[CorrelatedIncidentLink]
|
||||||
|
total_links: int
|
||||||
|
sentry_links_count: int
|
||||||
|
glitchtip_links_count: int
|
||||||
|
bridge_active: bool
|
||||||
|
|
||||||
|
def to_dict(self) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"schema_version": self.schema_version,
|
||||||
|
"providers": [p.to_dict() for p in self.providers],
|
||||||
|
"links": [link.to_dict() for link in self.links],
|
||||||
|
"metrics": {
|
||||||
|
"total_links": self.total_links,
|
||||||
|
"sentry_links_count": self.sentry_links_count,
|
||||||
|
"glitchtip_links_count": self.glitchtip_links_count,
|
||||||
|
"bridge_active": self.bridge_active,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def load_provider_health(
|
||||||
|
provider_name: str = "sentry",
|
||||||
|
env: dict[str, str] | None = None,
|
||||||
|
) -> ProviderHealth:
|
||||||
|
"""Inspect configuration and connection health for an observability provider."""
|
||||||
|
source_env = dict(env if env is not None else os.environ)
|
||||||
|
if provider_name.lower() == "sentry":
|
||||||
|
config = sentry_incident_bridge.load_bridge_config(source_env)
|
||||||
|
token = sentry_incident_bridge.resolve_token(source_env)
|
||||||
|
has_token = bool(token)
|
||||||
|
configured = bool(config.org and config.project and has_token)
|
||||||
|
|
||||||
|
if not config.org or not config.project:
|
||||||
|
status = "not_configured"
|
||||||
|
elif not has_token:
|
||||||
|
status = "missing_token"
|
||||||
|
elif not config.bridge_enabled:
|
||||||
|
status = "disabled"
|
||||||
|
else:
|
||||||
|
status = "healthy"
|
||||||
|
|
||||||
|
return ProviderHealth(
|
||||||
|
provider="sentry",
|
||||||
|
base_url=config.base_url,
|
||||||
|
org=config.org or "unconfigured",
|
||||||
|
project=config.project or "unconfigured",
|
||||||
|
configured=configured,
|
||||||
|
status=status,
|
||||||
|
bridge_enabled=config.bridge_enabled,
|
||||||
|
lookback=config.lookback,
|
||||||
|
min_events_for_issue=config.min_events_for_issue,
|
||||||
|
self_hosted=not config.base_url.rstrip("/").endswith("sentry.io"),
|
||||||
|
environment=config.environment,
|
||||||
|
credentials_present=has_token,
|
||||||
|
)
|
||||||
|
|
||||||
|
# GlitchTip or fallback provider configuration
|
||||||
|
glitchtip_url = (source_env.get("GLITCHTIP_BASE_URL") or "https://glitchtip.prgs.cc").strip()
|
||||||
|
glitchtip_org = (source_env.get("GLITCHTIP_ORG") or "").strip()
|
||||||
|
glitchtip_proj = (source_env.get("GLITCHTIP_PROJECT") or "").strip()
|
||||||
|
glitchtip_token = (source_env.get("GLITCHTIP_AUTH_TOKEN") or "").strip()
|
||||||
|
|
||||||
|
has_token = bool(glitchtip_token)
|
||||||
|
configured = bool(glitchtip_org and glitchtip_proj and has_token)
|
||||||
|
status = "healthy" if configured else ("missing_token" if glitchtip_org and glitchtip_proj else "not_configured")
|
||||||
|
|
||||||
|
return ProviderHealth(
|
||||||
|
provider="glitchtip",
|
||||||
|
base_url=glitchtip_url,
|
||||||
|
org=glitchtip_org or "unconfigured",
|
||||||
|
project=glitchtip_proj or "unconfigured",
|
||||||
|
configured=configured,
|
||||||
|
status=status,
|
||||||
|
bridge_enabled=configured,
|
||||||
|
lookback="24h",
|
||||||
|
min_events_for_issue=2,
|
||||||
|
self_hosted=True,
|
||||||
|
environment=source_env.get("GLITCHTIP_ENVIRONMENT"),
|
||||||
|
credentials_present=has_token,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_pr_numbers(raw: Any) -> list[int]:
|
||||||
|
if isinstance(raw, list):
|
||||||
|
return [int(x) for x in raw if str(x).isdigit()]
|
||||||
|
if isinstance(raw, str) and raw.strip():
|
||||||
|
import json
|
||||||
|
try:
|
||||||
|
parsed = json.loads(raw)
|
||||||
|
if isinstance(parsed, list):
|
||||||
|
return [int(x) for x in parsed if str(x).isdigit()]
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
def load_observability_snapshot(
|
||||||
|
db: ControlPlaneDB | None = None,
|
||||||
|
env: dict[str, str] | None = None,
|
||||||
|
) -> ObservabilitySnapshot:
|
||||||
|
"""Build a read-only snapshot of observability connection health and incident links."""
|
||||||
|
sentry_health = load_provider_health("sentry", env)
|
||||||
|
glitchtip_health = load_provider_health("glitchtip", env)
|
||||||
|
providers = [sentry_health, glitchtip_health]
|
||||||
|
|
||||||
|
target_db = db or ControlPlaneDB()
|
||||||
|
raw_links = target_db.list_incident_links(limit=100)
|
||||||
|
|
||||||
|
links: list[CorrelatedIncidentLink] = []
|
||||||
|
sentry_cnt = 0
|
||||||
|
glitchtip_cnt = 0
|
||||||
|
|
||||||
|
for r in raw_links:
|
||||||
|
prov = (r.get("provider") or "sentry").lower()
|
||||||
|
if prov == "sentry":
|
||||||
|
sentry_cnt += 1
|
||||||
|
elif prov == "glitchtip":
|
||||||
|
glitchtip_cnt += 1
|
||||||
|
|
||||||
|
pr_nums = _parse_pr_numbers(r.get("linked_pr_numbers"))
|
||||||
|
|
||||||
|
links.append(
|
||||||
|
CorrelatedIncidentLink(
|
||||||
|
link_id=int(r.get("link_id", 0)),
|
||||||
|
provider=prov,
|
||||||
|
provider_base_url=r.get("provider_base_url") or "",
|
||||||
|
provider_org=r.get("provider_org") or "",
|
||||||
|
provider_project=r.get("provider_project") or "",
|
||||||
|
provider_issue_id=str(r.get("provider_issue_id") or ""),
|
||||||
|
provider_short_id=r.get("provider_short_id"),
|
||||||
|
provider_permalink=r.get("provider_permalink"),
|
||||||
|
fingerprint=r.get("fingerprint"),
|
||||||
|
gitea_org=r.get("gitea_org") or "Scaled-Tech-Consulting",
|
||||||
|
gitea_repo=r.get("gitea_repo") or "Gitea-Tools",
|
||||||
|
gitea_issue_number=int(r.get("gitea_issue_number", 0)),
|
||||||
|
linked_pr_numbers=pr_nums,
|
||||||
|
last_seen=r.get("last_seen"),
|
||||||
|
event_count=int(r.get("event_count", 1)),
|
||||||
|
created_at=r.get("created_at"),
|
||||||
|
updated_at=r.get("updated_at"),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
bridge_active = any(p.bridge_enabled for p in providers)
|
||||||
|
|
||||||
|
return ObservabilitySnapshot(
|
||||||
|
schema_version=OBSERVABILITY_SCHEMA_VERSION,
|
||||||
|
providers=providers,
|
||||||
|
links=links,
|
||||||
|
total_links=len(links),
|
||||||
|
sentry_links_count=sentry_cnt,
|
||||||
|
glitchtip_links_count=glitchtip_cnt,
|
||||||
|
bridge_active=bridge_active,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def snapshot_to_dict(snapshot: ObservabilitySnapshot) -> dict[str, Any]:
|
||||||
|
return snapshot.to_dict()
|
||||||
@@ -0,0 +1,143 @@
|
|||||||
|
"""HTML view renderer for the Sentry/GlitchTip observability console (#649, Phase 4).
|
||||||
|
|
||||||
|
Renders connection status widgets, error correlation links, and gated issue creation
|
||||||
|
affordances over the read-only observability snapshot.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import html
|
||||||
|
from typing import Any
|
||||||
|
|
||||||
|
from webui.layout import render_page
|
||||||
|
from webui.observability_loader import ObservabilitySnapshot, snapshot_to_dict
|
||||||
|
|
||||||
|
|
||||||
|
def _badge(status: str) -> str:
|
||||||
|
st = (status or "").lower()
|
||||||
|
if st == "healthy":
|
||||||
|
return '<span class="badge badge-success">healthy</span>'
|
||||||
|
if st == "disabled":
|
||||||
|
return '<span class="badge badge-warning">disabled (dry-run)</span>'
|
||||||
|
if st in {"missing_token", "not_configured"}:
|
||||||
|
return f'<span class="badge badge-muted">{html.escape(st)}</span>'
|
||||||
|
return f'<span class="badge">{html.escape(st)}</span>'
|
||||||
|
|
||||||
|
|
||||||
|
def _provider_card(p: dict[str, Any]) -> str:
|
||||||
|
name = html.escape(str(p.get("provider", "provider")).upper())
|
||||||
|
base_url = html.escape(str(p.get("base_url", "")))
|
||||||
|
org = html.escape(str(p.get("org", "")))
|
||||||
|
proj = html.escape(str(p.get("project", "")))
|
||||||
|
status_badge = _badge(str(p.get("status", "")))
|
||||||
|
min_events = p.get("min_events_for_issue", 2)
|
||||||
|
lookback = html.escape(str(p.get("lookback", "24h")))
|
||||||
|
bridge_enabled = "yes" if p.get("bridge_enabled") else "no"
|
||||||
|
|
||||||
|
return f"""
|
||||||
|
<div class="card" style="margin-bottom: 1rem; padding: 1rem; border: 1px solid #ccc; border-radius: 6px;">
|
||||||
|
<div style="display: flex; justify-content: space-between; align-items: center;">
|
||||||
|
<h3 style="margin: 0;">{name} Connection</h3>
|
||||||
|
<div>{status_badge}</div>
|
||||||
|
</div>
|
||||||
|
<table style="width: 100%; margin-top: 0.5rem; border-collapse: collapse;">
|
||||||
|
<tr><td><strong>Base URL:</strong></td><td><code>{base_url}</code></td></tr>
|
||||||
|
<tr><td><strong>Scope:</strong></td><td><code>{org} / {proj}</code></td></tr>
|
||||||
|
<tr><td><strong>Bridge Enabled:</strong></td><td><code>{bridge_enabled}</code></td></tr>
|
||||||
|
<tr><td><strong>Min Events for Issue:</strong></td><td><code>{min_events}</code></td></tr>
|
||||||
|
<tr><td><strong>Lookback Window:</strong></td><td><code>{lookback}</code></td></tr>
|
||||||
|
</table>
|
||||||
|
</div>
|
||||||
|
"""
|
||||||
|
|
||||||
|
|
||||||
|
def render_observability_page(snapshot: ObservabilitySnapshot | dict[str, Any]) -> str:
|
||||||
|
"""Render the observability dashboard HTML page."""
|
||||||
|
data = snapshot.to_dict() if isinstance(snapshot, ObservabilitySnapshot) else dict(snapshot)
|
||||||
|
|
||||||
|
providers_raw = data.get("providers", [])
|
||||||
|
provider_cards = "".join(_provider_card(p) for p in providers_raw) if providers_raw else "<p>No providers configured.</p>"
|
||||||
|
|
||||||
|
links = data.get("links", [])
|
||||||
|
link_rows = []
|
||||||
|
|
||||||
|
for l in links:
|
||||||
|
prov = html.escape(str(l.get("provider", "")))
|
||||||
|
p_issue_id = html.escape(str(l.get("provider_issue_id", "")))
|
||||||
|
fingerprint = html.escape(str(l.get("fingerprint") or "—"))
|
||||||
|
g_issue_num = int(l.get("gitea_issue_number", 0))
|
||||||
|
g_org = html.escape(str(l.get("gitea_org", "")))
|
||||||
|
g_repo = html.escape(str(l.get("gitea_repo", "")))
|
||||||
|
g_issue_link = f'<strong>#{g_issue_num}</strong> ({g_org}/{g_repo})'
|
||||||
|
event_cnt = int(l.get("event_count", 1))
|
||||||
|
last_seen = html.escape(str(l.get("last_seen") or "—"))
|
||||||
|
short_id = html.escape(str(l.get("provider_short_id") or p_issue_id))
|
||||||
|
|
||||||
|
link_rows.append(f"""
|
||||||
|
<tr>
|
||||||
|
<td><code>{prov}</code></td>
|
||||||
|
<td><strong>{short_id}</strong><br><small style="color: #666;">id: {p_issue_id}</small></td>
|
||||||
|
<td><code>{fingerprint}</code></td>
|
||||||
|
<td>{g_issue_link}</td>
|
||||||
|
<td>{event_cnt}</td>
|
||||||
|
<td><small>{last_seen}</small></td>
|
||||||
|
</tr>
|
||||||
|
""")
|
||||||
|
|
||||||
|
table_body = "".join(link_rows) if link_rows else '<tr><td colspan="6" style="text-align: center; padding: 1.5rem; color: #666;">No correlated incident links stored. Bridge operates under dry-run default.</td></tr>'
|
||||||
|
|
||||||
|
metrics = data.get("metrics", {})
|
||||||
|
total_links = metrics.get("total_links", 0)
|
||||||
|
sentry_cnt = metrics.get("sentry_links_count", 0)
|
||||||
|
glitchtip_cnt = metrics.get("glitchtip_links_count", 0)
|
||||||
|
|
||||||
|
body_html = f"""
|
||||||
|
<h2>Observability & Incident Bridge (#649)</h2>
|
||||||
|
<p>Read-only console surface for Sentry/GlitchTip provider connections, error correlation,
|
||||||
|
and durable Gitea issue linkage.</p>
|
||||||
|
|
||||||
|
<div class="alert alert-info" style="background: #f0f4f8; padding: 1rem; border-left: 4px solid #0052cc; margin-bottom: 1.5rem;">
|
||||||
|
<strong>ADR Authority Model:</strong> Gitea records durable issue history. Control-plane DB coordinates incident links.
|
||||||
|
Sentry/GlitchTip observe errors. Raw monitoring incidents are <em>never</em> assignable control-plane work items.
|
||||||
|
Durable issue creation is gated and dry-runable via the <code>#612</code> bridge APIs.
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<h3>Provider Connections</h3>
|
||||||
|
<div style="display: grid; grid-template-columns: repeat(auto-fit, minmax(300px, 1fr)); gap: 1rem; margin-bottom: 2rem;">
|
||||||
|
{provider_cards}
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<div style="display: flex; justify-content: space-between; align-items: center; margin-bottom: 1rem;">
|
||||||
|
<h3 style="margin: 0;">Correlated Incidents ({total_links})</h3>
|
||||||
|
<div>
|
||||||
|
<span class="badge" style="margin-right: 0.5rem;">Sentry: {sentry_cnt}</span>
|
||||||
|
<span class="badge">GlitchTip: {glitchtip_cnt}</span>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<table class="table" style="width: 100%; border-collapse: collapse; border: 1px solid #ddd;">
|
||||||
|
<thead>
|
||||||
|
<tr style="background: #f9f9f9; text-align: left;">
|
||||||
|
<th style="padding: 0.5rem; border-bottom: 2px solid #ddd;">Provider</th>
|
||||||
|
<th style="padding: 0.5rem; border-bottom: 2px solid #ddd;">Incident ID</th>
|
||||||
|
<th style="padding: 0.5rem; border-bottom: 2px solid #ddd;">Fingerprint</th>
|
||||||
|
<th style="padding: 0.5rem; border-bottom: 2px solid #ddd;">Gitea Issue Link</th>
|
||||||
|
<th style="padding: 0.5rem; border-bottom: 2px solid #ddd;">Events</th>
|
||||||
|
<th style="padding: 0.5rem; border-bottom: 2px solid #ddd;">Last Seen</th>
|
||||||
|
</tr>
|
||||||
|
</thead>
|
||||||
|
<tbody>
|
||||||
|
{table_body}
|
||||||
|
</tbody>
|
||||||
|
</table>
|
||||||
|
|
||||||
|
<div style="margin-top: 2rem; padding: 1rem; background: #fafafa; border: 1px solid #eee; border-radius: 4px;">
|
||||||
|
<h4 style="margin-top: 0;">Reconcile & Link Controls (Gated)</h4>
|
||||||
|
<p style="margin-bottom: 0.5rem; color: #555;">
|
||||||
|
Create or reconcile durable Gitea issues from provider observations using the <code>#612</code> incident bridge:
|
||||||
|
</p>
|
||||||
|
<code>mcp call gitea_observability_reconcile_incident --provider sentry --apply false</code>
|
||||||
|
</div>
|
||||||
|
"""
|
||||||
|
|
||||||
|
return render_page(title="Observability", body_html=body_html)
|
||||||
@@ -0,0 +1,579 @@
|
|||||||
|
"""Read-only restart status, impact preview, and approval state (#667).
|
||||||
|
|
||||||
|
Phase 1 of the console restart surface. It *consumes* the #655 coordinator
|
||||||
|
substrate and renders it; it never restarts, reloads, drains, approves, or kills
|
||||||
|
anything. There is no apply path in this module, so there is no execution gate
|
||||||
|
here to arm incorrectly — the only writes the console could perform are the ones
|
||||||
|
it does not implement.
|
||||||
|
|
||||||
|
Sources, each independently fail-soft and each reported with its own
|
||||||
|
:class:`SourceStatus`:
|
||||||
|
|
||||||
|
* :mod:`restart_coordinator` — restart-class policy matrix (#663) and the
|
||||||
|
blast-radius impact report (#658).
|
||||||
|
* :mod:`drain_proof` — drain checklist and gate verdict (#661), verified
|
||||||
|
read-only against a caller-supplied proof.
|
||||||
|
* :mod:`post_restart_reconcile` — post-restart completion proof (#662).
|
||||||
|
* :mod:`webui.console_authz` — role authorization for the approval controls
|
||||||
|
(#633).
|
||||||
|
|
||||||
|
Three rules this module holds itself to, because a status surface that lies is
|
||||||
|
worse than one that is absent:
|
||||||
|
|
||||||
|
**A source that could not be read is reported unavailable, never green.** No
|
||||||
|
default, placeholder, or self-comparison is substituted for a reading that
|
||||||
|
failed. An unreadable control-plane DB yields ``inventory_complete=False``,
|
||||||
|
which the coordinator itself turns into a fail-closed verdict.
|
||||||
|
|
||||||
|
**Authorization is asked the way execution would ask it.** Every authorization
|
||||||
|
probe passes ``for_execution=True``, so the console reports whether the action
|
||||||
|
could actually run rather than the weaker "this principal is the right role".
|
||||||
|
While the console is in Phase 1 that answer is ``phase_not_active`` for every
|
||||||
|
phase-2 action, and the surface says so plainly instead of showing an allow.
|
||||||
|
|
||||||
|
**The database is opened read-only.** ``ControlPlaneDB()`` creates directories
|
||||||
|
and runs migrations on construction, which is a write; this module opens the
|
||||||
|
sqlite file with ``mode=ro`` exactly as :mod:`webui.inventory` does, and treats
|
||||||
|
a missing file as missing authority rather than an empty inventory.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
|
import sqlite3
|
||||||
|
from dataclasses import dataclass, field
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from typing import Any, Callable, Mapping
|
||||||
|
|
||||||
|
import control_plane_db
|
||||||
|
import drain_proof
|
||||||
|
import restart_coordinator
|
||||||
|
from webui import console_authz
|
||||||
|
from webui.inventory import redact_path, scrub
|
||||||
|
|
||||||
|
# --- Source status ----------------------------------------------------------
|
||||||
|
|
||||||
|
STATUS_OK = "ok"
|
||||||
|
STATUS_UNAVAILABLE = "unavailable"
|
||||||
|
|
||||||
|
#: Console actions whose authorization state this surface reports. Both are
|
||||||
|
#: pre-existing #642 actions; this module adds no new console action because it
|
||||||
|
#: performs no console action.
|
||||||
|
REPORTED_ACTIONS: tuple[str, ...] = (
|
||||||
|
"system.restart_namespace",
|
||||||
|
"system.reload_namespace",
|
||||||
|
)
|
||||||
|
|
||||||
|
#: The break-glass workflow (#664) is not consumed here. It is declared so the
|
||||||
|
#: surface is honest about the gap rather than silently omitting a governance
|
||||||
|
#: path the operator has been told exists.
|
||||||
|
BREAK_GLASS_ISSUE = 664
|
||||||
|
BREAK_GLASS_PENDING_REASON = (
|
||||||
|
"The break-glass workflow (#664) is not yet available on this branch's "
|
||||||
|
"base; no break-glass control is offered and none is implied."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class SourceStatus:
|
||||||
|
"""Whether one backing source could be read, and why not when it could not."""
|
||||||
|
|
||||||
|
name: str
|
||||||
|
status: str
|
||||||
|
detail: str = ""
|
||||||
|
|
||||||
|
@property
|
||||||
|
def available(self) -> bool:
|
||||||
|
return self.status == STATUS_OK
|
||||||
|
|
||||||
|
def as_dict(self) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"name": self.name,
|
||||||
|
"status": self.status,
|
||||||
|
"available": self.available,
|
||||||
|
"detail": self.detail,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class RestartClassView:
|
||||||
|
"""One row of the #663 restart-class matrix, scoped to the viewer's role."""
|
||||||
|
|
||||||
|
restart_class: str
|
||||||
|
required_permission: str
|
||||||
|
expected_blast_radius: str
|
||||||
|
drain_requirement: str
|
||||||
|
full_drain_required: bool
|
||||||
|
approval_requirement: str
|
||||||
|
request_roles: tuple[str, ...]
|
||||||
|
execution_roles: tuple[str, ...]
|
||||||
|
viewer_may_request: bool
|
||||||
|
viewer_may_execute: bool
|
||||||
|
|
||||||
|
def as_dict(self) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"restart_class": self.restart_class,
|
||||||
|
"required_permission": self.required_permission,
|
||||||
|
"expected_blast_radius": self.expected_blast_radius,
|
||||||
|
"drain_requirement": self.drain_requirement,
|
||||||
|
"full_drain_required": self.full_drain_required,
|
||||||
|
"approval_requirement": self.approval_requirement,
|
||||||
|
"request_roles": list(self.request_roles),
|
||||||
|
"execution_roles": list(self.execution_roles),
|
||||||
|
"viewer_may_request": self.viewer_may_request,
|
||||||
|
"viewer_may_execute": self.viewer_may_execute,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class ActionAuthorization:
|
||||||
|
"""Authorization state for one console action, asked as execution would."""
|
||||||
|
|
||||||
|
action_id: str
|
||||||
|
summary: str
|
||||||
|
required_role: str
|
||||||
|
allowed: bool
|
||||||
|
execution_enabled: bool
|
||||||
|
reason_code: str
|
||||||
|
detail: str
|
||||||
|
|
||||||
|
def as_dict(self) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"action_id": self.action_id,
|
||||||
|
"summary": self.summary,
|
||||||
|
"required_role": self.required_role,
|
||||||
|
"allowed": self.allowed,
|
||||||
|
"execution_enabled": self.execution_enabled,
|
||||||
|
"reason_code": self.reason_code,
|
||||||
|
"detail": self.detail,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class BreakGlassSurface:
|
||||||
|
"""Declared-but-unavailable break-glass panel (#664 is not on this base)."""
|
||||||
|
|
||||||
|
available: bool
|
||||||
|
issue: int
|
||||||
|
reason: str
|
||||||
|
viewer_is_privileged: bool
|
||||||
|
|
||||||
|
def as_dict(self) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"available": self.available,
|
||||||
|
"issue": self.issue,
|
||||||
|
"reason": self.reason,
|
||||||
|
"viewer_is_privileged": self.viewer_is_privileged,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class RestartConsoleSnapshot:
|
||||||
|
"""Everything the read-only restart console renders."""
|
||||||
|
|
||||||
|
generated_at: str
|
||||||
|
viewer_role: str
|
||||||
|
viewer_authenticated: bool
|
||||||
|
read_only: bool
|
||||||
|
impact: dict[str, Any] | None
|
||||||
|
impact_source: SourceStatus
|
||||||
|
drain: dict[str, Any] | None
|
||||||
|
drain_source: SourceStatus
|
||||||
|
reconcile: dict[str, Any] | None
|
||||||
|
reconcile_source: SourceStatus
|
||||||
|
restart_classes: tuple[RestartClassView, ...]
|
||||||
|
authorizations: tuple[ActionAuthorization, ...]
|
||||||
|
break_glass: BreakGlassSurface
|
||||||
|
notes: tuple[str, ...] = field(default_factory=tuple)
|
||||||
|
|
||||||
|
def as_dict(self) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"generated_at": self.generated_at,
|
||||||
|
"viewer_role": self.viewer_role,
|
||||||
|
"viewer_authenticated": self.viewer_authenticated,
|
||||||
|
"read_only": self.read_only,
|
||||||
|
"impact": self.impact,
|
||||||
|
"impact_source": self.impact_source.as_dict(),
|
||||||
|
"drain": self.drain,
|
||||||
|
"drain_source": self.drain_source.as_dict(),
|
||||||
|
"reconcile": self.reconcile,
|
||||||
|
"reconcile_source": self.reconcile_source.as_dict(),
|
||||||
|
"restart_classes": [c.as_dict() for c in self.restart_classes],
|
||||||
|
"authorizations": [a.as_dict() for a in self.authorizations],
|
||||||
|
"break_glass": self.break_glass.as_dict(),
|
||||||
|
"notes": list(self.notes),
|
||||||
|
"links": {
|
||||||
|
"issue": 667,
|
||||||
|
"extends": 642,
|
||||||
|
"umbrella": 655,
|
||||||
|
"coordinator": 658,
|
||||||
|
"drain_proof": 661,
|
||||||
|
"reconcile": 662,
|
||||||
|
"restart_classes": 663,
|
||||||
|
"break_glass": BREAK_GLASS_ISSUE,
|
||||||
|
"vision": 652,
|
||||||
|
"roadmap": 653,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _utc_now() -> datetime:
|
||||||
|
return datetime.now(timezone.utc)
|
||||||
|
|
||||||
|
|
||||||
|
# --- Control-plane inventory (read-only) ------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def read_control_plane_inventory(
|
||||||
|
*,
|
||||||
|
db_path: str | None = None,
|
||||||
|
limit: int = 200,
|
||||||
|
) -> dict[str, Any]:
|
||||||
|
"""Read sessions and leases for an impact evaluation, read-only.
|
||||||
|
|
||||||
|
Returns the inventory mapping
|
||||||
|
:func:`restart_coordinator.evaluate_restart_impact` expects.
|
||||||
|
``inventory_complete`` is True only when every read succeeded, so a partial
|
||||||
|
read denies rather than under-reporting the blast radius.
|
||||||
|
|
||||||
|
The database is never created, migrated, or written: a missing file means
|
||||||
|
the console has no session authority, which is not the same as there being
|
||||||
|
no sessions.
|
||||||
|
"""
|
||||||
|
|
||||||
|
path = (db_path or control_plane_db.default_db_path() or "").strip()
|
||||||
|
incomplete: list[str] = []
|
||||||
|
|
||||||
|
def _incomplete(reason: str) -> dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"sessions": [],
|
||||||
|
"leases": [],
|
||||||
|
"terminal_lock": None,
|
||||||
|
"prior_recovery_attempts": [],
|
||||||
|
"inventory_complete": False,
|
||||||
|
"incomplete_reasons": [reason],
|
||||||
|
}
|
||||||
|
|
||||||
|
if not path:
|
||||||
|
return _incomplete("control-plane database path is not configured")
|
||||||
|
if not os.path.exists(path):
|
||||||
|
return _incomplete(
|
||||||
|
f"control-plane database not present at {redact_path(path)}; "
|
||||||
|
"no session or lease authority available"
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
conn = sqlite3.connect(f"file:{path}?mode=ro", uri=True, timeout=5)
|
||||||
|
conn.row_factory = sqlite3.Row
|
||||||
|
except sqlite3.Error as exc:
|
||||||
|
return _incomplete(f"control-plane database could not be opened: {exc}")
|
||||||
|
|
||||||
|
sessions: list[dict[str, Any]] = []
|
||||||
|
leases: list[dict[str, Any]] = []
|
||||||
|
capped = max(1, int(limit))
|
||||||
|
try:
|
||||||
|
tables = {
|
||||||
|
str(row[0])
|
||||||
|
for row in conn.execute(
|
||||||
|
"SELECT name FROM sqlite_master WHERE type = 'table'"
|
||||||
|
).fetchall()
|
||||||
|
}
|
||||||
|
if "sessions" not in tables:
|
||||||
|
incomplete.append("control-plane database has no sessions table")
|
||||||
|
else:
|
||||||
|
sessions = [
|
||||||
|
dict(row)
|
||||||
|
for row in conn.execute(
|
||||||
|
"SELECT session_id, role, profile, pid, status,"
|
||||||
|
" last_heartbeat_at FROM sessions"
|
||||||
|
" WHERE status = 'active'"
|
||||||
|
" ORDER BY last_heartbeat_at DESC LIMIT ?",
|
||||||
|
(capped,),
|
||||||
|
).fetchall()
|
||||||
|
]
|
||||||
|
|
||||||
|
if "leases" not in tables:
|
||||||
|
incomplete.append("control-plane database has no leases table")
|
||||||
|
elif "work_items" not in tables:
|
||||||
|
incomplete.append(
|
||||||
|
"control-plane database has no work_items table; lease work "
|
||||||
|
"identity cannot be resolved"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
leases = [
|
||||||
|
dict(row)
|
||||||
|
for row in conn.execute(
|
||||||
|
"SELECT l.lease_id, l.session_id, l.role, l.phase,"
|
||||||
|
" l.status AS freshness, l.worktree_path,"
|
||||||
|
" w.kind AS work_kind, w.number AS work_number"
|
||||||
|
" FROM leases l"
|
||||||
|
" JOIN work_items w ON w.work_item_id = l.work_item_id"
|
||||||
|
" WHERE l.status = 'active'"
|
||||||
|
" ORDER BY l.expires_at DESC LIMIT ?",
|
||||||
|
(capped,),
|
||||||
|
).fetchall()
|
||||||
|
]
|
||||||
|
except sqlite3.Error as exc:
|
||||||
|
return _incomplete(f"control-plane database read failed: {exc}")
|
||||||
|
finally:
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
return {
|
||||||
|
"sessions": sessions,
|
||||||
|
"leases": leases,
|
||||||
|
"terminal_lock": None,
|
||||||
|
"prior_recovery_attempts": [],
|
||||||
|
"inventory_complete": not incomplete,
|
||||||
|
"incomplete_reasons": incomplete,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
# --- Composition ------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def build_restart_class_views(viewer_role: str | None) -> tuple[RestartClassView, ...]:
|
||||||
|
"""Render the #663 class matrix, marking what this viewer may request."""
|
||||||
|
|
||||||
|
normalized = str(viewer_role or "").strip().lower()
|
||||||
|
views: list[RestartClassView] = []
|
||||||
|
for policy in restart_coordinator.RESTART_CLASS_POLICIES.values():
|
||||||
|
views.append(
|
||||||
|
RestartClassView(
|
||||||
|
restart_class=policy.restart_class.value,
|
||||||
|
required_permission=policy.required_permission,
|
||||||
|
expected_blast_radius=policy.expected_blast_radius,
|
||||||
|
drain_requirement=policy.drain_requirement,
|
||||||
|
full_drain_required=policy.full_drain_required,
|
||||||
|
approval_requirement=policy.approval_requirement,
|
||||||
|
request_roles=tuple(policy.request_roles),
|
||||||
|
execution_roles=tuple(policy.execution_roles),
|
||||||
|
viewer_may_request=normalized in policy.request_roles,
|
||||||
|
viewer_may_execute=normalized in policy.execution_roles,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return tuple(views)
|
||||||
|
|
||||||
|
|
||||||
|
def build_action_authorizations(
|
||||||
|
principal: console_authz.Principal | None,
|
||||||
|
) -> tuple[ActionAuthorization, ...]:
|
||||||
|
"""Authorization state for the approval controls, asked as execution.
|
||||||
|
|
||||||
|
``for_execution=True`` is deliberate. Asking without it answers "is this
|
||||||
|
principal senior enough", which is not the question an operator looking at a
|
||||||
|
control needs answered; asking with it answers "would this run", and while
|
||||||
|
the console is in Phase 1 the honest answer is no.
|
||||||
|
"""
|
||||||
|
|
||||||
|
results: list[ActionAuthorization] = []
|
||||||
|
for action_id in REPORTED_ACTIONS:
|
||||||
|
action = console_authz.get_action(action_id)
|
||||||
|
decision = console_authz.authorize(action_id, principal, for_execution=True)
|
||||||
|
results.append(
|
||||||
|
ActionAuthorization(
|
||||||
|
action_id=action_id,
|
||||||
|
summary=action.summary if action else "",
|
||||||
|
required_role=(
|
||||||
|
action.minimum_role if action else console_authz.OPERATOR
|
||||||
|
),
|
||||||
|
allowed=bool(decision.allowed),
|
||||||
|
execution_enabled=bool(decision.execution_enabled),
|
||||||
|
reason_code=str(decision.reason_code or ""),
|
||||||
|
detail=str(decision.detail or ""),
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return tuple(results)
|
||||||
|
|
||||||
|
|
||||||
|
def viewer_is_privileged(principal: console_authz.Principal | None) -> bool:
|
||||||
|
"""True when the viewer holds at least the operator role."""
|
||||||
|
|
||||||
|
who = principal if principal is not None else console_authz.ANONYMOUS
|
||||||
|
if not who.authenticated:
|
||||||
|
return False
|
||||||
|
return who.rank >= console_authz.ROLE_ORDER.index(console_authz.OPERATOR)
|
||||||
|
|
||||||
|
|
||||||
|
def load_impact_report(
|
||||||
|
*,
|
||||||
|
principal: console_authz.Principal | None = None,
|
||||||
|
restart_class: str = restart_coordinator.RestartClass.FULL_MCP_RESTART.value,
|
||||||
|
db_path: str | None = None,
|
||||||
|
limit: int = 200,
|
||||||
|
read_inventory: Callable[..., Mapping[str, Any]] | None = None,
|
||||||
|
now: datetime | None = None,
|
||||||
|
) -> tuple[dict[str, Any] | None, SourceStatus]:
|
||||||
|
"""Evaluate the blast radius for *restart_class*, always dry-run."""
|
||||||
|
|
||||||
|
reader = read_inventory or read_control_plane_inventory
|
||||||
|
try:
|
||||||
|
inventory = dict(reader(db_path=db_path, limit=limit))
|
||||||
|
except Exception as exc: # noqa: BLE001
|
||||||
|
return None, SourceStatus(
|
||||||
|
"impact",
|
||||||
|
STATUS_UNAVAILABLE,
|
||||||
|
f"control-plane inventory failed: {type(exc).__name__}: {exc}",
|
||||||
|
)
|
||||||
|
|
||||||
|
who = principal if principal is not None else console_authz.ANONYMOUS
|
||||||
|
viewer_role = str(who.role or "").strip().lower()
|
||||||
|
try:
|
||||||
|
report = restart_coordinator.evaluate_restart_impact(
|
||||||
|
inventory,
|
||||||
|
now=now,
|
||||||
|
dry_run=True,
|
||||||
|
restart_class=restart_class,
|
||||||
|
requester_role=viewer_role,
|
||||||
|
requester_permissions=restart_coordinator.permissions_for_role(
|
||||||
|
viewer_role
|
||||||
|
),
|
||||||
|
)
|
||||||
|
except Exception as exc: # noqa: BLE001
|
||||||
|
return None, SourceStatus(
|
||||||
|
"impact",
|
||||||
|
STATUS_UNAVAILABLE,
|
||||||
|
f"impact evaluation failed: {type(exc).__name__}: {exc}",
|
||||||
|
)
|
||||||
|
|
||||||
|
payload = scrub(report.as_dict())
|
||||||
|
detail = ""
|
||||||
|
if not report.inventory_complete:
|
||||||
|
detail = "; ".join(report.incomplete_reasons) or "inventory incomplete"
|
||||||
|
return payload, SourceStatus("impact", STATUS_OK, detail)
|
||||||
|
|
||||||
|
|
||||||
|
def load_drain_status(
|
||||||
|
*,
|
||||||
|
proof: Mapping[str, Any] | None = None,
|
||||||
|
now: datetime | None = None,
|
||||||
|
expected_impact_fingerprint: str | None = None,
|
||||||
|
) -> tuple[dict[str, Any] | None, SourceStatus]:
|
||||||
|
"""Verify a supplied drain proof read-only and report the verdict.
|
||||||
|
|
||||||
|
No proof supplied is not a failure and not a pass: it is reported as the
|
||||||
|
absence of a proof, which is exactly what the #661 gate would deny on.
|
||||||
|
"""
|
||||||
|
|
||||||
|
if proof is None:
|
||||||
|
return None, SourceStatus(
|
||||||
|
"drain",
|
||||||
|
STATUS_UNAVAILABLE,
|
||||||
|
"no drain proof supplied; the #661 gate denies a restart without a "
|
||||||
|
"valid unexpired clean proof",
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
verified = drain_proof.verify_drain_proof(
|
||||||
|
proof,
|
||||||
|
now=now,
|
||||||
|
expected_impact_fingerprint=expected_impact_fingerprint,
|
||||||
|
)
|
||||||
|
except Exception as exc: # noqa: BLE001
|
||||||
|
return None, SourceStatus(
|
||||||
|
"drain",
|
||||||
|
STATUS_UNAVAILABLE,
|
||||||
|
f"drain proof verification failed: {type(exc).__name__}: {exc}",
|
||||||
|
)
|
||||||
|
return scrub(verified.as_dict()), SourceStatus("drain", STATUS_OK)
|
||||||
|
|
||||||
|
|
||||||
|
def load_reconcile_status(
|
||||||
|
*,
|
||||||
|
load_proof: Callable[[], Any] | None = None,
|
||||||
|
) -> tuple[dict[str, Any] | None, SourceStatus]:
|
||||||
|
"""Report the most recent post-restart completion proof (#662)."""
|
||||||
|
|
||||||
|
if load_proof is None:
|
||||||
|
return None, SourceStatus(
|
||||||
|
"reconcile",
|
||||||
|
STATUS_UNAVAILABLE,
|
||||||
|
"no post-restart completion proof source is wired into this view",
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
proof = load_proof()
|
||||||
|
except Exception as exc: # noqa: BLE001
|
||||||
|
return None, SourceStatus(
|
||||||
|
"reconcile",
|
||||||
|
STATUS_UNAVAILABLE,
|
||||||
|
f"reconcile proof unavailable: {type(exc).__name__}: {exc}",
|
||||||
|
)
|
||||||
|
if proof is None:
|
||||||
|
return None, SourceStatus(
|
||||||
|
"reconcile",
|
||||||
|
STATUS_UNAVAILABLE,
|
||||||
|
"no post-restart reconcile has been recorded",
|
||||||
|
)
|
||||||
|
payload = proof.as_dict() if hasattr(proof, "as_dict") else dict(proof)
|
||||||
|
return scrub(payload), SourceStatus("reconcile", STATUS_OK)
|
||||||
|
|
||||||
|
|
||||||
|
def load_restart_console_snapshot(
|
||||||
|
*,
|
||||||
|
principal: console_authz.Principal | None = None,
|
||||||
|
restart_class: str = restart_coordinator.RestartClass.FULL_MCP_RESTART.value,
|
||||||
|
db_path: str | None = None,
|
||||||
|
limit: int = 200,
|
||||||
|
drain_proof_payload: Mapping[str, Any] | None = None,
|
||||||
|
read_inventory: Callable[..., Mapping[str, Any]] | None = None,
|
||||||
|
load_reconcile_proof: Callable[[], Any] | None = None,
|
||||||
|
now: datetime | None = None,
|
||||||
|
) -> RestartConsoleSnapshot:
|
||||||
|
"""Compose the read-only restart console snapshot."""
|
||||||
|
|
||||||
|
who = principal if principal is not None else console_authz.ANONYMOUS
|
||||||
|
moment = now or _utc_now()
|
||||||
|
|
||||||
|
impact, impact_source = load_impact_report(
|
||||||
|
principal=who,
|
||||||
|
restart_class=restart_class,
|
||||||
|
db_path=db_path,
|
||||||
|
limit=limit,
|
||||||
|
read_inventory=read_inventory,
|
||||||
|
now=moment,
|
||||||
|
)
|
||||||
|
fingerprint = None
|
||||||
|
if impact is not None:
|
||||||
|
try:
|
||||||
|
fingerprint = drain_proof.impact_fingerprint(impact)
|
||||||
|
except Exception: # noqa: BLE001
|
||||||
|
fingerprint = None
|
||||||
|
|
||||||
|
drain, drain_source = load_drain_status(
|
||||||
|
proof=drain_proof_payload,
|
||||||
|
now=moment,
|
||||||
|
expected_impact_fingerprint=fingerprint,
|
||||||
|
)
|
||||||
|
reconcile, reconcile_source = load_reconcile_status(
|
||||||
|
load_proof=load_reconcile_proof
|
||||||
|
)
|
||||||
|
|
||||||
|
notes: list[str] = [
|
||||||
|
"This surface is read-only: it evaluates and displays, and performs no "
|
||||||
|
"restart, reload, drain, approval, or process action.",
|
||||||
|
]
|
||||||
|
if not impact_source.available:
|
||||||
|
notes.append(
|
||||||
|
"Impact preview unavailable — a restart decision must not be made "
|
||||||
|
"from this page while the blast radius is unknown."
|
||||||
|
)
|
||||||
|
|
||||||
|
return RestartConsoleSnapshot(
|
||||||
|
generated_at=moment.isoformat(),
|
||||||
|
viewer_role=str(who.role or "anonymous"),
|
||||||
|
viewer_authenticated=bool(who.authenticated),
|
||||||
|
read_only=True,
|
||||||
|
impact=impact,
|
||||||
|
impact_source=impact_source,
|
||||||
|
drain=drain,
|
||||||
|
drain_source=drain_source,
|
||||||
|
reconcile=reconcile,
|
||||||
|
reconcile_source=reconcile_source,
|
||||||
|
restart_classes=build_restart_class_views(who.role),
|
||||||
|
authorizations=build_action_authorizations(who),
|
||||||
|
break_glass=BreakGlassSurface(
|
||||||
|
available=False,
|
||||||
|
issue=BREAK_GLASS_ISSUE,
|
||||||
|
reason=BREAK_GLASS_PENDING_REASON,
|
||||||
|
viewer_is_privileged=viewer_is_privileged(who),
|
||||||
|
),
|
||||||
|
notes=tuple(notes),
|
||||||
|
)
|
||||||
@@ -0,0 +1,299 @@
|
|||||||
|
"""HTML views for the read-only restart console (#667).
|
||||||
|
|
||||||
|
Every interpolated value passes through :func:`_esc`. Values that can carry a
|
||||||
|
filesystem path or free-form operator text additionally pass through
|
||||||
|
:func:`webui.inventory.scrub_text`, which redacts credential-shaped tokens
|
||||||
|
*inside* a string rather than only at its start.
|
||||||
|
|
||||||
|
The page renders state and never offers a control that would mutate anything:
|
||||||
|
the approval and break-glass panels report authorization and availability, and
|
||||||
|
there is no form, button, or endpoint behind them.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import html
|
||||||
|
|
||||||
|
from webui.inventory import scrub_text
|
||||||
|
from webui.restart_console import RestartConsoleSnapshot, SourceStatus
|
||||||
|
|
||||||
|
|
||||||
|
def _esc(value: object) -> str:
|
||||||
|
"""Escape any value for HTML text or a quoted attribute."""
|
||||||
|
if value is None:
|
||||||
|
return ""
|
||||||
|
return html.escape(str(value), quote=True)
|
||||||
|
|
||||||
|
|
||||||
|
def _esc_text(value: object) -> str:
|
||||||
|
"""Escape free-form text after redacting secrets embedded inside it."""
|
||||||
|
if value is None:
|
||||||
|
return ""
|
||||||
|
return _esc(scrub_text(str(value)))
|
||||||
|
|
||||||
|
|
||||||
|
def _bool_badge(
|
||||||
|
value: bool, *, true_label: str = "yes", false_label: str = "no"
|
||||||
|
) -> str:
|
||||||
|
css = "badge-ok" if value else "badge-blocked"
|
||||||
|
label = true_label if value else false_label
|
||||||
|
return f'<span class="badge {css}">{_esc(label)}</span>'
|
||||||
|
|
||||||
|
|
||||||
|
def _source_badge(source: SourceStatus) -> str:
|
||||||
|
css = "badge-ok" if source.available else "badge-blocked"
|
||||||
|
badge = f'<span class="badge {css}">{_esc(source.status)}</span>'
|
||||||
|
if source.detail:
|
||||||
|
badge += f' <span class="muted">{_esc_text(source.detail)}</span>'
|
||||||
|
return badge
|
||||||
|
|
||||||
|
|
||||||
|
def _notes_block(snapshot: RestartConsoleSnapshot) -> str:
|
||||||
|
if not snapshot.notes:
|
||||||
|
return ""
|
||||||
|
items = "".join(f"<li>{_esc_text(note)}</li>" for note in snapshot.notes)
|
||||||
|
return f"<ul class='reasons'>{items}</ul>"
|
||||||
|
|
||||||
|
|
||||||
|
def _impact_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||||
|
head = (
|
||||||
|
"<section class='health-card'>"
|
||||||
|
f"<h3>Impact preview {_source_badge(snapshot.impact_source)}</h3>"
|
||||||
|
)
|
||||||
|
impact = snapshot.impact
|
||||||
|
if impact is None:
|
||||||
|
return (
|
||||||
|
head
|
||||||
|
+ "<p class='muted'>No impact preview is available, so the blast "
|
||||||
|
"radius of a restart is unknown. Treat this as unsafe.</p></section>"
|
||||||
|
)
|
||||||
|
|
||||||
|
counts = impact.get("counts") or {}
|
||||||
|
verdict = str(impact.get("verdict") or "unknown")
|
||||||
|
verdict_css = "badge-ok" if verdict == "safe" else "badge-blocked"
|
||||||
|
rows = "".join(
|
||||||
|
f"<tr><th>{_esc(key.replace('_', ' '))}</th><td>{_esc(value)}</td></tr>"
|
||||||
|
for key, value in sorted(counts.items())
|
||||||
|
)
|
||||||
|
reasons = "".join(
|
||||||
|
f"<li>{_esc_text(reason)}</li>" for reason in (impact.get("reasons") or [])
|
||||||
|
)
|
||||||
|
incomplete = ""
|
||||||
|
if not impact.get("inventory_complete", False):
|
||||||
|
detail = "; ".join(str(r) for r in (impact.get("incomplete_reasons") or []))
|
||||||
|
incomplete = (
|
||||||
|
"<p class='error'><strong>Inventory incomplete:</strong> "
|
||||||
|
f"{_esc_text(detail or 'unspecified')}. The coordinator fails "
|
||||||
|
"closed on an incomplete inventory.</p>"
|
||||||
|
)
|
||||||
|
|
||||||
|
sessions = impact.get("affected_sessions") or []
|
||||||
|
session_rows = "".join(
|
||||||
|
"<tr>"
|
||||||
|
f"<td><code>{_esc(s.get('session_id'))}</code></td>"
|
||||||
|
f"<td>{_esc(s.get('role'))}</td>"
|
||||||
|
f"<td>{_esc(s.get('pid'))}</td>"
|
||||||
|
f"<td>{_bool_badge(bool(s.get('live')), true_label='live', false_label='idle')}</td>"
|
||||||
|
f"<td>{_bool_badge(not s.get('heartbeat_stale'), true_label='fresh', false_label='stale')}</td>"
|
||||||
|
"</tr>"
|
||||||
|
for s in sessions[:50]
|
||||||
|
)
|
||||||
|
session_table = (
|
||||||
|
"<h4>Sessions a restart would terminate</h4>"
|
||||||
|
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||||
|
"<th>Session</th><th>Role</th><th>PID</th><th>State</th>"
|
||||||
|
"<th>Heartbeat</th></tr></thead><tbody>"
|
||||||
|
f"{session_rows}</tbody></table></div>"
|
||||||
|
if session_rows
|
||||||
|
else "<p class='muted'>No affected sessions reported.</p>"
|
||||||
|
)
|
||||||
|
truncated = (
|
||||||
|
f"<p class='muted'>Showing the first 50 of {_esc(len(sessions))} "
|
||||||
|
"affected sessions.</p>"
|
||||||
|
if len(sessions) > 50
|
||||||
|
else ""
|
||||||
|
)
|
||||||
|
|
||||||
|
return (
|
||||||
|
head
|
||||||
|
+ "<p class='health-headline'>Verdict "
|
||||||
|
f"<span class='badge {verdict_css}'>{_esc(verdict)}</span> · "
|
||||||
|
f"blast radius <code>{_esc(impact.get('blast_radius'))}</code> · "
|
||||||
|
f"class <code>{_esc(impact.get('restart_class'))}</code></p>"
|
||||||
|
+ incomplete
|
||||||
|
+ (f"<ul class='reasons'>{reasons}</ul>" if reasons else "")
|
||||||
|
+ (f"<table class='registry'><tbody>{rows}</tbody></table>" if rows else "")
|
||||||
|
+ session_table
|
||||||
|
+ truncated
|
||||||
|
+ "</section>"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _drain_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||||
|
head = (
|
||||||
|
"<section class='health-card'>"
|
||||||
|
f"<h3>Drain proof {_source_badge(snapshot.drain_source)}</h3>"
|
||||||
|
)
|
||||||
|
drain = snapshot.drain
|
||||||
|
if drain is None:
|
||||||
|
return (
|
||||||
|
head
|
||||||
|
+ "<p class='muted'>No drain proof has been presented to this view. "
|
||||||
|
"The #661 gate authorizes a restart only against a valid, unexpired, "
|
||||||
|
"clean proof, so the absence of one is a denial, not a pass.</p>"
|
||||||
|
"</section>"
|
||||||
|
)
|
||||||
|
reasons = "".join(
|
||||||
|
f"<li>{_esc_text(reason)}</li>" for reason in (drain.get("reasons") or [])
|
||||||
|
)
|
||||||
|
return (
|
||||||
|
head
|
||||||
|
+ "<table class='registry'><tbody>"
|
||||||
|
f"<tr><th>Valid</th><td>{_bool_badge(bool(drain.get('valid')))}</td></tr>"
|
||||||
|
f"<tr><th>Clean</th><td>{_bool_badge(bool(drain.get('clean')))}</td></tr>"
|
||||||
|
f"<tr><th>Expired</th><td>{_bool_badge(not drain.get('expired'), true_label='no', false_label='yes')}</td></tr>"
|
||||||
|
f"<tr><th>Tampered</th><td>{_bool_badge(not drain.get('tampered'), true_label='no', false_label='yes')}</td></tr>"
|
||||||
|
f"<tr><th>Proof id</th><td><code>{_esc(drain.get('proof_id'))}</code></td></tr>"
|
||||||
|
"</tbody></table>"
|
||||||
|
+ (f"<ul class='reasons'>{reasons}</ul>" if reasons else "")
|
||||||
|
+ "</section>"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _reconcile_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||||
|
head = (
|
||||||
|
"<section class='health-card'>"
|
||||||
|
f"<h3>Post-restart reconcile {_source_badge(snapshot.reconcile_source)}</h3>"
|
||||||
|
)
|
||||||
|
proof = snapshot.reconcile
|
||||||
|
if proof is None:
|
||||||
|
return (
|
||||||
|
head
|
||||||
|
+ "<p class='muted'>No post-restart completion proof is recorded. "
|
||||||
|
"Until one is, the last restart's recovery state is unproven.</p>"
|
||||||
|
"</section>"
|
||||||
|
)
|
||||||
|
items = "".join(
|
||||||
|
"<tr>"
|
||||||
|
f"<td>{_esc(item.get('dimension'))}</td>"
|
||||||
|
f"<td>{_esc(item.get('status'))}</td>"
|
||||||
|
f"<td>{_esc_text(item.get('summary'))}</td>"
|
||||||
|
f"<td>{_bool_badge(not item.get('follow_up_required'), true_label='no', false_label='yes')}</td>"
|
||||||
|
"</tr>"
|
||||||
|
for item in (proof.get("items") or [])
|
||||||
|
)
|
||||||
|
return (
|
||||||
|
head
|
||||||
|
+ "<p class='health-headline'>Status "
|
||||||
|
f"<code>{_esc(proof.get('overall_status'))}</code> · mode "
|
||||||
|
f"<code>{_esc(proof.get('mode'))}</code> · resolved "
|
||||||
|
f"{_esc(proof.get('resolved_count'))} · unresolved "
|
||||||
|
f"{_esc(proof.get('unresolved_count'))}</p>"
|
||||||
|
+ (
|
||||||
|
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||||
|
"<th>Dimension</th><th>Status</th><th>Summary</th>"
|
||||||
|
"<th>Follow-up required</th></tr></thead><tbody>"
|
||||||
|
f"{items}</tbody></table></div>"
|
||||||
|
if items
|
||||||
|
else "<p class='muted'>No reconcile dimensions reported.</p>"
|
||||||
|
)
|
||||||
|
+ "</section>"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _class_matrix_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||||
|
rows = "".join(
|
||||||
|
"<tr>"
|
||||||
|
f"<td><code>{_esc(view.restart_class)}</code></td>"
|
||||||
|
f"<td><code>{_esc(view.required_permission)}</code></td>"
|
||||||
|
f"<td>{_esc(view.expected_blast_radius)}</td>"
|
||||||
|
f"<td>{_esc(view.drain_requirement)}</td>"
|
||||||
|
f"<td>{_esc(view.approval_requirement)}</td>"
|
||||||
|
f"<td>{_bool_badge(view.viewer_may_request)}</td>"
|
||||||
|
f"<td>{_bool_badge(view.viewer_may_execute)}</td>"
|
||||||
|
"</tr>"
|
||||||
|
for view in snapshot.restart_classes
|
||||||
|
)
|
||||||
|
return (
|
||||||
|
"<section class='health-card'>"
|
||||||
|
"<h3>Restart classes</h3>"
|
||||||
|
"<p class='muted'>The least-privilege matrix each restart request is "
|
||||||
|
"resolved against. “You may request” and “you may "
|
||||||
|
"execute” are computed for the current viewer role, not for a "
|
||||||
|
"generic operator.</p>"
|
||||||
|
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||||
|
"<th>Class</th><th>Permission</th><th>Blast radius</th>"
|
||||||
|
"<th>Drain</th><th>Approval</th><th>You may request</th>"
|
||||||
|
"<th>You may execute</th></tr></thead><tbody>"
|
||||||
|
f"{rows}</tbody></table></div>"
|
||||||
|
"</section>"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _approval_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||||
|
rows = "".join(
|
||||||
|
"<tr>"
|
||||||
|
f"<td><code>{_esc(a.action_id)}</code></td>"
|
||||||
|
f"<td>{_esc(a.required_role)}</td>"
|
||||||
|
f"<td>{_bool_badge(a.allowed)}</td>"
|
||||||
|
f"<td>{_bool_badge(a.execution_enabled)}</td>"
|
||||||
|
f"<td><code>{_esc(a.reason_code)}</code></td>"
|
||||||
|
f"<td>{_esc_text(a.detail)}</td>"
|
||||||
|
"</tr>"
|
||||||
|
for a in snapshot.authorizations
|
||||||
|
)
|
||||||
|
return (
|
||||||
|
"<section class='health-card'>"
|
||||||
|
"<h3>Approval controls</h3>"
|
||||||
|
"<p class='muted'>Authorization is probed the way execution would probe "
|
||||||
|
"it, so “execution enabled” answers whether the action would "
|
||||||
|
"actually run — not merely whether this role outranks the requirement. "
|
||||||
|
"No control on this page performs the action.</p>"
|
||||||
|
"<div class='table-scroll'><table class='registry'><thead><tr>"
|
||||||
|
"<th>Action</th><th>Required role</th><th>Authorized</th>"
|
||||||
|
"<th>Execution enabled</th><th>Reason</th><th>Detail</th>"
|
||||||
|
"</tr></thead><tbody>"
|
||||||
|
f"{rows}</tbody></table></div>"
|
||||||
|
"</section>"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _break_glass_section(snapshot: RestartConsoleSnapshot) -> str:
|
||||||
|
bg = snapshot.break_glass
|
||||||
|
if not bg.viewer_is_privileged:
|
||||||
|
return (
|
||||||
|
"<section class='health-card'>"
|
||||||
|
"<h3>Break-glass</h3>"
|
||||||
|
"<p class='muted'>Break-glass status is visible to operator-class "
|
||||||
|
"roles only. Your role does not carry that authority, so no "
|
||||||
|
"emergency surface is shown.</p>"
|
||||||
|
"</section>"
|
||||||
|
)
|
||||||
|
return (
|
||||||
|
"<section class='health-card'>"
|
||||||
|
"<h3>Break-glass "
|
||||||
|
f"{_bool_badge(bg.available, true_label='available', false_label='unavailable')}"
|
||||||
|
"</h3>"
|
||||||
|
f"<p class='muted'>{_esc_text(bg.reason)}</p>"
|
||||||
|
f"<p class='meta'>Tracked by issue #{_esc(bg.issue)}.</p>"
|
||||||
|
"</section>"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def render_restart_console_page(snapshot: RestartConsoleSnapshot) -> str:
|
||||||
|
"""Render the whole read-only restart console body."""
|
||||||
|
|
||||||
|
return (
|
||||||
|
"<h2>Restart status and impact</h2>"
|
||||||
|
f"<p class='meta'>Generated <code>{_esc(snapshot.generated_at)}</code> · "
|
||||||
|
f"viewer role <code>{_esc(snapshot.viewer_role)}</code> · "
|
||||||
|
f"authenticated {_bool_badge(snapshot.viewer_authenticated)} · "
|
||||||
|
f"read-only {_bool_badge(snapshot.read_only)}</p>"
|
||||||
|
+ _notes_block(snapshot)
|
||||||
|
+ _impact_section(snapshot)
|
||||||
|
+ _drain_section(snapshot)
|
||||||
|
+ _reconcile_section(snapshot)
|
||||||
|
+ _class_matrix_section(snapshot)
|
||||||
|
+ _approval_section(snapshot)
|
||||||
|
+ _break_glass_section(snapshot)
|
||||||
|
)
|
||||||
+30
-1
@@ -279,6 +279,7 @@ def assess_root_source_mutation(
|
|||||||
locked_issue_number: int | None = None,
|
locked_issue_number: int | None = None,
|
||||||
role_kind: str | None = None,
|
role_kind: str | None = None,
|
||||||
mutation_task: str | None = None,
|
mutation_task: str | None = None,
|
||||||
|
bootstrap_assessment: Any | None = None,
|
||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""Fail closed for diagnostic/source edits on the control/root checkout.
|
"""Fail closed for diagnostic/source edits on the control/root checkout.
|
||||||
|
|
||||||
@@ -286,6 +287,13 @@ def assess_root_source_mutation(
|
|||||||
tracked source/test files on the control checkout always block, including
|
tracked source/test files on the control checkout always block, including
|
||||||
temporary/diagnostic/test-only intent.
|
temporary/diagnostic/test-only intent.
|
||||||
|
|
||||||
|
#941: ``bootstrap_author_issue_worktree`` is judged by the canonical
|
||||||
|
``create_issue_bootstrap.bootstrap_permits_control_checkout`` decision over
|
||||||
|
*bootstrap_assessment* — the same server-derived evidence the #274 and
|
||||||
|
#604 guards consume — instead of a task-name allowlist local to this
|
||||||
|
module. Evidence that is absent, malformed, wrongly scoped, or bound to
|
||||||
|
another workspace leaves the ordinary block in force.
|
||||||
|
|
||||||
#749: ``create_issue`` is a pure remote mutation with no local tree write.
|
#749: ``create_issue`` is a pure remote mutation with no local tree write.
|
||||||
When *mutation_task* is create_issue and the control checkout has no dirty
|
When *mutation_task* is create_issue and the control checkout has no dirty
|
||||||
source/test files, the missing-worktree signal is suppressed so the
|
source/test files, the missing-worktree signal is suppressed so the
|
||||||
@@ -336,6 +344,19 @@ def assess_root_source_mutation(
|
|||||||
if _cib is not None and _cib.is_create_issue_task(mutation_task):
|
if _cib is not None and _cib.is_create_issue_task(mutation_task):
|
||||||
# #749: clean-root create_issue is the sanctioned bootstrap path.
|
# #749: clean-root create_issue is the sanctioned bootstrap path.
|
||||||
create_issue_bootstrap = True
|
create_issue_bootstrap = True
|
||||||
|
elif _cib is not None and _cib.bootstrap_permits_control_checkout(
|
||||||
|
bootstrap_assessment,
|
||||||
|
task=mutation_task,
|
||||||
|
workspace_path=workspace,
|
||||||
|
canonical_repo_root=root,
|
||||||
|
):
|
||||||
|
# #941: the author issue-worktree bootstrap is authorized by the
|
||||||
|
# canonical shared decision over server-derived task-scope
|
||||||
|
# evidence, never by a task-name allowlist kept in this module.
|
||||||
|
# The predicate fails closed on missing, malformed, cross-scope,
|
||||||
|
# dirty, drifted, or wrongly bound evidence, so this arm cannot
|
||||||
|
# widen the waiver beyond the one sanctioned bootstrap task.
|
||||||
|
create_issue_bootstrap = True
|
||||||
else:
|
else:
|
||||||
# Explicit missing-worktree signal for force-on author entrypoints.
|
# Explicit missing-worktree signal for force-on author entrypoints.
|
||||||
reasons.append(
|
reasons.append(
|
||||||
@@ -393,8 +414,15 @@ def assess_production_mutation_guards(
|
|||||||
require_author_lock: bool = False,
|
require_author_lock: bool = False,
|
||||||
in_test_mode: bool = False,
|
in_test_mode: bool = False,
|
||||||
mutation_task: str | None = None,
|
mutation_task: str | None = None,
|
||||||
|
bootstrap_assessment: Any | None = None,
|
||||||
) -> dict[str, Any]:
|
) -> dict[str, Any]:
|
||||||
"""Compose root + scope production guards when they must be active (#683)."""
|
"""Compose root + scope production guards when they must be active (#683).
|
||||||
|
|
||||||
|
#941: *bootstrap_assessment* is the server-derived author-bootstrap
|
||||||
|
evidence, forwarded unchanged to :func:`assess_root_source_mutation` so
|
||||||
|
this guard reaches the same canonical decision as the #274 and #604
|
||||||
|
guards. Omitting it preserves the pre-existing behaviour.
|
||||||
|
"""
|
||||||
if not production_guards_active(in_test_mode=in_test_mode):
|
if not production_guards_active(in_test_mode=in_test_mode):
|
||||||
return {
|
return {
|
||||||
"proven": True,
|
"proven": True,
|
||||||
@@ -414,6 +442,7 @@ def assess_production_mutation_guards(
|
|||||||
locked_issue_number=locked_issue_number,
|
locked_issue_number=locked_issue_number,
|
||||||
role_kind=role_kind,
|
role_kind=role_kind,
|
||||||
mutation_task=mutation_task,
|
mutation_task=mutation_task,
|
||||||
|
bootstrap_assessment=bootstrap_assessment,
|
||||||
)
|
)
|
||||||
if root_assess["block"]:
|
if root_assess["block"]:
|
||||||
return {**root_assess, "skipped": False}
|
return {**root_assess, "skipped": False}
|
||||||
|
|||||||
Reference in New Issue
Block a user